assemblyai
Version:
The AssemblyAI JavaScript SDK provides an easy-to-use interface for interacting with the AssemblyAI API, which supports async and real-time transcription, as well as the latest LeMUR models.
36 lines (35 loc) • 1.74 kB
TypeScript
import { VadDetector, VadDetectorResult } from "../../types/streaming/dual-channel";
export type EnergyVadParams = {
/** Threshold = noiseFloor * thresholdRatio. Default 3.0 (~ +9.5 dB above noise). */
thresholdRatio?: number;
/** EMA smoothing for the noise-floor estimate when frame is non-speech. Default 0.05. */
noiseFloorAlpha?: number;
/** Hangover in frames: stay "active" this many frames after the last speech frame. Default 10 (\~200 ms at 20 ms frames). */
hangoverFrames?: number;
/** Initial noise floor estimate. Default 1e-4. Adaptive after the first non-speech frame. */
initialNoiseFloor?: number;
};
/**
* Energy-based VAD with adaptive noise-floor tracking and hangover. Pure JS,
* no dependencies. Suitable for the "which physical channel is speaking" task
* because the channels are already physically separated at capture — the harder
* problem (speech vs. non-speech in the wild) is one a customer can swap in a
* DNN VAD for via the `createVad` parameter.
*
* Tuning notes:
* - thresholdRatio below 2 will treat anything above noise as speech (too sensitive).
* - thresholdRatio above 6 will miss quiet utterance onsets/offsets.
* - noiseFloorAlpha above 0.1 makes the floor track quickly (good for non-stationary
* background) but risks slowly adapting *up* to a sustained low voice.
*/
export declare class EnergyVad implements VadDetector {
private readonly thresholdRatio;
private readonly noiseFloorAlpha;
private readonly hangoverFrames;
private readonly initialNoiseFloor;
private noiseFloor;
private hangoverRemaining;
constructor(params?: EnergyVadParams);
process(frame: Float32Array): VadDetectorResult;
reset(): void;
}