@omnimedia/omnitool
Version:
open source video processing tools
56 lines • 2.19 kB
JavaScript
import { WhisperTextStreamer } from "@huggingface/transformers";
export async function transcribe(options) {
const { pipe, spec, request, callbacks } = options;
if (!pipe.processor.feature_extractor)
throw new Error("no feature_extractor");
const timePrecision = (pipe.processor.feature_extractor?.config.chunk_length /
// @ts-ignore
pipe.model.config.max_source_positions);
let chunkCount = 0;
let startTime = null;
let tokenCount = 0;
let tokensPerSecond = 0;
const chunkDuration = spec.chunkLength - spec.strideLength;
const calculateProgress = () => {
const audioProgressSeconds = chunkCount * chunkDuration;
return Math.min(audioProgressSeconds / request.duration, 1);
};
// TODO type error on pipe.tokenizer
const tokenizer = pipe.tokenizer;
const streamer = new WhisperTextStreamer(tokenizer, {
time_precision: timePrecision,
token_callback_function: () => {
startTime ??= performance.now();
if (++tokenCount > 1) {
tokensPerSecond = (tokenCount / (performance.now() - startTime)) * 1000;
}
},
callback_function: (textChunk) => {
// TODO
callbacks.onTranscription(textChunk);
callbacks.onReport({ tokensPerSecond, progress: calculateProgress() });
},
on_finalize: () => {
startTime = null;
tokenCount = 0;
chunkCount++;
callbacks.onReport({ tokensPerSecond, progress: calculateProgress() });
},
});
const result = await pipe(new Float32Array(request.audio), {
top_k: 0,
do_sample: false,
chunk_length_s: spec.chunkLength,
stride_length_s: spec.strideLength,
language: request.language,
task: "transcribe",
return_timestamps: "word", // if using "word" the on_chunk_start & end is not called thus we cant retrieve timestamps, only after whole thing finishes
force_full_sequences: false,
streamer,
});
return {
text: result.text,
chunks: result.chunks
};
}
//# sourceMappingURL=transcribe.js.map