UNPKG

@omnimedia/omnitool

Version:

open source video processing tools

56 lines 2.19 kB
import { WhisperTextStreamer } from "@huggingface/transformers"; export async function transcribe(options) { const { pipe, spec, request, callbacks } = options; if (!pipe.processor.feature_extractor) throw new Error("no feature_extractor"); const timePrecision = (pipe.processor.feature_extractor?.config.chunk_length / // @ts-ignore pipe.model.config.max_source_positions); let chunkCount = 0; let startTime = null; let tokenCount = 0; let tokensPerSecond = 0; const chunkDuration = spec.chunkLength - spec.strideLength; const calculateProgress = () => { const audioProgressSeconds = chunkCount * chunkDuration; return Math.min(audioProgressSeconds / request.duration, 1); }; // TODO type error on pipe.tokenizer const tokenizer = pipe.tokenizer; const streamer = new WhisperTextStreamer(tokenizer, { time_precision: timePrecision, token_callback_function: () => { startTime ??= performance.now(); if (++tokenCount > 1) { tokensPerSecond = (tokenCount / (performance.now() - startTime)) * 1000; } }, callback_function: (textChunk) => { // TODO callbacks.onTranscription(textChunk); callbacks.onReport({ tokensPerSecond, progress: calculateProgress() }); }, on_finalize: () => { startTime = null; tokenCount = 0; chunkCount++; callbacks.onReport({ tokensPerSecond, progress: calculateProgress() }); }, }); const result = await pipe(new Float32Array(request.audio), { top_k: 0, do_sample: false, chunk_length_s: spec.chunkLength, stride_length_s: spec.strideLength, language: request.language, task: "transcribe", return_timestamps: "word", // if using "word" the on_chunk_start & end is not called thus we cant retrieve timestamps, only after whole thing finishes force_full_sequences: false, streamer, }); return { text: result.text, chunks: result.chunks }; } //# sourceMappingURL=transcribe.js.map