UNPKG

whisper-speech-to-text

Version:

A JavaScript library enabling in-browser audio recording and transcription using OpenAI's Whisper Speech-to-Text

190 lines (189 loc) 9.33 kB
"use strict"; var __awaiter = (this && this.__awaiter) || function (thisArg, _arguments, P, generator) { function adopt(value) { return value instanceof P ? value : new P(function (resolve) { resolve(value); }); } return new (P || (P = Promise))(function (resolve, reject) { function fulfilled(value) { try { step(generator.next(value)); } catch (e) { reject(e); } } function rejected(value) { try { step(generator["throw"](value)); } catch (e) { reject(e); } } function step(result) { result.done ? resolve(result.value) : adopt(result.value).then(fulfilled, rejected); } step((generator = generator.apply(thisArg, _arguments || [])).next()); }); }; var __generator = (this && this.__generator) || function (thisArg, body) { var _ = { label: 0, sent: function() { if (t[0] & 1) throw t[1]; return t[1]; }, trys: [], ops: [] }, f, y, t, g; return g = { next: verb(0), "throw": verb(1), "return": verb(2) }, typeof Symbol === "function" && (g[Symbol.iterator] = function() { return this; }), g; function verb(n) { return function (v) { return step([n, v]); }; } function step(op) { if (f) throw new TypeError("Generator is already executing."); while (g && (g = 0, op[0] && (_ = 0)), _) try { if (f = 1, y && (t = op[0] & 2 ? y["return"] : op[0] ? y["throw"] || ((t = y["return"]) && t.call(y), 0) : y.next) && !(t = t.call(y, op[1])).done) return t; if (y = 0, t) op = [op[0] & 2, t.value]; switch (op[0]) { case 0: case 1: t = op; break; case 4: _.label++; return { value: op[1], done: false }; case 5: _.label++; y = op[1]; op = [0]; continue; case 7: op = _.ops.pop(); _.trys.pop(); continue; default: if (!(t = _.trys, t = t.length > 0 && t[t.length - 1]) && (op[0] === 6 || op[0] === 2)) { _ = 0; continue; } if (op[0] === 3 && (!t || (op[1] > t[0] && op[1] < t[3]))) { _.label = op[1]; break; } if (op[0] === 6 && _.label < t[1]) { _.label = t[1]; t = op; break; } if (t && _.label < t[2]) { _.label = t[2]; _.ops.push(op); break; } if (t[2]) _.ops.pop(); _.trys.pop(); continue; } op = body.call(thisArg, _); } catch (e) { op = [6, e]; y = 0; } finally { f = t = 0; } if (op[0] & 5) throw op[1]; return { value: op[0] ? op[1] : void 0, done: true }; } }; var __importDefault = (this && this.__importDefault) || function (mod) { return (mod && mod.__esModule) ? mod : { "default": mod }; }; Object.defineProperty(exports, "__esModule", { value: true }); exports.WhisperSTT = void 0; var recordrtc_1 = require("recordrtc"); var axios_1 = __importDefault(require("axios")); var AUDIO_TYPE = 'audio'; var MODEL = 'whisper-1'; var TRANSCRIPTIONS_API_URL = 'https://api.openai.com/v1/audio/transcriptions'; var WhisperSTT = /** @class */ (function () { function WhisperSTT(apiKey) { var _this = this; this.pauseRecording = function () { return __awaiter(_this, void 0, void 0, function () { return __generator(this, function (_a) { switch (_a.label) { case 0: if (!this.recorder) { throw new Error('Cannot pause recording: no recorder'); } return [4 /*yield*/, this.recorder.pauseRecording()]; case 1: _a.sent(); this.isPaused = true; this.isRecording = false; return [2 /*return*/]; } }); }); }; this.resumeRecording = function () { return __awaiter(_this, void 0, void 0, function () { return __generator(this, function (_a) { switch (_a.label) { case 0: if (!this.recorder) { throw new Error('Cannot resume recording: no recorder'); } return [4 /*yield*/, this.recorder.resumeRecording()]; case 1: _a.sent(); this.isPaused = false; this.isRecording = true; return [2 /*return*/]; } }); }); }; this.startRecording = function () { return __awaiter(_this, void 0, void 0, function () { var _a, error_1; return __generator(this, function (_b) { switch (_b.label) { case 0: _b.trys.push([0, 2, , 3]); _a = this; return [4 /*yield*/, navigator.mediaDevices.getUserMedia({ audio: true })]; case 1: _a.stream = _b.sent(); this.recorder = new recordrtc_1.RecordRTCPromisesHandler(this.stream, { type: AUDIO_TYPE, }); this.recorder.startRecording(); this.isRecording = true; this.isStopped = false; return [3 /*break*/, 3]; case 2: error_1 = _b.sent(); this.isRecording = false; this.isStopped = true; throw new Error("Error starting recording: ".concat(error_1.message)); case 3: return [2 /*return*/]; } }); }); }; this.stopRecording = function (onFinish) { return __awaiter(_this, void 0, void 0, function () { var blob, error_2; var _a; return __generator(this, function (_b) { switch (_b.label) { case 0: if (!this.isRecording || !this.recorder) { throw new Error('Cannot stop recording: no recorder'); } _b.label = 1; case 1: _b.trys.push([1, 4, , 5]); return [4 /*yield*/, this.recorder.stopRecording()]; case 2: _b.sent(); return [4 /*yield*/, this.recorder.getBlob()]; case 3: blob = _b.sent(); this.transcribe(blob, onFinish); (_a = this.stream) === null || _a === void 0 ? void 0 : _a.getTracks().forEach(function (track) { track.stop(); }); this.recorder = null; this.stream = null; this.isRecording = false; this.isStopped = true; this.isPaused = false; return [3 /*break*/, 5]; case 4: error_2 = _b.sent(); this.isRecording = false; this.isStopped = true; throw new Error("Error stopping recording: ".concat(error_2.message)); case 5: return [2 /*return*/]; } }); }); }; this.transcribe = function (audioBlob, onFinish) { return __awaiter(_this, void 0, void 0, function () { var formData, headers, response, error_3; var _a; return __generator(this, function (_b) { switch (_b.label) { case 0: formData = new FormData(); formData.append('file', audioBlob, 'audio.wav'); formData.append('model', MODEL); headers = { Authorization: "Bearer ".concat(this.apiKey), 'Content-Type': 'multipart/form-data', }; _b.label = 1; case 1: _b.trys.push([1, 3, , 4]); return [4 /*yield*/, axios_1.default.post(TRANSCRIPTIONS_API_URL, formData, { headers: headers, })]; case 2: response = _b.sent(); onFinish(((_a = response.data) === null || _a === void 0 ? void 0 : _a.text) || ''); return [3 /*break*/, 4]; case 3: error_3 = _b.sent(); console.error('Error transcribing audio:', error_3); return [3 /*break*/, 4]; case 4: return [2 /*return*/]; } }); }); }; this.recorder = null; this.stream = null; this.isRecording = false; this.isStopped = true; this.isPaused = false; if (!apiKey) { throw new Error('API key is required'); } this.apiKey = apiKey; } return WhisperSTT; }()); exports.WhisperSTT = WhisperSTT;