74 lines
3.5 KiB
JavaScript
74 lines
3.5 KiB
JavaScript
import { EventEmitter } from 'node:events';
|
|
import { acquireQvac, releaseQvac, loadAuxiliaryModel, unloadAuxiliaryModel, qvacSdk, withQvacMaster, assertSdkVersion } from './qvac-master.js';
|
|
|
|
export function pcmS16le(samples, sampleRate = 44_100) {
|
|
if (samples instanceof Int16Array) return { samples, sampleRate };
|
|
const bytes = toUint8(samples);
|
|
const even = bytes.byteLength - (bytes.byteLength % 2);
|
|
const copy = bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + even);
|
|
return { samples: new Int16Array(copy), sampleRate };
|
|
}
|
|
|
|
function toUint8(samples) {
|
|
if (!samples) return new Uint8Array(0);
|
|
if (samples instanceof ArrayBuffer) return new Uint8Array(samples);
|
|
if (ArrayBuffer.isView(samples)) return new Uint8Array(samples.buffer, samples.byteOffset, samples.byteLength);
|
|
if (typeof samples === 'string') return Uint8Array.from(Buffer.from(samples));
|
|
if (Array.isArray(samples) || samples.length != null) return Uint8Array.from(Buffer.from(samples));
|
|
return new Uint8Array(0);
|
|
}
|
|
|
|
export class QvacVoiceAdapter extends EventEmitter {
|
|
constructor({ asrModel = process.env.JARVIS_ASR_MODEL || 'WHISPER_TINY', ttsModel = process.env.JARVIS_TTS_MODEL || 'TTS_EN_SUPERTONIC_Q8_0' } = {}) {
|
|
super(); this.asrModel = asrModel; this.ttsModel = ttsModel; this.asrId = null; this.ttsId = null; this.asrSession = null; this.acquired = false;
|
|
}
|
|
|
|
async start() {
|
|
if (this.acquired) return;
|
|
assertSdkVersion();
|
|
await acquireQvac(); this.acquired = true;
|
|
try {
|
|
this.asrId = await loadAuxiliaryModel(this.asrModel, { vadModelSrc: 'VAD_SILERO_5_1_2', audio_format: 's16le', language: 'en', no_timestamps: true, vad_params: { threshold: 0.6, min_speech_duration_ms: 300, min_silence_duration_ms: 700, max_speech_duration_s: 15, speech_pad_ms: 200 } }, 'whisper');
|
|
this.ttsId = await loadAuxiliaryModel(this.ttsModel, { ttsEngine: 'supertonic', language: 'en', voice: 'F1', ttsSpeed: 1.05, ttsNumInferenceSteps: 5 }, 'tts');
|
|
} catch (error) { await this.stop(); throw error; }
|
|
}
|
|
|
|
writeAudio(chunk) { this.asrSession?.write(Buffer.from(chunk)); }
|
|
async transcribeAudio(audio) {
|
|
const sdk = await qvacSdk();
|
|
return withQvacMaster(async () => {
|
|
const session = await sdk.transcribeStream({ modelId: this.asrId, emitVadEvents: true });
|
|
session.write(Buffer.from(audio)); session.end();
|
|
const parts = [];
|
|
for await (const event of session) parts.push(typeof event === 'string' ? event : event?.text || '');
|
|
return parts.join(' ').trim();
|
|
});
|
|
}
|
|
async *transcripts() { if (!this.asrSession) throw new Error('voice adapter is not started'); yield* this.asrSession; }
|
|
endAudio() { this.asrSession?.end(); }
|
|
|
|
async speak(text) {
|
|
const sdk = await qvacSdk();
|
|
const samples = await withQvacMaster(async () => {
|
|
const result = await sdk.textToSpeech({ modelId: this.ttsId, text: String(text), inputType: 'text', stream: false });
|
|
if (result?.buffer != null) return await result.buffer;
|
|
if (result?.bufferStream) {
|
|
const chunks = [];
|
|
for await (const chunk of result.bufferStream) chunks.push(Buffer.from(chunk));
|
|
return Buffer.concat(chunks);
|
|
}
|
|
return result;
|
|
});
|
|
return pcmS16le(samples, 44_100);
|
|
}
|
|
|
|
async stop() {
|
|
try { this.asrSession?.destroy?.(); } catch {}
|
|
this.asrSession = null;
|
|
if (this.ttsId) await unloadAuxiliaryModel(this.ttsId).catch(() => {});
|
|
if (this.asrId) await unloadAuxiliaryModel(this.asrId).catch(() => {});
|
|
this.ttsId = this.asrId = null;
|
|
if (this.acquired) { this.acquired = false; releaseQvac(); }
|
|
}
|
|
}
|