feat: 실시간 음성대화 GPU 엔진(whisper+XTTS) + Ollama 쿼터 자동 폴백
- voice_engine.py: faster-whisper(STT)+XTTS-v2(TTS)를 상시 로드해 로컬에서 서빙하는 WebSocket 엔진 (배치/스트리밍 프로토콜) - routes-voice-realtime.ts: 브라우저 WS를 voice_engine.py로 그대로 프록시 - voice-call.js/worklets: 실시간 연속 대화 모드(VAD 기반 발화 감지, barge-in), 마크다운/표/URL을 정리하고 읽는 sanitizeForSpeech 포함 - tts.ts/stt.ts: xtts_gpu/whisper_gpu provider 분기 추가 - ollama-client.ts/factory.ts: Ollama Cloud 세션 쿼터 초과 시 설정된 fallback 모델로 자동 재시도 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,81 @@
|
||||
// AudioWorkletProcessors for the real-time voice call mode.
|
||||
// Kept deliberately dumb: capture does downsampling + chunking + RMS only,
|
||||
// playback does ring-buffer PCM output only. All turn-taking / VAD / WS
|
||||
// logic lives in voice-call.js on the main thread.
|
||||
|
||||
class MicCaptureProcessor extends AudioWorkletProcessor {
|
||||
constructor() {
|
||||
super();
|
||||
this.targetRate = 16000;
|
||||
this.srcRate = sampleRate; // AudioWorkletGlobalScope — native context rate
|
||||
this.ratio = this.srcRate / this.targetRate;
|
||||
this.acc = 0; // fractional resample accumulator
|
||||
this.chunk = []; // pending int16 samples for the current outgoing chunk
|
||||
this.chunkSamples = Math.round(this.targetRate * 0.1); // ~100ms per chunk
|
||||
}
|
||||
|
||||
process(inputs) {
|
||||
const input = inputs[0];
|
||||
if (!input || !input[0]) return true;
|
||||
const ch = input[0];
|
||||
|
||||
for (let i = 0; i < ch.length; i++) {
|
||||
this.acc += 1;
|
||||
if (this.acc >= this.ratio) {
|
||||
this.acc -= this.ratio;
|
||||
const s = Math.max(-1, Math.min(1, ch[i]));
|
||||
this.chunk.push(s < 0 ? s * 0x8000 : s * 0x7fff);
|
||||
if (this.chunk.length >= this.chunkSamples) {
|
||||
const int16 = new Int16Array(this.chunk);
|
||||
let sumSq = 0;
|
||||
for (let j = 0; j < int16.length; j++) { const v = int16[j] / 32768; sumSq += v * v; }
|
||||
const rms = Math.sqrt(sumSq / int16.length);
|
||||
this.port.postMessage({ type: 'audio', pcm: int16, rms }, [int16.buffer]);
|
||||
this.chunk = [];
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
registerProcessor('mic-capture-processor', MicCaptureProcessor);
|
||||
|
||||
class PcmPlayerProcessor extends AudioWorkletProcessor {
|
||||
constructor() {
|
||||
super();
|
||||
this.queue = []; // array of Float32Array
|
||||
this.qi = 0; // read offset within queue[0]
|
||||
this.port.onmessage = (e) => {
|
||||
const msg = e.data;
|
||||
if (msg.type === 'push') {
|
||||
const int16 = msg.pcm;
|
||||
const f32 = new Float32Array(int16.length);
|
||||
for (let i = 0; i < int16.length; i++) f32[i] = int16[i] / 32768;
|
||||
this.queue.push(f32);
|
||||
} else if (msg.type === 'clear') {
|
||||
this.queue = [];
|
||||
this.qi = 0;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
process(_inputs, outputs) {
|
||||
const out = outputs[0][0];
|
||||
if (!out) return true;
|
||||
let oi = 0;
|
||||
while (oi < out.length) {
|
||||
if (this.queue.length === 0) {
|
||||
out[oi++] = 0;
|
||||
continue;
|
||||
}
|
||||
const cur = this.queue[0];
|
||||
out[oi++] = cur[this.qi++];
|
||||
if (this.qi >= cur.length) {
|
||||
this.queue.shift();
|
||||
this.qi = 0;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
registerProcessor('pcm-player-processor', PcmPlayerProcessor);
|
||||
Reference in New Issue
Block a user