feat: 실시간 음성대화 GPU 엔진(whisper+XTTS) + Ollama 쿼터 자동 폴백

- voice_engine.py: faster-whisper(STT)+XTTS-v2(TTS)를 상시 로드해 로컬에서
  서빙하는 WebSocket 엔진 (배치/스트리밍 프로토콜)
- routes-voice-realtime.ts: 브라우저 WS를 voice_engine.py로 그대로 프록시
- voice-call.js/worklets: 실시간 연속 대화 모드(VAD 기반 발화 감지,
  barge-in), 마크다운/표/URL을 정리하고 읽는 sanitizeForSpeech 포함
- tts.ts/stt.ts: xtts_gpu/whisper_gpu provider 분기 추가
- ollama-client.ts/factory.ts: Ollama Cloud 세션 쿼터 초과 시
  설정된 fallback 모델로 자동 재시도

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-09 18:05:55 +09:00
co-authored by Claude Sonnet 5
parent 94caf1c4da
commit 22c692fec3
14 changed files with 1153 additions and 14 deletions
+81
View File
@@ -0,0 +1,81 @@
// AudioWorkletProcessors for the real-time voice call mode.
// Kept deliberately dumb: capture does downsampling + chunking + RMS only,
// playback does ring-buffer PCM output only. All turn-taking / VAD / WS
// logic lives in voice-call.js on the main thread.
class MicCaptureProcessor extends AudioWorkletProcessor {
constructor() {
super();
this.targetRate = 16000;
this.srcRate = sampleRate; // AudioWorkletGlobalScope — native context rate
this.ratio = this.srcRate / this.targetRate;
this.acc = 0; // fractional resample accumulator
this.chunk = []; // pending int16 samples for the current outgoing chunk
this.chunkSamples = Math.round(this.targetRate * 0.1); // ~100ms per chunk
}
process(inputs) {
const input = inputs[0];
if (!input || !input[0]) return true;
const ch = input[0];
for (let i = 0; i < ch.length; i++) {
this.acc += 1;
if (this.acc >= this.ratio) {
this.acc -= this.ratio;
const s = Math.max(-1, Math.min(1, ch[i]));
this.chunk.push(s < 0 ? s * 0x8000 : s * 0x7fff);
if (this.chunk.length >= this.chunkSamples) {
const int16 = new Int16Array(this.chunk);
let sumSq = 0;
for (let j = 0; j < int16.length; j++) { const v = int16[j] / 32768; sumSq += v * v; }
const rms = Math.sqrt(sumSq / int16.length);
this.port.postMessage({ type: 'audio', pcm: int16, rms }, [int16.buffer]);
this.chunk = [];
}
}
}
return true;
}
}
registerProcessor('mic-capture-processor', MicCaptureProcessor);
class PcmPlayerProcessor extends AudioWorkletProcessor {
constructor() {
super();
this.queue = []; // array of Float32Array
this.qi = 0; // read offset within queue[0]
this.port.onmessage = (e) => {
const msg = e.data;
if (msg.type === 'push') {
const int16 = msg.pcm;
const f32 = new Float32Array(int16.length);
for (let i = 0; i < int16.length; i++) f32[i] = int16[i] / 32768;
this.queue.push(f32);
} else if (msg.type === 'clear') {
this.queue = [];
this.qi = 0;
}
};
}
process(_inputs, outputs) {
const out = outputs[0][0];
if (!out) return true;
let oi = 0;
while (oi < out.length) {
if (this.queue.length === 0) {
out[oi++] = 0;
continue;
}
const cur = this.queue[0];
out[oi++] = cur[this.qi++];
if (this.qi >= cur.length) {
this.queue.shift();
this.qi = 0;
}
}
return true;
}
}
registerProcessor('pcm-player-processor', PcmPlayerProcessor);
+300
View File
@@ -0,0 +1,300 @@
// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
(function () {
const RMS_SPEECH_ON = 0.020;
const RMS_SPEECH_OFF = 0.012;
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
const CHUNK_MS = 100;
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
// poisons the whole GPU context) on very short standalone text — merge short
// sentence fragments together before sending them off as one TTS request.
const MIN_TTS_CHARS = 15;
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
let captureNode = null, playerNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
let preRoll = [];
let ttsQueue = [];
let ttsBusy = false;
let currentTtsId = null;
let spokenUpTo = 0;
let pendingTts = '';
let active = false;
function setStatus(text) {
const el = document.getElementById('voice-call-status');
if (el) el.textContent = text;
}
function setCaption(text) {
const el = document.getElementById('voice-call-caption');
if (el) el.textContent = text || '';
}
window.toggleVoiceCall = function () {
if (active) voiceCallHangup();
else startVoiceCall();
};
window.voiceCallHangup = function () {
active = false;
state = 'idle';
try { ws && ws.close(); } catch {}
ws = null;
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
captureCtx = playCtx = micStream = captureNode = playerNode = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
const bar = document.getElementById('voice-call-bar');
if (bar) bar.style.display = 'none';
const btn = document.getElementById('voice-call-btn');
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
};
async function startVoiceCall() {
const btn = document.getElementById('voice-call-btn');
const bar = document.getElementById('voice-call-bar');
try {
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
} catch (e) {
alert('마이크 권한이 필요합니다: ' + e.message);
return;
}
active = true;
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
if (bar) bar.style.display = 'flex';
setStatus('통화 연결 중…');
setCaption('');
state = 'listening';
speechChunks = 0; silenceMs = 0; preRoll = [];
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
const source = captureCtx.createMediaStreamSource(micStream);
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
captureNode.port.onmessage = handleMicChunk;
source.connect(captureNode);
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
playerNode.connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
ws.binaryType = 'arraybuffer';
ws.onopen = () => setStatus('듣는 중…');
ws.onmessage = handleWsMessage;
ws.onerror = () => setStatus('연결 오류');
ws.onclose = () => { if (active) voiceCallHangup(); };
}
function handleMicChunk(e) {
if (!active) return;
const { pcm, rms } = e.data;
if (state === 'listening' || state === 'assistant_speaking') {
preRoll.push(pcm);
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
if (rms > RMS_SPEECH_ON) {
speechChunks++;
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
} else {
speechChunks = 0;
}
} else if (state === 'user_speaking') {
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
if (rms < RMS_SPEECH_OFF) {
silenceMs += CHUNK_MS;
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
} else {
silenceMs = 0;
}
}
}
function onSpeechStart() {
if (state === 'assistant_speaking') {
// barge-in: cut playback immediately and cancel whatever is still generating
ttsQueue = [];
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
}
ttsBusy = false;
currentTtsId = null;
}
state = 'user_speaking';
silenceMs = 0;
setStatus('듣는 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
for (const chunk of preRoll) ws.send(chunk.buffer);
}
preRoll = [];
}
function onSpeechEnd() {
state = 'processing';
speechChunks = 0;
setStatus('생각 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
}
}
function handleWsMessage(e) {
if (e.data instanceof ArrayBuffer) {
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
return;
}
let msg;
try { msg = JSON.parse(e.data); } catch { return; }
switch (msg.type) {
case 'stt_partial':
setCaption(msg.text || '');
break;
case 'stt_final': {
setCaption('');
const text = (msg.text || '').trim();
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
spokenUpTo = 0; pendingTts = '';
const input = document.getElementById('chat-input');
if (input) {
input.value = text;
input.dispatchEvent(new Event('input'));
}
setStatus('생각 중…');
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
break;
}
case 'tts_stream_start':
currentTtsId = msg.id;
state = 'assistant_speaking';
setStatus('말하는 중…');
break;
case 'tts_end':
ttsBusy = false;
currentTtsId = null;
pumpTtsQueue();
if (ttsQueue.length === 0) {
state = 'listening';
setStatus('듣는 중…');
}
break;
case 'error':
console.warn('[voice-call] engine error:', msg.message);
break;
}
}
function pumpTtsQueue() {
if (ttsBusy || ttsQueue.length === 0) return;
if (!ws || ws.readyState !== WebSocket.OPEN) return;
const text = ttsQueue.shift();
ttsBusy = true;
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
currentTtsId = id;
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
}
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
function sanitizeForSpeech(text) {
return text
.replace(/```[\s\S]*?```/g, ' ')
.replace(/`([^`]+)`/g, '$1')
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
// markdown table separator rows, e.g. "|---|:--:|---|"
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
.replace(/https?:\/\/\S+/g, ' ')
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
.replace(/^\s*[-*+]\s+/gm, '')
.replace(/^\s*>\s?/gm, '')
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/\*([^*]+)\*/g, '$1')
.replace(/__([^_]+)__/g, '$1')
.replace(/_([^_]+)_/g, '$1')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
.replace(/\s+/g, ' ')
.trim();
}
function enqueueSentence(text) {
text = sanitizeForSpeech((text || '').trim());
if (!text) return;
ttsQueue.push(text);
pumpTtsQueue();
}
// Accumulates sentence fragments until there's enough text to be safe to
// stream to XTTS, then enqueues the merged chunk.
function bufferSentence(sentence) {
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
enqueueSentence(pendingTts);
pendingTts = '';
}
}
function splitCompleteSentences(text) {
const sentences = [];
let start = 0;
for (let i = 0; i < text.length; i++) {
if (/[.!?\n]/.test(text[i])) {
let j = i + 1;
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
const piece = text.slice(start, j).trim();
if (piece) sentences.push(piece);
start = j;
i = j - 1;
}
}
return { sentences, consumedLength: start };
}
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;
const tail = fullPartialContent.slice(spokenUpTo);
const { sentences, consumedLength } = splitCompleteSentences(tail);
if (consumedLength > 0) {
spokenUpTo += consumedLength;
for (const s of sentences) bufferSentence(s);
}
};
// Hooked from app.js when the SSE stream for a turn finishes.
window._voiceCallOnTurnDone = function (finalText) {
if (!active) return;
const remaining = (finalText || '').slice(spokenUpTo).trim();
if (remaining) bufferSentence(remaining);
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
spokenUpTo = 0;
};
})();