feat: 실시간 음성대화 GPU 엔진(whisper+XTTS) + Ollama 쿼터 자동 폴백
- voice_engine.py: faster-whisper(STT)+XTTS-v2(TTS)를 상시 로드해 로컬에서 서빙하는 WebSocket 엔진 (배치/스트리밍 프로토콜) - routes-voice-realtime.ts: 브라우저 WS를 voice_engine.py로 그대로 프록시 - voice-call.js/worklets: 실시간 연속 대화 모드(VAD 기반 발화 감지, barge-in), 마크다운/표/URL을 정리하고 읽는 sanitizeForSpeech 포함 - tts.ts/stt.ts: xtts_gpu/whisper_gpu provider 분기 추가 - ollama-client.ts/factory.ts: Ollama Cloud 세션 쿼터 초과 시 설정된 fallback 모델로 자동 재시도 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,81 @@
|
||||
// AudioWorkletProcessors for the real-time voice call mode.
|
||||
// Kept deliberately dumb: capture does downsampling + chunking + RMS only,
|
||||
// playback does ring-buffer PCM output only. All turn-taking / VAD / WS
|
||||
// logic lives in voice-call.js on the main thread.
|
||||
|
||||
class MicCaptureProcessor extends AudioWorkletProcessor {
|
||||
constructor() {
|
||||
super();
|
||||
this.targetRate = 16000;
|
||||
this.srcRate = sampleRate; // AudioWorkletGlobalScope — native context rate
|
||||
this.ratio = this.srcRate / this.targetRate;
|
||||
this.acc = 0; // fractional resample accumulator
|
||||
this.chunk = []; // pending int16 samples for the current outgoing chunk
|
||||
this.chunkSamples = Math.round(this.targetRate * 0.1); // ~100ms per chunk
|
||||
}
|
||||
|
||||
process(inputs) {
|
||||
const input = inputs[0];
|
||||
if (!input || !input[0]) return true;
|
||||
const ch = input[0];
|
||||
|
||||
for (let i = 0; i < ch.length; i++) {
|
||||
this.acc += 1;
|
||||
if (this.acc >= this.ratio) {
|
||||
this.acc -= this.ratio;
|
||||
const s = Math.max(-1, Math.min(1, ch[i]));
|
||||
this.chunk.push(s < 0 ? s * 0x8000 : s * 0x7fff);
|
||||
if (this.chunk.length >= this.chunkSamples) {
|
||||
const int16 = new Int16Array(this.chunk);
|
||||
let sumSq = 0;
|
||||
for (let j = 0; j < int16.length; j++) { const v = int16[j] / 32768; sumSq += v * v; }
|
||||
const rms = Math.sqrt(sumSq / int16.length);
|
||||
this.port.postMessage({ type: 'audio', pcm: int16, rms }, [int16.buffer]);
|
||||
this.chunk = [];
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
registerProcessor('mic-capture-processor', MicCaptureProcessor);
|
||||
|
||||
class PcmPlayerProcessor extends AudioWorkletProcessor {
|
||||
constructor() {
|
||||
super();
|
||||
this.queue = []; // array of Float32Array
|
||||
this.qi = 0; // read offset within queue[0]
|
||||
this.port.onmessage = (e) => {
|
||||
const msg = e.data;
|
||||
if (msg.type === 'push') {
|
||||
const int16 = msg.pcm;
|
||||
const f32 = new Float32Array(int16.length);
|
||||
for (let i = 0; i < int16.length; i++) f32[i] = int16[i] / 32768;
|
||||
this.queue.push(f32);
|
||||
} else if (msg.type === 'clear') {
|
||||
this.queue = [];
|
||||
this.qi = 0;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
process(_inputs, outputs) {
|
||||
const out = outputs[0][0];
|
||||
if (!out) return true;
|
||||
let oi = 0;
|
||||
while (oi < out.length) {
|
||||
if (this.queue.length === 0) {
|
||||
out[oi++] = 0;
|
||||
continue;
|
||||
}
|
||||
const cur = this.queue[0];
|
||||
out[oi++] = cur[this.qi++];
|
||||
if (this.qi >= cur.length) {
|
||||
this.queue.shift();
|
||||
this.qi = 0;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
registerProcessor('pcm-player-processor', PcmPlayerProcessor);
|
||||
@@ -0,0 +1,300 @@
|
||||
// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
|
||||
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
|
||||
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
|
||||
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
|
||||
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
|
||||
|
||||
(function () {
|
||||
const RMS_SPEECH_ON = 0.020;
|
||||
const RMS_SPEECH_OFF = 0.012;
|
||||
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
|
||||
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
|
||||
const CHUNK_MS = 100;
|
||||
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
|
||||
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
|
||||
// poisons the whole GPU context) on very short standalone text — merge short
|
||||
// sentence fragments together before sending them off as one TTS request.
|
||||
const MIN_TTS_CHARS = 15;
|
||||
|
||||
let ws = null;
|
||||
let captureCtx = null, playCtx = null;
|
||||
let micStream = null;
|
||||
let captureNode = null, playerNode = null;
|
||||
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
|
||||
let speechChunks = 0;
|
||||
let silenceMs = 0;
|
||||
let preRoll = [];
|
||||
let ttsQueue = [];
|
||||
let ttsBusy = false;
|
||||
let currentTtsId = null;
|
||||
let spokenUpTo = 0;
|
||||
let pendingTts = '';
|
||||
let active = false;
|
||||
|
||||
function setStatus(text) {
|
||||
const el = document.getElementById('voice-call-status');
|
||||
if (el) el.textContent = text;
|
||||
}
|
||||
function setCaption(text) {
|
||||
const el = document.getElementById('voice-call-caption');
|
||||
if (el) el.textContent = text || '';
|
||||
}
|
||||
|
||||
window.toggleVoiceCall = function () {
|
||||
if (active) voiceCallHangup();
|
||||
else startVoiceCall();
|
||||
};
|
||||
|
||||
window.voiceCallHangup = function () {
|
||||
active = false;
|
||||
state = 'idle';
|
||||
try { ws && ws.close(); } catch {}
|
||||
ws = null;
|
||||
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
|
||||
try { captureCtx && captureCtx.close(); } catch {}
|
||||
try { playCtx && playCtx.close(); } catch {}
|
||||
captureCtx = playCtx = micStream = captureNode = playerNode = null;
|
||||
ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||||
const bar = document.getElementById('voice-call-bar');
|
||||
if (bar) bar.style.display = 'none';
|
||||
const btn = document.getElementById('voice-call-btn');
|
||||
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
|
||||
};
|
||||
|
||||
async function startVoiceCall() {
|
||||
const btn = document.getElementById('voice-call-btn');
|
||||
const bar = document.getElementById('voice-call-bar');
|
||||
try {
|
||||
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
|
||||
} catch (e) {
|
||||
alert('마이크 권한이 필요합니다: ' + e.message);
|
||||
return;
|
||||
}
|
||||
|
||||
active = true;
|
||||
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
|
||||
if (bar) bar.style.display = 'flex';
|
||||
setStatus('통화 연결 중…');
|
||||
setCaption('');
|
||||
state = 'listening';
|
||||
speechChunks = 0; silenceMs = 0; preRoll = [];
|
||||
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||||
|
||||
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
|
||||
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||||
const source = captureCtx.createMediaStreamSource(micStream);
|
||||
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
|
||||
captureNode.port.onmessage = handleMicChunk;
|
||||
source.connect(captureNode);
|
||||
|
||||
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
|
||||
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||||
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
|
||||
playerNode.connect(playCtx.destination);
|
||||
|
||||
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
||||
const token = getAuthToken();
|
||||
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
|
||||
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
|
||||
ws.binaryType = 'arraybuffer';
|
||||
ws.onopen = () => setStatus('듣는 중…');
|
||||
ws.onmessage = handleWsMessage;
|
||||
ws.onerror = () => setStatus('연결 오류');
|
||||
ws.onclose = () => { if (active) voiceCallHangup(); };
|
||||
}
|
||||
|
||||
function handleMicChunk(e) {
|
||||
if (!active) return;
|
||||
const { pcm, rms } = e.data;
|
||||
|
||||
if (state === 'listening' || state === 'assistant_speaking') {
|
||||
preRoll.push(pcm);
|
||||
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
|
||||
if (rms > RMS_SPEECH_ON) {
|
||||
speechChunks++;
|
||||
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
|
||||
} else {
|
||||
speechChunks = 0;
|
||||
}
|
||||
} else if (state === 'user_speaking') {
|
||||
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
|
||||
if (rms < RMS_SPEECH_OFF) {
|
||||
silenceMs += CHUNK_MS;
|
||||
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
|
||||
} else {
|
||||
silenceMs = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function onSpeechStart() {
|
||||
if (state === 'assistant_speaking') {
|
||||
// barge-in: cut playback immediately and cancel whatever is still generating
|
||||
ttsQueue = [];
|
||||
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
|
||||
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
|
||||
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
|
||||
}
|
||||
ttsBusy = false;
|
||||
currentTtsId = null;
|
||||
}
|
||||
state = 'user_speaking';
|
||||
silenceMs = 0;
|
||||
setStatus('듣는 중…');
|
||||
if (ws && ws.readyState === WebSocket.OPEN) {
|
||||
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
|
||||
for (const chunk of preRoll) ws.send(chunk.buffer);
|
||||
}
|
||||
preRoll = [];
|
||||
}
|
||||
|
||||
function onSpeechEnd() {
|
||||
state = 'processing';
|
||||
speechChunks = 0;
|
||||
setStatus('생각 중…');
|
||||
if (ws && ws.readyState === WebSocket.OPEN) {
|
||||
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
|
||||
}
|
||||
}
|
||||
|
||||
function handleWsMessage(e) {
|
||||
if (e.data instanceof ArrayBuffer) {
|
||||
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
|
||||
return;
|
||||
}
|
||||
let msg;
|
||||
try { msg = JSON.parse(e.data); } catch { return; }
|
||||
|
||||
switch (msg.type) {
|
||||
case 'stt_partial':
|
||||
setCaption(msg.text || '');
|
||||
break;
|
||||
|
||||
case 'stt_final': {
|
||||
setCaption('');
|
||||
const text = (msg.text || '').trim();
|
||||
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
|
||||
spokenUpTo = 0; pendingTts = '';
|
||||
const input = document.getElementById('chat-input');
|
||||
if (input) {
|
||||
input.value = text;
|
||||
input.dispatchEvent(new Event('input'));
|
||||
}
|
||||
setStatus('생각 중…');
|
||||
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
|
||||
break;
|
||||
}
|
||||
|
||||
case 'tts_stream_start':
|
||||
currentTtsId = msg.id;
|
||||
state = 'assistant_speaking';
|
||||
setStatus('말하는 중…');
|
||||
break;
|
||||
|
||||
case 'tts_end':
|
||||
ttsBusy = false;
|
||||
currentTtsId = null;
|
||||
pumpTtsQueue();
|
||||
if (ttsQueue.length === 0) {
|
||||
state = 'listening';
|
||||
setStatus('듣는 중…');
|
||||
}
|
||||
break;
|
||||
|
||||
case 'error':
|
||||
console.warn('[voice-call] engine error:', msg.message);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
function pumpTtsQueue() {
|
||||
if (ttsBusy || ttsQueue.length === 0) return;
|
||||
if (!ws || ws.readyState !== WebSocket.OPEN) return;
|
||||
const text = ttsQueue.shift();
|
||||
ttsBusy = true;
|
||||
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
|
||||
currentTtsId = id;
|
||||
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
|
||||
}
|
||||
|
||||
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
|
||||
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
|
||||
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
|
||||
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
|
||||
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
|
||||
function sanitizeForSpeech(text) {
|
||||
return text
|
||||
.replace(/```[\s\S]*?```/g, ' ')
|
||||
.replace(/`([^`]+)`/g, '$1')
|
||||
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
|
||||
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
|
||||
// markdown table separator rows, e.g. "|---|:--:|---|"
|
||||
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
|
||||
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
|
||||
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
|
||||
.replace(/https?:\/\/\S+/g, ' ')
|
||||
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
|
||||
.replace(/^\s*[-*+]\s+/gm, '')
|
||||
.replace(/^\s*>\s?/gm, '')
|
||||
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
||||
.replace(/\*([^*]+)\*/g, '$1')
|
||||
.replace(/__([^_]+)__/g, '$1')
|
||||
.replace(/_([^_]+)_/g, '$1')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
|
||||
function enqueueSentence(text) {
|
||||
text = sanitizeForSpeech((text || '').trim());
|
||||
if (!text) return;
|
||||
ttsQueue.push(text);
|
||||
pumpTtsQueue();
|
||||
}
|
||||
|
||||
// Accumulates sentence fragments until there's enough text to be safe to
|
||||
// stream to XTTS, then enqueues the merged chunk.
|
||||
function bufferSentence(sentence) {
|
||||
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
|
||||
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
|
||||
enqueueSentence(pendingTts);
|
||||
pendingTts = '';
|
||||
}
|
||||
}
|
||||
|
||||
function splitCompleteSentences(text) {
|
||||
const sentences = [];
|
||||
let start = 0;
|
||||
for (let i = 0; i < text.length; i++) {
|
||||
if (/[.!?\n]/.test(text[i])) {
|
||||
let j = i + 1;
|
||||
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
||||
const piece = text.slice(start, j).trim();
|
||||
if (piece) sentences.push(piece);
|
||||
start = j;
|
||||
i = j - 1;
|
||||
}
|
||||
}
|
||||
return { sentences, consumedLength: start };
|
||||
}
|
||||
|
||||
// Hooked from app.js's SSE token handler — only acts while a call is active.
|
||||
window._voiceCallOnToken = function (fullPartialContent) {
|
||||
if (!active) return;
|
||||
const tail = fullPartialContent.slice(spokenUpTo);
|
||||
const { sentences, consumedLength } = splitCompleteSentences(tail);
|
||||
if (consumedLength > 0) {
|
||||
spokenUpTo += consumedLength;
|
||||
for (const s of sentences) bufferSentence(s);
|
||||
}
|
||||
};
|
||||
|
||||
// Hooked from app.js when the SSE stream for a turn finishes.
|
||||
window._voiceCallOnTurnDone = function (finalText) {
|
||||
if (!active) return;
|
||||
const remaining = (finalText || '').slice(spokenUpTo).trim();
|
||||
if (remaining) bufferSentence(remaining);
|
||||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||||
spokenUpTo = 0;
|
||||
};
|
||||
})();
|
||||
Reference in New Issue
Block a user