OmniVoice가 "(" 문자(특히 한글에 바로 붙은 경우, 예: 영웅(Heroic))를 합성하다
멈춰버리는 문제가 있어 sanitizeForSpeech에서 괄호를 제거하도록 변경.
또한 TTS 엔진이 합성 실패 시 보내는 error 메시지를 클라이언트가 로그만 찍고
ttsBusy를 리셋하지 않아, 이후 통화 내내 TTS가 먹통이 되는 버그를 tts_end와
동일하게 큐를 복구하도록 수정.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
360 lines
14 KiB
JavaScript
360 lines
14 KiB
JavaScript
// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
|
|
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
|
|
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
|
|
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
|
|
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
|
|
|
|
(function () {
|
|
const RMS_SPEECH_ON = 0.020;
|
|
const RMS_SPEECH_OFF = 0.012;
|
|
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
|
|
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
|
|
const CHUNK_MS = 100;
|
|
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
|
|
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
|
|
// poisons the whole GPU context) on very short standalone text — merge short
|
|
// sentence fragments together before sending them off as one TTS request.
|
|
const MIN_TTS_CHARS = 15;
|
|
|
|
let ws = null;
|
|
let captureCtx = null, playCtx = null;
|
|
let micStream = null;
|
|
let captureNode = null, playerNode = null, gainNode = null;
|
|
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
|
|
let speechChunks = 0;
|
|
let silenceMs = 0;
|
|
let preRoll = [];
|
|
let ttsQueue = [];
|
|
let ttsBusy = false;
|
|
let currentTtsId = null;
|
|
let spokenUpTo = 0;
|
|
let pendingTts = '';
|
|
let active = false;
|
|
let wakeLock = null;
|
|
|
|
// The Wake Lock is auto-released whenever the tab loses visibility (screen off,
|
|
// app-switch) — without it, mobile browsers throttle/suspend the background tab
|
|
// and the call silently drops. Re-acquire it once the tab is visible again.
|
|
async function acquireWakeLock() {
|
|
if (!('wakeLock' in navigator)) return;
|
|
try {
|
|
wakeLock = await navigator.wakeLock.request('screen');
|
|
wakeLock.addEventListener('release', () => { wakeLock = null; });
|
|
} catch (e) {
|
|
console.warn('[voice-call] wakeLock request failed:', e.message);
|
|
}
|
|
}
|
|
document.addEventListener('visibilitychange', () => {
|
|
if (active && wakeLock === null && document.visibilityState === 'visible') acquireWakeLock();
|
|
});
|
|
|
|
function setStatus(text) {
|
|
const el = document.getElementById('voice-call-status');
|
|
if (el) el.textContent = text;
|
|
}
|
|
function setCaption(text) {
|
|
const el = document.getElementById('voice-call-caption');
|
|
if (el) el.textContent = text || '';
|
|
}
|
|
|
|
window.toggleVoiceCall = function () {
|
|
if (active) voiceCallHangup();
|
|
else startVoiceCall();
|
|
};
|
|
|
|
window.voiceCallSetVolume = function (v) {
|
|
const vol = parseFloat(v);
|
|
if (!Number.isFinite(vol)) return;
|
|
localStorage.setItem('voiceCallVolume', String(vol));
|
|
if (gainNode) gainNode.gain.value = vol;
|
|
};
|
|
|
|
window.voiceCallHangup = function () {
|
|
active = false;
|
|
state = 'idle';
|
|
try { ws && ws.close(); } catch {}
|
|
ws = null;
|
|
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
|
|
try { captureCtx && captureCtx.close(); } catch {}
|
|
try { playCtx && playCtx.close(); } catch {}
|
|
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
|
|
try { wakeLock && wakeLock.release(); } catch {}
|
|
wakeLock = null;
|
|
ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
|
const bar = document.getElementById('voice-call-bar');
|
|
if (bar) bar.style.display = 'none';
|
|
const btn = document.getElementById('voice-call-btn');
|
|
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
|
|
};
|
|
|
|
async function startVoiceCall() {
|
|
const btn = document.getElementById('voice-call-btn');
|
|
const bar = document.getElementById('voice-call-bar');
|
|
try {
|
|
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
|
|
} catch (e) {
|
|
alert('마이크 권한이 필요합니다: ' + e.message);
|
|
return;
|
|
}
|
|
|
|
active = true;
|
|
acquireWakeLock();
|
|
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
|
|
if (bar) bar.style.display = 'flex';
|
|
setStatus('통화 연결 중…');
|
|
setCaption('');
|
|
state = 'listening';
|
|
speechChunks = 0; silenceMs = 0; preRoll = [];
|
|
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
|
|
|
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
|
|
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
|
|
const source = captureCtx.createMediaStreamSource(micStream);
|
|
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
|
|
captureNode.port.onmessage = handleMicChunk;
|
|
source.connect(captureNode);
|
|
|
|
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
|
|
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
|
|
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
|
|
// A GainNode we control from the in-app slider — on some mobile browsers
|
|
// the hardware volume rocker maps to the call/mic audio session while a
|
|
// getUserMedia stream is open, not to this Web Audio output, so the phone's
|
|
// own volume slider doesn't reliably affect playback here.
|
|
gainNode = playCtx.createGain();
|
|
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
|
|
gainNode.gain.value = savedVolume;
|
|
const volumeSlider = document.getElementById('voice-call-volume');
|
|
if (volumeSlider) volumeSlider.value = String(savedVolume);
|
|
playerNode.connect(gainNode).connect(playCtx.destination);
|
|
|
|
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
|
const token = getAuthToken();
|
|
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
|
|
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
|
|
ws.binaryType = 'arraybuffer';
|
|
ws.onopen = () => setStatus('듣는 중…');
|
|
ws.onmessage = handleWsMessage;
|
|
ws.onerror = () => setStatus('연결 오류');
|
|
ws.onclose = () => { if (active) voiceCallHangup(); };
|
|
}
|
|
|
|
function handleMicChunk(e) {
|
|
if (!active) return;
|
|
const { pcm, rms } = e.data;
|
|
|
|
if (state === 'listening' || state === 'assistant_speaking') {
|
|
preRoll.push(pcm);
|
|
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
|
|
if (rms > RMS_SPEECH_ON) {
|
|
speechChunks++;
|
|
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
|
|
} else {
|
|
speechChunks = 0;
|
|
}
|
|
} else if (state === 'user_speaking') {
|
|
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
|
|
if (rms < RMS_SPEECH_OFF) {
|
|
silenceMs += CHUNK_MS;
|
|
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
|
|
} else {
|
|
silenceMs = 0;
|
|
}
|
|
}
|
|
}
|
|
|
|
function onSpeechStart() {
|
|
if (state === 'assistant_speaking') {
|
|
// barge-in: cut playback immediately and cancel whatever is still generating
|
|
ttsQueue = [];
|
|
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
|
|
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
|
|
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
|
|
}
|
|
ttsBusy = false;
|
|
currentTtsId = null;
|
|
}
|
|
state = 'user_speaking';
|
|
silenceMs = 0;
|
|
setStatus('듣는 중…');
|
|
if (ws && ws.readyState === WebSocket.OPEN) {
|
|
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
|
|
for (const chunk of preRoll) ws.send(chunk.buffer);
|
|
}
|
|
preRoll = [];
|
|
}
|
|
|
|
function onSpeechEnd() {
|
|
state = 'processing';
|
|
speechChunks = 0;
|
|
setStatus('생각 중…');
|
|
if (ws && ws.readyState === WebSocket.OPEN) {
|
|
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
|
|
}
|
|
}
|
|
|
|
function handleWsMessage(e) {
|
|
if (e.data instanceof ArrayBuffer) {
|
|
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
|
|
return;
|
|
}
|
|
let msg;
|
|
try { msg = JSON.parse(e.data); } catch { return; }
|
|
|
|
switch (msg.type) {
|
|
case 'stt_partial':
|
|
setCaption(msg.text || '');
|
|
break;
|
|
|
|
case 'stt_final': {
|
|
setCaption('');
|
|
const text = (msg.text || '').trim();
|
|
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
|
|
spokenUpTo = 0; pendingTts = '';
|
|
const input = document.getElementById('chat-input');
|
|
if (input) {
|
|
input.value = text;
|
|
input.dispatchEvent(new Event('input'));
|
|
}
|
|
setStatus('생각 중…');
|
|
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
|
|
break;
|
|
}
|
|
|
|
case 'tts_stream_start':
|
|
currentTtsId = msg.id;
|
|
state = 'assistant_speaking';
|
|
setStatus('말하는 중…');
|
|
break;
|
|
|
|
case 'tts_end':
|
|
ttsBusy = false;
|
|
currentTtsId = null;
|
|
pumpTtsQueue();
|
|
if (ttsQueue.length === 0) {
|
|
state = 'listening';
|
|
setStatus('듣는 중…');
|
|
}
|
|
break;
|
|
|
|
case 'error':
|
|
console.warn('[voice-call] engine error:', msg.message);
|
|
// The failed request may have been the one holding ttsBusy — without
|
|
// resetting it here, pumpTtsQueue() early-returns forever and the
|
|
// call goes silent for the rest of the session.
|
|
if (!msg.id || msg.id === currentTtsId) {
|
|
ttsBusy = false;
|
|
currentTtsId = null;
|
|
pumpTtsQueue();
|
|
if (ttsQueue.length === 0) {
|
|
state = 'listening';
|
|
setStatus('듣는 중…');
|
|
}
|
|
}
|
|
break;
|
|
}
|
|
}
|
|
|
|
function pumpTtsQueue() {
|
|
if (ttsBusy || ttsQueue.length === 0) return;
|
|
if (!ws || ws.readyState !== WebSocket.OPEN) return;
|
|
const text = ttsQueue.shift();
|
|
ttsBusy = true;
|
|
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
|
|
currentTtsId = id;
|
|
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
|
|
}
|
|
|
|
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
|
|
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
|
|
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
|
|
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
|
|
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
|
|
function sanitizeForSpeech(text) {
|
|
return text
|
|
.replace(/```[\s\S]*?```/g, ' ')
|
|
.replace(/`([^`]+)`/g, '$1')
|
|
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
|
|
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
|
|
// markdown table separator rows, e.g. "|---|:--:|---|"
|
|
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
|
|
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
|
|
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
|
|
.replace(/https?:\/\/\S+/g, ' ')
|
|
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
|
|
// "영웅(Heroic)" — and stops generating audio entirely instead of just
|
|
// mispronouncing it. Drop the parens but keep their contents.
|
|
.replace(/[()()]/g, ' ')
|
|
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
|
|
.replace(/^\s*[-*+]\s+/gm, '')
|
|
.replace(/^\s*>\s?/gm, '')
|
|
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
|
.replace(/\*([^*]+)\*/g, '$1')
|
|
.replace(/__([^_]+)__/g, '$1')
|
|
.replace(/_([^_]+)_/g, '$1')
|
|
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
|
|
.replace(/\s+/g, ' ')
|
|
.trim();
|
|
}
|
|
|
|
function enqueueSentence(text) {
|
|
text = sanitizeForSpeech((text || '').trim());
|
|
if (!text) return;
|
|
ttsQueue.push(text);
|
|
pumpTtsQueue();
|
|
}
|
|
|
|
// Accumulates sentence fragments until there's enough text to be safe to
|
|
// stream to XTTS, then enqueues the merged chunk.
|
|
function bufferSentence(sentence) {
|
|
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
|
|
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
|
|
enqueueSentence(pendingTts);
|
|
pendingTts = '';
|
|
}
|
|
}
|
|
|
|
function splitCompleteSentences(text) {
|
|
const sentences = [];
|
|
let start = 0;
|
|
for (let i = 0; i < text.length; i++) {
|
|
if (/[.!?\n]/.test(text[i])) {
|
|
let j = i + 1;
|
|
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
|
const piece = text.slice(start, j).trim();
|
|
if (piece) sentences.push(piece);
|
|
start = j;
|
|
i = j - 1;
|
|
}
|
|
}
|
|
return { sentences, consumedLength: start };
|
|
}
|
|
|
|
// Hooked from app.js when partialContent is replaced wholesale rather than
|
|
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
|
|
// offset into the old string and must be dropped before the new text arrives.
|
|
window._voiceCallResetSpoken = function () {
|
|
spokenUpTo = 0;
|
|
};
|
|
|
|
// Hooked from app.js's SSE token handler — only acts while a call is active.
|
|
window._voiceCallOnToken = function (fullPartialContent) {
|
|
if (!active) return;
|
|
const tail = fullPartialContent.slice(spokenUpTo);
|
|
const { sentences, consumedLength } = splitCompleteSentences(tail);
|
|
if (consumedLength > 0) {
|
|
spokenUpTo += consumedLength;
|
|
for (const s of sentences) bufferSentence(s);
|
|
}
|
|
};
|
|
|
|
// Hooked from app.js when the SSE stream for a turn finishes.
|
|
window._voiceCallOnTurnDone = function (finalText) {
|
|
if (!active) return;
|
|
const remaining = (finalText || '').slice(spokenUpTo).trim();
|
|
if (remaining) bufferSentence(remaining);
|
|
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
|
spokenUpTo = 0;
|
|
};
|
|
})();
|