// ── Voice: 실시간 연속 대화 모드 ─────────────────────────────────────────── // AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍. // VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고, // LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다. // barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다. (function () { const RMS_SPEECH_ON = 0.020; const RMS_SPEECH_OFF = 0.012; const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished const CHUNK_MS = 100; const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed // XTTS's streaming inference_stream() can crash (CUDA device-side assert, // poisons the whole GPU context) on very short standalone text — merge short // sentence fragments together before sending them off as one TTS request. const MIN_TTS_CHARS = 15; let ws = null; let captureCtx = null, playCtx = null; let micStream = null; let captureNode = null, playerNode = null, gainNode = null; let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking let speechChunks = 0; let silenceMs = 0; let preRoll = []; let ttsQueue = []; let ttsBusy = false; let currentTtsId = null; let spokenUpTo = 0; let pendingTts = ''; let active = false; let wakeLock = null; // The Wake Lock is auto-released whenever the tab loses visibility (screen off, // app-switch) — without it, mobile browsers throttle/suspend the background tab // and the call silently drops. Re-acquire it once the tab is visible again. async function acquireWakeLock() { if (!('wakeLock' in navigator)) return; try { wakeLock = await navigator.wakeLock.request('screen'); wakeLock.addEventListener('release', () => { wakeLock = null; }); } catch (e) { console.warn('[voice-call] wakeLock request failed:', e.message); } } document.addEventListener('visibilitychange', () => { if (active && wakeLock === null && document.visibilityState === 'visible') acquireWakeLock(); }); function setStatus(text) { const el = document.getElementById('voice-call-status'); if (el) el.textContent = text; } function setCaption(text) { const el = document.getElementById('voice-call-caption'); if (el) el.textContent = text || ''; } window.toggleVoiceCall = function () { if (active) voiceCallHangup(); else startVoiceCall(); }; window.voiceCallSetVolume = function (v) { const vol = parseFloat(v); if (!Number.isFinite(vol)) return; localStorage.setItem('voiceCallVolume', String(vol)); if (gainNode) gainNode.gain.value = vol; }; window.voiceCallHangup = function () { active = false; state = 'idle'; try { ws && ws.close(); } catch {} ws = null; try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {} try { captureCtx && captureCtx.close(); } catch {} try { playCtx && playCtx.close(); } catch {} captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null; try { wakeLock && wakeLock.release(); } catch {} wakeLock = null; ttsQueue = []; ttsBusy = false; currentTtsId = null; const bar = document.getElementById('voice-call-bar'); if (bar) bar.style.display = 'none'; const btn = document.getElementById('voice-call-btn'); if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); } }; async function startVoiceCall() { const btn = document.getElementById('voice-call-btn'); const bar = document.getElementById('voice-call-bar'); try { micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } }); } catch (e) { alert('마이크 권한이 필요합니다: ' + e.message); return; } active = true; acquireWakeLock(); if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); } if (bar) bar.style.display = 'flex'; setStatus('통화 연결 중…'); setCaption(''); state = 'listening'; speechChunks = 0; silenceMs = 0; preRoll = []; spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null; captureCtx = new (window.AudioContext || window.webkitAudioContext)(); await captureCtx.audioWorklet.addModule('voice-call-worklets.js'); const source = captureCtx.createMediaStreamSource(micStream); captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor'); captureNode.port.onmessage = handleMicChunk; source.connect(captureNode); playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 }); await playCtx.audioWorklet.addModule('voice-call-worklets.js'); playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor'); // A GainNode we control from the in-app slider — on some mobile browsers // the hardware volume rocker maps to the call/mic audio session while a // getUserMedia stream is open, not to this Web Audio output, so the phone's // own volume slider doesn't reliably affect playback here. gainNode = playCtx.createGain(); const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1'); gainNode.gain.value = savedVolume; const volumeSlider = document.getElementById('voice-call-volume'); if (volumeSlider) volumeSlider.value = String(savedVolume); playerNode.connect(gainNode).connect(playCtx.destination); const proto = location.protocol === 'https:' ? 'wss' : 'ws'; const token = getAuthToken(); const qs = token ? `?token=${encodeURIComponent(token)}` : ''; ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`); ws.binaryType = 'arraybuffer'; ws.onopen = () => setStatus('듣는 중…'); ws.onmessage = handleWsMessage; ws.onerror = () => setStatus('연결 오류'); ws.onclose = () => { if (active) voiceCallHangup(); }; } function handleMicChunk(e) { if (!active) return; const { pcm, rms } = e.data; if (state === 'listening' || state === 'assistant_speaking') { preRoll.push(pcm); if (preRoll.length > PREROLL_CHUNKS) preRoll.shift(); if (rms > RMS_SPEECH_ON) { speechChunks++; if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart(); } else { speechChunks = 0; } } else if (state === 'user_speaking') { if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer); if (rms < RMS_SPEECH_OFF) { silenceMs += CHUNK_MS; if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd(); } else { silenceMs = 0; } } } function onSpeechStart() { if (state === 'assistant_speaking') { // barge-in: cut playback immediately and cancel whatever is still generating ttsQueue = []; if (playerNode) playerNode.port.postMessage({ type: 'clear' }); if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) { ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId })); } ttsBusy = false; currentTtsId = null; } state = 'user_speaking'; silenceMs = 0; setStatus('듣는 중…'); if (ws && ws.readyState === WebSocket.OPEN) { ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' })); for (const chunk of preRoll) ws.send(chunk.buffer); } preRoll = []; } function onSpeechEnd() { state = 'processing'; speechChunks = 0; setStatus('생각 중…'); if (ws && ws.readyState === WebSocket.OPEN) { ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() })); } } function handleWsMessage(e) { if (e.data instanceof ArrayBuffer) { if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]); return; } let msg; try { msg = JSON.parse(e.data); } catch { return; } switch (msg.type) { case 'stt_partial': setCaption(msg.text || ''); break; case 'stt_final': { setCaption(''); const text = (msg.text || '').trim(); if (!text) { state = 'listening'; setStatus('듣는 중…'); break; } spokenUpTo = 0; pendingTts = ''; const input = document.getElementById('chat-input'); if (input) { input.value = text; input.dispatchEvent(new Event('input')); } setStatus('생각 중…'); setTimeout(() => (window._appSendFn || handleSendStop)(), 30); break; } case 'tts_queue': if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`); break; case 'tts_stream_start': currentTtsId = msg.id; state = 'assistant_speaking'; setStatus('말하는 중…'); break; case 'tts_end': ttsBusy = false; currentTtsId = null; pumpTtsQueue(); if (ttsQueue.length === 0) { state = 'listening'; setStatus('듣는 중…'); } break; case 'error': console.warn('[voice-call] engine error:', msg.message); // The failed request may have been the one holding ttsBusy — without // resetting it here, pumpTtsQueue() early-returns forever and the // call goes silent for the rest of the session. if (!msg.id || msg.id === currentTtsId) { ttsBusy = false; currentTtsId = null; pumpTtsQueue(); if (ttsQueue.length === 0) { state = 'listening'; setStatus('듣는 중…'); } } break; } } function pumpTtsQueue() { if (ttsBusy || ttsQueue.length === 0) return; if (!ws || ws.readyState !== WebSocket.OPEN) return; const text = ttsQueue.shift(); ttsBusy = true; const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7); currentTtsId = id; ws.send(JSON.stringify({ type: 'tts_start', text, id })); } // Strips markdown syntax, tables, URLs and emoji so raw LLM output // (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞 // etc.) doesn't get force-pronounced by XTTS under language='ko' — those // symbols were producing garbled non-Korean-sounding artifacts mid-sentence // (tables were the worst offender: pipes + dash separator rows + raw URLs). // Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles // 0-9999 (plenty for measurement-style numbers — temperatures, percentages, // speeds); larger numbers get an extra 만/억 grouping pass but aren't the // target use case. Mirrors app.js's _sinoKoreanInt. function _sinoKoreanInt(n) { if (n === 0) return '영'; const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구']; const smallUnits = ['', '십', '백', '천']; const bigUnits = ['', '만', '억', '조']; function fourDigit(num) { if (num === 0) return ''; let s = ''; const ds = String(num).padStart(4, '0').split('').map(Number); for (let i = 0; i < 4; i++) { const d = ds[i], unit = smallUnits[3 - i]; if (d === 0) continue; s += (d === 1 && unit !== '') ? unit : (digits[d] + unit); } return s; } const groups = []; let rem = n; while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); } let result = ''; for (let g = groups.length - 1; g >= 0; g--) { if (groups[g] === 0) continue; result += fourDigit(groups[g]) + bigUnits[g]; } return result || '영'; } // "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard // Korean convention), "-3.2" -> "마이너스 삼점이". function _numToHangulWord(numStr) { const neg = numStr[0] === '-'; if (neg) numStr = numStr.slice(1); const parts = numStr.split('.'); let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0); if (parts[1]) { const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구']; word += '점' + parts[1].split('').map((c) => d[+c]).join(''); } return (neg ? '마이너스 ' : '') + word; } // Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean // (일/이/삼...), specifically for clock hours and hour-durations — "9시" // is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분, // 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord. const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' }; function _hourReplacer(m, numStr, suffix) { const n = parseInt(numStr, 10); return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m; } function sanitizeForSpeech(text) { return text .replace(/```[\s\S]*?```/g, ' ') .replace(/`([^`]+)`/g, '$1') .replace(/!\[[^\]]*\]\([^)]*\)/g, ' ') .replace(/\[([^\]]+)\]\([^)]*\)/g, '$1') // markdown table separator rows, e.g. "|---|:--:|---|" .replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ') // remaining table rows: "| a | b |" -> "a, b" so cells read as a list .replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', ')) .replace(/https?:\/\/\S+/g, ' ') // OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees // — the digits (even with a decimal point) come through fine once the // symbol itself is spelled out in Korean (verified via STT round-trip, // 2026-07-10). Order matters: °C/°F before the bare ° fallback. // Range temperatures (e.g. "28~35°C") must be caught first — unlike // voice.js this file never strips "~" at all, so without this it survives // untouched into the final digit-spelling pass as a literal tilde // (reported 2026-08-08). .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도에서 $2도') .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도에서 $2도') .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°/g, '$1도에서 $2도') .replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도') .replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도') .replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도') // Spell out common units so a bare symbol doesn't get read wrong // (measured: numerals+symbols like "%"/"m/s" have a real error rate, // spelling everything in Hangul is dramatically more reliable, 2026-07-10). .replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트') .replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시') .replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초') .replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터') .replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터') // Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't // Korean words, so OmniVoice mangles them — spell out the compass point // instead. Longest abbreviations first so "SSW" doesn't partial-match as // "S" (reported 2026-07-10). Must run before the generic paren->period // conversion below, while the parens are still intact to anchor on. .replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => { const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' }; return d[dir] ? ' ' + d[dir] + ' ' : _m; }) // OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g. // "영웅(Heroic)" — and stops generating audio entirely instead of just // mispronouncing it. Swap for a period instead of just erasing: still // no literal "(" reaches the model, but the parenthetical aside gets a // clearer pause than a comma did (measured: comma ~+0.02-0.08s over a // bare space, period ~2-3x that, 2026-07-10). .replace(/[((]\s*/g, '. ') .replace(/\s*[))]\s*/g, '. ') .replace(/\.\s*\./g, '.') .replace(/^\s{0,3}#{1,6}\s+/gm, '') .replace(/^\s*[-*+]\s+/gm, '') .replace(/^\s*>\s?/gm, '') .replace(/\*\*([^*]+)\*\*/g, '$1') .replace(/\*([^*]+)\*/g, '$1') .replace(/__([^_]+)__/g, '$1') .replace(/_([^_]+)_/g, '$1') // Paragraph/line breaks were silently vanishing into the final \s+ -> // ' ' collapse below, so a blank line between two unrelated facts (e.g. // wind speed, then a separate rain/comfort sentence) got read with zero // pause at all — worse than the old paren-as-space bug. Must run after // all the ^...$/gm passes above (they still need real newlines). .replace(/(? _numToHangulWord(m)) .replace(/\s+/g, ' ') .trim(); } function enqueueSentence(text) { text = sanitizeForSpeech((text || '').trim()); if (!text) return; ttsQueue.push(text); pumpTtsQueue(); } function flushPendingTts() { if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; } } // Accumulates sentence fragments until there's enough text to be safe to // stream to XTTS, then enqueues the merged chunk. A line break (bullet // item, paragraph line) is an intentional structural separation, though — // never merge it with a neighbor, same reasoning as app.js's text-button // fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS) // silently absorbed the next bullet line and the pause between them // vanished (reported 2026-07-10). function bufferSentence(sentence, isLineBreak) { if (isLineBreak) { flushPendingTts(); enqueueSentence(sentence); return; } pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence; if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) { flushPendingTts(); } } function splitCompleteSentences(text) { const sentences = []; let start = 0; for (let i = 0; i < text.length; i++) { // A "." between two digits is a decimal point (e.g. "27.7"), not a // sentence end — without this guard, streamed numbers get sliced in // half mid-decimal into two separate TTS calls (found 2026-07-10 while // debugging why weather readouts sounded choppy). if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue; if (/[.!?\n]/.test(text[i])) { const boundaryChar = text[i]; let j = i + 1; while (j < text.length && /[\s.!?]/.test(text[j])) j++; let piece = text.slice(start, j).trim(); // bufferSentence() rejoins pieces with a plain space, not a real // "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm" // bullet-marker regex only sees ONE line (the whole joined string) // and misses every bullet after the first. Strip it per-piece here, // while a real line boundary still exists to anchor on (2026-07-10). piece = piece.replace(/^[-*+]\s+/, ''); // A "\n" boundary carries no punctuation of its own, so the .trim() // above silently erases the pause it implied — sanitizeForSpeech's // \n handling never gets a chance to see it since the newline is // gone by the time this piece reaches it. Put an explicit period // back so list items / paragraph breaks still read as a pause // instead of running straight into the next line (2026-07-10). if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) { piece += '.'; } if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' }); start = j; i = j - 1; } } return { sentences, consumedLength: start }; } // Hooked from app.js when partialContent is replaced wholesale rather than // appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an // offset into the old string and must be dropped before the new text arrives. window._voiceCallResetSpoken = function () { spokenUpTo = 0; }; // Hooked from app.js's SSE token handler — only acts while a call is active. window._voiceCallOnToken = function (fullPartialContent) { if (!active) return; const tail = fullPartialContent.slice(spokenUpTo); const { sentences, consumedLength } = splitCompleteSentences(tail); if (consumedLength > 0) { spokenUpTo += consumedLength; for (const s of sentences) bufferSentence(s.text, s.isLineBreak); } }; // Hooked from app.js when the SSE stream for a turn finishes. window._voiceCallOnTurnDone = function (finalText) { if (!active) return; const remaining = (finalText || '').slice(spokenUpTo).trim(); if (remaining) bufferSentence(remaining); flushPendingTts(); spokenUpTo = 0; }; })();