// voice.js — TTS + mic/STT extracted from helpers.js. // Korean number normalization for speech, playTTSText/playTTSMsg/initTTSToggle, // mic recording (initAppVoice, toggleMicRecording, ...). Shares _appSendFn. var _speechRecog = null; var _isRecording = false; var _micAutosend = localStorage.getItem('micAutosend') !== 'false'; // default on function updateMicAutosend() { _micAutosend = document.getElementById('mic-autosend-toggle').checked; localStorage.setItem('micAutosend', String(_micAutosend)); } function initMicAutosendToggle() { var toggle = document.getElementById('mic-autosend-toggle'); if (toggle) toggle.checked = _micAutosend; } function toggleMicRecording() { if (_isRecording) stopMicRecording(); else startMicRecording(); } function startMicRecording() { var SpeechRecog = window.SpeechRecognition || window.webkitSpeechRecognition; if (!SpeechRecog) { alert('이 브라우저는 음성 인식을 지원하지 않습니다.\nChrome을 사용해주세요.'); return; } _speechRecog = new SpeechRecog(); _speechRecog.lang = 'ko-KR'; _speechRecog.interimResults = false; _speechRecog.maxAlternatives = 1; _speechRecog.continuous = false; var btn = document.getElementById('mic-btn'); _speechRecog.onstart = function() { _isRecording = true; if (btn) { btn.textContent = '⏹'; btn.title = '말하는 중… (클릭 시 중지)'; btn.classList.add('recording'); } }; _speechRecog.onresult = function(e) { var transcript = e.results[0][0].transcript; var input = document.getElementById('chat-input'); if (input && transcript) { input.value = (input.value ? input.value + ' ' : '') + transcript; input.focus(); input.dispatchEvent(new Event('input')); if (_micAutosend) setTimeout(function() { (_appSendFn || handleSendStop)(); }, 50); } }; _speechRecog.onerror = function(e) { if (e.error !== 'no-speech') alert('음성 인식 오류: ' + e.error); }; _speechRecog.onend = function() { _isRecording = false; _speechRecog = null; if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력'; btn.classList.remove('recording'); } }; _speechRecog.start(); } function stopMicRecording() { if (_speechRecog) { _speechRecog.stop(); } } // ── Voice: TTS (음성 출력) ─────────────────────────────────────────────── var _ttsEnabled = localStorage.getItem('ttsEnabled') !== 'false'; // default on var _ttsAudio = null; var _ttsBtnActive = null; var _ttsAbort = null; var _ttsSeqResolve = null; // resolves the currently-awaited piece's completion, so _ttsStop() can unblock the playback loop var _appSendFn = null; function initAppVoice(sendFn) { _appSendFn = sendFn || null; } // Stops whatever is currently loading/playing (abort in-flight fetch, pause audio, reset button). function _ttsStop() { if (_ttsAbort) { _ttsAbort.abort(); _ttsAbort = null; } if (_ttsAudio) { _ttsAudio.pause(); _ttsAudio = null; } if (_ttsSeqResolve) { var r = _ttsSeqResolve; _ttsSeqResolve = null; r(); } if (_ttsBtnActive) { _ttsBtnActive.textContent = '🔊'; _ttsBtnActive.classList.remove('playing'); _ttsBtnActive = null; } } // Splits raw text (before per-piece cleaning) into sentence/line-sized // pieces synthesized and played back-to-back, instead of one big combined // call. A single long combined call to OmniVoice measurably drops/garbles // content (verified via STT round-trip, 2026-07-10, see project_tts_edge // memory) — short calls are far more reliable. Splits on "\n" too, not just // .!?, so bullet-list facts each become their own call rather than getting // joined into one comma-separated blob (a comma pause alone wasn't enough — // still failed reliably in testing). Tiny fragments get merged (mirrors // voice-call.js's MIN_TTS_CHARS=15) so we don't fire a round-trip per clause. // Splits raw text into pieces synthesized and played back-to-back, one pass: // short sentence/line fragments accumulate into a buffer and flush once they // cross MIN_CHARS (mirrors voice-call.js's MIN_TTS_CHARS=15, but lower — that // threshold dates back to XTTS's streaming inference crashing on very short // standalone text, which OmniVoice doesn't share; tested down to ~6 chars). // Markdown table cells are the exception: each one is pushed as its own // never-merged piece — tested 8/8 clean vs. frequent drops when cells get // joined into one comma-list or merged across a row boundary (2026-07-10, // e.g. "81%" vanishing, or "습도 바람 서울" nonsensically mixing header/data // cells together). A separator row ("|---|---|") is dropped, it has no // spoken content. Leading bullet markers get stripped per-piece too — once // merged with plain spaces (not a real "\n"), _cleanForSpeech's ^...$/gm // regexes only see the whole joined string as one line. function _splitForSpeech(text) { var out = []; var buf = ''; var MIN_CHARS = 6; function flush() { if (buf) { out.push(buf); buf = ''; } } function addPiece(s, isLineBreak) { s = s.trim(); if (!s) return; if (/^\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?$/.test(s)) return; // separator row var rowMatch = /^\|(.+)\|$/.exec(s); if (rowMatch) { flush(); // don't let a pending sentence fragment bleed into table cells // Numbers are where the model actually loses reliability (measured // repeatedly, project_tts_edge memory) — isolating every single cell // fixed that but made a whole table read out painfully slowly (one // network round-trip per cell). Words alone rarely garble, so merge // consecutive non-numeric cells (e.g. city + weather condition) and // only isolate cells that contain a digit — cuts call count roughly in // half on a typical weather table while keeping the error-prone cells // separate (reported 2026-07-10). var cellBuf = ''; rowMatch[1].split('|').map(function(c) { return c.trim(); }).filter(Boolean).forEach(function(cell) { if (/\d/.test(cell)) { if (cellBuf) { out.push(cellBuf); cellBuf = ''; } out.push(cell); } else { cellBuf = cellBuf ? cellBuf + ', ' + cell : cell; } }); if (cellBuf) out.push(cellBuf); return; } var bulletStripped = s.replace(/^[-*+]\s+/, ''); if (!bulletStripped) return; if (isLineBreak) { // A line break (bullet item, paragraph line) is an intentional // structural separation, same reasoning as table cells above — never // merge it with a neighbor. Without this, a short line like "습도: // 86%" (under MIN_CHARS) silently absorbed the next bullet line and // the pause between them vanished (reported 2026-07-10). Plain prose // sentences (.!? boundaries, not a line break) still merge below — // that's normal conversational flow, not an itemized fact list. flush(); out.push(bulletStripped); return; } buf = buf ? buf + ' ' + bulletStripped : bulletStripped; if (buf.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_CHARS) flush(); } var start = 0; for (var i = 0; i < text.length; i++) { if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue; // decimal point, not a boundary if (/[.!?\n]/.test(text[i])) { var j = i + 1; while (j < text.length && /[\s.!?]/.test(text[j])) j++; addPiece(text.slice(start, j), text[i] === '\n'); start = j; i = j - 1; } } addPiece(text.slice(start), false); flush(); return out.length ? out : [text]; } async function _ttsPlaySequence(btn, pieces) { for (var idx = 0; idx < pieces.length; idx++) { if (_ttsBtnActive !== btn) return; // stopped/superseded before this piece started var cleaned = _cleanForSpeech(pieces[idx]); if (!cleaned) continue; var ac = new AbortController(); _ttsAbort = ac; var blob; try { var r = await fetch('/api/voice/tts', { method: 'POST', headers: Object.assign({ 'Content-Type': 'application/json' }, authHeaders()), credentials: 'include', body: JSON.stringify({ text: cleaned.slice(0, 2000) }), signal: ac.signal, }); if (!r.ok) throw new Error('TTS 오류 ' + r.status); blob = await r.blob(); } catch (e) { if (e.name === 'AbortError' || _ttsBtnActive !== btn) return; _ttsStop(); alert('TTS 오류: ' + e.message); return; } if (_ttsBtnActive !== btn) return; // superseded while this piece was fetching var url = URL.createObjectURL(blob); var audio = new Audio(url); _ttsAudio = audio; btn.textContent = '⏹'; var done = new Promise(function(resolve) { _ttsSeqResolve = resolve; audio.onended = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); }; audio.onerror = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); }; }); audio.play(); await done; if (_ttsBtnActive !== btn) return; // stopped mid-playback } if (_ttsBtnActive === btn) _ttsStop(); // whole sequence finished naturally } // Shared core for playTTSText/playTTSMsg. Guards every async callback with a // "is this still the active button?" check so a slow/stale response from a // superseded click can never start overlapping playback (the echo bug). function _ttsPlay(btn, text) { if (!text) return; var wasActive = (_ttsBtnActive === btn); _ttsStop(); if (wasActive) return; // clicking the active button again = stop btn.textContent = '⌛'; btn.classList.add('playing'); _ttsBtnActive = btn; _ttsPlaySequence(btn, _splitForSpeech(text)); } // Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles // 0-9999 (plenty for the measurement-style numbers TTS text actually // contains — temperatures, percentages, speeds); larger numbers get an // extra 만/억 grouping pass but aren't the target use case. function _sinoKoreanInt(n) { if (n === 0) return '영'; var digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구']; var smallUnits = ['', '십', '백', '천']; var bigUnits = ['', '만', '억', '조']; function fourDigit(num) { if (num === 0) return ''; var s = ''; var ds = String(num).padStart(4, '0').split('').map(Number); for (var i = 0; i < 4; i++) { var d = ds[i], unit = smallUnits[3 - i]; if (d === 0) continue; s += (d === 1 && unit !== '') ? unit : (digits[d] + unit); } return s; } var groups = []; var rem = n; while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); } var result = ''; for (var g = groups.length - 1; g >= 0; g--) { if (groups[g] === 0) continue; result += fourDigit(groups[g]) + bigUnits[g]; } return result || '영'; } // "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard // Korean convention), "-3.2" -> "마이너스 삼점이". function _numToHangulWord(numStr) { var neg = numStr[0] === '-'; if (neg) numStr = numStr.slice(1); var parts = numStr.split('.'); var word = _sinoKoreanInt(parseInt(parts[0], 10) || 0); if (parts[1]) { var d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구']; word += '점' + parts[1].split('').map(function(c) { return d[+c]; }).join(''); } return (neg ? '마이너스 ' : '') + word; } // Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean // (일/이/삼...), specifically for clock hours and hour-durations — "9시" is // "아홉 시", never "구시" (reported 2026-07-10). Everything else (분, 초, // 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord. var _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' }; function _hourReplacer(m, numStr, suffix) { var n = parseInt(numStr, 10); return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m; } // Strips markdown, emoji and bare parens before handing text to the GPU // voice engine (OmniVoice). Mirrors voice-call.js's sanitizeForSpeech: emoji // get mispronounced as garbled syllables, a lone "(" makes OmniVoice // stop generating audio entirely instead of just mispronouncing it, and a // bare "°C"/"°F" gets read as "그램"(grams) instead of degrees. function _cleanForSpeech(text) { return (text || '') .replace(/!\[.*?\]\(.*?\)/g, '') .replace(/\[([^\]]+)\]\([^)]+\)/g, '$1') .replace(/```[\s\S]*?```/g, '') // Table handling was also never ported from voice-call.js — a leftover // "|" right before a cell's content (e.g. "| 81% |") is yet another // unknown-symbol case that makes OmniVoice choke and drop content // (reported 2026-07-10, e.g. "81%" vanishing entirely from a table row). .replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ') .replace(/^\s*\|(.+)\|\s*$/gm, function(_m, row) { return row.split('|').map(function(c) { return c.trim(); }).filter(Boolean).join(', '); }) // Bullet markers were never stripped here (unlike voice-call.js's // sanitizeForSpeech, which already had this) — a leftover "- " right // after the newline-to-comma pause below reads as a rushed, breathless // run-on (reported 2026-07-10). Must run before \n gets consumed. .replace(/^\s*[-*+]\s+/gm, '') // Range temperatures (e.g. "28~35°C") must be caught before the generic // symbol strip below eats the "~" — otherwise the first number loses its // unit entirely and reads as a disconnected "28 섭씨 35도" (reported 2026-08-08). .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도에서 $2도') .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도에서 $2도') .replace(/(\d+(?:\.\d+)?)\s*~\s*(\d+(?:\.\d+)?)\s*°/g, '$1도에서 $2도') .replace(/[#*`_~>]/g, ' ') .replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도') .replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도') .replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도') // Spell out common units so a bare symbol doesn't get read wrong // (measured: numerals+symbols like "%"/"m/s" have a real error rate, // spelling everything in Hangul is dramatically more reliable, 2026-07-10). .replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트') .replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시') .replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초') .replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터') .replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터') // Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't // Korean words, so OmniVoice mangles them — spell out the compass point // instead. Longest abbreviations first so "SSW" doesn't partial-match as // "S" (reported 2026-07-10). Must run before the generic paren->period // conversion below, while the parens are still intact to anchor on. .replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, function(_m, dir) { var d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' }; return d[dir] ? ' ' + d[dir] + ' ' : _m; }) // Swap parens for a period rather than just erasing them — still no bare // "(" reaches the model (the crash trigger), but a period gives the // parenthetical aside a clearer pause than a comma did (measured: comma // ~+0.02-0.08s over a bare space, period ~2-3x that, 2026-07-10). .replace(/[((]\s*/g, '. ') .replace(/\s*[))]\s*/g, '. ') .replace(/\.\s*\./g, '.') // Paragraph/line breaks were silently vanishing into the final \s+ -> // ' ' collapse below, so a blank line between two unrelated facts (e.g. // wind speed, then a separate rain/comfort sentence) got read with zero // pause at all — worse than the old paren-as-space bug. Give them a // pause proportional to how strong a break they represent. .replace(/(?