fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)

- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말)
- °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃
- 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환
- 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계
- 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합
- 소수점(27.7) 오탐 문장경계 버그 수정
- num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선)
- 대기열 위치 표시 (tts_queue 메시지)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-10 16:39:46 +09:00
co-authored by Claude Sonnet 5
parent adfa136597
commit 64f7e5ec3b
3 changed files with 450 additions and 40 deletions
+43 -2
View File
@@ -69,6 +69,25 @@ def crash_and_restart(where: str, e: Exception):
os._exit(1)
# Every TTS request (call streaming + the single-shot text button) serializes
# on the same GPU (measured: one request already saturates it, see
# project_tts_edge memory), so this counter doubles as an honest queue
# position — no separate scheduler needed.
_gpu_pending_lock = threading.Lock()
_gpu_pending = 0
def gpu_queue_enter() -> int:
global _gpu_pending
with _gpu_pending_lock:
_gpu_pending += 1
return _gpu_pending
def gpu_queue_exit():
global _gpu_pending
with _gpu_pending_lock:
_gpu_pending -= 1
def load_models(args):
log.info('Loading faster-whisper model=%s device=%s ...', args.stt_model, args.device)
from faster_whisper import WhisperModel
@@ -137,6 +156,8 @@ class Engine:
def synthesize_full(self, text: str, speed=None, num_step=None):
if num_step is None:
num_step = 16
if speed is None:
speed = 1.15
audio = self.tts_model.generate(
text=text[:4000],
language='Korean',
@@ -221,13 +242,29 @@ async def handle_tts_start(ws, engine: Engine, state: ConnectionState, msg: dict
loop.call_soon_threadsafe(queue.put_nowait, ('error', str(e)))
if is_unrecoverable_cuda_error(e):
crash_and_restart('handle_tts_start', e)
finally:
gpu_queue_exit()
position = gpu_queue_enter()
if position > 1:
# Someone else's synthesis is already running on the GPU (it's
# effectively serialized — see project_tts_edge memory); let the
# client show a queue indicator instead of silently hanging.
await ws.send(json.dumps({'type': 'tts_queue', 'aheadCount': position - 1, 'id': req_id}))
threading.Thread(target=produce, daemon=True).start()
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
try:
stream_started = False
while True:
kind, payload = await queue.get()
if kind == 'chunk':
if not stream_started:
# synthesize_stream yields the whole clip as one chunk
# (see its docstring) once GPU generation actually
# finishes, so this is the right moment to tell the
# client "now speaking" — sending it eagerly up front
# would stomp the tts_queue status above.
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
stream_started = True
await ws.send(payload)
elif kind == 'error':
await ws.send(json.dumps({'type': 'error', 'message': payload, 'id': req_id}))
@@ -268,7 +305,11 @@ async def handle_connection(ws, engine: Engine):
speed = msg.get('speed')
num_step = msg.get('num_step')
loop = asyncio.get_event_loop()
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
gpu_queue_enter()
try:
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
finally:
gpu_queue_exit()
await ws.send(json.dumps({
'type': 'tts_result',
'audioBase64': base64.b64encode(pcm).decode('ascii'),
+259 -27
View File
@@ -5697,6 +5697,7 @@ var _ttsEnabled = localStorage.getItem('ttsEnabled') !== 'false'; // default on
var _ttsAudio = null;
var _ttsBtnActive = null;
var _ttsAbort = null;
var _ttsSeqResolve = null; // resolves the currently-awaited piece's completion, so _ttsStop() can unblock the playback loop
var _appSendFn = null;
function initAppVoice(sendFn) {
@@ -5707,9 +5708,139 @@ function initAppVoice(sendFn) {
function _ttsStop() {
if (_ttsAbort) { _ttsAbort.abort(); _ttsAbort = null; }
if (_ttsAudio) { _ttsAudio.pause(); _ttsAudio = null; }
if (_ttsSeqResolve) { var r = _ttsSeqResolve; _ttsSeqResolve = null; r(); }
if (_ttsBtnActive) { _ttsBtnActive.textContent = '🔊'; _ttsBtnActive.classList.remove('playing'); _ttsBtnActive = null; }
}
// Splits raw text (before per-piece cleaning) into sentence/line-sized
// pieces synthesized and played back-to-back, instead of one big combined
// call. A single long combined call to OmniVoice measurably drops/garbles
// content (verified via STT round-trip, 2026-07-10, see project_tts_edge
// memory) — short calls are far more reliable. Splits on "\n" too, not just
// .!?, so bullet-list facts each become their own call rather than getting
// joined into one comma-separated blob (a comma pause alone wasn't enough —
// still failed reliably in testing). Tiny fragments get merged (mirrors
// voice-call.js's MIN_TTS_CHARS=15) so we don't fire a round-trip per clause.
// Splits raw text into pieces synthesized and played back-to-back, one pass:
// short sentence/line fragments accumulate into a buffer and flush once they
// cross MIN_CHARS (mirrors voice-call.js's MIN_TTS_CHARS=15, but lower — that
// threshold dates back to XTTS's streaming inference crashing on very short
// standalone text, which OmniVoice doesn't share; tested down to ~6 chars).
// Markdown table cells are the exception: each one is pushed as its own
// never-merged piece — tested 8/8 clean vs. frequent drops when cells get
// joined into one comma-list or merged across a row boundary (2026-07-10,
// e.g. "81%" vanishing, or "습도 바람 서울" nonsensically mixing header/data
// cells together). A separator row ("|---|---|") is dropped, it has no
// spoken content. Leading bullet markers get stripped per-piece too — once
// merged with plain spaces (not a real "\n"), _cleanForSpeech's ^...$/gm
// regexes only see the whole joined string as one line.
function _splitForSpeech(text) {
var out = [];
var buf = '';
var MIN_CHARS = 6;
function flush() { if (buf) { out.push(buf); buf = ''; } }
function addPiece(s, isLineBreak) {
s = s.trim();
if (!s) return;
if (/^\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?$/.test(s)) return; // separator row
var rowMatch = /^\|(.+)\|$/.exec(s);
if (rowMatch) {
flush(); // don't let a pending sentence fragment bleed into table cells
// Numbers are where the model actually loses reliability (measured
// repeatedly, project_tts_edge memory) — isolating every single cell
// fixed that but made a whole table read out painfully slowly (one
// network round-trip per cell). Words alone rarely garble, so merge
// consecutive non-numeric cells (e.g. city + weather condition) and
// only isolate cells that contain a digit — cuts call count roughly in
// half on a typical weather table while keeping the error-prone cells
// separate (reported 2026-07-10).
var cellBuf = '';
rowMatch[1].split('|').map(function(c) { return c.trim(); }).filter(Boolean).forEach(function(cell) {
if (/\d/.test(cell)) {
if (cellBuf) { out.push(cellBuf); cellBuf = ''; }
out.push(cell);
} else {
cellBuf = cellBuf ? cellBuf + ', ' + cell : cell;
}
});
if (cellBuf) out.push(cellBuf);
return;
}
var bulletStripped = s.replace(/^[-*+]\s+/, '');
if (!bulletStripped) return;
if (isLineBreak) {
// A line break (bullet item, paragraph line) is an intentional
// structural separation, same reasoning as table cells above — never
// merge it with a neighbor. Without this, a short line like "습도:
// 86%" (under MIN_CHARS) silently absorbed the next bullet line and
// the pause between them vanished (reported 2026-07-10). Plain prose
// sentences (.!? boundaries, not a line break) still merge below —
// that's normal conversational flow, not an itemized fact list.
flush();
out.push(bulletStripped);
return;
}
buf = buf ? buf + ' ' + bulletStripped : bulletStripped;
if (buf.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_CHARS) flush();
}
var start = 0;
for (var i = 0; i < text.length; i++) {
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue; // decimal point, not a boundary
if (/[.!?\n]/.test(text[i])) {
var j = i + 1;
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
addPiece(text.slice(start, j), text[i] === '\n');
start = j;
i = j - 1;
}
}
addPiece(text.slice(start), false);
flush();
return out.length ? out : [text];
}
async function _ttsPlaySequence(btn, pieces) {
for (var idx = 0; idx < pieces.length; idx++) {
if (_ttsBtnActive !== btn) return; // stopped/superseded before this piece started
var cleaned = _cleanForSpeech(pieces[idx]);
if (!cleaned) continue;
var ac = new AbortController();
_ttsAbort = ac;
var blob;
try {
var r = await fetch('/api/voice/tts', {
method: 'POST',
headers: Object.assign({ 'Content-Type': 'application/json' }, authHeaders()),
credentials: 'include',
body: JSON.stringify({ text: cleaned.slice(0, 2000) }),
signal: ac.signal,
});
if (!r.ok) throw new Error('TTS 오류 ' + r.status);
blob = await r.blob();
} catch (e) {
if (e.name === 'AbortError' || _ttsBtnActive !== btn) return;
_ttsStop();
alert('TTS 오류: ' + e.message);
return;
}
if (_ttsBtnActive !== btn) return; // superseded while this piece was fetching
var url = URL.createObjectURL(blob);
var audio = new Audio(url);
_ttsAudio = audio;
btn.textContent = '⏹';
var done = new Promise(function(resolve) {
_ttsSeqResolve = resolve;
audio.onended = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); };
audio.onerror = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); };
});
audio.play();
await done;
if (_ttsBtnActive !== btn) return; // stopped mid-playback
}
if (_ttsBtnActive === btn) _ttsStop(); // whole sequence finished naturally
}
// Shared core for playTTSText/playTTSMsg. Guards every async callback with a
// "is this still the active button?" check so a slow/stale response from a
// superseded click can never start overlapping playback (the echo bug).
@@ -5721,34 +5852,136 @@ function _ttsPlay(btn, text) {
btn.textContent = '⌛'; btn.classList.add('playing');
_ttsBtnActive = btn;
var ac = new AbortController();
_ttsAbort = ac;
fetch('/api/voice/tts', {
method: 'POST',
headers: Object.assign({ 'Content-Type': 'application/json' }, authHeaders()),
credentials: 'include',
body: JSON.stringify({ text: text.slice(0, 2000) }),
signal: ac.signal,
})
.then(function(r) { if (!r.ok) throw new Error('TTS 오류 ' + r.status); return r.blob(); })
.then(function(blob) {
if (_ttsBtnActive !== btn) return; // superseded by a newer click
var url = URL.createObjectURL(blob);
_ttsAudio = new Audio(url);
_ttsAudio.onended = function() { URL.revokeObjectURL(url); if (_ttsBtnActive === btn) _ttsStop(); };
_ttsAudio.onerror = function() { URL.revokeObjectURL(url); if (_ttsBtnActive === btn) _ttsStop(); };
btn.textContent = '⏹'; _ttsAudio.play();
})
.catch(function(e) {
if (e.name === 'AbortError' || _ttsBtnActive !== btn) return;
_ttsStop();
alert('TTS 오류: ' + e.message);
});
_ttsPlaySequence(btn, _splitForSpeech(text));
}
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
// 0-9999 (plenty for the measurement-style numbers TTS text actually
// contains — temperatures, percentages, speeds); larger numbers get an
// extra 만/억 grouping pass but aren't the target use case.
function _sinoKoreanInt(n) {
if (n === 0) return '영';
var digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
var smallUnits = ['', '십', '백', '천'];
var bigUnits = ['', '만', '억', '조'];
function fourDigit(num) {
if (num === 0) return '';
var s = '';
var ds = String(num).padStart(4, '0').split('').map(Number);
for (var i = 0; i < 4; i++) {
var d = ds[i], unit = smallUnits[3 - i];
if (d === 0) continue;
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
}
return s;
}
var groups = [];
var rem = n;
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
var result = '';
for (var g = groups.length - 1; g >= 0; g--) {
if (groups[g] === 0) continue;
result += fourDigit(groups[g]) + bigUnits[g];
}
return result || '영';
}
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
// Korean convention), "-3.2" -> "마이너스 삼점이".
function _numToHangulWord(numStr) {
var neg = numStr[0] === '-';
if (neg) numStr = numStr.slice(1);
var parts = numStr.split('.');
var word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
if (parts[1]) {
var d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
word += '점' + parts[1].split('').map(function(c) { return d[+c]; }).join('');
}
return (neg ? '마이너스 ' : '') + word;
}
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
// (일/이/삼...), specifically for clock hours and hour-durations — "9시" is
// "아홉 시", never "구시" (reported 2026-07-10). Everything else (분, 초,
// 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
var _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
function _hourReplacer(m, numStr, suffix) {
var n = parseInt(numStr, 10);
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
}
// Strips markdown, emoji and bare parens before handing text to the GPU
// voice engine (OmniVoice). Mirrors voice-call.js's sanitizeForSpeech: emoji
// get mispronounced as garbled syllables, a lone "(" makes OmniVoice
// stop generating audio entirely instead of just mispronouncing it, and a
// bare "°C"/"°F" gets read as "그램"(grams) instead of degrees.
function _cleanForSpeech(text) {
return (text || '')
.replace(/!\[.*?\]\(.*?\)/g, '')
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
.replace(/```[\s\S]*?```/g, '')
// Table handling was also never ported from voice-call.js — a leftover
// "|" right before a cell's content (e.g. "| 81% |") is yet another
// unknown-symbol case that makes OmniVoice choke and drop content
// (reported 2026-07-10, e.g. "81%" vanishing entirely from a table row).
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
.replace(/^\s*\|(.+)\|\s*$/gm, function(_m, row) { return row.split('|').map(function(c) { return c.trim(); }).filter(Boolean).join(', '); })
// Bullet markers were never stripped here (unlike voice-call.js's
// sanitizeForSpeech, which already had this) — a leftover "- " right
// after the newline-to-comma pause below reads as a rushed, breathless
// run-on (reported 2026-07-10). Must run before \n gets consumed.
.replace(/^\s*[-*+]\s+/gm, '')
.replace(/[#*`_~>]/g, ' ')
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
// Spell out common units so a bare symbol doesn't get read wrong
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
// Korean words, so OmniVoice mangles them — spell out the compass point
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
// "S" (reported 2026-07-10). Must run before the generic paren->period
// conversion below, while the parens are still intact to anchor on.
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, function(_m, dir) {
var d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
return d[dir] ? ' ' + d[dir] + ' ' : _m;
})
// Swap parens for a period rather than just erasing them — still no bare
// "(" reaches the model (the crash trigger), but a period gives the
// parenthetical aside a clearer pause than a comma did (measured: comma
// ~+0.02-0.08s over a bare space, period ~2-3x that, 2026-07-10).
.replace(/[((]\s*/g, '. ')
.replace(/\s*[))]\s*/g, '. ')
.replace(/\.\s*\./g, '.')
// Paragraph/line breaks were silently vanishing into the final \s+ ->
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
// wind speed, then a separate rain/comfort sentence) got read with zero
// pause at all — worse than the old paren-as-space bug. Give them a
// pause proportional to how strong a break they represent.
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
.replace(/(?<![.!?,:;])\n/g, ', ')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
// Hour numbers use native-Korean counting, not Sino-Korean — must run
// before the generic number pass below so "9시" doesn't become "구시".
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
// Final pass: every remaining numeral (including ones the unit
// conversions above just introduced digits next to, e.g. "27.7도")
// becomes spelled-out Hangul — this is the single biggest reliability
// win measured today (raw numerals: frequent digit swaps/drops;
// spelled-out: ~7/8 clean in repeated STT round-trip testing).
.replace(/-?\d+(?:\.\d+)?/g, function(m) { return _numToHangulWord(m); })
.replace(/\s+/g, ' ')
.trim();
}
function playTTSText(btn, text) {
text = (text||'').replace(/```[\s\S]*?```/g,'').replace(/[#*`_~>]/g,' ').replace(/\s+/g,' ').trim();
_ttsPlay(btn, text);
_ttsPlay(btn, text || '');
}
function updateTTSMode() {
@@ -5770,8 +6003,7 @@ function initTTSToggle() {
function playTTSMsg(btn, idx) {
var msg = chatHistory[idx];
if (!msg) return;
var text = (msg.content || '').replace(/!\[.*?\]\(.*?\)/g, '').replace(/\[([^\]]+)\]\([^)]+\)/g, '$1').replace(/```[\s\S]*?```/g, '').replace(/[#*`_~>]/g, ' ').replace(/\s+/g, ' ').trim();
_ttsPlay(btn, text);
_ttsPlay(btn, msg.content || '');
}
function saveLuckyModal() {
+148 -11
View File
@@ -221,6 +221,10 @@
break;
}
case 'tts_queue':
if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`);
break;
case 'tts_stream_start':
currentTtsId = msg.id;
state = 'assistant_speaking';
@@ -270,6 +274,61 @@
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
// 0-9999 (plenty for measurement-style numbers — temperatures, percentages,
// speeds); larger numbers get an extra 만/억 grouping pass but aren't the
// target use case. Mirrors app.js's _sinoKoreanInt.
function _sinoKoreanInt(n) {
if (n === 0) return '영';
const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
const smallUnits = ['', '십', '백', '천'];
const bigUnits = ['', '만', '억', '조'];
function fourDigit(num) {
if (num === 0) return '';
let s = '';
const ds = String(num).padStart(4, '0').split('').map(Number);
for (let i = 0; i < 4; i++) {
const d = ds[i], unit = smallUnits[3 - i];
if (d === 0) continue;
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
}
return s;
}
const groups = [];
let rem = n;
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
let result = '';
for (let g = groups.length - 1; g >= 0; g--) {
if (groups[g] === 0) continue;
result += fourDigit(groups[g]) + bigUnits[g];
}
return result || '영';
}
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
// Korean convention), "-3.2" -> "마이너스 삼점이".
function _numToHangulWord(numStr) {
const neg = numStr[0] === '-';
if (neg) numStr = numStr.slice(1);
const parts = numStr.split('.');
let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
if (parts[1]) {
const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
word += '점' + parts[1].split('').map((c) => d[+c]).join('');
}
return (neg ? '마이너스 ' : '') + word;
}
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
// (일/이/삼...), specifically for clock hours and hour-durations — "9시"
// is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분,
// 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
function _hourReplacer(m, numStr, suffix) {
const n = parseInt(numStr, 10);
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
}
function sanitizeForSpeech(text) {
return text
.replace(/```[\s\S]*?```/g, ' ')
@@ -281,10 +340,39 @@
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
.replace(/https?:\/\/\S+/g, ' ')
// OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees
// — the digits (even with a decimal point) come through fine once the
// symbol itself is spelled out in Korean (verified via STT round-trip,
// 2026-07-10). Order matters: °C/°F before the bare ° fallback.
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
// Spell out common units so a bare symbol doesn't get read wrong
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
// Korean words, so OmniVoice mangles them — spell out the compass point
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
// "S" (reported 2026-07-10). Must run before the generic paren->period
// conversion below, while the parens are still intact to anchor on.
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => {
const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
return d[dir] ? ' ' + d[dir] + ' ' : _m;
})
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
// "영웅(Heroic)" — and stops generating audio entirely instead of just
// mispronouncing it. Drop the parens but keep their contents.
.replace(/[()()]/g, ' ')
// mispronouncing it. Swap for a period instead of just erasing: still
// no literal "(" reaches the model, but the parenthetical aside gets a
// clearer pause than a comma did (measured: comma ~+0.02-0.08s over a
// bare space, period ~2-3x that, 2026-07-10).
.replace(/[((]\s*/g, '. ')
.replace(/\s*[))]\s*/g, '. ')
.replace(/\.\s*\./g, '.')
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
.replace(/^\s*[-*+]\s+/gm, '')
.replace(/^\s*>\s?/gm, '')
@@ -292,7 +380,22 @@
.replace(/\*([^*]+)\*/g, '$1')
.replace(/__([^_]+)__/g, '$1')
.replace(/_([^_]+)_/g, '$1')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
// Paragraph/line breaks were silently vanishing into the final \s+ ->
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
// wind speed, then a separate rain/comfort sentence) got read with zero
// pause at all — worse than the old paren-as-space bug. Must run after
// all the ^...$/gm passes above (they still need real newlines).
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
.replace(/(?<![.!?,:;])\n/g, ', ')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
// Hour numbers use native-Korean counting, not Sino-Korean — must run
// before the generic number pass below so "9시" doesn't become "구시".
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
// Final pass: every remaining numeral becomes spelled-out Hangul — the
// single biggest reliability win measured today (raw numerals: frequent
// digit swaps/drops; spelled-out: ~7/8 clean in repeated STT round-trip
// testing, 2026-07-10).
.replace(/-?\d+(?:\.\d+)?/g, (m) => _numToHangulWord(m))
.replace(/\s+/g, ' ')
.trim();
}
@@ -304,13 +407,26 @@
pumpTtsQueue();
}
function flushPendingTts() {
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
}
// Accumulates sentence fragments until there's enough text to be safe to
// stream to XTTS, then enqueues the merged chunk.
function bufferSentence(sentence) {
// stream to XTTS, then enqueues the merged chunk. A line break (bullet
// item, paragraph line) is an intentional structural separation, though —
// never merge it with a neighbor, same reasoning as app.js's text-button
// fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS)
// silently absorbed the next bullet line and the pause between them
// vanished (reported 2026-07-10).
function bufferSentence(sentence, isLineBreak) {
if (isLineBreak) {
flushPendingTts();
enqueueSentence(sentence);
return;
}
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
enqueueSentence(pendingTts);
pendingTts = '';
flushPendingTts();
}
}
@@ -318,11 +434,32 @@
const sentences = [];
let start = 0;
for (let i = 0; i < text.length; i++) {
// A "." between two digits is a decimal point (e.g. "27.7"), not a
// sentence end — without this guard, streamed numbers get sliced in
// half mid-decimal into two separate TTS calls (found 2026-07-10 while
// debugging why weather readouts sounded choppy).
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue;
if (/[.!?\n]/.test(text[i])) {
const boundaryChar = text[i];
let j = i + 1;
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
const piece = text.slice(start, j).trim();
if (piece) sentences.push(piece);
let piece = text.slice(start, j).trim();
// bufferSentence() rejoins pieces with a plain space, not a real
// "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm"
// bullet-marker regex only sees ONE line (the whole joined string)
// and misses every bullet after the first. Strip it per-piece here,
// while a real line boundary still exists to anchor on (2026-07-10).
piece = piece.replace(/^[-*+]\s+/, '');
// A "\n" boundary carries no punctuation of its own, so the .trim()
// above silently erases the pause it implied — sanitizeForSpeech's
// \n handling never gets a chance to see it since the newline is
// gone by the time this piece reaches it. Put an explicit period
// back so list items / paragraph breaks still read as a pause
// instead of running straight into the next line (2026-07-10).
if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) {
piece += '.';
}
if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' });
start = j;
i = j - 1;
}
@@ -344,7 +481,7 @@
const { sentences, consumedLength } = splitCompleteSentences(tail);
if (consumedLength > 0) {
spokenUpTo += consumedLength;
for (const s of sentences) bufferSentence(s);
for (const s of sentences) bufferSentence(s.text, s.isLineBreak);
}
};
@@ -353,7 +490,7 @@
if (!active) return;
const remaining = (finalText || '').slice(spokenUpTo).trim();
if (remaining) bufferSentence(remaining);
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
flushPendingTts();
spokenUpTo = 0;
};
})();