fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)

- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말)
- °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃
- 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환
- 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계
- 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합
- 소수점(27.7) 오탐 문장경계 버그 수정
- num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선)
- 대기열 위치 표시 (tts_queue 메시지)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-10 16:39:46 +09:00
co-authored by Claude Sonnet 5
parent adfa136597
commit 64f7e5ec3b
3 changed files with 450 additions and 40 deletions
+43 -2
View File
@@ -69,6 +69,25 @@ def crash_and_restart(where: str, e: Exception):
os._exit(1)
# Every TTS request (call streaming + the single-shot text button) serializes
# on the same GPU (measured: one request already saturates it, see
# project_tts_edge memory), so this counter doubles as an honest queue
# position — no separate scheduler needed.
_gpu_pending_lock = threading.Lock()
_gpu_pending = 0
def gpu_queue_enter() -> int:
global _gpu_pending
with _gpu_pending_lock:
_gpu_pending += 1
return _gpu_pending
def gpu_queue_exit():
global _gpu_pending
with _gpu_pending_lock:
_gpu_pending -= 1
def load_models(args):
log.info('Loading faster-whisper model=%s device=%s ...', args.stt_model, args.device)
from faster_whisper import WhisperModel
@@ -137,6 +156,8 @@ class Engine:
def synthesize_full(self, text: str, speed=None, num_step=None):
if num_step is None:
num_step = 16
if speed is None:
speed = 1.15
audio = self.tts_model.generate(
text=text[:4000],
language='Korean',
@@ -221,13 +242,29 @@ async def handle_tts_start(ws, engine: Engine, state: ConnectionState, msg: dict
loop.call_soon_threadsafe(queue.put_nowait, ('error', str(e)))
if is_unrecoverable_cuda_error(e):
crash_and_restart('handle_tts_start', e)
finally:
gpu_queue_exit()
position = gpu_queue_enter()
if position > 1:
# Someone else's synthesis is already running on the GPU (it's
# effectively serialized — see project_tts_edge memory); let the
# client show a queue indicator instead of silently hanging.
await ws.send(json.dumps({'type': 'tts_queue', 'aheadCount': position - 1, 'id': req_id}))
threading.Thread(target=produce, daemon=True).start()
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
try:
stream_started = False
while True:
kind, payload = await queue.get()
if kind == 'chunk':
if not stream_started:
# synthesize_stream yields the whole clip as one chunk
# (see its docstring) once GPU generation actually
# finishes, so this is the right moment to tell the
# client "now speaking" — sending it eagerly up front
# would stomp the tts_queue status above.
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
stream_started = True
await ws.send(payload)
elif kind == 'error':
await ws.send(json.dumps({'type': 'error', 'message': payload, 'id': req_id}))
@@ -268,7 +305,11 @@ async def handle_connection(ws, engine: Engine):
speed = msg.get('speed')
num_step = msg.get('num_step')
loop = asyncio.get_event_loop()
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
gpu_queue_enter()
try:
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
finally:
gpu_queue_exit()
await ws.send(json.dumps({
'type': 'tts_result',
'audioBase64': base64.b64encode(pcm).decode('ascii'),