fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)
- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말) - °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃 - 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환 - 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계 - 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합 - 소수점(27.7) 오탐 문장경계 버그 수정 - num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선) - 대기열 위치 표시 (tts_queue 메시지) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -69,6 +69,25 @@ def crash_and_restart(where: str, e: Exception):
|
||||
os._exit(1)
|
||||
|
||||
|
||||
# Every TTS request (call streaming + the single-shot text button) serializes
|
||||
# on the same GPU (measured: one request already saturates it, see
|
||||
# project_tts_edge memory), so this counter doubles as an honest queue
|
||||
# position — no separate scheduler needed.
|
||||
_gpu_pending_lock = threading.Lock()
|
||||
_gpu_pending = 0
|
||||
|
||||
def gpu_queue_enter() -> int:
|
||||
global _gpu_pending
|
||||
with _gpu_pending_lock:
|
||||
_gpu_pending += 1
|
||||
return _gpu_pending
|
||||
|
||||
def gpu_queue_exit():
|
||||
global _gpu_pending
|
||||
with _gpu_pending_lock:
|
||||
_gpu_pending -= 1
|
||||
|
||||
|
||||
def load_models(args):
|
||||
log.info('Loading faster-whisper model=%s device=%s ...', args.stt_model, args.device)
|
||||
from faster_whisper import WhisperModel
|
||||
@@ -137,6 +156,8 @@ class Engine:
|
||||
def synthesize_full(self, text: str, speed=None, num_step=None):
|
||||
if num_step is None:
|
||||
num_step = 16
|
||||
if speed is None:
|
||||
speed = 1.15
|
||||
audio = self.tts_model.generate(
|
||||
text=text[:4000],
|
||||
language='Korean',
|
||||
@@ -221,13 +242,29 @@ async def handle_tts_start(ws, engine: Engine, state: ConnectionState, msg: dict
|
||||
loop.call_soon_threadsafe(queue.put_nowait, ('error', str(e)))
|
||||
if is_unrecoverable_cuda_error(e):
|
||||
crash_and_restart('handle_tts_start', e)
|
||||
finally:
|
||||
gpu_queue_exit()
|
||||
|
||||
position = gpu_queue_enter()
|
||||
if position > 1:
|
||||
# Someone else's synthesis is already running on the GPU (it's
|
||||
# effectively serialized — see project_tts_edge memory); let the
|
||||
# client show a queue indicator instead of silently hanging.
|
||||
await ws.send(json.dumps({'type': 'tts_queue', 'aheadCount': position - 1, 'id': req_id}))
|
||||
threading.Thread(target=produce, daemon=True).start()
|
||||
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
|
||||
try:
|
||||
stream_started = False
|
||||
while True:
|
||||
kind, payload = await queue.get()
|
||||
if kind == 'chunk':
|
||||
if not stream_started:
|
||||
# synthesize_stream yields the whole clip as one chunk
|
||||
# (see its docstring) once GPU generation actually
|
||||
# finishes, so this is the right moment to tell the
|
||||
# client "now speaking" — sending it eagerly up front
|
||||
# would stomp the tts_queue status above.
|
||||
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
|
||||
stream_started = True
|
||||
await ws.send(payload)
|
||||
elif kind == 'error':
|
||||
await ws.send(json.dumps({'type': 'error', 'message': payload, 'id': req_id}))
|
||||
@@ -268,7 +305,11 @@ async def handle_connection(ws, engine: Engine):
|
||||
speed = msg.get('speed')
|
||||
num_step = msg.get('num_step')
|
||||
loop = asyncio.get_event_loop()
|
||||
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
|
||||
gpu_queue_enter()
|
||||
try:
|
||||
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
|
||||
finally:
|
||||
gpu_queue_exit()
|
||||
await ws.send(json.dumps({
|
||||
'type': 'tts_result',
|
||||
'audioBase64': base64.b64encode(pcm).decode('ascii'),
|
||||
|
||||
Reference in New Issue
Block a user