Fix voice: Web Speech API for STT, WAV for TTS

- STT: Replace MediaRecorder+Whisper with browser Web Speech API (ko-KR)
  Whisper base model hallucinated English for Korean speech; Chrome's
  built-in Google STT is far more accurate for Korean
- TTS: Skip ffmpeg OGG conversion, return WAV directly from Piper
  Avoids OGG/Opus codec compatibility issues in browsers

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-05-19 18:15:41 +09:00
co-authored by Claude Sonnet 4.6
parent de18ddc9f6
commit cd48a8ebe0
2 changed files with 61 additions and 66 deletions
+23 -5
View File
@@ -8742,15 +8742,33 @@ app.post('/api/voice/stt', async (req: express.Request, res: express.Response) =
req.on('error', (e: any) => res.status(500).json({ success: false, error: String(e?.message || e) }));
});
// ── Voice: TTS (Piper) ────────────────────────────────────────────────────
// ── Voice: TTS (Piper → WAV) ─────────────────────────────────────────────
app.post('/api/voice/tts', async (req: express.Request, res: express.Response) => {
const { text } = req.body || {};
if (!text || typeof text !== 'string') { res.status(400).json({ success: false, error: 'text required' }); return; }
if (!isTTSAvailable()) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; }
const cfg = (getConfig().getConfig() as any)?.voice;
if (!cfg?.tts) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; }
const piperPath = cfg.tts.piperPath || 'piper';
const modelPath = cfg.tts.modelPath || '';
const configPath = cfg.tts.configPath || '';
if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; }
try {
const audio = await synthesizeSpeech(text.slice(0, 2000));
res.set('Content-Type', 'audio/ogg');
res.send(audio);
const os = await import('os');
const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`);
const piperArgs = ['--model', modelPath, '--output_file', wavPath];
if (configPath) piperArgs.push('--config', configPath);
const { spawn: spawnProc } = await import('child_process');
await new Promise<void>((resolve, reject) => {
const piper = spawnProc(piperPath, piperArgs);
piper.stdin.write(text.slice(0, 2000), 'utf8');
piper.stdin.end();
piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`)));
piper.on('error', reject);
});
const wav = fs.readFileSync(wavPath);
try { fs.unlinkSync(wavPath); } catch {}
res.set('Content-Type', 'audio/wav');
res.send(wav);
} catch (e: any) {
res.status(500).json({ success: false, error: String(e?.message || e) });
}
+38 -61
View File
@@ -8832,11 +8832,9 @@ function initLuckyModalInteraction() {
});
}
// ── Voice: STT (마이크 입력) ─────────────────────────────────────────────
var _mediaRecorder = null;
var _audioChunks = [];
// ── Voice: STT (Web Speech API — Chrome 내장 Google STT) ────────────────
var _speechRecog = null;
var _isRecording = false;
var _micMimeType = '';
function toggleMicRecording() {
if (_isRecording) stopMicRecording();
@@ -8844,66 +8842,45 @@ function toggleMicRecording() {
}
function startMicRecording() {
navigator.mediaDevices.getUserMedia({ audio: true, video: false })
.then(function(stream) {
_audioChunks = [];
_micMimeType = MediaRecorder.isTypeSupported('audio/webm;codecs=opus') ? 'audio/webm;codecs=opus'
: MediaRecorder.isTypeSupported('audio/webm') ? 'audio/webm'
: MediaRecorder.isTypeSupported('audio/ogg;codecs=opus') ? 'audio/ogg;codecs=opus'
: '';
var opts = _micMimeType ? { mimeType: _micMimeType } : {};
_mediaRecorder = new MediaRecorder(stream, opts);
_mediaRecorder.ondataavailable = function(e) { if (e.data.size > 0) _audioChunks.push(e.data); };
_mediaRecorder.onstop = function() {
stream.getTracks().forEach(function(t) { t.stop(); });
var blob = new Blob(_audioChunks, { type: _micMimeType || 'audio/webm' });
sendToSTT(blob);
};
_mediaRecorder.start();
_isRecording = true;
var btn = document.getElementById('mic-btn');
if (btn) { btn.textContent = '⏹'; btn.title = '녹음 중지 (클릭)'; btn.classList.add('recording'); }
})
.catch(function(e) { alert('마이크 접근 실패: ' + e.message); });
var SpeechRecog = window.SpeechRecognition || window.webkitSpeechRecognition;
if (!SpeechRecog) {
alert('이 브라우저는 음성 인식을 지원하지 않습니다.\nChrome을 사용해주세요.');
return;
}
_speechRecog = new SpeechRecog();
_speechRecog.lang = 'ko-KR';
_speechRecog.interimResults = false;
_speechRecog.maxAlternatives = 1;
_speechRecog.continuous = false;
var btn = document.getElementById('mic-btn');
_speechRecog.onstart = function() {
_isRecording = true;
if (btn) { btn.textContent = '⏹'; btn.title = '말하는 중… (클릭 시 중지)'; btn.classList.add('recording'); }
};
_speechRecog.onresult = function(e) {
var transcript = e.results[0][0].transcript;
var input = document.getElementById('chat-input');
if (input && transcript) {
input.value = (input.value ? input.value + ' ' : '') + transcript;
input.focus();
input.dispatchEvent(new Event('input'));
}
};
_speechRecog.onerror = function(e) {
if (e.error !== 'no-speech') alert('음성 인식 오류: ' + e.error);
};
_speechRecog.onend = function() {
_isRecording = false;
_speechRecog = null;
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력'; btn.classList.remove('recording'); }
};
_speechRecog.start();
}
function stopMicRecording() {
if (_mediaRecorder && _isRecording) {
_mediaRecorder.stop();
_isRecording = false;
var btn = document.getElementById('mic-btn');
if (btn) { btn.textContent = '⌛'; btn.title = '인식 중…'; btn.classList.remove('recording'); }
}
}
function sendToSTT(blob) {
var ct = _micMimeType || 'audio/webm';
fetch('/api/voice/stt', {
method: 'POST',
headers: Object.assign({ 'Content-Type': ct }, authHeaders()),
credentials: 'include',
body: blob,
})
.then(function(r) { return r.json(); })
.then(function(data) {
var btn = document.getElementById('mic-btn');
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력 (클릭: 시작/중지)'; }
if (data.success && data.text) {
var input = document.getElementById('chat-input');
if (input) {
input.value = (input.value ? input.value + ' ' : '') + data.text;
input.focus();
input.dispatchEvent(new Event('input'));
}
} else if (!data.success) {
alert('음성 인식 실패: ' + (data.error || '알 수 없는 오류'));
}
})
.catch(function(e) {
var btn = document.getElementById('mic-btn');
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력 (클릭: 시작/중지)'; }
alert('STT 오류: ' + e.message);
});
if (_speechRecog) { _speechRecog.stop(); }
}
// ── Voice: TTS (음성 출력) ───────────────────────────────────────────────