Fix voice: Web Speech API for STT, WAV for TTS
- STT: Replace MediaRecorder+Whisper with browser Web Speech API (ko-KR) Whisper base model hallucinated English for Korean speech; Chrome's built-in Google STT is far more accurate for Korean - TTS: Skip ffmpeg OGG conversion, return WAV directly from Piper Avoids OGG/Opus codec compatibility issues in browsers Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -8742,15 +8742,33 @@ app.post('/api/voice/stt', async (req: express.Request, res: express.Response) =
|
||||
req.on('error', (e: any) => res.status(500).json({ success: false, error: String(e?.message || e) }));
|
||||
});
|
||||
|
||||
// ── Voice: TTS (Piper) ────────────────────────────────────────────────────
|
||||
// ── Voice: TTS (Piper → WAV) ─────────────────────────────────────────────
|
||||
app.post('/api/voice/tts', async (req: express.Request, res: express.Response) => {
|
||||
const { text } = req.body || {};
|
||||
if (!text || typeof text !== 'string') { res.status(400).json({ success: false, error: 'text required' }); return; }
|
||||
if (!isTTSAvailable()) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; }
|
||||
const cfg = (getConfig().getConfig() as any)?.voice;
|
||||
if (!cfg?.tts) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; }
|
||||
const piperPath = cfg.tts.piperPath || 'piper';
|
||||
const modelPath = cfg.tts.modelPath || '';
|
||||
const configPath = cfg.tts.configPath || '';
|
||||
if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; }
|
||||
try {
|
||||
const audio = await synthesizeSpeech(text.slice(0, 2000));
|
||||
res.set('Content-Type', 'audio/ogg');
|
||||
res.send(audio);
|
||||
const os = await import('os');
|
||||
const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`);
|
||||
const piperArgs = ['--model', modelPath, '--output_file', wavPath];
|
||||
if (configPath) piperArgs.push('--config', configPath);
|
||||
const { spawn: spawnProc } = await import('child_process');
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const piper = spawnProc(piperPath, piperArgs);
|
||||
piper.stdin.write(text.slice(0, 2000), 'utf8');
|
||||
piper.stdin.end();
|
||||
piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`)));
|
||||
piper.on('error', reject);
|
||||
});
|
||||
const wav = fs.readFileSync(wavPath);
|
||||
try { fs.unlinkSync(wavPath); } catch {}
|
||||
res.set('Content-Type', 'audio/wav');
|
||||
res.send(wav);
|
||||
} catch (e: any) {
|
||||
res.status(500).json({ success: false, error: String(e?.message || e) });
|
||||
}
|
||||
|
||||
+38
-61
@@ -8832,11 +8832,9 @@ function initLuckyModalInteraction() {
|
||||
});
|
||||
}
|
||||
|
||||
// ── Voice: STT (마이크 입력) ─────────────────────────────────────────────
|
||||
var _mediaRecorder = null;
|
||||
var _audioChunks = [];
|
||||
// ── Voice: STT (Web Speech API — Chrome 내장 Google STT) ────────────────
|
||||
var _speechRecog = null;
|
||||
var _isRecording = false;
|
||||
var _micMimeType = '';
|
||||
|
||||
function toggleMicRecording() {
|
||||
if (_isRecording) stopMicRecording();
|
||||
@@ -8844,66 +8842,45 @@ function toggleMicRecording() {
|
||||
}
|
||||
|
||||
function startMicRecording() {
|
||||
navigator.mediaDevices.getUserMedia({ audio: true, video: false })
|
||||
.then(function(stream) {
|
||||
_audioChunks = [];
|
||||
_micMimeType = MediaRecorder.isTypeSupported('audio/webm;codecs=opus') ? 'audio/webm;codecs=opus'
|
||||
: MediaRecorder.isTypeSupported('audio/webm') ? 'audio/webm'
|
||||
: MediaRecorder.isTypeSupported('audio/ogg;codecs=opus') ? 'audio/ogg;codecs=opus'
|
||||
: '';
|
||||
var opts = _micMimeType ? { mimeType: _micMimeType } : {};
|
||||
_mediaRecorder = new MediaRecorder(stream, opts);
|
||||
_mediaRecorder.ondataavailable = function(e) { if (e.data.size > 0) _audioChunks.push(e.data); };
|
||||
_mediaRecorder.onstop = function() {
|
||||
stream.getTracks().forEach(function(t) { t.stop(); });
|
||||
var blob = new Blob(_audioChunks, { type: _micMimeType || 'audio/webm' });
|
||||
sendToSTT(blob);
|
||||
};
|
||||
_mediaRecorder.start();
|
||||
_isRecording = true;
|
||||
var btn = document.getElementById('mic-btn');
|
||||
if (btn) { btn.textContent = '⏹'; btn.title = '녹음 중지 (클릭)'; btn.classList.add('recording'); }
|
||||
})
|
||||
.catch(function(e) { alert('마이크 접근 실패: ' + e.message); });
|
||||
var SpeechRecog = window.SpeechRecognition || window.webkitSpeechRecognition;
|
||||
if (!SpeechRecog) {
|
||||
alert('이 브라우저는 음성 인식을 지원하지 않습니다.\nChrome을 사용해주세요.');
|
||||
return;
|
||||
}
|
||||
_speechRecog = new SpeechRecog();
|
||||
_speechRecog.lang = 'ko-KR';
|
||||
_speechRecog.interimResults = false;
|
||||
_speechRecog.maxAlternatives = 1;
|
||||
_speechRecog.continuous = false;
|
||||
|
||||
var btn = document.getElementById('mic-btn');
|
||||
|
||||
_speechRecog.onstart = function() {
|
||||
_isRecording = true;
|
||||
if (btn) { btn.textContent = '⏹'; btn.title = '말하는 중… (클릭 시 중지)'; btn.classList.add('recording'); }
|
||||
};
|
||||
_speechRecog.onresult = function(e) {
|
||||
var transcript = e.results[0][0].transcript;
|
||||
var input = document.getElementById('chat-input');
|
||||
if (input && transcript) {
|
||||
input.value = (input.value ? input.value + ' ' : '') + transcript;
|
||||
input.focus();
|
||||
input.dispatchEvent(new Event('input'));
|
||||
}
|
||||
};
|
||||
_speechRecog.onerror = function(e) {
|
||||
if (e.error !== 'no-speech') alert('음성 인식 오류: ' + e.error);
|
||||
};
|
||||
_speechRecog.onend = function() {
|
||||
_isRecording = false;
|
||||
_speechRecog = null;
|
||||
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력'; btn.classList.remove('recording'); }
|
||||
};
|
||||
_speechRecog.start();
|
||||
}
|
||||
|
||||
function stopMicRecording() {
|
||||
if (_mediaRecorder && _isRecording) {
|
||||
_mediaRecorder.stop();
|
||||
_isRecording = false;
|
||||
var btn = document.getElementById('mic-btn');
|
||||
if (btn) { btn.textContent = '⌛'; btn.title = '인식 중…'; btn.classList.remove('recording'); }
|
||||
}
|
||||
}
|
||||
|
||||
function sendToSTT(blob) {
|
||||
var ct = _micMimeType || 'audio/webm';
|
||||
fetch('/api/voice/stt', {
|
||||
method: 'POST',
|
||||
headers: Object.assign({ 'Content-Type': ct }, authHeaders()),
|
||||
credentials: 'include',
|
||||
body: blob,
|
||||
})
|
||||
.then(function(r) { return r.json(); })
|
||||
.then(function(data) {
|
||||
var btn = document.getElementById('mic-btn');
|
||||
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력 (클릭: 시작/중지)'; }
|
||||
if (data.success && data.text) {
|
||||
var input = document.getElementById('chat-input');
|
||||
if (input) {
|
||||
input.value = (input.value ? input.value + ' ' : '') + data.text;
|
||||
input.focus();
|
||||
input.dispatchEvent(new Event('input'));
|
||||
}
|
||||
} else if (!data.success) {
|
||||
alert('음성 인식 실패: ' + (data.error || '알 수 없는 오류'));
|
||||
}
|
||||
})
|
||||
.catch(function(e) {
|
||||
var btn = document.getElementById('mic-btn');
|
||||
if (btn) { btn.textContent = '🎤'; btn.title = '음성 입력 (클릭: 시작/중지)'; }
|
||||
alert('STT 오류: ' + e.message);
|
||||
});
|
||||
if (_speechRecog) { _speechRecog.stop(); }
|
||||
}
|
||||
|
||||
// ── Voice: TTS (음성 출력) ───────────────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user