#!/bin/bash # SmallClaw Voice Dependencies Installer (fixed) # Run: bash scripts/install-voice-deps.sh set -e echo "=== SmallClaw Voice Dependencies Installer ===" echo "" # 1. ffmpeg echo "[1/5] Installing ffmpeg..." if command -v ffmpeg &>/dev/null; then echo " ffmpeg already installed: $(ffmpeg -version 2>&1 | head -1)" else sudo apt update && sudo apt install -y ffmpeg echo " ffmpeg installed: $(ffmpeg -version 2>&1 | head -1)" fi # 2. whisper.cpp — build from source echo "" echo "[2/5] Installing whisper-cli..." WHISPER_BIN="/usr/local/bin/whisper-cli" if [ -x "$WHISPER_BIN" ]; then echo " whisper-cli already installed" else echo " Building whisper.cpp from source (requires cmake, git, build-essential)..." sudo apt install -y cmake git build-essential TMPDIR=$(mktemp -d) git clone https://github.com/ggml-org/whisper.cpp.git "$TMPDIR/whisper.cpp" cd "$TMPDIR/whisper.cpp" cmake -B build cmake --build build --config Release -j$(nproc) sudo cp build/bin/whisper-cli "$WHISPER_BIN" cd - rm -rf "$TMPDIR" echo " whisper-cli installed to $WHISPER_BIN" fi # 3. Whisper model (ggml-medium.bin for better Korean) echo "" echo "[3/5] Downloading Whisper model (ggml-medium.bin ~1.5GB)..." WHISPER_MODEL_DIR="/usr/local/share/whisper-models" sudo mkdir -p "$WHISPER_MODEL_DIR" WHISPER_MODEL="$WHISPER_MODEL_DIR/ggml-medium.bin" if [ -f "$WHISPER_MODEL" ] && [ "$(stat -c%s "$WHISPER_MODEL" 2>/dev/null || stat -f%z "$WHISPER_MODEL" 2>/dev/null)" -gt 1000000 ]; then echo " Model already exists: $WHISPER_MODEL" else echo " Downloading... (this may take a few minutes)" curl -L "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.bin" \ -o /tmp/ggml-medium.bin && \ sudo mv /tmp/ggml-medium.bin "$WHISPER_MODEL" echo " Model saved to $WHISPER_MODEL" fi # 4. Piper TTS — use architecture-specific filename echo "" echo "[4/5] Installing Piper TTS..." PIPER_BIN="/usr/local/bin/piper" if [ -x "$PIPER_BIN" ]; then echo " piper already installed" else ARCH=$(dpkg --print-architecture) if [ "$ARCH" = "amd64" ]; then PIPER_ARCH="x86_64" elif [ "$ARCH" = "arm64" ]; then PIPER_ARCH="aarch64" elif [ "$ARCH" = "armhf" ]; then PIPER_ARCH="armv7l" else PIPER_ARCH="$ARCH" fi PIPER_URL="https://github.com/rhasspy/piper/releases/download/2023.11.14-2/piper_linux_${PIPER_ARCH}.tar.gz" echo " Downloading piper for $PIPER_ARCH from $PIPER_URL..." cd /tmp curl -L "$PIPER_URL" -o piper.tar.gz tar xzf piper.tar.gz sudo cp piper/piper "$PIPER_BIN" sudo chmod +x "$PIPER_BIN" # Install shared libraries required by piper sudo cp piper/libespeak-ng.so* /usr/local/lib/ sudo cp piper/libonnxruntime.so* /usr/local/lib/ sudo cp piper/libpiper_phonemize.so* /usr/local/lib/ sudo cp piper/espeak-ng /usr/local/lib/ sudo cp -r piper/espeak-ng-data /usr/local/lib/ sudo ldconfig rm -rf piper piper.tar.gz echo " piper installed to $PIPER_BIN" fi # 5. Piper Korean voice model (from neurlang — rhasspy repo returns 404 for Korean) echo "" echo "[5/5] Downloading Piper Korean voice model..." PIPER_VOICES_DIR="/usr/local/share/piper-voices" sudo mkdir -p "$PIPER_VOICES_DIR" ONNX_FILE="$PIPER_VOICES_DIR/ko_KR-kss-medium.onnx" ONNX_JSON="$PIPER_VOICES_DIR/ko_KR-kss-medium.onnx.json" if [ -f "$ONNX_FILE" ] && [ "$(stat -c%s "$ONNX_FILE" 2>/dev/null || stat -f%z "$ONNX_FILE" 2>/dev/null)" -gt 1000000 ]; then echo " Korean voice model already exists" else echo " Downloading ko_KR-kss-medium model (~63MB)..." curl -L "https://huggingface.co/neurlang/piper-onnx-kss-korean/resolve/main/piper-kss-korean.onnx" \ -o /tmp/ko_KR-kss-medium.onnx && \ curl -L "https://huggingface.co/neurlang/piper-onnx-kss-korean/raw/main/piper-kss-korean.onnx.json" \ -o /tmp/ko_KR-kss-medium.onnx.json && \ sudo mv /tmp/ko_KR-kss-medium.onnx "$ONNX_FILE" && \ sudo mv /tmp/ko_KR-kss-medium.onnx.json "$ONNX_JSON" echo " Korean voice model saved" fi echo "" echo "=== Installation complete! ===" echo "" echo "Verify installations:" echo " ffmpeg -version" echo " whisper-cli --help" echo " piper --version" echo "" echo "Add to your config.json to enable voice:" echo "" echo ' "voice": {' echo ' "enabled": true,' echo ' "stt": {' echo ' "model": "ggml-medium.bin",' echo ' "modelPath": "/usr/local/share/whisper-models/ggml-medium.bin",' echo ' "language": "ko"' echo ' },' echo ' "tts": {' echo ' "model": "ko_KR-kss-medium",' echo ' "modelPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx",' echo ' "configPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx.json"' echo ' }' echo ' }'