Files
kingston b459533770 Initial standalone vox TTS server extracted from the ttstoy bot cog
Self-contained HTTP service with engine, data, start script, systemd unit,
and documentation. Runs independently of the Discord bot on its fixed port.
2026-09-14 14:31:16 -05:00

105 lines
2.9 KiB
Python

"""
Bundled VOX engine for ttstoy.
Based on VOXGen by wphillips (https://github.com/wphillips/VOXGen)
Black Mesa / Half-Life VOX announcer text-to-speech via word-level WAV concatenation.
"""
import os
import wave
import struct
from pathlib import Path
from typing import List, Optional
_BASE_DIR = Path(__file__).resolve().parent
VOX_PACKS = {
"vox": _BASE_DIR / "vox_words",
"vox2": _BASE_DIR / "vox2_words",
}
def get_available_packs() -> List[str]:
"""Return list of installed VOX packs."""
return [name for name, p in VOX_PACKS.items() if p.exists()]
def get_available_words(pack: str = "vox") -> List[str]:
"""Return sorted list of available words for a VOX pack."""
pack_dir = VOX_PACKS.get(pack)
if not pack_dir or not pack_dir.exists():
return []
return sorted(
p.stem for p in pack_dir.glob("*.wav")
if not p.stem.startswith("00_") and not p.stem.startswith("_")
)
def generate_vox(text: str, output_path: str, pack: str = "vox") -> List[str]:
"""
Generate VOX audio from text by concatenating word WAV files.
Args:
text: Input sentence (words separated by spaces).
output_path: Path to write the output WAV/MP3 file.
pack: Which VOX pack to use ("vox" or "vox2").
Returns:
List of words that were actually found and used.
Raises:
RuntimeError: If no words matched or pack not found.
"""
pack_dir = VOX_PACKS.get(pack)
if not pack_dir or not pack_dir.exists():
raise RuntimeError(f"VOX pack '{pack}' not found at {pack_dir}")
words = text.lower().split()
accepted = []
splice = bytes()
sample_rate = 11025
sample_width = 1
channels = 1
for word in words:
wav_path = pack_dir / f"{word}.wav"
if not wav_path.exists():
continue
with wave.open(str(wav_path), 'rb') as snd:
sample_rate = snd.getframerate()
sample_width = snd.getsampwidth()
channels = snd.getnchannels()
splice += snd.readframes(snd.getnframes())
accepted.append(word)
if not accepted:
raise RuntimeError(
f"No matching VOX words found. Available words include: "
f"{', '.join(get_available_words(pack)[:20])}..."
)
# Write intermediate WAV
wav_out = output_path.rsplit('.', 1)[0] + '_tmp.wav'
with wave.open(wav_out, 'wb') as out:
out.setnchannels(channels)
out.setsampwidth(sample_width)
out.setframerate(sample_rate)
out.writeframes(splice)
# Convert to MP3 using ffmpeg for consistency with the rest of ttstoy
import subprocess
cmd = [
"ffmpeg", "-y", "-i", wav_out,
"-ac", "1", "-ar", "32000", "-b:a", "128k",
output_path,
]
subprocess.run(cmd, check=True, capture_output=True, text=True)
# Clean up temp WAV
try:
os.remove(wav_out)
except OSError:
pass
return accepted