Initial standalone vox TTS server extracted from the ttstoy bot cog
Self-contained HTTP service with engine, data, start script, systemd unit, and documentation. Runs independently of the Discord bot on its fixed port.
This commit is contained in:
+104
@@ -0,0 +1,104 @@
|
||||
"""
|
||||
Bundled VOX engine for ttstoy.
|
||||
Based on VOXGen by wphillips (https://github.com/wphillips/VOXGen)
|
||||
Black Mesa / Half-Life VOX announcer text-to-speech via word-level WAV concatenation.
|
||||
"""
|
||||
import os
|
||||
import wave
|
||||
import struct
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
_BASE_DIR = Path(__file__).resolve().parent
|
||||
|
||||
VOX_PACKS = {
|
||||
"vox": _BASE_DIR / "vox_words",
|
||||
"vox2": _BASE_DIR / "vox2_words",
|
||||
}
|
||||
|
||||
|
||||
def get_available_packs() -> List[str]:
|
||||
"""Return list of installed VOX packs."""
|
||||
return [name for name, p in VOX_PACKS.items() if p.exists()]
|
||||
|
||||
|
||||
def get_available_words(pack: str = "vox") -> List[str]:
|
||||
"""Return sorted list of available words for a VOX pack."""
|
||||
pack_dir = VOX_PACKS.get(pack)
|
||||
if not pack_dir or not pack_dir.exists():
|
||||
return []
|
||||
return sorted(
|
||||
p.stem for p in pack_dir.glob("*.wav")
|
||||
if not p.stem.startswith("00_") and not p.stem.startswith("_")
|
||||
)
|
||||
|
||||
|
||||
def generate_vox(text: str, output_path: str, pack: str = "vox") -> List[str]:
|
||||
"""
|
||||
Generate VOX audio from text by concatenating word WAV files.
|
||||
|
||||
Args:
|
||||
text: Input sentence (words separated by spaces).
|
||||
output_path: Path to write the output WAV/MP3 file.
|
||||
pack: Which VOX pack to use ("vox" or "vox2").
|
||||
|
||||
Returns:
|
||||
List of words that were actually found and used.
|
||||
|
||||
Raises:
|
||||
RuntimeError: If no words matched or pack not found.
|
||||
"""
|
||||
pack_dir = VOX_PACKS.get(pack)
|
||||
if not pack_dir or not pack_dir.exists():
|
||||
raise RuntimeError(f"VOX pack '{pack}' not found at {pack_dir}")
|
||||
|
||||
words = text.lower().split()
|
||||
accepted = []
|
||||
splice = bytes()
|
||||
|
||||
sample_rate = 11025
|
||||
sample_width = 1
|
||||
channels = 1
|
||||
|
||||
for word in words:
|
||||
wav_path = pack_dir / f"{word}.wav"
|
||||
if not wav_path.exists():
|
||||
continue
|
||||
|
||||
with wave.open(str(wav_path), 'rb') as snd:
|
||||
sample_rate = snd.getframerate()
|
||||
sample_width = snd.getsampwidth()
|
||||
channels = snd.getnchannels()
|
||||
splice += snd.readframes(snd.getnframes())
|
||||
accepted.append(word)
|
||||
|
||||
if not accepted:
|
||||
raise RuntimeError(
|
||||
f"No matching VOX words found. Available words include: "
|
||||
f"{', '.join(get_available_words(pack)[:20])}..."
|
||||
)
|
||||
|
||||
# Write intermediate WAV
|
||||
wav_out = output_path.rsplit('.', 1)[0] + '_tmp.wav'
|
||||
with wave.open(wav_out, 'wb') as out:
|
||||
out.setnchannels(channels)
|
||||
out.setsampwidth(sample_width)
|
||||
out.setframerate(sample_rate)
|
||||
out.writeframes(splice)
|
||||
|
||||
# Convert to MP3 using ffmpeg for consistency with the rest of ttstoy
|
||||
import subprocess
|
||||
cmd = [
|
||||
"ffmpeg", "-y", "-i", wav_out,
|
||||
"-ac", "1", "-ar", "32000", "-b:a", "128k",
|
||||
output_path,
|
||||
]
|
||||
subprocess.run(cmd, check=True, capture_output=True, text=True)
|
||||
|
||||
# Clean up temp WAV
|
||||
try:
|
||||
os.remove(wav_out)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
return accepted
|
||||
Reference in New Issue
Block a user