105 lines
2.9 KiB
Python
105 lines
2.9 KiB
Python
"""
|
|
Bundled VOX engine for ttstoy.
|
|
Based on VOXGen by wphillips (https://github.com/wphillips/VOXGen)
|
|
Black Mesa / Half-Life VOX announcer text-to-speech via word-level WAV concatenation.
|
|
"""
|
|
import os
|
|
import wave
|
|
import struct
|
|
from pathlib import Path
|
|
from typing import List, Optional
|
|
|
|
_BASE_DIR = Path(__file__).resolve().parent
|
|
|
|
VOX_PACKS = {
|
|
"vox": _BASE_DIR / "vox_words",
|
|
"vox2": _BASE_DIR / "vox2_words",
|
|
}
|
|
|
|
|
|
def get_available_packs() -> List[str]:
|
|
"""Return list of installed VOX packs."""
|
|
return [name for name, p in VOX_PACKS.items() if p.exists()]
|
|
|
|
|
|
def get_available_words(pack: str = "vox") -> List[str]:
|
|
"""Return sorted list of available words for a VOX pack."""
|
|
pack_dir = VOX_PACKS.get(pack)
|
|
if not pack_dir or not pack_dir.exists():
|
|
return []
|
|
return sorted(
|
|
p.stem for p in pack_dir.glob("*.wav")
|
|
if not p.stem.startswith("00_") and not p.stem.startswith("_")
|
|
)
|
|
|
|
|
|
def generate_vox(text: str, output_path: str, pack: str = "vox") -> List[str]:
|
|
"""
|
|
Generate VOX audio from text by concatenating word WAV files.
|
|
|
|
Args:
|
|
text: Input sentence (words separated by spaces).
|
|
output_path: Path to write the output WAV/MP3 file.
|
|
pack: Which VOX pack to use ("vox" or "vox2").
|
|
|
|
Returns:
|
|
List of words that were actually found and used.
|
|
|
|
Raises:
|
|
RuntimeError: If no words matched or pack not found.
|
|
"""
|
|
pack_dir = VOX_PACKS.get(pack)
|
|
if not pack_dir or not pack_dir.exists():
|
|
raise RuntimeError(f"VOX pack '{pack}' not found at {pack_dir}")
|
|
|
|
words = text.lower().split()
|
|
accepted = []
|
|
splice = bytes()
|
|
|
|
sample_rate = 11025
|
|
sample_width = 1
|
|
channels = 1
|
|
|
|
for word in words:
|
|
wav_path = pack_dir / f"{word}.wav"
|
|
if not wav_path.exists():
|
|
continue
|
|
|
|
with wave.open(str(wav_path), 'rb') as snd:
|
|
sample_rate = snd.getframerate()
|
|
sample_width = snd.getsampwidth()
|
|
channels = snd.getnchannels()
|
|
splice += snd.readframes(snd.getnframes())
|
|
accepted.append(word)
|
|
|
|
if not accepted:
|
|
raise RuntimeError(
|
|
f"No matching VOX words found. Available words include: "
|
|
f"{', '.join(get_available_words(pack)[:20])}..."
|
|
)
|
|
|
|
# Write intermediate WAV
|
|
wav_out = output_path.rsplit('.', 1)[0] + '_tmp.wav'
|
|
with wave.open(wav_out, 'wb') as out:
|
|
out.setnchannels(channels)
|
|
out.setsampwidth(sample_width)
|
|
out.setframerate(sample_rate)
|
|
out.writeframes(splice)
|
|
|
|
# Convert to MP3 using ffmpeg for consistency with the rest of ttstoy
|
|
import subprocess
|
|
cmd = [
|
|
"ffmpeg", "-y", "-i", wav_out,
|
|
"-ac", "1", "-ar", "32000", "-b:a", "128k",
|
|
output_path,
|
|
]
|
|
subprocess.run(cmd, check=True, capture_output=True, text=True)
|
|
|
|
# Clean up temp WAV
|
|
try:
|
|
os.remove(wav_out)
|
|
except OSError:
|
|
pass
|
|
|
|
return accepted
|