feat: add VOX TTS server on port 33003 with webui integration
- New vox-server/server.py (HTTP API matching morshu-server pattern) - Added voxstart/voxstop/voxstatus commands to ttstoy cog - WebUI /api/tts/generate and /api/mc/tts now support VOX mode - Cog_unload stops VOX server automatically
This commit is contained in:
@@ -0,0 +1,102 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
VOX TTS Server — standalone HTTP API for Half-Life VOX engine.
|
||||
Runs on port 33003. GET /say?text=...&pack=vox returns WAV audio.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import io
|
||||
import wave
|
||||
from pathlib import Path
|
||||
from http.server import HTTPServer, BaseHTTPRequestHandler
|
||||
from urllib.parse import urlparse, parse_qs
|
||||
|
||||
# Add ttstoy root to path so we can import the engine
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
TTSTOY_DIR = SCRIPT_DIR.parent
|
||||
sys.path.insert(0, str(TTSTOY_DIR))
|
||||
|
||||
from vox_engine import VOX_PACKS, get_available_words, get_available_packs
|
||||
|
||||
PORT = int(os.environ.get("PORT", 33003))
|
||||
|
||||
|
||||
def generate_wav(text, pack="vox"):
|
||||
"""Concatenate word WAVs and return raw WAV bytes."""
|
||||
pack_dir = VOX_PACKS.get(pack)
|
||||
if not pack_dir or not pack_dir.exists():
|
||||
raise RuntimeError(f"VOX pack '{pack}' not found")
|
||||
|
||||
words = text.lower().split()
|
||||
splice = bytes()
|
||||
sample_rate = 11025
|
||||
sample_width = 1
|
||||
channels = 1
|
||||
found = False
|
||||
|
||||
for word in words:
|
||||
wav_path = pack_dir / f"{word}.wav"
|
||||
if not wav_path.exists():
|
||||
continue
|
||||
with wave.open(str(wav_path), 'rb') as snd:
|
||||
sample_rate = snd.getframerate()
|
||||
sample_width = snd.getsampwidth()
|
||||
channels = snd.getnchannels()
|
||||
splice += snd.readframes(snd.getnframes())
|
||||
found = True
|
||||
|
||||
if not found:
|
||||
raise RuntimeError("No matching VOX words found")
|
||||
|
||||
buf = io.BytesIO()
|
||||
with wave.open(buf, 'wb') as out:
|
||||
out.setnchannels(channels)
|
||||
out.setsampwidth(sample_width)
|
||||
out.setframerate(sample_rate)
|
||||
out.writeframes(splice)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
parsed = urlparse(self.path)
|
||||
if parsed.path == "/say":
|
||||
params = parse_qs(parsed.query)
|
||||
text = params.get("text", [""])[0]
|
||||
if not text:
|
||||
self.send_error(400, "Missing 'text' parameter")
|
||||
return
|
||||
pack = params.get("pack", ["vox"])[0]
|
||||
try:
|
||||
wav_bytes = generate_wav(text, pack)
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "audio/wav")
|
||||
self.send_header("Content-Length", str(len(wav_bytes)))
|
||||
self.end_headers()
|
||||
self.wfile.write(wav_bytes)
|
||||
except Exception as e:
|
||||
self.send_error(500, str(e))
|
||||
elif parsed.path == "/health":
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.end_headers()
|
||||
self.wfile.write(b"ok")
|
||||
elif parsed.path == "/packs":
|
||||
packs = get_available_packs()
|
||||
body = ",".join(packs).encode()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/plain")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
else:
|
||||
self.send_error(404)
|
||||
|
||||
def log_message(self, format, *args):
|
||||
print(f"[VOX] {args[0]}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"[VOX] Starting server on port {PORT}...")
|
||||
server = HTTPServer(("0.0.0.0", PORT), Handler)
|
||||
print(f"[VOX] Ready — http://127.0.0.1:{PORT}/say?text=hello+world")
|
||||
server.serve_forever()
|
||||
Reference in New Issue
Block a user