feat: add VOX TTS server on port 33003 with webui integration

- New vox-server/server.py (HTTP API matching morshu-server pattern)
- Added voxstart/voxstop/voxstatus commands to ttstoy cog
- WebUI /api/tts/generate and /api/mc/tts now support VOX mode
- Cog_unload stops VOX server automatically
This commit is contained in:
2026-06-09 11:07:59 -05:00
parent fa8c904c6f
commit 2f1e66cef4
4 changed files with 233 additions and 0 deletions
+109
View File
@@ -321,6 +321,11 @@ class TtsToy(Cog):
self.morshu_process = None
self.morshu_port = 33002
# VOX server
self.vox_dir = Path(__file__).resolve().parent / "vox-server"
self.vox_process = None
self.vox_port = 33003
# TTS queue system - one queue per guild
self.tts_queues = {} # guild_id -> asyncio.Queue
self.tts_locks = {} # guild_id -> asyncio.Lock
@@ -386,6 +391,8 @@ class TtsToy(Cog):
self._webui_state_processor.stop()
await self._stop_webui()
await self._stop_dectalk_server()
await self._stop_morshu_server()
await self._stop_vox_server()
if self._health_server:
self._health_server.shutdown()
@@ -572,6 +579,56 @@ class TtsToy(Cog):
except Exception as e:
log.exception(f"Error stopping Morshu server: {e}")
async def _start_vox_server(self) -> tuple:
"""Start the built-in VOX TTS server. Returns (success, error_msg)."""
try:
if self.vox_process and self.vox_process.poll() is None:
return True, None
server_py = self.vox_dir / "server.py"
if not server_py.exists():
return False, f"server.py not found: {server_py}"
env = os.environ.copy()
env['PORT'] = str(self.vox_port)
self.vox_process = await asyncio.to_thread(
subprocess.Popen,
[sys.executable, str(server_py)],
cwd=str(self.vox_dir),
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
await asyncio.sleep(2)
if self.vox_process.poll() is not None:
stderr = self.vox_process.stderr.read()
stdout = self.vox_process.stdout.read()
err = stderr or stdout or "Unknown error"
log.error(f"VOX server failed (exit {self.vox_process.returncode}): {err}")
return False, err
log.info(f"VOX server started on port {self.vox_port}")
return True, None
except Exception as e:
log.exception(f"Failed to start VOX server: {e}")
return False, str(e)
async def _stop_vox_server(self):
"""Stop the built-in VOX server."""
if self.vox_process and self.vox_process.poll() is None:
try:
self.vox_process.terminate()
await asyncio.sleep(1)
if self.vox_process.poll() is None:
self.vox_process.kill()
log.info("VOX server stopped")
except Exception as e:
log.exception(f"Error stopping VOX server: {e}")
async def _get_effective_minimax_voice(self, user: discord.abc.User) -> str:
mode = await self._get_tts_mode()
@@ -2181,6 +2238,58 @@ class TtsToy(Cog):
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="voxstart")
@commands.is_owner()
async def vox_start(self, ctx: commands.Context):
"""Start the built-in VOX TTS server."""
await ctx.send("🚀 Starting VOX server...")
success, error = await self._start_vox_server()
if success:
await ctx.send(
f"✅ VOX server started on port {self.vox_port}!\n"
f"Switch to VOX mode: `{ctx.clean_prefix}ttstoy mode vox`"
)
else:
err_msg = (error or "Unknown error")[:1500]
await ctx.send(f"❌ Failed to start VOX server:\n```\n{err_msg}\n```")
@ttstoy_group.command(name="voxstop")
@commands.is_owner()
async def vox_stop(self, ctx: commands.Context):
"""Stop the built-in VOX server."""
if not self.vox_process or self.vox_process.poll() is not None:
return await ctx.send("❌ VOX server is not running")
await self._stop_vox_server()
await ctx.send("✅ VOX server stopped")
@ttstoy_group.command(name="voxstatus")
@commands.is_owner()
async def vox_status(self, ctx: commands.Context):
"""Check VOX server status."""
process_running = self.vox_process and self.vox_process.poll() is None
api_responding = False
try:
r = requests.get(f"http://127.0.0.1:{self.vox_port}/health", timeout=2)
api_responding = r.status_code == 200
except:
pass
lines = [
"**VOX Server Status**",
f"Process Running: {'✅ Yes' if process_running else '❌ No'}",
f"API Responding: {'✅ Yes' if api_responding else '❌ No'}",
f"URL: `http://127.0.0.1:{self.vox_port}`",
]
if not process_running:
lines.append(f"\nStart with: `{ctx.clean_prefix}ttstoy voxstart`")
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="accessibility")
@commands.is_owner()
async def accessibility_toggle(self, ctx: commands.Context):
+102
View File
@@ -0,0 +1,102 @@
#!/usr/bin/env python3
"""
VOX TTS Server — standalone HTTP API for Half-Life VOX engine.
Runs on port 33003. GET /say?text=...&pack=vox returns WAV audio.
"""
import os
import sys
import io
import wave
from pathlib import Path
from http.server import HTTPServer, BaseHTTPRequestHandler
from urllib.parse import urlparse, parse_qs
# Add ttstoy root to path so we can import the engine
SCRIPT_DIR = Path(__file__).resolve().parent
TTSTOY_DIR = SCRIPT_DIR.parent
sys.path.insert(0, str(TTSTOY_DIR))
from vox_engine import VOX_PACKS, get_available_words, get_available_packs
PORT = int(os.environ.get("PORT", 33003))
def generate_wav(text, pack="vox"):
"""Concatenate word WAVs and return raw WAV bytes."""
pack_dir = VOX_PACKS.get(pack)
if not pack_dir or not pack_dir.exists():
raise RuntimeError(f"VOX pack '{pack}' not found")
words = text.lower().split()
splice = bytes()
sample_rate = 11025
sample_width = 1
channels = 1
found = False
for word in words:
wav_path = pack_dir / f"{word}.wav"
if not wav_path.exists():
continue
with wave.open(str(wav_path), 'rb') as snd:
sample_rate = snd.getframerate()
sample_width = snd.getsampwidth()
channels = snd.getnchannels()
splice += snd.readframes(snd.getnframes())
found = True
if not found:
raise RuntimeError("No matching VOX words found")
buf = io.BytesIO()
with wave.open(buf, 'wb') as out:
out.setnchannels(channels)
out.setsampwidth(sample_width)
out.setframerate(sample_rate)
out.writeframes(splice)
return buf.getvalue()
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
parsed = urlparse(self.path)
if parsed.path == "/say":
params = parse_qs(parsed.query)
text = params.get("text", [""])[0]
if not text:
self.send_error(400, "Missing 'text' parameter")
return
pack = params.get("pack", ["vox"])[0]
try:
wav_bytes = generate_wav(text, pack)
self.send_response(200)
self.send_header("Content-Type", "audio/wav")
self.send_header("Content-Length", str(len(wav_bytes)))
self.end_headers()
self.wfile.write(wav_bytes)
except Exception as e:
self.send_error(500, str(e))
elif parsed.path == "/health":
self.send_response(200)
self.send_header("Content-Type", "text/plain")
self.end_headers()
self.wfile.write(b"ok")
elif parsed.path == "/packs":
packs = get_available_packs()
body = ",".join(packs).encode()
self.send_response(200)
self.send_header("Content-Type", "text/plain")
self.end_headers()
self.wfile.write(body)
else:
self.send_error(404)
def log_message(self, format, *args):
print(f"[VOX] {args[0]}")
if __name__ == "__main__":
print(f"[VOX] Starting server on port {PORT}...")
server = HTTPServer(("0.0.0.0", PORT), Handler)
print(f"[VOX] Ready — http://127.0.0.1:{PORT}/say?text=hello+world")
server.serve_forever()
+21
View File
@@ -428,6 +428,20 @@ def api_tts_generate():
except Exception as e:
return jsonify({"error": f"DECTalk server error: {e}"}), 500
if mode == "vox":
try:
pack = gcfg.get("vox_pack", "vox")
r = requests.get("http://127.0.0.1:33003/say", params={"text": text, "pack": pack}, timeout=60)
r.raise_for_status()
job_id = str(uuid.uuid4())
out_path = tempfile.mktemp(suffix=".wav")
with open(out_path, "wb") as f:
f.write(r.content)
_tts_jobs[job_id] = {"status": "done", "path": out_path, "text": text, "user": discord_name, "engine": "vox"}
return jsonify({"job_id": job_id})
except Exception as e:
return jsonify({"error": f"VOX server error: {e}"}), 500
if mode != "chatterbox":
return jsonify({"error": f"Generate Only is not available for '{mode}' mode via web UI."}), 400
@@ -684,6 +698,13 @@ def api_mc_tts():
r.raise_for_status()
return r.content, 200, {"Content-Type": "audio/wav"}
elif mode == "vox":
gcfg = get_global_config()
pack = gcfg.get("vox_pack", "vox")
r = requests.get("http://127.0.0.1:33003/say", params={"text": text, "pack": pack}, timeout=60)
r.raise_for_status()
return r.content, 200, {"Content-Type": "audio/wav"}
else:
return jsonify({"error": f"Unknown mode: {mode}"}), 400