diff --git a/.ai-context.md b/.ai-context.md index cd7e7df..8d6342b 100644 --- a/.ai-context.md +++ b/.ai-context.md @@ -21,6 +21,7 @@ Multi-engine TTS cog with Discord slash commands, web UI, and Minecraft server i - `webui/app.py` — Flask web UI for browser-based TTS with login system - `dectalk-server/` — Node.js DECTalk wrapper (Express, port 33001) - `morshu-server/server.py` — Python HTTP server for Morshuspeak (port 33002) +- `vox-server/server.py` — Python HTTP server for Half-Life VOX (port 33003) - `morshutalk_engine.py` — Audio concatenation engine for Morshu voice - `vox_engine.py` — Half-Life VOX word concatenation - `vox_words/`, `vox2_words/` — VOX audio sample packs diff --git a/ttstoy/ttstoy.py b/ttstoy/ttstoy.py index e6f311a..297ecfa 100644 --- a/ttstoy/ttstoy.py +++ b/ttstoy/ttstoy.py @@ -321,6 +321,11 @@ class TtsToy(Cog): self.morshu_process = None self.morshu_port = 33002 + # VOX server + self.vox_dir = Path(__file__).resolve().parent / "vox-server" + self.vox_process = None + self.vox_port = 33003 + # TTS queue system - one queue per guild self.tts_queues = {} # guild_id -> asyncio.Queue self.tts_locks = {} # guild_id -> asyncio.Lock @@ -386,6 +391,8 @@ class TtsToy(Cog): self._webui_state_processor.stop() await self._stop_webui() await self._stop_dectalk_server() + await self._stop_morshu_server() + await self._stop_vox_server() if self._health_server: self._health_server.shutdown() @@ -572,6 +579,56 @@ class TtsToy(Cog): except Exception as e: log.exception(f"Error stopping Morshu server: {e}") + async def _start_vox_server(self) -> tuple: + """Start the built-in VOX TTS server. Returns (success, error_msg).""" + try: + if self.vox_process and self.vox_process.poll() is None: + return True, None + + server_py = self.vox_dir / "server.py" + if not server_py.exists(): + return False, f"server.py not found: {server_py}" + + env = os.environ.copy() + env['PORT'] = str(self.vox_port) + + self.vox_process = await asyncio.to_thread( + subprocess.Popen, + [sys.executable, str(server_py)], + cwd=str(self.vox_dir), + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True + ) + + await asyncio.sleep(2) + + if self.vox_process.poll() is not None: + stderr = self.vox_process.stderr.read() + stdout = self.vox_process.stdout.read() + err = stderr or stdout or "Unknown error" + log.error(f"VOX server failed (exit {self.vox_process.returncode}): {err}") + return False, err + + log.info(f"VOX server started on port {self.vox_port}") + return True, None + except Exception as e: + log.exception(f"Failed to start VOX server: {e}") + return False, str(e) + + async def _stop_vox_server(self): + """Stop the built-in VOX server.""" + if self.vox_process and self.vox_process.poll() is None: + try: + self.vox_process.terminate() + await asyncio.sleep(1) + if self.vox_process.poll() is None: + self.vox_process.kill() + log.info("VOX server stopped") + except Exception as e: + log.exception(f"Error stopping VOX server: {e}") + async def _get_effective_minimax_voice(self, user: discord.abc.User) -> str: mode = await self._get_tts_mode() @@ -2181,6 +2238,58 @@ class TtsToy(Cog): await ctx.send("\n".join(lines)) + @ttstoy_group.command(name="voxstart") + @commands.is_owner() + async def vox_start(self, ctx: commands.Context): + """Start the built-in VOX TTS server.""" + await ctx.send("🚀 Starting VOX server...") + + success, error = await self._start_vox_server() + + if success: + await ctx.send( + f"✅ VOX server started on port {self.vox_port}!\n" + f"Switch to VOX mode: `{ctx.clean_prefix}ttstoy mode vox`" + ) + else: + err_msg = (error or "Unknown error")[:1500] + await ctx.send(f"❌ Failed to start VOX server:\n```\n{err_msg}\n```") + + @ttstoy_group.command(name="voxstop") + @commands.is_owner() + async def vox_stop(self, ctx: commands.Context): + """Stop the built-in VOX server.""" + if not self.vox_process or self.vox_process.poll() is not None: + return await ctx.send("❌ VOX server is not running") + + await self._stop_vox_server() + await ctx.send("✅ VOX server stopped") + + @ttstoy_group.command(name="voxstatus") + @commands.is_owner() + async def vox_status(self, ctx: commands.Context): + """Check VOX server status.""" + process_running = self.vox_process and self.vox_process.poll() is None + + api_responding = False + try: + r = requests.get(f"http://127.0.0.1:{self.vox_port}/health", timeout=2) + api_responding = r.status_code == 200 + except: + pass + + lines = [ + "**VOX Server Status**", + f"Process Running: {'✅ Yes' if process_running else '❌ No'}", + f"API Responding: {'✅ Yes' if api_responding else '❌ No'}", + f"URL: `http://127.0.0.1:{self.vox_port}`", + ] + + if not process_running: + lines.append(f"\nStart with: `{ctx.clean_prefix}ttstoy voxstart`") + + await ctx.send("\n".join(lines)) + @ttstoy_group.command(name="accessibility") @commands.is_owner() async def accessibility_toggle(self, ctx: commands.Context): diff --git a/ttstoy/vox-server/server.py b/ttstoy/vox-server/server.py new file mode 100644 index 0000000..a2b35ab --- /dev/null +++ b/ttstoy/vox-server/server.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +""" +VOX TTS Server — standalone HTTP API for Half-Life VOX engine. +Runs on port 33003. GET /say?text=...&pack=vox returns WAV audio. +""" +import os +import sys +import io +import wave +from pathlib import Path +from http.server import HTTPServer, BaseHTTPRequestHandler +from urllib.parse import urlparse, parse_qs + +# Add ttstoy root to path so we can import the engine +SCRIPT_DIR = Path(__file__).resolve().parent +TTSTOY_DIR = SCRIPT_DIR.parent +sys.path.insert(0, str(TTSTOY_DIR)) + +from vox_engine import VOX_PACKS, get_available_words, get_available_packs + +PORT = int(os.environ.get("PORT", 33003)) + + +def generate_wav(text, pack="vox"): + """Concatenate word WAVs and return raw WAV bytes.""" + pack_dir = VOX_PACKS.get(pack) + if not pack_dir or not pack_dir.exists(): + raise RuntimeError(f"VOX pack '{pack}' not found") + + words = text.lower().split() + splice = bytes() + sample_rate = 11025 + sample_width = 1 + channels = 1 + found = False + + for word in words: + wav_path = pack_dir / f"{word}.wav" + if not wav_path.exists(): + continue + with wave.open(str(wav_path), 'rb') as snd: + sample_rate = snd.getframerate() + sample_width = snd.getsampwidth() + channels = snd.getnchannels() + splice += snd.readframes(snd.getnframes()) + found = True + + if not found: + raise RuntimeError("No matching VOX words found") + + buf = io.BytesIO() + with wave.open(buf, 'wb') as out: + out.setnchannels(channels) + out.setsampwidth(sample_width) + out.setframerate(sample_rate) + out.writeframes(splice) + return buf.getvalue() + + +class Handler(BaseHTTPRequestHandler): + def do_GET(self): + parsed = urlparse(self.path) + if parsed.path == "/say": + params = parse_qs(parsed.query) + text = params.get("text", [""])[0] + if not text: + self.send_error(400, "Missing 'text' parameter") + return + pack = params.get("pack", ["vox"])[0] + try: + wav_bytes = generate_wav(text, pack) + self.send_response(200) + self.send_header("Content-Type", "audio/wav") + self.send_header("Content-Length", str(len(wav_bytes))) + self.end_headers() + self.wfile.write(wav_bytes) + except Exception as e: + self.send_error(500, str(e)) + elif parsed.path == "/health": + self.send_response(200) + self.send_header("Content-Type", "text/plain") + self.end_headers() + self.wfile.write(b"ok") + elif parsed.path == "/packs": + packs = get_available_packs() + body = ",".join(packs).encode() + self.send_response(200) + self.send_header("Content-Type", "text/plain") + self.end_headers() + self.wfile.write(body) + else: + self.send_error(404) + + def log_message(self, format, *args): + print(f"[VOX] {args[0]}") + + +if __name__ == "__main__": + print(f"[VOX] Starting server on port {PORT}...") + server = HTTPServer(("0.0.0.0", PORT), Handler) + print(f"[VOX] Ready — http://127.0.0.1:{PORT}/say?text=hello+world") + server.serve_forever() diff --git a/ttstoy/webui/app.py b/ttstoy/webui/app.py index d6e0809..974a507 100644 --- a/ttstoy/webui/app.py +++ b/ttstoy/webui/app.py @@ -428,6 +428,20 @@ def api_tts_generate(): except Exception as e: return jsonify({"error": f"DECTalk server error: {e}"}), 500 + if mode == "vox": + try: + pack = gcfg.get("vox_pack", "vox") + r = requests.get("http://127.0.0.1:33003/say", params={"text": text, "pack": pack}, timeout=60) + r.raise_for_status() + job_id = str(uuid.uuid4()) + out_path = tempfile.mktemp(suffix=".wav") + with open(out_path, "wb") as f: + f.write(r.content) + _tts_jobs[job_id] = {"status": "done", "path": out_path, "text": text, "user": discord_name, "engine": "vox"} + return jsonify({"job_id": job_id}) + except Exception as e: + return jsonify({"error": f"VOX server error: {e}"}), 500 + if mode != "chatterbox": return jsonify({"error": f"Generate Only is not available for '{mode}' mode via web UI."}), 400 @@ -684,6 +698,13 @@ def api_mc_tts(): r.raise_for_status() return r.content, 200, {"Content-Type": "audio/wav"} + elif mode == "vox": + gcfg = get_global_config() + pack = gcfg.get("vox_pack", "vox") + r = requests.get("http://127.0.0.1:33003/say", params={"text": text, "pack": pack}, timeout=60) + r.raise_for_status() + return r.content, 200, {"Content-Type": "audio/wav"} + else: return jsonify({"error": f"Unknown mode: {mode}"}), 400