Add standalone morshu-server, update webui to support morshu/dectalk generate

This commit is contained in:
2026-06-05 19:05:00 -05:00
parent 27b24701e8
commit 7d9c6c0edb
4 changed files with 117 additions and 70 deletions
+60
View File
@@ -0,0 +1,60 @@
#!/usr/bin/env python3
"""
Morshu TTS Server — standalone HTTP API for MorshuTalk engine.
Runs on port 3002. GET /say?text=... returns WAV audio.
"""
import sys
import io
from pathlib import Path
from http.server import HTTPServer, BaseHTTPRequestHandler
from urllib.parse import urlparse, parse_qs
# Add ttstoy to path so we can import the engine
sys.path.insert(0, str(Path(__file__).parent.parent))
from morshutalk_engine import Morshu
morshu = Morshu()
PORT = 3002
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
parsed = urlparse(self.path)
if parsed.path == "/say":
params = parse_qs(parsed.query)
text = params.get("text", [""])[0]
if not text:
self.send_error(400, "Missing 'text' parameter")
return
try:
audio = morshu.load_text(text)
if audio is False or len(audio) == 0:
self.send_error(500, "Failed to generate audio")
return
buf = io.BytesIO()
audio.export(buf, format="wav")
wav_bytes = buf.getvalue()
self.send_response(200)
self.send_header("Content-Type", "audio/wav")
self.send_header("Content-Length", str(len(wav_bytes)))
self.end_headers()
self.wfile.write(wav_bytes)
except Exception as e:
self.send_error(500, str(e))
elif parsed.path == "/health":
self.send_response(200)
self.send_header("Content-Type", "text/plain")
self.end_headers()
self.wfile.write(b"ok")
else:
self.send_error(404)
def log_message(self, format, *args):
print(f"[Morshu] {args[0]}")
if __name__ == "__main__":
print(f"[Morshu] Starting server on port {PORT}...")
server = HTTPServer(("0.0.0.0", PORT), Handler)
print(f"[Morshu] Ready — http://127.0.0.1:{PORT}/say?text=hello")
server.serve_forever()
+4
View File
@@ -0,0 +1,4 @@
#!/bin/bash
# Start Morshu TTS Server
cd "$(dirname "$0")"
exec python3 server.py
+21 -11
View File
@@ -621,14 +621,26 @@ class TtsToy(Cog):
def _save_morshu_tts(self, text: str, audio_path: str):
"""
Generate TTS audio using MorshuTalk.
Generate TTS audio using MorshuTalk server (HTTP).
Falls back to in-process engine if server is unavailable.
Args:
text: Text to synthesize
audio_path: Output file path
"""
morshu_url = "http://127.0.0.1:3002"
try:
r = requests.get(f"{morshu_url}/say", params={"text": text}, timeout=60)
r.raise_for_status()
with open(audio_path, "wb") as f:
f.write(r.content)
return
except Exception as e:
log.warning(f"[Morshu] Server unavailable ({e}), falling back to in-process engine")
# Fallback to in-process
if not self.morshu:
raise RuntimeError("MorshuTalk engine is not available. Check that g2p_en, numpy, and pydub are installed.")
raise RuntimeError("MorshuTalk engine is not available. Start the morshu-server or install g2p_en, numpy, pydub.")
audio = self.morshu.load_text(text)
if audio is False or audio is None:
@@ -2972,9 +2984,8 @@ class TtsToy(Cog):
mode_override = payload.get("mode_override") # per-request mode from web UI
voice_override = payload.get("voice_override") # per-request voice from web UI
skip_vc = payload.get("skip_vc", False)
skip_post = payload.get("skip_post", False)
guild_id_str = payload.get("guild_id", "")
log.info(f"[WebUI] speak_in_vc: user={user} job={job_id} guild={guild_id_str} mode_override={mode_override} voice_override={voice_override} skip_vc={skip_vc} skip_post={skip_post} text={text[:60]!r}")
log.info(f"[WebUI] speak_in_vc: user={user} job={job_id} guild={guild_id_str} mode_override={mode_override} voice_override={voice_override} skip_vc={skip_vc} text={text[:60]!r}")
webui_url = self._get_webui_url()
secret = self._get_internal_secret()
@@ -3086,14 +3097,13 @@ class TtsToy(Cog):
log.exception(f"[WebUI] VC playback failed for job {job_id}: {e}")
# Post transcript to Discord channel and mark job done
# Post transcript to Discord channel (unless skip_post)
# Resolve the best guild ID for posting: login guild > VC guild > empty
post_guild_id = guild_id_str or (str(member_guild.id) if member_guild else "")
if not skip_post:
await self._handle_webui_post({
"job_id": job_id, "path": audio_path,
"text": text, "user": discord_name, "engine": mode,
"guild_id": post_guild_id,
})
await self._handle_webui_post({
"job_id": job_id, "path": audio_path,
"text": text, "user": discord_name, "engine": mode,
"guild_id": post_guild_id,
})
await asyncio.to_thread(_update_job, "done", path=audio_path)
async def _handle_webui_post(self, post: dict):
+32 -59
View File
@@ -386,7 +386,7 @@ def tts_job_update(job_id):
@app.route("/api/tts/generate", methods=["POST"])
@login_required
def api_tts_generate():
"""Generate TTS via the bot (no VC playback). All modes supported."""
"""Generate TTS via the bot or directly via TTS servers."""
data = request.get_json(force=True)
text = (data.get("text") or "").strip()
if not text:
@@ -400,8 +400,38 @@ def api_tts_generate():
global_mode = gcfg.get("tts_mode", "minimax")
mode = data.get("mode") or user_mode_overrides.get(str(user_id), global_mode)
# Direct server modes — generate audio without the bot
if mode == "morshu":
try:
r = requests.get("http://127.0.0.1:3002/say", params={"text": text}, timeout=60)
r.raise_for_status()
job_id = str(uuid.uuid4())
out_path = tempfile.mktemp(suffix=".wav")
with open(out_path, "wb") as f:
f.write(r.content)
_tts_jobs[job_id] = {"status": "done", "path": out_path, "text": text, "user": discord_name, "engine": "morshu"}
return jsonify({"job_id": job_id})
except Exception as e:
return jsonify({"error": f"Morshu server error: {e}"}), 500
if mode == "dectalk":
try:
dectalk_url = gcfg.get("dectalk_api_url", "http://127.0.0.1:3001")
r = requests.get(f"{dectalk_url}/say", params={"text": text}, timeout=30)
r.raise_for_status()
job_id = str(uuid.uuid4())
out_path = tempfile.mktemp(suffix=".wav")
with open(out_path, "wb") as f:
f.write(r.content)
_tts_jobs[job_id] = {"status": "done", "path": out_path, "text": text, "user": discord_name, "engine": "dectalk"}
return jsonify({"job_id": job_id})
except Exception as e:
return jsonify({"error": f"DECTalk server error: {e}"}), 500
if mode != "chatterbox":
return jsonify({"error": f"Generate Only is not available for '{mode}' mode via web UI."}), 400
voice = data.get("voice") or None
post_to_discord = data.get("post_to_discord", True)
job_id = str(uuid.uuid4())
_tts_jobs[job_id] = {"status": "pending", "text": text, "user": discord_name, "engine": mode}
@@ -412,7 +442,6 @@ def api_tts_generate():
"mode_override": mode,
"voice_override": voice,
"skip_vc": True,
"skip_post": not post_to_discord,
"guild_id": _resolve_guild_id(),
})
@@ -565,48 +594,11 @@ def api_chatterbox_voices():
user_id = session["user_id"]
ucfg = get_user_config(user_id)
voices = ucfg.get("chatterbox_voices", [])
# Validate voices exist on the Chatterbox server
cb_url = get_chatterbox_url()
server_voices = set()
try:
r = requests.get(f"{cb_url.rstrip('/')}/api/ui/initial-data", timeout=3)
if r.status_code == 200:
data = r.json()
server_voices = set(data.get("predefined_voices", []))
except Exception:
pass # If server unreachable, show all voices without filtering
result = []
stale = []
for v in voices:
if server_voices and v not in server_voices:
stale.append(v)
continue
parts = v.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
result.append({"filename": v, "display": display})
# Auto-clean stale voices from config
if stale:
try:
p = COG_CONFIG_PATH / "settings.json"
data = _read_json(p)
if data:
ucfg_data = data.get(COG_IDENTIFIER, {}).get("USER", {}).get(str(user_id), {})
cfg_voices = ucfg_data.get("chatterbox_voices", [])
for s in stale:
if s in cfg_voices:
cfg_voices.remove(s)
ucfg_data["chatterbox_voices"] = cfg_voices
if ucfg_data.get("minimax_voice") in stale:
ucfg_data["minimax_voice"] = cfg_voices[0] if cfg_voices else None
with open(p, "w") as f:
json.dump(data, f, indent=2)
log.info(f"Auto-cleaned {len(stale)} stale voices for user {user_id}: {stale}")
except Exception as ex:
log.warning(f"Could not auto-clean stale voices: {ex}")
return jsonify(result)
@@ -670,25 +662,6 @@ def api_chatterbox_delete_voice(filename):
return jsonify({"error": f"Server returned {r.status_code}"}), 500
except Exception as ex:
return jsonify({"error": str(ex)}), 500
# Immediately remove from settings.json so the UI updates without waiting for bot
try:
p = COG_CONFIG_PATH / "settings.json"
data = _read_json(p)
if data:
ucfg = data.get(COG_IDENTIFIER, {}).get("USER", {}).get(str(user_id), {})
voices = ucfg.get("chatterbox_voices", [])
if filename in voices:
voices.remove(filename)
ucfg["chatterbox_voices"] = voices
# If active voice was deleted, switch to first remaining or clear
if ucfg.get("minimax_voice") == filename:
ucfg["minimax_voice"] = voices[0] if voices else None
with open(p, "w") as f:
json.dump(data, f, indent=2)
except Exception as ex:
log.warning(f"Could not update settings.json directly: {ex}")
_write_bot_command(user_id, "remove_voice", {"filename": filename})
return jsonify({"ok": True})