From 79d149f3b8754b10101e6011d3e2508c6f86ca2a Mon Sep 17 00:00:00 2001 From: Kingston-SCYD Date: Tue, 9 Jun 2026 16:06:54 -0500 Subject: [PATCH] feat: add minimax mode to /api/mc/tts endpoint --- ttstoy/webui/app.py | 33 ++++++++++++++++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/ttstoy/webui/app.py b/ttstoy/webui/app.py index b98c372..33a2099 100644 --- a/ttstoy/webui/app.py +++ b/ttstoy/webui/app.py @@ -686,7 +686,7 @@ def api_mc_tts(): r"(|:[A-Za-z0-9_]+:|[\U0001F300-\U0001FAFF\U00002600-\U000026FF\U00002700-\U000027BF][\uFE0E\uFE0F]?(?:\u200D[\U0001F300-\U0001FAFF\U00002600-\U000026FF\U00002700-\U000027BF][\uFE0E\uFE0F]?)*)" ) VOICE_SWITCH_RE = re.compile(r"\[([a-zA-Z]+)(?:\|([^\]]+))?\]") - VALID_MODES = {"chatterbox", "clone", "dectalk", "morshu", "vox"} + VALID_MODES = {"chatterbox", "clone", "dectalk", "morshu", "vox", "minimax"} # Turbo tokens — these look like voice switches but should pass through to chatterbox TURBO_TOKENS = {"laugh", "chuckle", "sigh", "gasp", "cough", "cry", "groan", "yawn", "sniff"} @@ -738,6 +738,37 @@ def api_mc_tts(): r = requests.get("http://127.0.0.1:33003/say", params={"text": seg_text, "pack": pack}, timeout=60) r.raise_for_status() return r.content + elif mode == "minimax": + api_key = gcfg.get("minimax_api_key") + if not api_key: + raise RuntimeError("MiniMax API key not configured in ttstoy") + model = gcfg.get("minimax_model", "speech-01-turbo") + payload = { + "model": model, + "text": seg_text, + "stream": False, + "language_boost": "auto", + "output_format": "hex", + "voice_setting": { + "voice_id": voice or "male-qn-qingse", + "speed": 1, + "vol": 1, + "pitch": 0, + }, + "audio_setting": { + "sample_rate": 32000, + "bitrate": 128000, + "format": "mp3", + }, + } + headers = {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"} + r = requests.post("https://api.minimax.io/v1/t2a_v2", json=payload, headers=headers, timeout=60) + r.raise_for_status() + resp = r.json() + audio_hex = resp.get("data", {}).get("audio", "") + if not audio_hex: + raise RuntimeError("MiniMax returned no audio") + return bytes.fromhex(audio_hex) else: raise RuntimeError(f"Unknown mode: {mode}")