Files
scrapyard-cogworks/ttstoy/ttstoy.py
T
kingston 9ca1511f82 ttstoy: decouple TTS services from the bot lifecycle
The dectalk, morshu, and vox TTS servers now run as independent systemd
services on their fixed ports (33001/33002/33003). The cog no longer spawns
or kills them: start/stop helpers became reachability checks and no-ops, the
cog_load auto-start and cog_unload teardown were removed, and the start/stop/
status commands now point at systemctl. The HTTP client calls are unchanged.
2026-09-14 14:35:53 -05:00

3375 lines
143 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import os
import sys
import shutil
import logging
import asyncio
import subprocess
import re
import threading
import json
from http.server import BaseHTTPRequestHandler, HTTPServer
from pathlib import Path
import lavalink
from typing import Optional
try:
import fcntl as _fcntl
except ImportError: # pragma: no cover - Windows
_fcntl = None
import discord
from discord.ext import tasks
import requests
from redbot.core import commands, Config
from redbot.core.bot import Red, cog_data_path
from redbot.core.commands import Cog
from redbot.cogs.audio import Audio
try:
from .morshutalk_engine import Morshu
MORSHU_AVAILABLE = True
except Exception:
try:
from morshutalk_engine import Morshu
MORSHU_AVAILABLE = True
except Exception:
MORSHU_AVAILABLE = False
try:
from .vox_engine import generate_vox, get_available_words, get_available_packs
VOX_AVAILABLE = True
except Exception:
try:
from vox_engine import generate_vox, get_available_words, get_available_packs
VOX_AVAILABLE = True
except Exception:
VOX_AVAILABLE = False
log = logging.getLogger("red.ttstoy")
# ----------------------------
# MiniMax defaults
# ----------------------------
DEFAULT_MINIMAX_MODEL = "speech-01-turbo"
DEFAULT_MINIMAX_VOICE = "moss_audio_b304011b-11b2-11f1-90fe-36953c023630"
MINIMAX_VOICES = {
"BigMan": "moss_audio_54f969bf-12c1-11f1-93de-a6f4120d2cc7",
"BlueGnome": "moss_audio_166dfe7c-1390-11f1-b6f2-729162d0a8d2",
"Dracafow": "moss_audio_13cb66d2-12c7-11f1-b6f2-729162d0a8d2",
"Gaben": "moss_audio_5239ca93-115b-11f1-b9c4-4ea5324904c7",
"Grigori": "moss_audio_63a397aa-24c9-11f1-918f-5a2de67f838a",
"Gregori": "moss_audio_63a397aa-24c9-11f1-918f-5a2de67f838a",
"Gnome": "moss_audio_b304011b-11b2-11f1-90fe-36953c023630",
"King": "moss_audio_39c16b2c-1158-11f1-995d-b25e2f30d3fa",
"Peppa": "moss_audio_d656489f-114b-11f1-995d-b25e2f30d3fa",
"Robotnik": "moss_audio_123443a8-11b4-11f1-bfa6-763108879732",
}
DECTALK_VOICES = {
"paul": "paul", # Perfect Paul (default, male)
"betty": "betty", # Beautiful Betty (female)
"harry": "harry", # Huge Harry (male)
"frank": "frank", # Frail Frank (male)
"dennis": "dennis", # Doctor Dennis (male)
"kit": "kit", # Kit the Kid (child)
"ursula": "ursula", # Uppity Ursula (female)
"rita": "rita", # Rough Rita (female)
"wendy": "wendy", # Whispering Wendy (female)
}
CHATTERBOX_DEFAULT_URL = "http://127.0.0.1:8099"
_DEFAULT_EMOJI_SFX = {
"🎉": "party",
"😂": "laugh",
"🥖": "spy",
"👏": "clap",
"🔥": "fire",
"💀": "skull",
"✅": "check",
"❌": "error",
"📢": "airhorn",
"🚢": "boathorn",
"😶": "drum",
"👼": "angel",
"🥜": "cashew",
"💪": "physical",
"🧠": "intelligence",
"👁": "psychic",
"✍️": "motor",
}
_SFX_MAP_PATH = Path(__file__).resolve().parent / "sfx" / "emoji_map.json"
def _load_emoji_sfx_map() -> dict:
"""Load emoji→folder mapping from JSON file, falling back to defaults."""
try:
if _SFX_MAP_PATH.exists():
with open(_SFX_MAP_PATH) as f:
return json.load(f)
except Exception:
pass
return dict(_DEFAULT_EMOJI_SFX)
def _save_emoji_sfx_map(mapping: dict):
"""Persist emoji→folder mapping to JSON file."""
_SFX_MAP_PATH.parent.mkdir(parents=True, exist_ok=True)
with open(_SFX_MAP_PATH, "w") as f:
json.dump(mapping, f, ensure_ascii=False, indent=2)
# Live reference — all code reads from this dict; _reload_emoji_sfx_map() refreshes it.
EMOJI_SFX_FOLDERS: dict = _load_emoji_sfx_map()
def _reload_emoji_sfx_map():
"""Reload the mapping from disk into the live dict."""
EMOJI_SFX_FOLDERS.clear()
EMOJI_SFX_FOLDERS.update(_load_emoji_sfx_map())
EMOJI_CANDIDATE_PATTERN = re.compile(
r"(<a?:\w+:\d+>|:[A-Za-z0-9_]+:|[\U0001F300-\U0001FAFF\U00002600-\U000026FF\U00002700-\U000027BF][\uFE0E\uFE0F]?(?:\u200D[\U0001F300-\U0001FAFF\U00002600-\U000026FF\U00002700-\U000027BF][\uFE0E\uFE0F]?)*)"
)
# Pattern for inline voice switches: [mode|voice] or [mode] (voice optional)
VOICE_SWITCH_PATTERN = re.compile(
r"\[([a-zA-Z]+)(?:\|([^\]]+))?\]"
)
VALID_VOICE_MODES = {"minimax", "chatterbox", "dectalk", "morshu", "vox"}
def resolve_voice_for_mode(mode: str, voice_name: str, chatterbox_voices: list = None) -> str:
"""Resolve a voice name/label to the engine-specific voice ID, case-insensitive.
For chatterbox, pass the list of known voice filenames (e.g. ['123_Emily.wav']).
The function matches by the display portion of the filename (after the first underscore,
without extension), case-insensitively.
"""
name_lower = voice_name.lower().strip()
if mode == "minimax":
for label, vid in MINIMAX_VOICES.items():
if label.lower() == name_lower:
return vid
return voice_name # pass through raw IDs
if mode == "dectalk":
return DECTALK_VOICES.get(name_lower, name_lower)
if mode == "chatterbox" and chatterbox_voices:
# Try exact match first
for fn in chatterbox_voices:
if fn.lower() == name_lower or fn == voice_name:
return fn
# Match by display name (the part after userid_)
for fn in chatterbox_voices:
parts = fn.split("_", 1)
if len(parts) > 1:
display = parts[1].rsplit(".", 1)[0] # strip extension
else:
display = fn.rsplit(".", 1)[0]
if display.lower() == name_lower:
return fn
return voice_name
def _normalize_chatterbox_reference(
audio_seg, min_ms: int = 6000, max_ms: int = 30000
):
"""
Normalize a voice reference clip to a usable length for Chatterbox cloning.
- Clips shorter than ``min_ms`` are looped until they reach the minimum
duration required for a usable voice reference (>5s).
- Clips longer than ``max_ms`` are trimmed to the server's
``max_reference_duration_sec`` limit, preferring a silence boundary near
the cut so words aren't chopped mid-syllable.
Returns the processed AudioSegment.
"""
if audio_seg is None:
return audio_seg
if len(audio_seg) < min_ms:
loops_needed = (min_ms // len(audio_seg)) + 1 if len(audio_seg) > 0 else 1
audio_seg = audio_seg * loops_needed
log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s")
if len(audio_seg) > max_ms:
trimmed = audio_seg[:max_ms]
try:
from pydub.silence import detect_silence
search_start = max(0, max_ms - 1500)
silences = detect_silence(
audio_seg[search_start:max_ms],
min_silence_len=300,
silence_thresh=-40,
)
if silences:
cutoff_local = silences[0][0]
if cutoff_local > 100:
trimmed = audio_seg[: search_start + cutoff_local]
except Exception:
pass
log.info(
f"[Chatterbox] Trimmed long clip from {len(audio_seg)/1000:.1f}s "
f"to {len(trimmed)/1000:.1f}s"
)
audio_seg = trimmed
return audio_seg
def _split_voice_segments(text: str):
"""
Split text on [mode|voice] tags into segments.
Returns a list of (mode_override, voice_override, text) tuples.
mode_override/voice_override are None for the initial segment (use caller defaults).
"""
segments = []
last_end = 0
for m in VOICE_SWITCH_PATTERN.finditer(text):
# Text before this tag belongs to the current (last) voice segment
before = text[last_end:m.start()]
if before:
if not segments:
segments.append((None, None, before))
else:
prev = segments[-1]
segments[-1] = (prev[0], prev[1], prev[2] + before)
mode_tag = m.group(1).lower()
voice_tag = (m.group(2) or "").strip()
if mode_tag in VALID_VOICE_MODES:
segments.append((mode_tag, voice_tag, ""))
else:
# Invalid mode, treat the whole tag as literal text
if not segments:
segments.append((None, None, m.group(0)))
else:
prev = segments[-1]
segments[-1] = (prev[0], prev[1], prev[2] + m.group(0))
last_end = m.end()
# Remaining text after the last tag
tail = text[last_end:]
if tail:
if not segments:
segments.append((None, None, tail))
else:
prev = segments[-1]
segments[-1] = (prev[0], prev[1], prev[2] + tail)
if not segments:
segments.append((None, None, text))
return segments
def resolve_minimax_voice_id(label_or_id: str) -> str:
return MINIMAX_VOICES.get(label_or_id, label_or_id)
def resolve_dectalk_voice_id(label_or_id: str) -> str:
# Case-insensitive lookup
label_lower = label_or_id.lower()
return DECTALK_VOICES.get(label_lower, label_lower)
def _normalize_sfx_trigger(token: str) -> str:
if token.startswith("<") and token.endswith(">"):
# Discord custom emoji format: <:name:id> or <a:name:id>
parts = token.strip("<>").split(":")
if len(parts) == 3:
return f":{parts[1]}:"
return token
# ---------------------------------------------------------------------
# HEALTH SERVER
# ---------------------------------------------------------------------
TTSTOY_HEALTH_PORT = 8097
class _HealthHandler(BaseHTTPRequestHandler):
cog_ref = None # set by TtsToy
def do_GET(self):
if self.path == "/health":
cog = _HealthHandler.cog_ref
payload = {
"status": "ok",
"service": "ttstoy",
}
body = json.dumps(payload).encode()
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.send_header("Access-Control-Allow-Origin", "*")
self.end_headers()
self.wfile.write(body)
else:
self.send_response(404)
self.end_headers()
def log_message(self, format, *args):
pass # suppress access logs
# ---------------------------------------------------------------------
# MODAL FOR API KEY ENTRY
# ---------------------------------------------------------------------
class MinimaxKeyModal(discord.ui.Modal, title="Set MiniMax API Key"):
api_key = discord.ui.TextInput(
label="Enter MiniMax API Key",
style=discord.TextStyle.short,
placeholder="paste your API key",
required=True,
)
def __init__(self, config: Config):
super().__init__()
self.config = config
async def on_submit(self, interaction: discord.Interaction):
await self.config.minimax_api_key.set(self.api_key.value.strip())
await interaction.response.send_message("✅ MiniMax API key saved.", ephemeral=True)
class MinimaxKeyButton(discord.ui.View):
def __init__(self, config: Config):
super().__init__(timeout=None)
self.config = config
@discord.ui.button(label="Enter MiniMax API Key", style=discord.ButtonStyle.primary)
async def enter_key(self, interaction: discord.Interaction, button: discord.ui.Button): # type: ignore[override]
modal = MinimaxKeyModal(self.config)
await interaction.response.send_modal(modal)
# ---------------------------------------------------------------------
# MAIN COG
# ---------------------------------------------------------------------
class TtsToy(Cog):
"""Multi-engine TTS with voice cloning, emoji SFX, and a web UI.
Engines: MiniMax · Chatterbox · DECTalk · Morshu · VOX
"""
def __init__(self, bot: Red):
super().__init__()
self.bot = bot
self.tts_storage = cog_data_path(cog_instance=self).joinpath("ttstoy")
self.sfx_root = Path(__file__).resolve().parent / "sfx"
self.dectalk_dir = Path(__file__).resolve().parent / "dectalk-server"
self.dectalk_process = None
self.dectalk_port = 33001
# Morshu TTS engine + server
self.morshu = Morshu() if MORSHU_AVAILABLE else None
self.morshu_dir = Path(__file__).resolve().parent / "morshu-server"
self.morshu_process = None
self.morshu_port = 33002
# VOX server
self.vox_dir = Path(__file__).resolve().parent / "vox-server"
self.vox_process = None
self.vox_port = 33003
# TTS queue system - one queue per guild
self.tts_queues = {} # guild_id -> asyncio.Queue
self.tts_locks = {} # guild_id -> asyncio.Lock
self.tts_processors = {} # guild_id -> asyncio.Task
self._ensure_sfx_folders()
self.clear_old_tts.start()
self._health_server: HTTPServer | None = None
self._health_thread: threading.Thread | None = None
self.config: Config = Config.get_conf(
self, identifier=0x0A0A0A0A, force_registration=True
)
self.config.register_global(
minimax_api_key=None,
minimax_model=DEFAULT_MINIMAX_MODEL,
minimax_voice=DEFAULT_MINIMAX_VOICE,
sfx_volume=100,
tts_mode="minimax",
chatterbox_api_url=CHATTERBOX_DEFAULT_URL,
dectalk_api_url=f"http://127.0.0.1:33001",
dectalk_auto_start=True,
accessibility_mode=False,
vox_pack="vox",
webui_tts_channel_id=None, # channel to post web UI TTS transcripts/audio
webui_public_url="https://ttstoy.kingstons-scrapyard.net", # public URL shown in login DMs
)
self.config.register_guild(
webui_tts_channel_id=None, # per-guild channel for web UI TTS posts
)
self.config.register_user(
minimax_voice=None,
chatterbox_voices=[], # list of voice filenames this user has uploaded
chatterbox_temperature=None, # DEPRECATED: kept for migration
chatterbox_exaggeration=None, # DEPRECATED: kept for migration
chatterbox_temperature_offsets={}, # voice_name -> temperature (e.g. {"Emily.wav": 0.8})
chatterbox_exaggeration_offsets={}, # voice_name -> exaggeration (e.g. {"Emily.wav": 1.5})
chatterbox_volume_offsets={}, # voice_name -> dB offset (e.g. {"Emily.wav": -3.0})
chatterbox_speed_offsets={}, # voice_name -> speed factor (e.g. {"Emily.wav": 1.2})
)
async def red_delete_data_for_user(self, **kwargs):
# only per-user voice; nothing special to wipe beyond config
pass
# -----------------------------------------
@tasks.loop(hours=1)
async def clear_old_tts(self):
try:
if os.path.exists(self.tts_storage):
shutil.rmtree(self.tts_storage)
except OSError:
log.exception("Trying to clear old TTS audio files")
async def cog_unload(self):
self.clear_old_tts.stop()
if self._webui_state_processor.is_running():
self._webui_state_processor.stop()
self._kill_leftover_webui()
# TTS services (dectalk/morshu/vox) now run as independent systemd
# services and are no longer started or stopped by this cog.
if self._health_server:
self._health_server.shutdown()
self._health_server = None
# Cancel all TTS processor tasks
for task in self.tts_processors.values():
if not task.done():
task.cancel()
# Wait for all tasks to finish
if self.tts_processors:
await asyncio.gather(*self.tts_processors.values(), return_exceptions=True)
def _kill_leftover_webui(self):
"""Terminate any web UI subprocess left over from before the web UI
moved out of the bot. It ran from <cog_dir>/webui/app.py, which no
longer exists — so a matching process is guaranteed stale."""
try:
pattern = str(Path(__file__).resolve().parent / "webui" / "app.py")
result = subprocess.run(
["pkill", "-f", pattern], capture_output=True, text=True
)
if result.returncode == 0:
log.info("[WebUI] Killed leftover web UI subprocess")
except Exception as e:
log.warning(f"[WebUI] Could not kill leftover web UI subprocess: {e}")
# -----------------------------------------
async def _get_minimax_settings(self):
return (
await self.config.minimax_api_key(),
await self.config.minimax_model(),
await self.config.minimax_voice(),
)
async def _get_tts_mode(self):
"""Get current TTS mode."""
mode = await self.config.tts_mode()
# Migrate old "local" mode to "chatterbox"
if mode == "local":
await self.config.tts_mode.set("chatterbox")
return "chatterbox"
return mode
async def _get_chatterbox_api_url(self):
"""Get Chatterbox TTS Server API URL."""
return await self.config.chatterbox_api_url()
async def _get_dectalk_api_url(self):
"""Get DECTalk API URL."""
return await self.config.dectalk_api_url()
async def _install_dectalk_server(self) -> bool:
"""Install DECTalk server dependencies."""
try:
if not self.dectalk_dir.exists():
log.error(f"DECTalk server directory not found: {self.dectalk_dir}")
return False
# Run npm install
result = await asyncio.to_thread(
subprocess.run,
["npm", "install"],
cwd=str(self.dectalk_dir),
capture_output=True,
text=True,
timeout=300
)
if result.returncode != 0:
log.error(f"npm install failed: {result.stderr}")
return False
log.info("DECTalk server installed successfully")
return True
except Exception as e:
log.exception(f"Failed to install DECTalk server: {e}")
return False
async def _start_dectalk_server(self) -> bool:
"""Check whether the independent DECTalk service is reachable.
The DECTalk server now runs as a standalone systemd service
(dectalk-tts-server) and is no longer spawned by the bot. This method
only confirms the service is up on its fixed port.
"""
try:
def _check():
try:
r = requests.get(
f"http://127.0.0.1:{self.dectalk_port}/health", timeout=5
)
return r.status_code == 200
except Exception:
return False
reachable = await asyncio.to_thread(_check)
if reachable:
log.info(f"DECTalk service reachable on port {self.dectalk_port}")
else:
log.warning(
f"DECTalk service not reachable on port {self.dectalk_port}. "
"Ensure the dectalk-tts-server systemd service is running."
)
return reachable
except Exception as e:
log.exception(f"Failed to check DECTalk service: {e}")
return False
async def _stop_dectalk_server(self):
"""No-op. The DECTalk service is managed by systemd, not the bot."""
return
async def _start_morshu_server(self) -> tuple:
"""Check whether the independent Morshu service is reachable.
The Morshu server now runs as a standalone systemd service
(morshu-tts-server) and is no longer spawned by the bot. Returns
(reachable, error_msg).
"""
try:
def _check():
try:
r = requests.get(
f"http://127.0.0.1:{self.morshu_port}/health", timeout=5
)
return r.status_code == 200
except Exception:
return False
reachable = await asyncio.to_thread(_check)
if reachable:
log.info(f"Morshu service reachable on port {self.morshu_port}")
return True, None
return False, (
f"Morshu service not reachable on port {self.morshu_port}. "
"Ensure the morshu-tts-server systemd service is running."
)
except Exception as e:
log.exception(f"Failed to check Morshu service: {e}")
return False, str(e)
async def _stop_morshu_server(self):
"""No-op. The Morshu service is managed by systemd, not the bot."""
return
async def _start_vox_server(self) -> tuple:
"""Check whether the independent VOX service is reachable.
The VOX server now runs as a standalone systemd service
(vox-tts-server) and is no longer spawned by the bot. Returns
(reachable, error_msg).
"""
try:
def _check():
try:
r = requests.get(
f"http://127.0.0.1:{self.vox_port}/health", timeout=5
)
return r.status_code == 200
except Exception:
return False
reachable = await asyncio.to_thread(_check)
if reachable:
log.info(f"VOX service reachable on port {self.vox_port}")
return True, None
return False, (
f"VOX service not reachable on port {self.vox_port}. "
"Ensure the vox-tts-server systemd service is running."
)
except Exception as e:
log.exception(f"Failed to check VOX service: {e}")
return False, str(e)
async def _stop_vox_server(self):
"""No-op. The VOX service is managed by systemd, not the bot."""
return
async def _get_effective_minimax_voice(self, user: discord.abc.User) -> str:
mode = await self._get_tts_mode()
# DECTalk doesn't use voice selection - users control voices with [:n*] commands
if mode == "dectalk":
return "paul" # Default, but users can override with commands
# Morshu doesn't use voice selection - it's always Morshu
if mode == "morshu":
return "morshu"
# VOX doesn't use voice selection - pack is chosen separately
if mode == "vox":
return "vox"
# Chatterbox supports per-user voice selection
if mode == "chatterbox":
u = await self.config.user(user).minimax_voice()
# Only use saved voice if it looks like a Chatterbox filename (has a file extension)
if u and ("." in u):
return u
g = await self.config.minimax_voice()
if g and ("." in g):
return g
return "Emily.wav"
u = await self.config.user(user).minimax_voice()
if u:
return resolve_minimax_voice_id(u)
g = await self.config.minimax_voice()
return resolve_minimax_voice_id(g)
async def _get_all_chatterbox_voices(self) -> list:
"""Collect all chatterbox voice filenames across all users."""
all_voices = []
all_users = await self.config.all_users()
for uid, udata in all_users.items():
voices = udata.get("chatterbox_voices", [])
all_voices.extend(voices)
return all_voices
def _save_minimax_tts(self, api_key, model, voice_id, text, audio_path, mode="minimax", dectalk_url=None):
"""
Save TTS audio using MiniMax API or DECTalk API.
Args:
api_key: API key (required for MiniMax)
model: Model name
voice_id: Voice identifier
text: Text to synthesize
audio_path: Output file path
mode: "minimax" or "dectalk"
dectalk_url: DECTalk API URL (required if mode="dectalk")
Returns:
dict: Response data including detected language (if available)
"""
if mode == "dectalk":
if not dectalk_url:
raise RuntimeError("DECTalk API URL not configured")
url = f"{dectalk_url.rstrip('/')}/v1/t2a_v2"
headers = {"Content-Type": "application/json"}
else:
url = "https://api.minimax.io/v1/t2a_v2"
headers = {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"}
payload = {
"model": model,
"text": text,
"stream": False,
"language_boost": "auto",
"output_format": "hex",
"voice_setting": {
"voice_id": voice_id,
"speed": 1,
"vol": 1,
"pitch": 0,
},
"audio_setting": {
"sample_rate": 32000,
"bitrate": 128000,
"format": "mp3",
"channel": 1,
},
}
r = requests.post(url, headers=headers, json=payload, timeout=60)
r.raise_for_status()
data = r.json()
base = data.get("base_resp") or {}
if base.get("status_code") != 0:
raise RuntimeError(
f"MiniMax error {base.get('status_code')}: {base.get('status_msg')}"
)
hex_audio = data.get("data", {}).get("audio")
if not hex_audio:
raise RuntimeError("MiniMax returned no audio data.")
with open(audio_path, "wb") as f:
f.write(bytes.fromhex(hex_audio))
# Return response data for language detection
return data
def _save_morshu_tts(self, text: str, audio_path: str):
"""
Generate TTS audio using MorshuTalk server (HTTP).
Falls back to in-process engine if server is unavailable.
Args:
text: Text to synthesize
audio_path: Output file path
"""
morshu_url = "http://127.0.0.1:33002"
try:
r = requests.get(f"{morshu_url}/say", params={"text": text}, timeout=60)
r.raise_for_status()
with open(audio_path, "wb") as f:
f.write(r.content)
return
except Exception as e:
log.warning(f"[Morshu] Server unavailable ({e}), falling back to in-process engine")
# Fallback to in-process
if not self.morshu:
raise RuntimeError("MorshuTalk engine is not available. Start the morshu-server or install g2p_en, numpy, pydub.")
audio = self.morshu.load_text(text)
if audio is False or audio is None:
raise RuntimeError("MorshuTalk failed to generate audio.")
audio.export(audio_path, format="mp3", bitrate="128k")
def _save_vox_tts(self, text: str, audio_path: str, pack: str = "vox"):
"""
Generate TTS audio using Black Mesa VOX engine.
Args:
text: Text to synthesize (words matched against VOX dictionary)
audio_path: Output file path
pack: VOX pack to use ("vox" or "vox2")
"""
if not VOX_AVAILABLE:
raise RuntimeError("VOX engine is not available.")
generate_vox(text, audio_path, pack=pack)
def _save_chatterbox_tts(self, text: str, audio_path: str, chatterbox_url: str, voice: str = "Emily.wav", temperature: float = None, exaggeration: float = None, volume_db: float = 0.0, speed_factor: float = None):
"""
Generate TTS audio using Chatterbox TTS Server's /tts endpoint.
"""
url = f"{chatterbox_url.rstrip('/')}/tts"
payload = {
"text": text,
"voice_mode": "predefined",
"predefined_voice_id": voice if voice else "Emily.wav",
"output_format": "wav",
"split_text": False,
}
if temperature is not None:
payload["temperature"] = temperature
if exaggeration is not None:
payload["exaggeration"] = exaggeration
if speed_factor is not None:
payload["speed_factor"] = speed_factor
log.info(f"[Chatterbox] POST {url} | voice={voice} | temp={temperature} | exag={exaggeration} | vol_db={volume_db} | speed={speed_factor} | text={text[:80]!r}")
try:
r = requests.post(url, json=payload, timeout=120)
log.info(f"[Chatterbox] Response: {r.status_code} | {len(r.content)} bytes")
r.raise_for_status()
except requests.exceptions.HTTPError as e:
# Log response body for debugging
try:
body = r.text[:500]
except Exception:
body = "(could not read body)"
log.error(f"[Chatterbox] HTTP {r.status_code} error. Body: {body}")
if r.status_code == 500 and "failed to synthesize" in body.lower():
raise RuntimeError(
f"Chatterbox couldn't generate audio with voice `{voice}`. "
"The voice clip may be too short, too long, or in a bad format. "
"Try uploading a clear 5-15 second .wav clip."
)
raise
if len(r.content) < 100:
raise RuntimeError(f"Chatterbox returned too little audio data ({len(r.content)} bytes)")
# Apply volume offset if non-zero
if volume_db and volume_db != 0.0:
tmp_raw = audio_path + ".raw.wav"
with open(tmp_raw, "wb") as f:
f.write(r.content)
try:
subprocess.run(
[
"ffmpeg", "-y", "-i", tmp_raw,
"-filter:a", f"volume={volume_db}dB",
"-vn", audio_path,
],
check=True, capture_output=True, text=True,
)
finally:
if os.path.exists(tmp_raw):
os.remove(tmp_raw)
else:
with open(audio_path, "wb") as f:
f.write(r.content)
def _ensure_sfx_folders(self):
self.sfx_root.mkdir(parents=True, exist_ok=True)
# Seed the JSON map file if it doesn't exist yet
if not _SFX_MAP_PATH.exists():
_save_emoji_sfx_map(dict(_DEFAULT_EMOJI_SFX))
for folder in sorted(set(EMOJI_SFX_FOLDERS.values())):
(self.sfx_root / folder).mkdir(parents=True, exist_ok=True)
def _get_sfx_source_for_trigger(self, trigger: str) -> Optional[Path]:
normalized = _normalize_sfx_trigger(trigger)
folder_name = EMOJI_SFX_FOLDERS.get(normalized)
if not folder_name:
return None
folder_path = self.sfx_root / folder_name
folder_path.mkdir(parents=True, exist_ok=True)
# Search for any supported audio format
for ext in ("*.mp3", "*.wav", "*.ogg"):
candidates = sorted(folder_path.glob(ext))
if candidates:
return candidates[0]
return None
def _split_prompt_segments(self, text: str):
"""
Split prompt into ordered TTS/SFX segments.
SFX triggers are read from EMOJI_SFX_FOLDERS and support:
- Unicode emoji keys (example: "🎉")
- Colon-name keys (example: ":airhorn:")
- Discord custom emoji tokens in prompts (`<:airhorn:123>` / `<a:airhorn:123>`)
when their normalized `:airhorn:` key exists in EMOJI_SFX_FOLDERS.
"""
segments = []
text_buffer = []
def flush_text_buffer():
if text_buffer:
combined = "".join(text_buffer)
if combined:
segments.append(("tts", combined))
text_buffer.clear()
cursor = 0
for match in EMOJI_CANDIDATE_PATTERN.finditer(text):
if match.start() > cursor:
text_buffer.append(text[cursor : match.start()])
token = match.group(0)
normalized = _normalize_sfx_trigger(token)
if normalized in EMOJI_SFX_FOLDERS:
flush_text_buffer()
segments.append(("sfx", token))
else:
text_buffer.append(token)
cursor = match.end()
if cursor < len(text):
text_buffer.append(text[cursor:])
flush_text_buffer()
return segments
def _render_sfx_part(self, sfx_source: Path, output_path: str, volume_percent: int):
volume_multiplier = max(0, min(100, volume_percent)) / 100
cmd = [
"ffmpeg",
"-y",
"-i",
str(sfx_source),
"-filter:a",
f"volume={volume_multiplier}",
"-vn",
"-ac",
"1",
"-ar",
"32000",
"-b:a",
"128k",
output_path,
]
subprocess.run(cmd, check=True, capture_output=True, text=True)
def _concat_audio_parts(self, part_paths, output_path: str):
if len(part_paths) == 1:
shutil.copyfile(part_paths[0], output_path)
return
cmd = ["ffmpeg", "-y"]
for path in part_paths:
cmd.extend(["-i", path])
filter_inputs = "".join(f"[{i}:a]" for i in range(len(part_paths)))
cmd.extend(
[
"-filter_complex",
f"{filter_inputs}concat=n={len(part_paths)}:v=0:a=1[outa]",
"-map",
"[outa]",
"-ac",
"1",
"-ar",
"32000",
"-b:a",
"128k",
output_path,
]
)
subprocess.run(cmd, check=True, capture_output=True, text=True)
def _save_prompt_audio(self, api_key, model, voice_id, text, output_path: str, sfx_volume: int, mode="minimax", dectalk_url=None, vox_pack="vox", chatterbox_url=None, chatterbox_temperature=None, chatterbox_exaggeration=None, chatterbox_volume_db=0.0, chatterbox_speed=None, all_chatterbox_voices=None):
"""
Save prompt audio with SFX and multi-voice support.
Supports inline voice switches via [mode|voice] or [mode] tags, e.g.:
[minimax|robotnik] Hello [dectalk] [:nh]Deep voice here
Returns:
dict: Response data including detected language (if available)
"""
voice_segments = _split_voice_segments(text)
has_voice_switches = len(voice_segments) > 1 or voice_segments[0][0] is not None
# Fast path: no voice switches — use the legacy single-engine path
if not has_voice_switches:
return self._render_single_mode(
api_key, model, voice_id, text, output_path, sfx_volume,
mode, dectalk_url, vox_pack, chatterbox_url,
chatterbox_temperature, chatterbox_exaggeration,
chatterbox_volume_db, chatterbox_speed,
)
# Multi-voice path: render each voice segment separately, then concat
self.tts_storage.mkdir(parents=True, exist_ok=True)
temp_dir = os.path.join(self.tts_storage, f"parts_{os.path.basename(output_path)}")
os.makedirs(temp_dir, exist_ok=True)
all_parts = []
try:
for seg_idx, (seg_mode, seg_voice, seg_text) in enumerate(voice_segments):
seg_text = seg_text.strip()
if not seg_text:
continue
# Resolve effective mode and voice for this segment
eff_mode = seg_mode or mode
if seg_voice:
eff_voice = resolve_voice_for_mode(eff_mode, seg_voice, all_chatterbox_voices)
else:
eff_voice = voice_id
# Split this segment's text on SFX emojis
sfx_segments = self._split_prompt_segments(seg_text)
if not sfx_segments:
continue
for part_idx, (kind, value) in enumerate(sfx_segments):
part_path = os.path.join(temp_dir, f"seg{seg_idx:02}_{part_idx:03}.mp3")
if kind == "sfx":
sfx_source = self._get_sfx_source_for_trigger(value)
if sfx_source:
self._render_sfx_part(sfx_source, part_path, sfx_volume)
all_parts.append(part_path)
continue
# Fallback: treat as text
value = _normalize_sfx_trigger(value)
if not value.strip():
continue
self._render_tts_part(
value, part_path, eff_mode, eff_voice,
api_key, model, dectalk_url, vox_pack, chatterbox_url,
chatterbox_temperature, chatterbox_exaggeration,
chatterbox_volume_db, chatterbox_speed,
)
all_parts.append(part_path)
if not all_parts:
raise RuntimeError("Prompt resulted in no playable audio segments.")
self._concat_audio_parts(all_parts, output_path)
finally:
shutil.rmtree(temp_dir, ignore_errors=True)
return {"language": "MULTI"}
def _render_tts_part(self, text, part_path, mode, voice_id,
api_key, model, dectalk_url, vox_pack, chatterbox_url,
cb_temp, cb_exag, cb_vol_db, cb_speed):
"""Render a single TTS text segment with the given engine and voice."""
if mode == "morshu":
self._save_morshu_tts(text, part_path)
elif mode == "vox":
self._save_vox_tts(text, part_path, pack=vox_pack or "vox")
elif mode == "chatterbox":
if not chatterbox_url:
raise RuntimeError("Chatterbox API URL not configured")
self._save_chatterbox_tts(
text, part_path, chatterbox_url, voice_id,
temperature=cb_temp, exaggeration=cb_exag,
volume_db=cb_vol_db, speed_factor=cb_speed,
)
else:
# minimax or dectalk
self._save_minimax_tts(api_key, model, voice_id, text, part_path, mode, dectalk_url)
def _render_single_mode(self, api_key, model, voice_id, text, output_path, sfx_volume,
mode, dectalk_url, vox_pack, chatterbox_url,
cb_temp, cb_exag, cb_vol_db, cb_speed):
"""Original single-mode render path (no inline voice switches)."""
# Morshu mode bypasses SFX splitting
if mode == "morshu":
self._save_morshu_tts(text, output_path)
return {"language": "MORSHU"}
# VOX mode bypasses SFX splitting
if mode == "vox":
self._save_vox_tts(text, output_path, pack=vox_pack or "vox")
return {"language": "VOX"}
# Chatterbox mode
if mode == "chatterbox":
if not chatterbox_url:
raise RuntimeError("Chatterbox API URL not configured")
segments = self._split_prompt_segments(text)
if not segments:
raise RuntimeError("Prompt had no content to synthesize.")
cb_kwargs = {"temperature": cb_temp, "exaggeration": cb_exag, "volume_db": cb_vol_db, "speed_factor": cb_speed}
if not any(kind == "sfx" for kind, _ in segments):
self._save_chatterbox_tts(text, output_path, chatterbox_url, voice_id, **cb_kwargs)
return {"language": "CHATTERBOX"}
part_paths = []
temp_dir = os.path.join(self.tts_storage, f"parts_{os.path.basename(output_path)}")
os.makedirs(temp_dir, exist_ok=True)
try:
for index, (kind, value) in enumerate(segments):
part_path = os.path.join(temp_dir, f"part_{index:03}.mp3")
if kind == "tts":
if not value.strip():
continue
self._save_chatterbox_tts(value, part_path, chatterbox_url, voice_id, **cb_kwargs)
else:
sfx_source = self._get_sfx_source_for_trigger(value)
if sfx_source is None:
normalized_text = _normalize_sfx_trigger(value)
self._save_chatterbox_tts(normalized_text, part_path, chatterbox_url, voice_id, **cb_kwargs)
else:
self._render_sfx_part(sfx_source, part_path, sfx_volume)
part_paths.append(part_path)
if not part_paths:
raise RuntimeError("Prompt resulted in no playable audio segments.")
self._concat_audio_parts(part_paths, output_path)
finally:
shutil.rmtree(temp_dir, ignore_errors=True)
return {"language": "CHATTERBOX"}
# MiniMax / DECTalk path
segments = self._split_prompt_segments(text)
if not segments:
raise RuntimeError("Prompt had no content to synthesize.")
detected_language = None
if not any(kind == "sfx" for kind, _ in segments):
response = self._save_minimax_tts(api_key, model, voice_id, text, output_path, mode, dectalk_url)
if response and isinstance(response, dict):
detected_language = response.get("data", {}).get("extra_info", {}).get("language")
return {"language": detected_language}
part_paths = []
temp_dir = os.path.join(self.tts_storage, f"parts_{os.path.basename(output_path)}")
os.makedirs(temp_dir, exist_ok=True)
try:
for index, (kind, value) in enumerate(segments):
part_path = os.path.join(temp_dir, f"part_{index:03}.mp3")
if kind == "tts":
if not value.strip():
continue
response = self._save_minimax_tts(api_key, model, voice_id, value, part_path, mode, dectalk_url)
if detected_language is None and response and isinstance(response, dict):
detected_language = response.get("data", {}).get("extra_info", {}).get("language")
else:
sfx_source = self._get_sfx_source_for_trigger(value)
if sfx_source is None:
normalized_text = _normalize_sfx_trigger(value)
response = self._save_minimax_tts(api_key, model, voice_id, normalized_text, part_path, mode, dectalk_url)
if detected_language is None and response and isinstance(response, dict):
detected_language = response.get("data", {}).get("extra_info", {}).get("language")
else:
self._render_sfx_part(sfx_source, part_path, sfx_volume)
part_paths.append(part_path)
if not part_paths:
raise RuntimeError("Prompt resulted in no playable audio segments.")
self._concat_audio_parts(part_paths, output_path)
finally:
shutil.rmtree(temp_dir, ignore_errors=True)
return {"language": detected_language}
# ---------------------------------------------------------------------
# TOP-LEVEL LOGIN (shortcut for [p]ttstoy login)
# ---------------------------------------------------------------------
@commands.command(name="login")
@commands.guild_only()
async def login_shortcut(self, ctx: commands.Context):
"""Get one-time login keys for Kingston's Scrapyard sites. Keys are DM'd to you."""
await self.webui_login(ctx)
# ---------------------------------------------------------------------
# CONFIG GROUP
# ---------------------------------------------------------------------
@commands.group(name="ttstoy", invoke_without_command=True)
async def ttstoy_group(self, ctx: commands.Context):
"""TTS Toy configuration and management.
Use `[p]help ttstoy` to see all subcommands.
"""
if ctx.invoked_subcommand is None:
await ctx.send_help(ctx.command)
@commands.group(name="chatterbox", invoke_without_command=True)
async def chatterbox_group(self, ctx: commands.Context):
"""Chatterbox TTS voice management — upload, tune, and share cloned voices.
Use `[p]help chatterbox` to see all subcommands.
"""
if ctx.invoked_subcommand is None:
await ctx.send_help(ctx.command)
@chatterbox_group.command(name="guide")
async def chatterbox_guide(self, ctx: commands.Context):
"""Full guide to Chatterbox TTS features, voice cloning, and special tokens."""
prefix = ctx.clean_prefix
part1 = [
"**🏠 Chatterbox TTS Guide**",
"",
"**── Voice Cloning ──**",
"Upload a voice clip and Chatterbox will clone it for TTS.",
f"• `{prefix}chatterbox addvoice <name>` — attach a .wav or .mp3 (5+ sec)",
f"• `{prefix}chatterbox addvoice <name> <url>` — or paste a direct link",
f"• `{prefix}chatterbox removevoice <name>` — delete a voice",
f"• `{prefix}chatterbox myvoices` — see your uploaded voices",
f"• `{prefix}ttstoy myvoice <name>` — switch your active voice",
"",
"Tips: 5–15s of clear speech, one speaker, no background noise.",
".wav works best, .mp3 accepted. Short clips auto-loop to 5s.",
"",
"**── Special Tokens (Turbo model) ──**",
"`[laugh]` `[chuckle]` `[sigh]` `[gasp]` `[cough]`",
"`[clear throat]` `[sniff]` `[groan]` `[shush]`",
f"Example: `{prefix}tts Hey [chuckle] thanks for calling back`",
]
part2 = [
"**── Voice Settings ──**",
"All saved per voice per user. Each voice keeps its own tuning.",
"",
f"• `{prefix}chatterbox temp [0.0–1.5]` — randomness",
f"• `{prefix}chatterbox exag [0.25–2.0]` — expressiveness",
f"• `{prefix}chatterbox volume [-20–20]` — dB loudness offset",
"Append `reset` to any of the above to clear it.",
"",
"**── Voice Sharing ──**",
f"• `{prefix}chatterbox sharevoice <name> @User`",
"",
"**── Quick Start ──**",
f"1. Upload: `{prefix}chatterbox addvoice MyVoice` (attach audio)",
f"2. Speak: `{prefix}tts Hello world [laugh] cloned voice`",
f"3. Tweak: `{prefix}chatterbox exag 1.5`",
]
await ctx.send("\n".join(part1))
await ctx.send("\n".join(part2))
@chatterbox_group.command(name="model")
@commands.is_owner()
async def chatterbox_model_cmd(self, ctx: commands.Context, model: Optional[str] = None):
"""
Show or switch the Chatterbox model (owner only).
- `[p]chatterbox model` → show current model
- `[p]chatterbox model turbo` → fast, supports [laugh] tags
- `[p]chatterbox model original` → better voice cloning, slower
"""
chatterbox_url = await self._get_chatterbox_api_url()
if model is None:
# Fetch current model info
try:
r = await asyncio.to_thread(
requests.get, f"{chatterbox_url.rstrip('/')}/api/model-info", timeout=5
)
r.raise_for_status()
info = r.json()
current = info.get("type", "unknown")
class_name = info.get("class_name", "unknown")
loaded = info.get("loaded", False)
except Exception as e:
return await ctx.send(f"❌ Cannot reach Chatterbox server: {e}")
lines = [
f"**Current model:** `{current}` ({class_name})",
f"**Loaded:** {'✅' if loaded else '❌'}",
"",
"Available models:",
"• `turbo` — fast, 350M params, supports [laugh]/[cough]/etc tags",
"• `original` — better voice cloning, 0.5B params, stronger emotion control",
"",
f"Switch: `{ctx.clean_prefix}chatterbox model turbo` or `{ctx.clean_prefix}chatterbox model original`",
]
return await ctx.send("\n".join(lines))
model = model.lower().strip()
model_map = {
"turbo": "chatterbox-turbo",
"original": "chatterbox",
"chatterbox-turbo": "chatterbox-turbo",
"chatterbox": "chatterbox",
}
repo_id = model_map.get(model)
if not repo_id:
return await ctx.send("❌ Model must be `turbo` or `original`.")
msg = await ctx.send(f"🔄 Switching to `{model}`... this may take a moment.")
try:
# Save the new model to config
r = await asyncio.to_thread(
requests.post,
f"{chatterbox_url.rstrip('/')}/save_settings",
json={"model": {"repo_id": repo_id}},
timeout=10,
)
r.raise_for_status()
# Trigger hot-swap
r2 = await asyncio.to_thread(
requests.post,
f"{chatterbox_url.rstrip('/')}/restart_server",
timeout=120,
)
r2.raise_for_status()
result = r2.json()
await msg.edit(content=f"✅ {result.get('message', f'Switched to {model}')}")
except Exception as e:
log.exception("Model switch error: %s", e)
await msg.edit(content=f"❌ Failed to switch model: {e}")
@ttstoy_group.command(name="sfxvolume")
@commands.is_owner()
async def sfx_volume(self, ctx: commands.Context, volume: Optional[int] = None):
"""Show or set sound effect volume (0-100)."""
if volume is None:
current = await self.config.sfx_volume()
return await ctx.send(f"Current SFX volume is `{current}`.")
if not 0 <= volume <= 100:
return await ctx.send("❌ SFX volume must be between 0 and 100.")
await self.config.sfx_volume.set(volume)
await ctx.send(f"✅ SFX volume set to `{volume}`.")
@ttstoy_group.command(name="sfx")
async def sfx_list(self, ctx: commands.Context):
"""List all available sound effects and their triggers."""
lines = ["**Available Sound Effects:**", ""]
# Group by folder and show emoji triggers
folder_to_emojis = {}
for emoji, folder in EMOJI_SFX_FOLDERS.items():
if folder not in folder_to_emojis:
folder_to_emojis[folder] = []
folder_to_emojis[folder].append(emoji)
# Check which folders have audio files
for folder_name in sorted(folder_to_emojis.keys()):
folder_path = self.sfx_root / folder_name
audio_files = list(folder_path.glob("*.mp3")) if folder_path.exists() else []
emojis = folder_to_emojis[folder_name]
emoji_str = " ".join(emojis)
if audio_files:
file_count = len(audio_files)
status = f"✅ ({file_count} file{'s' if file_count > 1 else ''})"
else:
status = "❌ (no audio)"
lines.append(f"{emoji_str} → `{folder_name}` {status}")
lines.append("")
lines.append("**Usage:** Include emoji in your TTS text")
lines.append(f"**Example:** `{ctx.clean_prefix}tts Hello 🎉 world 😂`")
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="mode")
@commands.is_owner()
async def tts_mode(self, ctx: commands.Context, mode: Optional[str] = None):
"""
Show or set TTS mode (minimax, chatterbox, dectalk, morshu, or vox).
- `[p]ttstoy mode` → show current mode
- `[p]ttstoy mode minimax` → use MiniMax API
- `[p]ttstoy mode chatterbox` → use local Chatterbox TTS server
- `[p]ttstoy mode dectalk` → use DECTalk API
- `[p]ttstoy mode morshu` → use MorshuTalk
- `[p]ttstoy mode vox` → use Black Mesa VOX announcer
"""
if mode is None:
current = await self._get_tts_mode()
dectalk_url = await self._get_dectalk_api_url()
chatterbox_url = await self._get_chatterbox_api_url()
mode_emoji = {"minimax": "🌐", "chatterbox": "🏠", "dectalk": "🤖", "morshu": "🛒", "vox": "📢"}.get(current, "❓")
lines = [
f"**Current TTS Mode: {mode_emoji} {current.upper()}**",
"",
"Available modes:",
"- `minimax` – Use MiniMax cloud API (requires API key)",
"- `chatterbox` – Use local Chatterbox TTS server (AI voice cloning)",
"- `dectalk` – Use DECTalk API (classic robotic voice)",
f"- `morshu` – Use MorshuTalk ({'available' if MORSHU_AVAILABLE else 'not installed'})",
f"- `vox` – Black Mesa VOX announcer ({'available' if VOX_AVAILABLE else 'not installed'})",
"",
]
if current == "chatterbox":
lines.append(f"Chatterbox URL: `{chatterbox_url}`")
lines.append(f"Change with: `{ctx.clean_prefix}ttstoy chatterboxurl <url>`")
elif current == "dectalk":
lines.append(f"DECTalk API URL: `{dectalk_url}`")
lines.append(f"Change with: `{ctx.clean_prefix}ttstoy dectalkurl <url>`")
elif current == "morshu":
lines.append("MorshuTalk generates speech from Morshu's voice lines.")
lines.append("No voice selection needed - it's always Morshu.")
elif current == "vox":
vox_pack = await self.config.vox_pack()
lines.append(f"Black Mesa VOX announcer. Current pack: `{vox_pack}`")
lines.append(f"Change pack: `{ctx.clean_prefix}ttstoy voxpack [vox|vox2]`")
lines.append(f"List words: `{ctx.clean_prefix}ttstoy voxwords`")
else:
lines.append("Switch to chatterbox, dectalk, morshu, or vox mode to use alternative TTS.")
return await ctx.send("\n".join(lines))
mode = mode.lower().strip()
# Accept "local" as alias for "chatterbox" for backwards compat
if mode == "local":
mode = "chatterbox"
if mode not in ["minimax", "chatterbox", "dectalk", "morshu", "vox"]:
return await ctx.send("Mode must be `minimax`, `chatterbox`, `dectalk`, `morshu`, or `vox`.")
if mode == "morshu" and not MORSHU_AVAILABLE:
return await ctx.send(
"❌ MorshuTalk engine failed to load.\n"
"Make sure `g2p_en`, `numpy`, and `pydub` are installed."
)
if mode == "vox" and not VOX_AVAILABLE:
return await ctx.send("❌ VOX engine is not available. Check bot logs for details.")
await self.config.tts_mode.set(mode)
mode_emoji = {"minimax": "🌐", "chatterbox": "🏠", "dectalk": "🤖", "morshu": "🛒", "vox": "📢"}.get(mode, "❓")
if mode == "morshu":
await ctx.send(
f"TTS mode set to {mode_emoji} **MORSHU**\n"
f"Lamp, oil, rope, bombs? You want it? It's yours, my friend."
)
elif mode == "vox":
vox_pack = await self.config.vox_pack()
await ctx.send(
f"TTS mode set to {mode_emoji} **VOX**\n"
f"Black Mesa VOX announcer (pack: `{vox_pack}`).\n"
f"Only words in the VOX dictionary will be spoken. Unknown words are skipped."
)
elif mode == "chatterbox":
chatterbox_url = await self._get_chatterbox_api_url()
await ctx.send(
f"✅ TTS mode set to {mode_emoji} **CHATTERBOX**\n"
f"Using Chatterbox TTS at: `{chatterbox_url}`\n"
f"Make sure your Chatterbox TTS server is running!"
)
elif mode == "dectalk":
dectalk_url = await self._get_dectalk_api_url()
auto_start = await self.config.dectalk_auto_start()
msg = f"✅ TTS mode set to {mode_emoji} **DECTALK**\n"
msg += f"Using DECTalk API at: `{dectalk_url}`\n"
if auto_start:
msg += "🚀 Auto-starting DECTalk server..."
await ctx.send(msg)
if await self._start_dectalk_server():
await ctx.send("✅ DECTalk server started successfully!")
else:
await ctx.send(
"❌ Failed to start DECTalk server.\n"
f"Try manually: `{ctx.clean_prefix}ttstoy dectalkstart`\n"
f"Or install: `{ctx.clean_prefix}ttstoy dectalkinstall`"
)
else:
msg += f"Make sure your DECTalk server is running!\n"
msg += f"Or enable auto-start: `{ctx.clean_prefix}ttstoy dectalkautotoggle`"
await ctx.send(msg)
else:
await ctx.send(
f"✅ TTS mode set to {mode_emoji} **MINIMAX**\n"
f"Using MiniMax cloud API. Make sure your API key is set."
)
@chatterbox_group.command(name="url")
@commands.is_owner()
async def chatterbox_url(self, ctx: commands.Context, url: Optional[str] = None):
"""
Show or set Chatterbox TTS Server URL.
- `[p]chatterbox url` → show current URL
- `[p]chatterbox url http://127.0.0.1:8099` → set URL
"""
if url is None:
current = await self._get_chatterbox_api_url()
mode = await self._get_tts_mode()
lines = [
f"**Chatterbox TTS URL:** `{current}`",
"",
]
if mode == "chatterbox":
lines.append("✅ Currently using chatterbox mode")
try:
test_url = f"{current.rstrip('/')}/api/ui/initial-data"
r = requests.get(test_url, timeout=3)
if r.status_code == 200:
lines.append("✅ Chatterbox server is reachable")
else:
lines.append(f"⚠️ Chatterbox returned status {r.status_code}")
except Exception as e:
lines.append(f"❌ Cannot reach Chatterbox: {e}")
else:
lines.append(f"ℹ️ Currently using {mode} mode")
lines.append(f"Switch with: `{ctx.clean_prefix}ttstoy mode chatterbox`")
return await ctx.send("\n".join(lines))
url = url.strip().rstrip('/')
if not url.startswith(('http://', 'https://')):
return await ctx.send("❌ URL must start with http:// or https://")
try:
test_url = f"{url}/api/ui/initial-data"
r = requests.get(test_url, timeout=5)
if r.status_code != 200:
return await ctx.send(
f"⚠️ Warning: URL returned status {r.status_code}\n"
f"Saving anyway. Make sure your Chatterbox server is running."
)
except Exception as e:
return await ctx.send(
f"⚠️ Warning: Cannot reach URL: {e}\n"
f"Saving anyway. Make sure your Chatterbox server is running."
)
await self.config.chatterbox_api_url.set(url)
await ctx.send(f"✅ Chatterbox URL set to: `{url}`")
@chatterbox_group.command(name="addvoice")
async def add_voice(self, ctx: commands.Context, name: Optional[str] = None, url: Optional[str] = None):
"""
Upload a voice clip to Chatterbox and add it to your voice list.
Attach a .wav or .mp3 file, or paste a direct link to one.
The voice is automatically set as your active voice.
- `[p]chatterbox addvoice MyVoice` (with audio attachment)
- `[p]chatterbox addvoice MyVoice https://example.com/clip.mp3`
"""
# Detect if name is actually a URL
if name and re.match(r'<?https?://', name, re.IGNORECASE):
url = name
name = None
# Strip Discord's angle-bracket embed suppression from URLs
if url:
url = url.strip("<>")
has_attachment = bool(ctx.message.attachments)
has_url = bool(url and re.match(r'https?://', url, re.IGNORECASE))
if not has_attachment and not has_url:
return await ctx.send(
"❌ Attach a `.wav` or `.mp3` audio clip, or paste a direct link.\n"
f"Example: `{ctx.clean_prefix}chatterbox addvoice CoolVoice` with a file attached\n"
f"Example: `{ctx.clean_prefix}chatterbox addvoice CoolVoice https://example.com/clip.mp3`"
)
if has_attachment:
attachment = ctx.message.attachments[0]
lower_name = attachment.filename.lower()
if not (lower_name.endswith(".wav") or lower_name.endswith(".mp3")):
return await ctx.send("❌ Only `.wav` and `.mp3` files are supported.")
source_filename = attachment.filename
else:
# Extract filename from URL path
url_path = url.split("?")[0].split("#")[0]
source_filename = url_path.rsplit("/", 1)[-1]
lower_url = source_filename.lower()
# Accept .wav, .mp3, or extensionless (we'll convert later via pydub)
if "." in source_filename and not (lower_url.endswith(".wav") or lower_url.endswith(".mp3")):
return await ctx.send("❌ URL must point to a `.wav` or `.mp3` file.")
if "." not in source_filename:
source_filename = source_filename + ".wav"
# Build filename: prefix with user ID to avoid collisions
if name:
safe_name = re.sub(r'[^a-zA-Z0-9_]', '', name)
if not safe_name:
return await ctx.send("❌ Voice name must contain at least one letter or number.")
else:
safe_name = re.sub(r'[^a-zA-Z0-9_]', '', source_filename.rsplit('.', 1)[0])
if not safe_name:
safe_name = "voice"
ext = ".wav" if source_filename.lower().endswith(".wav") else ".mp3"
filename = f"{ctx.author.id}_{safe_name}{ext}"
chatterbox_url = await self._get_chatterbox_api_url()
async with ctx.typing():
try:
if has_attachment:
audio_bytes = await ctx.message.attachments[0].read()
else:
try:
r = await asyncio.to_thread(requests.get, url, timeout=30)
r.raise_for_status()
audio_bytes = r.content
except Exception as e:
return await ctx.send(f"❌ Failed to download audio from URL: {e}")
# Ensure reference clip is a usable length for voice cloning:
# loop short clips up to the minimum duration, trim long clips to the max
try:
from pydub import AudioSegment
import io as _io
audio_seg = AudioSegment.from_file(_io.BytesIO(audio_bytes))
audio_seg = _normalize_chatterbox_reference(audio_seg)
# Always export as wav for best compatibility
buf = _io.BytesIO()
audio_seg.export(buf, format="wav")
audio_bytes = buf.getvalue()
ext = ".wav"
filename = f"{ctx.author.id}_{safe_name}{ext}"
except ImportError:
log.warning("[Chatterbox] pydub not available, skipping duration check")
except Exception as e_dur:
log.warning(f"[Chatterbox] Could not check/loop audio duration: {e_dur}")
upload_url = f"{chatterbox_url.rstrip('/')}/upload_predefined_voice"
files = {"files": (filename, audio_bytes, "audio/wav")}
r = await asyncio.to_thread(
requests.post, upload_url, files=files, timeout=30
)
r.raise_for_status()
data = r.json()
errors = data.get("errors", [])
uploaded = data.get("uploaded_files", [])
if errors and not uploaded:
error_msg = errors[0].get("error", "Unknown error")
return await ctx.send(f"❌ Upload failed: {error_msg}")
# Add to user's voice list
user_voices = await self.config.user(ctx.author).chatterbox_voices()
if filename not in user_voices:
user_voices.append(filename)
await self.config.user(ctx.author).chatterbox_voices.set(user_voices)
# Auto-set as active voice
await self.config.user(ctx.author).minimax_voice.set(filename)
await ctx.send(
f"✅ Voice `{safe_name}` uploaded and set as your active voice!\n"
f"Switch voices: `{ctx.clean_prefix}chatterbox myvoices`"
)
except Exception as e:
log.exception("Voice upload error: %s", e)
await ctx.send(f"❌ Failed to upload voice: {e}")
@chatterbox_group.command(name="removevoice")
async def remove_voice(self, ctx: commands.Context, name: str):
"""
Remove a voice from your personal voice list and delete it from the server.
- `[p]chatterbox removevoice CoolVoice`
"""
user_voices = await self.config.user(ctx.author).chatterbox_voices()
if not user_voices:
return await ctx.send("You don't have any uploaded voices.")
# Find matching voice (user can type just the name without ID prefix/extension)
match = None
for v in user_voices:
# Match by exact filename, or by the name part after the user ID prefix
parts = v.split("_", 1)
voice_name = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
if v == name or voice_name.lower() == name.lower() or v.lower() == name.lower():
match = v
break
if not match:
voice_names = ", ".join(
v.split("_", 1)[1].rsplit(".", 1)[0] if "_" in v else v.rsplit(".", 1)[0]
for v in user_voices
)
return await ctx.send(f"❌ Voice `{name}` not found in your list.\nYour voices: {voice_names}")
user_voices.remove(match)
await self.config.user(ctx.author).chatterbox_voices.set(user_voices)
# Clean up per-voice settings
for offsets_key in ("chatterbox_temperature_offsets", "chatterbox_exaggeration_offsets", "chatterbox_volume_offsets"):
offsets = await getattr(self.config.user(ctx.author), offsets_key)()
if match in offsets:
offsets.pop(match)
await getattr(self.config.user(ctx.author), offsets_key).set(offsets)
# Only delete from server if this user is the original uploader
# (filename starts with their user ID). Shared voices stay on the server.
is_owner_of_file = match.startswith(f"{ctx.author.id}_")
if is_owner_of_file:
chatterbox_url = await self._get_chatterbox_api_url()
try:
del_url = f"{chatterbox_url.rstrip('/')}/delete_predefined_voice/{match}"
r = await asyncio.to_thread(requests.delete, del_url, timeout=10)
if r.status_code not in (200, 204, 404):
log.warning(f"[Chatterbox] Failed to delete voice file {match}: HTTP {r.status_code}")
except Exception as e:
log.warning(f"[Chatterbox] Could not delete voice file {match} from server: {e}")
# If this was the active voice, clear it
current = await self.config.user(ctx.author).minimax_voice()
if current == match:
if user_voices:
await self.config.user(ctx.author).minimax_voice.set(user_voices[0])
new_name = user_voices[0].split("_", 1)[1].rsplit(".", 1)[0] if "_" in user_voices[0] else user_voices[0]
await ctx.send(f"✅ Voice `{name}` removed. Switched to `{new_name}`.")
else:
await self.config.user(ctx.author).minimax_voice.set(None)
await ctx.send(f"✅ Voice `{name}` removed. Using default voice now.")
else:
await ctx.send(f"✅ Voice `{name}` removed from your list.")
@chatterbox_group.command(name="myvoices")
async def my_voices(self, ctx: commands.Context):
"""List your uploaded Chatterbox voices and switch between them."""
mode = await self._get_tts_mode()
if mode != "chatterbox":
return await ctx.send(f"Voice management is only available in chatterbox mode.\nSwitch with: `{ctx.clean_prefix}ttstoy mode chatterbox`")
user_voices = await self.config.user(ctx.author).chatterbox_voices()
current = await self.config.user(ctx.author).minimax_voice()
if not user_voices:
return await ctx.send(
"You haven't uploaded any voices yet.\n"
f"Upload one: `{ctx.clean_prefix}chatterbox addvoice MyVoice` (attach a .wav or .mp3)"
)
lines = ["**Your Voices:**", ""]
for v in user_voices:
# Extract display name from filename (strip user ID prefix and extension)
parts = v.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
marker = " ← active" if v == current else ""
lines.append(f" `{display}`{marker}")
lines.append("")
lines.append(f"Switch: `{ctx.clean_prefix}ttstoy myvoice <name>`")
lines.append(f"Add: `{ctx.clean_prefix}chatterbox addvoice <name>` (attach audio)")
lines.append(f"Remove: `{ctx.clean_prefix}chatterbox removevoice <name>`")
await ctx.send("\n".join(lines))
@chatterbox_group.command(name="temp")
async def chatterbox_temperature(self, ctx: commands.Context, value: Optional[float] = None):
"""
Show or set Chatterbox temperature for your current voice (0.0–1.5).
Saved per voice per user — each voice remembers its own setting.
Lower = more consistent, higher = more varied.
- `[p]chatterbox temp` → show current
- `[p]chatterbox temp 0.8` → set value
- `[p]chatterbox temp reset` → use server default
"""
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_temperature_offsets()
# Migrate old global setting on first use
if not offsets:
old_val = await self.config.user(ctx.author).chatterbox_temperature()
if old_val is not None:
offsets[voice] = old_val
await self.config.user(ctx.author).chatterbox_temperature_offsets.set(offsets)
await self.config.user(ctx.author).chatterbox_temperature.set(None)
if value is None:
current = offsets.get(voice)
if current is None:
return await ctx.send(
f"Temperature for `{voice}`: `server default`\n"
f"Set with: `{ctx.clean_prefix}chatterbox temp 0.8`"
)
return await ctx.send(
f"Temperature for `{voice}`: `{current}`\n"
f"Reset with: `{ctx.clean_prefix}chatterbox temp reset`"
)
offsets[voice] = value
await self.config.user(ctx.author).chatterbox_temperature_offsets.set(offsets)
await ctx.send(f"✅ Temperature for `{voice}` set to `{value}`.")
@chatterbox_temperature.error
async def chatterbox_temperature_error(self, ctx, error):
"""Handle 'reset' being passed as a non-float."""
if isinstance(error, commands.BadArgument):
msg_content = ctx.message.content.lower()
if "reset" in msg_content:
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_temperature_offsets()
offsets.pop(voice, None)
await self.config.user(ctx.author).chatterbox_temperature_offsets.set(offsets)
await ctx.send(f"✅ Temperature for `{voice}` reset to server default.")
else:
await ctx.send("❌ Value must be a number (0.0–1.5) or `reset`.")
@chatterbox_group.command(name="exag")
async def chatterbox_exaggeration(self, ctx: commands.Context, value: Optional[float] = None):
"""
Show or set Chatterbox exaggeration for your current voice (0.25–2.0).
Saved per voice per user — each voice remembers its own setting.
Higher = more expressive/dramatic delivery.
- `[p]chatterbox exag` → show current
- `[p]chatterbox exag 1.5` → set value
- `[p]chatterbox exag reset` → use server default
"""
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_exaggeration_offsets()
# Migrate old global setting on first use
if not offsets:
old_val = await self.config.user(ctx.author).chatterbox_exaggeration()
if old_val is not None:
offsets[voice] = old_val
await self.config.user(ctx.author).chatterbox_exaggeration_offsets.set(offsets)
await self.config.user(ctx.author).chatterbox_exaggeration.set(None)
if value is None:
current = offsets.get(voice)
if current is None:
return await ctx.send(
f"Exaggeration for `{voice}`: `server default`\n"
f"Set with: `{ctx.clean_prefix}chatterbox exag 1.5`"
)
return await ctx.send(
f"Exaggeration for `{voice}`: `{current}`\n"
f"Reset with: `{ctx.clean_prefix}chatterbox exag reset`"
)
offsets[voice] = value
await self.config.user(ctx.author).chatterbox_exaggeration_offsets.set(offsets)
await ctx.send(f"✅ Exaggeration for `{voice}` set to `{value}`.")
@chatterbox_exaggeration.error
async def chatterbox_exaggeration_error(self, ctx, error):
"""Handle 'reset' being passed as a non-float."""
if isinstance(error, commands.BadArgument):
msg_content = ctx.message.content.lower()
if "reset" in msg_content:
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_exaggeration_offsets()
offsets.pop(voice, None)
await self.config.user(ctx.author).chatterbox_exaggeration_offsets.set(offsets)
await ctx.send(f"✅ Exaggeration for `{voice}` reset to server default.")
else:
await ctx.send("❌ Value must be a number (0.25–2.0) or `reset`.")
@chatterbox_group.command(name="volume")
async def chatterbox_volume(self, ctx: commands.Context, value: Optional[float] = None):
"""
Show or set a volume offset (in dB) for your current Chatterbox voice.
Each voice can have its own offset so they all sound equally loud.
- `[p]chatterbox volume` → show offset for your active voice
- `[p]chatterbox volume -3` → make current voice 3 dB quieter
- `[p]chatterbox volume 4.5` → make current voice 4.5 dB louder
- `[p]chatterbox volume reset` → remove offset for current voice
Range: -20.0 to +20.0 dB.
"""
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_volume_offsets()
if value is None:
current = offsets.get(voice, 0.0)
if current == 0.0:
return await ctx.send(
f"🔊 Voice `{voice}` has no volume offset (0 dB).\n"
f"Set one with: `{ctx.clean_prefix}chatterbox volume <dB>`"
)
sign = "+" if current > 0 else ""
return await ctx.send(f"🔊 Voice `{voice}` volume offset: `{sign}{current} dB`")
if value < -20.0 or value > 20.0:
return await ctx.send("❌ Volume offset must be between -20.0 and +20.0 dB.")
offsets[voice] = value
await self.config.user(ctx.author).chatterbox_volume_offsets.set(offsets)
sign = "+" if value > 0 else ""
await ctx.send(f"✅ Volume offset for `{voice}` set to `{sign}{value} dB`.")
@chatterbox_volume.error
async def chatterbox_volume_error(self, ctx, error):
"""Handle 'reset' being passed as a non-float."""
if isinstance(error, commands.BadArgument):
msg_content = ctx.message.content.lower()
if "reset" in msg_content:
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_volume_offsets()
offsets.pop(voice, None)
await self.config.user(ctx.author).chatterbox_volume_offsets.set(offsets)
await ctx.send(f"✅ Volume offset for `{voice}` reset to 0 dB.")
else:
await ctx.send("❌ Value must be a number (-20.0 to +20.0) or `reset`.")
@chatterbox_group.command(name="speed")
async def chatterbox_speed(self, ctx: commands.Context, value: Optional[float] = None):
"""
Show or set speed for your current voice (0.25–4.0).
Saved per-voice — each voice remembers its own speed.
- `[p]chatterbox speed` → show current voice's speed
- `[p]chatterbox speed 1.2` → set speed (1.0 = normal)
- `[p]chatterbox speed reset` → remove speed override
"""
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_speed_offsets()
parts = voice.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 and parts[0].isdigit() else voice.rsplit(".", 1)[0]
if value is None:
current = offsets.get(voice)
if current is None:
return await ctx.send(
f"Speed for `{display}`: `default (1.0)`\n"
f"Set with: `{ctx.clean_prefix}chatterbox speed 1.2`"
)
return await ctx.send(
f"Speed for `{display}`: `{current}`\n"
f"Reset with: `{ctx.clean_prefix}chatterbox speed reset`"
)
if not 0.25 <= value <= 4.0:
return await ctx.send("❌ Speed must be between 0.25 and 4.0.")
offsets[voice] = value
await self.config.user(ctx.author).chatterbox_speed_offsets.set(offsets)
await ctx.send(f"✅ Speed for `{display}` set to `{value}`.")
@chatterbox_speed.error
async def chatterbox_speed_error(self, ctx, error):
"""Handle 'reset' being passed as a non-float."""
if isinstance(error, commands.BadArgument):
msg_content = ctx.message.content.lower()
if "reset" in msg_content:
voice = await self._get_effective_minimax_voice(ctx.author)
offsets = await self.config.user(ctx.author).chatterbox_speed_offsets()
offsets.pop(voice, None)
await self.config.user(ctx.author).chatterbox_speed_offsets.set(offsets)
parts = voice.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 and parts[0].isdigit() else voice.rsplit(".", 1)[0]
await ctx.send(f"✅ Speed for `{display}` reset to default.")
else:
await ctx.send("❌ Value must be a number (0.25–4.0) or `reset`.")
@chatterbox_group.command(name="sharevoice")
async def share_voice(self, ctx: commands.Context, name: str, target: discord.Member):
"""
Share one of your Chatterbox voices with another user.
Adds the voice to their library so they can use it too.
- `[p]chatterbox sharevoice CoolVoice @Friend`
"""
if target.bot:
return await ctx.send("❌ You can't share voices with bots.")
if target.id == ctx.author.id:
return await ctx.send("❌ You already have that voice.")
user_voices = await self.config.user(ctx.author).chatterbox_voices()
if not user_voices:
return await ctx.send("You don't have any uploaded voices to share.")
# Find matching voice
match = None
for v in user_voices:
parts = v.split("_", 1)
voice_name = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
if v == name or voice_name.lower() == name.lower() or v.lower() == name.lower():
match = v
break
if not match:
voice_names = ", ".join(
v.split("_", 1)[1].rsplit(".", 1)[0] if "_" in v else v.rsplit(".", 1)[0]
for v in user_voices
)
return await ctx.send(f"❌ Voice `{name}` not found in your list.\nYour voices: {voice_names}")
# Check if target already has this voice
target_voices = await self.config.user(target).chatterbox_voices()
if match in target_voices:
parts = match.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else match.rsplit(".", 1)[0]
return await ctx.send(f"❌ {target.display_name} already has `{display}`.")
# Add the same voice file to the target user's library
target_voices.append(match)
await self.config.user(target).chatterbox_voices.set(target_voices)
parts = match.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else match.rsplit(".", 1)[0]
await ctx.send(
f"✅ Voice `{display}` shared with {target.display_name}!\n"
f"They can switch to it with: `{ctx.clean_prefix}ttstoy myvoice {display}`"
)
@chatterbox_group.command(name="reset")
async def chatterbox_reset(self, ctx: commands.Context):
"""
Reset all per-voice parameters (temp, exag, volume, speed) for your current voice to server defaults.
- `[p]chatterbox reset`
"""
voice = await self._get_effective_minimax_voice(ctx.author)
parts = voice.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 and parts[0].isdigit() else voice.rsplit(".", 1)[0]
cleared = []
for key in ("chatterbox_temperature_offsets", "chatterbox_exaggeration_offsets", "chatterbox_volume_offsets", "chatterbox_speed_offsets"):
offsets = await getattr(self.config.user(ctx.author), key)()
if voice in offsets:
offsets.pop(voice)
await getattr(self.config.user(ctx.author), key).set(offsets)
cleared.append(key.replace("chatterbox_", "").replace("_offsets", ""))
if cleared:
await ctx.send(f"✅ Reset {', '.join(cleared)} for `{display}` to server defaults.")
else:
await ctx.send(f"ℹ️ `{display}` is already using all server defaults.")
@ttstoy_group.command(name="dectalkurl")
@commands.is_owner()
async def dectalk_url(self, ctx: commands.Context, url: Optional[str] = None):
"""
Show or set DECTalk API URL.
- `[p]ttstoy dectalkurl` → show current URL
- `[p]ttstoy dectalkurl http://127.0.0.1:33001` → set URL
"""
if url is None:
current = await self._get_dectalk_api_url()
mode = await self._get_tts_mode()
lines = [
f"**DECTalk API URL:** `{current}`",
"",
]
if mode == "dectalk":
lines.append("✅ Currently using DECTalk mode")
# Test connection
try:
test_url = f"{current.rstrip('/')}/health"
r = requests.get(test_url, timeout=2)
if r.status_code == 200:
lines.append("✅ DECTalk API is reachable")
else:
lines.append(f"⚠️ DECTalk API returned status {r.status_code}")
except Exception as e:
lines.append(f"❌ Cannot reach DECTalk API: {e}")
else:
lines.append("ℹ️ Not currently using DECTalk mode")
lines.append(f"Switch with: `{ctx.clean_prefix}ttstoy mode dectalk`")
return await ctx.send("\n".join(lines))
url = url.strip().rstrip('/')
if not url.startswith(('http://', 'https://')):
return await ctx.send("❌ URL must start with http:// or https://")
# Test the URL
try:
test_url = f"{url}/health"
r = requests.get(test_url, timeout=5)
if r.status_code != 200:
return await ctx.send(
f"⚠️ Warning: URL returned status {r.status_code}\n"
f"Saving anyway. Make sure your DECTalk server is running."
)
except Exception as e:
return await ctx.send(
f"⚠️ Warning: Cannot reach URL: {e}\n"
f"Saving anyway. Make sure your DECTalk server is running."
)
await self.config.dectalk_api_url.set(url)
await ctx.send(f"✅ DECTalk API URL set to: `{url}`")
@ttstoy_group.command(name="dectalkinstall")
@commands.is_owner()
async def dectalk_install(self, ctx: commands.Context):
"""Install the built-in DECTalk server."""
await ctx.send("📦 Installing DECTalk server (this may take a minute)...")
success = await self._install_dectalk_server()
if success:
await ctx.send(
"✅ DECTalk server installed successfully!\n"
f"Start it with: `{ctx.clean_prefix}ttstoy dectalkstart`\n"
f"Or enable auto-start: `{ctx.clean_prefix}ttstoy dectalkautotoggle`"
)
else:
await ctx.send(
"❌ Failed to install DECTalk server.\n"
"Make sure Node.js and npm are installed.\n"
"Check the bot logs for details."
)
@ttstoy_group.command(name="dectalkstart")
@commands.is_owner()
async def dectalk_start(self, ctx: commands.Context):
"""Start the built-in DECTalk server."""
await ctx.send("🚀 Starting DECTalk server...")
success = await self._start_dectalk_server()
if success:
await ctx.send(
f"✅ DECTalk server started on port {self.dectalk_port}!\n"
f"Switch to DECTalk mode: `{ctx.clean_prefix}ttstoy mode dectalk`"
)
else:
await ctx.send(
"❌ Failed to start DECTalk server.\n"
f"Try installing first: `{ctx.clean_prefix}ttstoy dectalkinstall`\n"
"Check the bot logs for details."
)
@ttstoy_group.command(name="dectalkstop")
@commands.is_owner()
async def dectalk_stop(self, ctx: commands.Context):
"""The DECTalk server is managed by systemd, not the bot."""
await ctx.send(
"The DECTalk server now runs as an independent systemd service "
"(dectalk-tts-server). Manage it with: "
"`sudo systemctl stop dectalk-tts-server`"
)
@ttstoy_group.command(name="dectalkstatus")
@commands.is_owner()
async def dectalk_status(self, ctx: commands.Context):
"""Check DECTalk server status."""
auto_start = await self.config.dectalk_auto_start()
dectalk_url = await self._get_dectalk_api_url()
# Check if installed
server_js = self.dectalk_dir / "server.js"
node_modules = self.dectalk_dir / "node_modules"
installed = server_js.exists() and node_modules.exists()
# The service is managed by systemd; report API reachability.
# Check if API is responding
api_responding = False
try:
test_url = f"{dectalk_url.rstrip('/')}/health"
r = requests.get(test_url, timeout=2)
api_responding = r.status_code == 200
except:
pass
lines = [
"**DECTalk Server Status**",
f"Installed: {'✅ Yes' if installed else '❌ No'}",
f"API Responding: {'✅ Yes' if api_responding else '❌ No'}",
f"Auto-start: {'✅ Enabled' if auto_start else '❌ Disabled'}",
f"URL: `{dectalk_url}`",
"",
]
if not api_responding:
lines.append(
"Service managed by systemd. Start with: "
"`sudo systemctl start dectalk-tts-server`"
)
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="dectalkautotoggle")
@commands.is_owner()
async def dectalk_auto_toggle(self, ctx: commands.Context):
"""Toggle auto-start for built-in DECTalk server."""
current = await self.config.dectalk_auto_start()
new_state = not current
await self.config.dectalk_auto_start.set(new_state)
status = "enabled" if new_state else "disabled"
await ctx.send(f"✅ DECTalk auto-start {status}")
if new_state:
await ctx.send("DECTalk server will start automatically when switching to DECTalk mode.")
else:
await ctx.send(f"You'll need to manually start DECTalk with `{ctx.clean_prefix}ttstoy dectalkstart`")
@ttstoy_group.command(name="morshustart")
@commands.is_owner()
async def morshu_start(self, ctx: commands.Context):
"""Start the built-in Morshu TTS server."""
await ctx.send("🚀 Starting Morshu server...")
success, error = await self._start_morshu_server()
if success:
await ctx.send(
f"✅ Morshu server started on port {self.morshu_port}!\n"
f"Switch to Morshu mode: `{ctx.clean_prefix}ttstoy mode morshu`"
)
else:
err_msg = (error or "Unknown error")[:1500]
await ctx.send(f"❌ Failed to start Morshu server:\n```\n{err_msg}\n```")
@ttstoy_group.command(name="morshustop")
@commands.is_owner()
async def morshu_stop(self, ctx: commands.Context):
"""The Morshu server is managed by systemd, not the bot."""
await ctx.send(
"The Morshu server now runs as an independent systemd service "
"(morshu-tts-server). Manage it with: "
"`sudo systemctl stop morshu-tts-server`"
)
@ttstoy_group.command(name="morshustatus")
@commands.is_owner()
async def morshu_status(self, ctx: commands.Context):
"""Check Morshu server status."""
api_responding = False
try:
r = requests.get(f"http://127.0.0.1:{self.morshu_port}/health", timeout=2)
api_responding = r.status_code == 200
except:
pass
lines = [
"**Morshu Server Status**",
f"API Responding: {'✅ Yes' if api_responding else '❌ No'}",
f"URL: `http://127.0.0.1:{self.morshu_port}`",
]
if not api_responding:
lines.append(
"\nService managed by systemd. Start with: "
"`sudo systemctl start morshu-tts-server`"
)
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="voxstart")
@commands.is_owner()
async def vox_start(self, ctx: commands.Context):
"""Start the built-in VOX TTS server."""
await ctx.send("🚀 Starting VOX server...")
success, error = await self._start_vox_server()
if success:
await ctx.send(
f"✅ VOX server started on port {self.vox_port}!\n"
f"Switch to VOX mode: `{ctx.clean_prefix}ttstoy mode vox`"
)
else:
err_msg = (error or "Unknown error")[:1500]
await ctx.send(f"❌ Failed to start VOX server:\n```\n{err_msg}\n```")
@ttstoy_group.command(name="voxstop")
@commands.is_owner()
async def vox_stop(self, ctx: commands.Context):
"""The VOX server is managed by systemd, not the bot."""
await ctx.send(
"The VOX server now runs as an independent systemd service "
"(vox-tts-server). Manage it with: "
"`sudo systemctl stop vox-tts-server`"
)
@ttstoy_group.command(name="voxstatus")
@commands.is_owner()
async def vox_status(self, ctx: commands.Context):
"""Check VOX server status."""
api_responding = False
try:
r = requests.get(f"http://127.0.0.1:{self.vox_port}/health", timeout=2)
api_responding = r.status_code == 200
except:
pass
lines = [
"**VOX Server Status**",
f"API Responding: {'✅ Yes' if api_responding else '❌ No'}",
f"URL: `http://127.0.0.1:{self.vox_port}`",
]
if not api_responding:
lines.append(
"\nService managed by systemd. Start with: "
"`sudo systemctl start vox-tts-server`"
)
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="accessibility")
@commands.is_owner()
async def accessibility_toggle(self, ctx: commands.Context):
"""Toggle accessibility mode (forces DECTalk, disables SFX emojis)."""
current = await self.config.accessibility_mode()
new_state = not current
await self.config.accessibility_mode.set(new_state)
if new_state:
await ctx.send(
"♿ **Accessibility mode enabled.**\n"
"• TTS forced to DECTalk\n"
"• Emoji SFX disabled (emojis read as text instead)"
)
else:
await ctx.send("♿ **Accessibility mode disabled.** Normal TTS settings restored.")
@ttstoy_group.command(name="voxpack")
@commands.is_owner()
async def vox_pack_cmd(self, ctx: commands.Context, pack: Optional[str] = None):
"""Show or set the VOX voice pack (vox or vox2)."""
if pack is None:
current = await self.config.vox_pack()
packs = get_available_packs() if VOX_AVAILABLE else []
return await ctx.send(
f"Current VOX pack: `{current}`\n"
f"Available packs: {', '.join(f'`{p}`' for p in packs) or 'none'}"
)
pack = pack.lower().strip()
if not VOX_AVAILABLE:
return await ctx.send("❌ VOX engine is not available.")
available = get_available_packs()
if pack not in available:
return await ctx.send(f"❌ Pack `{pack}` not found. Available: {', '.join(f'`{p}`' for p in available)}")
await self.config.vox_pack.set(pack)
await ctx.send(f"✅ VOX pack set to `{pack}`.")
@ttstoy_group.command(name="voxwords")
async def vox_words_cmd(self, ctx: commands.Context):
"""List available words in the current VOX pack."""
if not VOX_AVAILABLE:
return await ctx.send("❌ VOX engine is not available.")
pack = await self.config.vox_pack()
words = get_available_words(pack)
if not words:
return await ctx.send(f"No words found in VOX pack `{pack}`.")
# Split into chunks to avoid message length limits
header = f"**VOX Words ({pack}) — {len(words)} words:**\n"
chunk = header
for w in words:
entry = f"`{w}` "
if len(chunk) + len(entry) > 1900:
await ctx.send(chunk)
chunk = ""
chunk += entry
if chunk:
await ctx.send(chunk)
@ttstoy_group.command(name="info")
async def ttstoy_info(self, ctx: commands.Context):
"""DM usage instructions for TTS Toy."""
prefix = ctx.clean_prefix
is_owner = await self.bot.is_owner(ctx.author)
mode = await self._get_tts_mode()
lines = []
if is_owner:
lines.extend(
[
"## Setup Instructions (Owner)",
f"1. Choose TTS mode with `{prefix}ttstoy mode [minimax|chatterbox|dectalk|morshu|vox]`.",
"",
"**For MiniMax mode:**",
f"- Set your MiniMax API key with `{prefix}ttstoy key`.",
f"- Pick a model with `{prefix}ttstoy model <model_name>`.",
"",
"**For Chatterbox mode:**",
f"- Set Chatterbox URL with `{prefix}ttstoy chatterboxurl <url>` (default: http://127.0.0.1:8004).",
f"- Make sure your Chatterbox TTS server is running.",
"",
"**For DECTalk mode:**",
f"- Set DECTalk API URL with `{prefix}ttstoy dectalkurl <url>`.",
f"- Make sure your DECTalk server is running.",
f"- Use voice commands in your text: `[:np]` Paul, `[:nb]` Betty, `[:nh]` Harry, etc.",
f"- Run `{prefix}ttstoy myvoice` to see all voice commands.",
"",
"**For Morshu mode:**",
f"- Just switch to morshu mode — no setup needed.",
f"- All text is spoken in Morshu's voice from the CD-i Zelda games.",
"",
"**For VOX mode:**",
f"- Black Mesa VOX announcer from Half-Life.",
f"- Only words in the VOX dictionary are spoken; unknown words are skipped.",
f"- Choose pack with `{prefix}ttstoy voxpack [vox|vox2]`.",
f"- List words with `{prefix}ttstoy voxwords`.",
"",
f"2. Set a global default voice with `{prefix}ttstoy voice <name or voice_id>`.",
f"3. (Optional) Tune emoji SFX volume with `{prefix}ttstoy sfxvolume <0-100>`.",
"",
]
)
mode_emoji = {"minimax": "🌐", "chatterbox": "🏠", "dectalk": "🤖", "morshu": "🛒", "vox": "📢"}.get(mode, "❓")
lines.extend(
[
"## End User Instructions",
f"Current mode: **{mode.upper()}** {mode_emoji}",
"",
f"1. (Optional) Set your own voice with `{prefix}ttstoy myvoice <name or voice_id>`.",
f"2. Check your current/personal voice with `{prefix}ttstoy myvoice`.",
f"3. Speak in voice chat with `{prefix}tts <text>`.",
"4. Include supported emoji in your text to trigger SFX clips.",
]
)
try:
await ctx.author.send("\n".join(lines))
except discord.Forbidden:
await ctx.send("❌ I couldn't DM you. Please enable DMs and try again.")
return
await ctx.send("✅ I sent you a DM with ttstoy instructions.")
# --------- API KEY VIA BUTTON + MODAL ---------
@ttstoy_group.command(name="key")
@commands.is_owner()
async def minimax_key(self, ctx: commands.Context):
"""
Opens a button that launches a dialog to enter the MiniMax API key.
"""
view = MinimaxKeyButton(self.config)
await ctx.send("Click the button to enter your MiniMax API key:", view=view)
# --------- MODEL SET / LIST ---------
@ttstoy_group.command(name="model")
@commands.is_owner()
async def minimax_model(self, ctx: commands.Context, model: Optional[str] = None):
"""
Show or set the MiniMax TTS model.
- `[p]ttstoy model` → show known models & current setting
- `[p]ttstoy model speech-01-turbo` → set model
"""
# If no model given, show list + current
if model is None:
current = await self.config.minimax_model()
lines = [
"**MiniMax TTS Models**",
"",
"These are example model names you can use (subject to your MiniMax account):",
"- `speech-01-turbo` – fast & cheap",
"- `speech-01-hd` – higher quality, slower",
"- `speech-02-turbo` – newer fast model",
"- `speech-02-hd` – newer high quality model",
"",
f"Current configured model: `{current}`",
"",
f"To change it, run:\n`{ctx.clean_prefix}ttstoy model <model_name>`",
]
return await ctx.send("\n".join(lines))
# Model given: set it
model = model.strip()
await self.config.minimax_model.set(model)
await ctx.send(f"✅ MiniMax model set to `{model}`.")
# --------- GLOBAL VOICE ---------
@ttstoy_group.command(name="voice")
@commands.is_owner()
async def minimax_voice(self, ctx: commands.Context, *, voice_id: Optional[str] = None):
"""
Set or show the global default voice.
Usage: [p]ttstoy voice [voice_name]
"""
mode = await self._get_tts_mode()
# Show voice list if no voice provided
if not voice_id:
current = await self.config.minimax_voice()
if mode == "dectalk":
lines = [
"**DECTalk Voice Commands** (use in text):",
"`[:np]` Paul (default) • `[:nb]` Betty • `[:nh]` Harry • `[:nf]` Frank",
"`[:nd]` Dennis • `[:nk]` Kit • `[:nu]` Ursula • `[:nr]` Rita • `[:nw]` Wendy",
"",
"**Example:**",
f"`{ctx.clean_prefix}tts [:nh]Deep voice [:nb]now female`",
"",
"ℹ️ DECTalk doesn't use voice selection - control voices in your text."
]
return await ctx.send("\n".join(lines))
elif mode == "chatterbox":
return await ctx.send(
f"Current global voice: `{current or 'Emily.wav'}`\n"
f"List voices: `{ctx.clean_prefix}ttstoy voices`\n"
f"Set with: `{ctx.clean_prefix}ttstoy voice <filename>`"
)
elif mode == "morshu":
return await ctx.send(
"**MorshuTalk Mode**\n"
"No voice selection - it's always Morshu.\n"
"Lamp, oil, rope, bombs? You want it? It's yours, my friend."
)
elif mode == "vox":
vox_pack = await self.config.vox_pack()
words = get_available_words(vox_pack) if VOX_AVAILABLE else []
return await ctx.send(
f"**📢 VOX Mode** (pack: `{vox_pack}`)\n"
f"No voice selection. {len(words)} words available.\n"
f"Switch pack: `{ctx.clean_prefix}ttstoy voxpack [vox|vox2]`\n"
f"List words: `{ctx.clean_prefix}ttstoy voxwords`"
)
else: # minimax mode
voice_list = "\n".join([f"• `{name}` → {vid}" for name, vid in sorted(MINIMAX_VOICES.items())])
lines = [
"**Available MiniMax Voices:**",
voice_list,
"",
f"Current global voice: `{current or DEFAULT_MINIMAX_VOICE}`",
f"Set with: `{ctx.clean_prefix}ttstoy voice <voice_name>`"
]
return await ctx.send("\n".join(lines))
# DECTalk doesn't support voice selection - users use commands in text
if mode == "dectalk":
return await ctx.send(
"DECTalk mode doesn't use voice selection.\n"
"Users control voices with commands in their text:\n"
"`[:np]` Paul, `[:nb]` Betty, `[:nh]` Harry, `[:nf]` Frank, `[:nd]` Dennis, "
"`[:nk]` Kit, `[:nu]` Ursula, `[:nr]` Rita, `[:nw]` Wendy\n\n"
"Example: `[p]tts [:nh]Deep voice [:nb]now female`"
)
# Morshu doesn't support voice selection
if mode == "morshu":
return await ctx.send("MorshuTalk doesn't use voice selection - it's always Morshu.")
# VOX doesn't support voice selection - use voxpack instead
if mode == "vox":
return await ctx.send("📢 VOX mode doesn't use voice selection. Use `[p]ttstoy voxpack` to switch packs.")
voice_id = voice_id.strip()
await self.config.minimax_voice.set(voice_id)
if mode == "chatterbox":
resolved = voice_id
else:
resolved = resolve_minimax_voice_id(voice_id)
await ctx.send(f"✅ Global voice set → `{resolved}`.")
# --------- USER VOICE ---------
@ttstoy_group.command(name="myvoice")
@commands.guild_only()
async def my_minimax_voice(self, ctx: commands.Context, *, voice_id: Optional[str] = None):
"""
Set or show your personal voice.
Usage: [p]ttstoy myvoice [voice_name]
"""
mode = await self._get_tts_mode()
# Show voice list if no voice provided
if not voice_id:
cur = await self.config.user(ctx.author).minimax_voice()
glob = await self.config.minimax_voice()
if mode == "dectalk":
lines = [
"**DECTalk Voice Commands** (use in your text):",
"`[:np]` Paul (default) • `[:nb]` Betty • `[:nh]` Harry • `[:nf]` Frank",
"`[:nd]` Dennis • `[:nk]` Kit • `[:nu]` Ursula • `[:nr]` Rita • `[:nw]` Wendy",
"",
"**Example:**",
f"`{ctx.clean_prefix}tts [:nh]I'm gonna eat a pizza. [:dial67589340] Hi, can i order a pizza?`",
"",
"ℹ️ DECTalk doesn't use voice selection - control voices in your text."
]
return await ctx.send("\n".join(lines))
elif mode == "chatterbox":
user_voices = await self.config.user(ctx.author).chatterbox_voices()
cur_display = "None (using default)"
if cur:
parts = cur.split("_", 1)
cur_display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else cur.rsplit(".", 1)[0]
lines = [f"**🏠 Chatterbox Mode**"]
lines.append(f"Your active voice: `{cur_display}`")
if user_voices:
lines.append("\n**Your voices:**")
for v in user_voices:
parts = v.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
marker = " ← active" if v == cur else ""
lines.append(f" `{display}`{marker}")
else:
lines.append("\nNo uploaded voices yet.")
lines.append(f"\nSwitch: `{ctx.clean_prefix}ttstoy myvoice <name>`")
lines.append(f"Upload: `{ctx.clean_prefix}chatterbox addvoice <name>` (attach audio)")
return await ctx.send("\n".join(lines))
elif mode == "morshu":
return await ctx.send(
"**🛒 MorshuTalk Mode**\n"
"No voice selection - it's always Morshu.\n"
"Lamp, oil, rope, bombs? You want it? It's yours, my friend."
)
elif mode == "vox":
vox_pack = await self.config.vox_pack()
return await ctx.send(
f"**📢 VOX Mode** (pack: `{vox_pack}`)\n"
f"No voice selection. Use `{ctx.clean_prefix}ttstoy voxpack` to switch packs."
)
else: # minimax mode
lines = [f"**🌐 MiniMax Mode**"]
lines.append(f"Your voice: `{cur or 'None (using global)'}`")
lines.append(f"Global default: `{glob or DEFAULT_MINIMAX_VOICE}`")
lines.append("\n**Available voice labels:**")
for k, v in sorted(MINIMAX_VOICES.items()):
lines.append(f"• `{k}` → {v}")
lines.append(f"\nSet with: `{ctx.clean_prefix}ttstoy myvoice <voice_name>`")
return await ctx.send("\n".join(lines))
# Setting a voice - DECTalk doesn't support voice selection
if mode == "dectalk":
return await ctx.send(
"ℹ️ DECTalk mode doesn't use voice selection.\n"
"Control voices with commands in your text instead."
)
# Morshu doesn't support voice selection
if mode == "morshu":
return await ctx.send("🛒 MorshuTalk doesn't use voice selection - it's always Morshu.")
# VOX doesn't support voice selection
if mode == "vox":
return await ctx.send("📢 VOX mode doesn't use voice selection. Use `[p]ttstoy voxpack` to switch packs.")
voice_id = voice_id.strip()
# Chatterbox: match against user's uploaded voices by display name
if mode == "chatterbox":
user_voices = await self.config.user(ctx.author).chatterbox_voices()
if not user_voices:
return await ctx.send(
f"You don't have any uploaded voices.\n"
f"Upload one: `{ctx.clean_prefix}chatterbox addvoice <name>` (attach audio)"
)
match = None
for v in user_voices:
parts = v.split("_", 1)
display = parts[1].rsplit(".", 1)[0] if len(parts) > 1 else v.rsplit(".", 1)[0]
if display.lower() == voice_id.lower() or v.lower() == voice_id.lower():
match = v
break
if not match:
voice_names = ", ".join(
v.split("_", 1)[1].rsplit(".", 1)[0] if "_" in v else v.rsplit(".", 1)[0]
for v in user_voices
)
return await ctx.send(f"❌ Voice `{voice_id}` not found.\nYour voices: {voice_names}")
await self.config.user(ctx.author).minimax_voice.set(match)
display = match.split("_", 1)[1].rsplit(".", 1)[0] if "_" in match else match.rsplit(".", 1)[0]
return await ctx.send(f"✅ Your voice → `{display}`.")
await self.config.user(ctx.author).minimax_voice.set(voice_id.strip())
if mode == "chatterbox":
resolved = voice_id.strip()
else:
resolved = resolve_minimax_voice_id(voice_id.strip())
return await ctx.send(f"✅ Your voice → `{resolved}`.")
async def _process_tts_queue(self, guild_id: int):
"""Process TTS queue for a guild - plays TTS items in order."""
queue = self.tts_queues[guild_id]
# Track music state across all TTS in a batch
music_paused = False
saved_state = None
while True:
try:
# Wait for next TTS item
tts_item = await queue.get()
if tts_item is None: # Shutdown signal
break
ctx, audio_path = tts_item
try:
player = lavalink.get_player(guild_id)
player.store("channel", ctx.channel.id)
loaded = await player.load_tracks(audio_path)
if loaded.has_error or loaded.load_type != lavalink.enums.LoadType.TRACK_LOADED:
log.error(f"Failed to load TTS audio: {audio_path}")
queue.task_done()
continue
# Only pause and save state on the FIRST TTS in a batch
if not music_paused:
was_playing = player.is_playing
current_position = player.position if player.current else 0
original_queue = list(player.queue)
original_current = player.current
saved_state = {
'was_playing': was_playing,
'position': current_position,
'queue': original_queue,
'current': original_current
}
if was_playing:
await player.pause()
# Store current track temporarily and clear queue
if original_current:
player.queue.clear()
music_paused = True
# Play TTS
player.add(requester=ctx.author, track=loaded.tracks[0])
await player.play()
# Wait for TTS to finish
while player.is_playing:
await asyncio.sleep(0.1)
except Exception as e:
log.exception(f"Error processing TTS for guild {guild_id}: {e}")
finally:
queue.task_done()
# Check if queue is empty - if so, restore music
if queue.empty() and music_paused and saved_state:
try:
# Restore the original queue and position
if saved_state['current']:
player.add(requester=saved_state['current'].requester, track=saved_state['current'])
for track in saved_state['queue']:
player.add(requester=track.requester, track=track)
# Resume from where we left off
if saved_state['was_playing']:
await player.play()
await player.seek(saved_state['position'])
except Exception as e:
log.exception(f"Error restoring music for guild {guild_id}: {e}")
finally:
music_paused = False
saved_state = None
except asyncio.CancelledError:
break
except Exception as e:
log.exception(f"Unexpected error in TTS queue processor for guild {guild_id}: {e}")
def _get_or_create_tts_queue(self, guild_id: int):
"""Get or create TTS queue and processor for a guild."""
if guild_id not in self.tts_queues:
self.tts_queues[guild_id] = asyncio.Queue()
self.tts_locks[guild_id] = asyncio.Lock()
# Start processor task for this guild
self.tts_processors[guild_id] = asyncio.create_task(
self._process_tts_queue(guild_id)
)
return self.tts_queues[guild_id]
# ---------------------------------------------------------------------
# SPEAK COMMAND (JUST [p]tts)
# ---------------------------------------------------------------------
@commands.hybrid_command(name="tts")
@commands.guild_only()
async def tts(self, ctx: commands.Context, *, text: str):
"""
Generate TTS audio and upload the MP3. Also plays in voice chat if you're in one.
"""
async with ctx.typing():
# Check if user is in a VC — we'll play there if so, but it's not required
user_in_vc = ctx.author.voice and ctx.author.voice.channel
audio: Optional[Audio] = self.bot.get_cog("Audio")
if user_in_vc and audio and not ctx.guild.me.voice:
await ctx.invoke(audio.command_summon)
mode = await self._get_tts_mode()
# Not in VC: only chatterbox is allowed
if not user_in_vc and mode != "chatterbox":
return await ctx.send("❌ You must be in a voice channel to use **{}** mode. Join a VC, or switch to **chatterbox** mode (`[p]ttstoy mode chatterbox`).".format(mode))
chatterbox_url = await self._get_chatterbox_api_url()
dectalk_url = await self._get_dectalk_api_url()
api_key, model, _ = await self._get_minimax_settings()
# Accessibility mode: force dectalk, strip emoji SFX
accessibility = await self.config.accessibility_mode()
if accessibility:
mode = "dectalk"
# Strip emoji triggers so they're read as text, not played as SFX
text = EMOJI_CANDIDATE_PATTERN.sub(
lambda m: _normalize_sfx_trigger(m.group(0)) if _normalize_sfx_trigger(m.group(0)) in EMOJI_SFX_FOLDERS else m.group(0),
text
)
text = re.sub(r'[\U0001F300-\U0001FAFF\U00002600-\U000026FF\U00002700-\U000027BF]', '', text).strip()
# Check requirements based on mode
if mode == "minimax" and not api_key:
return await ctx.send("❌ MiniMax API key not set. Use `[p]ttstoy key` or switch to chatterbox/dectalk mode.")
if mode == "morshu" and not MORSHU_AVAILABLE:
return await ctx.send("❌ MorshuTalk engine failed to load. Check bot logs for details (missing g2p_en, numpy, or pydub?).")
if mode == "vox" and not VOX_AVAILABLE:
return await ctx.send("❌ VOX engine is not available.")
if mode == "chatterbox":
try:
test_url = f"{chatterbox_url.rstrip('/')}/api/ui/initial-data"
r = requests.get(test_url, timeout=3)
if r.status_code != 200:
return await ctx.send(f"❌ Chatterbox TTS not responding (status {r.status_code}). Check if server is running.")
# Also verify /tts endpoint exists
openapi_url = f"{chatterbox_url.rstrip('/')}/openapi.json"
r2 = requests.get(openapi_url, timeout=3)
if r2.status_code == 200:
paths = list(r2.json().get("paths", {}).keys())
log.info(f"[Chatterbox] Server routes: {paths}")
if "/tts" not in paths:
return await ctx.send(f"❌ Chatterbox server is running but `/tts` endpoint not found.\nAvailable routes: {', '.join(paths[:10])}")
except Exception as e:
return await ctx.send(f"❌ Cannot reach Chatterbox TTS: {e}\nMake sure the server is running at `{chatterbox_url}`")
if mode == "dectalk":
# Test DECTalk API availability
try:
test_url = f"{dectalk_url.rstrip('/')}/health"
r = requests.get(test_url, timeout=2)
if r.status_code != 200:
return await ctx.send(f"❌ DECTalk API not responding (status {r.status_code}). Check if server is running.")
except Exception as e:
return await ctx.send(f"❌ Cannot reach DECTalk API: {e}\nMake sure the server is running at `{dectalk_url}`")
voice_id = await self._get_effective_minimax_voice(ctx.author)
sfx_volume = await self.config.sfx_volume()
vox_pack = await self.config.vox_pack()
# Fetch per-user chatterbox settings (per-voice)
cb_temp = None
cb_exag = None
cb_vol_db = 0.0
cb_speed = None
if mode == "chatterbox":
temp_offsets = await self.config.user(ctx.author).chatterbox_temperature_offsets()
exag_offsets = await self.config.user(ctx.author).chatterbox_exaggeration_offsets()
vol_offsets = await self.config.user(ctx.author).chatterbox_volume_offsets()
speed_offsets = await self.config.user(ctx.author).chatterbox_speed_offsets()
cb_temp = temp_offsets.get(voice_id)
cb_exag = exag_offsets.get(voice_id)
cb_vol_db = vol_offsets.get(voice_id, 0.0)
cb_speed = speed_offsets.get(voice_id)
# Fallback: check old global settings for migration
if cb_temp is None:
cb_temp = await self.config.user(ctx.author).chatterbox_temperature()
if cb_exag is None:
cb_exag = await self.config.user(ctx.author).chatterbox_exaggeration()
self.tts_storage.mkdir(parents=True, exist_ok=True)
audio_path = str(self.tts_storage.joinpath(f"{ctx.message.id}.mp3"))
# Collect all chatterbox voices for inline voice resolution
all_cb_voices = await self._get_all_chatterbox_voices()
try:
result = await asyncio.to_thread(
self._save_prompt_audio, api_key, model, voice_id, text, audio_path, sfx_volume, mode, dectalk_url, vox_pack, chatterbox_url, cb_temp, cb_exag, cb_vol_db, cb_speed, all_cb_voices
)
detected_language = result.get("language") if result else None
except Exception as e:
log.exception("TTS error: %s", e)
mode_name = {"minimax": "MiniMax", "chatterbox": "Chatterbox", "dectalk": "DECTalk", "morshu": "MorshuTalk", "vox": "VOX"}.get(mode, "TTS")
return await ctx.send(f"{mode_name} error generating audio: {e}")
view = discord.ui.View()
view.add_item(
discord.ui.Button(
label="PLS DONATE",
style=discord.ButtonStyle.link,
url="https://account.venmo.com/u/kingstonscyd",
)
)
filename = f"{mode}-tts-{ctx.message.id}.mp3"
# Build response message with language detection
LANG_NAMES = {
"EN": "English", "ZH": "Chinese", "JA": "Japanese",
"ES": "Spanish", "FR": "French", "DE": "German",
"KO": "Korean", "PT": "Portuguese", "RU": "Russian",
"IT": "Italian", "AR": "Arabic", "HI": "Hindi",
"TH": "Thai", "TR": "Turkish", "VI": "Vietnamese",
"NL": "Dutch", "PL": "Polish", "SV": "Swedish",
}
if detected_language:
code = detected_language.upper()
lang_display = LANG_NAMES.get(code, code)
lang_line = f"Language detected: {lang_display} ({code})"
else:
lang_line = "Language detected: Unknown"
await ctx.send(
content=lang_line,
file=discord.File(audio_path, filename=filename),
view=view,
)
# Play in VC only if user is in a voice channel
if user_in_vc and audio and ctx.guild.me.voice:
queue = self._get_or_create_tts_queue(ctx.guild.id)
queue_position = queue.qsize() + 1
if queue_position > 1:
await ctx.send(f"Added to TTS queue (position {queue_position})")
await queue.put((ctx, audio_path))
if ctx.interaction:
await ctx.reply("🗣")
else:
await ctx.react_quietly("🗣")
# =========================================================================
# WEB UI — login, webui commands, and shared-state processor
# =========================================================================
WEBUI_SHARED_STATE = Path(os.environ.get("TTSTOY_SHARED_STATE", "/tmp/ttstoy_webui_state.json"))
def _read_shared_state(self) -> dict:
try:
with open(self.WEBUI_SHARED_STATE) as f:
return json.load(f)
except Exception:
return {}
def _write_shared_state(self, state: dict):
try:
self.WEBUI_SHARED_STATE.parent.mkdir(parents=True, exist_ok=True)
with open(self.WEBUI_SHARED_STATE, "w") as f:
json.dump(state, f)
except Exception as e:
log.warning(f"[WebUI] Could not write shared state: {e}")
def _get_webui_url(self) -> str:
state = self._read_shared_state()
return state.get("webui_url", os.environ.get("WEBUI_URL", "https://ttstoy.kingstons-scrapyard.net"))
def _get_webui_internal_url(self) -> str:
port = os.environ.get("WEBUI_PORT", "8098")
return f"http://127.0.0.1:{port}"
def _get_internal_secret(self) -> str:
state = self._read_shared_state()
return state.get("internal_secret", "")
@ttstoy_group.command(name="login")
@commands.guild_only()
async def webui_login(self, ctx: commands.Context):
"""Get a one-time login key for Kingston's Scrapyard. Must be used in a server so the key knows which server you're in. The key is DM'd to you."""
webui_url = self._get_webui_internal_url()
secret = self._get_internal_secret()
if not secret:
return await ctx.send(
"❌ The web UI is not running. Ask the bot owner to start it.",
ephemeral=True,
)
# Register token with ttstoy webui — it writes to shared_auth.json.
# The user logs in once on the Homepage; the session cookie is shared
# across all *.kingstons-scrapyard.net subdomains.
avatar_url = str(ctx.author.display_avatar.url) if ctx.author.display_avatar else ""
try:
r = await asyncio.to_thread(
requests.post,
f"{webui_url.rstrip('/')}/api/register_token",
headers={"X-Internal-Secret": secret, "Content-Type": "application/json"},
json={
"user_id": str(ctx.author.id),
"discord_name": str(ctx.author),
"avatar_url": avatar_url,
"is_owner": await self.bot.is_owner(ctx.author),
"guild_id": str(ctx.guild.id),
"guild_name": ctx.guild.name,
},
timeout=5,
)
r.raise_for_status()
token = r.json()["token"]
except Exception as e:
log.exception("[WebUI] Failed to get login token: %s", e)
return await ctx.send("❌ Could not reach the web UI. Is it running?", ephemeral=True)
try:
await ctx.author.send(
f"**Kingston's Scrapyard — Login**\n\n"
f"Go to: https://homepage.kingstons-scrapyard.net/login\n\n"
f"Your key (valid for 5 minutes):\n```\n{token}\n```\n"
f"Log in once and you'll be signed in across all Scrapyard sites."
)
except discord.Forbidden:
return await ctx.send("❌ I couldn't DM you. Enable DMs from server members and try again.")
await ctx.send("✅ I sent you a DM with your login key.", ephemeral=True)
@ttstoy_group.command(name="webui")
async def webui_url_cmd(self, ctx: commands.Context):
"""Get the link to the TtsToy web UI."""
webui_url = self._get_webui_url()
await ctx.send(
f"🌐 **TtsToy Web UI:** {webui_url}\n"
f"Run `{ctx.clean_prefix}ttstoy login` to get your login key.",
ephemeral=True,
)
@ttstoy_group.command(name="webuistatus")
@commands.is_owner()
async def webui_status_cmd(self, ctx: commands.Context):
"""Check the status of the TtsToy web UI (now a scrapyard service)."""
webui_url = self._get_webui_url()
# Try to hit the login page
reachable = False
try:
r = await asyncio.to_thread(requests.get, f"{webui_url.rstrip('/')}/login", timeout=3)
reachable = r.status_code == 200
except Exception:
pass
lines = [
"**TtsToy Web UI Status**",
f"Reachable at URL: {'✅ Yes' if reachable else '❌ No'}",
f"URL: {webui_url}",
"The web UI now runs as a scrapyard service (not a bot subprocess).",
f"Restart with: `sudo systemctl restart scrapyard`",
]
await ctx.send("\n".join(lines))
@ttstoy_group.command(name="webuirestart")
@commands.is_owner()
async def webui_restart_cmd(self, ctx: commands.Context):
"""Restart the TtsToy web UI (now a scrapyard service)."""
await ctx.send("🔄 The web UI is now a scrapyard service.\nRestart it with: `sudo systemctl restart scrapyard`")
# -------------------------------------------------------------------------
# Background task — poll shared state for pending commands and posts
# -------------------------------------------------------------------------
@tasks.loop(seconds=3)
async def _webui_state_processor(self):
"""Poll the shared state file and execute pending commands/posts from the web UI."""
try:
pending_cmds, pending_posts = await asyncio.to_thread(self._take_pending_state_updates)
for cmd in pending_cmds:
await self._handle_webui_command(cmd)
for post in pending_posts:
await self._handle_webui_post(post)
# Reload emoji→SFX mapping in case it was changed via the web UI
_reload_emoji_sfx_map()
except Exception as e:
log.exception("[WebUI] State processor error: %s", e)
def _take_pending_state_updates(self):
"""Under a cross-process lock: pop pending commands/posts and refresh user info.
Returns (pending_cmds, pending_posts). Runs in a worker thread — the
lock matches the one the web UI uses when appending, so no updates are
lost between the read and the write.
"""
lock_file = Path(os.environ.get("TTSTOY_SHARED_STATE_LOCK", "/tmp/ttstoy_webui_state.lock"))
with open(lock_file, "w") as lf:
if _fcntl is not None:
_fcntl.flock(lf, _fcntl.LOCK_EX)
try:
state = self._read_shared_state()
pending_cmds = state.pop("pending_commands", [])
pending_posts = state.pop("pending_posts", [])
# Refresh user info so the profile page stays current
users_info = state.get("users", {})
for guild in self.bot.guilds:
for member in guild.members:
uid = str(member.id)
info = users_info.setdefault(uid, {})
info["display_name"] = member.display_name
info["username"] = str(member)
info["joined_at"] = str(member.joined_at)[:10] if member.joined_at else "unknown"
info["roles"] = ", ".join(r.name for r in member.roles if r.name != "@everyone")
state["users"] = users_info
self._write_shared_state(state)
finally:
if _fcntl is not None:
_fcntl.flock(lf, _fcntl.LOCK_UN)
return pending_cmds, pending_posts
async def _handle_webui_command(self, cmd: dict):
"""Execute a command queued by the web UI."""
user_id = int(cmd.get("user_id", 0))
action = cmd.get("cmd")
payload = cmd.get("payload", {})
user = self.bot.get_user(user_id)
if not user:
log.warning(f"[WebUI] Unknown user {user_id} for command {action}")
return
try:
if action == "set_voice":
voice = payload.get("voice", "")
await self.config.user(user).minimax_voice.set(voice)
log.info(f"[WebUI] Set voice for {user} → {voice}")
elif action == "set_mode":
if await self.bot.is_owner(user):
mode = payload.get("mode", "minimax")
await self.config.tts_mode.set(mode)
log.info(f"[WebUI] Set global TTS mode → {mode}")
elif action == "set_chatterbox_params":
voice = payload.get("voice", "")
params = payload.get("params", {})
if "temperature" in params:
offsets = await self.config.user(user).chatterbox_temperature_offsets()
offsets[voice] = params["temperature"]
await self.config.user(user).chatterbox_temperature_offsets.set(offsets)
if "exaggeration" in params:
offsets = await self.config.user(user).chatterbox_exaggeration_offsets()
offsets[voice] = params["exaggeration"]
await self.config.user(user).chatterbox_exaggeration_offsets.set(offsets)
if "volume_db" in params:
offsets = await self.config.user(user).chatterbox_volume_offsets()
offsets[voice] = params["volume_db"]
await self.config.user(user).chatterbox_volume_offsets.set(offsets)
if "speed" in params:
offsets = await self.config.user(user).chatterbox_speed_offsets()
offsets[voice] = params["speed"]
await self.config.user(user).chatterbox_speed_offsets.set(offsets)
log.info(f"[WebUI] Updated chatterbox params for {user} voice={voice}")
elif action == "add_voice":
filename = payload.get("filename", "")
voices = await self.config.user(user).chatterbox_voices()
if filename and filename not in voices:
voices.append(filename)
await self.config.user(user).chatterbox_voices.set(voices)
await self.config.user(user).minimax_voice.set(filename)
log.info(f"[WebUI] Added voice {filename} for {user}")
elif action == "remove_voice":
filename = payload.get("filename", "")
voices = await self.config.user(user).chatterbox_voices()
if filename in voices:
voices.remove(filename)
await self.config.user(user).chatterbox_voices.set(voices)
for key in ("chatterbox_temperature_offsets", "chatterbox_exaggeration_offsets",
"chatterbox_volume_offsets", "chatterbox_speed_offsets"):
offsets = await getattr(self.config.user(user), key)()
offsets.pop(filename, None)
await getattr(self.config.user(user), key).set(offsets)
current = await self.config.user(user).minimax_voice()
if current == filename:
await self.config.user(user).minimax_voice.set(voices[0] if voices else None)
log.info(f"[WebUI] Removed voice {filename} for {user}")
except Exception as e:
log.exception(f"[WebUI] Error handling command {action} for user {user_id}: {e}")
async def _handle_webui_post(self, post: dict):
"""Deliver a pre-generated TTS file queued by the web UI.
If `play_vc` is set and the user is in a voice channel, the audio is
queued for VC playback (via the standard ttstoy queue). Either way it is
also posted as a message to the configured TTS channel.
"""
try:
audio_path = post.get("path")
text = post.get("text", "")
user = post.get("user", "Unknown")
engine = post.get("engine", "tts")
guild_id_str = post.get("guild_id", "")
user_id_str = post.get("user_id", "")
play_vc = bool(post.get("play_vc", False))
if not audio_path or not os.path.exists(audio_path):
log.warning(f"[WebUI] Audio file not found: {audio_path}")
return
target_guild_id = int(guild_id_str) if guild_id_str else None
# --- VC playback (optional) ---
if play_vc and user_id_str:
await self._play_webui_audio_in_vc(user_id_str, audio_path, target_guild_id)
# --- Post transcript to the configured TTS channel ---
channel = await self._get_webui_tts_channel(target_guild_id)
if not channel:
log.warning("[WebUI] No TTS channel configured. Set TTSTOY_WEBUI_CHANNEL_ID env var.")
return
suffix = Path(audio_path).suffix or ".mp3"
filename = f"webui-{engine}-tts{suffix}"
content = f"🗣 **{discord.utils.escape_markdown(user)}** via web UI [{engine}]\n> {discord.utils.escape_markdown(text[:200])}"
await channel.send(content=content, file=discord.File(audio_path, filename=filename))
log.info(f"[WebUI] Posted TTS from {user} to channel {channel.id}")
except Exception as e:
log.exception(f"[WebUI] Error posting TTS: {e}")
async def _play_webui_audio_in_vc(self, user_id_str: str, audio_path: str, target_guild_id: int = None):
"""Queue a pre-generated audio file for VC playback in the user's current VC."""
try:
user_id = int(user_id_str)
member_vc = None
member_guild = None
if target_guild_id:
guild = self.bot.get_guild(target_guild_id)
if guild:
member_guild = guild
member = guild.get_member(user_id)
if member and member.voice and member.voice.channel:
member_vc = member.voice.channel
if not member_vc:
for guild in self.bot.guilds:
member = guild.get_member(user_id)
if member and member.voice and member.voice.channel:
member_vc = member.voice.channel
if not member_guild:
member_guild = guild
break
if not member_vc or not member_guild:
log.info(f"[WebUI] {user_id} is not in a VC — skipping playback")
return
member = member_guild.get_member(user_id)
audio_cog: Optional[Audio] = self.bot.get_cog("Audio")
if not audio_cog:
log.warning("[WebUI] Audio cog not loaded — skipping VC playback")
return
text_channel = member_guild.system_channel or next(
(c for c in member_guild.text_channels
if c.permissions_for(member_guild.me).send_messages), None
)
if not text_channel:
log.warning("[WebUI] No text channel available for VC playback")
return
if not member_guild.me.voice or member_guild.me.voice.channel != member_vc:
await member_vc.connect()
class _FakeCtx:
guild = member_guild
channel = text_channel
author = member or user_id
interaction = None
async def react_quietly(self, *a, **kw): pass
queue = self._get_or_create_tts_queue(member_guild.id)
await queue.put((_FakeCtx(), audio_path))
log.info(f"[WebUI] Queued audio for VC playback (user={user_id} guild={member_guild.id})")
except Exception as e:
log.exception(f"[WebUI] VC playback failed: {e}")
async def _get_webui_tts_channel(self, guild_id: int = None) -> Optional[discord.TextChannel]:
"""Get the channel to post web UI TTS messages to.
Checks per-guild configured channel, then global configured channel,
then looks for a channel named tts/ttstoy/tts-toy in the target guild.
"""
# Per-guild configured channel
if guild_id:
guild = self.bot.get_guild(guild_id)
if guild:
guild_channel_id = await self.config.guild(guild).webui_tts_channel_id()
if guild_channel_id:
ch = self.bot.get_channel(int(guild_channel_id))
if ch:
return ch
# Global configured channel (use if it's in the target guild, or no guild specified)
channel_id = await self.config.webui_tts_channel_id()
if not channel_id:
channel_id = int(os.environ.get("TTSTOY_WEBUI_CHANNEL_ID", 0))
if channel_id:
ch = self.bot.get_channel(int(channel_id))
if ch:
if not guild_id or (hasattr(ch, 'guild') and ch.guild.id == guild_id):
return ch
# Search the target guild for a tts-named channel
if guild_id:
guild = self.bot.get_guild(guild_id)
if guild:
for ch in guild.text_channels:
if ch.name.lower() in ("tts", "ttstoy", "tts-toy"):
return ch
# No guild specified — search all guilds
for guild in self.bot.guilds:
for ch in guild.text_channels:
if ch.name.lower() in ("tts", "ttstoy", "tts-toy"):
return ch
return None
@ttstoy_group.command(name="webuichannel")
@commands.is_owner()
@commands.guild_only()
async def webui_channel_cmd(self, ctx: commands.Context, channel: Optional[discord.TextChannel] = None):
"""Set or show the channel where web UI TTS audio is posted in this server.
Usage:
[p]ttstoy webuichannel — show current channel
[p]ttstoy webuichannel #channel — set channel for this server
"""
if channel is None:
guild_channel_id = await self.config.guild(ctx.guild).webui_tts_channel_id()
if guild_channel_id:
ch = self.bot.get_channel(int(guild_channel_id))
name = ch.mention if ch else f"ID {guild_channel_id} (not found)"
else:
name = "Not set (falls back to any channel named `tts`)"
return await ctx.send(f"Web UI TTS channel for this server: {name}")
await self.config.guild(ctx.guild).webui_tts_channel_id.set(channel.id)
await ctx.send(f"✅ Web UI TTS audio will be posted to {channel.mention} in this server")
# -------------------------------------------------------------------------
# Cog lifecycle
# -------------------------------------------------------------------------
async def cog_load(self):
"""Called when cog is loaded."""
# The dectalk/morshu/vox TTS services now run as independent systemd
# services (managed outside the bot), so the cog no longer starts them.
# It simply connects to them over HTTP on their fixed ports.
# Start health HTTP server
try:
_HealthHandler.cog_ref = self
self._health_server = HTTPServer(("0.0.0.0", TTSTOY_HEALTH_PORT), _HealthHandler)
self._health_thread = threading.Thread(
target=self._health_server.serve_forever, daemon=True
)
self._health_thread.start()
log.info(f"TtsToy health server started on port {TTSTOY_HEALTH_PORT}")
except Exception:
log.exception("Failed to start TtsToy health server")
# Start web UI state processor
self._webui_state_processor.start()
log.info("[WebUI] State processor started")