commit e61a320499fb7b620062ed119b3cb0921fd74e7f Author: owen Date: Mon Sep 14 14:31:15 2026 -0500 Initial standalone morshu TTS server extracted from the ttstoy bot cog Self-contained HTTP service with engine, data, start script, systemd unit, and documentation. Runs independently of the Discord bot on its fixed port. diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..58ae165 --- /dev/null +++ b/.env.example @@ -0,0 +1,2 @@ +# Port the Morshu TTS server listens on +PORT=33002 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..150526a --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +__pycache__/ +*.pyc +venv/ +.venv/ +*.log diff --git a/INSTALL.md b/INSTALL.md new file mode 100644 index 0000000..002b0f7 --- /dev/null +++ b/INSTALL.md @@ -0,0 +1,48 @@ +# Installing the Morshu TTS Server as a system service + +These steps install the server as a systemd service that starts on boot and +restarts automatically if it crashes. Commands that need root are shown with +`sudo`; run them in your own terminal. + +## 1. Install prerequisites + +```bash +sudo apt install python3 python3-venv +``` + +## 2. Create the virtual environment + +The first run of `start.sh` creates a local venv and installs the Python +dependencies from requirements.txt. You can prime it now: + +```bash +cd /home/owen/morshu-tts-server +./start.sh +``` + +Stop it with Ctrl+C once you see the ready message, then install the unit. + +## 3. Install the systemd unit + +```bash +sudo cp /home/owen/morshu-tts-server/morshu-tts-server.service /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now morshu-tts-server +``` + +## 4. Verify + +```bash +systemctl status morshu-tts-server +curl -s http://127.0.0.1:33002/health +``` + +The health endpoint should return `ok`. + +## Managing the service + +```bash +sudo systemctl restart morshu-tts-server +sudo systemctl stop morshu-tts-server +journalctl -u morshu-tts-server -f +``` diff --git a/README.md b/README.md new file mode 100644 index 0000000..aeb64ce --- /dev/null +++ b/README.md @@ -0,0 +1,67 @@ +# Morshu TTS Server + +A standalone HTTP server that generates speech in the voice of Morshu (from the +Zelda CD-i games) using phoneme matching against sampled voice lines. It is +based on MorshuTalk by jalenluorion (https://github.com/jalenluorion/MorshuTalk). + +This service was extracted from the ttstoy Discord bot cog so it can run on its +own as a system service, independent of the bot. + +## How it works + +Input text is converted to phonemes with a grapheme-to-phoneme model, then the +engine splices together matching phoneme segments taken from a sample recording +(`morshutalk_morshu.wav`) to build the output audio. + +## Requirements + +- Python 3.9 or newer +- Python packages: `numpy`, `pydub`, `nltk`, `g2p_en` (see requirements.txt) +- The bundled `morshutalk_morshu.wav` sample (included in this repo) + +On first import, NLTK downloads a few small data packages +(`averaged_perceptron_tagger_eng`, `cmudict`, `punkt`, `punkt_tab`). + +## Running + +```bash +./start.sh +``` + +`start.sh` creates a local virtual environment on first launch, installs the +dependencies from requirements.txt, and starts the server. The listening port +defaults to `33002` and can be overridden with the `PORT` environment variable. + +You can also run it directly against an existing interpreter that has the +dependencies installed: + +```bash +PORT=33002 python3 server.py +``` + +## API + +### GET /say + +Generate speech and return a WAV file. + +``` +GET /say?text=Hello%20it%20is%20me%20Morshu +``` + +Response: `audio/wav` (WAV bytes). + +Example: + +```bash +curl "http://127.0.0.1:33002/say?text=lamp+oil+rope+bombs" -o morshu.wav +``` + +### GET /health + +Returns `ok`. + +## Running as a system service + +A systemd unit file is provided (`morshu-tts-server.service`). See INSTALL.md +for setup steps. diff --git a/morshu-tts-server.service b/morshu-tts-server.service new file mode 100644 index 0000000..66a87b6 --- /dev/null +++ b/morshu-tts-server.service @@ -0,0 +1,16 @@ +[Unit] +Description=Morshu TTS Server +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=owen +WorkingDirectory=/home/owen/morshu-tts-server +Environment=PORT=33002 +ExecStart=/home/owen/morshu-tts-server/start.sh +Restart=always +RestartSec=3 + +[Install] +WantedBy=multi-user.target diff --git a/morshutalk_engine.py b/morshutalk_engine.py new file mode 100644 index 0000000..2f0c333 --- /dev/null +++ b/morshutalk_engine.py @@ -0,0 +1,306 @@ +""" +Bundled MorshuTalk engine for ttstoy. +Based on MorshuTalk by jalenluorion (https://github.com/jalenluorion/MorshuTalk) +Generates speech audio from Morshu's voice lines using phoneme matching. +""" +import re +import unicodedata +from os import path +from typing import List, Tuple, Callable, Literal + +import numpy as np +import random +import warnings +from pydub import AudioSegment + +# --------------------------------------------------------------------------- +# G2P (Grapheme-to-Phoneme) wrapper with progress + cancel support +# --------------------------------------------------------------------------- +import nltk +for _res in ('averaged_perceptron_tagger_eng', 'cmudict', 'punkt', 'punkt_tab'): + try: + nltk.data.find(f'taggers/{_res}' if 'tagger' in _res else _res) + except LookupError: + nltk.download(_res, quiet=True) + +from g2p_en.g2p import G2p, unicode, normalize_numbers, word_tokenize, pos_tag + + +class G2pProgress(G2p): + def __init__(self): + super().__init__() + self.cancelled = False + + def cancel(self): + self.cancelled = True + + def run_with_progress(self, text, callback: Callable[[int, int], None] = None): + self.cancelled = False + + text = unicode(text) + text = normalize_numbers(text) + text = ''.join(char for char in unicodedata.normalize('NFD', text) + if unicodedata.category(char) != 'Mn') + text = text.lower() + text = re.sub("[^ a-z'.,?!\\-]", "", text) + text = text.replace("i.e.", "that is") + text = text.replace("e.g.", "for example") + + words = word_tokenize(text) + tokens = pos_tag(words) + + step = 0 + total = len(tokens) + prons = [] + + for word, pos in tokens: + if self.cancelled: + return + + if callback: + callback(step, total) + step += 1 + + if re.search("[a-z]", word) is None: + pron = [word] + elif word in self.homograph2features: + pron1, pron2, pos1 = self.homograph2features[word] + pron = pron1 if pos.startswith(pos1) else pron2 + elif word in self.cmu: + pron = self.cmu[word][0] + else: + pron = self.predict(word) + + prons.extend(pron) + prons.extend([" "]) + + if callback: + callback(total, total) + + return prons[:-1] + + +# --------------------------------------------------------------------------- +# Singleton G2P instance and audio data +# --------------------------------------------------------------------------- +g2p = G2pProgress() + +_WAV_PATH = path.join(path.dirname(__file__), 'morshutalk_morshu.wav') +morshu_wav = AudioSegment.from_wav(_WAV_PATH) + +# Phoneme timing record from the morshu audio +morshu_rec = np.rec.array([ + ('', 160, 0), ('L', 250, 2), ('AE', 348, 2), ('M', 420, 2), ('P', 510, 1), + ('OY', 700, 2), ('L', 835, 1), ('', 1090, 0), + ('R', 1180, 2), ('OW', 1300, 2), ('', 1390, 0), ('P', 1490, 2), ('', 1850, 0), + ('B', 1895, 2), ('AA', 2090, 2), ('M', 2235, 2), ('Z', 2390, 2), + ('', 2780, 0), ('Y', 2840, 2), ('UW', 2960, 2), + ('W', 3030, 2), ('AA', 3110, 2), ('N', 3150, 1), ('IH', 3240, 2), ('T', 3370, 2), ('', 3810, 0), + ('IH', 3960, 2), ('T', 4070, 2), ('Y', 4260, 2), ('UH', 4400, 2), ('R', 4510, 2), ('Z', 4600, 2), + ('M', 4675, 2), ('AY', 4810, 2), ('', 4885, 0), + ('F', 4930, 2), ('R', 4980, 2), ('EH', 5100, 2), ('N', 5240, 2), ('D', 5300, 2), ('', 5520, 0), + ('AE', 5630, 2), ('Z', 5740, 2), ('L', 5870, 2), ('AO', 6000, 2), ('NG', 6140, 2), + ('AE', 6170, 1), ('Z', 6265, 2), ('Y', 6300, 2), ('UW', 6380, 2), + ('HH', 6450, 2), ('AE', 6510, 1), ('V', 6580, 2), + ('IH', 6640, 2), ('N', 6670, 2), ('AH', 6747, 2), ('F', 6855, 2), + ('R', 6960, 2), ('UW', 7060, 2), ('B', 7170, 1), ('IY', 7340, 2), ('Z', 7520, 2), ('', 8236, 0), + ('S', 8407, 2), ('AA', 8495, 2), ('R', 8570, 2), ('IY', 8630, 1), + ('L', 8740, 2), ('IH', 8811, 2), ('NG', 8942, 2), ('K', 9014, 2), ('', 9251, 0), + ('AY', 9384, 2), ('', 9467, 0), ('K', 9512, 2), ('AE', 9640, 2), ('N', 9716, 2), ('', 9844, 0), + ('G', 9894, 2), ('IH', 9985, 2), ('V', 10060, 2), ('', 10149, 0), + ('K', 10256, 2), ('R', 10297, 2), ('EH', 10383, 2), ('IH', 10482, 1), ('', 10564, 0), ('T', 10617, 2), + ('', 10962, 0), ('K', 11019, 2), ('AH', 11100, 2), ('M', 11229, 2), ('B', 11246, 2), ('AE', 11369, 2), + ('', 11511, 0), ('W', 11590, 2), ('EH', 11622, 1), ('N', 11705, 2), + ('Y', 11755, 2), ('UH', 11808, 2), ('R', 11864, 2), ('AH', 11959, 2), + ('L', 12095, 2), ('IH', 12202, 2), ('L', 12386, 2), + ('', 12596, 0), ('M', 12748, 2), ('M', 12888, 2), ('M', 13037, 2), ('M', 13196, 2), ('', 13426, 0), + ('R', 13494, 2), ('IH', 13589, 2), ('', 13632, 0), ('CH', 13773, 2), ('ER', 13991, 2), ('', 13992, 0) +], names=('phoneme', 'timing', 'priority')) + +similar_phonemes = { + 'AW': ['AE', 'UW'], + 'DH': ['D'], + 'EY': ['EH', 'IY'], + 'JH': ['CH'], + 'SH': ['CH'], + 'TH': ['D'], + 'ZH': ['CH'], +} + + +# --------------------------------------------------------------------------- +# Morshu TTS Engine +# --------------------------------------------------------------------------- +class Morshu: + def __init__(self): + self.input_str = "" + self.input_phonemes = [] + self.stop_chars = '.,?!:;()\n' + self.space_length = 20 + self.stop_length = 100 + self.use_phoneme_priority = True + self.out_audio = AudioSegment.empty() + self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu')) + self.canceled = False + + def cancel(self): + g2p.cancel() + self.canceled = True + + def load_text(self, text: str = None, progress_callback: Callable[[int, int, int], None] = None) \ + -> AudioSegment | Literal[False]: + """Generate audio from text. Returns AudioSegment or False if cancelled.""" + self.canceled = False + + if progress_callback is None: + progress_callback = lambda major_step, minor_step, minor_total: None + + if text is None: + text = self.input_str + self.input_str = text + text = text.replace('\n', ',,,') + + phonemes = g2p.run_with_progress(text, lambda step, total: progress_callback(0, step, total)) + if g2p.cancelled: + return False + + progress_step = 0 + progress_total = len(phonemes) + + output = AudioSegment.empty().set_frame_rate(morshu_wav.frame_rate) + audio_out_millis = [] + audio_morshu_millis = [] + phoneme_segment = [] + + while len(phonemes) > 0: + if self.canceled: + return False + + progress_callback(1, progress_step, progress_total) + progress_step += 1 + + p = phonemes.pop(0) + if p in g2p.phonemes: + phoneme_segment.append(p) + if p not in g2p.phonemes or len(phonemes) == 0: + output = self.append_best_morshu_phoneme_segment( + output, phoneme_segment, audio_out_millis, audio_morshu_millis) + phoneme_segment = [] + if p == ' ': + output = self.append_audio_segment( + output, AudioSegment.silent(self.space_length), -1, + audio_out_millis, audio_morshu_millis) + elif p in self.stop_chars: + output = self.append_audio_segment( + output, AudioSegment.silent(self.stop_length), -1, + audio_out_millis, audio_morshu_millis) + + if len(output) == 0: + warnings.warn('returned audio segment is empty', UserWarning) + self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu')) + else: + self.audio_segment_timings = np.rec.array( + tuple(zip(audio_out_millis, audio_morshu_millis)), + names=('output', 'morshu')) + + progress_callback(1, progress_total, progress_total) + self.out_audio = output + return output + + @staticmethod + def substitute_similar_phonemes(phonemes: List[str]): + i = 0 + while i < len(phonemes): + if phonemes[i].endswith('0') or phonemes[i].endswith('1') or phonemes[i].endswith('2'): + phonemes[i] = phonemes[i][:len(phonemes[i]) - 1] + if phonemes[i] in similar_phonemes: + phonemes = phonemes[0:i] + similar_phonemes[phonemes[i]] + phonemes[i + 1:] + i += 1 + return phonemes + + @staticmethod + def append_audio_segment(audio_out: AudioSegment, audio_segment: AudioSegment, + morshu_millis_start: int, + audio_out_millis: List[int], + audio_morshu_millis: List[int]) -> AudioSegment: + audio_out_millis.append(len(audio_out)) + audio_morshu_millis.append(morshu_millis_start) + audio_out += audio_segment + return audio_out + + @staticmethod + def get_phoneme_sequence_occurrences(phonemes: List[str]) -> List[Tuple[int, int]]: + occurrences = [] + for i in range(len(morshu_rec) - len(phonemes)): + if (morshu_rec['phoneme'][i:i + len(phonemes)] == phonemes).all(): + start = morshu_rec['timing'][i - 1] + end = morshu_rec['timing'][i + len(phonemes) - 1] + occurrences.append((start, end)) + return occurrences + + def get_best_morshu_single_phoneme(self, phoneme: str, preceding: str = "", + succeeding: str = "") -> Tuple[AudioSegment, int]: + best_indices = [] + phoneme_indices = np.where(morshu_rec['phoneme'] == phoneme)[0] + if len(phoneme_indices) == 0: + return AudioSegment.empty(), 0 + + highest_priority = 0 + for i in phoneme_indices: + morshu_preceding = morshu_rec['phoneme'][i - 1] + priority = morshu_rec['priority'][i] if self.use_phoneme_priority else 0 + + if morshu_preceding == preceding: + priority += 10 + elif any(c in morshu_preceding for c in "AEIOU") and any(c in preceding for c in "AEIOU"): + priority += 5 + + morshu_succeeding = morshu_rec['phoneme'][i + 1] + if morshu_succeeding == succeeding: + priority += 10 + elif any(c in morshu_succeeding for c in "AEIOU") and any(c in succeeding for c in "AEIOU"): + priority += 1 + + if priority < highest_priority: + continue + if priority > highest_priority: + highest_priority = priority + best_indices = [] + best_indices.append(i) + + index = random.choice(best_indices) + segment = morshu_wav[morshu_rec['timing'][index - 1]: morshu_rec['timing'][index]] + return segment, morshu_rec['timing'][index - 1] + + def append_best_morshu_phoneme_segment(self, output: AudioSegment, phonemes: List[str], + audio_out_millis: List[int] = None, + audio_morshu_millis: List[int] = None) -> AudioSegment: + phonemes = Morshu.substitute_similar_phonemes(phonemes) + if len(phonemes) == 1: + segment, start = self.get_best_morshu_single_phoneme(phonemes[0]) + return Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis) + + preceding = "" + while len(phonemes) > 0: + sequence_length = 1 + segment = AudioSegment.empty() + start = 0 + + while sequence_length <= len(phonemes): + occurrences = Morshu.get_phoneme_sequence_occurrences(phonemes[:sequence_length]) + if len(occurrences) == 0: + break + start, end = random.choice(occurrences) + segment = morshu_wav[start:end] + sequence_length += 1 + sequence_length -= 1 + + if sequence_length == 1: + succeeding = phonemes[sequence_length + 1] if sequence_length + 1 < len(phonemes) else "" + segment, start = self.get_best_morshu_single_phoneme(phonemes[0], preceding, succeeding) + + output = Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis) + preceding = phonemes[sequence_length - 1] + del phonemes[:sequence_length] + + return output diff --git a/morshutalk_morshu.wav b/morshutalk_morshu.wav new file mode 100644 index 0000000..e0db031 Binary files /dev/null and b/morshutalk_morshu.wav differ diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..adeecb2 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +numpy>=1.24 +pydub>=0.25 +nltk>=3.8 +g2p_en>=2.1 diff --git a/server.py b/server.py new file mode 100644 index 0000000..de2de97 --- /dev/null +++ b/server.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +""" +Morshu TTS Server — standalone HTTP API for MorshuTalk engine. +Runs on port 33002. GET /say?text=... returns WAV audio. +""" +import os +import sys +import io +from pathlib import Path +from http.server import HTTPServer, BaseHTTPRequestHandler +from urllib.parse import urlparse, parse_qs + +# Engine lives alongside this server (self-contained) +SCRIPT_DIR = Path(__file__).resolve().parent +sys.path.insert(0, str(SCRIPT_DIR)) + +from morshutalk_engine import Morshu + +morshu = Morshu() +PORT = int(os.environ.get("PORT", 33002)) + + +class Handler(BaseHTTPRequestHandler): + def do_GET(self): + parsed = urlparse(self.path) + if parsed.path == "/say": + params = parse_qs(parsed.query) + text = params.get("text", [""])[0] + if not text: + self.send_error(400, "Missing 'text' parameter") + return + try: + audio = morshu.load_text(text) + if audio is False or len(audio) == 0: + self.send_error(500, "Failed to generate audio") + return + buf = io.BytesIO() + audio.export(buf, format="wav") + wav_bytes = buf.getvalue() + self.send_response(200) + self.send_header("Content-Type", "audio/wav") + self.send_header("Content-Length", str(len(wav_bytes))) + self.end_headers() + self.wfile.write(wav_bytes) + except Exception as e: + self.send_error(500, str(e)) + elif parsed.path == "/health": + self.send_response(200) + self.send_header("Content-Type", "text/plain") + self.end_headers() + self.wfile.write(b"ok") + else: + self.send_error(404) + + def log_message(self, format, *args): + print(f"[Morshu] {args[0]}") + + +if __name__ == "__main__": + print(f"[Morshu] Starting server on port {PORT}...") + server = HTTPServer(("0.0.0.0", PORT), Handler) + print(f"[Morshu] Ready — http://127.0.0.1:{PORT}/say?text=hello") + server.serve_forever() diff --git a/start.sh b/start.sh new file mode 100755 index 0000000..1d1efe3 --- /dev/null +++ b/start.sh @@ -0,0 +1,17 @@ +#!/bin/bash +# Start script for the Morshu TTS server. +# Creates a local venv on first run, installs deps, then runs the server. +set -e + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +cd "$SCRIPT_DIR" + +if [ ! -d venv ]; then + echo "Creating virtual environment..." + python3 -m venv venv + ./venv/bin/pip install --upgrade pip + ./venv/bin/pip install -r requirements.txt +fi + +export PORT="${PORT:-33002}" +exec ./venv/bin/python "$SCRIPT_DIR/server.py"