Files
2026-06-05 16:41:55 -05:00

307 lines
12 KiB
Python

"""
Bundled MorshuTalk engine for ttstoy.
Based on MorshuTalk by jalenluorion (https://github.com/jalenluorion/MorshuTalk)
Generates speech audio from Morshu's voice lines using phoneme matching.
"""
import re
import unicodedata
from os import path
from typing import List, Tuple, Callable, Literal
import numpy as np
import random
import warnings
from pydub import AudioSegment
# ---------------------------------------------------------------------------
# G2P (Grapheme-to-Phoneme) wrapper with progress + cancel support
# ---------------------------------------------------------------------------
import nltk
for _res in ('averaged_perceptron_tagger_eng', 'cmudict', 'punkt', 'punkt_tab'):
try:
nltk.data.find(f'taggers/{_res}' if 'tagger' in _res else _res)
except LookupError:
nltk.download(_res, quiet=True)
from g2p_en.g2p import G2p, unicode, normalize_numbers, word_tokenize, pos_tag
class G2pProgress(G2p):
def __init__(self):
super().__init__()
self.cancelled = False
def cancel(self):
self.cancelled = True
def run_with_progress(self, text, callback: Callable[[int, int], None] = None):
self.cancelled = False
text = unicode(text)
text = normalize_numbers(text)
text = ''.join(char for char in unicodedata.normalize('NFD', text)
if unicodedata.category(char) != 'Mn')
text = text.lower()
text = re.sub("[^ a-z'.,?!\\-]", "", text)
text = text.replace("i.e.", "that is")
text = text.replace("e.g.", "for example")
words = word_tokenize(text)
tokens = pos_tag(words)
step = 0
total = len(tokens)
prons = []
for word, pos in tokens:
if self.cancelled:
return
if callback:
callback(step, total)
step += 1
if re.search("[a-z]", word) is None:
pron = [word]
elif word in self.homograph2features:
pron1, pron2, pos1 = self.homograph2features[word]
pron = pron1 if pos.startswith(pos1) else pron2
elif word in self.cmu:
pron = self.cmu[word][0]
else:
pron = self.predict(word)
prons.extend(pron)
prons.extend([" "])
if callback:
callback(total, total)
return prons[:-1]
# ---------------------------------------------------------------------------
# Singleton G2P instance and audio data
# ---------------------------------------------------------------------------
g2p = G2pProgress()
_WAV_PATH = path.join(path.dirname(__file__), 'morshutalk_morshu.wav')
morshu_wav = AudioSegment.from_wav(_WAV_PATH)
# Phoneme timing record from the morshu audio
morshu_rec = np.rec.array([
('', 160, 0), ('L', 250, 2), ('AE', 348, 2), ('M', 420, 2), ('P', 510, 1),
('OY', 700, 2), ('L', 835, 1), ('', 1090, 0),
('R', 1180, 2), ('OW', 1300, 2), ('', 1390, 0), ('P', 1490, 2), ('', 1850, 0),
('B', 1895, 2), ('AA', 2090, 2), ('M', 2235, 2), ('Z', 2390, 2),
('', 2780, 0), ('Y', 2840, 2), ('UW', 2960, 2),
('W', 3030, 2), ('AA', 3110, 2), ('N', 3150, 1), ('IH', 3240, 2), ('T', 3370, 2), ('', 3810, 0),
('IH', 3960, 2), ('T', 4070, 2), ('Y', 4260, 2), ('UH', 4400, 2), ('R', 4510, 2), ('Z', 4600, 2),
('M', 4675, 2), ('AY', 4810, 2), ('', 4885, 0),
('F', 4930, 2), ('R', 4980, 2), ('EH', 5100, 2), ('N', 5240, 2), ('D', 5300, 2), ('', 5520, 0),
('AE', 5630, 2), ('Z', 5740, 2), ('L', 5870, 2), ('AO', 6000, 2), ('NG', 6140, 2),
('AE', 6170, 1), ('Z', 6265, 2), ('Y', 6300, 2), ('UW', 6380, 2),
('HH', 6450, 2), ('AE', 6510, 1), ('V', 6580, 2),
('IH', 6640, 2), ('N', 6670, 2), ('AH', 6747, 2), ('F', 6855, 2),
('R', 6960, 2), ('UW', 7060, 2), ('B', 7170, 1), ('IY', 7340, 2), ('Z', 7520, 2), ('', 8236, 0),
('S', 8407, 2), ('AA', 8495, 2), ('R', 8570, 2), ('IY', 8630, 1),
('L', 8740, 2), ('IH', 8811, 2), ('NG', 8942, 2), ('K', 9014, 2), ('', 9251, 0),
('AY', 9384, 2), ('', 9467, 0), ('K', 9512, 2), ('AE', 9640, 2), ('N', 9716, 2), ('', 9844, 0),
('G', 9894, 2), ('IH', 9985, 2), ('V', 10060, 2), ('', 10149, 0),
('K', 10256, 2), ('R', 10297, 2), ('EH', 10383, 2), ('IH', 10482, 1), ('', 10564, 0), ('T', 10617, 2),
('', 10962, 0), ('K', 11019, 2), ('AH', 11100, 2), ('M', 11229, 2), ('B', 11246, 2), ('AE', 11369, 2),
('', 11511, 0), ('W', 11590, 2), ('EH', 11622, 1), ('N', 11705, 2),
('Y', 11755, 2), ('UH', 11808, 2), ('R', 11864, 2), ('AH', 11959, 2),
('L', 12095, 2), ('IH', 12202, 2), ('L', 12386, 2),
('', 12596, 0), ('M', 12748, 2), ('M', 12888, 2), ('M', 13037, 2), ('M', 13196, 2), ('', 13426, 0),
('R', 13494, 2), ('IH', 13589, 2), ('', 13632, 0), ('CH', 13773, 2), ('ER', 13991, 2), ('', 13992, 0)
], names=('phoneme', 'timing', 'priority'))
similar_phonemes = {
'AW': ['AE', 'UW'],
'DH': ['D'],
'EY': ['EH', 'IY'],
'JH': ['CH'],
'SH': ['CH'],
'TH': ['D'],
'ZH': ['CH'],
}
# ---------------------------------------------------------------------------
# Morshu TTS Engine
# ---------------------------------------------------------------------------
class Morshu:
def __init__(self):
self.input_str = ""
self.input_phonemes = []
self.stop_chars = '.,?!:;()\n'
self.space_length = 20
self.stop_length = 100
self.use_phoneme_priority = True
self.out_audio = AudioSegment.empty()
self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu'))
self.canceled = False
def cancel(self):
g2p.cancel()
self.canceled = True
def load_text(self, text: str = None, progress_callback: Callable[[int, int, int], None] = None) \
-> AudioSegment | Literal[False]:
"""Generate audio from text. Returns AudioSegment or False if cancelled."""
self.canceled = False
if progress_callback is None:
progress_callback = lambda major_step, minor_step, minor_total: None
if text is None:
text = self.input_str
self.input_str = text
text = text.replace('\n', ',,,')
phonemes = g2p.run_with_progress(text, lambda step, total: progress_callback(0, step, total))
if g2p.cancelled:
return False
progress_step = 0
progress_total = len(phonemes)
output = AudioSegment.empty().set_frame_rate(morshu_wav.frame_rate)
audio_out_millis = []
audio_morshu_millis = []
phoneme_segment = []
while len(phonemes) > 0:
if self.canceled:
return False
progress_callback(1, progress_step, progress_total)
progress_step += 1
p = phonemes.pop(0)
if p in g2p.phonemes:
phoneme_segment.append(p)
if p not in g2p.phonemes or len(phonemes) == 0:
output = self.append_best_morshu_phoneme_segment(
output, phoneme_segment, audio_out_millis, audio_morshu_millis)
phoneme_segment = []
if p == ' ':
output = self.append_audio_segment(
output, AudioSegment.silent(self.space_length), -1,
audio_out_millis, audio_morshu_millis)
elif p in self.stop_chars:
output = self.append_audio_segment(
output, AudioSegment.silent(self.stop_length), -1,
audio_out_millis, audio_morshu_millis)
if len(output) == 0:
warnings.warn('returned audio segment is empty', UserWarning)
self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu'))
else:
self.audio_segment_timings = np.rec.array(
tuple(zip(audio_out_millis, audio_morshu_millis)),
names=('output', 'morshu'))
progress_callback(1, progress_total, progress_total)
self.out_audio = output
return output
@staticmethod
def substitute_similar_phonemes(phonemes: List[str]):
i = 0
while i < len(phonemes):
if phonemes[i].endswith('0') or phonemes[i].endswith('1') or phonemes[i].endswith('2'):
phonemes[i] = phonemes[i][:len(phonemes[i]) - 1]
if phonemes[i] in similar_phonemes:
phonemes = phonemes[0:i] + similar_phonemes[phonemes[i]] + phonemes[i + 1:]
i += 1
return phonemes
@staticmethod
def append_audio_segment(audio_out: AudioSegment, audio_segment: AudioSegment,
morshu_millis_start: int,
audio_out_millis: List[int],
audio_morshu_millis: List[int]) -> AudioSegment:
audio_out_millis.append(len(audio_out))
audio_morshu_millis.append(morshu_millis_start)
audio_out += audio_segment
return audio_out
@staticmethod
def get_phoneme_sequence_occurrences(phonemes: List[str]) -> List[Tuple[int, int]]:
occurrences = []
for i in range(len(morshu_rec) - len(phonemes)):
if (morshu_rec['phoneme'][i:i + len(phonemes)] == phonemes).all():
start = morshu_rec['timing'][i - 1]
end = morshu_rec['timing'][i + len(phonemes) - 1]
occurrences.append((start, end))
return occurrences
def get_best_morshu_single_phoneme(self, phoneme: str, preceding: str = "",
succeeding: str = "") -> Tuple[AudioSegment, int]:
best_indices = []
phoneme_indices = np.where(morshu_rec['phoneme'] == phoneme)[0]
if len(phoneme_indices) == 0:
return AudioSegment.empty(), 0
highest_priority = 0
for i in phoneme_indices:
morshu_preceding = morshu_rec['phoneme'][i - 1]
priority = morshu_rec['priority'][i] if self.use_phoneme_priority else 0
if morshu_preceding == preceding:
priority += 10
elif any(c in morshu_preceding for c in "AEIOU") and any(c in preceding for c in "AEIOU"):
priority += 5
morshu_succeeding = morshu_rec['phoneme'][i + 1]
if morshu_succeeding == succeeding:
priority += 10
elif any(c in morshu_succeeding for c in "AEIOU") and any(c in succeeding for c in "AEIOU"):
priority += 1
if priority < highest_priority:
continue
if priority > highest_priority:
highest_priority = priority
best_indices = []
best_indices.append(i)
index = random.choice(best_indices)
segment = morshu_wav[morshu_rec['timing'][index - 1]: morshu_rec['timing'][index]]
return segment, morshu_rec['timing'][index - 1]
def append_best_morshu_phoneme_segment(self, output: AudioSegment, phonemes: List[str],
audio_out_millis: List[int] = None,
audio_morshu_millis: List[int] = None) -> AudioSegment:
phonemes = Morshu.substitute_similar_phonemes(phonemes)
if len(phonemes) == 1:
segment, start = self.get_best_morshu_single_phoneme(phonemes[0])
return Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis)
preceding = ""
while len(phonemes) > 0:
sequence_length = 1
segment = AudioSegment.empty()
start = 0
while sequence_length <= len(phonemes):
occurrences = Morshu.get_phoneme_sequence_occurrences(phonemes[:sequence_length])
if len(occurrences) == 0:
break
start, end = random.choice(occurrences)
segment = morshu_wav[start:end]
sequence_length += 1
sequence_length -= 1
if sequence_length == 1:
succeeding = phonemes[sequence_length + 1] if sequence_length + 1 < len(phonemes) else ""
segment, start = self.get_best_morshu_single_phoneme(phonemes[0], preceding, succeeding)
output = Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis)
preceding = phonemes[sequence_length - 1]
del phonemes[:sequence_length]
return output