e61a320499
Self-contained HTTP service with engine, data, start script, systemd unit, and documentation. Runs independently of the Discord bot on its fixed port.
307 lines
12 KiB
Python
307 lines
12 KiB
Python
"""
|
|
Bundled MorshuTalk engine for ttstoy.
|
|
Based on MorshuTalk by jalenluorion (https://github.com/jalenluorion/MorshuTalk)
|
|
Generates speech audio from Morshu's voice lines using phoneme matching.
|
|
"""
|
|
import re
|
|
import unicodedata
|
|
from os import path
|
|
from typing import List, Tuple, Callable, Literal
|
|
|
|
import numpy as np
|
|
import random
|
|
import warnings
|
|
from pydub import AudioSegment
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# G2P (Grapheme-to-Phoneme) wrapper with progress + cancel support
|
|
# ---------------------------------------------------------------------------
|
|
import nltk
|
|
for _res in ('averaged_perceptron_tagger_eng', 'cmudict', 'punkt', 'punkt_tab'):
|
|
try:
|
|
nltk.data.find(f'taggers/{_res}' if 'tagger' in _res else _res)
|
|
except LookupError:
|
|
nltk.download(_res, quiet=True)
|
|
|
|
from g2p_en.g2p import G2p, unicode, normalize_numbers, word_tokenize, pos_tag
|
|
|
|
|
|
class G2pProgress(G2p):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.cancelled = False
|
|
|
|
def cancel(self):
|
|
self.cancelled = True
|
|
|
|
def run_with_progress(self, text, callback: Callable[[int, int], None] = None):
|
|
self.cancelled = False
|
|
|
|
text = unicode(text)
|
|
text = normalize_numbers(text)
|
|
text = ''.join(char for char in unicodedata.normalize('NFD', text)
|
|
if unicodedata.category(char) != 'Mn')
|
|
text = text.lower()
|
|
text = re.sub("[^ a-z'.,?!\\-]", "", text)
|
|
text = text.replace("i.e.", "that is")
|
|
text = text.replace("e.g.", "for example")
|
|
|
|
words = word_tokenize(text)
|
|
tokens = pos_tag(words)
|
|
|
|
step = 0
|
|
total = len(tokens)
|
|
prons = []
|
|
|
|
for word, pos in tokens:
|
|
if self.cancelled:
|
|
return
|
|
|
|
if callback:
|
|
callback(step, total)
|
|
step += 1
|
|
|
|
if re.search("[a-z]", word) is None:
|
|
pron = [word]
|
|
elif word in self.homograph2features:
|
|
pron1, pron2, pos1 = self.homograph2features[word]
|
|
pron = pron1 if pos.startswith(pos1) else pron2
|
|
elif word in self.cmu:
|
|
pron = self.cmu[word][0]
|
|
else:
|
|
pron = self.predict(word)
|
|
|
|
prons.extend(pron)
|
|
prons.extend([" "])
|
|
|
|
if callback:
|
|
callback(total, total)
|
|
|
|
return prons[:-1]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Singleton G2P instance and audio data
|
|
# ---------------------------------------------------------------------------
|
|
g2p = G2pProgress()
|
|
|
|
_WAV_PATH = path.join(path.dirname(__file__), 'morshutalk_morshu.wav')
|
|
morshu_wav = AudioSegment.from_wav(_WAV_PATH)
|
|
|
|
# Phoneme timing record from the morshu audio
|
|
morshu_rec = np.rec.array([
|
|
('', 160, 0), ('L', 250, 2), ('AE', 348, 2), ('M', 420, 2), ('P', 510, 1),
|
|
('OY', 700, 2), ('L', 835, 1), ('', 1090, 0),
|
|
('R', 1180, 2), ('OW', 1300, 2), ('', 1390, 0), ('P', 1490, 2), ('', 1850, 0),
|
|
('B', 1895, 2), ('AA', 2090, 2), ('M', 2235, 2), ('Z', 2390, 2),
|
|
('', 2780, 0), ('Y', 2840, 2), ('UW', 2960, 2),
|
|
('W', 3030, 2), ('AA', 3110, 2), ('N', 3150, 1), ('IH', 3240, 2), ('T', 3370, 2), ('', 3810, 0),
|
|
('IH', 3960, 2), ('T', 4070, 2), ('Y', 4260, 2), ('UH', 4400, 2), ('R', 4510, 2), ('Z', 4600, 2),
|
|
('M', 4675, 2), ('AY', 4810, 2), ('', 4885, 0),
|
|
('F', 4930, 2), ('R', 4980, 2), ('EH', 5100, 2), ('N', 5240, 2), ('D', 5300, 2), ('', 5520, 0),
|
|
('AE', 5630, 2), ('Z', 5740, 2), ('L', 5870, 2), ('AO', 6000, 2), ('NG', 6140, 2),
|
|
('AE', 6170, 1), ('Z', 6265, 2), ('Y', 6300, 2), ('UW', 6380, 2),
|
|
('HH', 6450, 2), ('AE', 6510, 1), ('V', 6580, 2),
|
|
('IH', 6640, 2), ('N', 6670, 2), ('AH', 6747, 2), ('F', 6855, 2),
|
|
('R', 6960, 2), ('UW', 7060, 2), ('B', 7170, 1), ('IY', 7340, 2), ('Z', 7520, 2), ('', 8236, 0),
|
|
('S', 8407, 2), ('AA', 8495, 2), ('R', 8570, 2), ('IY', 8630, 1),
|
|
('L', 8740, 2), ('IH', 8811, 2), ('NG', 8942, 2), ('K', 9014, 2), ('', 9251, 0),
|
|
('AY', 9384, 2), ('', 9467, 0), ('K', 9512, 2), ('AE', 9640, 2), ('N', 9716, 2), ('', 9844, 0),
|
|
('G', 9894, 2), ('IH', 9985, 2), ('V', 10060, 2), ('', 10149, 0),
|
|
('K', 10256, 2), ('R', 10297, 2), ('EH', 10383, 2), ('IH', 10482, 1), ('', 10564, 0), ('T', 10617, 2),
|
|
('', 10962, 0), ('K', 11019, 2), ('AH', 11100, 2), ('M', 11229, 2), ('B', 11246, 2), ('AE', 11369, 2),
|
|
('', 11511, 0), ('W', 11590, 2), ('EH', 11622, 1), ('N', 11705, 2),
|
|
('Y', 11755, 2), ('UH', 11808, 2), ('R', 11864, 2), ('AH', 11959, 2),
|
|
('L', 12095, 2), ('IH', 12202, 2), ('L', 12386, 2),
|
|
('', 12596, 0), ('M', 12748, 2), ('M', 12888, 2), ('M', 13037, 2), ('M', 13196, 2), ('', 13426, 0),
|
|
('R', 13494, 2), ('IH', 13589, 2), ('', 13632, 0), ('CH', 13773, 2), ('ER', 13991, 2), ('', 13992, 0)
|
|
], names=('phoneme', 'timing', 'priority'))
|
|
|
|
similar_phonemes = {
|
|
'AW': ['AE', 'UW'],
|
|
'DH': ['D'],
|
|
'EY': ['EH', 'IY'],
|
|
'JH': ['CH'],
|
|
'SH': ['CH'],
|
|
'TH': ['D'],
|
|
'ZH': ['CH'],
|
|
}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Morshu TTS Engine
|
|
# ---------------------------------------------------------------------------
|
|
class Morshu:
|
|
def __init__(self):
|
|
self.input_str = ""
|
|
self.input_phonemes = []
|
|
self.stop_chars = '.,?!:;()\n'
|
|
self.space_length = 20
|
|
self.stop_length = 100
|
|
self.use_phoneme_priority = True
|
|
self.out_audio = AudioSegment.empty()
|
|
self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu'))
|
|
self.canceled = False
|
|
|
|
def cancel(self):
|
|
g2p.cancel()
|
|
self.canceled = True
|
|
|
|
def load_text(self, text: str = None, progress_callback: Callable[[int, int, int], None] = None) \
|
|
-> AudioSegment | Literal[False]:
|
|
"""Generate audio from text. Returns AudioSegment or False if cancelled."""
|
|
self.canceled = False
|
|
|
|
if progress_callback is None:
|
|
progress_callback = lambda major_step, minor_step, minor_total: None
|
|
|
|
if text is None:
|
|
text = self.input_str
|
|
self.input_str = text
|
|
text = text.replace('\n', ',,,')
|
|
|
|
phonemes = g2p.run_with_progress(text, lambda step, total: progress_callback(0, step, total))
|
|
if g2p.cancelled:
|
|
return False
|
|
|
|
progress_step = 0
|
|
progress_total = len(phonemes)
|
|
|
|
output = AudioSegment.empty().set_frame_rate(morshu_wav.frame_rate)
|
|
audio_out_millis = []
|
|
audio_morshu_millis = []
|
|
phoneme_segment = []
|
|
|
|
while len(phonemes) > 0:
|
|
if self.canceled:
|
|
return False
|
|
|
|
progress_callback(1, progress_step, progress_total)
|
|
progress_step += 1
|
|
|
|
p = phonemes.pop(0)
|
|
if p in g2p.phonemes:
|
|
phoneme_segment.append(p)
|
|
if p not in g2p.phonemes or len(phonemes) == 0:
|
|
output = self.append_best_morshu_phoneme_segment(
|
|
output, phoneme_segment, audio_out_millis, audio_morshu_millis)
|
|
phoneme_segment = []
|
|
if p == ' ':
|
|
output = self.append_audio_segment(
|
|
output, AudioSegment.silent(self.space_length), -1,
|
|
audio_out_millis, audio_morshu_millis)
|
|
elif p in self.stop_chars:
|
|
output = self.append_audio_segment(
|
|
output, AudioSegment.silent(self.stop_length), -1,
|
|
audio_out_millis, audio_morshu_millis)
|
|
|
|
if len(output) == 0:
|
|
warnings.warn('returned audio segment is empty', UserWarning)
|
|
self.audio_segment_timings = np.rec.array((0, 0), names=('output', 'morshu'))
|
|
else:
|
|
self.audio_segment_timings = np.rec.array(
|
|
tuple(zip(audio_out_millis, audio_morshu_millis)),
|
|
names=('output', 'morshu'))
|
|
|
|
progress_callback(1, progress_total, progress_total)
|
|
self.out_audio = output
|
|
return output
|
|
|
|
@staticmethod
|
|
def substitute_similar_phonemes(phonemes: List[str]):
|
|
i = 0
|
|
while i < len(phonemes):
|
|
if phonemes[i].endswith('0') or phonemes[i].endswith('1') or phonemes[i].endswith('2'):
|
|
phonemes[i] = phonemes[i][:len(phonemes[i]) - 1]
|
|
if phonemes[i] in similar_phonemes:
|
|
phonemes = phonemes[0:i] + similar_phonemes[phonemes[i]] + phonemes[i + 1:]
|
|
i += 1
|
|
return phonemes
|
|
|
|
@staticmethod
|
|
def append_audio_segment(audio_out: AudioSegment, audio_segment: AudioSegment,
|
|
morshu_millis_start: int,
|
|
audio_out_millis: List[int],
|
|
audio_morshu_millis: List[int]) -> AudioSegment:
|
|
audio_out_millis.append(len(audio_out))
|
|
audio_morshu_millis.append(morshu_millis_start)
|
|
audio_out += audio_segment
|
|
return audio_out
|
|
|
|
@staticmethod
|
|
def get_phoneme_sequence_occurrences(phonemes: List[str]) -> List[Tuple[int, int]]:
|
|
occurrences = []
|
|
for i in range(len(morshu_rec) - len(phonemes)):
|
|
if (morshu_rec['phoneme'][i:i + len(phonemes)] == phonemes).all():
|
|
start = morshu_rec['timing'][i - 1]
|
|
end = morshu_rec['timing'][i + len(phonemes) - 1]
|
|
occurrences.append((start, end))
|
|
return occurrences
|
|
|
|
def get_best_morshu_single_phoneme(self, phoneme: str, preceding: str = "",
|
|
succeeding: str = "") -> Tuple[AudioSegment, int]:
|
|
best_indices = []
|
|
phoneme_indices = np.where(morshu_rec['phoneme'] == phoneme)[0]
|
|
if len(phoneme_indices) == 0:
|
|
return AudioSegment.empty(), 0
|
|
|
|
highest_priority = 0
|
|
for i in phoneme_indices:
|
|
morshu_preceding = morshu_rec['phoneme'][i - 1]
|
|
priority = morshu_rec['priority'][i] if self.use_phoneme_priority else 0
|
|
|
|
if morshu_preceding == preceding:
|
|
priority += 10
|
|
elif any(c in morshu_preceding for c in "AEIOU") and any(c in preceding for c in "AEIOU"):
|
|
priority += 5
|
|
|
|
morshu_succeeding = morshu_rec['phoneme'][i + 1]
|
|
if morshu_succeeding == succeeding:
|
|
priority += 10
|
|
elif any(c in morshu_succeeding for c in "AEIOU") and any(c in succeeding for c in "AEIOU"):
|
|
priority += 1
|
|
|
|
if priority < highest_priority:
|
|
continue
|
|
if priority > highest_priority:
|
|
highest_priority = priority
|
|
best_indices = []
|
|
best_indices.append(i)
|
|
|
|
index = random.choice(best_indices)
|
|
segment = morshu_wav[morshu_rec['timing'][index - 1]: morshu_rec['timing'][index]]
|
|
return segment, morshu_rec['timing'][index - 1]
|
|
|
|
def append_best_morshu_phoneme_segment(self, output: AudioSegment, phonemes: List[str],
|
|
audio_out_millis: List[int] = None,
|
|
audio_morshu_millis: List[int] = None) -> AudioSegment:
|
|
phonemes = Morshu.substitute_similar_phonemes(phonemes)
|
|
if len(phonemes) == 1:
|
|
segment, start = self.get_best_morshu_single_phoneme(phonemes[0])
|
|
return Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis)
|
|
|
|
preceding = ""
|
|
while len(phonemes) > 0:
|
|
sequence_length = 1
|
|
segment = AudioSegment.empty()
|
|
start = 0
|
|
|
|
while sequence_length <= len(phonemes):
|
|
occurrences = Morshu.get_phoneme_sequence_occurrences(phonemes[:sequence_length])
|
|
if len(occurrences) == 0:
|
|
break
|
|
start, end = random.choice(occurrences)
|
|
segment = morshu_wav[start:end]
|
|
sequence_length += 1
|
|
sequence_length -= 1
|
|
|
|
if sequence_length == 1:
|
|
succeeding = phonemes[sequence_length + 1] if sequence_length + 1 < len(phonemes) else ""
|
|
segment, start = self.get_best_morshu_single_phoneme(phonemes[0], preceding, succeeding)
|
|
|
|
output = Morshu.append_audio_segment(output, segment, start, audio_out_millis, audio_morshu_millis)
|
|
preceding = phonemes[sequence_length - 1]
|
|
del phonemes[:sequence_length]
|
|
|
|
return output
|