From 6aac9bcae916da2326ae74b3ab14a23f33fcf8e6 Mon Sep 17 00:00:00 2001 From: owen Date: Mon, 7 Sep 2026 23:38:56 -0500 Subject: [PATCH] fix: trim long chatterbox voice clips to 30s on upload --- ttstoy/ttstoy.py | 58 ++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 51 insertions(+), 7 deletions(-) diff --git a/ttstoy/ttstoy.py b/ttstoy/ttstoy.py index 297ecfa..7b040c7 100644 --- a/ttstoy/ttstoy.py +++ b/ttstoy/ttstoy.py @@ -169,6 +169,54 @@ def resolve_voice_for_mode(mode: str, voice_name: str, chatterbox_voices: list = return voice_name +def _normalize_chatterbox_reference( + audio_seg, min_ms: int = 6000, max_ms: int = 30000 +): + """ + Normalize a voice reference clip to a usable length for Chatterbox cloning. + + - Clips shorter than ``min_ms`` are looped until they reach the minimum + duration required for a usable voice reference (>5s). + - Clips longer than ``max_ms`` are trimmed to the server's + ``max_reference_duration_sec`` limit, preferring a silence boundary near + the cut so words aren't chopped mid-syllable. + + Returns the processed AudioSegment. + """ + if audio_seg is None: + return audio_seg + + if len(audio_seg) < min_ms: + loops_needed = (min_ms // len(audio_seg)) + 1 if len(audio_seg) > 0 else 1 + audio_seg = audio_seg * loops_needed + log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s") + + if len(audio_seg) > max_ms: + trimmed = audio_seg[:max_ms] + try: + from pydub.silence import detect_silence + + search_start = max(0, max_ms - 1500) + silences = detect_silence( + audio_seg[search_start:max_ms], + min_silence_len=300, + silence_thresh=-40, + ) + if silences: + cutoff_local = silences[0][0] + if cutoff_local > 100: + trimmed = audio_seg[: search_start + cutoff_local] + except Exception: + pass + log.info( + f"[Chatterbox] Trimmed long clip from {len(audio_seg)/1000:.1f}s " + f"to {len(trimmed)/1000:.1f}s" + ) + audio_seg = trimmed + + return audio_seg + + def _split_voice_segments(text: str): """ Split text on [mode|voice] tags into segments. @@ -1591,17 +1639,13 @@ class TtsToy(Cog): except Exception as e: return await ctx.send(f"❌ Failed to download audio from URL: {e}") - # Ensure minimum duration (Chatterbox requires >5 seconds) - # Loop short clips until they're long enough + # Ensure reference clip is a usable length for voice cloning: + # loop short clips up to the minimum duration, trim long clips to the max try: from pydub import AudioSegment import io as _io audio_seg = AudioSegment.from_file(_io.BytesIO(audio_bytes)) - min_ms = 6000 # 6 seconds to be safe - if len(audio_seg) < min_ms: - loops_needed = (min_ms // len(audio_seg)) + 1 - audio_seg = audio_seg * loops_needed - log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s") + audio_seg = _normalize_chatterbox_reference(audio_seg) # Always export as wav for best compatibility buf = _io.BytesIO() audio_seg.export(buf, format="wav")