fix: trim long chatterbox voice clips to 30s on upload

This commit is contained in:
owen
2026-09-07 23:38:56 -05:00
parent e0ff129b6b
commit 6aac9bcae9
+51 -7
View File
@@ -169,6 +169,54 @@ def resolve_voice_for_mode(mode: str, voice_name: str, chatterbox_voices: list =
return voice_name
def _normalize_chatterbox_reference(
audio_seg, min_ms: int = 6000, max_ms: int = 30000
):
"""
Normalize a voice reference clip to a usable length for Chatterbox cloning.
- Clips shorter than ``min_ms`` are looped until they reach the minimum
duration required for a usable voice reference (>5s).
- Clips longer than ``max_ms`` are trimmed to the server's
``max_reference_duration_sec`` limit, preferring a silence boundary near
the cut so words aren't chopped mid-syllable.
Returns the processed AudioSegment.
"""
if audio_seg is None:
return audio_seg
if len(audio_seg) < min_ms:
loops_needed = (min_ms // len(audio_seg)) + 1 if len(audio_seg) > 0 else 1
audio_seg = audio_seg * loops_needed
log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s")
if len(audio_seg) > max_ms:
trimmed = audio_seg[:max_ms]
try:
from pydub.silence import detect_silence
search_start = max(0, max_ms - 1500)
silences = detect_silence(
audio_seg[search_start:max_ms],
min_silence_len=300,
silence_thresh=-40,
)
if silences:
cutoff_local = silences[0][0]
if cutoff_local > 100:
trimmed = audio_seg[: search_start + cutoff_local]
except Exception:
pass
log.info(
f"[Chatterbox] Trimmed long clip from {len(audio_seg)/1000:.1f}s "
f"to {len(trimmed)/1000:.1f}s"
)
audio_seg = trimmed
return audio_seg
def _split_voice_segments(text: str):
"""
Split text on [mode|voice] tags into segments.
@@ -1591,17 +1639,13 @@ class TtsToy(Cog):
except Exception as e:
return await ctx.send(f"❌ Failed to download audio from URL: {e}")
# Ensure minimum duration (Chatterbox requires >5 seconds)
# Loop short clips until they're long enough
# Ensure reference clip is a usable length for voice cloning:
# loop short clips up to the minimum duration, trim long clips to the max
try:
from pydub import AudioSegment
import io as _io
audio_seg = AudioSegment.from_file(_io.BytesIO(audio_bytes))
min_ms = 6000 # 6 seconds to be safe
if len(audio_seg) < min_ms:
loops_needed = (min_ms // len(audio_seg)) + 1
audio_seg = audio_seg * loops_needed
log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s")
audio_seg = _normalize_chatterbox_reference(audio_seg)
# Always export as wav for best compatibility
buf = _io.BytesIO()
audio_seg.export(buf, format="wav")