fix: trim long chatterbox voice clips to 30s on upload
This commit is contained in:
+51
-7
@@ -169,6 +169,54 @@ def resolve_voice_for_mode(mode: str, voice_name: str, chatterbox_voices: list =
|
|||||||
return voice_name
|
return voice_name
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_chatterbox_reference(
|
||||||
|
audio_seg, min_ms: int = 6000, max_ms: int = 30000
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
Normalize a voice reference clip to a usable length for Chatterbox cloning.
|
||||||
|
|
||||||
|
- Clips shorter than ``min_ms`` are looped until they reach the minimum
|
||||||
|
duration required for a usable voice reference (>5s).
|
||||||
|
- Clips longer than ``max_ms`` are trimmed to the server's
|
||||||
|
``max_reference_duration_sec`` limit, preferring a silence boundary near
|
||||||
|
the cut so words aren't chopped mid-syllable.
|
||||||
|
|
||||||
|
Returns the processed AudioSegment.
|
||||||
|
"""
|
||||||
|
if audio_seg is None:
|
||||||
|
return audio_seg
|
||||||
|
|
||||||
|
if len(audio_seg) < min_ms:
|
||||||
|
loops_needed = (min_ms // len(audio_seg)) + 1 if len(audio_seg) > 0 else 1
|
||||||
|
audio_seg = audio_seg * loops_needed
|
||||||
|
log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s")
|
||||||
|
|
||||||
|
if len(audio_seg) > max_ms:
|
||||||
|
trimmed = audio_seg[:max_ms]
|
||||||
|
try:
|
||||||
|
from pydub.silence import detect_silence
|
||||||
|
|
||||||
|
search_start = max(0, max_ms - 1500)
|
||||||
|
silences = detect_silence(
|
||||||
|
audio_seg[search_start:max_ms],
|
||||||
|
min_silence_len=300,
|
||||||
|
silence_thresh=-40,
|
||||||
|
)
|
||||||
|
if silences:
|
||||||
|
cutoff_local = silences[0][0]
|
||||||
|
if cutoff_local > 100:
|
||||||
|
trimmed = audio_seg[: search_start + cutoff_local]
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
log.info(
|
||||||
|
f"[Chatterbox] Trimmed long clip from {len(audio_seg)/1000:.1f}s "
|
||||||
|
f"to {len(trimmed)/1000:.1f}s"
|
||||||
|
)
|
||||||
|
audio_seg = trimmed
|
||||||
|
|
||||||
|
return audio_seg
|
||||||
|
|
||||||
|
|
||||||
def _split_voice_segments(text: str):
|
def _split_voice_segments(text: str):
|
||||||
"""
|
"""
|
||||||
Split text on [mode|voice] tags into segments.
|
Split text on [mode|voice] tags into segments.
|
||||||
@@ -1591,17 +1639,13 @@ class TtsToy(Cog):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return await ctx.send(f"❌ Failed to download audio from URL: {e}")
|
return await ctx.send(f"❌ Failed to download audio from URL: {e}")
|
||||||
|
|
||||||
# Ensure minimum duration (Chatterbox requires >5 seconds)
|
# Ensure reference clip is a usable length for voice cloning:
|
||||||
# Loop short clips until they're long enough
|
# loop short clips up to the minimum duration, trim long clips to the max
|
||||||
try:
|
try:
|
||||||
from pydub import AudioSegment
|
from pydub import AudioSegment
|
||||||
import io as _io
|
import io as _io
|
||||||
audio_seg = AudioSegment.from_file(_io.BytesIO(audio_bytes))
|
audio_seg = AudioSegment.from_file(_io.BytesIO(audio_bytes))
|
||||||
min_ms = 6000 # 6 seconds to be safe
|
audio_seg = _normalize_chatterbox_reference(audio_seg)
|
||||||
if len(audio_seg) < min_ms:
|
|
||||||
loops_needed = (min_ms // len(audio_seg)) + 1
|
|
||||||
audio_seg = audio_seg * loops_needed
|
|
||||||
log.info(f"[Chatterbox] Looped short clip {loops_needed}x to {len(audio_seg)/1000:.1f}s")
|
|
||||||
# Always export as wav for best compatibility
|
# Always export as wav for best compatibility
|
||||||
buf = _io.BytesIO()
|
buf = _io.BytesIO()
|
||||||
audio_seg.export(buf, format="wav")
|
audio_seg.export(buf, format="wav")
|
||||||
|
|||||||
Reference in New Issue
Block a user