From 7664005a5d0bbfbf8b877b94efdf49cf6c421d84 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:23:19 -0700 Subject: [PATCH] refactor(tools): pack tts constant tables, writelines for elevenlabs, lease release result literal --- tools/tts_tool_delivery.py | 16 +++++----------- tools/tts_tool_lifecycle.py | 6 ++---- tools/tts_tool_providers.py | 5 ++--- 3 files changed, 9 insertions(+), 18 deletions(-) diff --git a/tools/tts_tool_delivery.py b/tools/tts_tool_delivery.py index bae7da2bf0..6836b693ee 100644 --- a/tools/tts_tool_delivery.py +++ b/tools/tts_tool_delivery.py @@ -59,14 +59,10 @@ PROVIDER_MAX_TEXT_LENGTH: Dict[str, int] = { # ElevenLabs caps vary by model_id. https://elevenlabs.io/docs/overview/models ELEVENLABS_MODEL_MAX_TEXT_LENGTH: Dict[str, int] = { - "eleven_v3": 5000, - "eleven_ttv_v3": 5000, - "eleven_multilingual_v2": 10000, - "eleven_multilingual_v1": 10000, - "eleven_english_sts_v2": 10000, - "eleven_english_sts_v1": 10000, - "eleven_flash_v2": 30000, - "eleven_flash_v2_5": 40000} + "eleven_v3": 5000, "eleven_ttv_v3": 5000, + "eleven_multilingual_v2": 10000, "eleven_multilingual_v1": 10000, + "eleven_english_sts_v2": 10000, "eleven_english_sts_v1": 10000, + "eleven_flash_v2": 30000, "eleven_flash_v2_5": 40000} def _positive_int(value: Any) -> Optional[int]: @@ -103,9 +99,7 @@ def _resolve_max_text_length(provider: Optional[str], tts_config: Optional[Dict[ # PCM output specs for Gemini TTS (fixed by the API): 24kHz mono 16-bit (L16). -GEMINI_TTS_SAMPLE_RATE = 24000 -GEMINI_TTS_CHANNELS = 1 -GEMINI_TTS_SAMPLE_WIDTH = 2 +GEMINI_TTS_SAMPLE_RATE, GEMINI_TTS_CHANNELS, GEMINI_TTS_SAMPLE_WIDTH = 24000, 1, 2 # ffmpeg args producing the Ogg/Opus voice-bubble encoding Telegram & co expect. _OPUS_VOICE_ARGS = [ diff --git a/tools/tts_tool_lifecycle.py b/tools/tts_tool_lifecycle.py index 4df1f1bb6a..7c30bb059c 100644 --- a/tools/tts_tool_lifecycle.py +++ b/tools/tts_tool_lifecycle.py @@ -153,10 +153,8 @@ def release_tts_lease(lease: str) -> Dict[str, Any]: with _tts_lease_lock: _tts_leases.discard(lease) holders = len(_tts_leases) - result: Dict[str, Any] = {"leases": holders, "released": 0} - if holders == 0: - result["released"] = release_tts_provider()["released"] - return result + released = release_tts_provider()["released"] if holders == 0 else 0 + return {"leases": holders, "released": released} def tts_lease_holders() -> List[str]: diff --git a/tools/tts_tool_providers.py b/tools/tts_tool_providers.py index d6d2a438c7..2e1dc2a8ec 100644 --- a/tools/tts_tool_providers.py +++ b/tools/tts_tool_providers.py @@ -225,8 +225,7 @@ def _generate_elevenlabs(text: str, output_path: str, tts_config: Dict[str, Any] model_id=el_config.get("model_id", DEFAULT_ELEVENLABS_MODEL_ID), output_format="opus_48000_64" if output_path.endswith(".ogg") else "mp3_44100_128") with open(output_path, "wb") as f: - for chunk in audio_generator: - f.write(chunk) + f.writelines(audio_generator) return output_path @@ -238,7 +237,7 @@ _XAI_WRAPPING_SPEECH_TAGS = ( "soft", "whisper", "loud", "build-intensity", "decrease-intensity", "higher-pitch", "lower-pitch", "slow", "fast", "sing-song", "singing", "laugh-speak", "emphasis") _XAI_SPEECH_TAG_RE = re.compile( - r"(\[(?:" + "|".join(_XAI_INLINE_SPEECH_TAGS) + r")\]|)", + rf"(\[(?:{'|'.join(_XAI_INLINE_SPEECH_TAGS)})\]|)", flags=re.IGNORECASE) _XAI_FIRST_SENTENCE_RE = re.compile(r"^(.{12,120}?[.!?…])\s+(?=\S)", flags=re.DOTALL)