1860 lines
72 KiB
Python
1860 lines
72 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Text-to-Speech Tool Module
|
|
|
|
Built-in TTS providers:
|
|
- Edge TTS (default, free, no API key): Microsoft Edge neural voices
|
|
- ElevenLabs (premium): High-quality voices, needs ELEVENLABS_API_KEY
|
|
- OpenAI TTS: Good quality, needs OPENAI_API_KEY
|
|
- MiniMax TTS: High-quality with voice cloning, needs the selected region's key
|
|
- Mistral (Voxtral TTS): Multilingual, native Opus, needs MISTRAL_API_KEY
|
|
- Google Gemini TTS: Controllable, 30 prebuilt voices, needs GEMINI_API_KEY
|
|
- xAI TTS: Grok voices, uses xAI Grok OAuth credentials or XAI_API_KEY
|
|
- NeuTTS (local, free, no API key): On-device TTS via neutts
|
|
- KittenTTS (local, free, no API key): On-device 25MB model
|
|
- Piper (local, free, no API key): OHF-Voice/piper1-gpl neural VITS, 44 languages
|
|
|
|
Custom command providers: any number of named ``type: command`` providers under
|
|
``tts.providers.<name>`` in ``~/.hermes/config.yaml``; Hermes writes the text to
|
|
a temp file and runs the shell template (see the Local Command section of
|
|
``website/docs/user-guide/features/tts.md``).
|
|
|
|
Output: Opus (.ogg) for voice-bubble platforms (Telegram etc.), MP3 elsewhere.
|
|
Configuration lives under the ``tts:`` key; the user chooses provider/voice,
|
|
the model just sends text.
|
|
|
|
Module layout: this file owns config resolution, the command/plugin provider
|
|
layers, the OpenAI/DeepInfra backends (managed-gateway aware), provider
|
|
dispatch, the lifecycle leases and the tool registration. Sibling modules:
|
|
``tts_tool_providers`` (cloud backends), ``tts_tool_local`` (on-device engines
|
|
+ model caches), ``tts_tool_delivery`` (chunking / ffmpeg / packing),
|
|
``tts_tool_speaker`` (streaming speaker pipeline). Their names are re-imported
|
|
here so ``tools.tts_tool.<name>`` keeps resolving.
|
|
|
|
Usage:
|
|
from tools.tts_tool import text_to_speech_tool, check_tts_requirements
|
|
|
|
result = text_to_speech_tool(text="Hello world")
|
|
"""
|
|
|
|
import asyncio
|
|
import datetime
|
|
import importlib.util
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import subprocess
|
|
import tempfile
|
|
import threading
|
|
import time
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Callable, Dict, Any, List, Optional
|
|
from urllib.parse import urljoin
|
|
|
|
from hermes_constants import display_hermes_home
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def get_env_value(name, default=None):
|
|
"""Read env values through the live config module.
|
|
|
|
Resolved at call time: tests monkeypatch/restore
|
|
``hermes_cli.config.get_env_value`` and must not leave TTS holding a stale
|
|
function for the rest of the process.
|
|
"""
|
|
try:
|
|
from hermes_cli.config import get_env_value as _get_env_value
|
|
except ImportError:
|
|
return os.getenv(name, default)
|
|
value = _get_env_value(name)
|
|
return default if value is None else value
|
|
|
|
|
|
def _resolve_provider_key(env_var: str, provider_id: str) -> str:
|
|
"""Resolve a TTS provider API key via the shared voice-key resolver.
|
|
|
|
``tools.tool_backend_helpers.resolve_provider_secret`` is the single owner
|
|
of STT/TTS key resolution (config > env/.env > credential pool). Resolved
|
|
at call time so tests that reload the helpers module see the live function.
|
|
"""
|
|
try:
|
|
from tools.tool_backend_helpers import resolve_provider_secret
|
|
except ImportError: # pragma: no cover — helpers are in-repo
|
|
return str(get_env_value(env_var) or "").strip()
|
|
return resolve_provider_secret(env_var, provider_id, env_getter=get_env_value)
|
|
|
|
|
|
from tools.managed_tool_gateway import resolve_managed_tool_gateway
|
|
from tools.tts_command_provider import (
|
|
command_env_passthrough as _command_provider_env_passthrough,
|
|
render_command_template as _render_command_tts_template,
|
|
run_command_provider as _run_command_tts,
|
|
shell_quote_context as _shell_quote_context, # noqa: F401 — tests import via this module
|
|
)
|
|
from tools.tool_backend_helpers import (
|
|
NOUS_MANAGED_PROVIDER,
|
|
managed_nous_tools_enabled,
|
|
nous_tool_gateway_unavailable_message,
|
|
read_selection,
|
|
resolve_openai_audio_api_key,
|
|
selection_error,
|
|
)
|
|
from tools.tts_tool_delivery import ( # noqa: F401 — historical names re-exported
|
|
FALLBACK_MAX_TEXT_LENGTH,
|
|
AudioDeliveryProfile,
|
|
_build_audio_delivery_files,
|
|
_concat_audio_files,
|
|
_convert_to_opus,
|
|
_has_ffmpeg,
|
|
_pack_audio_files_for_delivery,
|
|
_repair_ogg_container,
|
|
_resolve_audio_delivery_profile,
|
|
_sniff_audio_container,
|
|
_split_oversized_sentence,
|
|
_split_text_for_tts,
|
|
_wrap_pcm_as_wav,
|
|
)
|
|
from tools.tts_tool_providers import ( # noqa: F401 — historical names re-exported
|
|
DEFAULT_ELEVENLABS_MODEL_ID,
|
|
DEFAULT_ELEVENLABS_STREAMING_MODEL_ID,
|
|
DEFAULT_ELEVENLABS_VOICE_ID,
|
|
DEFAULT_GEMINI_TTS_BASE_URL,
|
|
DEFAULT_GEMINI_TTS_MODEL,
|
|
DEFAULT_GEMINI_TTS_VOICE,
|
|
DEFAULT_MINIMAX_BASE_URL,
|
|
DEFAULT_MINIMAX_CN_BASE_URL,
|
|
DEFAULT_XAI_BASE_URL,
|
|
DEFAULT_XAI_VOICE_ID,
|
|
TTS_RESPONSE_BODY_LIMIT_BYTES,
|
|
_XAI_FIRST_SENTENCE_RE,
|
|
_XAI_INLINE_SPEECH_TAGS,
|
|
_XAI_WRAPPING_SPEECH_TAGS,
|
|
_apply_xai_auto_speech_tags,
|
|
_elevenlabs_environment_kwargs,
|
|
_generate_edge_tts,
|
|
_generate_elevenlabs,
|
|
_generate_gemini_tts,
|
|
_generate_minimax_tts,
|
|
_generate_mistral_tts,
|
|
_generate_xai_tts,
|
|
_read_tts_response_bytes,
|
|
_resolve_minimax_tts_runtime,
|
|
_tts_response_format_from_path,
|
|
)
|
|
from tools.tts_tool_local import ( # noqa: F401 — historical names re-exported
|
|
DEFAULT_PIPER_VOICE,
|
|
_LOCAL_TTS_MODEL_CACHES,
|
|
_TTS_MODEL_CACHE_MAX,
|
|
_generate_kittentts,
|
|
_generate_neutts,
|
|
_generate_piper_tts,
|
|
_kittentts_model_cache,
|
|
_load_kittentts_model_for_config,
|
|
_load_piper_voice_for_config,
|
|
_piper_voice_cache,
|
|
_resolve_piper_voice_path,
|
|
_tts_cache_get_or_load,
|
|
)
|
|
from tools.tts_tool_speaker import ( # noqa: F401 — historical names re-exported
|
|
stream_tts_to_speaker,
|
|
)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Lazy imports -- providers are imported only when actually used to avoid
|
|
# crashing in headless environments (SSH, Docker, WSL, no PortAudio).
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _lazy_ensure(feature: str) -> None:
|
|
"""Best-effort ``tools.lazy_deps.ensure`` so an SDK installs on first use.
|
|
|
|
Users who enabled a provider by editing config.yaml never ran the
|
|
post-setup hook. Any failure (lazy_deps missing, install refused) falls
|
|
through so the raw import below still raises a clean ImportError.
|
|
"""
|
|
try:
|
|
from tools.lazy_deps import ensure
|
|
ensure(feature, prompt=False)
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _import_edge_tts():
|
|
"""Lazy import edge_tts. Returns the module or raises ImportError."""
|
|
_lazy_ensure("tts.edge")
|
|
import edge_tts
|
|
return edge_tts
|
|
|
|
|
|
def _import_elevenlabs():
|
|
"""Lazy import the ElevenLabs client class or raise ImportError."""
|
|
_lazy_ensure("tts.elevenlabs")
|
|
from elevenlabs.client import ElevenLabs
|
|
return ElevenLabs
|
|
|
|
|
|
def _import_openai_client():
|
|
from openai import OpenAI as OpenAIClient
|
|
return OpenAIClient
|
|
|
|
|
|
def _import_mistral_client():
|
|
"""Lazy import the Mistral client class or raise ImportError."""
|
|
_lazy_ensure("tts.mistral")
|
|
from mistralai.client import Mistral
|
|
return Mistral
|
|
|
|
|
|
def _import_sounddevice():
|
|
"""Raises ImportError/OSError when PortAudio is unavailable."""
|
|
import sounddevice as sd
|
|
return sd
|
|
|
|
|
|
def _import_kittentts():
|
|
from kittentts import KittenTTS
|
|
return KittenTTS
|
|
|
|
|
|
def _import_piper():
|
|
"""``pip install piper-tts`` ships cross-platform wheels with embedded espeak-ng."""
|
|
from piper import PiperVoice
|
|
return PiperVoice
|
|
|
|
|
|
def _package_installed(name: str) -> bool:
|
|
try:
|
|
return importlib.util.find_spec(name) is not None
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _check_neutts_available() -> bool:
|
|
return _package_installed("neutts")
|
|
|
|
|
|
def _check_kittentts_available() -> bool:
|
|
return _package_installed("kittentts")
|
|
|
|
|
|
def _check_piper_available() -> bool:
|
|
return _package_installed("piper")
|
|
|
|
|
|
# ===========================================================================
|
|
# Defaults
|
|
# ===========================================================================
|
|
DEFAULT_PROVIDER = "edge"
|
|
DEFAULT_OPENAI_MODEL = "gpt-4o-mini-tts"
|
|
# The managed OpenAI audio gateway (Nous portal proxy) only proxies these
|
|
# speech models; anything else is 400 "Unsupported managed OpenAI speech model".
|
|
MANAGED_OPENAI_TTS_MODELS = frozenset({"gpt-4o-mini-tts"})
|
|
DEFAULT_OPENAI_VOICE = "alloy"
|
|
DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1"
|
|
# DeepInfra base URL is resolved via hermes_cli.models.deepinfra_base_url (shared).
|
|
DEFAULT_DEEPINFRA_TTS_VOICE = "default"
|
|
|
|
|
|
def _get_default_output_dir() -> str:
|
|
from hermes_constants import get_hermes_dir
|
|
return str(get_hermes_dir("cache/audio", "audio_cache"))
|
|
|
|
|
|
DEFAULT_OUTPUT_DIR = _get_default_output_dir()
|
|
_DEFAULT_OUTPUT_DIR_AT_IMPORT = DEFAULT_OUTPUT_DIR
|
|
|
|
|
|
def _default_output_dir() -> str:
|
|
"""Return the active profile's audio output dir at call time.
|
|
|
|
Long-lived multi-profile runtimes (dashboard, TUI/Desktop backend, cron)
|
|
import this module once and later switch profiles via
|
|
``set_hermes_home_override()``; a frozen constant would keep writing into
|
|
the launch profile's cache. ``DEFAULT_OUTPUT_DIR`` stays as a module
|
|
attribute for tests/patchers and wins whenever it has been patched.
|
|
"""
|
|
configured = DEFAULT_OUTPUT_DIR
|
|
if configured != _DEFAULT_OUTPUT_DIR_AT_IMPORT:
|
|
return configured
|
|
return _get_default_output_dir()
|
|
|
|
|
|
# Per-provider input-character caps (from official provider docs); override
|
|
# via ``tts.<provider>.max_text_length``.
|
|
PROVIDER_MAX_TEXT_LENGTH: Dict[str, int] = {
|
|
"edge": 5000, # edge-tts practical sync limit
|
|
"openai": 4096, # https://platform.openai.com/docs/guides/text-to-speech
|
|
"xai": 15000, # https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
"minimax": 10000, # https://platform.minimax.io/docs/api-reference/speech-t2a-http (sync)
|
|
"mistral": 4000, # conservative; no published per-request cap
|
|
"gemini": 32000, # 32k-token context window; char cap is conservative
|
|
"elevenlabs": 10000, # fallback when model-aware lookup can't resolve (multilingual_v2)
|
|
"neutts": 2000, # local model, quality falls off on long text
|
|
"kittentts": 2000, # local 25MB model
|
|
"piper": 5000, # local VITS model, phoneme-based; practical cap
|
|
}
|
|
|
|
# ElevenLabs caps vary by model_id. https://elevenlabs.io/docs/overview/models
|
|
ELEVENLABS_MODEL_MAX_TEXT_LENGTH: Dict[str, int] = {
|
|
"eleven_v3": 5000,
|
|
"eleven_ttv_v3": 5000,
|
|
"eleven_multilingual_v2": 10000,
|
|
"eleven_multilingual_v1": 10000,
|
|
"eleven_english_sts_v2": 10000,
|
|
"eleven_english_sts_v1": 10000,
|
|
"eleven_flash_v2": 30000,
|
|
"eleven_flash_v2_5": 40000,
|
|
}
|
|
|
|
# Back-compat alias. Prefer ``_resolve_max_text_length()`` for new code.
|
|
MAX_TEXT_LENGTH = FALLBACK_MAX_TEXT_LENGTH
|
|
|
|
|
|
def _positive_int_override(value: Any) -> Optional[int]:
|
|
"""A user ``max_text_length`` override, or None when absent/bool/non-positive."""
|
|
if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
|
|
return None
|
|
return value
|
|
|
|
|
|
def _resolve_max_text_length(
|
|
provider: Optional[str],
|
|
tts_config: Optional[Dict[str, Any]] = None,
|
|
) -> int:
|
|
"""Return the input-character cap for *provider*.
|
|
|
|
Order: ``tts.<provider>.max_text_length`` > ElevenLabs model table >
|
|
``PROVIDER_MAX_TEXT_LENGTH`` > command provider's own ``max_text_length``
|
|
(else ``DEFAULT_COMMAND_TTS_MAX_TEXT_LENGTH``) > ``FALLBACK_MAX_TEXT_LENGTH``.
|
|
Non-positive / non-int overrides fall through so a broken config can't
|
|
disable truncation.
|
|
"""
|
|
if not provider:
|
|
return FALLBACK_MAX_TEXT_LENGTH
|
|
key = provider.lower().strip()
|
|
cfg = tts_config or {}
|
|
|
|
prov_cfg = cfg.get(key) if isinstance(cfg.get(key), dict) else {}
|
|
override = _positive_int_override(prov_cfg.get("max_text_length") if prov_cfg else None)
|
|
if override:
|
|
return override
|
|
|
|
if key == "elevenlabs":
|
|
model_id = (prov_cfg or {}).get("model_id") or DEFAULT_ELEVENLABS_MODEL_ID
|
|
mapped = ELEVENLABS_MODEL_MAX_TEXT_LENGTH.get(str(model_id).strip())
|
|
if mapped:
|
|
return mapped
|
|
|
|
if key in PROVIDER_MAX_TEXT_LENGTH:
|
|
return PROVIDER_MAX_TEXT_LENGTH[key]
|
|
|
|
if key not in BUILTIN_TTS_PROVIDERS:
|
|
named = _get_named_provider_config(cfg, key)
|
|
if _is_command_provider_config(named):
|
|
return _positive_int_override(named.get("max_text_length")) or DEFAULT_COMMAND_TTS_MAX_TEXT_LENGTH
|
|
|
|
return FALLBACK_MAX_TEXT_LENGTH
|
|
|
|
|
|
# ===========================================================================
|
|
# Config loader -- reads tts: section from ~/.hermes/config.yaml
|
|
# ===========================================================================
|
|
def _load_tts_config() -> Dict[str, Any]:
|
|
"""Return the ``tts`` config section ({} when unavailable)."""
|
|
try:
|
|
from hermes_cli.config import load_config
|
|
config = load_config()
|
|
return config.get("tts") or {}
|
|
except ImportError:
|
|
logger.debug("hermes_cli.config not available, using default TTS config")
|
|
return {}
|
|
except Exception as e:
|
|
logger.warning("Failed to load TTS config: %s", e, exc_info=True)
|
|
return {}
|
|
|
|
|
|
def _get_provider(tts_config: Dict[str, Any]) -> str:
|
|
"""The explicitly configured TTS provider, or the free default.
|
|
|
|
Inference credentials do not imply consent to paid speech generation:
|
|
cloud TTS is opt-in via ``tts.provider``. The managed selection
|
|
(``tts.provider: nous``) is serviced by the OpenAI implementation, routed
|
|
through the managed openai-audio gateway by
|
|
``_resolve_openai_audio_client_config``.
|
|
"""
|
|
provider = (tts_config.get("provider") or DEFAULT_PROVIDER).lower().strip()
|
|
if provider == NOUS_MANAGED_PROVIDER:
|
|
return "openai"
|
|
return provider
|
|
|
|
|
|
# ===========================================================================
|
|
# Custom command providers (type: command under tts.providers.<name>)
|
|
# ===========================================================================
|
|
#
|
|
# Config shape::
|
|
#
|
|
# tts:
|
|
# provider: piper-en
|
|
# providers:
|
|
# piper-en:
|
|
# type: command
|
|
# command: "piper -m ~/model.onnx -f {output_path} < {input_path}"
|
|
# output_format: wav
|
|
#
|
|
# Placeholders: ``{input_path}``, ``{text_path}`` (alias), ``{output_path}``,
|
|
# ``{format}``, ``{voice}``, ``{model}``, ``{speed}``; ``{{``/``}}`` for literal
|
|
# braces. Values are shell-quoted for their surrounding quote context. Built-in
|
|
# provider names always win over a same-named entry under ``tts.providers``.
|
|
|
|
# Any ``tts.provider`` value NOT in this set refers to ``tts.providers.<name>``.
|
|
BUILTIN_TTS_PROVIDERS = frozenset({
|
|
"edge",
|
|
"elevenlabs",
|
|
"openai",
|
|
"minimax",
|
|
"xai",
|
|
"mistral",
|
|
"gemini",
|
|
"neutts",
|
|
"kittentts",
|
|
"piper",
|
|
"deepinfra",
|
|
})
|
|
|
|
DEFAULT_COMMAND_TTS_TIMEOUT_SECONDS = 120
|
|
DEFAULT_COMMAND_TTS_OUTPUT_FORMAT = "mp3"
|
|
COMMAND_TTS_OUTPUT_FORMATS = frozenset(
|
|
{"mp3", "wav", "ogg", "flac", "m4a", "aac", "amr", "opus"}
|
|
)
|
|
DEFAULT_COMMAND_TTS_MAX_TEXT_LENGTH = 5000
|
|
|
|
# Platforms whose native voice-bubble delivery requires Ogg/Opus audio
|
|
# (MP3 renders as a broken attachment there).
|
|
OPUS_VOICE_PLATFORMS = frozenset({
|
|
"telegram",
|
|
"matrix",
|
|
"feishu",
|
|
"whatsapp",
|
|
"signal",
|
|
})
|
|
|
|
# Built-ins that emit Opus natively when asked for .ogg (no ffmpeg needed).
|
|
_NATIVE_OPUS_PROVIDERS = frozenset({"openai", "elevenlabs", "mistral", "gemini"})
|
|
# Built-ins whose native output (MP3/WAV) needs ffmpeg for voice-bubble delivery.
|
|
_FFMPEG_OPUS_PROVIDERS = frozenset({"edge", "neutts", "minimax", "xai", "kittentts", "piper"})
|
|
|
|
|
|
def _get_provider_section(tts_config: Dict[str, Any], name: str) -> Dict[str, Any]:
|
|
"""Return a provider config block if it's a dict, else an empty dict."""
|
|
if not isinstance(tts_config, dict):
|
|
return {}
|
|
section = tts_config.get(name)
|
|
return section if isinstance(section, dict) else {}
|
|
|
|
|
|
def _get_named_provider_config(
|
|
tts_config: Dict[str, Any],
|
|
name: str,
|
|
) -> Dict[str, Any]:
|
|
"""Config dict for a user-declared provider, or {}.
|
|
|
|
``tts.providers.<name>`` is canonical; ``tts.<name>`` is accepted as
|
|
back-compat only for non-built-in names (so a user's ``tts.openai`` block
|
|
still means the OpenAI provider, not a custom command).
|
|
"""
|
|
providers = _get_provider_section(tts_config, "providers")
|
|
section = providers.get(name) if isinstance(providers, dict) else None
|
|
if isinstance(section, dict):
|
|
return section
|
|
if name.lower() not in BUILTIN_TTS_PROVIDERS:
|
|
legacy = _get_provider_section(tts_config, name)
|
|
if legacy:
|
|
return legacy
|
|
return {}
|
|
|
|
|
|
def _is_command_provider_config(config: Dict[str, Any]) -> bool:
|
|
"""True when *config* declares a command-type provider (has a non-empty ``command``)."""
|
|
if not isinstance(config, dict):
|
|
return False
|
|
ptype = str(config.get("type") or "").strip().lower()
|
|
if ptype and ptype != "command":
|
|
return False
|
|
command = config.get("command")
|
|
return isinstance(command, str) and bool(command.strip())
|
|
|
|
|
|
def _resolve_command_provider_config(
|
|
provider: str,
|
|
tts_config: Dict[str, Any],
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""The provider config when *provider* is a user-declared command provider.
|
|
|
|
None for built-in names (native handlers win), unknown names, or
|
|
non-command types.
|
|
"""
|
|
if not provider:
|
|
return None
|
|
key = provider.lower().strip()
|
|
if key in BUILTIN_TTS_PROVIDERS:
|
|
return None
|
|
config = _get_named_provider_config(tts_config, key)
|
|
if _is_command_provider_config(config):
|
|
return config
|
|
return None
|
|
|
|
|
|
def _dispatch_to_plugin_provider(
|
|
text: str,
|
|
output_path: str,
|
|
provider: str,
|
|
tts_config: Dict[str, Any],
|
|
) -> Optional[str]:
|
|
"""Route to a plugin-registered TTS provider; None means "fall through".
|
|
|
|
Invariants enforced here even though the caller checks them too, so a
|
|
caller refactor can't silently break them:
|
|
|
|
1. Built-in names never reach the plugin registry.
|
|
2. A same-named ``type: command`` provider wins over a plugin.
|
|
3. Dispatch fires only for a registered :class:`TTSProvider` whose name
|
|
equals the configured value; unknown names return None.
|
|
|
|
Plugin exceptions propagate — the outer ``text_to_speech_tool`` converts
|
|
them to the standard error envelope.
|
|
"""
|
|
if not provider:
|
|
return None
|
|
key = provider.lower().strip()
|
|
if key in BUILTIN_TTS_PROVIDERS:
|
|
return None
|
|
if _is_command_provider_config(_get_named_provider_config(tts_config, key)):
|
|
return None
|
|
try:
|
|
from agent.tts_registry import get_provider
|
|
from hermes_cli.plugins import _ensure_plugins_discovered
|
|
|
|
_ensure_plugins_discovered()
|
|
plugin_provider = get_provider(key)
|
|
if plugin_provider is None:
|
|
# Long-lived sessions may have discovered plugins before this one
|
|
# was installed/enabled; retry once with a forced refresh.
|
|
_ensure_plugins_discovered(force=True)
|
|
plugin_provider = get_provider(key)
|
|
except Exception as exc: # noqa: BLE001 — discovery failure is non-fatal
|
|
logger.debug("tts plugin dispatch skipped (discovery failed): %s", exc)
|
|
return None
|
|
if plugin_provider is None:
|
|
return None
|
|
|
|
# voice/model/speed/format are optional per the TTSProvider.synthesize
|
|
# contract; providers fall back to their own defaults on None.
|
|
cfg = tts_config if isinstance(tts_config, dict) else {}
|
|
voice = cfg.get("voice")
|
|
model = cfg.get("model")
|
|
speed = cfg.get("speed")
|
|
fmt = cfg.get("output_format", DEFAULT_COMMAND_TTS_OUTPUT_FORMAT)
|
|
|
|
logger.info("Generating speech with plugin TTS provider '%s'...", key)
|
|
written = plugin_provider.synthesize(
|
|
text,
|
|
output_path,
|
|
voice=voice if isinstance(voice, str) and voice else None,
|
|
model=model if isinstance(model, str) and model else None,
|
|
speed=float(speed) if isinstance(speed, (int, float)) else None,
|
|
format=str(fmt).lower() if fmt else "mp3",
|
|
)
|
|
# Contract: returns the (possibly rewritten) output path; tolerate None.
|
|
return written if isinstance(written, str) and written else output_path
|
|
|
|
|
|
def _plugin_provider_is_voice_compatible(provider: str) -> bool:
|
|
"""True when the registered plugin provider opts into voice-bubble delivery.
|
|
|
|
Any registry/property failure means False (safe default, like command providers).
|
|
"""
|
|
if not provider:
|
|
return False
|
|
key = provider.lower().strip()
|
|
if key in BUILTIN_TTS_PROVIDERS:
|
|
return False
|
|
try:
|
|
from agent.tts_registry import get_provider
|
|
|
|
plugin_provider = get_provider(key)
|
|
if plugin_provider is None:
|
|
return False
|
|
return bool(plugin_provider.voice_compatible)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.debug("tts plugin voice_compatible check failed for '%s': %s", key, exc)
|
|
return False
|
|
|
|
|
|
def _iter_command_providers(tts_config: Dict[str, Any]):
|
|
"""Yield (name, config) pairs for every declared command-type provider."""
|
|
if not isinstance(tts_config, dict):
|
|
return
|
|
providers = _get_provider_section(tts_config, "providers")
|
|
for name, cfg in (providers or {}).items():
|
|
if (
|
|
isinstance(name, str)
|
|
and name.lower() not in BUILTIN_TTS_PROVIDERS
|
|
and _is_command_provider_config(cfg)
|
|
):
|
|
yield name, cfg
|
|
|
|
|
|
def _get_command_tts_timeout(config: Dict[str, Any]) -> float:
|
|
"""Timeout in seconds; invalid or non-positive values fall back to the default."""
|
|
raw = config.get("timeout", config.get("timeout_seconds", DEFAULT_COMMAND_TTS_TIMEOUT_SECONDS))
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError):
|
|
return float(DEFAULT_COMMAND_TTS_TIMEOUT_SECONDS)
|
|
if value <= 0:
|
|
return float(DEFAULT_COMMAND_TTS_TIMEOUT_SECONDS)
|
|
return value
|
|
|
|
|
|
def _get_command_tts_output_format(
|
|
config: Dict[str, Any],
|
|
output_path: Optional[str] = None,
|
|
) -> str:
|
|
"""Validated output format: the output path's suffix wins, then ``format``/``output_format``."""
|
|
if output_path:
|
|
suffix = Path(output_path).suffix.lower().strip().lstrip(".")
|
|
if suffix in COMMAND_TTS_OUTPUT_FORMATS:
|
|
return suffix
|
|
raw = (
|
|
config.get("format")
|
|
or config.get("output_format")
|
|
or DEFAULT_COMMAND_TTS_OUTPUT_FORMAT
|
|
)
|
|
fmt = str(raw).lower().strip().lstrip(".")
|
|
return fmt if fmt in COMMAND_TTS_OUTPUT_FORMATS else DEFAULT_COMMAND_TTS_OUTPUT_FORMAT
|
|
|
|
|
|
def _is_command_tts_voice_compatible(config: Dict[str, Any]) -> bool:
|
|
"""True only when the user explicitly opted in to voice delivery."""
|
|
value = config.get("voice_compatible", False)
|
|
if isinstance(value, str):
|
|
return value.strip().lower() in {"1", "true", "yes", "on"}
|
|
return bool(value)
|
|
|
|
|
|
def _configured_command_tts_output_path(path: Path, config: Dict[str, Any]) -> Path:
|
|
"""Return an output path whose extension matches the provider's output_format."""
|
|
fmt = _get_command_tts_output_format(config)
|
|
return path.with_suffix(f".{fmt}")
|
|
|
|
|
|
def _generate_command_tts(
|
|
text: str,
|
|
output_path: str,
|
|
provider_name: str,
|
|
config: Dict[str, Any],
|
|
tts_config: Dict[str, Any],
|
|
) -> str:
|
|
"""Generate speech by running a user-configured shell command.
|
|
|
|
Returns the absolute path of the audio file the command wrote. Raises
|
|
``ValueError`` for invalid provider config and ``RuntimeError`` for
|
|
timeouts / non-zero exits / empty output.
|
|
"""
|
|
command_template = str(config.get("command") or "").strip()
|
|
if not command_template:
|
|
raise ValueError(
|
|
f"tts.providers.{provider_name}.command is not configured"
|
|
)
|
|
|
|
output = Path(output_path).expanduser()
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
if output.exists():
|
|
output.unlink()
|
|
|
|
timeout = _get_command_tts_timeout(config)
|
|
output_format = _get_command_tts_output_format(config, str(output))
|
|
speed = config.get("speed", tts_config.get("speed", ""))
|
|
|
|
with tempfile.TemporaryDirectory() as tmpdir:
|
|
text_path = Path(tmpdir) / "input.txt"
|
|
text_path.write_text(text, encoding="utf-8")
|
|
|
|
placeholders = {
|
|
"input_path": str(text_path),
|
|
"text_path": str(text_path),
|
|
"output_path": str(output),
|
|
"format": output_format,
|
|
"voice": str(config.get("voice", "")),
|
|
"model": str(config.get("model", "")),
|
|
"speed": str(speed),
|
|
}
|
|
command = _render_command_tts_template(command_template, placeholders)
|
|
|
|
try:
|
|
_run_command_tts(
|
|
command,
|
|
timeout,
|
|
env_passthrough=_command_provider_env_passthrough(config),
|
|
)
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise RuntimeError(
|
|
f"TTS provider '{provider_name}' timed out after {timeout:g}s"
|
|
) from exc
|
|
except subprocess.CalledProcessError as exc:
|
|
detail_parts = []
|
|
if exc.stderr:
|
|
detail_parts.append(f"stderr: {exc.stderr.strip()}")
|
|
if exc.stdout:
|
|
detail_parts.append(f"stdout: {exc.stdout.strip()}")
|
|
detail = "; ".join(detail_parts) or "no command output"
|
|
raise RuntimeError(
|
|
f"TTS provider '{provider_name}' exited with code "
|
|
f"{exc.returncode}: {detail}"
|
|
) from exc
|
|
|
|
if not output.exists() or output.stat().st_size <= 0:
|
|
raise RuntimeError(
|
|
f"TTS provider '{provider_name}' produced no output at {output}"
|
|
)
|
|
return str(output)
|
|
|
|
|
|
def _has_any_command_tts_provider(tts_config: Optional[Dict[str, Any]] = None) -> bool:
|
|
"""Return True when any command-type TTS provider is configured."""
|
|
if tts_config is None:
|
|
tts_config = _load_tts_config()
|
|
for _name, _cfg in _iter_command_providers(tts_config):
|
|
return True
|
|
return False
|
|
|
|
|
|
# ===========================================================================
|
|
# Provider: OpenAI TTS (also every OpenAI-compatible endpoint — DeepInfra
|
|
# delegates here). Kept in the origin module: it shares the managed-gateway
|
|
# selection logic below.
|
|
# ===========================================================================
|
|
def _generate_openai_tts(
|
|
text: str,
|
|
output_path: str,
|
|
tts_config: Dict[str, Any],
|
|
*,
|
|
api_key: Optional[str] = None,
|
|
base_url: Optional[str] = None,
|
|
model: Optional[str] = None,
|
|
voice: Optional[str] = None,
|
|
speed: Optional[float] = None,
|
|
instructions: Optional[str] = None,
|
|
) -> str:
|
|
"""Generate audio via the OpenAI ``audio.speech.create`` SDK shape.
|
|
|
|
Explicit kwargs let OpenAI-compatible backends (DeepInfra) pass their own
|
|
credentials/model/voice and skip ``_resolve_openai_audio_client_config``
|
|
(the managed-gateway path). When None: ``api_key`` comes from the OpenAI
|
|
auth chain, ``base_url`` from ``tts.openai.base_url`` then the auth-chain
|
|
fallback then the OpenAI default, model/voice/speed from ``tts.openai``
|
|
(speed falling back to global ``tts.speed``). ``instructions`` is
|
|
forwarded only when truthy so ``tts-1`` and strict OpenAI-compatible
|
|
servers that reject unknown kwargs are unaffected.
|
|
"""
|
|
fallback_base: Optional[str] = None
|
|
is_managed = False
|
|
explicit_base_url = base_url is not None
|
|
if api_key is None:
|
|
api_key, fallback_base, is_managed = _resolve_openai_audio_client_config()
|
|
|
|
# ``tts.openai: null`` in YAML yields None — coalesce so .get() is safe.
|
|
oai_config = (tts_config.get("openai") if isinstance(tts_config, dict) else None) or {}
|
|
if model is None:
|
|
model = oai_config.get("model", DEFAULT_OPENAI_MODEL)
|
|
if voice is None:
|
|
voice = oai_config.get("voice", DEFAULT_OPENAI_VOICE)
|
|
config_base_url = oai_config.get("base_url")
|
|
if base_url is None:
|
|
# Config override beats the auth-chain fallback; an explicit arg
|
|
# (DeepInfra) skipped this block and always wins.
|
|
base_url = config_base_url or fallback_base or DEFAULT_OPENAI_BASE_URL
|
|
if speed is None:
|
|
speed_default = tts_config.get("speed", 1.0) if isinstance(tts_config, dict) else 1.0
|
|
speed = float(oai_config.get("speed", speed_default))
|
|
language = oai_config.get("language")
|
|
|
|
# The managed gateway only proxies MANAGED_OPENAI_TTS_MODELS; coerce a
|
|
# direct-OpenAI model (e.g. "tts-1-hd") unless the user redirected
|
|
# base_url to their own endpoint.
|
|
if (
|
|
is_managed
|
|
and not explicit_base_url
|
|
and not config_base_url
|
|
and model not in MANAGED_OPENAI_TTS_MODELS
|
|
):
|
|
logger.warning(
|
|
"TTS: managed OpenAI audio gateway does not support model %r; "
|
|
"falling back to %s. Set VOICE_TOOLS_OPENAI_KEY or OPENAI_API_KEY "
|
|
"to use %r directly.",
|
|
model, DEFAULT_OPENAI_MODEL, model,
|
|
)
|
|
model = DEFAULT_OPENAI_MODEL
|
|
|
|
response_format = _tts_response_format_from_path(output_path)
|
|
|
|
OpenAIClient = _import_openai_client()
|
|
client = OpenAIClient(api_key=api_key, base_url=base_url)
|
|
try:
|
|
create_kwargs: Dict[str, Any] = {
|
|
"model": model,
|
|
"voice": voice,
|
|
"input": text,
|
|
"response_format": response_format,
|
|
"extra_headers": {"x-idempotency-key": str(uuid.uuid4())},
|
|
}
|
|
if speed != 1.0:
|
|
create_kwargs["speed"] = max(0.25, min(4.0, speed))
|
|
if instructions:
|
|
create_kwargs["instructions"] = instructions
|
|
if language:
|
|
create_kwargs["extra_body"] = {"lang_code": language}
|
|
response = client.audio.speech.create(**create_kwargs)
|
|
|
|
response.stream_to_file(output_path)
|
|
return output_path
|
|
finally:
|
|
close = getattr(client, "close", None)
|
|
if callable(close):
|
|
close()
|
|
|
|
|
|
def _generate_deepinfra_tts(text: str, output_path: str, tts_config: Dict[str, Any]) -> str:
|
|
"""Resolve DeepInfra credentials/model, then delegate to the OpenAI handler.
|
|
|
|
DeepInfra's audio endpoint is OpenAI-compatible. Model ids come live from
|
|
the shared ``hermes_cli.models`` catalog helpers (no hardcoded ids, so
|
|
retired models disappear without a patch).
|
|
"""
|
|
api_key = _resolve_provider_key("DEEPINFRA_API_KEY", "deepinfra")
|
|
if not api_key:
|
|
raise ValueError(
|
|
"DEEPINFRA_API_KEY not set. Run `hermes setup` to configure, "
|
|
"or set the env var directly."
|
|
)
|
|
|
|
# ``tts.deepinfra: null`` yields None (no DEFAULT_CONFIG block to merge over).
|
|
di_config = tts_config.get("deepinfra") if isinstance(tts_config, dict) else None
|
|
if not isinstance(di_config, dict):
|
|
di_config = {}
|
|
|
|
from hermes_cli.models import deepinfra_base_url, deepinfra_model_ids
|
|
|
|
model = di_config.get("model")
|
|
if not isinstance(model, str) or not model.strip():
|
|
candidates = deepinfra_model_ids("tts")
|
|
if not candidates:
|
|
raise ValueError(
|
|
"No DeepInfra TTS model available. Pin one in config.yaml "
|
|
"under tts.deepinfra.model, or check connectivity to "
|
|
"api.deepinfra.com so the live catalog can be fetched."
|
|
)
|
|
model = candidates[0]
|
|
return _generate_openai_tts(
|
|
text,
|
|
output_path,
|
|
tts_config,
|
|
api_key=api_key,
|
|
base_url=deepinfra_base_url(di_config),
|
|
model=model,
|
|
voice=di_config.get("voice", DEFAULT_DEEPINFRA_TTS_VOICE),
|
|
speed=float(di_config.get("speed", tts_config.get("speed", 1.0))),
|
|
)
|
|
|
|
|
|
# ===========================================================================
|
|
# Local-engine lifecycle: warm-up / release driven by TTS-output toggles
|
|
# ===========================================================================
|
|
#
|
|
# Local engines load their model lazily on first synthesis, so the first spoken
|
|
# reply after a user turns speech output on pays the whole load as dead air,
|
|
# and the model then stays resident forever. The toggles ARE the intent
|
|
# signal: every surface that flips speech output on holds a *lease* here
|
|
# (warming the configured engine); when the last lease is released the local
|
|
# model caches are dropped. Lease-counting keeps one surface's "off" from
|
|
# unloading a model another surface in this process still needs. Cloud
|
|
# providers have nothing resident; warming them only ensures the lazily
|
|
# installed SDK is importable.
|
|
|
|
def _local_tts_warmers() -> Dict[str, Callable[[Dict[str, Any]], Any]]:
|
|
"""Provider name → loader populating that engine's cache slot (same key synthesis uses)."""
|
|
return {
|
|
"piper": lambda cfg: _load_piper_voice_for_config(cfg)[0],
|
|
"kittentts": lambda cfg: _load_kittentts_model_for_config(cfg)[0],
|
|
}
|
|
|
|
|
|
def _lazy_sdk_feature_for_provider(provider: str) -> Optional[str]:
|
|
"""tools.lazy_deps feature key for providers whose SDK installs on first use."""
|
|
return {
|
|
"edge": "tts.edge",
|
|
"elevenlabs": "tts.elevenlabs",
|
|
"mistral": "tts.mistral",
|
|
}.get(provider)
|
|
|
|
|
|
_tts_lease_lock = threading.Lock()
|
|
_tts_leases: set = set()
|
|
|
|
|
|
def _signal_user_tts_provider(name: str, tts_config: Dict[str, Any], hook: str) -> Optional[str]:
|
|
"""Forward a lease ``hook`` (``"warm"`` / ``"release"``) to a user-declared provider.
|
|
|
|
Command providers run their optional ``warm_command`` / ``release_command``
|
|
(same template/env/timeout rules as ``command``; output discarded) on a
|
|
background thread so a toggle never waits on a model server. Plugin
|
|
providers get :meth:`TTSProvider.warm` / :meth:`TTSProvider.release`.
|
|
Best-effort: failures are logged at debug. Returns the action taken.
|
|
"""
|
|
if not name or name in BUILTIN_TTS_PROVIDERS:
|
|
return None
|
|
cfg = _get_named_provider_config(tts_config, name)
|
|
try:
|
|
if _is_command_provider_config(cfg):
|
|
template = str(cfg.get(f"{hook}_command") or "").strip()
|
|
if not template:
|
|
return None
|
|
command = _render_command_tts_template(template, {
|
|
"voice": str(cfg.get("voice", "")),
|
|
"model": str(cfg.get("model", "")),
|
|
"speed": str(cfg.get("speed", tts_config.get("speed", ""))),
|
|
})
|
|
|
|
def _run() -> None:
|
|
try:
|
|
_run_command_tts(command, _get_command_tts_timeout(cfg),
|
|
env_passthrough=_command_provider_env_passthrough(cfg))
|
|
except Exception as exc: # noqa: BLE001 — best-effort hook
|
|
logger.debug("[TTS] %s_command for %s failed: %s", hook, name, exc)
|
|
|
|
threading.Thread(target=_run, name=f"tts-{hook}-{name}", daemon=True).start()
|
|
return hook
|
|
from agent.tts_registry import get_provider
|
|
from hermes_cli.plugins import _ensure_plugins_discovered
|
|
|
|
_ensure_plugins_discovered()
|
|
plugin_provider = get_provider(name)
|
|
if plugin_provider is None:
|
|
return None
|
|
getattr(plugin_provider, hook)()
|
|
return hook
|
|
except Exception as exc: # noqa: BLE001 — best-effort hook
|
|
logger.debug("[TTS] %s hook for %s failed: %s", hook, name, exc)
|
|
return "error"
|
|
|
|
|
|
def warm_tts_provider(
|
|
tts_config: Optional[Dict[str, Any]] = None,
|
|
provider: Optional[str] = None,
|
|
) -> Dict[str, Any]:
|
|
"""Pre-load the configured TTS provider so the next synthesis starts hot.
|
|
|
|
Local engines load their voice/model into the same LRU slot synthesis
|
|
reads (including first-use download); lazily-installed cloud SDKs are
|
|
made importable; user-declared providers get ``warm_command`` /
|
|
:meth:`TTSProvider.warm`; everything else is ``action: "noop"``.
|
|
Never raises — the result dict carries ``warmed`` / ``action`` /
|
|
``error``. Blocking; UI threads should run it in the background.
|
|
"""
|
|
if tts_config is None:
|
|
tts_config = _load_tts_config()
|
|
name = (provider or _get_provider(tts_config) or "").lower().strip()
|
|
result: Dict[str, Any] = {"provider": name, "warmed": False, "action": "noop"}
|
|
|
|
warmer = _local_tts_warmers().get(name)
|
|
if warmer is not None:
|
|
cache = _LOCAL_TTS_MODEL_CACHES.get(name)
|
|
before = len(cache) if cache is not None else 0
|
|
started = time.monotonic()
|
|
try:
|
|
warmer(tts_config)
|
|
except Exception as exc: # engine missing, download failed, bad voice…
|
|
logger.warning("[TTS] warm-up for %s failed: %s", name, exc)
|
|
result.update(action="error", error=str(exc))
|
|
return result
|
|
after = len(cache) if cache is not None else 0
|
|
result.update(
|
|
warmed=True,
|
|
action="loaded" if after > before else "cached",
|
|
elapsed_ms=int((time.monotonic() - started) * 1000),
|
|
)
|
|
logger.info("[TTS] warm-up %s: %s in %dms", name, result["action"], result["elapsed_ms"])
|
|
return result
|
|
|
|
signalled = _signal_user_tts_provider(name, tts_config, "warm")
|
|
if signalled is not None:
|
|
result.update(warmed=signalled != "error", action="warmed" if signalled != "error" else "error")
|
|
return result
|
|
|
|
feature = _lazy_sdk_feature_for_provider(name)
|
|
if feature is not None:
|
|
try:
|
|
from tools.lazy_deps import ensure, is_available
|
|
|
|
if is_available(feature):
|
|
result.update(warmed=True, action="cached")
|
|
else:
|
|
ensure(feature, prompt=False)
|
|
result.update(warmed=True, action="installed")
|
|
except Exception as exc:
|
|
logger.debug("[TTS] SDK warm-up for %s skipped: %s", name, exc)
|
|
result.update(action="error", error=str(exc))
|
|
return result
|
|
|
|
|
|
def release_tts_provider(provider: Optional[str] = None) -> Dict[str, Any]:
|
|
"""Drop resident local TTS models so their memory is returned.
|
|
|
|
With ``provider`` given only that engine's cache is cleared; otherwise
|
|
every local cache is, and the configured user-declared provider is
|
|
signalled (plugin ``release()`` / command ``release_command``). Returns
|
|
``{"released": <model instances dropped>}``.
|
|
"""
|
|
name = (provider or "").lower().strip()
|
|
if not name:
|
|
tts_config = _load_tts_config()
|
|
_signal_user_tts_provider(_get_provider(tts_config), tts_config, "release")
|
|
released = 0
|
|
for cache_name, cache in _LOCAL_TTS_MODEL_CACHES.items():
|
|
if name and cache_name != name:
|
|
continue
|
|
released += len(cache)
|
|
cache.clear()
|
|
if released:
|
|
logger.info("[TTS] released %d resident local model(s)", released)
|
|
return {"released": released}
|
|
|
|
|
|
def acquire_tts_lease(lease: str, tts_config: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
|
"""Register ``lease`` (e.g. ``"desktop:read-aloud"``) as a live consumer and warm the provider.
|
|
|
|
Re-acquiring is idempotent but still re-warms (cheap on a cache hit, and
|
|
heals a cache cleared elsewhere).
|
|
"""
|
|
with _tts_lease_lock:
|
|
_tts_leases.add(lease)
|
|
holders = len(_tts_leases)
|
|
result = warm_tts_provider(tts_config)
|
|
result["leases"] = holders
|
|
return result
|
|
|
|
|
|
def release_tts_lease(lease: str) -> Dict[str, Any]:
|
|
"""Drop ``lease``; when it was the last one, unload resident local models.
|
|
|
|
Releasing a never-acquired lease is a no-op (still reports the holder
|
|
count) so surfaces can call it unconditionally on their "off" path.
|
|
"""
|
|
with _tts_lease_lock:
|
|
_tts_leases.discard(lease)
|
|
holders = len(_tts_leases)
|
|
result: Dict[str, Any] = {"leases": holders, "released": 0}
|
|
if holders == 0:
|
|
result["released"] = release_tts_provider()["released"]
|
|
return result
|
|
|
|
|
|
def tts_lease_holders() -> List[str]:
|
|
"""Snapshot of live lease names (diagnostics / tests)."""
|
|
with _tts_lease_lock:
|
|
return sorted(_tts_leases)
|
|
|
|
|
|
def _reset_tts_leases_for_tests() -> None:
|
|
with _tts_lease_lock:
|
|
_tts_leases.clear()
|
|
|
|
|
|
# ===========================================================================
|
|
# Built-in provider dispatch
|
|
# ===========================================================================
|
|
# provider -> (importer-name or None, "package missing" error, log line,
|
|
# generator-name). Names are looked up in module globals at call time so
|
|
# tests that monkeypatch ``tools.tts_tool._import_x`` / ``_generate_x`` apply.
|
|
_BUILTIN_DISPATCH: Dict[str, tuple] = {
|
|
"elevenlabs": (
|
|
"_import_elevenlabs",
|
|
"ElevenLabs provider selected but 'elevenlabs' package not installed. Run: pip install elevenlabs",
|
|
"Generating speech with ElevenLabs...",
|
|
"_generate_elevenlabs",
|
|
),
|
|
"openai": (
|
|
"_import_openai_client",
|
|
"OpenAI provider selected but 'openai' package not installed.",
|
|
"Generating speech with OpenAI TTS...",
|
|
"_generate_openai_tts",
|
|
),
|
|
"deepinfra": (
|
|
"_import_openai_client",
|
|
"DeepInfra TTS uses the 'openai' SDK but it isn't installed.",
|
|
"Generating speech with DeepInfra TTS...",
|
|
"_generate_deepinfra_tts",
|
|
),
|
|
"minimax": (None, None, "Generating speech with MiniMax TTS...", "_generate_minimax_tts"),
|
|
"xai": (None, None, "Generating speech with xAI TTS...", "_generate_xai_tts"),
|
|
"mistral": (
|
|
"_import_mistral_client",
|
|
"Mistral provider selected but 'mistralai' package not installed. "
|
|
"Run `hermes setup` to install Mistral support.",
|
|
"Generating speech with Mistral Voxtral TTS...",
|
|
"_generate_mistral_tts",
|
|
),
|
|
"gemini": (None, None, "Generating speech with Google Gemini TTS...", "_generate_gemini_tts"),
|
|
"kittentts": (
|
|
"_import_kittentts",
|
|
"KittenTTS provider selected but 'kittentts' package not installed. "
|
|
"Run 'hermes setup tts' and choose KittenTTS, or install manually: "
|
|
"pip install https://github.com/KittenML/KittenTTS/releases/download/0.8.1/kittentts-0.8.1-py3-none-any.whl",
|
|
"Generating speech with KittenTTS (local, ~25MB)...",
|
|
"_generate_kittentts",
|
|
),
|
|
"piper": (
|
|
"_import_piper",
|
|
"Piper provider selected but 'piper-tts' package not installed. "
|
|
"Run 'hermes tools' and select Piper under TTS, or install manually: "
|
|
"pip install piper-tts",
|
|
"Generating speech with Piper (local)...",
|
|
"_generate_piper_tts",
|
|
),
|
|
}
|
|
_NEUTTS_MISSING_ERROR = (
|
|
"NeuTTS provider selected but neutts is not installed. "
|
|
"Run hermes setup and choose NeuTTS, or install espeak-ng and run python -m pip install -U neutts[all]."
|
|
)
|
|
|
|
|
|
def _error_json(message: str) -> str:
|
|
return json.dumps({"success": False, "error": message}, ensure_ascii=False)
|
|
|
|
|
|
def _run_edge_tts(text: str, file_str: str, tts_config: Dict[str, Any]) -> None:
|
|
"""Run the async Edge generator from sync code (worker thread; direct run if that fails)."""
|
|
try:
|
|
import concurrent.futures
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
pool.submit(
|
|
lambda: asyncio.run(_generate_edge_tts(text, file_str, tts_config))
|
|
).result(timeout=60)
|
|
except RuntimeError:
|
|
asyncio.run(_generate_edge_tts(text, file_str, tts_config))
|
|
|
|
|
|
def _select_builtin_engine(provider: str) -> tuple:
|
|
"""Check a built-in provider's SDK. Returns ``(engine, None)`` or ``(provider, error_json)``.
|
|
|
|
Unknown names take the Edge default; when edge-tts is missing, NeuTTS is
|
|
the local fallback (``engine`` then differs from ``provider``).
|
|
"""
|
|
entry = _BUILTIN_DISPATCH.get(provider)
|
|
if entry is not None:
|
|
importer_name, missing_error = entry[0], entry[1]
|
|
if importer_name is not None and not _importable(globals()[importer_name]):
|
|
return provider, _error_json(missing_error)
|
|
return provider, None
|
|
if provider == "neutts":
|
|
if not _check_neutts_available():
|
|
return provider, _error_json(_NEUTTS_MISSING_ERROR)
|
|
logger.info("Generating speech with NeuTTS (local)...")
|
|
return provider, None
|
|
if _importable(_import_edge_tts):
|
|
return provider, None # Edge default; the reported provider stays as configured
|
|
if _check_neutts_available():
|
|
logger.info("Edge TTS not available, falling back to NeuTTS (local)...")
|
|
return "neutts", None
|
|
return provider, _error_json(
|
|
"No TTS provider available. Install edge-tts (pip install edge-tts) "
|
|
"or set up NeuTTS for local synthesis."
|
|
)
|
|
|
|
|
|
def _synthesize_builtin(engine: str, text: str, file_str: str, tts_config: Dict[str, Any], instructions: Optional[str]) -> None:
|
|
"""Run the already-selected built-in *engine* (the caller logs the engine-selection line)."""
|
|
entry = _BUILTIN_DISPATCH.get(engine)
|
|
if entry is not None:
|
|
logger.info(entry[2])
|
|
if engine == "openai":
|
|
_generate_openai_tts(text, file_str, tts_config, instructions=instructions)
|
|
else:
|
|
globals()[entry[3]](text, file_str, tts_config)
|
|
elif engine == "neutts":
|
|
_generate_neutts(text, file_str, tts_config)
|
|
else:
|
|
logger.info("Generating speech with Edge TTS...")
|
|
_run_edge_tts(text, file_str, tts_config)
|
|
|
|
|
|
def _finalize_voice_delivery(
|
|
file_str: str,
|
|
provider: str,
|
|
command_provider_config: Optional[Dict[str, Any]],
|
|
want_opus: bool,
|
|
) -> tuple:
|
|
"""Decide voice-bubble eligibility and Opus-convert when needed.
|
|
|
|
Command and plugin providers are documents by default and opt in via
|
|
``voice_compatible``; native-Opus built-ins are voice-compatible when the
|
|
platform wants Opus and they wrote .ogg; MP3/WAV built-ins are converted
|
|
with ffmpeg only when the platform needs Opus. Returns ``(path, voice_compatible)``.
|
|
"""
|
|
voice_compatible = False
|
|
if command_provider_config is not None:
|
|
opted_in = _is_command_tts_voice_compatible(command_provider_config)
|
|
elif provider not in BUILTIN_TTS_PROVIDERS:
|
|
opted_in = _plugin_provider_is_voice_compatible(provider)
|
|
elif want_opus and provider in _FFMPEG_OPUS_PROVIDERS and not file_str.endswith(".ogg"):
|
|
opus_path = _convert_to_opus(file_str)
|
|
if opus_path:
|
|
return opus_path, True
|
|
return file_str, False
|
|
elif provider in _NATIVE_OPUS_PROVIDERS:
|
|
return file_str, want_opus and file_str.endswith(".ogg")
|
|
else:
|
|
return file_str, False
|
|
|
|
if opted_in:
|
|
if not file_str.endswith(".ogg"):
|
|
opus_path = _convert_to_opus(file_str)
|
|
if opus_path:
|
|
file_str = opus_path
|
|
voice_compatible = file_str.endswith(".ogg")
|
|
return file_str, voice_compatible
|
|
|
|
|
|
# ===========================================================================
|
|
# Main tool function
|
|
# ===========================================================================
|
|
|
|
def _apply_call_overrides(tts_config: Dict[str, Any], speed: Optional[float], provider: Optional[str]):
|
|
"""Apply per-call ``speed`` (clamped, on a shallow copy) and resolve the provider name."""
|
|
if speed is not None:
|
|
clamped = max(0.25, min(4.0, float(speed)))
|
|
tts_config = dict(tts_config) # shallow copy to avoid mutating the cache
|
|
tts_config["speed"] = clamped
|
|
provider = provider.lower().strip() if provider else _get_provider(tts_config)
|
|
return tts_config, provider
|
|
|
|
|
|
def _session_platform() -> tuple:
|
|
"""``(platform, wants_opus)`` — platforms delivering voice bubbles only as Ogg/Opus want Opus."""
|
|
from gateway.session_context import get_session_env
|
|
platform = get_session_env("HERMES_SESSION_PLATFORM", "").lower()
|
|
return platform, platform in OPUS_VOICE_PLATFORMS
|
|
|
|
|
|
def _resolve_output_base(
|
|
output_path: Optional[str],
|
|
provider: str,
|
|
command_provider_config: Optional[Dict[str, Any]],
|
|
want_opus: bool,
|
|
) -> tuple:
|
|
"""Pick the output file. Returns ``(Path, None)`` or ``(None, error_json)``.
|
|
|
|
A caller-supplied path is rejected on ``..`` traversal (bug or
|
|
prompt-injection; an absolute path is fine) and on protected credential/
|
|
system locations. Command providers get their configured extension.
|
|
Default: ``<audio cache>/tts_<timestamp>.<ext>`` where ext is the command
|
|
format, ``.ogg`` for native-Opus providers on Opus platforms, else ``.mp3``.
|
|
"""
|
|
if output_path:
|
|
from tools.path_security import has_traversal_component
|
|
if has_traversal_component(output_path):
|
|
return None, _error_json(
|
|
f"output_path contains '..' traversal component: {output_path}. "
|
|
"Use an absolute path or one relative to the current directory "
|
|
"without '..'."
|
|
)
|
|
file_path = Path(output_path).expanduser()
|
|
if command_provider_config is not None:
|
|
file_path = _configured_command_tts_output_path(file_path, command_provider_config)
|
|
from agent.file_safety import is_write_approval_required, is_write_denied
|
|
if is_write_denied(str(file_path)) or is_write_approval_required(str(file_path)):
|
|
return None, _error_json(
|
|
f"output_path targets a protected credential or system path: "
|
|
f"{file_path}. Choose a normal audio output location."
|
|
)
|
|
else:
|
|
timestamp = datetime.datetime.now().strftime("%Y%m%d_%H%M%S_%f")
|
|
out_dir = Path(_default_output_dir())
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
if command_provider_config is not None:
|
|
ext = _get_command_tts_output_format(command_provider_config)
|
|
elif want_opus and provider in _NATIVE_OPUS_PROVIDERS:
|
|
ext = "ogg"
|
|
else:
|
|
ext = "mp3"
|
|
file_path = out_dir / f"tts_{timestamp}.{ext}"
|
|
file_path.parent.mkdir(parents=True, exist_ok=True)
|
|
return file_path, None
|
|
|
|
|
|
def _text_to_speech_single(
|
|
text: str,
|
|
output_path: Optional[str] = None,
|
|
*,
|
|
speed: Optional[float] = None,
|
|
instructions: Optional[str] = None,
|
|
provider: Optional[str] = None,
|
|
tts_config_override: Optional[Dict[str, Any]] = None,
|
|
) -> str:
|
|
"""Synthesize one provider-safe text chunk and return one final-encoded file.
|
|
|
|
Text arrives already normalized; :func:`text_to_speech_tool` owns
|
|
long-form splitting, delivery packing and size enforcement.
|
|
"""
|
|
if not text or not text.strip():
|
|
return tool_error("Text is required", success=False)
|
|
|
|
tts_config = tts_config_override if tts_config_override is not None else _load_tts_config()
|
|
tts_config, provider = _apply_call_overrides(tts_config, speed, provider)
|
|
|
|
# Command providers resolve BEFORE built-in dispatch; built-in names
|
|
# short-circuit so ``tts.providers.openai.command`` can't shadow OpenAI.
|
|
command_provider_config = _resolve_command_provider_config(provider, tts_config)
|
|
|
|
max_len = _resolve_max_text_length(provider, tts_config)
|
|
if len(text) > max_len:
|
|
logger.warning(
|
|
"TTS text exceeds provider %s cap (%d > %d chars) — "
|
|
"use text_to_speech_tool() for automatic chunking",
|
|
provider, len(text), max_len,
|
|
)
|
|
|
|
_platform, want_opus = _session_platform()
|
|
file_path, error = _resolve_output_base(output_path, provider, command_provider_config, want_opus)
|
|
if error:
|
|
return error
|
|
file_str = str(file_path)
|
|
|
|
try:
|
|
if command_provider_config is not None:
|
|
logger.info("Generating speech with command TTS provider '%s'...", provider)
|
|
file_str = _generate_command_tts(text, file_str, provider, command_provider_config, tts_config)
|
|
# Plugin provider: only for names that are neither built-in nor command;
|
|
# a None return falls through to built-in dispatch (unknown -> Edge default).
|
|
elif provider not in BUILTIN_TTS_PROVIDERS and (
|
|
_plugin_path := _dispatch_to_plugin_provider(text, file_str, provider, tts_config)
|
|
) is not None:
|
|
file_str = _plugin_path
|
|
else:
|
|
provider, error = _select_builtin_engine(provider)
|
|
if error:
|
|
return error
|
|
_synthesize_builtin(provider, text, file_str, tts_config, instructions)
|
|
|
|
if not os.path.exists(file_str) or os.path.getsize(file_str) == 0:
|
|
return _error_json(f"TTS generation produced no output (provider: {provider})")
|
|
|
|
# Sniff once for every provider: MP3/WAV bytes in a .ogg path render
|
|
# as broken 0-second voice bubbles.
|
|
file_str = _repair_ogg_container(file_str)
|
|
file_str, voice_compatible = _finalize_voice_delivery(
|
|
file_str, provider, command_provider_config, want_opus,
|
|
)
|
|
|
|
file_size = os.path.getsize(file_str)
|
|
logger.info("TTS audio saved: %s (%s bytes, provider: %s)", file_str, f"{file_size:,}", provider)
|
|
|
|
media_tag = f"MEDIA:{file_str}"
|
|
if voice_compatible:
|
|
media_tag = f"[[audio_as_voice]]\n{media_tag}"
|
|
|
|
return json.dumps({
|
|
"success": True,
|
|
"file_path": file_str,
|
|
"media_tag": media_tag,
|
|
"provider": provider,
|
|
"voice_compatible": voice_compatible,
|
|
}, ensure_ascii=False)
|
|
|
|
except ValueError as e:
|
|
error_msg = f"TTS configuration error ({provider}): {e}"
|
|
logger.error("%s", error_msg)
|
|
return tool_error(error_msg, success=False)
|
|
except FileNotFoundError as e:
|
|
error_msg = f"TTS dependency missing ({provider}): {e}"
|
|
logger.error("%s", error_msg, exc_info=True)
|
|
return tool_error(error_msg, success=False)
|
|
except Exception as e:
|
|
error_msg = f"TTS generation failed ({provider}): {e}"
|
|
logger.error("%s", error_msg, exc_info=True)
|
|
return tool_error(error_msg, success=False)
|
|
|
|
|
|
def text_to_speech_tool(
|
|
text: str,
|
|
output_path: Optional[str] = None,
|
|
speed: Optional[float] = None,
|
|
instructions: Optional[str] = None,
|
|
provider: Optional[str] = None,
|
|
) -> str:
|
|
"""Convert text to speech audio with long-form chunking.
|
|
|
|
Text is normalized, split into provider-safe chunks, synthesized
|
|
sequentially (each chunk final-encoded), then packed against the
|
|
destination platform's upload limit. Multi-chunk voice output is
|
|
re-encoded when combined; a failed combine keeps separate valid files;
|
|
no over-limit artifact is returned. On messaging platforms the
|
|
``MEDIA:<path>`` tag is delivered as a native voice message.
|
|
|
|
Args:
|
|
text: Text to speak; longer input is split into ordered chunks, never
|
|
silently truncated.
|
|
output_path: Optional custom save path.
|
|
speed: Optional playback speed multiplier (0.25-4.0).
|
|
instructions: Optional voice-design guidance (tone, emotion, pacing).
|
|
provider: Optional TTS provider override.
|
|
|
|
Returns:
|
|
str: JSON result with success, file_path, file_paths, and MEDIA tag.
|
|
"""
|
|
if not text or not text.strip():
|
|
return tool_error("Text is required", success=False)
|
|
|
|
# Shared cleaner: markdown, emoji, think blocks, verifier footer, units, newlines.
|
|
try:
|
|
from tools.tts_text_normalize import prepare_spoken_text
|
|
text = prepare_spoken_text(text, max_chars=None)
|
|
except Exception:
|
|
text = text.strip()
|
|
if not text:
|
|
return tool_error("Text is empty after TTS cleanup", success=False)
|
|
|
|
tts_config, provider = _apply_call_overrides(_load_tts_config(), speed, provider)
|
|
|
|
command_provider_config = _resolve_command_provider_config(provider, tts_config)
|
|
max_len = _resolve_max_text_length(provider, tts_config)
|
|
chunks = _split_text_for_tts(text, max_len)
|
|
if not chunks:
|
|
return tool_error("Text is required", success=False)
|
|
if len(chunks) > 1:
|
|
logger.info(
|
|
"TTS text for provider %s split into %d chunks (input=%d chars, cap=%d)",
|
|
provider,
|
|
len(chunks),
|
|
len(text),
|
|
max_len,
|
|
)
|
|
|
|
platform, want_opus = _session_platform()
|
|
delivery_profile = _resolve_audio_delivery_profile(platform, tts_config)
|
|
|
|
base_path, error = _resolve_output_base(output_path, provider, command_provider_config, want_opus)
|
|
if error:
|
|
return error
|
|
|
|
generated_artifacts: set[str] = set()
|
|
final_paths: List[str] = []
|
|
chunk_results: List[Dict[str, Any]] = []
|
|
try:
|
|
encoded_paths: List[str] = []
|
|
for index, chunk in enumerate(chunks, start=1):
|
|
if len(chunks) == 1:
|
|
chunk_path = base_path
|
|
else:
|
|
chunk_path = base_path.with_name(
|
|
f"{base_path.stem}.chunk{index:03d}{base_path.suffix}"
|
|
)
|
|
generated_artifacts.add(str(chunk_path))
|
|
raw_result = _text_to_speech_single(
|
|
text=chunk,
|
|
output_path=str(chunk_path),
|
|
speed=speed,
|
|
instructions=instructions,
|
|
provider=provider,
|
|
tts_config_override=tts_config,
|
|
)
|
|
try:
|
|
chunk_result = json.loads(raw_result)
|
|
except (json.JSONDecodeError, TypeError):
|
|
raise RuntimeError(
|
|
f"TTS chunk {index} returned invalid JSON: {str(raw_result)[:200]}"
|
|
)
|
|
if not chunk_result.get("success"):
|
|
error_msg = chunk_result.get("error", "unknown error")
|
|
return tool_error(
|
|
f"TTS chunk {index} failed ({provider}): {error_msg}",
|
|
success=False,
|
|
)
|
|
actual_path = str(chunk_result.get("file_path") or chunk_path)
|
|
if not os.path.isfile(actual_path) or os.path.getsize(actual_path) <= 0:
|
|
raise RuntimeError(
|
|
f"TTS chunk {index} produced no final audio: {actual_path}"
|
|
)
|
|
generated_artifacts.add(actual_path)
|
|
encoded_paths.append(actual_path)
|
|
chunk_results.append(chunk_result)
|
|
|
|
voice_compatible = bool(chunk_results) and all(
|
|
bool(result.get("voice_compatible")) for result in chunk_results
|
|
)
|
|
delivery_base = base_path.with_suffix(Path(encoded_paths[0]).suffix)
|
|
final_paths, combined_chunks = _build_audio_delivery_files(
|
|
encoded_paths,
|
|
str(delivery_base),
|
|
delivery_profile,
|
|
voice_compatible=voice_compatible,
|
|
)
|
|
|
|
for path in final_paths:
|
|
logger.info(
|
|
"TTS audio saved: %s (%s bytes, provider: %s)",
|
|
path,
|
|
f"{os.path.getsize(path):,}",
|
|
provider,
|
|
)
|
|
media_tag = "\n".join(f"MEDIA:{path}" for path in final_paths)
|
|
if voice_compatible:
|
|
media_tag = f"[[audio_as_voice]]\n{media_tag}"
|
|
|
|
return json.dumps({
|
|
"success": True,
|
|
"file_path": final_paths[0],
|
|
"file_paths": final_paths,
|
|
"media_tag": media_tag,
|
|
"provider": chunk_results[0].get("provider", provider),
|
|
"voice_compatible": voice_compatible,
|
|
"chunk_count": len(chunks),
|
|
"delivery_file_count": len(final_paths),
|
|
"combined_chunks": bool(combined_chunks),
|
|
"delivery_profile": {
|
|
"platform": delivery_profile.platform,
|
|
"max_file_bytes": delivery_profile.max_file_bytes,
|
|
"target_file_bytes": delivery_profile.target_file_bytes,
|
|
},
|
|
}, ensure_ascii=False)
|
|
except ValueError as exc:
|
|
error_msg = f"TTS delivery error ({provider}): {exc}"
|
|
logger.error("%s", error_msg)
|
|
return tool_error(error_msg, success=False)
|
|
except Exception as exc:
|
|
error_msg = f"TTS long-form generation failed ({provider}): {exc}"
|
|
logger.error("%s", error_msg, exc_info=True)
|
|
return tool_error(error_msg, success=False)
|
|
finally:
|
|
final_absolute = {os.path.abspath(path) for path in final_paths}
|
|
for artifact in generated_artifacts:
|
|
if os.path.abspath(artifact) in final_absolute:
|
|
continue
|
|
try:
|
|
os.unlink(artifact)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
# ===========================================================================
|
|
# Requirements check
|
|
# ===========================================================================
|
|
|
|
def _importable(importer: Callable[[], Any]) -> bool:
|
|
try:
|
|
importer()
|
|
return True
|
|
except ImportError:
|
|
return False
|
|
|
|
|
|
def _edge_requirements() -> bool:
|
|
return _importable(_import_edge_tts) or _check_neutts_available()
|
|
|
|
|
|
def _elevenlabs_requirements() -> bool:
|
|
return _importable(_import_elevenlabs) and bool(_resolve_provider_key("ELEVENLABS_API_KEY", "elevenlabs"))
|
|
|
|
|
|
def _openai_requirements() -> bool:
|
|
return importlib.util.find_spec("openai") is not None and _has_openai_audio_backend()
|
|
|
|
|
|
def _deepinfra_requirements() -> bool:
|
|
return importlib.util.find_spec("openai") is not None and bool(
|
|
_resolve_provider_key("DEEPINFRA_API_KEY", "deepinfra")
|
|
)
|
|
|
|
|
|
def _minimax_requirements() -> bool:
|
|
try:
|
|
_resolve_minimax_tts_runtime(_load_tts_config())
|
|
except ValueError:
|
|
return False
|
|
return True
|
|
|
|
|
|
def _xai_requirements() -> bool:
|
|
try:
|
|
from tools.xai_http import resolve_xai_http_credentials
|
|
|
|
return bool(resolve_xai_http_credentials().get("api_key"))
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _gemini_requirements() -> bool:
|
|
return bool(
|
|
_resolve_provider_key("GEMINI_API_KEY", "gemini")
|
|
or _resolve_provider_key("GOOGLE_API_KEY", "gemini")
|
|
)
|
|
|
|
|
|
def _mistral_requirements() -> bool:
|
|
return _importable(_import_mistral_client) and bool(_resolve_provider_key("MISTRAL_API_KEY", "mistral"))
|
|
|
|
|
|
# Must mirror text_to_speech_tool dispatch: unrelated cloud credentials never
|
|
# make the Edge default usable, and an explicit provider is checked on its own.
|
|
_BUILTIN_REQUIREMENTS: Dict[str, Callable[[], bool]] = {
|
|
"edge": _edge_requirements,
|
|
"elevenlabs": _elevenlabs_requirements,
|
|
"openai": _openai_requirements,
|
|
"deepinfra": _deepinfra_requirements,
|
|
"minimax": _minimax_requirements,
|
|
"xai": _xai_requirements,
|
|
"gemini": _gemini_requirements,
|
|
"mistral": _mistral_requirements,
|
|
"neutts": lambda: _check_neutts_available(),
|
|
"kittentts": lambda: _check_kittentts_available(),
|
|
"piper": lambda: _check_piper_available(),
|
|
}
|
|
|
|
|
|
def check_tts_requirements() -> bool:
|
|
"""Return whether the explicitly resolved TTS provider can run."""
|
|
tts_config = _load_tts_config()
|
|
provider = _get_provider(tts_config)
|
|
if _resolve_command_provider_config(provider, tts_config) is not None:
|
|
return True
|
|
|
|
check = _BUILTIN_REQUIREMENTS.get(provider)
|
|
if check is not None:
|
|
return check()
|
|
|
|
try:
|
|
from agent.tts_registry import get_provider
|
|
from hermes_cli.plugins import _ensure_plugins_discovered
|
|
|
|
_ensure_plugins_discovered()
|
|
plugin = get_provider(provider)
|
|
return bool(plugin and plugin.is_available())
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _managed_openai_audio_route() -> Optional[tuple]:
|
|
gateway = resolve_managed_tool_gateway("openai-audio")
|
|
if gateway is None:
|
|
return None
|
|
return gateway.nous_user_token, urljoin(f"{gateway.gateway_origin.rstrip('/')}/", "v1"), True
|
|
|
|
|
|
def _resolve_openai_audio_client_config() -> tuple[str, str, bool]:
|
|
"""Return ``(api_key, base_url, is_managed)`` for the OpenAI audio client.
|
|
|
|
``is_managed`` marks the Nous managed audio gateway (a restricted proxy)
|
|
so callers can coerce the request to what it supports. Strict selection
|
|
semantics on the stored ``tts`` provider:
|
|
- ``"nous"`` → managed gateway ONLY; unentitled/unreachable is an error.
|
|
- any other stored provider → direct credentials ONLY (``tts.openai.api_key``
|
|
then ``VOICE_TOOLS_OPENAI_KEY``/``OPENAI_API_KEY``); no silent managed fallback.
|
|
- never-configured tts section → legacy ladder: config key → env key → managed.
|
|
"""
|
|
tts_config = _load_tts_config()
|
|
openai_cfg = (tts_config.get("openai") if isinstance(tts_config, dict) else None) or {}
|
|
cfg_api_key = openai_cfg.get("api_key") or ""
|
|
cfg_base_url = openai_cfg.get("base_url") or ""
|
|
direct_base = cfg_base_url or DEFAULT_OPENAI_BASE_URL
|
|
|
|
selected = read_selection("tts")
|
|
|
|
if selected == NOUS_MANAGED_PROVIDER:
|
|
route = _managed_openai_audio_route()
|
|
if route is None:
|
|
raise ValueError(selection_error(
|
|
"tts",
|
|
NOUS_MANAGED_PROVIDER,
|
|
"the Nous Tool Gateway is not available (not entitled or "
|
|
"unreachable)",
|
|
))
|
|
return route
|
|
|
|
if cfg_api_key:
|
|
return cfg_api_key, direct_base, False
|
|
direct_api_key = resolve_openai_audio_api_key()
|
|
if direct_api_key:
|
|
return direct_api_key, direct_base, False
|
|
|
|
if selected is not None:
|
|
raise ValueError(selection_error(
|
|
"tts",
|
|
selected,
|
|
"neither tts.openai.api_key in config nor "
|
|
"VOICE_TOOLS_OPENAI_KEY/OPENAI_API_KEY is set",
|
|
))
|
|
|
|
route = _managed_openai_audio_route()
|
|
if route is None:
|
|
message = (
|
|
"Neither tts.openai.api_key in config nor "
|
|
"VOICE_TOOLS_OPENAI_KEY/OPENAI_API_KEY is set"
|
|
)
|
|
if managed_nous_tools_enabled():
|
|
message += (
|
|
". "
|
|
+ nous_tool_gateway_unavailable_message(
|
|
"managed OpenAI audio for TTS",
|
|
)
|
|
)
|
|
raise ValueError(message)
|
|
return route
|
|
|
|
|
|
def _has_openai_audio_backend() -> bool:
|
|
"""Return True when the selected OpenAI audio route is usable."""
|
|
try:
|
|
_resolve_openai_audio_client_config()
|
|
return True
|
|
except ValueError:
|
|
return False
|
|
|
|
|
|
# ===========================================================================
|
|
# Speech text cleanup (shared by voice-mode streaming and gateway auto-TTS)
|
|
# ===========================================================================
|
|
# Legacy regex fallback, only used if the shared normalizer raises.
|
|
_LEGACY_TTS_STRIP_STEPS = (
|
|
(re.compile(r'<think[\s>].*?</think>', flags=re.DOTALL), ' '),
|
|
(re.compile(r'```[\s\S]*?```'), ' '),
|
|
(re.compile(r'\[([^\]]+)\]\([^)]+\)'), r'\1'),
|
|
(re.compile(r'https?://\S+'), ''),
|
|
(re.compile(r'\*\*(.+?)\*\*'), r'\1'),
|
|
(re.compile(r'\*(.+?)\*'), r'\1'),
|
|
(re.compile(r'`(.+?)`'), r'\1'),
|
|
(re.compile(r'^#+\s*', flags=re.MULTILINE), ''),
|
|
(re.compile(r'^\s*[-*]\s+', flags=re.MULTILINE), ''),
|
|
(re.compile(r'---+'), ''),
|
|
# Emoji + variation selectors/ZWJ: providers speak them as awkward labels.
|
|
(re.compile('[\U0001F000-\U0001FAFF\u2600-\u27BF\uFE0F\u200D\U000E0020-\U000E007F]+'), ' '),
|
|
(re.compile(r'\n{3,}'), '\n\n'),
|
|
)
|
|
|
|
|
|
def _strip_markdown_for_tts(text: str) -> str:
|
|
"""Prepare text for speech via the shared cleaner in tts_text_normalize.
|
|
|
|
One cleaner for every TTS path (tool, gateway auto-TTS, voice-mode
|
|
streaming, web dashboard): strips <think> blocks, the verifier footer,
|
|
markdown and emoji; expands units; flattens newlines so newline-sensitive
|
|
providers (Kokoro) speak the whole script. Falls back to the legacy regex
|
|
pipeline if the normalizer ever fails.
|
|
"""
|
|
try:
|
|
from tools.tts_text_normalize import prepare_spoken_text
|
|
return prepare_spoken_text(text, max_chars=None)
|
|
except Exception:
|
|
pass
|
|
for pattern, repl in _LEGACY_TTS_STRIP_STEPS:
|
|
text = pattern.sub(repl, text)
|
|
return text.strip()
|
|
|
|
|
|
# ===========================================================================
|
|
# Main -- quick diagnostics
|
|
# ===========================================================================
|
|
if __name__ == "__main__":
|
|
print("🔊 Text-to-Speech Tool Module")
|
|
print("=" * 50)
|
|
|
|
print("\nProvider availability:")
|
|
print(f" Edge TTS: {'installed' if _importable(_import_edge_tts) else 'not installed (pip install edge-tts)'}")
|
|
print(f" ElevenLabs: {'installed' if _importable(_import_elevenlabs) else 'not installed (pip install elevenlabs)'}")
|
|
print(f" API Key: {'set' if _resolve_provider_key('ELEVENLABS_API_KEY', 'elevenlabs') else 'not set'}")
|
|
print(f" OpenAI: {'installed' if _importable(_import_openai_client) else 'not installed'}")
|
|
print(
|
|
" API Key: "
|
|
f"{'set' if resolve_openai_audio_api_key() else 'not set (VOICE_TOOLS_OPENAI_KEY or OPENAI_API_KEY)'}"
|
|
)
|
|
config = _load_tts_config()
|
|
try:
|
|
minimax_runtime = _resolve_minimax_tts_runtime(config)
|
|
minimax_status = (
|
|
f"API key set ({minimax_runtime.region}, "
|
|
f"{minimax_runtime.credential_source})"
|
|
)
|
|
except ValueError as exc:
|
|
minimax_status = f"unavailable ({exc})"
|
|
print(f" MiniMax: {minimax_status}")
|
|
print(f" Piper: {'installed' if _check_piper_available() else 'not installed (pip install piper-tts)'}")
|
|
print(f" ffmpeg: {'✅ found' if _has_ffmpeg() else '❌ not found (needed for Telegram Opus)'}")
|
|
print(f"\n Output dir: {_default_output_dir()}")
|
|
|
|
provider = _get_provider(config)
|
|
print(f" Configured provider: {provider}")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Registry
|
|
# ---------------------------------------------------------------------------
|
|
from tools.registry import registry, tool_error
|
|
|
|
TTS_SCHEMA = {
|
|
"name": "text_to_speech",
|
|
"description": "Convert text to speech audio. Returns a MEDIA: path that the platform delivers as native audio. Compatible providers render as a voice bubble on Telegram; otherwise audio is sent as a regular attachment. In CLI mode, saves to ~/voice-memos/. Voice and provider are user-configured (built-in providers like edge/openai or custom command providers under tts.providers.<name>), not model-selected.",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {
|
|
"text": {
|
|
"type": "string",
|
|
"description": "The text to convert to speech. Provider-specific per-request character caps apply automatically (OpenAI 4096, xAI 15000, MiniMax 10000, ElevenLabs 5k-40k depending on model); longer input is split into ordered chunks without silent truncation."
|
|
},
|
|
"output_path": {
|
|
"type": "string",
|
|
"description": f"Optional custom file path to save the audio. Defaults to {display_hermes_home()}/audio_cache/<timestamp>.mp3"
|
|
},
|
|
"speed": {
|
|
"type": "number",
|
|
"description": "Playback speed multiplier. 1.0 = normal, 0.5 = very slow (language learning), 2.0 = fast. Range: 0.25-4.0. Overrides the speed configured in config.yaml."
|
|
},
|
|
"instructions": {
|
|
"type": "string",
|
|
"description": (
|
|
"Optional voice-design guidance: tone, emotion, pacing, accent, "
|
|
"whispering, impressions (e.g. 'Speak in a cheerful, excited whisper'). "
|
|
"Forwarded to the OpenAI backend (gpt-4o-mini-tts and OpenAI-compatible "
|
|
"voice-design servers). Silently ignored by backends that don't support it."
|
|
)
|
|
},
|
|
"provider": {
|
|
"type": "string",
|
|
"description": (
|
|
"Optional TTS provider override. Accepts built-in names "
|
|
"(edge, openai, elevenlabs, minimax, xai, mistral, gemini, "
|
|
"neutts, kittentts, piper), user-declared command provider "
|
|
"names from tts.providers.<name>, or plugin-registered names. "
|
|
"When omitted, the configured tts.provider from config.yaml is used."
|
|
)
|
|
}
|
|
},
|
|
"required": ["text"]
|
|
}
|
|
}
|
|
|
|
registry.register(
|
|
name="text_to_speech",
|
|
toolset="tts",
|
|
schema=TTS_SCHEMA,
|
|
handler=lambda args, **kw: text_to_speech_tool(
|
|
text=args.get("text", ""),
|
|
output_path=args.get("output_path"),
|
|
speed=args.get("speed"),
|
|
instructions=args.get("instructions"),
|
|
provider=args.get("provider")),
|
|
check_fn=check_tts_requirements,
|
|
emoji="🔊",
|
|
)
|