From ce963a46b13b60604679a4b24f41be253a67e54f Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 18 Sep 2026 19:32:57 +0800 Subject: [PATCH] fix(tts): play the artifact the tool reported in the sync speaker pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit text_to_speech_tool reports its artifacts (file_path/file_paths) in a JSON envelope, and they often land beside — not at — the requested path: a command provider's declared format rewrites the suffix, and voice-compatible delivery ffmpeg-converts to .ogg. _synthesize_to_tmp ignored the envelope and returned its own mkstemp path, so _drain's playback gate was false for every such provider: voice mode was silent, nothing was logged, and the real audio files leaked into TMPDIR. Prefer the first reported artifact that exists and is non-empty, falling back to the requested path. Fixes #115029 --- tests/tools/test_tts_streaming.py | 64 +++++++++++++++++++++++++++++++ tools/tts_tool_speaker.py | 28 +++++++++++++- 2 files changed, 90 insertions(+), 2 deletions(-) diff --git a/tests/tools/test_tts_streaming.py b/tests/tools/test_tts_streaming.py index 83e605fe38..4eff79d2b1 100644 --- a/tests/tools/test_tts_streaming.py +++ b/tests/tools/test_tts_streaming.py @@ -7,6 +7,7 @@ the chunked-streamer playback path, and the universal per-sentence sync fallback """ import os +import json import queue import sys import tempfile @@ -1193,3 +1194,66 @@ def test_speaker_output_stream_opens_at_rate_learned_from_first_chunk(monkeypatc assert done.is_set() assert [c.kwargs["samplerate"] for c in sd.OutputStream.call_args_list] == [44100] assert out.write.call_count == 2 + + +def test_sync_pipeline_plays_the_artifact_the_tool_reported(monkeypatch, tmp_path): + """A provider whose artifact lands off the requested path (command ``format`` suffix + rewrite, or voice-compatible ffmpeg conversion) must still play: follow the reported + ``file_path``/``file_paths`` instead of gating on the requested path (#115029).""" + from tools import tts_tool + from tools.tts_tool_speaker import stream_tts_to_speaker + + ogg = tmp_path / "sentence.ogg" + ogg.write_bytes(b"x" * 32) + + def fake_synth(text, output_path): + # The requested .mp3 stays a zero-byte mkstemp file; only the reported artifact is real. + ogg_str = str(ogg) + return json.dumps({ + "success": True, + "file_path": ogg_str, + "file_paths": [ogg_str], + }) + + played = [] + fake_vm = MagicMock() + fake_vm.play_audio_file.side_effect = played.append + monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth) + monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm) + + q = _drain_queue(["Hello there. "]) + with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None): + stream_tts_to_speaker(q, threading.Event(), threading.Event()) + assert played == [str(ogg)], ( + "sentence dropped: playback ignored the reported artifact" + ) + + +def test_sync_pipeline_falls_back_to_requested_path_when_reported_missing(monkeypatch): + """A tool envelope that reports nothing usable (None, non-JSON, missing files) keeps the + legacy behavior: play the requested path when the tool wrote it there.""" + from tools import tts_tool + from tools.tts_tool_speaker import stream_tts_to_speaker + + def fake_synth(text, output_path): + with open(output_path, "wb") as fh: + fh.write(b"x" * 100) + return json.dumps({"success": False, "error": "shape without paths"}) + + played = [] + fake_vm = MagicMock() + + def _record(path): + played.append((path, os.path.getsize(path))) + + fake_vm.play_audio_file.side_effect = _record + monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth) + monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm) + + q = _drain_queue(["Hello there. "]) + with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None): + stream_tts_to_speaker(q, threading.Event(), threading.Event()) + # The temp file is unlinked after playback, so capture its size at play time. + assert len(played) == 1 and played[0][1] > 0, ( + "requested-path fallback no longer plays" + ) diff --git a/tools/tts_tool_speaker.py b/tools/tts_tool_speaker.py index 043273bf48..e8ccd41bcd 100644 --- a/tools/tts_tool_speaker.py +++ b/tools/tts_tool_speaker.py @@ -12,6 +12,7 @@ from __future__ import annotations import contextlib import itertools +import json import logging import os import platform @@ -73,6 +74,29 @@ def _drain_chunks(chunk_queue: "queue.Queue[Optional[bytes]]") -> List[bytes]: return list(iter(chunk_queue.get, None)) +def _first_written_artifact(raw: object, requested: str) -> str: + """The audio path the TTS tool actually wrote, falling back to the requested one. + + ``text_to_speech_tool`` reports its artifacts (``file_path``/``file_paths``) in a JSON + envelope, and they often land beside — not at — the requested path: a command provider's + declared ``format`` rewrites the suffix, and voice-compatible delivery ffmpeg-converts to + ``.ogg``. Gating playback on the requested path alone drops every such sentence silently, + so prefer the first reported artifact that actually exists and is non-empty.""" + candidates: List[str] = [] + try: + payload = json.loads(raw) if isinstance(raw, str) else (raw or {}) + if isinstance(payload, dict): + reported = payload.get("file_paths") or ([payload["file_path"]] if payload.get("file_path") else []) + if isinstance(reported, list): + candidates = [p for p in reported if isinstance(p, str)] + except (ValueError, TypeError): + pass + for path in candidates: + if os.path.isfile(path) and os.path.getsize(path) > 0: + return path + return requested + + class _SyncSentencePipeline: """Overlap per-sentence synthesis with playback for non-streaming providers. @@ -106,8 +130,8 @@ class _SyncSentencePipeline: try: fd, tmp_path = tempfile.mkstemp(suffix=".mp3") os.close(fd) - _origin().text_to_speech_tool(text=cleaned, output_path=tmp_path) - return tmp_path + raw = _origin().text_to_speech_tool(text=cleaned, output_path=tmp_path) + return _first_written_artifact(raw, tmp_path) except Exception as exc: logger.warning("Sync per-sentence TTS synthesis failed: %s", exc) _unlink_quietly(tmp_path)