fix(tts): play the artifact the tool reported in the sync speaker pipeline

text_to_speech_tool reports its artifacts (file_path/file_paths) in a JSON
envelope, and they often land beside — not at — the requested path: a command
provider's declared format rewrites the suffix, and voice-compatible delivery
ffmpeg-converts to .ogg. _synthesize_to_tmp ignored the envelope and returned
its own mkstemp path, so _drain's playback gate was false for every such
provider: voice mode was silent, nothing was logged, and the real audio files
leaked into TMPDIR. Prefer the first reported artifact that exists and is
non-empty, falling back to the requested path.

Fixes #115029
This commit is contained in:
liuhao1024
2026-09-18 19:32:57 +08:00
committed by Teknium
parent 8c9b674622
commit ce963a46b1
2 changed files with 90 additions and 2 deletions

View File

@@ -7,6 +7,7 @@ the chunked-streamer playback path, and the universal per-sentence sync fallback
"""
import os
import json
import queue
import sys
import tempfile
@@ -1193,3 +1194,66 @@ def test_speaker_output_stream_opens_at_rate_learned_from_first_chunk(monkeypatc
assert done.is_set()
assert [c.kwargs["samplerate"] for c in sd.OutputStream.call_args_list] == [44100]
assert out.write.call_count == 2
def test_sync_pipeline_plays_the_artifact_the_tool_reported(monkeypatch, tmp_path):
"""A provider whose artifact lands off the requested path (command ``format`` suffix
rewrite, or voice-compatible ffmpeg conversion) must still play: follow the reported
``file_path``/``file_paths`` instead of gating on the requested path (#115029)."""
from tools import tts_tool
from tools.tts_tool_speaker import stream_tts_to_speaker
ogg = tmp_path / "sentence.ogg"
ogg.write_bytes(b"x" * 32)
def fake_synth(text, output_path):
# The requested .mp3 stays a zero-byte mkstemp file; only the reported artifact is real.
ogg_str = str(ogg)
return json.dumps({
"success": True,
"file_path": ogg_str,
"file_paths": [ogg_str],
})
played = []
fake_vm = MagicMock()
fake_vm.play_audio_file.side_effect = played.append
monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth)
monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm)
q = _drain_queue(["Hello there. "])
with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None):
stream_tts_to_speaker(q, threading.Event(), threading.Event())
assert played == [str(ogg)], (
"sentence dropped: playback ignored the reported artifact"
)
def test_sync_pipeline_falls_back_to_requested_path_when_reported_missing(monkeypatch):
"""A tool envelope that reports nothing usable (None, non-JSON, missing files) keeps the
legacy behavior: play the requested path when the tool wrote it there."""
from tools import tts_tool
from tools.tts_tool_speaker import stream_tts_to_speaker
def fake_synth(text, output_path):
with open(output_path, "wb") as fh:
fh.write(b"x" * 100)
return json.dumps({"success": False, "error": "shape without paths"})
played = []
fake_vm = MagicMock()
def _record(path):
played.append((path, os.path.getsize(path)))
fake_vm.play_audio_file.side_effect = _record
monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth)
monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm)
q = _drain_queue(["Hello there. "])
with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None):
stream_tts_to_speaker(q, threading.Event(), threading.Event())
# The temp file is unlinked after playback, so capture its size at play time.
assert len(played) == 1 and played[0][1] > 0, (
"requested-path fallback no longer plays"
)

View File

@@ -12,6 +12,7 @@ from __future__ import annotations
import contextlib
import itertools
import json
import logging
import os
import platform
@@ -73,6 +74,29 @@ def _drain_chunks(chunk_queue: "queue.Queue[Optional[bytes]]") -> List[bytes]:
return list(iter(chunk_queue.get, None))
def _first_written_artifact(raw: object, requested: str) -> str:
"""The audio path the TTS tool actually wrote, falling back to the requested one.
``text_to_speech_tool`` reports its artifacts (``file_path``/``file_paths``) in a JSON
envelope, and they often land beside — not at — the requested path: a command provider's
declared ``format`` rewrites the suffix, and voice-compatible delivery ffmpeg-converts to
``.ogg``. Gating playback on the requested path alone drops every such sentence silently,
so prefer the first reported artifact that actually exists and is non-empty."""
candidates: List[str] = []
try:
payload = json.loads(raw) if isinstance(raw, str) else (raw or {})
if isinstance(payload, dict):
reported = payload.get("file_paths") or ([payload["file_path"]] if payload.get("file_path") else [])
if isinstance(reported, list):
candidates = [p for p in reported if isinstance(p, str)]
except (ValueError, TypeError):
pass
for path in candidates:
if os.path.isfile(path) and os.path.getsize(path) > 0:
return path
return requested
class _SyncSentencePipeline:
"""Overlap per-sentence synthesis with playback for non-streaming providers.
@@ -106,8 +130,8 @@ class _SyncSentencePipeline:
try:
fd, tmp_path = tempfile.mkstemp(suffix=".mp3")
os.close(fd)
_origin().text_to_speech_tool(text=cleaned, output_path=tmp_path)
return tmp_path
raw = _origin().text_to_speech_tool(text=cleaned, output_path=tmp_path)
return _first_written_artifact(raw, tmp_path)
except Exception as exc:
logger.warning("Sync per-sentence TTS synthesis failed: %s", exc)
_unlink_quietly(tmp_path)