fix(tts): play the artifact the tool reported in the sync speaker pipeline
text_to_speech_tool reports its artifacts (file_path/file_paths) in a JSON envelope, and they often land beside — not at — the requested path: a command provider's declared format rewrites the suffix, and voice-compatible delivery ffmpeg-converts to .ogg. _synthesize_to_tmp ignored the envelope and returned its own mkstemp path, so _drain's playback gate was false for every such provider: voice mode was silent, nothing was logged, and the real audio files leaked into TMPDIR. Prefer the first reported artifact that exists and is non-empty, falling back to the requested path. Fixes #115029
This commit is contained in:
@@ -7,6 +7,7 @@ the chunked-streamer playback path, and the universal per-sentence sync fallback
|
||||
"""
|
||||
|
||||
import os
|
||||
import json
|
||||
import queue
|
||||
import sys
|
||||
import tempfile
|
||||
@@ -1193,3 +1194,66 @@ def test_speaker_output_stream_opens_at_rate_learned_from_first_chunk(monkeypatc
|
||||
assert done.is_set()
|
||||
assert [c.kwargs["samplerate"] for c in sd.OutputStream.call_args_list] == [44100]
|
||||
assert out.write.call_count == 2
|
||||
|
||||
|
||||
def test_sync_pipeline_plays_the_artifact_the_tool_reported(monkeypatch, tmp_path):
|
||||
"""A provider whose artifact lands off the requested path (command ``format`` suffix
|
||||
rewrite, or voice-compatible ffmpeg conversion) must still play: follow the reported
|
||||
``file_path``/``file_paths`` instead of gating on the requested path (#115029)."""
|
||||
from tools import tts_tool
|
||||
from tools.tts_tool_speaker import stream_tts_to_speaker
|
||||
|
||||
ogg = tmp_path / "sentence.ogg"
|
||||
ogg.write_bytes(b"x" * 32)
|
||||
|
||||
def fake_synth(text, output_path):
|
||||
# The requested .mp3 stays a zero-byte mkstemp file; only the reported artifact is real.
|
||||
ogg_str = str(ogg)
|
||||
return json.dumps({
|
||||
"success": True,
|
||||
"file_path": ogg_str,
|
||||
"file_paths": [ogg_str],
|
||||
})
|
||||
|
||||
played = []
|
||||
fake_vm = MagicMock()
|
||||
fake_vm.play_audio_file.side_effect = played.append
|
||||
monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth)
|
||||
monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm)
|
||||
|
||||
q = _drain_queue(["Hello there. "])
|
||||
with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None):
|
||||
stream_tts_to_speaker(q, threading.Event(), threading.Event())
|
||||
assert played == [str(ogg)], (
|
||||
"sentence dropped: playback ignored the reported artifact"
|
||||
)
|
||||
|
||||
|
||||
def test_sync_pipeline_falls_back_to_requested_path_when_reported_missing(monkeypatch):
|
||||
"""A tool envelope that reports nothing usable (None, non-JSON, missing files) keeps the
|
||||
legacy behavior: play the requested path when the tool wrote it there."""
|
||||
from tools import tts_tool
|
||||
from tools.tts_tool_speaker import stream_tts_to_speaker
|
||||
|
||||
def fake_synth(text, output_path):
|
||||
with open(output_path, "wb") as fh:
|
||||
fh.write(b"x" * 100)
|
||||
return json.dumps({"success": False, "error": "shape without paths"})
|
||||
|
||||
played = []
|
||||
fake_vm = MagicMock()
|
||||
|
||||
def _record(path):
|
||||
played.append((path, os.path.getsize(path)))
|
||||
|
||||
fake_vm.play_audio_file.side_effect = _record
|
||||
monkeypatch.setattr(tts_tool, "text_to_speech_tool", fake_synth)
|
||||
monkeypatch.setitem(sys.modules, "tools.voice_mode", fake_vm)
|
||||
|
||||
q = _drain_queue(["Hello there. "])
|
||||
with patch("tools.tts_streaming.resolve_streaming_provider", return_value=None):
|
||||
stream_tts_to_speaker(q, threading.Event(), threading.Event())
|
||||
# The temp file is unlinked after playback, so capture its size at play time.
|
||||
assert len(played) == 1 and played[0][1] > 0, (
|
||||
"requested-path fallback no longer plays"
|
||||
)
|
||||
|
||||
@@ -12,6 +12,7 @@ from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import itertools
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import platform
|
||||
@@ -73,6 +74,29 @@ def _drain_chunks(chunk_queue: "queue.Queue[Optional[bytes]]") -> List[bytes]:
|
||||
return list(iter(chunk_queue.get, None))
|
||||
|
||||
|
||||
def _first_written_artifact(raw: object, requested: str) -> str:
|
||||
"""The audio path the TTS tool actually wrote, falling back to the requested one.
|
||||
|
||||
``text_to_speech_tool`` reports its artifacts (``file_path``/``file_paths``) in a JSON
|
||||
envelope, and they often land beside — not at — the requested path: a command provider's
|
||||
declared ``format`` rewrites the suffix, and voice-compatible delivery ffmpeg-converts to
|
||||
``.ogg``. Gating playback on the requested path alone drops every such sentence silently,
|
||||
so prefer the first reported artifact that actually exists and is non-empty."""
|
||||
candidates: List[str] = []
|
||||
try:
|
||||
payload = json.loads(raw) if isinstance(raw, str) else (raw or {})
|
||||
if isinstance(payload, dict):
|
||||
reported = payload.get("file_paths") or ([payload["file_path"]] if payload.get("file_path") else [])
|
||||
if isinstance(reported, list):
|
||||
candidates = [p for p in reported if isinstance(p, str)]
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
for path in candidates:
|
||||
if os.path.isfile(path) and os.path.getsize(path) > 0:
|
||||
return path
|
||||
return requested
|
||||
|
||||
|
||||
class _SyncSentencePipeline:
|
||||
"""Overlap per-sentence synthesis with playback for non-streaming providers.
|
||||
|
||||
@@ -106,8 +130,8 @@ class _SyncSentencePipeline:
|
||||
try:
|
||||
fd, tmp_path = tempfile.mkstemp(suffix=".mp3")
|
||||
os.close(fd)
|
||||
_origin().text_to_speech_tool(text=cleaned, output_path=tmp_path)
|
||||
return tmp_path
|
||||
raw = _origin().text_to_speech_tool(text=cleaned, output_path=tmp_path)
|
||||
return _first_written_artifact(raw, tmp_path)
|
||||
except Exception as exc:
|
||||
logger.warning("Sync per-sentence TTS synthesis failed: %s", exc)
|
||||
_unlink_quietly(tmp_path)
|
||||
|
||||
Reference in New Issue
Block a user