fix(agent): refusal handler reports Anthropic stop_details instead of "(no text)"

A stop_reason=refusal from Anthropic's streaming classifier arrives with an
empty body; the reason lives on the message's stop_details (category and,
when present, an explanation). handle_content_policy_refusal now uses that
explanation (or the category) as the refusal text shown to the user and
recorded in the content_policy_blocked error, and the warning line carries
native_stop_reason + stop_details so a classifier refusal is
distinguishable from a Bedrock guardrail block (both map to content_filter;
that mapping is unchanged because turn_response_check routes it here).

The main-loop Anthropic stream returns the SDK's get_final_message()
snapshot, which copies only stop_reason/stop_sequence from message_delta;
the accumulator now keeps stop_details and _call_anthropic restores it on
the snapshot, mirroring the aux-client path fixed in the previous commit.

Part of #113689. The native-stop-reason logging follows the direction of
PR #113699 by @liuhao1024, minus the refusal content block, which the
Messages API does not emit.

Co-authored-by: liuhao1024 <sunsky.lau@gmail.com>
This commit is contained in:
teknium1
2026-09-18 00:46:08 -07:00
committed by Teknium
parent c5cd426a03
commit 6ed8bea80b
4 changed files with 32 additions and 2 deletions

View File

@@ -3172,6 +3172,11 @@ class _StreamingCall(StreamingWaitMonitor):
if not self.agent._interrupt_requested and raw_stream is not None:
try:
base_final_message = raw_stream.get_final_message()
# The SDK snapshot keeps only stop_reason/stop_sequence from message_delta; the
# refusal's stop_details (category/explanation) survives only in our accumulator.
_stop_details = accumulator.finalize().get("stop_details")
if _stop_details is not None and getattr(base_final_message, "stop_details", None) is None:
base_final_message.stop_details = _stop_details
except AssertionError:
if not saw_stream_event:
raise EmptyStreamError(

View File

@@ -583,7 +583,7 @@ class AnthropicStreamAccumulator:
def _on_message_delta(self, payload: dict[str, Any]) -> None:
delta = payload.get("delta")
if isinstance(delta, dict):
self._message.update({k: delta[k] for k in ("stop_reason", "stop_sequence") if k in delta})
self._message.update({k: delta[k] for k in ("stop_reason", "stop_sequence", "stop_details") if k in delta})
if "usage" in payload:
usage, current_usage = payload["usage"], self._message.get("usage")
if isinstance(current_usage, dict) and isinstance(usage, dict):

View File

@@ -584,6 +584,13 @@ def handle_content_policy_refusal(
_refusal_text = (getattr(_refusal_result, "content", None) or "").strip()
if not _refusal_text:
_refusal_text = (agent._extract_reasoning(_refusal_result) or "").strip()
# Anthropic stop_reason=refusal carries its reason on stop_details (category + optional explanation),
# not in a content block — without it a classifier halt reads as "(no text)" (#113689).
_stop_details = (getattr(_refusal_result, "provider_data", None) or {}).get("stop_details")
if not _refusal_text and isinstance(_stop_details, dict):
_refusal_text = str(_stop_details.get("explanation") or "").strip() or (
f"provider refusal category: {_stop_details['category']}" if _stop_details.get("category") else ""
)
agent._invoke_api_request_error_hook(
task_id=effective_task_id, turn_id=turn_id, api_request_id=api_request_id,
@@ -603,9 +610,13 @@ def handle_content_policy_refusal(
agent._flush_status_buffer()
_refusal_log = _refusal_text[:500] + "..." if len(_refusal_text) > 500 else _refusal_text
# native_stop_reason tells an Anthropic classifier refusal (``refusal``) from a Bedrock guardrail
# block (``end_turn``); both arrive here as content_filter.
logger.warning(
"%sModel declined to respond (finish_reason=content_filter). model=%s provider=%s refusal=%s",
"%sModel declined to respond (finish_reason=content_filter). model=%s provider=%s "
"native_stop_reason=%s stop_details=%s refusal=%s",
agent.log_prefix, agent.model, agent.provider,
getattr(response, "stop_reason", None) or "n/a", _stop_details or "n/a",
_refusal_log or "(no text)",
)
agent._emit_diagnostic_status("⚠️ The model declined to respond to this request (safety refusal).")

View File

@@ -132,6 +132,20 @@ class TestAnthropicTransport:
assert nr.tool_calls is None or nr.tool_calls == []
assert nr.finish_reason == "stop"
def test_normalize_response_refusal_surfaces_stop_details(self, transport):
"""stop_reason=refusal maps to content_filter and carries the message's stop_details (the
SDK exposes it only as an extra field); a plain end_turn adds no stop_details key."""
refusal = SimpleNamespace(
content=[], stop_reason="refusal", usage=None, model="claude",
stop_details={"type": "refusal", "category": "general_harms", "explanation": "classifier halt"},
)
nr = transport.normalize_response(refusal)
assert nr.finish_reason == "content_filter"
assert nr.provider_data["stop_details"]["explanation"] == "classifier halt"
plain = transport.normalize_response(
SimpleNamespace(content=[SimpleNamespace(type="text", text="ok")], stop_reason="end_turn", usage=None, model="claude"))
assert "stop_details" not in (plain.provider_data or {})
def test_normalize_response_tool_calls(self, transport):
"""Test normalization of a tool-use response."""
r = SimpleNamespace(