fix(agent): honour an explicit provider refusal field (from #118406)
OpenAI-style structured-output refusals arrive in `choices[0].message.refusal` with `content` empty or filler, so the prose detector never sees the refusal and the filler would be committed as the compaction checkpoint. Read that field (str, or the dict shapes some proxies emit) in both the batch (`_call_summary_llm`) and micro (`_micro_summarize_one`) paths and route it through the same "refusal content" failure so it can never become `_previous_summary`. Also widen the prose opener with #118406's extra shapes — `we` as subject, `refuse to`, `could not` / `couldn't`, `apologise`, and a bare `I'm unable` / `I am not able` opener. #118406's `could(?:not|['’]t)` missed "couldn't" (it only matched "couldnot"/"could't"); spelled out here as `could\s*not|couldn['’]t`. Co-authored-by: Mohamad Kanso <91088196+MohamadKanso@users.noreply.github.com>
This commit is contained in:
@@ -165,9 +165,11 @@ _TRUNCATED_SUMMARY_MARKER = "finish_reason=length"
|
||||
# whole response begins with one of these phrases and refers to the requested
|
||||
# summary/checkpoint.
|
||||
_SUMMARY_REFUSAL_PREFIX_RE = re.compile(
|
||||
r"^\s*(?:(?:sorry|i(?:['’]m| am)\s+sorry|i\s+apologize|as\s+an\s+ai)"
|
||||
r"\s*[,;:]?\s*(?:but\s+)?)?i\s+"
|
||||
r"(?:can(?:not|['’]t)|won['’]t|will not|must decline|am unable to|am not able to)\b",
|
||||
r"^\s*(?:(?:sorry|i(?:['’]m| am)\s+sorry|i\s+apologi[sz]e|as\s+an\s+ai)"
|
||||
r"\s*[,;:]?\s*(?:but\s+)?)?(?:i|we)\s+"
|
||||
r"(?:can(?:not|['’]t)|could\s*not|couldn['’]t|won['’]t|will\s+not|must\s+decline|"
|
||||
r"refuse\s+to|am\s+unable\s+to|am\s+not\s+able\s+to)\b"
|
||||
r"|^\s*(?:i['’]?m|i\s+am)\s+(?:unable|not\s+able)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
@@ -182,6 +184,23 @@ def _is_summary_refusal(content: str) -> bool:
|
||||
return any(term in normalized[:400].casefold() for term in ("summary", "summarize", "checkpoint"))
|
||||
|
||||
|
||||
def _response_refusal_text(response: Any) -> str:
|
||||
"""Explicit provider ``choices[0].message.refusal`` (str, or dict with message/reason/text); ``""`` when absent.
|
||||
|
||||
OpenAI-style structured-output refusals put the refusal here and leave ``content`` as filler or
|
||||
empty, so the prose detector never sees it.
|
||||
"""
|
||||
choices = (response.get("choices") if isinstance(response, dict) else getattr(response, "choices", None)) or []
|
||||
if not choices:
|
||||
return ""
|
||||
first = choices[0]
|
||||
message = first.get("message") if isinstance(first, dict) else getattr(first, "message", None)
|
||||
refusal = message.get("refusal") if isinstance(message, dict) else getattr(message, "refusal", None)
|
||||
if isinstance(refusal, dict):
|
||||
refusal = refusal.get("message") or refusal.get("reason") or refusal.get("text")
|
||||
return refusal.strip() if isinstance(refusal, str) else ""
|
||||
|
||||
|
||||
def _is_summary_access_or_quota_error(exc: Exception) -> bool:
|
||||
"""Return True for non-retryable summary auth, permission, or quota errors."""
|
||||
|
||||
@@ -3608,7 +3627,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
|
||||
# error, rather than replacing real context with an empty summary.
|
||||
if not content.strip():
|
||||
raise RuntimeError(f"Context compression LLM returned empty content {where}")
|
||||
if _is_summary_refusal(content):
|
||||
if _response_refusal_text(response) or _is_summary_refusal(content):
|
||||
# Treat a refusal as unusable content. This deliberately reuses the
|
||||
# established fallback/cooldown/abort path for an empty body, so it
|
||||
# can never be committed as `_previous_summary`.
|
||||
|
||||
@@ -154,7 +154,7 @@ class MicroCompactionMixin:
|
||||
if not content:
|
||||
logger.info("micro-summarization returned empty content")
|
||||
return None
|
||||
if _cc()._is_summary_refusal(content):
|
||||
if _cc()._response_refusal_text(response) or _cc()._is_summary_refusal(content):
|
||||
logger.warning("micro-summarization returned refusal content — discarding unusable summary")
|
||||
return None
|
||||
return content
|
||||
|
||||
@@ -196,6 +196,27 @@ class TestGenerateSummaryTruncationGuard:
|
||||
assert c._last_compress_aborted is True
|
||||
assert c._previous_summary is None or refusal not in (c._previous_summary or "")
|
||||
|
||||
def test_provider_refusal_field_is_rejected_even_with_summary_shaped_content(self):
|
||||
"""An explicit ``message.refusal`` wins over plausible-looking content (#118406)."""
|
||||
with patch("agent.context_compressor.get_model_context_length", return_value=100000):
|
||||
c = ContextCompressor(
|
||||
model="test", quiet_mode=True,
|
||||
protect_first_n=2, protect_last_n=2,
|
||||
abort_on_summary_failure=False,
|
||||
)
|
||||
msgs = _msgs()
|
||||
filler = "## Goal\nContinue the task.\n\n## Completed Actions\n1. Nothing yet."
|
||||
response = _mock_response(filler, "stop")
|
||||
response.choices[0].message.refusal = "policy refusal"
|
||||
with patch("agent.context_compressor.call_llm", return_value=response):
|
||||
result = c.compress(msgs, current_tokens=999999, force=True)
|
||||
|
||||
assert result == msgs
|
||||
assert c._last_summary_empty_content_failure is True
|
||||
assert c._last_compress_aborted is True
|
||||
assert "refusal content" in (c._last_summary_error or "")
|
||||
assert c._previous_summary is None
|
||||
|
||||
def test_missing_finish_reason_still_succeeds(self):
|
||||
"""Providers that omit finish_reason entirely must not be rejected."""
|
||||
with patch("agent.context_compressor.get_model_context_length", return_value=100000):
|
||||
|
||||
Reference in New Issue
Block a user