fd3565deec removed every user-facing output-cap control, but the unparseable
output-cap failure in agent/turn_overflow.py::_recover_context_length still told users
to "Lower model.max_tokens in config.yaml" — a key Hermes no longer reads. The notice
now explains what happened (the provider's cap was exceeded and the error did not
state the limit) and where the cap actually lives (the endpoint's default max output
tokens on the server or proxy) without naming a config key that does not exist.
Tests: one parametrized invariant for the Azure and SGLang ceiling wordings plus the
SGLang input-alone-over-window control (red on base).
506 lines
26 KiB
Python
506 lines
26 KiB
Python
"""Overflow recovery for the conversation turn loop: 413 payload-too-large and
|
|
context-length errors after ``classify_api_error``.
|
|
|
|
Each path either compresses and signals a restart, defers softly (compression lock /
|
|
transient block), or ends the turn with a typed result. Nothing here imports
|
|
``agent.conversation_loop`` at module level (cycle); loop-internal helpers and the token
|
|
estimators that tests patch on the loop module are imported lazily inside the handlers.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
import time
|
|
from dataclasses import dataclass
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from agent.conversation_compression import (
|
|
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE, COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE,
|
|
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE, compression_blocked_transiently,
|
|
compression_skipped_due_to_lock, context_compression_timed_out,
|
|
)
|
|
from agent.error_classifier import FailoverReason
|
|
from agent.message_sanitization import serialized_messages_bytes
|
|
from agent.model_metadata import (
|
|
get_context_length_from_provider_error, is_output_cap_error,
|
|
parse_available_output_tokens_from_error,
|
|
)
|
|
from agent.turn_failure_copy import site_copy, stamp_failure
|
|
from agent.turn_retry_state import TurnRetryState
|
|
from utils import base_url_host_matches
|
|
|
|
logger = logging.getLogger("agent.conversation_loop")
|
|
|
|
_RETRY_HINT = " 💡 Try /new to start a fresh conversation, or /compress to retry compression."
|
|
|
|
# A provider "context exceeded" whose request (incl. the output reservation) is under this share
|
|
# of the known window is not explained by the transcript; the rough estimator is ±30%, so a real
|
|
# overflow (request ≈ window) never lands below half.
|
|
_UNEXPLAINED_REJECTION_FRACTION = 0.5
|
|
|
|
_GITHUB_MODELS_HINT = (
|
|
" 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every",
|
|
" request at ~8K tokens. Hermes' system prompt + tool schemas baseline",
|
|
" exceeds that floor, so this endpoint cannot run an agentic loop.",
|
|
" Use the `copilot` provider with a Copilot subscription token (`hermes",
|
|
" setup` → GitHub Copilot), or pick any other provider.",
|
|
)
|
|
|
|
_MINIMAX_ANTHROPIC_URLS = ("https://api.minimax.io/anthropic", "https://api.minimaxi.com/anthropic")
|
|
|
|
|
|
@dataclass
|
|
class OverflowVerdict:
|
|
"""Outcome of ``recover_from_overflow``.
|
|
|
|
``action`` is one of ``"return"`` (end the turn with ``result``), ``"break"``
|
|
(restart the API call — a ``_retry.restart_with_*`` flag is set), ``"continue"``
|
|
(retry the call immediately) or ``"fallthrough"`` (not an overflow error, or
|
|
overflow recovery declined — continue generic error handling). The remaining
|
|
fields are the loop locals the handler may have rebound."""
|
|
|
|
action: str
|
|
result: Optional[Dict[str, Any]]
|
|
messages: List[Dict[str, Any]]
|
|
active_system_prompt: Any
|
|
conversation_history: Any
|
|
approx_tokens: int
|
|
compression_attempts: int
|
|
provider_overflow_recovery_pending: bool
|
|
is_context_length_error: bool
|
|
|
|
|
|
@dataclass(kw_only=True)
|
|
class _Recovery(OverflowVerdict):
|
|
"""Working state shared by the overflow sub-handlers — the verdict itself, plus the
|
|
read-only call context; handlers mutate the loop-local fields and ``done()`` stamps
|
|
the action."""
|
|
|
|
agent: Any
|
|
api_messages: Any
|
|
system_message: Any
|
|
effective_task_id: Any
|
|
api_call_count: int
|
|
max_compression_attempts: int
|
|
action: str = "fallthrough"
|
|
result: Optional[Dict[str, Any]] = None
|
|
provider_overflow_recovery_pending: bool = False
|
|
is_context_length_error: bool = False
|
|
|
|
def done(self, action: str, result: Optional[Dict[str, Any]] = None) -> OverflowVerdict:
|
|
self.action, self.result = action, result
|
|
return self
|
|
|
|
def fail_turn(
|
|
self, final_response: str, *, notices: tuple = (), log: Optional[tuple] = None,
|
|
compression_exhausted: bool = True, reason: str = "context_overflow",
|
|
retryable: bool = False, **extra: Any,
|
|
) -> OverflowVerdict:
|
|
"""End the turn as failed/partial. ``notices`` flush the buffered retry trace
|
|
first so the user sees what compression attempts were made."""
|
|
agent = self.agent
|
|
if notices:
|
|
agent._flush_status_buffer()
|
|
for line in notices:
|
|
agent._vprint(f"{agent.log_prefix}{line}", force=True, diagnostic=True)
|
|
if log:
|
|
logger.error(*log)
|
|
agent._persist_session(self.messages, self.conversation_history)
|
|
result = stamp_failure({
|
|
"final_response": final_response,
|
|
"messages": self.messages,
|
|
"completed": False,
|
|
"api_calls": self.api_call_count,
|
|
"error": final_response,
|
|
"partial": True,
|
|
"failed": True,
|
|
}, reason, retryable)
|
|
if compression_exhausted:
|
|
# Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The
|
|
# bloated transcript remains intact while future input can move to a clean session instead of
|
|
# replaying the summarize-timeout loop.
|
|
result["compression_exhausted"] = True
|
|
result.update(extra)
|
|
return self.done("return", result)
|
|
|
|
def count_attempt(self, *, payload_too_large: bool = False) -> Optional[OverflowVerdict]:
|
|
"""Bump ``compression_attempts``; the terminal verdict once the cap is exceeded."""
|
|
self.compression_attempts += 1
|
|
cap = self.max_compression_attempts
|
|
if self.compression_attempts <= cap:
|
|
return None
|
|
if payload_too_large:
|
|
return self.fail_turn(
|
|
site_copy("payload_too_large", model=self.agent.model),
|
|
notices=(
|
|
f"❌ Max compression attempts ({cap}) reached for payload-too-large error.",
|
|
_RETRY_HINT,
|
|
),
|
|
log=("%s413 compression failed after %d attempts.", self.agent.log_prefix, cap),
|
|
)
|
|
return self.fail_turn(
|
|
site_copy("context_overflow", model=self.agent.model),
|
|
notices=(f"❌ Max compression attempts ({cap}) reached.", _RETRY_HINT),
|
|
log=("%sContext compression failed after %d attempts.", self.agent.log_prefix, cap),
|
|
)
|
|
|
|
def compress(self, request_tokens: int, *, fail_on_timeout: bool = False) -> Optional[OverflowVerdict]:
|
|
"""One compression pass with the summary-failure cooldown bypassed (the
|
|
provider proved the request doesn't fit). Returns ``None`` when history was
|
|
compressed, or a soft-defer verdict when another path holds the compression
|
|
lock or a timed guard no-oped the pass: the attempt is refunded and the turn
|
|
ends WITHOUT ``compression_exhausted`` so the gateway does not auto-reset. With
|
|
``fail_on_timeout`` a host timeout (recovery spent its wait budget with no
|
|
committed summary) ends the turn via the typed contract, since re-sending would
|
|
hit the same overflow."""
|
|
from agent.conversation_compression import conversation_history_after_compression
|
|
from agent.conversation_loop import _COMPRESSION_TIMEOUT_FINAL_RESPONSE, _compression_deferred_result
|
|
|
|
agent = self.agent
|
|
before = self.messages
|
|
self.messages, self.active_system_prompt = agent._compress_context(
|
|
before, self.system_message, approx_tokens=request_tokens,
|
|
task_id=self.effective_task_id, bypass_cooldown=True,
|
|
)
|
|
if self.messages is before:
|
|
deferred = None
|
|
if compression_skipped_due_to_lock(agent):
|
|
deferred = _compression_deferred_result(agent, self.messages, self.api_call_count)
|
|
elif compression_blocked_transiently(agent):
|
|
deferred = _compression_deferred_result(
|
|
agent, self.messages, self.api_call_count, reason="transient_block",
|
|
)
|
|
if deferred is not None:
|
|
self.compression_attempts -= 1
|
|
agent._persist_session(self.messages, self.conversation_history)
|
|
return self.done("return", deferred)
|
|
if fail_on_timeout and context_compression_timed_out(agent):
|
|
return self.fail_turn(
|
|
_COMPRESSION_TIMEOUT_FINAL_RESPONSE, turn_exit_reason="context_compression_timeout"
|
|
)
|
|
self.conversation_history = conversation_history_after_compression(
|
|
agent, self.messages, self.conversation_history
|
|
)
|
|
return None
|
|
|
|
def compress_scored_by_tokens(
|
|
self, request_tokens: int, *, fail_on_timeout: bool = False,
|
|
) -> Tuple[Optional[OverflowVerdict], bool, int]:
|
|
"""``compress`` scored in message count / tokens (context-overflow errors ARE
|
|
token-budget errors). Same-message-count compression (tool-result pruning,
|
|
in-place summarization) can shrink the request, so re-estimate rather than trust
|
|
the array length. Returns ``(deferred_verdict, shrank, new_tokens)``."""
|
|
from agent.model_metadata import estimate_messages_tokens_rough
|
|
|
|
original_len = len(self.messages)
|
|
original_tokens = estimate_messages_tokens_rough(self.messages)
|
|
deferred = self.compress(request_tokens, fail_on_timeout=fail_on_timeout)
|
|
if deferred is not None:
|
|
return deferred, False, original_tokens
|
|
messages = self.messages
|
|
# Re-measure after compression. Same-message-count compression (tool-result pruning, in-place
|
|
# summarization) can materially reduce request size without reducing the message array (#39550), and
|
|
# — the image-dominated case — compaction's historical-media aging (#97160) can free megabytes of
|
|
# base64 that the token estimate never counted. Bytes are the yardstick for a 413; tokens are kept
|
|
# only for status display.
|
|
# Re-estimate tokens after compression. Same-message-count compression (tool-result pruning,
|
|
# in-place summarization) can materially reduce request size without reducing the message array.
|
|
# (#39550)
|
|
new_tokens = estimate_messages_tokens_rough(messages)
|
|
shrank_tokens = new_tokens > 0 and new_tokens < original_tokens * 0.95
|
|
if len(messages) < original_len:
|
|
self.agent._buffer_diagnostic_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
|
|
elif shrank_tokens:
|
|
self.agent._buffer_diagnostic_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
|
|
return None, len(messages) < original_len or shrank_tokens, new_tokens
|
|
|
|
def request_tokens(self) -> int:
|
|
"""Overhead-aware request size (msgs + tools + system) so LCM forced-overflow
|
|
recovery arms on the TRUE request, not the tool-blind message count."""
|
|
from agent.model_metadata import estimate_request_tokens_rough
|
|
|
|
return estimate_request_tokens_rough(self.api_messages, tools=self.agent.tools or None)
|
|
|
|
|
|
def _recover_payload_too_large(st: _Recovery, _retry: TurnRetryState) -> OverflowVerdict:
|
|
"""413: compress and retry. A 413 is a BYTE-size error, so progress is scored in
|
|
payload bytes — never the token estimate, which is deliberately byte-blind to images
|
|
and wedged sessions on "no progress"."""
|
|
from agent.model_metadata import estimate_messages_tokens_rough
|
|
|
|
agent = st.agent
|
|
exhausted = st.count_attempt(payload_too_large=True)
|
|
if exhausted is not None:
|
|
return exhausted
|
|
agent._buffer_diagnostic_status(
|
|
f"⚠️ Request payload too large (413) — compression attempt "
|
|
f"{st.compression_attempts}/{st.max_compression_attempts}..."
|
|
)
|
|
|
|
messages = st.messages
|
|
original_len = len(messages)
|
|
# A 413 is a BYTE-size error, so this branch scores progress in BYTES of the serialized messages payload
|
|
# — exact and free — never the token estimate. The estimator prices every image at a flat per-image
|
|
# token cost (see estimate_messages_tokens_rough) so screenshots don't trigger premature compaction;
|
|
# that deliberate byte-blindness means compaction can free megabytes of base64 (real case: two vision
|
|
# results = 96.6% of the request body but ~3.7% of the estimate) while the token delta stays under any
|
|
# threshold. Token-scored progress here burned all attempts on "no progress" and wedged the session
|
|
# permanently. (#88960 / #47339)
|
|
original_bytes = serialized_messages_bytes(messages)
|
|
deferred = st.compress(st.request_tokens())
|
|
if deferred is not None:
|
|
return deferred
|
|
|
|
# Re-measure: same-count compression and media aging can shrink the request
|
|
# without shrinking the array. Tokens only for status display.
|
|
messages = st.messages
|
|
st.approx_tokens = estimate_messages_tokens_rough(messages)
|
|
new_bytes = serialized_messages_bytes(messages)
|
|
if len(messages) < original_len or (new_bytes > 0 and new_bytes < original_bytes * 0.95):
|
|
if len(messages) < original_len:
|
|
agent._buffer_diagnostic_status(COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=original_len, after=len(messages)))
|
|
else:
|
|
agent._buffer_diagnostic_status(
|
|
f"🗜️ Compressed {original_bytes:,} → {new_bytes:,} " f"payload bytes, retrying..."
|
|
)
|
|
time.sleep(2) # Brief pause between compression retries
|
|
_retry.restart_with_compressed_messages = True
|
|
return st.done("break")
|
|
|
|
if agent._try_strip_image_parts_from_tool_messages(st.api_messages, remember_model=False):
|
|
agent._buffer_diagnostic_status(
|
|
"📐 Compression could not reduce the request further — "
|
|
"removed retained vision payloads and retrying..."
|
|
)
|
|
return st.done("continue")
|
|
|
|
return st.fail_turn(
|
|
site_copy("payload_too_large", model=agent.model),
|
|
notices=("❌ Payload too large and cannot compress further.", _RETRY_HINT),
|
|
log=("%s413 payload too large. Cannot compress further.", agent.log_prefix),
|
|
)
|
|
|
|
|
|
def _clamp_output_cap(st: _Recovery, _retry: TurnRetryState, available_out: int, old_ctx: int) -> OverflowVerdict:
|
|
"""Output-cap error ("max_tokens too large": input fits but input + max_tokens >
|
|
window). The provider's available_tokens is the authoritative bound; also estimate
|
|
the real request shape (API-only content) and use the smaller minus a margin."""
|
|
agent = st.agent
|
|
request_input_estimate = st.request_tokens()
|
|
local_available_out = old_ctx - request_input_estimate
|
|
if local_available_out > 0:
|
|
safe_out = max(1, min(available_out, local_available_out) - 64)
|
|
else:
|
|
# Local estimate can overshoot; fall back to the provider-reported budget.
|
|
safe_out = max(1, available_out - 64)
|
|
agent._ephemeral_max_output_tokens = safe_out
|
|
agent._buffer_vprint(
|
|
f"⚠️ Output cap too large for current prompt — retrying with max_tokens={safe_out:,} "
|
|
f"(provider_available={available_out:,}, estimated_request_tokens={request_input_estimate:,}; "
|
|
f"context_length unchanged at {old_ctx:,})"
|
|
)
|
|
# Still count against compression_attempts so a recurring error can't loop forever.
|
|
exhausted = st.count_attempt()
|
|
if exhausted is not None:
|
|
return exhausted
|
|
# Also compress history so the retry doesn't spin on max_tokens alone; dropping the
|
|
# middle window makes the total fit. Compression must never turn an output-cap error
|
|
# fatal — on error, fall through and retry on max_tokens alone.
|
|
try:
|
|
deferred, _shrank, _new_tokens = st.compress_scored_by_tokens(request_input_estimate)
|
|
if deferred is not None:
|
|
return deferred
|
|
except Exception:
|
|
logger.warning(
|
|
"%sOutput-cap compression hit an error; retrying on max_tokens only.", agent.log_prefix
|
|
)
|
|
_retry.restart_with_compressed_messages = True
|
|
return st.done("break")
|
|
|
|
|
|
def _adopt_provider_context_limit(st: _Recovery, error_msg: str, old_ctx: int) -> Optional[int]:
|
|
"""Shrink context_length only when the provider reports the real limit; else keep
|
|
the window and compress. Guessed probe tiers can turn a configured 1M window into
|
|
256K/128K/64K. Returns the provider-reported limit, or ``None``."""
|
|
from agent.model_metadata import save_provider_context_length
|
|
|
|
agent = st.agent
|
|
compressor = agent.context_compressor
|
|
new_ctx = get_context_length_from_provider_error(error_msg, old_ctx)
|
|
if new_ctx is not None:
|
|
agent._buffer_vprint(f"Context limit detected from API: {new_ctx:,} tokens (was {old_ctx:,})")
|
|
compressor.update_model(
|
|
model=agent.model, context_length=new_ctx, base_url=agent.base_url,
|
|
api_key=getattr(agent, "api_key", ""), provider=agent.provider, api_mode=agent.api_mode,
|
|
)
|
|
# Persist the provider-reported limit BEFORE compression/retry: rate limit,
|
|
# missing usage, or restart must not lose confirmed metadata. Probe flags
|
|
# remain a fallback if this write fails.
|
|
save_provider_context_length(agent.model, agent.base_url, new_ctx, agent.provider)
|
|
# Probe flags only on the built-in compressor (plugin engines manage their
|
|
# own); provider-sourced value, so safe to cache.
|
|
if hasattr(compressor, "_context_probed"):
|
|
compressor._context_probed = True
|
|
compressor._context_probe_persistable = True
|
|
agent._buffer_vprint(f"⚠️ Context length exceeded — using provider limit: {old_ctx:,} → {new_ctx:,} tokens")
|
|
return new_ctx
|
|
|
|
is_minimax_provider = (
|
|
(getattr(agent, "provider", "") or "").lower() in {"minimax", "minimax-cn"}
|
|
or (getattr(agent, "base_url", "") or "").rstrip("/").lower().startswith(_MINIMAX_ANTHROPIC_URLS)
|
|
)
|
|
if is_minimax_provider and "context window exceeds limit (" in error_msg:
|
|
agent._buffer_vprint(
|
|
f"Provider reported overflow amount only; "
|
|
f"keeping context_length at {old_ctx:,} tokens."
|
|
)
|
|
else:
|
|
agent._buffer_vprint(
|
|
f"⚠️ Context length exceeded, but provider did not report a max context length; "
|
|
f"keeping context_length at {old_ctx:,} tokens."
|
|
)
|
|
return None
|
|
|
|
|
|
def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: str) -> OverflowVerdict:
|
|
"""Context-length error. Two shapes: "prompt too long" = INPUT overflows the window
|
|
(shrink context_length + compress); "max_tokens too large" = input fits but
|
|
input + max_tokens > window (shrink the OUTPUT cap only)."""
|
|
agent = st.agent
|
|
old_ctx = agent.context_compressor.context_length
|
|
|
|
available_out = parse_available_output_tokens_from_error(error_msg)
|
|
if available_out is not None:
|
|
return _clamp_output_cap(st, _retry, available_out, old_ctx)
|
|
|
|
# Output-cap error with unparseable budget: compression can't help (input already
|
|
# fits) and would death-loop on the same 400. Fail fast.
|
|
if is_output_cap_error(error_msg):
|
|
return st.fail_turn(
|
|
"The requested output length exceeds the provider's output cap for this model, "
|
|
"and the error did not state the allowed limit.",
|
|
notices=(
|
|
"❌ The provider rejected the request because the requested output length exceeds its "
|
|
"output cap for this model, and the error did not state the allowed limit.",
|
|
" 💡 Hermes has no user setting for the output cap — check the endpoint's default max output "
|
|
"(completion) tokens for this model on the server or proxy. "
|
|
"(This is an output-cap error, not a context overflow — compression cannot fix it.)",
|
|
),
|
|
log=(
|
|
f"{agent.log_prefix}Output-cap error not routed into compression "
|
|
f"(max_tokens over provider cap): {error_msg[:200]}",
|
|
),
|
|
compression_exhausted=False,
|
|
)
|
|
|
|
new_ctx = _adopt_provider_context_limit(st, error_msg, old_ctx)
|
|
|
|
# A rejection the transcript cannot explain (#114644): the request sits far below the window
|
|
# Hermes knows for this model (after adopting any limit the server reported), so compressing
|
|
# would destroy history for nothing. Single-slot local servers reject like this while ANOTHER
|
|
# request — a background review from an earlier session — holds their context. Name that,
|
|
# keep the turn retryable and transient: no "conversation too long", no gateway auto-reset.
|
|
# Only when the server quoted NO measurement of its own: "prompt is too long: 233153 tokens
|
|
# > 200000" is the server's count and beats the local estimate.
|
|
window = agent.context_compressor.context_length
|
|
request_tokens = st.request_tokens() + max(0, int(getattr(agent, "max_tokens", 0) or 0))
|
|
if (
|
|
not re.search(r"\d{4,}", error_msg)
|
|
and isinstance(window, int) and window > 0
|
|
and request_tokens < window * _UNEXPLAINED_REJECTION_FRACTION
|
|
):
|
|
return st.fail_turn(
|
|
site_copy("server_context_rejection", model=agent.model, tokens=request_tokens, window=window),
|
|
notices=(
|
|
f"❌ The server rejected the request as too large, but it is only ~{request_tokens:,} "
|
|
f"tokens against a {window:,}-token window — not compressing.",
|
|
" 💡 Wait a moment and /retry; another request on the same server (e.g. a background "
|
|
"review) may have been holding its context.",
|
|
),
|
|
log=(
|
|
"%sProvider context rejection not explained by request size (~%s of %s tokens); "
|
|
"skipping compression: %s", agent.log_prefix, f"{request_tokens:,}", f"{window:,}", error_msg[:200],
|
|
),
|
|
compression_exhausted=False, reason=FailoverReason.server_error.value, retryable=True,
|
|
)
|
|
|
|
exhausted = st.count_attempt()
|
|
if exhausted is not None:
|
|
return exhausted
|
|
agent._buffer_diagnostic_status(COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE.format(
|
|
tokens=st.approx_tokens, attempt=st.compression_attempts, cap=st.max_compression_attempts,
|
|
))
|
|
|
|
deferred, shrank, new_tokens = st.compress_scored_by_tokens(st.request_tokens(), fail_on_timeout=True)
|
|
if deferred is not None:
|
|
return deferred
|
|
st.approx_tokens = new_tokens
|
|
if shrank or (new_ctx and new_ctx < old_ctx):
|
|
time.sleep(2) # Brief pause between compression retries
|
|
# Rebuild the full request and force normal preflight to honor it; message
|
|
# count alone doesn't prove system/tool-inclusive pressure fell.
|
|
st.provider_overflow_recovery_pending = True
|
|
_retry.restart_with_compressed_messages = True
|
|
return st.done("break")
|
|
|
|
# Can't compress further and already at minimum tier.
|
|
return st.fail_turn(
|
|
site_copy("context_overflow", model=agent.model),
|
|
notices=(
|
|
f"❌ The conversation is too long for the model ({new_tokens:,} tokens) and cannot be shrunk further.",
|
|
_RETRY_HINT,
|
|
),
|
|
log=("%sContext length exceeded: %s tokens. Cannot compress further.", agent.log_prefix, f"{new_tokens:,}"),
|
|
)
|
|
|
|
|
|
def recover_from_overflow(
|
|
agent: Any, api_error: Exception, classified: Any, _retry: TurnRetryState, *,
|
|
status_code: Optional[int], error_msg: str, wrapped_output_cap_budget: Optional[int],
|
|
messages: List[Dict[str, Any]], api_messages: Any, system_message: Any,
|
|
active_system_prompt: Any, conversation_history: Any, approx_tokens: int,
|
|
compression_attempts: int, max_compression_attempts: int, api_call_count: int,
|
|
effective_task_id: Any,
|
|
) -> OverflowVerdict:
|
|
"""413 payload-too-large and context-length recovery (compress + retry, output-cap
|
|
clamp, provider-reported context limit, GitHub Models free-tier hint). Order is
|
|
load-bearing: 413 is checked BEFORE the generic 4xx handler, and context-length
|
|
errors (incl. relay-wrapped output-cap 429s) BEFORE non-retryable client errors.
|
|
Compression progress is scored in payload BYTES for 413 (never the byte-blind token
|
|
estimate) and in tokens/message count for context overflow."""
|
|
st = _Recovery(
|
|
agent=agent, api_messages=api_messages, system_message=system_message,
|
|
effective_task_id=effective_task_id, api_call_count=api_call_count,
|
|
max_compression_attempts=max_compression_attempts, messages=messages,
|
|
active_system_prompt=active_system_prompt, conversation_history=conversation_history,
|
|
approx_tokens=approx_tokens, compression_attempts=compression_attempts,
|
|
)
|
|
|
|
# GitHub Models free tier caps requests at 8K tokens, under the system prompt +
|
|
# tool schema floor; compression can't help, so say so.
|
|
if (
|
|
status_code == 413
|
|
and isinstance(agent.base_url, str)
|
|
and base_url_host_matches(agent.base_url, "models.inference.ai.azure.com")
|
|
):
|
|
for line in _GITHUB_MODELS_HINT:
|
|
agent._vprint(f"{agent.log_prefix}{line}", force=True, diagnostic=True)
|
|
|
|
if classified.reason == FailoverReason.payload_too_large:
|
|
return _recover_payload_too_large(st, _retry)
|
|
|
|
# Relay-wrapped output-cap 429s (parsed by the caller) go to the clamp, not
|
|
# failover or generic retries. The classifier also covers 400/disconnect +
|
|
# large-session heuristics.
|
|
st.is_context_length_error = (
|
|
# Check for context-length errors BEFORE generic 4xx handler. The classifier detects context
|
|
# overflow from: explicit error messages, generic 400 + large session heuristic (#1630), and server
|
|
# disconnect + large session pattern (#2153).
|
|
classified.reason == FailoverReason.context_overflow
|
|
or wrapped_output_cap_budget is not None
|
|
)
|
|
if st.is_context_length_error:
|
|
return _recover_context_length(st, _retry, error_msg)
|
|
return st.done("fallthrough")
|