From 0ebe70d574246cb0fdcff9d73af60b1fb03fda18 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Tue, 1 Sep 2026 15:46:06 -0700 Subject: [PATCH] fix(agent): long-context tier recovery also rechecks the rebuilt request The Anthropic long-context 429 handler restarts on row count alone, the same shape #100614 fixed in the generic overflow handler. Arm the same provider-overflow recovery flag there so the rebuilt request is measured against the reduced window before the provider is retried. The 413 (byte-scored) and output-cap (max_tokens) handlers are a different yardstick and are left as-is. --- agent/conversation_loop.py | 6 ++ tests/run_agent/test_413_compression.py | 78 +++++++++++++++++++++++++ 2 files changed, 84 insertions(+) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 26df1c362a..4469ab395e 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -5825,6 +5825,12 @@ def run_conversation( ) ) time.sleep(2) + # Same class as the generic overflow handler below: + # the provider proved the request does not fit the + # (now-reduced) window, and row count alone is not + # proof the rebuilt request does. Recheck the + # complete request before the next provider call. + _provider_overflow_recovery_pending = True _retry.restart_with_compressed_messages = True break # Fall through to normal error handling if compression diff --git a/tests/run_agent/test_413_compression.py b/tests/run_agent/test_413_compression.py index 20a386d6e7..2c99ff03a0 100644 --- a/tests/run_agent/test_413_compression.py +++ b/tests/run_agent/test_413_compression.py @@ -1051,6 +1051,84 @@ class TestPreflightCompression: assert mock_compress.call_count == 2 assert agent.client.chat.completions.create.call_count == 1 + def test_long_context_tier_recovery_rechecks_complete_request_before_retry(self, agent): + """The Anthropic long-context 429 handler is the same recovery class. + + It compacts and restarts on row count alone, exactly like the generic + overflow handler. The rebuilt request must be measured against the + (now-reduced) window before the provider is retried, so a compaction + that drops rows but stays oversized fails closed instead of being + sent again. + """ + agent.compression_enabled = True + agent.max_compression_attempts = 2 + agent.context_compressor.context_length = 1_000_000 + agent.context_compressor.threshold_tokens = 500_000 + + tier_error = Exception( + "Extra usage is required for long context requests." + ) + tier_error.status_code = 429 + agent.client.chat.completions.create.side_effect = [tier_error] + + history = [ + {"role": "user", "content": "earlier question"}, + {"role": "assistant", "content": "earlier answer"}, + ] + compress_calls = 0 + + def _request_pressure(*_args, **_kwargs): + if agent.client.chat.completions.create.call_count == 0: + return 30_000 + return 250_000 + + def _compress(_messages, *_args, **_kwargs): + nonlocal compress_calls + compress_calls += 1 + return ( + [ + {"role": "user", "content": f"summary {compress_calls}"}, + {"role": "assistant", "content": "summary acknowledged"}, + ], + "rebuilt prompt remains oversized", + ) + + def _update_model(*, context_length, **_kwargs): + agent.context_compressor.context_length = context_length + agent.context_compressor.threshold_tokens = context_length // 2 + + with ( + patch( + "agent.turn_context.estimate_request_tokens_rough", + return_value=30_000, + ), + patch( + "agent.conversation_loop._midturn_request_pressure_tokens", + side_effect=_request_pressure, + ), + patch.object( + agent.context_compressor, + "should_defer_preflight_to_real_usage", + return_value=True, + ), + patch.object( + agent.context_compressor, "update_model", side_effect=_update_model + ), + patch.object(agent, "_compress_context", side_effect=_compress) as mock_compress, + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + result = agent.run_conversation( + "continue", + conversation_history=history, + ) + + assert result["completed"] is False + assert result["compression_exhausted"] is True + assert mock_compress.call_count == 2 + assert agent.client.chat.completions.create.call_count == 1 + def test_interrupt_before_first_provider_call_restores_preflight_display_seed(self, agent): """Interrupted turns must not keep a speculative preflight display seed.