From 4252aecc2ed88dc69e9b0f60af796612f1feeada Mon Sep 17 00:00:00 2001 From: Darafei Praliaskouski Date: Sun, 16 Aug 2026 10:30:03 +0400 Subject: [PATCH] fix(agent): cap compaction threshold floor at 85% of the context window MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The MINIMUM_CONTEXT_LENGTH floor in _compute_threshold_tokens only degraded to the 85% trigger when it met or exceeded the effective window exactly (#14690). Near-minimum windows slipped through: at context_length=65536 the threshold passed through at 64,000 — 97.7% of the window, ~1.5K tokens of output room — so pre-API compaction effectively could not fire. Providers that silently truncate over-window prompts instead of rejecting them (e.g. ollama's OpenAI-compatible /v1 endpoint) never deliver the reactive context-overflow backstop either. Observed live on a 65,536-token local model: the session rode into the window ceiling and each length-continuation retry re-sent a window-filling prompt (65,120 -> 65,273 prompt tokens, 263 output tokens of room) until the turn died with "Response remained truncated after 4 continuation attempts" — every retry paying a full multi-minute prefill. Cap the floored threshold at _MIN_CTX_TRIGGER_RATIO (85%) of the effective input budget whenever the floor is the binding term. An explicit threshold_percent above 85% is user intent and stays uncapped; windows where the floor lands at/below the cap are unchanged. --- agent/context_compressor.py | 34 ++++++++++++++++++-------- tests/agent/test_context_compressor.py | 25 +++++++++++++++++++ 2 files changed, 49 insertions(+), 10 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 5c613b8618..b20199620e 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -3315,10 +3315,20 @@ class ContextCompressor(ContextEngine): equal the ENTIRE window — auto-compression can never fire because the provider rejects the request before usage reaches 100% (#14690). - When the floor would meet or exceed the context window, trigger at - ``_MIN_CTX_TRIGGER_RATIO`` (85%) of the window — high enough that a - small model uses most of its context before compacting, but below - 100% so compaction fires before the provider rejects the request. + Near-minimum windows degenerate the same way without ever tripping an + equality check: at ``context_length == 65536`` the floored threshold + used to pass through at 64,000 — 97.7% of the window, ~1.5K tokens of + output room. Providers that silently truncate over-window prompts + instead of rejecting them (e.g. ollama's OpenAI-compatible endpoint) + never deliver the reactive context-overflow backstop either, so a + session rides into the window ceiling and every length-continuation + retry re-sends a window-filling prompt for a shrinking sliver of + output. Whenever the floor is the binding term, it is therefore capped + at ``_MIN_CTX_TRIGGER_RATIO`` (85%) of the effective input budget — + high enough that a small model uses most of its context before + compacting, but low enough that compaction fires while output room + remains. An explicit ``threshold_percent`` above 85% is user intent, + not the floor, and is not capped. The provider reserves ``max_tokens`` of output space out of the same window, so the usable INPUT budget is ``context_length - max_tokens``. @@ -3334,13 +3344,17 @@ class ContextCompressor(ContextEngine): effective_window = context_length pct_value = int(effective_window * threshold_percent) floored = max(pct_value, MINIMUM_CONTEXT_LENGTH) - # If flooring pushed the threshold to/over the effective window it can - # never be reached. Trigger at 85% of the effective input budget so a - # minimum-context model rides most of its budget before compacting - # instead of wasting half. + # The floor must not consume the window's output headroom: cap it at + # 85% of the effective input budget whenever it is the binding term. + # (An explicit threshold_percent above 85% is user intent — kept.) + trigger_cap = int(effective_window * ContextCompressor._MIN_CTX_TRIGGER_RATIO) + if effective_window > 0 and floored > pct_value and floored > trigger_cap: + floored = max(pct_value, trigger_cap) + # If the percentage itself reaches the effective window it can never + # be reached — trigger at 85% of the window, below 100% so compaction + # fires before the provider rejects (or silently clips) the request. if effective_window > 0 and floored >= effective_window: - return max(1, min(int(effective_window * ContextCompressor._MIN_CTX_TRIGGER_RATIO), - effective_window - 1)) + return max(1, min(trigger_cap, effective_window - 1)) return floored def __init__( self, diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index ca99799361..2b822a10cd 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -421,6 +421,31 @@ class TestCompress: assert t < MINIMUM_CONTEXT_LENGTH assert t == 54400 # 85% of 64000 + def test_threshold_floor_capped_at_85_percent_of_window(self): + """The MINIMUM_CONTEXT_LENGTH floor must not consume the window's + output headroom. At context_length == 65,536 (a common local-model + window) the floored threshold used to pass through at 64,000 — 97.7% + of the window, ~1.5K tokens of output room — so pre-API compaction + effectively could not fire. Providers that silently truncate + over-window prompts instead of rejecting them (e.g. ollama's + OpenAI-compatible endpoint) never delivered the reactive + context-overflow backstop either: a live session rode into the window + ceiling and each length-continuation retry re-sent a window-filling + prompt (observed 65,120 -> 65,273 prompt tokens against 65,536, + leaving 263 output tokens) until the turn died with "Response + remained truncated after 4 continuation attempts". The floor is now + capped at 85% of the effective input budget whenever it is the + binding term.""" + t = ContextCompressor._compute_threshold_tokens(65_536, 0.50) + assert t == int(65_536 * 0.85) # 55,705 + # Any window where the floor lands above 85% is capped the same way. + assert ContextCompressor._compute_threshold_tokens(70_000, 0.50) == 59_500 + # Floor binding but at/under the 85% cap: unchanged. + assert ContextCompressor._compute_threshold_tokens(100_000, 0.50) == 64_000 + # An explicit threshold_percent above 85% is user intent, not the + # floor — it is not capped. + assert ContextCompressor._compute_threshold_tokens(372_000, 0.90) == 334_800 +