fix(agent): guard against uncompressed session overflow when compression is disabled (#89297)

When compression is explicitly disabled (compression.enabled: false), conversations can grow past the model's context window across hundreds of messages (e.g., 824 messages / 460K+ tokens in #89297). Serializing massive JSON payloads repeatedly under memory-constrained environments leads to swap thrashing (STAT=U) and unhandled provider errors.

Add a pre-flight uncompressed context overflow guardrail in build_turn_context and a deduped _warn_uncompressed_context_overflow method on AIAgent to alert users to run /compact or enable compression before unmanageable payloads freeze the process.
This commit is contained in:
joaomarcos
2026-08-18 17:18:21 -03:00
committed by kshitij
parent fc9b4186a0
commit db5d5dffea
3 changed files with 134 additions and 0 deletions

View File

@@ -1163,6 +1163,49 @@ def build_turn_context(
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
agent._mute_post_response = False
elif not agent.compression_enabled:
# Uncompressed session guard (#89297): when compression is explicitly
# disabled, sessions can grow past the model's context window across
# hundreds of messages without compression to shrink them.
# Run a cheap character pre-check before computing rough tokens.
_raw_chars = sum(
len(m.get("content") or "") for m in messages if isinstance(m, dict)
)
if _raw_chars > 20_000:
_uncompressed_tokens = estimate_request_tokens_rough(
messages,
system_prompt=active_system_prompt or "",
tools=agent.tools or None,
)
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if not isinstance(_ctx_len, int) or _ctx_len <= 0:
try:
from agent.model_metadata import get_model_context_length
_ctx_len = get_model_context_length(
agent.model,
getattr(agent, "base_url", "") or "",
provider=getattr(agent, "provider", "") or "",
)
except Exception:
_ctx_len = None
if _ctx_len and _uncompressed_tokens > _ctx_len:
_warn_fn = getattr(
agent, "_warn_uncompressed_context_overflow", None
)
if callable(_warn_fn):
_warn_fn(_uncompressed_tokens, _ctx_len)
else:
_emit_w = getattr(agent, "_emit_warning", None)
if callable(_emit_w):
_emit_w(
f"⚠️ Session context (~{_uncompressed_tokens:,} tokens) exceeds the "
f"model context window (~{_ctx_len:,} tokens) with compression disabled "
f"(compression.enabled: false). Use /compact to compress history or "
f"enable compression in config.yaml."
)
if _preflight_compressed:
# Compression rebuilt the list (tail messages are fresh compaction

View File

@@ -1042,6 +1042,26 @@ class AIAgent:
)
)
def _warn_uncompressed_context_overflow(
self, preflight_tokens: int, context_length: int
) -> None:
"""Surface a deduped warning when uncompressed context exceeds model limit.
When compression is explicitly disabled (compression.enabled: false), long
sessions can grow past the model context window with no compression to shrink
them (#89297). Surface an actionable warning so the user knows to run /compact
or enable compression.
"""
_warn_key = ("uncompressed_ctx_overflow", context_length)
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
self._last_ctx_overflow_warn = _warn_key
self._emit_warning(
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
f"context window (~{context_length:,} tokens) with compression disabled "
f"(compression.enabled: false). Use /compact to compress history or "
f"enable compression in config.yaml."
)
def _clear_context_overflow_warn(self) -> None:
"""Reset the dedup state for the blocked-overflow warning.

View File

@@ -0,0 +1,71 @@
"""Unit tests for uncompressed context overflow guardrail (Issue #89297)."""
from __future__ import annotations
import types
from unittest.mock import MagicMock, patch
import pytest
from agent.turn_context import TurnContext, build_turn_context
from tests.agent.test_turn_context import _FakeAgent, _build
class _FakeUncompressedAgent(_FakeAgent):
"""Agent stub with compression disabled (compression.enabled: False)."""
def __init__(self, model="deepseek-v4-flash", context_length=10_000):
super().__init__()
self.model = model
self.provider = "deepseek"
self.compression_enabled = False
self.context_compressor = types.SimpleNamespace(
protect_first_n=2,
protect_last_n=2,
context_length=context_length,
threshold_tokens=int(context_length * 0.75),
last_prompt_tokens=-1,
)
def _warn_uncompressed_context_overflow(self, preflight_tokens: int, context_length: int) -> None:
_warn_key = ("uncompressed_ctx_overflow", context_length)
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
self._last_ctx_overflow_warn = _warn_key
msg = (
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
f"context window (~{context_length:,} tokens) with compression disabled "
f"(compression.enabled: false). Use /compact to compress history or "
f"enable compression in config.yaml."
)
self._emit_warning(msg)
def test_uncompressed_session_within_limits_emits_no_warning():
agent = _FakeUncompressedAgent(context_length=128_000)
history = [
{"role": "user", "content": "hello"},
{"role": "assistant", "content": "hi there"},
]
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
agent._emit_warning.assert_not_called()
def test_uncompressed_session_exceeding_context_limit_warns():
# Model context length is 10,000 tokens (~40,000 chars)
agent = _FakeUncompressedAgent(context_length=10_000)
# Construct an oversized uncompressed history of ~15,000 tokens (>60,000 chars)
large_turn = "Large context content " * 500 # ~2,500 tokens
history = []
for i in range(10):
history.append({"role": "user", "content": f"Turn {i}: {large_turn}"})
history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"})
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
agent._emit_warning.assert_called_once()
warning_msg = agent._emit_warning.call_args[0][0]
assert "exceeds the model context window" in warning_msg
assert "compression.enabled: false" in warning_msg
assert "10,000 tokens" in warning_msg