fix(agent): guard against uncompressed session overflow when compression is disabled (#89297)
When compression is explicitly disabled (compression.enabled: false), conversations can grow past the model's context window across hundreds of messages (e.g., 824 messages / 460K+ tokens in #89297). Serializing massive JSON payloads repeatedly under memory-constrained environments leads to swap thrashing (STAT=U) and unhandled provider errors. Add a pre-flight uncompressed context overflow guardrail in build_turn_context and a deduped _warn_uncompressed_context_overflow method on AIAgent to alert users to run /compact or enable compression before unmanageable payloads freeze the process.
This commit is contained in:
@@ -1163,6 +1163,49 @@ def build_turn_context(
|
||||
agent._last_content_with_tools = None
|
||||
agent._last_content_tools_all_housekeeping = False
|
||||
agent._mute_post_response = False
|
||||
elif not agent.compression_enabled:
|
||||
# Uncompressed session guard (#89297): when compression is explicitly
|
||||
# disabled, sessions can grow past the model's context window across
|
||||
# hundreds of messages without compression to shrink them.
|
||||
# Run a cheap character pre-check before computing rough tokens.
|
||||
_raw_chars = sum(
|
||||
len(m.get("content") or "") for m in messages if isinstance(m, dict)
|
||||
)
|
||||
if _raw_chars > 20_000:
|
||||
_uncompressed_tokens = estimate_request_tokens_rough(
|
||||
messages,
|
||||
system_prompt=active_system_prompt or "",
|
||||
tools=agent.tools or None,
|
||||
)
|
||||
_ctx_len = getattr(
|
||||
getattr(agent, "context_compressor", None), "context_length", None
|
||||
)
|
||||
if not isinstance(_ctx_len, int) or _ctx_len <= 0:
|
||||
try:
|
||||
from agent.model_metadata import get_model_context_length
|
||||
|
||||
_ctx_len = get_model_context_length(
|
||||
agent.model,
|
||||
getattr(agent, "base_url", "") or "",
|
||||
provider=getattr(agent, "provider", "") or "",
|
||||
)
|
||||
except Exception:
|
||||
_ctx_len = None
|
||||
if _ctx_len and _uncompressed_tokens > _ctx_len:
|
||||
_warn_fn = getattr(
|
||||
agent, "_warn_uncompressed_context_overflow", None
|
||||
)
|
||||
if callable(_warn_fn):
|
||||
_warn_fn(_uncompressed_tokens, _ctx_len)
|
||||
else:
|
||||
_emit_w = getattr(agent, "_emit_warning", None)
|
||||
if callable(_emit_w):
|
||||
_emit_w(
|
||||
f"⚠️ Session context (~{_uncompressed_tokens:,} tokens) exceeds the "
|
||||
f"model context window (~{_ctx_len:,} tokens) with compression disabled "
|
||||
f"(compression.enabled: false). Use /compact to compress history or "
|
||||
f"enable compression in config.yaml."
|
||||
)
|
||||
|
||||
if _preflight_compressed:
|
||||
# Compression rebuilt the list (tail messages are fresh compaction
|
||||
|
||||
20
run_agent.py
20
run_agent.py
@@ -1042,6 +1042,26 @@ class AIAgent:
|
||||
)
|
||||
)
|
||||
|
||||
def _warn_uncompressed_context_overflow(
|
||||
self, preflight_tokens: int, context_length: int
|
||||
) -> None:
|
||||
"""Surface a deduped warning when uncompressed context exceeds model limit.
|
||||
|
||||
When compression is explicitly disabled (compression.enabled: false), long
|
||||
sessions can grow past the model context window with no compression to shrink
|
||||
them (#89297). Surface an actionable warning so the user knows to run /compact
|
||||
or enable compression.
|
||||
"""
|
||||
_warn_key = ("uncompressed_ctx_overflow", context_length)
|
||||
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
|
||||
self._last_ctx_overflow_warn = _warn_key
|
||||
self._emit_warning(
|
||||
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
|
||||
f"context window (~{context_length:,} tokens) with compression disabled "
|
||||
f"(compression.enabled: false). Use /compact to compress history or "
|
||||
f"enable compression in config.yaml."
|
||||
)
|
||||
|
||||
def _clear_context_overflow_warn(self) -> None:
|
||||
"""Reset the dedup state for the blocked-overflow warning.
|
||||
|
||||
|
||||
71
tests/agent/test_uncompressed_context_guardrail.py
Normal file
71
tests/agent/test_uncompressed_context_guardrail.py
Normal file
@@ -0,0 +1,71 @@
|
||||
"""Unit tests for uncompressed context overflow guardrail (Issue #89297)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import types
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.turn_context import TurnContext, build_turn_context
|
||||
from tests.agent.test_turn_context import _FakeAgent, _build
|
||||
|
||||
|
||||
class _FakeUncompressedAgent(_FakeAgent):
|
||||
"""Agent stub with compression disabled (compression.enabled: False)."""
|
||||
|
||||
def __init__(self, model="deepseek-v4-flash", context_length=10_000):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
self.provider = "deepseek"
|
||||
self.compression_enabled = False
|
||||
self.context_compressor = types.SimpleNamespace(
|
||||
protect_first_n=2,
|
||||
protect_last_n=2,
|
||||
context_length=context_length,
|
||||
threshold_tokens=int(context_length * 0.75),
|
||||
last_prompt_tokens=-1,
|
||||
)
|
||||
|
||||
def _warn_uncompressed_context_overflow(self, preflight_tokens: int, context_length: int) -> None:
|
||||
_warn_key = ("uncompressed_ctx_overflow", context_length)
|
||||
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
|
||||
self._last_ctx_overflow_warn = _warn_key
|
||||
msg = (
|
||||
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
|
||||
f"context window (~{context_length:,} tokens) with compression disabled "
|
||||
f"(compression.enabled: false). Use /compact to compress history or "
|
||||
f"enable compression in config.yaml."
|
||||
)
|
||||
self._emit_warning(msg)
|
||||
|
||||
|
||||
def test_uncompressed_session_within_limits_emits_no_warning():
|
||||
agent = _FakeUncompressedAgent(context_length=128_000)
|
||||
history = [
|
||||
{"role": "user", "content": "hello"},
|
||||
{"role": "assistant", "content": "hi there"},
|
||||
]
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
assert isinstance(tctx, TurnContext)
|
||||
agent._emit_warning.assert_not_called()
|
||||
|
||||
|
||||
def test_uncompressed_session_exceeding_context_limit_warns():
|
||||
# Model context length is 10,000 tokens (~40,000 chars)
|
||||
agent = _FakeUncompressedAgent(context_length=10_000)
|
||||
|
||||
# Construct an oversized uncompressed history of ~15,000 tokens (>60,000 chars)
|
||||
large_turn = "Large context content " * 500 # ~2,500 tokens
|
||||
history = []
|
||||
for i in range(10):
|
||||
history.append({"role": "user", "content": f"Turn {i}: {large_turn}"})
|
||||
history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"})
|
||||
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
assert isinstance(tctx, TurnContext)
|
||||
agent._emit_warning.assert_called_once()
|
||||
warning_msg = agent._emit_warning.call_args[0][0]
|
||||
assert "exceeds the model context window" in warning_msg
|
||||
assert "compression.enabled: false" in warning_msg
|
||||
assert "10,000 tokens" in warning_msg
|
||||
Reference in New Issue
Block a user