fix(agent): stop length continuation when the prompt filled the context window

finish_reason='length' has two causes: the answer was long (max_tokens reached),
or the prompt itself left no room to generate. _continue_text treated both the
same: append the fragment + a continuation nudge and retry, up to 4 times. In the
second case every retry sends a strictly longer prompt, so each attempt is worse
(Ollama n_ctx=32768: 32,638 -> 32,685 -> 32,732 prompt tokens, all truncated), the
user is told "model hit max output tokens", and max_tokens is not the lever.

The response's usage already carries prompt_tokens and the compressor already
resolves the model's context window; compare them once per truncation. Under
_MIN_CONTINUATION_HEADROOM (512) free tokens the turn ends on the first
truncation, keeps the partial text, names the context window as the cause and
points at /compress or a larger window. Unknown usage or window keeps today's
behaviour. max_tokens semantics untouched.

Co-authored-by: gaoanze888 <214786078+gaoanze888@users.noreply.github.com>
Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
This commit is contained in:
Teknium
2026-09-09 05:11:49 -07:00
parent 95cfcba353
commit 68bc4b216e
4 changed files with 135 additions and 1 deletions

View File

@@ -20,6 +20,7 @@ from agent.message_sanitization import close_interrupted_tool_sequence
from agent.repetition_guard import is_repetition_dominated
from agent.turn_api_call import stop_thinking_spinner
from agent.turn_retry_state import TurnRetryState
from agent.usage_pricing import normalize_usage
from hermes_constants import PARTIAL_STREAM_STUB_ID
logger = logging.getLogger("agent.conversation_loop")
@@ -60,6 +61,27 @@ _CEILING_NO_TEXT = (
"continuation attempt — its reasoning consumed the entire budget each time.\n\nTo fix this:\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning none`\n→ Or raise max_tokens for this model"
)
# Below this many free tokens the prompt itself filled the window: a continuation nudge +
# fragment costs ~100 tokens per attempt, so retrying only shrinks the room (#106120).
_MIN_CONTINUATION_HEADROOM = 512
_WINDOW_FILLED = (
"⚠️ **Context window full.** The prompt used {prompt:,} of this model's {ctx:,}-token "
"context window, leaving no room to answer in. This is a context-window limit, not an "
"output-length limit.\n\nTo fix this:\n→ Compress the conversation with `/compress` or start "
"a new session\n→ Or raise the model's context window (e.g. Ollama `num_ctx`)"
)
def _prompt_filled_window(agent: Any, response: Any) -> Optional[tuple[int, int]]:
"""``(prompt_tokens, context_length)`` when this response's usage shows the prompt left
less than ``_MIN_CONTINUATION_HEADROOM`` in the window compression resolves for the
model; ``None`` (keep continuing) when either number is unknown."""
ctx = int(getattr(getattr(agent, "context_compressor", None), "context_length", 0) or 0)
usage = getattr(response, "usage", None)
if not (ctx and usage):
return None
prompt = normalize_usage(usage, provider=agent.provider, api_mode=agent.api_mode).prompt_tokens
return (prompt, ctx) if prompt and ctx - prompt < _MIN_CONTINUATION_HEADROOM else None
def normalize_response_for_agent(agent: Any, response: Any) -> Any:
@@ -125,6 +147,7 @@ class _Trunc(TruncationVerdict):
current_turn_user_idx: Any
action: str = "fallthrough"
result: Optional[Dict[str, Any]] = None
window_filled: Optional[tuple[int, int]] = None # (prompt_tokens, context_length)
def done(self, action: str, result: Optional[Dict[str, Any]] = None) -> TruncationVerdict:
self.action, self.result = action, result
@@ -230,7 +253,8 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) -
append_message(messages, interim_msg)
st.truncated_response_parts.append(_interim_content)
if n < 4:
filled = st.window_filled
if n < 4 and filled is None:
_dropped_tools = getattr(st.response, "_dropped_tool_names", None)
if st.is_stub and _dropped_tools:
agent._vprint(
@@ -253,6 +277,8 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) -
# The one-shot reasoning-off override must not leak into the next turn.
agent._ephemeral_reasoning_off = False
agent._vprint(
f"{agent.log_prefix}⚠️ Not continuing — each attempt would only grow the prompt."
if filled is not None else
f"{agent.log_prefix}⚠️ Response still truncated after {n} continuation attempts — "
+ ("keeping the partial response received so far." if partial_response
else "no visible text was produced."),
@@ -272,6 +298,12 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) -
"role": "assistant", "content": partial_response, "finish_reason": "length"
})
agent._session_messages = messages
if filled is not None:
notice = _WINDOW_FILLED.format(prompt=filled[0], ctx=filled[1])
return st.end_turn(
f"{partial_response}\n\n{notice}" if partial_response else notice,
f"Prompt used {filled[0]} of {filled[1]} context tokens; no room to answer",
)
return st.end_turn(
partial_response or _CEILING_NO_TEXT,
"Response remained truncated after 4 continuation attempts",
@@ -335,9 +367,13 @@ def recover_from_truncation(
truncated_tool_call_retries=truncated_tool_call_retries, retry_count=retry_count,
compression_attempts=compression_attempts,
)
st.window_filled = _prompt_filled_window(agent, response)
agent._vprint(
f"{agent.log_prefix}⚠️ Response truncated — stream ended before completion"
if st.is_stub else
f"{agent.log_prefix}⚠️ Response truncated (finish_reason='length') - the prompt filled the "
f"context window ({st.window_filled[0]:,}/{st.window_filled[1]:,} tokens)"
if st.window_filled else
f"{agent.log_prefix}⚠️ Response truncated (finish_reason='length') - model hit max output tokens",
force=True,
)

View File

@@ -0,0 +1,2 @@
KoNit-K
# PR #106233 co-author

View File

@@ -0,0 +1,2 @@
gaoanze888
# PR #106223 co-author

View File

@@ -0,0 +1,94 @@
"""Length continuation stops when the PROMPT filled the context window (#106120).
``finish_reason='length'`` with ``usage.prompt_tokens`` ≈ context length means there was
no room to generate, not that the answer was long. Continuing appends a fragment + nudge
— strictly more prompt — so every retry is worse. The turn must end on the first
truncation and name the context window as the cause; a real output-cap truncation (plenty
of headroom) keeps continuing.
"""
from __future__ import annotations
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
from hermes_constants import FINISH_REASON_LENGTH
@pytest.fixture()
def loop_agent():
from run_agent import AIAgent
with (
patch("model_tools.get_tool_definitions", return_value=[]),
patch("model_tools.check_toolset_requirements", return_value={}),
patch("agent.process_bootstrap.OpenAI"),
):
a = AIAgent(
api_key="test-key-1234567890",
base_url="https://openrouter.ai/api/v1",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.compression_enabled = False
a.save_trajectories = False
a.context_compressor.context_length = 32768
return a
def _length_response(content: str, prompt_tokens: int):
from tests.run_agent.test_run_agent import _mock_assistant_msg
return SimpleNamespace(
id="resp",
model="test/model",
choices=[SimpleNamespace(
index=0, message=_mock_assistant_msg(content=content), finish_reason=FINISH_REASON_LENGTH,
)],
usage=SimpleNamespace(prompt_tokens=prompt_tokens, completion_tokens=40,
total_tokens=prompt_tokens + 40),
)
def _run(agent, message):
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
):
return agent.run_conversation(message)
def test_prompt_filling_the_window_ends_the_turn_on_first_truncation(loop_agent):
# Reporter's live numbers: 32,638-token prompt in a 32,768 window (~130 tokens of room).
loop_agent.client.chat.completions.create.side_effect = [
_length_response(f"part {i} ", prompt_tokens=32638 + 47 * i) for i in range(4)
]
result = _run(loop_agent, "summarize everything so far")
assert loop_agent.client.chat.completions.create.call_count == 1
assert result["partial"] is True
assert "part 0" in result["final_response"]
assert "context window" in result["final_response"].lower()
assert "32,638" in result["final_response"] and "32,768" in result["final_response"]
assert "continuation attempts" not in (result.get("error") or "")
# No continuation trail is left behind for the next turn.
assert not any(
m.get("_length_continuation_fragment") or m.get("_length_continuation_nudge")
for m in result["messages"] if isinstance(m, dict)
)
def test_output_cap_truncation_with_headroom_still_continues(loop_agent):
loop_agent.client.chat.completions.create.side_effect = [
_length_response(f"part {i} ", prompt_tokens=4_000 + 500 * i) for i in range(4)
]
result = _run(loop_agent, "write me a long report")
assert loop_agent.client.chat.completions.create.call_count == 4
assert "truncated after 4 continuation attempts" in (result.get("error") or "")
assert "part 3" in result["final_response"]