fix(agent_init): reserve Gemini's default maxOutputTokens in the compressor when max_tokens is unset

The native generateContent adapter never runs uncapped: when
model.max_tokens is unset it sends maxOutputTokens=65,535
(GEMINI_DEFAULT_MAX_OUTPUT_TOKENS) because Gemini treats an omitted cap
as a low internal default. The context compressor's trigger is
pct×(window − max_tokens), and constructing it with max_tokens=None
reserved 0 — so on a 128K Gemma window the trigger landed at 98,304
while the real safe input budget was 65,537, and the provider 400'd
before compaction fired.

Live repro (real imports, temp HERMES_HOME, native Gemini base_url,
window=131072, max_tokens unset):
  before: compressor.max_tokens=None, threshold_tokens=98304,
          wire maxOutputTokens=65535 → trigger ABOVE the safe budget
  after:  compressor.max_tokens=65535, threshold_tokens=64000 → below it

Scoped to the native Gemini wiring (provider names + native base_url via
is_native_gemini_base_url; the /openai compat endpoint is excluded). The
generic provider-default reservation gap remains tracked in #63839.

Reported by @Artemonim in #57275 (residual claim 4).
This commit is contained in:
Teknium
2026-08-31 11:54:41 -07:00
parent cb71d5f1b1
commit 7cefa87ea7
2 changed files with 125 additions and 1 deletions

View File

@@ -2777,6 +2777,34 @@ def init_agent(
if not agent.quiet_mode:
_ra().logger.info("Using context engine: %s", _selected_engine.name)
else:
# Native Gemini output reservation (#57275 claim 4): when
# model.max_tokens is unset, the native generateContent adapter does
# NOT run uncapped — it sends maxOutputTokens=65,535
# (GEMINI_DEFAULT_MAX_OUTPUT_TOKENS, see
# _effective_gemini_max_output_tokens). The compressor's threshold is
# pct×(window − max_tokens); passing None here meant it reserved 0
# while the wire reserved 65,535, so on a 128K window the trigger
# landed at ~96K against a real safe input budget of ~65K and the
# provider 400'd before compaction fired. Mirror the adapter's
# default so the reservation matches what is actually sent. The
# generic provider-default gap is #63839; this wires only the native
# Gemini path, where the default is a documented constant.
_compressor_max_tokens = agent.max_tokens
if _compressor_max_tokens is None:
try:
from agent.gemini_native_adapter import (
GEMINI_DEFAULT_MAX_OUTPUT_TOKENS,
is_native_gemini_base_url,
)
_gemini_provider = str(
getattr(agent, "provider", "") or ""
).strip().lower() in {
"gemini", "google", "google-gemini", "google-ai-studio",
}
if _gemini_provider or is_native_gemini_base_url(agent.base_url):
_compressor_max_tokens = GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
except Exception:
pass
agent.context_compressor = ContextCompressor(
model=agent.model,
threshold_percent=compression_threshold,
@@ -2791,7 +2819,7 @@ def init_agent(
provider=agent.provider,
api_mode=agent.api_mode,
abort_on_summary_failure=compression_abort_on_summary_failure,
max_tokens=agent.max_tokens,
max_tokens=_compressor_max_tokens,
model_thresholds=compression_model_thresholds,
threshold_tokens_cap=compression_threshold_tokens,
proactive_prune_tokens=compression_proactive_prune_tokens,

View File

@@ -0,0 +1,96 @@
"""Native Gemini output-token reservation at agent init (#57275 claim 4).
When ``model.max_tokens`` is unset, the native generateContent adapter still
sends ``maxOutputTokens=65,535`` (GEMINI_DEFAULT_MAX_OUTPUT_TOKENS) — Gemini
treats an omitted cap as a low internal default, not "unlimited", so the
adapter always sends an explicit cap. The compressor's trigger is
``pct × (window − max_tokens)``; constructing it with ``max_tokens=None``
reserved 0 while the wire reserved 65,535: on a 128K window the trigger
landed at ~96K against a real safe input budget of ~65K, and the provider
400'd before compaction ever fired.
These tests assert the compressor's reservation mirrors the adapter default
on the native Gemini path, and ONLY there.
"""
from unittest.mock import patch
import agent.context_compressor as cc_mod
from agent.gemini_native_adapter import GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
CFG = {"agent": {}}
def _build_agent(model, base_url, provider="", max_tokens=None, window=131072):
with (
patch("run_agent.get_tool_definitions", return_value=[]),
patch("run_agent.check_toolset_requirements", return_value={}),
patch("run_agent.OpenAI"),
patch("hermes_cli.config.load_config", return_value=CFG),
patch("hermes_cli.config.load_config_readonly", return_value=CFG),
patch(
"agent.model_metadata.get_model_context_length", return_value=window,
),
patch.object(cc_mod, "get_model_context_length", return_value=window),
):
from run_agent import AIAgent
return AIAgent(
model=model,
api_key="test-key-1234567890",
base_url=base_url,
provider=provider,
max_tokens=max_tokens,
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
def test_native_gemini_unset_max_tokens_reserves_adapter_default():
agent = _build_agent(
"gemma-3-27b-it",
"https://generativelanguage.googleapis.com/v1beta",
)
cc = agent.context_compressor
assert cc.max_tokens == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
# Trigger must sit at/below the real safe input budget the wire leaves.
assert cc.threshold_tokens <= cc.context_length - GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
def test_gemini_provider_name_also_reserves_default():
agent = _build_agent(
"gemini-3.7-flash", "https://example-proxy.invalid/v1", provider="google",
)
assert agent.context_compressor.max_tokens == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
def test_explicit_max_tokens_wins_over_adapter_default():
agent = _build_agent(
"gemma-3-27b-it",
"https://generativelanguage.googleapis.com/v1beta",
max_tokens=8192,
)
assert agent.context_compressor.max_tokens == 8192
def test_non_gemini_paths_keep_no_reservation():
agent = _build_agent(
"openai/gpt-4.1", "https://openrouter.ai/api/v1",
)
assert agent.context_compressor.max_tokens is None
def test_gemini_openai_compat_endpoint_not_treated_as_native():
# The /openai compatibility endpoint does not use the native adapter.
agent = _build_agent(
"gemma-3-27b-it",
"https://generativelanguage.googleapis.com/v1beta/openai",
)
assert agent.context_compressor.max_tokens is None