fix(review): cap default background input budget

This commit is contained in:
KoNit-K
2026-09-18 12:14:54 +08:00
committed by Teknium
parent ffe69fd711
commit f797a23c09
3 changed files with 89 additions and 14 deletions

View File

@@ -151,9 +151,12 @@ _REVIEW_MAX_ITERATIONS = 16
# Aggregate INPUT-token budget for one review fork (checked in conversation_loop's
# ``_review_input_budget_exhausted``). Request #1 replays the full snapshot as a warm cache read
# (both compression gates deferred until the first response); compaction then bounds each
# request, but nothing else caps the SUM across the tool loop. 2x the historical 300k foreground
# trigger. Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables.
_REVIEW_MAX_INPUT_TOKENS_DEFAULT = 600_000
# request, but nothing else caps the SUM across the tool loop. The default leaves 25% of the
# review model's context window available and never exceeds the historical cloud-scale ceiling.
# Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables.
_REVIEW_MAX_INPUT_TOKENS_CAP = 600_000
_REVIEW_INPUT_CONTEXT_FRACTION = 0.75
_REVIEW_MAX_INPUT_TOKENS_FALLBACK = 120_000
def _task_block(cfg: Any) -> Dict[str, Any]:
@@ -175,13 +178,45 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) ->
return {}
def _review_input_token_budget(task_cfg: Optional[Dict[str, Any]] = None) -> Optional[int]:
def _context_derived_review_input_budget(runtime: Optional[Dict[str, Any]] = None) -> int:
"""Return a bounded default based on the active review runtime when known."""
runtime = runtime if isinstance(runtime, dict) else {}
model = str(runtime.get("model") or "").strip()
if not model:
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
try:
from agent.model_metadata import get_model_context_length
context_window = get_model_context_length(
model,
base_url=str(runtime.get("base_url") or ""),
api_key=str(runtime.get("api_key") or ""),
provider=str(runtime.get("provider") or ""),
)
except Exception:
logger.debug("Background review context-window resolution failed", exc_info=True)
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
if not isinstance(context_window, int) or context_window <= 0:
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
return min(
_REVIEW_MAX_INPUT_TOKENS_CAP,
max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION)),
)
def _review_input_token_budget(
task_cfg: Optional[Dict[str, Any]] = None,
runtime: Optional[Dict[str, Any]] = None,
) -> Optional[int]:
"""Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables)."""
raw = _background_review_task_config(task_cfg).get("max_input_tokens", _REVIEW_MAX_INPUT_TOKENS_DEFAULT)
task = _background_review_task_config(task_cfg)
if "max_input_tokens" not in task:
return _context_derived_review_input_budget(runtime)
raw = task["max_input_tokens"]
try:
budget = int(raw)
except (TypeError, ValueError):
budget = _REVIEW_MAX_INPUT_TOKENS_DEFAULT
return _context_derived_review_input_budget(runtime)
return budget if budget > 0 else None
@@ -978,7 +1013,7 @@ def build_cache_parity_fork(
_detach_fork_compression(review_agent)
# Compaction bounds a single request; this bounds the WHOLE review (checked in
# conversation_loop via _review_input_budget_exhausted).
review_agent._review_input_token_budget = _review_input_token_budget(task_cfg)
review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, _rt)
return review_agent, _rt, _routed

View File

@@ -747,14 +747,15 @@ DEFAULT_CONFIG = {
"monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine
# Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying
# the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper).
# enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of
# replayed input tokens over the review loop (iterations capped at 16); the loop stops
# before crossing it. <= 0 = unlimited.
# enabled=false skips auto spawns (/refine still works). An explicit max_input_tokens caps
# the SUM of replayed input tokens over the review loop (iterations capped at 16); the loop
# stops before crossing it. When unset, the runtime derives a budget from the active model
# context window. <= 0 = unlimited.
# reasoning_effort is IGNORED while the review stays on the main model: the fork inherits the
# conversation's reasoning config verbatim so its request bytes keep the parent's warm
# prompt-cache prefix (#30532). Set provider/model below to route the review to another model
# if you want a different effort level; a one-time warning says so when the key is set.
"background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000},
"background_review": {"enabled": True, **_aux(120)},
# No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset
# (moa.presets.<name>.reference_models[].reasoning_effort / aggregator.reasoning_effort).
"moa_reference": _aux(900, reasoning_effort=False),

View File

@@ -224,16 +224,55 @@ def test_review_input_budget_exhausted_predicate_edge_cases():
@pytest.mark.parametrize(
("config_value", "expected"),
[
({}, 600_000),
({"max_input_tokens": 1_000_000}, 1_000_000),
({"max_input_tokens": 0}, None),
({"max_input_tokens": -5}, None),
({"max_input_tokens": "not-a-number"}, 600_000),
({"max_input_tokens": "300000"}, 300_000),
],
)
def test_review_input_token_budget_resolution(config_value, expected):
"""Config parsing: default, override, explicit disable, garbage fallback."""
"""Explicit settings retain their established override and unlimited semantics."""
from agent.background_review import _review_input_token_budget
assert _review_input_token_budget(config_value) == expected
@pytest.mark.parametrize(
("context_window", "expected"),
[
(65_536, 49_152),
(4_096, 3_072),
],
)
def test_review_input_token_budget_default_tracks_active_context(context_window, expected):
"""An unset budget leaves room below the review model's context window."""
from agent.background_review import _review_input_token_budget
runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"}
with patch("agent.model_metadata.get_model_context_length", return_value=context_window):
assert _review_input_token_budget({}, runtime) == expected
def test_review_input_token_budget_malformed_value_uses_context_derived_default():
"""A malformed explicit value is safe, rather than restoring the old 600k default."""
from agent.background_review import _review_input_token_budget
runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"}
with patch("agent.model_metadata.get_model_context_length", return_value=65_536):
assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, runtime) == 49_152
def test_review_input_token_budget_unknown_context_uses_conservative_fallback():
"""Failed context discovery must still bound unattended review work."""
from agent.background_review import _review_input_token_budget
runtime = {"provider": "local", "model": "unknown", "base_url": "http://localhost:1234"}
with patch("agent.model_metadata.get_model_context_length", side_effect=RuntimeError("unavailable")):
assert _review_input_token_budget({}, runtime) == 120_000
def test_background_review_config_does_not_freeze_a_fixed_input_budget():
"""The config default must leave the budget resolver access to the active runtime."""
from hermes_cli.config_defaults import DEFAULT_CONFIG
assert "max_input_tokens" not in DEFAULT_CONFIG["auxiliary"]["background_review"]