fix(review): cap default background input budget
This commit is contained in:
@@ -151,9 +151,12 @@ _REVIEW_MAX_ITERATIONS = 16
|
||||
# Aggregate INPUT-token budget for one review fork (checked in conversation_loop's
|
||||
# ``_review_input_budget_exhausted``). Request #1 replays the full snapshot as a warm cache read
|
||||
# (both compression gates deferred until the first response); compaction then bounds each
|
||||
# request, but nothing else caps the SUM across the tool loop. 2x the historical 300k foreground
|
||||
# trigger. Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables.
|
||||
_REVIEW_MAX_INPUT_TOKENS_DEFAULT = 600_000
|
||||
# request, but nothing else caps the SUM across the tool loop. The default leaves 25% of the
|
||||
# review model's context window available and never exceeds the historical cloud-scale ceiling.
|
||||
# Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables.
|
||||
_REVIEW_MAX_INPUT_TOKENS_CAP = 600_000
|
||||
_REVIEW_INPUT_CONTEXT_FRACTION = 0.75
|
||||
_REVIEW_MAX_INPUT_TOKENS_FALLBACK = 120_000
|
||||
|
||||
|
||||
def _task_block(cfg: Any) -> Dict[str, Any]:
|
||||
@@ -175,13 +178,45 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) ->
|
||||
return {}
|
||||
|
||||
|
||||
def _review_input_token_budget(task_cfg: Optional[Dict[str, Any]] = None) -> Optional[int]:
|
||||
def _context_derived_review_input_budget(runtime: Optional[Dict[str, Any]] = None) -> int:
|
||||
"""Return a bounded default based on the active review runtime when known."""
|
||||
runtime = runtime if isinstance(runtime, dict) else {}
|
||||
model = str(runtime.get("model") or "").strip()
|
||||
if not model:
|
||||
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
|
||||
try:
|
||||
from agent.model_metadata import get_model_context_length
|
||||
|
||||
context_window = get_model_context_length(
|
||||
model,
|
||||
base_url=str(runtime.get("base_url") or ""),
|
||||
api_key=str(runtime.get("api_key") or ""),
|
||||
provider=str(runtime.get("provider") or ""),
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Background review context-window resolution failed", exc_info=True)
|
||||
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
|
||||
if not isinstance(context_window, int) or context_window <= 0:
|
||||
return _REVIEW_MAX_INPUT_TOKENS_FALLBACK
|
||||
return min(
|
||||
_REVIEW_MAX_INPUT_TOKENS_CAP,
|
||||
max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION)),
|
||||
)
|
||||
|
||||
|
||||
def _review_input_token_budget(
|
||||
task_cfg: Optional[Dict[str, Any]] = None,
|
||||
runtime: Optional[Dict[str, Any]] = None,
|
||||
) -> Optional[int]:
|
||||
"""Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables)."""
|
||||
raw = _background_review_task_config(task_cfg).get("max_input_tokens", _REVIEW_MAX_INPUT_TOKENS_DEFAULT)
|
||||
task = _background_review_task_config(task_cfg)
|
||||
if "max_input_tokens" not in task:
|
||||
return _context_derived_review_input_budget(runtime)
|
||||
raw = task["max_input_tokens"]
|
||||
try:
|
||||
budget = int(raw)
|
||||
except (TypeError, ValueError):
|
||||
budget = _REVIEW_MAX_INPUT_TOKENS_DEFAULT
|
||||
return _context_derived_review_input_budget(runtime)
|
||||
return budget if budget > 0 else None
|
||||
|
||||
|
||||
@@ -978,7 +1013,7 @@ def build_cache_parity_fork(
|
||||
_detach_fork_compression(review_agent)
|
||||
# Compaction bounds a single request; this bounds the WHOLE review (checked in
|
||||
# conversation_loop via _review_input_budget_exhausted).
|
||||
review_agent._review_input_token_budget = _review_input_token_budget(task_cfg)
|
||||
review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, _rt)
|
||||
return review_agent, _rt, _routed
|
||||
|
||||
|
||||
|
||||
@@ -747,14 +747,15 @@ DEFAULT_CONFIG = {
|
||||
"monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine
|
||||
# Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying
|
||||
# the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper).
|
||||
# enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of
|
||||
# replayed input tokens over the review loop (iterations capped at 16); the loop stops
|
||||
# before crossing it. <= 0 = unlimited.
|
||||
# enabled=false skips auto spawns (/refine still works). An explicit max_input_tokens caps
|
||||
# the SUM of replayed input tokens over the review loop (iterations capped at 16); the loop
|
||||
# stops before crossing it. When unset, the runtime derives a budget from the active model
|
||||
# context window. <= 0 = unlimited.
|
||||
# reasoning_effort is IGNORED while the review stays on the main model: the fork inherits the
|
||||
# conversation's reasoning config verbatim so its request bytes keep the parent's warm
|
||||
# prompt-cache prefix (#30532). Set provider/model below to route the review to another model
|
||||
# if you want a different effort level; a one-time warning says so when the key is set.
|
||||
"background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000},
|
||||
"background_review": {"enabled": True, **_aux(120)},
|
||||
# No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset
|
||||
# (moa.presets.<name>.reference_models[].reasoning_effort / aggregator.reasoning_effort).
|
||||
"moa_reference": _aux(900, reasoning_effort=False),
|
||||
|
||||
@@ -224,16 +224,55 @@ def test_review_input_budget_exhausted_predicate_edge_cases():
|
||||
@pytest.mark.parametrize(
|
||||
("config_value", "expected"),
|
||||
[
|
||||
({}, 600_000),
|
||||
({"max_input_tokens": 1_000_000}, 1_000_000),
|
||||
({"max_input_tokens": 0}, None),
|
||||
({"max_input_tokens": -5}, None),
|
||||
({"max_input_tokens": "not-a-number"}, 600_000),
|
||||
({"max_input_tokens": "300000"}, 300_000),
|
||||
],
|
||||
)
|
||||
def test_review_input_token_budget_resolution(config_value, expected):
|
||||
"""Config parsing: default, override, explicit disable, garbage fallback."""
|
||||
"""Explicit settings retain their established override and unlimited semantics."""
|
||||
from agent.background_review import _review_input_token_budget
|
||||
|
||||
assert _review_input_token_budget(config_value) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("context_window", "expected"),
|
||||
[
|
||||
(65_536, 49_152),
|
||||
(4_096, 3_072),
|
||||
],
|
||||
)
|
||||
def test_review_input_token_budget_default_tracks_active_context(context_window, expected):
|
||||
"""An unset budget leaves room below the review model's context window."""
|
||||
from agent.background_review import _review_input_token_budget
|
||||
|
||||
runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"}
|
||||
with patch("agent.model_metadata.get_model_context_length", return_value=context_window):
|
||||
assert _review_input_token_budget({}, runtime) == expected
|
||||
|
||||
|
||||
def test_review_input_token_budget_malformed_value_uses_context_derived_default():
|
||||
"""A malformed explicit value is safe, rather than restoring the old 600k default."""
|
||||
from agent.background_review import _review_input_token_budget
|
||||
|
||||
runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"}
|
||||
with patch("agent.model_metadata.get_model_context_length", return_value=65_536):
|
||||
assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, runtime) == 49_152
|
||||
|
||||
|
||||
def test_review_input_token_budget_unknown_context_uses_conservative_fallback():
|
||||
"""Failed context discovery must still bound unattended review work."""
|
||||
from agent.background_review import _review_input_token_budget
|
||||
|
||||
runtime = {"provider": "local", "model": "unknown", "base_url": "http://localhost:1234"}
|
||||
with patch("agent.model_metadata.get_model_context_length", side_effect=RuntimeError("unavailable")):
|
||||
assert _review_input_token_budget({}, runtime) == 120_000
|
||||
|
||||
|
||||
def test_background_review_config_does_not_freeze_a_fixed_input_budget():
|
||||
"""The config default must leave the budget resolver access to the active runtime."""
|
||||
from hermes_cli.config_defaults import DEFAULT_CONFIG
|
||||
|
||||
assert "max_input_tokens" not in DEFAULT_CONFIG["auxiliary"]["background_review"]
|
||||
|
||||
Reference in New Issue
Block a user