diff --git a/agent/background_review.py b/agent/background_review.py index 5979b1134b..2d0db4ea5d 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -151,9 +151,12 @@ _REVIEW_MAX_ITERATIONS = 16 # Aggregate INPUT-token budget for one review fork (checked in conversation_loop's # ``_review_input_budget_exhausted``). Request #1 replays the full snapshot as a warm cache read # (both compression gates deferred until the first response); compaction then bounds each -# request, but nothing else caps the SUM across the tool loop. 2x the historical 300k foreground -# trigger. Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. -_REVIEW_MAX_INPUT_TOKENS_DEFAULT = 600_000 +# request, but nothing else caps the SUM across the tool loop. The default leaves 25% of the +# review model's context window available and never exceeds the historical cloud-scale ceiling. +# Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. +_REVIEW_MAX_INPUT_TOKENS_CAP = 600_000 +_REVIEW_INPUT_CONTEXT_FRACTION = 0.75 +_REVIEW_MAX_INPUT_TOKENS_FALLBACK = 120_000 def _task_block(cfg: Any) -> Dict[str, Any]: @@ -175,13 +178,45 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) -> return {} -def _review_input_token_budget(task_cfg: Optional[Dict[str, Any]] = None) -> Optional[int]: +def _context_derived_review_input_budget(runtime: Optional[Dict[str, Any]] = None) -> int: + """Return a bounded default based on the active review runtime when known.""" + runtime = runtime if isinstance(runtime, dict) else {} + model = str(runtime.get("model") or "").strip() + if not model: + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + try: + from agent.model_metadata import get_model_context_length + + context_window = get_model_context_length( + model, + base_url=str(runtime.get("base_url") or ""), + api_key=str(runtime.get("api_key") or ""), + provider=str(runtime.get("provider") or ""), + ) + except Exception: + logger.debug("Background review context-window resolution failed", exc_info=True) + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + if not isinstance(context_window, int) or context_window <= 0: + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + return min( + _REVIEW_MAX_INPUT_TOKENS_CAP, + max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION)), + ) + + +def _review_input_token_budget( + task_cfg: Optional[Dict[str, Any]] = None, + runtime: Optional[Dict[str, Any]] = None, +) -> Optional[int]: """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables).""" - raw = _background_review_task_config(task_cfg).get("max_input_tokens", _REVIEW_MAX_INPUT_TOKENS_DEFAULT) + task = _background_review_task_config(task_cfg) + if "max_input_tokens" not in task: + return _context_derived_review_input_budget(runtime) + raw = task["max_input_tokens"] try: budget = int(raw) except (TypeError, ValueError): - budget = _REVIEW_MAX_INPUT_TOKENS_DEFAULT + return _context_derived_review_input_budget(runtime) return budget if budget > 0 else None @@ -978,7 +1013,7 @@ def build_cache_parity_fork( _detach_fork_compression(review_agent) # Compaction bounds a single request; this bounds the WHOLE review (checked in # conversation_loop via _review_input_budget_exhausted). - review_agent._review_input_token_budget = _review_input_token_budget(task_cfg) + review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, _rt) return review_agent, _rt, _routed diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 94a1e465bf..80a3bfbafd 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -747,14 +747,15 @@ DEFAULT_CONFIG = { "monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine # Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying # the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper). - # enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of - # replayed input tokens over the review loop (iterations capped at 16); the loop stops - # before crossing it. <= 0 = unlimited. + # enabled=false skips auto spawns (/refine still works). An explicit max_input_tokens caps + # the SUM of replayed input tokens over the review loop (iterations capped at 16); the loop + # stops before crossing it. When unset, the runtime derives a budget from the active model + # context window. <= 0 = unlimited. # reasoning_effort is IGNORED while the review stays on the main model: the fork inherits the # conversation's reasoning config verbatim so its request bytes keep the parent's warm # prompt-cache prefix (#30532). Set provider/model below to route the review to another model # if you want a different effort level; a one-time warning says so when the key is set. - "background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000}, + "background_review": {"enabled": True, **_aux(120)}, # No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset # (moa.presets..reference_models[].reasoning_effort / aggregator.reasoning_effort). "moa_reference": _aux(900, reasoning_effort=False), diff --git a/tests/agent/test_background_review_input_budget.py b/tests/agent/test_background_review_input_budget.py index b0d483f93d..0c556111a7 100644 --- a/tests/agent/test_background_review_input_budget.py +++ b/tests/agent/test_background_review_input_budget.py @@ -224,16 +224,55 @@ def test_review_input_budget_exhausted_predicate_edge_cases(): @pytest.mark.parametrize( ("config_value", "expected"), [ - ({}, 600_000), ({"max_input_tokens": 1_000_000}, 1_000_000), ({"max_input_tokens": 0}, None), ({"max_input_tokens": -5}, None), - ({"max_input_tokens": "not-a-number"}, 600_000), ({"max_input_tokens": "300000"}, 300_000), ], ) def test_review_input_token_budget_resolution(config_value, expected): - """Config parsing: default, override, explicit disable, garbage fallback.""" + """Explicit settings retain their established override and unlimited semantics.""" from agent.background_review import _review_input_token_budget assert _review_input_token_budget(config_value) == expected + + +@pytest.mark.parametrize( + ("context_window", "expected"), + [ + (65_536, 49_152), + (4_096, 3_072), + ], +) +def test_review_input_token_budget_default_tracks_active_context(context_window, expected): + """An unset budget leaves room below the review model's context window.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", return_value=context_window): + assert _review_input_token_budget({}, runtime) == expected + + +def test_review_input_token_budget_malformed_value_uses_context_derived_default(): + """A malformed explicit value is safe, rather than restoring the old 600k default.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", return_value=65_536): + assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, runtime) == 49_152 + + +def test_review_input_token_budget_unknown_context_uses_conservative_fallback(): + """Failed context discovery must still bound unattended review work.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "local", "model": "unknown", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", side_effect=RuntimeError("unavailable")): + assert _review_input_token_budget({}, runtime) == 120_000 + + +def test_background_review_config_does_not_freeze_a_fixed_input_budget(): + """The config default must leave the budget resolver access to the active runtime.""" + from hermes_cli.config_defaults import DEFAULT_CONFIG + + assert "max_input_tokens" not in DEFAULT_CONFIG["auxiliary"]["background_review"]