From 93af3db01dffd62132de645dbb0803b1d863a68a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Mon, 7 Sep 2026 02:13:48 -0700 Subject: [PATCH] fix: checkpoint Kanban completion before tool access expires Give dispatcher-owned workers a tool-capable reporting opportunity before the hard iteration cap, without accepting arbitrary diffs or weakening failure counting. Add opt-in per-turn iteration checkpoints for ordinary agents. Persist checkpoint text with the fresh tool result, never rewrite cached rows. Salvages the opt-in ratio and per-turn reset implementation from #104683; credits the earlier default-off signpost proposal in #92438. Local fixture wire A/B: Kanban ready/1 failure -> done/0; deliberately stuck workers still reach blocked/2 after two runs. Default-off control unchanged. Targeted and affected-directory suites queued behind campaign test lock. Co-authored-by: fangliquanflq Co-authored-by: C. Michael Gibbs <252231331+MikeGibbsOnyx@users.noreply.github.com> --- agent/agent_init.py | 20 ++++++- agent/tool_executor.py | 5 ++ agent/turn_context.py | 1 + agent/turn_iteration_prep.py | 59 ++++++++++++++++++++ hermes_cli/config_defaults.py | 3 + tests/agent/test_iteration_budget_warning.py | 52 +++++++++++++++++ tests/agent/test_kanban_budget_checkpoint.py | 19 +++++++ website/docs/user-guide/configuration.md | 3 + website/docs/user-guide/features/kanban.md | 16 ++++++ 9 files changed, 176 insertions(+), 2 deletions(-) create mode 100644 tests/agent/test_iteration_budget_warning.py create mode 100644 tests/agent/test_kanban_budget_checkpoint.py diff --git a/agent/agent_init.py b/agent/agent_init.py index bd63274cb8..8d424a054b 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -10,6 +10,7 @@ Symbols that tests patch on ``run_agent.*`` (``OpenAI``, ``get_tool_definitions` from __future__ import annotations import logging +import math import os import re import sys @@ -317,6 +318,17 @@ def _normalize_run_budget_seconds(value) -> Optional[float]: return seconds if seconds > 0 else None # NaN compares False → None +def _normalize_budget_warning_ratio(value) -> Optional[float]: + """A finite ratio strictly between zero and one, or None (feature off).""" + if value is None or isinstance(value, bool): + return None + try: + ratio = float(value) + except (TypeError, ValueError): + return None + return ratio if math.isfinite(ratio) and 0 < ratio < 1 else None + + def _refuse_checkpoint_required_on_codex_app_server( checkpoint_required: bool, api_mode: Optional[str] ) -> None: @@ -536,8 +548,9 @@ _CONTROL_STATE: Dict[str, Any] = { # Per-turn bookkeeping: budgets, activity tracking, rate-limit/credits telemetry. _TURN_STATE: Dict[str, Any] = { - # Iteration budget: notify the LLM only on exhaustion (one message, one grace call, then - # a forced summary) — intermediate pressure warnings made models give up early. + # Intermediate pressure warnings made models give up early; ordinary conversations + # remain opt-in. Dispatcher workers receive a bounded completion checkpoint. + "_iteration_budget_warning_injected": False, "_budget_exhausted_injected": False, "_budget_grace_call": False, "_run_budget_started_at": None, # set by turn_context.prepare_turn when a budget is active @@ -1307,6 +1320,9 @@ def _apply_agent_section(agent, _agent_cfg): agent._skill_nudge_interval = int(_agent_cfg.get("skills", {}).get("creation_nudge_interval", 10)) _agent_section = _cfg_dict(_agent_cfg, "agent") + agent.budget_warning_ratio = _normalize_budget_warning_ratio( + _agent_section.get("budget_warning_ratio") + ) # Both: "auto" (model-list match), true, false, or list of model substrings; independent # of each other (gates in agent/system_prompt.py). agent._tool_use_enforcement = _agent_section.get("tool_use_enforcement", "auto") diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 799199c10f..a34d284312 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -173,6 +173,11 @@ def _resolve_concurrent_tool_timeout() -> float | None: def _flush_session_db_after_tool_progress(agent, messages: list, *, stage: str) -> bool: """Flush tool-call progress to the session DB before projecting it to any UI: tool side effects can kill/restart the process before turn-end persistence runs.""" + from agent.turn_iteration_prep import _maybe_inject_iteration_budget_warning + + # Persist exactly the checkpoint text the next model call will see, before stamping + # this tool result as durable. Already-written rows must never be rewritten later. + _maybe_inject_iteration_budget_warning(agent, messages) try: persisted = agent._flush_messages_to_session_db(messages) is not False if not persisted: diff --git a/agent/turn_context.py b/agent/turn_context.py index 16b3015656..6f36e3524b 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -442,6 +442,7 @@ _PER_TURN_RESET_STATE: Tuple[Tuple[str, Any], ...] = ( ("_last_content_with_tools", None), ("_last_content_tools_all_housekeeping", False), ("_mute_post_response", False), ("_unicode_sanitization_passes", 0), ("_tool_guardrail_halt_decision", None), ("_vision_supported", True), + ("_iteration_budget_warning_injected", False), ("_run_budget_wrapup_injected", False), ("_verification_stop_nudges", 0), ("_pre_verify_nudges", 0), ) diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index ff25f98ea7..7790fcca66 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -11,6 +11,7 @@ from __future__ import annotations import logging import random +import sys from contextlib import suppress from dataclasses import dataclass from typing import Any, Dict @@ -20,6 +21,61 @@ from agent.turn_context import reanchor_current_turn_user_idx logger = logging.getLogger("agent.conversation_loop") +ITERATION_BUDGET_WARNING_TEMPLATE = ( + "[SYSTEM NOTICE — iteration budget checkpoint] You have used {used} of {maximum} " + "iterations. Checkpoint durable progress now, then continue the task; do not stop " + "solely because of this warning." +) + + +def _maybe_inject_iteration_budget_warning(agent: Any, messages: Any) -> bool: + """Append the opt-in one-shot warning to the newest tool result.""" + import os + from agent.delegation_context import is_dispatcher_owned_worker_context + + ratio = getattr(agent, "budget_warning_ratio", None) + kanban_worker = ( + bool(os.environ.get("HERMES_KANBAN_TASK")) + and is_dispatcher_owned_worker_context() + and "kanban_complete" in getattr(agent, "valid_tool_names", ()) + ) + if ratio is None and kanban_worker: + ratio = 0.9 + budget = getattr(agent, "iteration_budget", None) + if ( + ratio is None + or budget is None + or budget.max_total <= 1 + or budget.max_total >= sys.maxsize + or getattr(agent, "_iteration_budget_warning_injected", False) + or budget.used < min(ratio * budget.max_total, budget.max_total - 1) + ): + return False + notice = ITERATION_BUDGET_WARNING_TEMPLATE.format( + used=budget.used, maximum=budget.max_total + ) + if kanban_worker: + notice += ( + " While tools are still available, call kanban_complete only if all task " + "requirements are verified; otherwise persist a kanban_comment handoff and " + "continue. A diff or commit alone is not completion evidence." + ) + # Only the current tool-result tail is mutable; an older turn may already be cached. + from agent.context_compressor import _DB_PERSISTED_MARKER + if (not messages or messages[-1].get("role") != "tool" + or messages[-1].get(_DB_PERSISTED_MARKER)): + return False + message = messages[-1] + content = message.get("content", "") + if isinstance(content, str): + message["content"] = content + f"\n\n{notice}" + elif isinstance(content, list) or content is None: + message["content"] = [*(content or []), {"type": "text", "text": notice}] + else: + return False + agent._iteration_budget_warning_injected = True + return True + @dataclass class IterationPrep: @@ -76,6 +132,9 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera if getattr(agent, "run_budget_seconds", None): _maybe_inject_run_budget_wrapup(agent, messages) + # Use the same cache-safe channel as /steer; never add a synthetic user/system row. + _maybe_inject_iteration_budget_warning(agent, messages) + request_logger = getattr(agent, "logger", None) or logger # same name as the origin module # Per-agent validation cursor skips re-parsing tool_call args already validated. # Identity-keyed; a rewritten list breaks the prefix match and forces a re-scan. diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 9fc9d384ea..771bff9320 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -50,6 +50,9 @@ DEFAULT_CONFIG = { # Turn cap. null = unlimited (default; caps caused silent mid-task truncation). Positive int # caps; "none"/"unlimited"/"inf"/0/-1 also mean unlimited (resolve_turn_limit). "max_turns": None, + # Optional one-time model-visible checkpoint warning before a finite turn cap is exhausted. + # null = off; set a ratio strictly between 0 and 1 (for example, 0.75). + "budget_warning_ratio": None, # Wall-clock budget (seconds) per run. null = off. When set: one-time wrap-up notice at 80% # elapsed; implicit provider stale timeouts capped to remaining budget. CLI equivalent: # `hermes chat --run-budget N`. diff --git a/tests/agent/test_iteration_budget_warning.py b/tests/agent/test_iteration_budget_warning.py new file mode 100644 index 0000000000..90b46ee807 --- /dev/null +++ b/tests/agent/test_iteration_budget_warning.py @@ -0,0 +1,52 @@ +"""Iteration checkpoints preserve the transcript and the hard budget.""" +from copy import deepcopy + +import pytest + + +def _agent(tmp_path, monkeypatch, ratio): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text( + f"agent:\n budget_warning_ratio: {ratio}\n", encoding="utf-8" + ) + from run_agent import AIAgent + return AIAgent(model="test-model", provider="openai-compat", api_key="test", + base_url="http://127.0.0.1:1/v1", max_iterations=4, + quiet_mode=True, skip_context_files=True, skip_memory=True) + + +@pytest.mark.parametrize("ratio", ["null", "true", "0", "1", ".nan", "junk", "0.75"]) +def test_checkpoint_is_opt_in_and_does_not_rewrite_prior_turns(tmp_path, monkeypatch, ratio): + from agent.turn_iteration_prep import prepare_iteration + agent = _agent(tmp_path, monkeypatch, ratio) + for _ in range(3): + agent.iteration_budget.consume() + messages = [{"role": "user", "content": "work"}, + {"role": "assistant", "tool_calls": [{"id": "t", "type": "function", "function": {"name": "read_file", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "t", "content": "result"}] + original = deepcopy(messages) + prepare_iteration(agent, messages=messages, api_call_count=3) + assert messages[:-1] == original[:-1] + assert (messages[-1]["content"] != "result") == (ratio == "0.75") + snapshot = deepcopy(messages) + prepare_iteration(agent, messages=messages, api_call_count=3) + assert messages == snapshot + assert agent.iteration_budget.used == 3 + + +@pytest.mark.parametrize("content", ["result", [{"type": "text", "text": "result"}]]) +def test_checkpoint_rearms_per_turn_with_tools_still_available(tmp_path, monkeypatch, content): + from agent.turn_context import _reset_per_turn_agent_state + from agent.turn_iteration_prep import prepare_iteration + agent = _agent(tmp_path, monkeypatch, "0.75") + for _ in range(2): + _reset_per_turn_agent_state(agent) + for _ in range(3): + agent.iteration_budget.consume() + messages = [{"role": "user", "content": "work"}, + {"role": "assistant", "tool_calls": [{"id": "t", "type": "function", "function": {"name": "read_file", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "t", "content": deepcopy(content)}] + prepare_iteration(agent, messages=messages, api_call_count=3) + assert "3 of 4" in str(messages[-1]["content"]) + assert agent.iteration_budget.consume() + assert not agent.iteration_budget.consume() diff --git a/tests/agent/test_kanban_budget_checkpoint.py b/tests/agent/test_kanban_budget_checkpoint.py new file mode 100644 index 0000000000..c3ea13eca0 --- /dev/null +++ b/tests/agent/test_kanban_budget_checkpoint.py @@ -0,0 +1,19 @@ +import pytest +from tests.agent.test_iteration_budget_warning import _agent + +@pytest.mark.parametrize("owned", [True, False]) +def test_kanban_checkpoint_is_only_for_dispatcher_owner(tmp_path, monkeypatch, owned): + from agent.delegation_context import non_dispatcher_owned_context + from contextlib import nullcontext + from agent.turn_iteration_prep import prepare_iteration + monkeypatch.setenv("HERMES_KANBAN_TASK", "t_checkpoint") + with nullcontext() if owned else non_dispatcher_owned_context(): + agent = _agent(tmp_path, monkeypatch, "null") + for _ in range(4): + agent.iteration_budget.consume() + messages = [{"role": "user", "content": "work"}, + {"role": "assistant", "tool_calls": [{"id": "t", "type": "function", "function": {"name": "read_file", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "t", "content": "verified artifact"}] + prepare_iteration(agent, messages=messages, api_call_count=4) + assert ("kanban_complete" in messages[-1]["content"]) is owned + assert agent.iteration_budget.remaining == 0 diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 8562713058..d0f20e555f 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1133,11 +1133,14 @@ agent: max_turns: none # Iterations per conversation turn (default: none = unlimited) # Set a positive integer to cap; "none"/"null"/ # "unlimited"/"inf"/"infinity"/"infinite"/0/-1 = no limit + budget_warning_ratio: null # Optional one-time checkpoint warning, e.g. 0.75 api_max_retries: 3 # Retries per provider before fallback engages (default: 3) ``` `agent.max_turns` is **unlimited by default** — the turn cap caused more problems than it solved (silent mid-task truncation), so out of the box Hermes runs a conversation turn to completion. To impose a cap, set a positive integer. To be explicit about "no limit", any of these case-insensitive spellings work: `"none"`, `"null"`, `"unlimited"`, `"infinite"`, `"infinity"`, `"inf"`, `0`, `-1` (they resolve to a `sys.maxsize` sentinel so the loop never exits on a turn count). +`agent.budget_warning_ratio` is off by default for ordinary and delegated conversations. When set to a value strictly between `0` and `1` alongside a finite `max_turns`, Hermes appends one model-visible checkpoint notice to the latest tool result after the threshold is reached. The notice rearms each conversation turn and uses each agent's own iteration budget. It only appends to a current tool-result tail, never an older turn, and does not add a synthetic user/system message or change the existing exhaustion grace call. Dispatcher-owned Kanban workers receive a completion checkpoint at 90% by default (an explicit ratio changes that threshold), while their tools are still available. The checkpoint asks for verified completion or a durable progress comment, not premature success. + `agent.api_max_retries` controls how many times Hermes retries a provider API call on transient errors (rate limits, connection drops, 5xx) **before** fallback-provider switching engages. The default is `3` — four attempts total. If you have [fallback providers](/user-guide/features/fallback-providers) configured and want to fail over faster, drop this to `0` so the first transient error on your primary immediately hands off to the fallback instead of churning retries against the flaky endpoint. ## Wall-Clock Run Budget diff --git a/website/docs/user-guide/features/kanban.md b/website/docs/user-guide/features/kanban.md index 47ccb1cb74..8374942f28 100644 --- a/website/docs/user-guide/features/kanban.md +++ b/website/docs/user-guide/features/kanban.md @@ -10,6 +10,22 @@ description: "Durable SQLite-backed task board for coordinating multiple Hermes Hermes Kanban is a durable task board, shared across all your Hermes profiles, that lets multiple named agents collaborate on work without fragile in-process subagent swarms. Every task is a row in `~/.hermes/kanban.db`; every handoff is a row anyone can read and write; every worker is a full OS process with its own identity. +### Completion checkpoints before the iteration cap + +Dispatcher-owned workers get one checkpoint notice near 90% of their finite iteration +budget, attached to a fresh tool result while another tool-capable call remains. Use +`agent.budget_warning_ratio` to choose an earlier threshold. Tiny budgets warn no later +than their penultimate iteration; a one-iteration run has no pre-cap checkpoint window. +The notice is saved in the session transcript before the next request. Workers should +call `kanban_complete` only after verifying the task contract, or persist a progress +comment and continue. A commit or diff alone never automatically completes a task. + +The hard cap, toolless final summary, and consecutive-failure circuit breaker are +unchanged: workers that still exhaust their budget remain subject to bounded retries. +This is a reporting opportunity, not a guarantee that a model will heed the notice. +Ordinary conversations and delegated children do not inherit the automatic Kanban +checkpoint; their iteration warning remains opt-in. + ### Two surfaces: the model talks through tools, you talk through the CLI The board has two front doors, both backed by the same `~/.hermes/kanban.db`: