From a79d1d3a71f40e62b58e811d4fe955104f33c0be Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 19 Sep 2026 03:19:03 -0700 Subject: [PATCH] feat: one-shot runs drop the self-improvement footprint (no skill authoring, fewer process skills, delegation cap) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A finite `hermes chat -q` / `--oneshot` run has no later session in its HERMES_HOME to learn for, yet it ran the full interactive self-improvement loop. Measured over 21 one-shot benchmark trajectories: 7 skills created and a bundled one patched mid-task, 37 of ~215 tool calls on skill_view/skill_manage, skill text = 34% of all tool-result bytes fed back into context, plus reviewer subagents spawned on the agent's own diff (one task: 5 delegations, 62 subagent API calls, each re-paying a cold system prompt). Keyed on the existing HERMES_SINGLE_QUERY_SESSION marker (approval gate, delegation dispatcher), so interactive and gateway sessions are byte-identical: * agent/oneshot_footprint.py (new sibling): skill_manage is pruned from the tool set; the ## Skills block keeps the index + skill_view but drops the record/patch/ offer-to-save coaching and the "load process skills for work you already know" push (SKILLS_GUIDANCE follows because it is gated on skill_manage). * delegation.oneshot_max_children (default 2, 0 = unlimited): total children a one-shot run may spawn; past it delegate_task returns a tool error telling the model to finish inline. * requesting-code-review skill: reviewer/fixer subagents (Steps 5 and 7) are interactive-only; one-shot applies the checklist inline. Live: `hermes chat -q "list tools starting with skill_"` on the portal — base "skill_manage, skill_view, skills_list", fix "skill_view, skills_list"; a real AIAgent under a temp HERMES_HOME shows the interactive prompt unchanged (6643 chars both) and the one-shot prompt without skill_manage / offer-to-save. --- agent/agent_init.py | 3 + agent/oneshot_footprint.py | 48 ++++++++++++++ agent/prompt_builder.py | 13 ++++ hermes_cli/config_defaults.py | 4 ++ .../requesting-code-review/SKILL.md | 9 ++- tests/agent/test_oneshot_footprint.py | 65 +++++++++++++++++++ tools/delegate_tool.py | 25 ++++++- tools/delegate_tool_config.py | 8 +++ website/docs/user-guide/configuration.md | 3 + .../docs/user-guide/features/delegation.md | 2 + 10 files changed, 177 insertions(+), 3 deletions(-) create mode 100644 agent/oneshot_footprint.py create mode 100644 tests/agent/test_oneshot_footprint.py diff --git a/agent/agent_init.py b/agent/agent_init.py index a7b7b64c6a..246ab57af4 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1059,6 +1059,9 @@ def _load_tools(agent, enabled_toolsets, disabled_toolsets): enabled_toolsets=enabled_toolsets, disabled_toolsets=disabled_toolsets, quiet_mode=agent.quiet_mode, ) + # A finite -q run has no later session to learn for: no skill authoring tool (agent/oneshot_footprint.py). + from agent.oneshot_footprint import prune_oneshot_tools + agent.tools = prune_oneshot_tools(agent.tools or []) agent.valid_tool_names = {tool["function"]["name"] for tool in agent.tools} if agent.tools else set() # Kanban guidance is session-static for the dispatcher-owned worker only. Profiles may diff --git a/agent/oneshot_footprint.py b/agent/oneshot_footprint.py new file mode 100644 index 0000000000..61e7545f5d --- /dev/null +++ b/agent/oneshot_footprint.py @@ -0,0 +1,48 @@ +"""What a finite one-shot session (``hermes chat -q`` / ``--oneshot``, ``hermes -z``) does NOT do. + +A one-shot run has no later session in its HERMES_HOME to learn for: the process answers one query and +exits. The interactive self-improvement loop is pure overhead there, and a measured one — across 21 +one-shot benchmark trajectories the agent authored 7 new skills and patched a bundled one mid-task, 37 of +~215 tool calls were ``skill_view``/``skill_manage``, and skill text was 34% of every tool-result byte fed +back into context. Three rules follow, all keyed on the same session marker the approval gate and the +delegation dispatcher already read (``HERMES_SINGLE_QUERY_SESSION``), so interactive sessions are untouched: + +* ``skill_manage`` is not offered (``skills_list``/``skill_view`` stay: reading a domain skill can still win); +* the ## Skills prompt drops the "record it / patch it / offer to save" coaching and the "load process skills + even for tasks you already know" push, keeping only "load a skill when it adds knowledge you lack"; +* delegation is capped per session (``delegation.oneshot_max_children``): subagents each re-pay a cold + system prompt and re-explore the repo, and the observed spawns were mostly "independent review of my own + work" rather than parallel work. +""" +from __future__ import annotations + +from typing import Any, Dict, Iterable, List + +ONESHOT_HIDDEN_TOOLS = frozenset({"skill_manage"}) + + +def is_single_query_session() -> bool: + """The finite ``-q`` marker, read through the session env so gateway-bound sessions never see it.""" + try: + from gateway.session_context import get_session_env + except Exception: + import os + get_session_env = os.environ.get + return str(get_session_env("HERMES_SINGLE_QUERY_SESSION", "") or "") == "1" + + +def prune_oneshot_tools(tools: Iterable[Dict[str, Any]]) -> List[Dict[str, Any]]: + """*tools* minus ``ONESHOT_HIDDEN_TOOLS``; identity when the session is not one-shot.""" + tools = list(tools) + if not is_single_query_session(): + return tools + return [t for t in tools if (t.get("function") or {}).get("name") not in ONESHOT_HIDDEN_TOOLS] + + +ONESHOT_SKILLS_LOAD_GUIDANCE = ( + "## Skills\n" + "Scan the skills below and load one with skill_view(name) only when it carries domain knowledge you lack " + "for THIS task (an API, a tool's commands, a project's conventions). Do not load general process skills " + "(testing, debugging, review methodology) for work you already know how to do, and do not create or edit " + "skills: this is a one-shot run with no later session to reuse them.\n" +) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index a1287768ce..49a63d6881 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -1354,6 +1354,13 @@ def _render_skills_index( if name not in seen: seen.add(name) index_lines.append(f" - {name}: {desc}" if desc else f" - {name}") + from agent.oneshot_footprint import ONESHOT_SKILLS_LOAD_GUIDANCE, is_single_query_session + if is_single_query_session(): + return ( + ONESHOT_SKILLS_LOAD_GUIDANCE + + "\n\n" + "\n".join(index_lines) + "\n" + + hidden_note + ) return ( "## Skills\n" "Before replying, scan the skills below. If a skill matches or is even partially relevant to your " @@ -1377,6 +1384,11 @@ def _render_skills_index( ) +def _oneshot_prompt_variant() -> bool: + from agent.oneshot_footprint import is_single_query_session + return is_single_query_session() + + def _build_skills_system_prompt_inner( skills_dir: "Path", external_dirs: "list[Path]", available_tools: "set[str] | None", available_toolsets: "set[str] | None", compact_categories: "frozenset[str] | None", @@ -1391,6 +1403,7 @@ def _build_skills_system_prompt_inner( tuple(sorted(str(t) for t in (available_tools or set()))), tuple(sorted(str(ts) for ts in (available_toolsets or set()))), _platform_hint, tuple(sorted(disabled)), tuple(sorted(compact_categories or ())), + _oneshot_prompt_variant(), ) with _SKILLS_PROMPT_CACHE_LOCK: cached = _SKILLS_PROMPT_CACHE.get(cache_key) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 4c80450f44..2f663a683d 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1308,6 +1308,10 @@ DEFAULT_CONFIG = { # Orchestrator role controls. Depth floored at 1, no ceiling; each level multiplies cost. "max_spawn_depth": 1, # 1 = flat, 2 = orchestrator→leaf, 3+ = deeper "orchestrator_enabled": True, # kill switch for role="orchestrator" + # Total subagents a finite one-shot run (hermes chat -q / --oneshot) may spawn; 0 = unlimited. + # Each child re-pays a cold system prompt and re-explores the repo, and one-shot spawns are mostly + # "review my own work" rather than parallel work (agent/oneshot_footprint.py). + "oneshot_max_children": 2, # Subagent threads ALWAYS resolve approvals non-interactively (the parent TUI owns stdin; # input() from a worker would deadlock). false = auto-deny, true = auto-approve "once"; both # log a warning audit line. true only for trusted batch work. diff --git a/skills/software-development/requesting-code-review/SKILL.md b/skills/software-development/requesting-code-review/SKILL.md index 1209c3411c..24c20b8a1b 100644 --- a/skills/software-development/requesting-code-review/SKILL.md +++ b/skills/software-development/requesting-code-review/SKILL.md @@ -1,7 +1,7 @@ --- name: requesting-code-review description: "Pre-commit review: security scan, quality gates, auto-fix." -version: 2.0.0 +version: 2.1.0 author: Hermes Agent (adapted from obra/superpowers + MorAlekss) license: MIT platforms: [linux, macos, windows] @@ -124,6 +124,11 @@ Quick scan before dispatching the reviewer: ## Step 5 — Independent reviewer subagent +**Interactive sessions only.** In a one-shot run (`hermes chat -q`, `--oneshot`, a +benchmark harness) there is no one to hand the verdict to and a fresh subagent re-pays +the whole system prompt plus a repo re-read: skip Steps 5 and 7, apply the Step 4 +checklist to the diff yourself, run the tests, and go to Step 8. + Call `delegate_task` directly — it is NOT available inside execute_code or scripts. The reviewer gets ONLY the diff and static scan results. No shared context with @@ -193,7 +198,7 @@ Suggestions (non-blocking): [list] ## Step 7 — Auto-fix loop -**Maximum 2 fix-and-reverify cycles.** +**Maximum 2 fix-and-reverify cycles. Interactive sessions only (see Step 5).** Spawn a THIRD agent context — not you (the implementer), not the reviewer. It fixes ONLY the reported issues: diff --git a/tests/agent/test_oneshot_footprint.py b/tests/agent/test_oneshot_footprint.py new file mode 100644 index 0000000000..ecc8d16b03 --- /dev/null +++ b/tests/agent/test_oneshot_footprint.py @@ -0,0 +1,65 @@ +"""One-shot sessions (``hermes chat -q``) drop the self-improvement footprint. + +Across 21 one-shot benchmark trajectories the agent created 7 skills and patched a bundled one +mid-task, spent 37 of ~215 tool calls on skill_view/skill_manage, and spawned review subagents of +its own work. None of that has a consumer in a finite run. The marker is the same +``HERMES_SINGLE_QUERY_SESSION`` the approval gate and delegation dispatcher read, so an interactive +session — the control in every test here — keeps the full surface. +""" + +import pytest + +from agent import oneshot_footprint +from agent.prompt_builder import build_skills_system_prompt + + +@pytest.fixture +def oneshot(monkeypatch): + monkeypatch.setenv("HERMES_SINGLE_QUERY_SESSION", "1") + + +def _tools(*names): + return [{"type": "function", "function": {"name": n}} for n in names] + + +def test_oneshot_hides_skill_manage_and_skill_authoring_coaching(oneshot, interactive_prompt, tmp_path): + """-q: no skill_manage tool and a skills prompt that neither asks to save/patch skills nor pushes process + skills; skill reading stays. The interactive prompt for the same skills dir is the control.""" + kept = {t["function"]["name"] for t in oneshot_footprint.prune_oneshot_tools( + _tools("skill_manage", "skill_view", "skills_list", "terminal"))} + assert "skill_manage" not in kept and {"skill_view", "skills_list", "terminal"} <= kept + + prompt = build_skills_system_prompt(available_tools={"skill_view", "skills_list"}, skills_dir_override=_skills_dir(tmp_path)) + assert "demo-skill" in prompt and "skill_view" in prompt + assert "skill_manage" not in prompt and "offer to save as a skill" not in prompt + assert "skill_manage" in interactive_prompt and "offer to save as a skill" in interactive_prompt + + +def _skills_dir(tmp_path): + d = tmp_path / "skills" / "misc" / "demo-skill" + d.mkdir(parents=True, exist_ok=True) + (d / "SKILL.md").write_text("---\nname: demo-skill\ndescription: Demo skill for tests.\n---\n# Demo\n", encoding="utf-8") + return tmp_path / "skills" + + +@pytest.fixture +def interactive_prompt(tmp_path, monkeypatch): + monkeypatch.delenv("HERMES_SINGLE_QUERY_SESSION", raising=False) + prompt = build_skills_system_prompt(available_tools={"skill_view", "skills_list", "skill_manage"}, + skills_dir_override=_skills_dir(tmp_path)) + monkeypatch.setenv("HERMES_SINGLE_QUERY_SESSION", "1") + return prompt + + +def test_oneshot_delegation_budget_charges_total_children_then_refuses(oneshot, monkeypatch): + from tools import delegate_tool + + monkeypatch.setattr(delegate_tool, "_get_oneshot_max_children", lambda: 2) + parent = type("P", (), {})() + assert delegate_tool._oneshot_spawn_budget(parent, 1) is None + assert delegate_tool._oneshot_spawn_budget(parent, 1) is None + err = delegate_tool._oneshot_spawn_budget(parent, 1) + assert err and "oneshot_max_children" in err + # Interactive sessions are never charged, whatever the count. + monkeypatch.delenv("HERMES_SINGLE_QUERY_SESSION") + assert delegate_tool._oneshot_spawn_budget(parent, 50) is None diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 9cc8fd7f7e..f2c7c67053 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -29,7 +29,7 @@ from tools.delegate_tool_child_run import ( # noqa: F401 ) from tools.delegate_tool_config import ( # noqa: F401 _DEFAULT_MAX_CONCURRENT_CHILDREN, _get_child_timeout, _get_max_async_children, _get_max_concurrent_children, - _get_max_spawn_depth, _get_orchestrator_enabled, _get_subagent_approval_callback, _get_worktree_isolation, + _get_max_spawn_depth, _get_oneshot_max_children, _get_orchestrator_enabled, _get_subagent_approval_callback, _get_worktree_isolation, _inherit_parent_capabilities, _load_config, _merge_request_overrides, _resolve_child_credential_pool, _resolve_child_runtime, _resolve_delegation_credentials, _subagent_auto_approve, _subagent_auto_deny, @@ -414,6 +414,26 @@ def _build_children( return children, None +def _oneshot_spawn_budget(parent_agent: Any, requested: int) -> Optional[str]: + """Charge *requested* children against the finite one-shot session's total (delegation.oneshot_max_children); + the error text tells the model to do the work inline. Interactive and gateway sessions are never charged.""" + from agent.oneshot_footprint import is_single_query_session + if not is_single_query_session(): + return None + cap = _get_oneshot_max_children() + if cap <= 0: + return None + spent = getattr(parent_agent, "_oneshot_children_spawned", 0) + if spent + requested > cap: + return ( + f"Delegation budget for this one-shot run is exhausted ({spent}/{cap} subagents used; " + f"delegation.oneshot_max_children). Do the remaining work yourself in this session — reviewing " + f"your own diff and running the tests inline is expected here, not a delegated review." + ) + parent_agent._oneshot_children_spawned = spent + requested + return None + + def delegate_task( goal: Optional[str] = None, context: Optional[str] = None, tasks: Optional[List[Dict[str, Any]]] = None, max_iterations: Optional[int] = None, role: Optional[str] = None, background: Optional[bool] = None, @@ -482,6 +502,9 @@ def delegate_task( task_images, err = _coerce_task_images(task_list, images) if err: return tool_error(err) + err = _oneshot_spawn_budget(parent_agent, len(task_list)) + if err: + return tool_error(err) overall_start = time.monotonic() # Live transcripts: cache/delegation/live//task-.log per task, a side channel with zero effect on message diff --git a/tools/delegate_tool_config.py b/tools/delegate_tool_config.py index 6fe745e7f9..e8c399a2e6 100644 --- a/tools/delegate_tool_config.py +++ b/tools/delegate_tool_config.py @@ -82,6 +82,14 @@ def _warn_once(flag_name: str, message: str, *args: Any) -> None: globals()[flag_name] = True logger.warning(message, *args) +def _get_oneshot_max_children() -> int: + """delegation.oneshot_max_children (total children per finite one-shot session; 0 = unlimited).""" + return _knob( + "oneshot_max_children", None, lambda v: max(0, int(v)), 2, + "delegation.oneshot_max_children=%r is not a valid integer; using default 2", + ) + + def _get_max_concurrent_children() -> int: """delegation.max_concurrent_children > DELEGATION_MAX_CONCURRENT_CHILDREN env > 10. diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 3e1116b62c..f1fb6d7a1f 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2797,6 +2797,7 @@ delegation: worktree_isolation: false # Give each child its own git worktree branched from HEAD (local backend + git repos only; inspired by Muse Code). See Subagent Delegation → Worktree Isolation. max_spawn_depth: 1 # Delegation tree depth cap (1-3, clamped). 1 = flat (default): parent spawns leaves that cannot delegate. 2 = orchestrator children can spawn leaf grandchildren. 3 = three levels. orchestrator_enabled: true # Global kill switch. When false, role="orchestrator" is ignored and every child is forced to leaf regardless of max_spawn_depth. + oneshot_max_children: 2 # Total subagents a one-shot run (hermes chat -q / --oneshot) may spawn; 0 = unlimited. Interactive and gateway sessions are never capped by this. ``` **Subagent provider:model override:** By default, subagents inherit the parent agent's provider and model. Set `delegation.provider` and `delegation.model` to route subagents to a different provider:model pair — e.g., use a cheap/fast model for narrowly-scoped subtasks while your primary agent runs an expensive reasoning model. @@ -2823,6 +2824,8 @@ The delegation provider uses the same credential resolution as CLI/gateway start **Precedence:** `delegation.base_url` in config → `delegation.provider` in config → parent provider (inherited). `delegation.model` in config → parent model (inherited). Setting just `model` without `provider` changes only the model name while keeping the parent's credentials (useful for switching models within the same provider like OpenRouter). +**One-shot runs:** a finite `hermes chat -q` / `--oneshot` session has no later turn to consume delegated results and no later session to learn for, so it runs a smaller footprint: `skill_manage` is not offered (skills are still listed and loadable with `skill_view`), the skills prompt asks for domain skills only rather than process skills, and `oneshot_max_children` caps the total subagents the run may spawn (default `2`, `0` = unlimited). Past the cap `delegate_task` returns a tool error telling the agent to finish inline. + **Width and depth:** `max_concurrent_children` caps how many subagents run in parallel per batch (default `3`, floor of 1, no ceiling). Can also be set via the `DELEGATION_MAX_CONCURRENT_CHILDREN` env var. When the model submits a `tasks` array longer than the cap, `delegate_task` returns a tool error explaining the limit rather than silently truncating. `max_spawn_depth` controls the delegation tree depth (clamped to 1-3). At the default `1`, delegation is flat: children cannot spawn grandchildren, and passing `role="orchestrator"` silently degrades to `leaf`. Raise to `2` so orchestrator children can spawn leaf grandchildren; `3` for three-level trees. The agent opts into orchestration per call via `role="orchestrator"`; `orchestrator_enabled: false` forces every child back to leaf regardless. Cost scales multiplicatively — at `max_spawn_depth: 3` with `max_concurrent_children: 3`, the tree can reach 3×3×3 = 27 concurrent leaf agents. See [Subagent Delegation → Depth Limit and Nested Orchestration](features/delegation.md#depth-limit-and-nested-orchestration) for usage patterns. **Child process notifications:** background processes started by subagents route their completion/watch notifications to the parent conversation, but those are **suppressed** there by default — the child's consolidated result is the deliverable. Set `delegation.surface_child_process_notifications: true` to deliver them (with subagent attribution). Delegation results themselves are never suppressed. See [Subagent Delegation → Child background-process notifications](features/delegation.md#child-background-process-notifications). diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index b5795358ad..78c6f07d44 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -543,6 +543,8 @@ delegate_task( **Cost warning:** With `max_spawn_depth: 3` and `max_concurrent_children: 3`, the tree can reach 3×3×3 = 27 concurrent leaf agents. Each extra level multiplies spend — raise `max_spawn_depth` intentionally. +**One-shot runs are capped separately.** `hermes chat -q` / `--oneshot` sessions may spawn at most `delegation.oneshot_max_children` subagents in total (default `2`, `0` = unlimited). A one-shot run has no later turn to receive results, and in benchmark trajectories most of its spawns were "independently review my own work" rather than parallel work — each such child re-pays a cold system prompt and re-reads the repo. Interactive and gateway sessions are unaffected. + ## Lifetime and Durability :::warning Background completion durability is not durable execution