fix(kanban): stop-guard nudge names every worker exit; sync terminal-tool copies

Follow-up to the two salvaged commits. The nudge text asserted the card was
"still `running`" and offered only kanban_complete / kanban_block, so even
when it fired legitimately it steered a review-bound card toward a false
completion. It now states what the transcript actually shows (no terminal
board call yet) and lists kanban_complete / kanban_request_review /
kanban_block, plus the reviewer exits.

Every remaining copy of the "terminal tools" knowledge is brought in line:
turn_stop_gates docstring + diagnostic status, goals.py finalize comment,
kanban_db_dispatch grace/concurrency comments, the orchestrator-only
refusal in kanban_tools, and the kanban / tutorial / codex-runtime docs.

Fixes #114598
This commit is contained in:
teknium1
2026-09-18 00:38:52 -07:00
committed by Teknium
parent edd14100f9
commit 41fe679d96
9 changed files with 38 additions and 21 deletions

View File

@@ -77,16 +77,21 @@ def build_kanban_stop_nudge(
return None
tid = (task_id or os.environ.get("HERMES_KANBAN_TASK") or "").strip() or "this task"
# The transcript is the status source: this text is only reached when the session made no
# handoff call, so it never tells a worker to close a card it already sent to review.
return (
"[System: You are a Hermes kanban worker. A plain-text reply is NOT a "
"terminal state for the board.\n\n"
f"Task `{tid}` is still `running`. Ending now without a board tool "
"causes a protocol violation (clean exit with no "
"`kanban_complete` / `kanban_block`).\n\n"
f"Task `{tid}` has not been handed off: this session made no terminal board "
"call (`kanban_complete` / `kanban_request_review` / `kanban_block`). Ending now "
"causes a protocol violation (clean exit with the card still `running`).\n\n"
"Do this immediately in your next response — do not narrate intent:\n"
"1. Finish any remaining deliverable (write the required file(s) now).\n"
"2. Call `kanban_complete(summary=..., artifacts=[...])` if the work "
"is done, OR `kanban_block(reason=...)` if you are blocked.\n\n"
"2. Call `kanban_complete(summary=..., artifacts=[...])` if the work is done "
"and needs no review, `kanban_request_review(summary=...)` if it is a code "
"change that needs same-card review, OR `kanban_block(reason=...)` if you are "
"blocked. Reviewers approve with `kanban_complete` or send the card back with "
"`kanban_request_changes(reason=...)`.\n\n"
"Never end a turn with only a promise of future action. Repeated "
"protocol violations will block this task and require manual intervention.]"
)

View File

@@ -77,7 +77,8 @@ def _pre_verify_nudge(agent, final_response, attempt: int) -> Optional[str]:
def _kanban_stop_nudge(agent, messages) -> Optional[str]:
"""Workers must end with kanban_complete / kanban_block; a narrated stop is recorded
"""Workers must end with a terminal board tool (kanban_complete / kanban_block /
kanban_request_review / kanban_request_changes); a narrated stop is recorded
as protocol_violation, so nudge once or twice first."""
try:
from agent.kanban_stop import build_kanban_stop_nudge
@@ -163,8 +164,8 @@ def apply_stop_gates(
os.environ.get("HERMES_KANBAN_TASK", ""),
)
agent._emit_diagnostic_status(
"⚠️ Kanban worker tried to exit without "
"kanban_complete/kanban_block — nudging to finish"
"⚠️ Kanban worker tried to exit without a terminal board call "
"(kanban_complete/kanban_request_review/kanban_block) — nudging to finish"
)
return verdict
return StopGateVerdict(

View File

@@ -1563,7 +1563,8 @@ KANBAN_GOAL_CONTINUATION_TEMPLATE = (
"calling one of them."
)
# Judge says done but the worker never called kanban_complete/kanban_block: one explicit nudge.
# Judge says done but the worker never made a terminal board call
# (kanban_complete/kanban_request_review/kanban_block): one explicit nudge.
KANBAN_GOAL_FINALIZE_TEMPLATE = (
"[The work looks complete, but the task is still open]\n"
"Reason: {reason}\n\n"

View File

@@ -38,7 +38,8 @@ DEFAULT_LOG_ROTATE_BYTES = 2 * 1024 * 1024 # 2 MiB
DEFAULT_LOG_BACKUP_COUNT = 1
# Keep a little wall-clock budget for the worker to observe a terminal timeout
# and call kanban_block/kanban_complete before max_runtime_seconds kills it.
# and make a terminal board call (kanban_block/kanban_complete/kanban_request_review)
# before max_runtime_seconds kills it.
KANBAN_TERMINAL_TIMEOUT_GRACE_SECONDS = 30
# A healthy worker is still alive for a while after kanban_complete /
@@ -1991,8 +1992,8 @@ def _tick_spawn_budget(
cap by N — exactly the fan-out the memory-derived default exists to prevent.
"""
# Count already-running tasks so max_spawn enforces concurrency, not a
# per-tick budget: "running" tasks stay running until the worker calls
# kanban_complete/kanban_block or the TTL reclaims them.
# per-tick budget: "running" tasks stay running until the worker makes a terminal
# board call (kanban_complete/kanban_block/kanban_request_review) or the TTL reclaims them.
running_count = 0
spawn_budget: Optional[int] = None
if max_spawn is not None or max_in_progress is not None:

View File

@@ -167,4 +167,8 @@ def test_nudge_still_fires_for_non_terminal_kanban_tool(clear_kanban_env):
{"role": "tool", "name": "kanban_comment", "tool_call_id": "1", "content": "ok"},
]
assert session_called_kanban_terminal(messages) is False
assert build_kanban_stop_nudge(messages=messages) is not None
nudge = build_kanban_stop_nudge(messages=messages)
assert nudge is not None
# The nudge offers every worker exit, not just close-out; a card that must go
# through review must never be steered to ``kanban_complete`` alone.
assert "kanban_request_review" in nudge and "kanban_block" in nudge

View File

@@ -213,8 +213,8 @@ def _require_orchestrator_tool(tool_name: str) -> None:
if os.environ.get("HERMES_KANBAN_TASK"):
raise _Reject(
f"{tool_name} is orchestrator-only; dispatcher-spawned workers must use "
"kanban_complete, kanban_block, kanban_heartbeat, or kanban_comment for their "
"assigned task.")
"kanban_complete, kanban_request_review, kanban_request_changes, kanban_block, "
"kanban_heartbeat, or kanban_comment for their assigned task.")
@contextmanager

View File

@@ -95,7 +95,7 @@ What works inside a codex-runtime worker:
- The Hermes tool callback for browser_*, vision, image_gen, skills, TTS
What also works because the MCP callback exposes them:
- **`kanban_complete` / `kanban_block` / `kanban_comment` / `kanban_heartbeat`** — the worker handoff tools. These read `HERMES_KANBAN_TASK` from env (set by the dispatcher), gate access correctly, and write to the per-board SQLite DB pinned by `HERMES_KANBAN_DB`. Without these in the callback, a worker on this runtime could do its task but couldn't report back, hanging until the dispatcher's timeout.
- **`kanban_complete` / `kanban_request_review` / `kanban_request_changes` / `kanban_block` / `kanban_comment` / `kanban_heartbeat`** — the worker handoff tools. These read `HERMES_KANBAN_TASK` from env (set by the dispatcher), gate access correctly, and write to the per-board SQLite DB pinned by `HERMES_KANBAN_DB`. Without these in the callback, a worker on this runtime could do its task but couldn't report back, hanging until the dispatcher's timeout.
- **`kanban_show` / `kanban_list`** — read-only board queries for the worker to check its own context.
- **`kanban_create` / `kanban_unblock` / `kanban_link`** — orchestrator-only operations. Available for orchestrator agents running on the codex runtime that need to dispatch new tasks.

View File

@@ -10,7 +10,7 @@ hermes dashboard # opens http://127.0.0.1:9119 in your browser
# click Kanban in the left nav
```
The dashboard is the most comfortable place for **you** to watch the system. Agent workers the dispatcher spawns never see the dashboard or the CLI — they drive the board through a dedicated `kanban_*` [toolset](./kanban#how-workers-interact-with-the-board) (`kanban_show`, `kanban_list`, `kanban_complete`, `kanban_block`, `kanban_heartbeat`, `kanban_comment`, `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_create`, `kanban_link`, `kanban_unblock`). All three surfaces — dashboard, CLI, worker tools — route through the same per-board SQLite DB (`~/.hermes/kanban.db` for the default board, `~/.hermes/kanban/boards/<slug>/kanban.db` for any board you create later), so each board is consistent no matter which side of the fence a change came from.
The dashboard is the most comfortable place for **you** to watch the system. Agent workers the dispatcher spawns never see the dashboard or the CLI — they drive the board through a dedicated `kanban_*` [toolset](./kanban#how-workers-interact-with-the-board) (`kanban_show`, `kanban_list`, `kanban_complete`, `kanban_request_review`, `kanban_request_changes`, `kanban_block`, `kanban_heartbeat`, `kanban_comment`, `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_create`, `kanban_link`, `kanban_unblock`). All three surfaces — dashboard, CLI, worker tools — route through the same per-board SQLite DB (`~/.hermes/kanban.db` for the default board, `~/.hermes/kanban/boards/<slug>/kanban.db` for any board you create later), so each board is consistent no matter which side of the fence a change came from.
This tutorial uses the `default` board throughout. If you want multiple isolated queues (one per project / repo / domain), see [Boards (multi-project)](./kanban#boards-multi-project) in the overview — the same CLI / dashboard / worker flows apply per board, and workers physically cannot see tasks on other boards.

View File

@@ -498,9 +498,11 @@ Every profile that works kanban tasks automatically gets the worker lifecycle
1. On spawn, call `kanban_show()` to read title + body + parent handoffs + prior attempts + full comment thread.
2. `cd $HERMES_KANBAN_WORKSPACE` (via the terminal tool) and do the work there.
3. Call `kanban_heartbeat(note="...")` every few minutes during long operations. **If your work may run longer than 1 hour, call `kanban_heartbeat` at least once an hour** — the dispatcher reclaims tasks that have been running past `kanban.dispatch_stale_timeout_seconds` (default 4 h) with no heartbeat in the last hour, on the assumption the worker crashed without cleanup. A reclaim is benign (the task goes back to `ready` for re-dispatch without a failure-counter tick) but you lose your current run's progress.
4. Complete with `kanban_complete(summary="...", metadata={...})`, or `kanban_block(reason="...")` if stuck.
4. Complete with `kanban_complete(summary="...", metadata={...})`, hand a code change off for same-card review with `kanban_request_review(summary="...")`, or `kanban_block(reason="...")` if stuck.
That final `kanban_complete` / `kanban_block` call is part of the worker
That final terminal board call (`kanban_complete` / `kanban_request_review` /
`kanban_block`; reviewers end with `kanban_complete` or `kanban_request_changes`)
is part of the worker
protocol. If the worker process exits with status 0 while the task is still
`running`, the dispatcher treats that as a protocol violation and emits a
`protocol_violation` event. A dispatcher-spawned worker whose turn failed
@@ -514,7 +516,10 @@ booked as a protocol violation.
synthetic nudges when it detects the model is about to stop without a terminal
board tool call. This catches the common case where the model narrates the next
step ("Let me write the report") and stops with `finish_reason=stop`. The nudge
reminds the model to call `kanban_complete` or `kanban_block` immediately. This
reminds the model to call `kanban_complete`, `kanban_request_review` or `kanban_block`
immediately. A worker that already handed its card off (`kanban_request_review`, or
`kanban_request_changes` from a reviewer) is never nudged — the handoff is its terminal
call, and the nudge never asks it to `kanban_complete` a card that is under review. This
guard is active only for the dispatcher-spawned worker itself (`HERMES_KANBAN_TASK` is
set and the run owns that task) — `delegate_task` children and cron jobs run inside
the worker inherit the variable but are never nudged, since they have no board tools —
@@ -1321,7 +1326,7 @@ Every transition appends a row to `task_events`. Each row carries an optional `r
| `reconciled` | `{reason, claim_lock, claim_expires, worker_pid}` | Orphaned-card reconciliation: the card was `running` with broken claim bookkeeping (`claim_lock` or `claim_expires` NULL — crash mid-claim, manual SQL, DB restore) and no live worker, so none of the TTL/crash/stale paths could ever recover it. The dispatcher requeued it to `ready` with an explanatory comment. Gated by `kanban.reconcile_orphans` in config.yaml (default `true`). |
| `respawn_guarded` | `{reason}` | Dispatcher refused to re-spawn this ready task this tick. Reasons: `blocker_auth` (last failure was a quota/auth/429 error — wait for the rate window to reset), `recent_success` (a completed run happened in the last hour — wait for review before re-running), `active_pr` (a GitHub PR URL appears in a recent comment — a prior worker already opened a PR). The task stays in `ready`; the next tick gets another chance to spawn. If the underlying condition persists, the normal `consecutive_failures` circuit breaker will auto-block via `gave_up` after `failure_limit` failures. |
| `spawn_failed` | `{error, failures}` | One spawn attempt failed (missing PATH, workspace unmountable, …). Counter increments; task returns to `ready` for retry. |
| `protocol_violation` | `{pid, claimer, exit_code, protocol_violation, worker_output?}` | Worker exited successfully while the task was still `running`, usually because it answered without calling `kanban_complete` or `kanban_block`. Emitted on every violation (the payload's `protocol_violation: true` marker is copied into the run metadata and feeds the violation-only retry budget). Below the budget — up to `_PROTOCOL_VIOLATION_FAILURE_LIMIT` (default 3) *consecutive* violations, per-task `max_retries` overriding — the task simply returns to `ready` for another attempt; when the streak reaches the bound the dispatcher also emits `gave_up` and auto-blocks. `worker_output` carries the worker's own last printed text (usually its explanation of why it stopped), also folded into `last_failure_error` and shown to the retry worker as the prior-attempt error. |
| `protocol_violation` | `{pid, claimer, exit_code, protocol_violation, worker_output?}` | Worker exited successfully while the task was still `running`, usually because it answered without a terminal board call (`kanban_complete`, `kanban_request_review` or `kanban_block`). Emitted on every violation (the payload's `protocol_violation: true` marker is copied into the run metadata and feeds the violation-only retry budget). Below the budget — up to `_PROTOCOL_VIOLATION_FAILURE_LIMIT` (default 3) *consecutive* violations, per-task `max_retries` overriding — the task simply returns to `ready` for another attempt; when the streak reaches the bound the dispatcher also emits `gave_up` and auto-blocks. `worker_output` carries the worker's own last printed text (usually its explanation of why it stopped), also folded into `last_failure_error` and shown to the retry worker as the prior-attempt error. |
| `gave_up` | `{failures, effective_limit, limit_source, error}` | Circuit breaker fired after N consecutive non-successful attempts. Task auto-blocks with the last error. The effective limit resolves as task `max_retries`, then dispatcher `failure_limit` / `kanban.failure_limit`, then the built-in default. |
`hermes kanban tail <id>` shows these for a single task. `hermes kanban watch` streams them board-wide.