5725 lines
222 KiB
Python
5725 lines
222 KiB
Python
"""SQLite-backed Kanban board for multi-profile, multi-project collaboration.
|
||
|
||
The board lives under the **shared Hermes root** ``<root>`` (the parent of any
|
||
active profile; ``HERMES_HOME`` itself in Docker / custom deployments).
|
||
Profiles intentionally collapse onto a shared board — it IS the cross-profile
|
||
coordination primitive: a worker spawned with ``hermes -p <profile>`` joins the
|
||
same board as the dispatcher that claimed its task.
|
||
|
||
**Boards:** each extra board is ``<root>/kanban/boards/<slug>/`` with its own
|
||
``kanban.db``, ``workspaces/`` and ``logs/``; a worker on one board cannot see
|
||
or enumerate others and its dispatcher ticks never touch their DBs. The first
|
||
board is ``default`` and, for back-compat, its DB stays at ``<root>/kanban.db``
|
||
so pre-boards installs need zero migration (see :func:`kanban_db_path`).
|
||
|
||
Board resolution order (highest precedence first, all optional):
|
||
|
||
* ``board=`` argument to :func:`connect` / :func:`init_db` (CLI ``--board``,
|
||
dashboard ``?board=``).
|
||
* ``HERMES_KANBAN_BOARD`` env var (dispatcher pins workers to their board).
|
||
* ``HERMES_KANBAN_DB`` env var (pins the DB file path directly; legacy
|
||
override, wins when the file path itself is what the caller forces).
|
||
* ``<root>/kanban/current`` — one-line slug file written by
|
||
``hermes kanban boards switch``; absent → ``default``.
|
||
|
||
Legacy overrides ``HERMES_KANBAN_DB`` / ``HERMES_KANBAN_WORKSPACES_ROOT`` /
|
||
``HERMES_KANBAN_HOME`` (umbrella root; tests and unusual deployments) still
|
||
work. The dispatcher injects the DB, workspaces-root and board env vars into
|
||
worker subprocesses so they converge on the exact DB it claimed from, even
|
||
under unusual symlink or Docker layouts.
|
||
|
||
Schema: tasks, task_links, task_comments, task_events (+ runs, attachments,
|
||
notify subs). ``workspace_kind`` decouples coordination from git worktrees so
|
||
research / ops workloads run alongside coding. See
|
||
``docs/hermes-kanban-v1-spec.pdf``.
|
||
|
||
Concurrency: WAL + ``BEGIN IMMEDIATE`` write transactions + compare-and-swap
|
||
updates on ``tasks.status`` / ``tasks.claim_lock``. SQLite serializes writers,
|
||
so at most one claimer wins a task; losers see zero affected rows and move on
|
||
— no retry loops, no distributed locks. CAS is per-board (one DB per board).
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import contextlib
|
||
import json
|
||
import os
|
||
import re
|
||
import secrets
|
||
import sqlite3
|
||
import subprocess
|
||
import sys
|
||
import logging
|
||
import time
|
||
from contextvars import ContextVar, Token
|
||
from dataclasses import dataclass
|
||
from pathlib import Path
|
||
from typing import Any, Iterable, Optional
|
||
|
||
from toolsets import get_toolset_names
|
||
|
||
_log = logging.getLogger(__name__)
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Shared micro-helpers (row access, JSON, env, git)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _row_get(row: Any, col: str, default: Any = None) -> Any:
|
||
"""``row[col]`` tolerant of the column being absent from the SELECT / schema."""
|
||
if row is None or col not in row.keys():
|
||
return default
|
||
return row[col]
|
||
|
||
|
||
def _json_or(value: Any, default: Any = None) -> Any:
|
||
"""Decode a JSON text column; any decode failure or empty value yields ``default``."""
|
||
if not value:
|
||
return default
|
||
try:
|
||
return json.loads(value)
|
||
except Exception:
|
||
return default
|
||
|
||
|
||
def _json_dict(value: Any) -> dict:
|
||
"""Decode a JSON text column that must be an object; anything else yields ``{}``."""
|
||
parsed = _json_or(value, {})
|
||
return parsed if isinstance(parsed, dict) else {}
|
||
|
||
|
||
def _env_int(name: str, default: int, *, minimum: int = 0) -> int:
|
||
"""Integer env override: absent/empty/non-integer/below ``minimum`` falls back to ``default``."""
|
||
raw = os.environ.get(name, "").strip()
|
||
if raw:
|
||
try:
|
||
parsed = int(raw)
|
||
except ValueError:
|
||
return default
|
||
if parsed >= minimum:
|
||
return parsed
|
||
return default
|
||
|
||
|
||
def _git_out(cwd: Path, *args: str, timeout: int = 30) -> Optional[str]:
|
||
"""Run ``git -C cwd args`` and return stripped stdout, or ``None`` on any failure / empty output."""
|
||
try:
|
||
result = subprocess.run(
|
||
["git", "-C", str(cwd), *args],
|
||
capture_output=True, text=True, encoding="utf-8", errors="replace",
|
||
timeout=timeout, check=False,
|
||
)
|
||
except Exception:
|
||
return None
|
||
if result.returncode != 0:
|
||
return None
|
||
return (result.stdout or "").strip() or None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Constants
|
||
# ---------------------------------------------------------------------------
|
||
|
||
VALID_STATUSES = {"triage", "todo", "scheduled", "ready", "running", "blocked", "review", "done", "archived"}
|
||
VALID_INITIAL_STATUSES = {"running", "blocked"}
|
||
|
||
# Typed block reasons (routing rules live on ``block_task``): ``dependency``
|
||
# -> ``todo`` (parent gating promotes it, no human/cron/retry storm);
|
||
# ``needs_input`` / ``capability`` -> ``blocked`` for a human; ``transient`` =
|
||
# may clear on retry. ``None`` = legacy un-typed block (generic human blocker).
|
||
VALID_BLOCK_KINDS = {"dependency", "needs_input", "capability", "transient"}
|
||
|
||
# Same-reason block -> unblock -> re-block cycles tolerated before the loop
|
||
# breaker stops trusting the unblocker (usually a cron) and routes to ``triage``
|
||
# for a human decision. Counts manual unblock recurrences, NOT dispatcher
|
||
# spawn/crash/timeout failures (that is ``DEFAULT_FAILURE_LIMIT``).
|
||
BLOCK_RECURRENCE_LIMIT = 2
|
||
VALID_WORKSPACE_KINDS = {"scratch", "worktree", "dir"}
|
||
|
||
|
||
def normalize_reasoning_effort(effort: Optional[str]) -> Optional[str]:
|
||
"""Normalize a per-task reasoning effort into a storable level.
|
||
|
||
Accepts any level in ``hermes_constants.VALID_REASONING_EFFORTS`` plus
|
||
``"none"`` (thinking disabled), case-insensitively. Empty / None means
|
||
"inherit the worker profile's own ``agent.reasoning_effort``" and stores
|
||
NULL. Anything else is rejected rather than silently dropped — a typo'd
|
||
level must not quietly hand the task back to the profile default.
|
||
"""
|
||
from hermes_constants import VALID_REASONING_EFFORTS
|
||
|
||
value = str(effort or "").strip().lower()
|
||
if not value:
|
||
return None
|
||
if value == "none" or value in VALID_REASONING_EFFORTS:
|
||
return value
|
||
allowed = ", ".join(("none", *VALID_REASONING_EFFORTS))
|
||
raise ValueError(
|
||
f"reasoning_effort must be one of {allowed}, got {effort!r}"
|
||
)
|
||
|
||
|
||
KNOWN_TOOLSET_NAMES = frozenset(name.casefold() for name in get_toolset_names())
|
||
_IS_WINDOWS = sys.platform == "win32"
|
||
KANBAN_ATTACHMENT_MAX_BYTES = 25 * 1024 * 1024
|
||
|
||
|
||
def _assert_not_delegated_child_mutation() -> None:
|
||
"""Reject Kanban state mutations from ``delegate_task`` child contexts.
|
||
|
||
The structured kanban tools and CLI dispatch layer both have fast-fail
|
||
guards for better UX, but neither is a trust boundary: a delegated child can
|
||
still shell out to the CLI or import this module directly. The actual
|
||
invariant belongs at the DB/filesystem mutation layer so every public
|
||
mutator that uses ``write_txn`` (tasks, runs, comments, attachments,
|
||
dispatcher claims, repair events, subscriptions, GC, etc.) and every board
|
||
metadata mutator fails closed before touching durable state.
|
||
"""
|
||
try:
|
||
from agent.delegation_context import is_delegated_child_process_context
|
||
|
||
delegated = is_delegated_child_process_context()
|
||
except Exception:
|
||
delegated = bool(os.environ.get("HERMES_DELEGATED_CHILD_CONTEXT"))
|
||
if delegated:
|
||
raise PermissionError(
|
||
"delegate_task child contexts cannot mutate Kanban tasks or boards"
|
||
)
|
||
|
||
|
||
def _fire_kanban_lifecycle_hook(event: str, task_id: str, **fields: Any) -> None:
|
||
"""Fire a lifecycle plugin hook, best-effort. Callers invoke it AFTER their
|
||
write txn commits so plugins never run under a SQLite write lock and
|
||
always see durable state; any failure is swallowed so an observer can
|
||
never break a transition. ``profile_name`` comes from the active
|
||
HERMES_HOME so dispatcher- and worker-side hooks agree without plumbing.
|
||
"""
|
||
try:
|
||
from hermes_cli.lifecycle import invoke_hook
|
||
|
||
invoke_hook(event, task_id=task_id, profile_name=_hook_profile_name(), **fields)
|
||
except Exception as exc: # pragma: no cover - defensive
|
||
_log.debug("kanban lifecycle hook %s failed: %s", event, exc)
|
||
|
||
|
||
def _hook_profile_name() -> str:
|
||
"""Active profile for hook payloads; ``"default"`` when it cannot be resolved."""
|
||
from hermes_cli.profiles import get_active_profile_name
|
||
|
||
try:
|
||
return get_active_profile_name()
|
||
except Exception:
|
||
return "default"
|
||
|
||
|
||
def _kanban_observer_consumed(event: str) -> bool:
|
||
"""Whether any observer/plugin consumes *event* — hot-path short-circuit
|
||
so per-tick / per-write observers skip payload assembly when nothing
|
||
subscribes. If inspection fails the event counts as unconsumed (the
|
||
invoke path would fail identically; dropping an observer is always safe).
|
||
"""
|
||
try:
|
||
from hermes_cli.lifecycle import has_hook
|
||
|
||
return has_hook(event)
|
||
except Exception: # pragma: no cover - defensive
|
||
return False
|
||
|
||
|
||
def _fire_worker_spawned_hook(
|
||
conn: sqlite3.Connection,
|
||
task: "Task",
|
||
workspace_path: str,
|
||
pid: Optional[int],
|
||
*,
|
||
board: Optional[str] = None,
|
||
) -> None:
|
||
"""Fire ``on_kanban_worker_spawned`` AFTER ``spawn_fn`` returned and the
|
||
reported PID is durably persisted. Best-effort: a misbehaving observer can
|
||
never break the dispatch loop.
|
||
"""
|
||
if not _kanban_observer_consumed("on_kanban_worker_spawned"):
|
||
return
|
||
try:
|
||
_fire_kanban_lifecycle_hook(
|
||
"on_kanban_worker_spawned",
|
||
task.id,
|
||
board=board or get_current_board(),
|
||
assignee=task.assignee,
|
||
run_id=_current_run_id(conn, task.id),
|
||
worker_pid=int(pid) if pid else None,
|
||
workspace_path=str(workspace_path),
|
||
)
|
||
except Exception as exc: # pragma: no cover - defensive
|
||
_log.debug("kanban worker spawned hook failed: %s", exc)
|
||
|
||
|
||
def notify_task_updated(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
changed_fields: Iterable[str],
|
||
*,
|
||
board: Optional[str] = None,
|
||
) -> None:
|
||
"""Fire ``on_kanban_task_updated`` AFTER a task-row mutation outside the
|
||
claim/complete/block lifecycle has committed — including direct-SQL
|
||
surfaces that bypass every ``kanban_db`` mutator (dashboard field
|
||
editors). ``changed_fields`` carries field NAMES only, never values.
|
||
Observer-only, best-effort; one ``has_hook`` probe when nothing subscribes.
|
||
"""
|
||
if not _kanban_observer_consumed("on_kanban_task_updated"):
|
||
return
|
||
try:
|
||
row = conn.execute(
|
||
"SELECT assignee, current_run_id FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
_fire_kanban_lifecycle_hook(
|
||
"on_kanban_task_updated",
|
||
task_id,
|
||
board=board or get_current_board(),
|
||
assignee=row["assignee"] if row else None,
|
||
run_id=row["current_run_id"] if row else None,
|
||
changed_fields=list(changed_fields),
|
||
)
|
||
except Exception as exc: # pragma: no cover - defensive
|
||
_log.debug("kanban task updated hook failed: %s", exc)
|
||
|
||
|
||
def _fire_dispatch_tick_hook(
|
||
result: "DispatchResult",
|
||
*,
|
||
board: Optional[str] = None,
|
||
dry_run: bool = False,
|
||
) -> None:
|
||
"""Fire ``on_kanban_dispatch_tick`` after one tick — strictly AFTER
|
||
``_dispatch_tick_lock`` is released, so a slow subscriber cannot extend
|
||
the single-writer critical section and stall a sibling dispatcher.
|
||
Observer-only, best-effort: subscriber failures are swallowed.
|
||
"""
|
||
if not _kanban_observer_consumed("on_kanban_dispatch_tick"):
|
||
return
|
||
try:
|
||
from hermes_cli.lifecycle import invoke_hook
|
||
|
||
profile_name = _hook_profile_name()
|
||
if board is None:
|
||
try:
|
||
board = get_current_board()
|
||
except Exception:
|
||
board = None
|
||
outcome = "ok"
|
||
if result.skipped_locked:
|
||
outcome = "skipped_locked"
|
||
elif not any((
|
||
result.spawned,
|
||
result.reclaimed,
|
||
result.promoted,
|
||
result.reconciled_orphans,
|
||
result.crashed,
|
||
result.stale,
|
||
result.timed_out,
|
||
result.auto_blocked,
|
||
result.rate_limited,
|
||
result.auto_assigned_default,
|
||
result.respawn_guarded,
|
||
result.skipped_per_profile_capped,
|
||
result.skipped_unassigned,
|
||
result.skipped_nonspawnable,
|
||
)):
|
||
outcome = "idle"
|
||
invoke_hook(
|
||
"on_kanban_dispatch_tick",
|
||
board=board,
|
||
profile_name=profile_name,
|
||
dry_run=bool(dry_run),
|
||
outcome=outcome,
|
||
result=result,
|
||
)
|
||
except Exception as exc: # pragma: no cover - defensive
|
||
_log.debug("kanban dispatch tick hook failed: %s", exc)
|
||
|
||
|
||
# Claim window before the next tick reclaims a running task; long workers
|
||
# ``heartbeat_claim`` or raise it via HERMES_KANBAN_CLAIM_TTL_SECONDS.
|
||
DEFAULT_CLAIM_TTL_SECONDS = 15 * 60
|
||
|
||
# A live PID whose ``last_heartbeat_at`` is older than this is treated as
|
||
# wedged (logic loop, no observable progress) and reclaimed regardless of
|
||
# liveness. ``_touch_activity`` bridges chunk-level API liveness into the
|
||
# heartbeat, so a genuinely active worker never trips this.
|
||
DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS = 60 * 60
|
||
|
||
# Grace added when a reclaim is deferred because the host-local worker
|
||
# survived a termination attempt (cgroup memory.high throttle parking it in
|
||
# D state, SIGKILL pending). Releasing the claim then would spawn a duplicate
|
||
# beside the survivor; holding it a tick lets the signal land.
|
||
RECLAIM_DEFER_GRACE_SECONDS = 120
|
||
|
||
|
||
def _resolve_claim_ttl_seconds(ttl_seconds: Optional[int] = None) -> int:
|
||
"""Explicit ``ttl_seconds`` wins; else a positive
|
||
``HERMES_KANBAN_CLAIM_TTL_SECONDS`` overrides ``DEFAULT_CLAIM_TTL_SECONDS``
|
||
(invalid values fall back silently so existing installs keep working)."""
|
||
if ttl_seconds is not None:
|
||
return max(1, int(ttl_seconds))
|
||
|
||
return _env_int("HERMES_KANBAN_CLAIM_TTL_SECONDS", DEFAULT_CLAIM_TTL_SECONDS, minimum=1)
|
||
|
||
|
||
# After ``running`` starts, ``detect_crashed_workers`` skips ``_pid_alive`` for
|
||
# this long: the fork() -> /proc-visibility window can transiently report a
|
||
# fresh worker dead. The claim TTL still catches real crashes.
|
||
DEFAULT_CRASH_GRACE_SECONDS = 30
|
||
|
||
# Worker exit code meaning "provider rate-limited / quota exhausted, not a task
|
||
# failure": the reap classifier maps it to ``rate_limited`` so the task is
|
||
# released WITHOUT counting a failure (the breaker must never trip on a
|
||
# throttle). 75 == BSD EX_TEMPFAIL, clear of the worker's 0/1/2 codes.
|
||
KANBAN_RATE_LIMIT_EXIT_CODE = 75
|
||
|
||
|
||
def _resolve_crash_grace_seconds() -> int:
|
||
"""``HERMES_KANBAN_CRASH_GRACE_SECONDS`` (0 = immediate reclaim, for tests)
|
||
else ``DEFAULT_CRASH_GRACE_SECONDS``."""
|
||
return _env_int("HERMES_KANBAN_CRASH_GRACE_SECONDS", DEFAULT_CRASH_GRACE_SECONDS)
|
||
|
||
|
||
def _resolve_rate_limit_cooldown_seconds() -> int:
|
||
"""``HERMES_KANBAN_RATE_LIMIT_COOLDOWN_SECONDS`` (0 = respawn next tick, for
|
||
tests) else ``DEFAULT_RATE_LIMIT_COOLDOWN_SECONDS``."""
|
||
return _env_int("HERMES_KANBAN_RATE_LIMIT_COOLDOWN_SECONDS", DEFAULT_RATE_LIMIT_COOLDOWN_SECONDS)
|
||
|
||
|
||
# build_worker_context() caps (independently tunable) so the prompt stays
|
||
# bounded on pathological boards; sized for a ~100k-char prompt with headroom.
|
||
_CTX_MAX_PRIOR_ATTEMPTS = 10 # most recent N prior runs shown in full
|
||
_CTX_MAX_COMMENTS = 30 # most recent N comments shown in full
|
||
_CTX_MAX_FIELD_BYTES = 4 * 1024 # per summary/error/metadata/result
|
||
_CTX_MAX_BODY_BYTES = 8 * 1024 # per task.body (opening post)
|
||
_CTX_MAX_COMMENT_BYTES = 2 * 1024 # per comment
|
||
|
||
|
||
def _relative_age(ts: Optional[int], now: Optional[int] = None) -> str:
|
||
"""Coarse relative age (``just now`` / ``18h ago`` / ``3d ago``); "" for a
|
||
missing/invalid timestamp so callers append unconditionally. An LLM reads
|
||
a bare absolute timestamp as current fact; the relative age is what
|
||
prompts a worker to re-verify stale sibling work before acting on it.
|
||
"""
|
||
if ts is None:
|
||
return ""
|
||
try:
|
||
ts = int(ts)
|
||
except (TypeError, ValueError):
|
||
return ""
|
||
if now is None:
|
||
now = int(time.time())
|
||
delta = now - ts
|
||
if delta < 0:
|
||
# Clock skew across machines/profiles — don't claim "in the future".
|
||
return "just now"
|
||
if delta < 60:
|
||
return "just now"
|
||
if delta < 3600:
|
||
m = delta // 60
|
||
return f"{m}m ago"
|
||
if delta < 86400:
|
||
h = delta // 3600
|
||
return f"{h}h ago"
|
||
d = delta // 86400
|
||
return f"{d}d ago"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Paths
|
||
# ---------------------------------------------------------------------------
|
||
|
||
DEFAULT_BOARD = "default"
|
||
_CURRENT_BOARD_OVERRIDE: ContextVar[str | None] = ContextVar(
|
||
"hermes_kanban_current_board_override",
|
||
default=None,
|
||
)
|
||
|
||
|
||
@contextlib.contextmanager
|
||
def scoped_current_board(slug: str):
|
||
"""Temporarily pin the active board for the current context only."""
|
||
token: Token[str | None] = _CURRENT_BOARD_OVERRIDE.set(slug)
|
||
try:
|
||
yield
|
||
finally:
|
||
_CURRENT_BOARD_OVERRIDE.reset(token)
|
||
|
||
# Slug validator: lowercase alphanumerics, digits, hyphens; 1–64 chars.
|
||
# Strict enough to stop traversal (`..`) and embedded path separators, loose
|
||
# enough that kebab-case names like ``atm10-server`` or ``hermes-agent``
|
||
# pass without fuss. Board names with display formatting (spaces, emoji)
|
||
# live in ``board.json``; the slug is just the directory name.
|
||
_BOARD_SLUG_RE = re.compile(r"^[a-z0-9][a-z0-9\-_]{0,63}$")
|
||
|
||
|
||
def _normalize_board_slug(slug: Optional[str]) -> Optional[str]:
|
||
"""Lowercase + strip a slug; validate; return ``None`` for empty."""
|
||
if slug is None:
|
||
return None
|
||
s = str(slug).strip().lower()
|
||
if not s:
|
||
return None
|
||
if not _BOARD_SLUG_RE.match(s):
|
||
raise ValueError(
|
||
f"invalid board slug {slug!r}: must be 1-64 chars, lowercase "
|
||
f"alphanumerics / hyphens / underscores, not starting with '-' or '_'"
|
||
)
|
||
return s
|
||
|
||
|
||
def kanban_home() -> Path:
|
||
"""Shared Hermes root anchoring the board: ``HERMES_KANBAN_HOME`` if set,
|
||
else ``get_default_hermes_root()`` (``<root>`` for a
|
||
``<root>/profiles/<name>`` HERMES_HOME, HERMES_HOME itself otherwise).
|
||
|
||
The board is shared across profiles BY DESIGN; resolving through the
|
||
active profile's HERMES_HOME would silently fork it per profile and break
|
||
the dispatcher / worker handoff.
|
||
"""
|
||
override = os.environ.get("HERMES_KANBAN_HOME", "").strip()
|
||
if override:
|
||
return Path(override).expanduser()
|
||
from hermes_constants import get_default_hermes_root
|
||
return get_default_hermes_root()
|
||
|
||
|
||
def boards_root() -> Path:
|
||
"""Return ``<root>/kanban/boards`` — the parent of non-default board dirs.
|
||
|
||
``default`` is intentionally NOT under this directory — its DB lives at
|
||
``<root>/kanban.db`` for back-compat with pre-boards installs. This
|
||
function returns the directory where *additional* named boards live,
|
||
used by :func:`list_boards` to enumerate them.
|
||
"""
|
||
return kanban_home() / "kanban" / "boards"
|
||
|
||
|
||
def current_board_path() -> Path:
|
||
"""Return the path to ``<root>/kanban/current``.
|
||
|
||
One-line text file written by ``hermes kanban boards switch <slug>``
|
||
to persist the user's board selection across CLI invocations. Absent
|
||
by default (meaning: active board is ``default``).
|
||
"""
|
||
return kanban_home() / "kanban" / "current"
|
||
|
||
|
||
def get_current_board() -> str:
|
||
"""Active board slug: ``HERMES_KANBAN_BOARD`` env (dispatcher injects it
|
||
on spawn) -> ``<root>/kanban/current`` (``boards switch``, only while that
|
||
board still exists) -> ``DEFAULT_BOARD``.
|
||
|
||
A malformed/stale slug falls through to the next layer with a warning —
|
||
the dispatcher must never crash because a user hand-edited a file or
|
||
removed a board directory.
|
||
"""
|
||
def _existing(candidate: str) -> Optional[str]:
|
||
if not candidate:
|
||
return None
|
||
try:
|
||
normed = _normalize_board_slug(candidate)
|
||
except ValueError:
|
||
return None
|
||
return normed if normed and board_exists(normed) else None
|
||
|
||
for candidate in (
|
||
(_CURRENT_BOARD_OVERRIDE.get() or "").strip(),
|
||
os.environ.get("HERMES_KANBAN_BOARD", "").strip(),
|
||
):
|
||
found = _existing(candidate)
|
||
if found:
|
||
return found
|
||
try:
|
||
f = current_board_path()
|
||
if f.exists():
|
||
found = _existing(f.read_text(encoding="utf-8").strip())
|
||
if found:
|
||
return found
|
||
except OSError:
|
||
pass
|
||
return DEFAULT_BOARD
|
||
|
||
|
||
def set_current_board(slug: str) -> Path:
|
||
"""Persist ``slug`` as the active board. Returns the file written.
|
||
|
||
Writes ``<root>/kanban/current``. The caller should validate the slug
|
||
exists first (via :func:`board_exists`) — this function does not —
|
||
so that ``hermes kanban boards switch <typo>`` returns an error
|
||
instead of silently pointing at nothing.
|
||
"""
|
||
_assert_not_delegated_child_mutation()
|
||
normed = _normalize_board_slug(slug)
|
||
if not normed:
|
||
raise ValueError("board slug is required")
|
||
path = current_board_path()
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(normed + "\n", encoding="utf-8")
|
||
return path
|
||
|
||
|
||
def clear_current_board() -> None:
|
||
"""Remove ``<root>/kanban/current`` so the active board reverts to ``default``."""
|
||
_assert_not_delegated_child_mutation()
|
||
with contextlib.suppress(FileNotFoundError):
|
||
current_board_path().unlink()
|
||
|
||
|
||
def board_dir(board: Optional[str] = None) -> Path:
|
||
"""``<root>/kanban/boards/<slug>/``. For ``default`` this holds metadata
|
||
only (board.json, workspaces/, logs/) — its DB stays at ``<root>/kanban.db``
|
||
for back-compat (:func:`kanban_db_path`).
|
||
"""
|
||
slug = _normalize_board_slug(board) or DEFAULT_BOARD
|
||
return boards_root() / slug
|
||
|
||
|
||
def board_exists(board: Optional[str] = None) -> bool:
|
||
"""Return True if the board has persisted metadata or a DB on disk.
|
||
|
||
``default`` is considered to always exist — its DB is created
|
||
on first :func:`connect` and there's no way for it to be missing
|
||
in a configuration where the kanban feature is usable at all.
|
||
"""
|
||
slug = _normalize_board_slug(board) or DEFAULT_BOARD
|
||
if slug == DEFAULT_BOARD:
|
||
return True
|
||
d = board_dir(slug)
|
||
return (d / "board.json").exists() or (d / "kanban.db").exists()
|
||
|
||
|
||
def _board_path(
|
||
env_var: Optional[str], board: Optional[str], default_parts: tuple[str, ...], leaf: str,
|
||
) -> Path:
|
||
"""Shared resolver: ``env_var`` override, else legacy ``<root>/<default_parts>``
|
||
for the ``default`` board, else ``board_dir(slug)/leaf``."""
|
||
if env_var:
|
||
override = os.environ.get(env_var, "").strip()
|
||
if override:
|
||
return Path(override).expanduser()
|
||
slug = _normalize_board_slug(board)
|
||
if slug is None:
|
||
slug = get_current_board()
|
||
if slug == DEFAULT_BOARD:
|
||
return kanban_home().joinpath(*default_parts)
|
||
return board_dir(slug) / leaf
|
||
|
||
|
||
def kanban_db_path(board: Optional[str] = None) -> Path:
|
||
"""``kanban.db`` path for ``board``: ``HERMES_KANBAN_DB`` pins it (the
|
||
dispatcher injects it into worker env so workers are immune to any
|
||
path-resolution disagreement); else ``default`` -> ``<root>/kanban.db``
|
||
(back-compat), other boards -> ``<root>/kanban/boards/<slug>/kanban.db``.
|
||
"""
|
||
return _board_path("HERMES_KANBAN_DB", board, ("kanban.db",), "kanban.db")
|
||
|
||
|
||
def workspaces_root(board: Optional[str] = None) -> Path:
|
||
"""Per-board ``scratch`` workspace root (``HERMES_KANBAN_WORKSPACES_ROOT``
|
||
wins — the dispatcher injects it into worker env). ``default`` keeps the
|
||
legacy ``<root>/kanban/workspaces/`` so pre-boards workspaces survive;
|
||
other boards use ``<root>/kanban/boards/<slug>/workspaces/``.
|
||
"""
|
||
return _board_path("HERMES_KANBAN_WORKSPACES_ROOT", board, ("kanban", "workspaces"), "workspaces")
|
||
|
||
|
||
def attachments_root(board: Optional[str] = None) -> Path:
|
||
"""Per-board attachments root (``HERMES_KANBAN_ATTACHMENTS_ROOT`` wins).
|
||
|
||
``default`` -> ``<root>/kanban/attachments/``, other boards ->
|
||
``<root>/kanban/boards/<slug>/attachments/``; each task gets its own
|
||
``<task_id>/`` subdir. Workers read attachments by the absolute path
|
||
surfaced in :func:`build_worker_context`, so remote terminal backends
|
||
(Docker/Modal) need this directory mounted.
|
||
"""
|
||
return _board_path("HERMES_KANBAN_ATTACHMENTS_ROOT", board, ("kanban", "attachments"), "attachments")
|
||
|
||
|
||
def task_attachments_dir(task_id: str, board: Optional[str] = None) -> Path:
|
||
"""Return the per-task attachment directory ``<root>/<task_id>/``."""
|
||
return attachments_root(board=board) / task_id
|
||
|
||
|
||
def worker_logs_dir(board: Optional[str] = None) -> Path:
|
||
"""Return the directory under which per-task worker logs are written.
|
||
|
||
``default`` keeps the legacy path ``<root>/kanban/logs/``. Other
|
||
boards use ``<root>/kanban/boards/<slug>/logs/``. Logs follow the
|
||
board — makes ``hermes kanban log`` unambiguous even when multiple
|
||
boards have tasks with the same id.
|
||
"""
|
||
return _board_path(None, board, ("kanban", "logs"), "logs")
|
||
|
||
|
||
def board_metadata_path(board: Optional[str] = None) -> Path:
|
||
"""Return the path to ``board.json`` for ``board``.
|
||
|
||
Stores display metadata (display name, description, icon, color,
|
||
created_at). The on-disk slug is the canonical identity; this file
|
||
is purely for presentation in the CLI / dashboard.
|
||
"""
|
||
slug = _normalize_board_slug(board) or DEFAULT_BOARD
|
||
return board_dir(slug) / "board.json"
|
||
|
||
|
||
def _default_board_display_name(slug: str) -> str:
|
||
"""Turn a slug into a reasonable default display name.
|
||
|
||
``atm10-server`` → ``Atm10 Server``. Users can override via
|
||
``board.json`` but the default should look presentable in the
|
||
dashboard without any follow-up editing.
|
||
"""
|
||
return " ".join(part.capitalize() for part in slug.replace("_", "-").split("-") if part) or slug
|
||
|
||
|
||
def read_board_metadata(board: Optional[str] = None) -> dict:
|
||
"""Return ``board.json`` contents (or synthesized defaults).
|
||
|
||
Never raises — a missing / malformed ``board.json`` falls back to a
|
||
synthesised entry so the dashboard always has something to render.
|
||
Includes the canonical ``slug`` and ``db_path`` so the caller
|
||
doesn't need to reconstruct them.
|
||
"""
|
||
slug = _normalize_board_slug(board) or DEFAULT_BOARD
|
||
meta: dict[str, Any] = {
|
||
"slug": slug,
|
||
"name": _default_board_display_name(slug),
|
||
"description": "",
|
||
"icon": "",
|
||
"color": "",
|
||
"default_workdir": None,
|
||
# Optional first-class Project this board is scoped to. When set, new
|
||
# tasks inherit it (deterministic worktree + branch under the project's
|
||
# primary repo) and ``default_workdir`` mirrors the project's primary
|
||
# path so the persistent-workspace inheritance path keeps working.
|
||
"project_id": None,
|
||
"created_at": None,
|
||
"archived": False,
|
||
}
|
||
try:
|
||
p = board_metadata_path(slug)
|
||
if p.exists():
|
||
raw = json.loads(p.read_text(encoding="utf-8"))
|
||
if isinstance(raw, dict):
|
||
# Never let the metadata file claim a different slug than
|
||
# its directory — trust the filesystem.
|
||
raw["slug"] = slug
|
||
meta.update(raw)
|
||
except (OSError, json.JSONDecodeError):
|
||
pass
|
||
meta["db_path"] = str(kanban_db_path(slug))
|
||
return meta
|
||
|
||
|
||
def write_board_metadata(
|
||
board: Optional[str],
|
||
*,
|
||
name: Optional[str] = None,
|
||
description: Optional[str] = None,
|
||
icon: Optional[str] = None,
|
||
color: Optional[str] = None,
|
||
archived: Optional[bool] = None,
|
||
default_workdir: Optional[str] = None,
|
||
project_id: Optional[str] = None,
|
||
) -> dict:
|
||
"""Create / update ``board.json`` for ``board``.
|
||
|
||
Preserves any existing fields not mentioned in the call. Sets
|
||
``created_at`` on first write. Returns the resulting metadata dict.
|
||
|
||
``project_id``: ``None`` leaves it unchanged; empty string clears the
|
||
project scope; a value sets it (not validated here — the caller resolves
|
||
it against ``projects_db``).
|
||
"""
|
||
_assert_not_delegated_child_mutation()
|
||
slug = _normalize_board_slug(board) or DEFAULT_BOARD
|
||
meta = read_board_metadata(slug)
|
||
# Preserve existing DB-derived fields — they get re-computed each
|
||
# read but shouldn't be written into board.json.
|
||
meta.pop("db_path", None)
|
||
if name is not None:
|
||
meta["name"] = str(name).strip() or _default_board_display_name(slug)
|
||
if description is not None:
|
||
meta["description"] = str(description)
|
||
if icon is not None:
|
||
meta["icon"] = str(icon)
|
||
if color is not None:
|
||
meta["color"] = str(color)
|
||
if archived is not None:
|
||
meta["archived"] = bool(archived)
|
||
if default_workdir is not None:
|
||
meta["default_workdir"] = str(default_workdir) if default_workdir else None
|
||
if project_id is not None:
|
||
meta["project_id"] = str(project_id) if project_id else None
|
||
if not meta.get("created_at"):
|
||
meta["created_at"] = int(time.time())
|
||
path = board_metadata_path(slug)
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
path.write_text(
|
||
json.dumps(meta, indent=2, ensure_ascii=False) + "\n",
|
||
encoding="utf-8",
|
||
)
|
||
meta["db_path"] = str(kanban_db_path(slug))
|
||
return meta
|
||
|
||
|
||
def create_board(
|
||
slug: str,
|
||
*,
|
||
name: Optional[str] = None,
|
||
description: Optional[str] = None,
|
||
icon: Optional[str] = None,
|
||
color: Optional[str] = None,
|
||
default_workdir: Optional[str] = None,
|
||
project_id: Optional[str] = None,
|
||
) -> dict:
|
||
"""Create a new board directory + DB + metadata. Idempotent.
|
||
|
||
Returns the resulting metadata. Raises :class:`ValueError` for a
|
||
malformed slug; returns the existing metadata (not an error) if the
|
||
board already exists — matching ``mkdir -p`` semantics.
|
||
"""
|
||
normed = _normalize_board_slug(slug)
|
||
if not normed:
|
||
raise ValueError("board slug is required")
|
||
meta = write_board_metadata(
|
||
normed,
|
||
name=name,
|
||
description=description,
|
||
icon=icon,
|
||
color=color,
|
||
default_workdir=default_workdir,
|
||
project_id=project_id,
|
||
)
|
||
# Touch the DB so list_boards() sees it immediately.
|
||
init_db(board=normed)
|
||
return meta
|
||
|
||
|
||
def list_boards(*, include_archived: bool = True) -> list[dict]:
|
||
"""Metadata dicts for every board on disk, ``default`` first (always
|
||
present — its DB lives at the legacy path, so no ``boards/default/`` dir
|
||
is needed) then alphabetical; a ``boards/<slug>/`` counts when it holds a
|
||
``kanban.db`` or ``board.json``.
|
||
"""
|
||
entries: list[dict] = []
|
||
seen: set[str] = set()
|
||
|
||
# Default board is always first.
|
||
entries.append(read_board_metadata(DEFAULT_BOARD))
|
||
seen.add(DEFAULT_BOARD)
|
||
|
||
root = boards_root()
|
||
if root.is_dir():
|
||
for child in sorted(root.iterdir(), key=lambda p: p.name.lower()):
|
||
if not child.is_dir():
|
||
continue
|
||
slug = child.name
|
||
# Keep slug normalisation soft for discovery — but skip dirs
|
||
# that don't parse as valid slugs so we don't surface junk.
|
||
try:
|
||
normed = _normalize_board_slug(slug)
|
||
except ValueError:
|
||
continue
|
||
if not normed or normed in seen:
|
||
continue
|
||
has_db = (child / "kanban.db").exists()
|
||
has_meta = (child / "board.json").exists()
|
||
if not (has_db or has_meta):
|
||
continue
|
||
meta = read_board_metadata(normed)
|
||
if meta.get("archived") and not include_archived:
|
||
continue
|
||
entries.append(meta)
|
||
seen.add(normed)
|
||
return entries
|
||
|
||
|
||
def remove_board(slug: str, *, archive: bool = True) -> dict:
|
||
"""Archive (move to ``boards/_archived/<slug>-<timestamp>/``, recoverable)
|
||
or, with ``archive=False``, delete a board. ``default`` cannot be removed
|
||
(ValueError). Returns ``{"slug", "action", "new_path"}``.
|
||
"""
|
||
_assert_not_delegated_child_mutation()
|
||
normed = _normalize_board_slug(slug)
|
||
if not normed:
|
||
raise ValueError("board slug is required")
|
||
if normed == DEFAULT_BOARD:
|
||
raise ValueError("the 'default' board cannot be removed")
|
||
d = board_dir(normed)
|
||
if not d.exists():
|
||
raise ValueError(f"board {normed!r} does not exist")
|
||
|
||
# If the user removed the currently-active board, revert to default.
|
||
if get_current_board() == normed:
|
||
clear_current_board()
|
||
|
||
# A concurrent connect(board=normed) after the rename/delete recreates
|
||
# an empty sqlite file via mkdir(exist_ok=True); the cache entry must be
|
||
# dropped first so the schema init pass re-runs on that fresh file.
|
||
_INITIALIZED_PATHS.discard(str((d / "kanban.db").resolve()))
|
||
|
||
if archive:
|
||
archive_root = boards_root() / "_archived"
|
||
archive_root.mkdir(parents=True, exist_ok=True)
|
||
ts = int(time.time())
|
||
target = archive_root / f"{normed}-{ts}"
|
||
# Avoid collision on rapid double-archives.
|
||
suffix = 1
|
||
while target.exists():
|
||
target = archive_root / f"{normed}-{ts}-{suffix}"
|
||
suffix += 1
|
||
d.rename(target)
|
||
return {"slug": normed, "action": "archived", "new_path": str(target)}
|
||
import shutil
|
||
shutil.rmtree(d)
|
||
return {"slug": normed, "action": "deleted", "new_path": ""}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Data classes
|
||
# ---------------------------------------------------------------------------
|
||
|
||
@dataclass
|
||
class Task:
|
||
"""In-memory view of a row from the ``tasks`` table."""
|
||
|
||
id: str
|
||
title: str
|
||
body: Optional[str]
|
||
assignee: Optional[str]
|
||
status: str
|
||
priority: int
|
||
created_by: Optional[str]
|
||
created_at: int
|
||
started_at: Optional[int]
|
||
completed_at: Optional[int]
|
||
workspace_kind: str
|
||
workspace_path: Optional[str]
|
||
claim_lock: Optional[str]
|
||
claim_expires: Optional[int]
|
||
tenant: Optional[str]
|
||
branch_name: Optional[str] = None
|
||
project_id: Optional[str] = None
|
||
result: Optional[str] = None
|
||
idempotency_key: Optional[str] = None
|
||
# Column semantics are documented on SCHEMA_SQL. Pre-rename columns:
|
||
# ``spawn_failures`` -> consecutive_failures, ``last_spawn_error`` ->
|
||
# last_failure_error (see ``from_row`` fallbacks).
|
||
consecutive_failures: int = 0
|
||
worker_pid: Optional[int] = None
|
||
last_failure_error: Optional[str] = None
|
||
max_runtime_seconds: Optional[int] = None
|
||
last_heartbeat_at: Optional[int] = None
|
||
current_run_id: Optional[int] = None
|
||
workflow_template_id: Optional[str] = None
|
||
current_step_key: Optional[str] = None
|
||
skills: Optional[list] = None # None = defaults only; [] = explicitly none
|
||
model_override: Optional[str] = None
|
||
provider_override: Optional[str] = None # provider ``model_override`` belongs to
|
||
reasoning_effort: Optional[str] = None # VALID_REASONING_EFFORTS | "none"; NULL = profile's
|
||
# Failure count at which the breaker trips (1 = block on first failure);
|
||
# None -> ``kanban.failure_limit`` config -> DEFAULT_FAILURE_LIMIT.
|
||
max_retries: Optional[int] = None
|
||
# Ralph-style goal loop (same engine as ``/goal``): a judge model re-checks
|
||
# the worker's response against title/body each turn and feeds a
|
||
# continuation prompt IN THE SAME SESSION until done, budget exhausted
|
||
# (-> kanban_block) or explicit block/complete. ``goal_max_turns`` None ->
|
||
# ``goals.DEFAULT_MAX_TURNS``.
|
||
goal_mode: bool = False
|
||
goal_max_turns: Optional[int] = None
|
||
# Originating agent session (``HERMES_SESSION_ID``); NULL from CLI/dashboard.
|
||
session_id: Optional[str] = None
|
||
# Typed block reason (VALID_BLOCK_KINDS) or None for legacy blocks; kept
|
||
# across unblock so a same-kind re-block is recognisable as a loop.
|
||
block_kind: Optional[str] = None
|
||
block_recurrences: int = 0 # unblock-loop counter, see BLOCK_RECURRENCE_LIMIT
|
||
|
||
@classmethod
|
||
def from_row(cls, row: sqlite3.Row) -> "Task":
|
||
g = lambda col, default=None: _row_get(row, col, default) # noqa: E731
|
||
parsed = _json_or(g("skills"))
|
||
skills_value = [str(s) for s in parsed if s] if isinstance(parsed, list) else None
|
||
return cls(
|
||
id=row["id"],
|
||
title=row["title"],
|
||
body=row["body"],
|
||
assignee=row["assignee"],
|
||
status=row["status"],
|
||
priority=row["priority"],
|
||
created_by=row["created_by"],
|
||
created_at=row["created_at"],
|
||
started_at=row["started_at"],
|
||
completed_at=row["completed_at"],
|
||
workspace_kind=row["workspace_kind"],
|
||
workspace_path=row["workspace_path"],
|
||
branch_name=g("branch_name"),
|
||
project_id=g("project_id"),
|
||
claim_lock=row["claim_lock"],
|
||
claim_expires=row["claim_expires"],
|
||
tenant=g("tenant"),
|
||
result=g("result"),
|
||
idempotency_key=g("idempotency_key"),
|
||
# Pre-migration fallbacks (spawn_failures / last_spawn_error) are only
|
||
# reachable on a DB never opened since the rename migration landed.
|
||
consecutive_failures=g("consecutive_failures", g("spawn_failures", 0)),
|
||
worker_pid=g("worker_pid"),
|
||
last_failure_error=g("last_failure_error", g("last_spawn_error")),
|
||
max_runtime_seconds=g("max_runtime_seconds"),
|
||
last_heartbeat_at=g("last_heartbeat_at"),
|
||
current_run_id=g("current_run_id"),
|
||
workflow_template_id=g("workflow_template_id"),
|
||
current_step_key=g("current_step_key"),
|
||
skills=skills_value,
|
||
model_override=g("model_override") or None,
|
||
provider_override=g("provider_override") or None,
|
||
reasoning_effort=g("reasoning_effort") or None,
|
||
max_retries=g("max_retries"),
|
||
goal_mode=bool(g("goal_mode")),
|
||
goal_max_turns=g("goal_max_turns") or None,
|
||
session_id=g("session_id"),
|
||
block_kind=g("block_kind") or None,
|
||
block_recurrences=(
|
||
int(g("block_recurrences")) if g("block_recurrences") is not None else 0
|
||
),
|
||
)
|
||
|
||
|
||
@dataclass
|
||
class Run:
|
||
"""In-memory view of a ``task_runs`` row.
|
||
|
||
A run is one attempt to execute a task — created on claim, closed
|
||
on complete/block/crash/timeout/spawn_failure/reclaim. Multiple runs
|
||
per task when retries happen. Carries the claim machinery, PID,
|
||
heartbeat, and the structured handoff summary that downstream workers
|
||
read via ``build_worker_context``.
|
||
"""
|
||
|
||
id: int
|
||
task_id: str
|
||
profile: Optional[str]
|
||
step_key: Optional[str]
|
||
status: str
|
||
claim_lock: Optional[str]
|
||
claim_expires: Optional[int]
|
||
worker_pid: Optional[int]
|
||
max_runtime_seconds: Optional[int]
|
||
last_heartbeat_at: Optional[int]
|
||
started_at: int
|
||
ended_at: Optional[int]
|
||
outcome: Optional[str]
|
||
summary: Optional[str]
|
||
metadata: Optional[dict]
|
||
error: Optional[str]
|
||
|
||
@classmethod
|
||
def from_row(cls, row: sqlite3.Row) -> "Run":
|
||
return cls(
|
||
id=int(row["id"]),
|
||
task_id=row["task_id"],
|
||
profile=row["profile"],
|
||
step_key=row["step_key"],
|
||
status=row["status"],
|
||
claim_lock=row["claim_lock"],
|
||
claim_expires=row["claim_expires"],
|
||
worker_pid=row["worker_pid"],
|
||
max_runtime_seconds=row["max_runtime_seconds"],
|
||
last_heartbeat_at=row["last_heartbeat_at"],
|
||
started_at=int(row["started_at"]),
|
||
ended_at=_opt_int(row["ended_at"]),
|
||
outcome=row["outcome"],
|
||
summary=row["summary"],
|
||
metadata=_json_or(row["metadata"]),
|
||
error=row["error"],
|
||
)
|
||
|
||
|
||
@dataclass
|
||
class Comment:
|
||
id: int
|
||
task_id: str
|
||
author: str
|
||
body: str
|
||
created_at: int
|
||
|
||
@classmethod
|
||
def from_row(cls, r: sqlite3.Row) -> "Comment":
|
||
return cls(
|
||
id=r["id"], task_id=r["task_id"], author=r["author"],
|
||
body=r["body"], created_at=r["created_at"],
|
||
)
|
||
|
||
|
||
@dataclass
|
||
class Attachment:
|
||
"""In-memory view of a row from the ``task_attachments`` table."""
|
||
|
||
id: int
|
||
task_id: str
|
||
filename: str
|
||
stored_path: str
|
||
content_type: Optional[str]
|
||
size: int
|
||
uploaded_by: Optional[str]
|
||
created_at: int
|
||
|
||
@classmethod
|
||
def from_row(cls, r: sqlite3.Row) -> "Attachment":
|
||
return cls(
|
||
id=r["id"], task_id=r["task_id"], filename=r["filename"],
|
||
stored_path=r["stored_path"], content_type=r["content_type"],
|
||
size=r["size"] or 0, uploaded_by=r["uploaded_by"],
|
||
created_at=r["created_at"],
|
||
)
|
||
|
||
|
||
@dataclass
|
||
class Event:
|
||
id: int
|
||
task_id: str
|
||
kind: str
|
||
payload: Optional[dict]
|
||
created_at: int
|
||
run_id: Optional[int] = None
|
||
|
||
@classmethod
|
||
def from_row(cls, row: sqlite3.Row) -> "Event":
|
||
run_id = _row_get(row, "run_id")
|
||
return cls(
|
||
id=row["id"], task_id=row["task_id"], kind=row["kind"],
|
||
payload=_json_or(row["payload"]), created_at=row["created_at"],
|
||
run_id=_opt_int(run_id),
|
||
)
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Schema
|
||
# ---------------------------------------------------------------------------
|
||
|
||
SCHEMA_SQL = """
|
||
CREATE TABLE IF NOT EXISTS tasks (
|
||
id TEXT PRIMARY KEY,
|
||
title TEXT NOT NULL,
|
||
body TEXT,
|
||
assignee TEXT,
|
||
status TEXT NOT NULL,
|
||
priority INTEGER DEFAULT 0,
|
||
created_by TEXT,
|
||
created_at INTEGER NOT NULL,
|
||
started_at INTEGER,
|
||
completed_at INTEGER,
|
||
workspace_kind TEXT NOT NULL DEFAULT 'scratch',
|
||
workspace_path TEXT,
|
||
branch_name TEXT,
|
||
-- Optional link to a first-class Project (hermes_cli/projects_db). When set,
|
||
-- the task's worktree is anchored under the project's primary repo with a
|
||
-- deterministic branch name instead of a random wt/<task-id> fallback.
|
||
project_id TEXT,
|
||
claim_lock TEXT,
|
||
claim_expires INTEGER,
|
||
tenant TEXT,
|
||
result TEXT,
|
||
idempotency_key TEXT,
|
||
-- Unified consecutive-failure counter. Incremented on spawn
|
||
-- failure, timeout, or crash; reset only on successful completion.
|
||
-- The circuit breaker in _record_task_failure trips when this
|
||
-- exceeds DEFAULT_FAILURE_LIMIT consecutive non-successes.
|
||
consecutive_failures INTEGER NOT NULL DEFAULT 0,
|
||
worker_pid INTEGER,
|
||
-- Short excerpt of the most recent failure's error text.
|
||
last_failure_error TEXT,
|
||
max_runtime_seconds INTEGER,
|
||
last_heartbeat_at INTEGER,
|
||
-- Pointer into task_runs for the currently-active run (NULL if no
|
||
-- run is in-flight). Denormalised for cheap reads.
|
||
current_run_id INTEGER,
|
||
-- Forward-compat for v2 workflow routing. In v1 the kernel writes
|
||
-- these when the task is opted into a template but otherwise ignores
|
||
-- them; the dispatcher doesn't consult them for routing yet.
|
||
workflow_template_id TEXT,
|
||
current_step_key TEXT,
|
||
-- Force-loaded skills for the worker on this task, stored as JSON.
|
||
-- Passed to the worker via `--skills`. NULL or empty array = no extras.
|
||
skills TEXT,
|
||
-- Per-task model override. When set, the dispatcher passes -m <model>
|
||
-- to the worker, overriding the profile's default model. NULL = use
|
||
-- the profile default.
|
||
model_override TEXT,
|
||
-- Provider the model override belongs to. When set (alongside
|
||
-- model_override), the dispatcher passes --provider <name> so the
|
||
-- worker resolves the model against the right backend instead of the
|
||
-- profile's configured provider. NULL = profile provider.
|
||
provider_override TEXT,
|
||
-- Per-task reasoning effort for the worker (minimal|low|medium|high|
|
||
-- xhigh|max|ultra, or 'none' for thinking off). When set, the dispatcher
|
||
-- passes --reasoning <level> so the worker runs at that depth regardless
|
||
-- of the profile's agent.reasoning_effort. NULL = profile setting.
|
||
reasoning_effort TEXT,
|
||
-- Per-task override for the consecutive-failure circuit breaker.
|
||
-- The value is the failure count at which the breaker trips — e.g.
|
||
-- ``max_retries=1`` blocks on the first failure. NULL (the common
|
||
-- case) falls through to the dispatcher-level ``kanban.failure_limit``
|
||
-- config and then ``DEFAULT_FAILURE_LIMIT``.
|
||
max_retries INTEGER,
|
||
-- When 1, the dispatched worker runs in a Ralph-style goal loop: an
|
||
-- auxiliary judge re-evaluates the worker's response against the
|
||
-- card title/body after each turn and feeds a continuation prompt
|
||
-- back into the SAME session until the judge agrees the work is done
|
||
-- or ``goal_max_turns`` is exhausted. NULL/0 = classic single-shot
|
||
-- worker (the default).
|
||
goal_mode INTEGER NOT NULL DEFAULT 0,
|
||
-- Goal-loop turn budget for ``goal_mode`` workers. NULL = use the
|
||
-- goals-engine default.
|
||
goal_max_turns INTEGER,
|
||
-- Originating chat/agent session id when the task was created from
|
||
-- inside an agent loop that propagated ``HERMES_SESSION_ID``. NULL
|
||
-- for tasks created from the CLI, dashboard, or any path that doesn't
|
||
-- set the env var. Indexed so per-session list queries stay cheap on
|
||
-- larger boards.
|
||
session_id TEXT,
|
||
-- Typed block reason set by ``block_task`` (one of VALID_BLOCK_KINDS, or
|
||
-- NULL for legacy/un-typed blocks). Drives routing: ``dependency`` never
|
||
-- sits in ``blocked`` (goes to ``todo`` for parent-gating); the others go
|
||
-- to ``blocked`` for a human. Preserved across unblock so a re-block for
|
||
-- the SAME kind can be recognised as a loop.
|
||
block_kind TEXT,
|
||
-- Unblock-loop counter. Incremented each time a task is re-blocked for the
|
||
-- same truly-blocked reason after having been unblocked. When it reaches
|
||
-- BLOCK_RECURRENCE_LIMIT the task is routed to ``triage`` instead of
|
||
-- ``blocked`` so a cron can't spin it forever. Reset to 0 only on a
|
||
-- successful completion — NOT on unblock (resetting on unblock is exactly
|
||
-- the amnesia that let the loop run unbounded).
|
||
block_recurrences INTEGER NOT NULL DEFAULT 0
|
||
);
|
||
|
||
CREATE TABLE IF NOT EXISTS task_links (
|
||
parent_id TEXT NOT NULL,
|
||
child_id TEXT NOT NULL,
|
||
PRIMARY KEY (parent_id, child_id)
|
||
);
|
||
|
||
CREATE TABLE IF NOT EXISTS task_comments (
|
||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||
task_id TEXT NOT NULL,
|
||
author TEXT NOT NULL,
|
||
body TEXT NOT NULL,
|
||
created_at INTEGER NOT NULL
|
||
);
|
||
|
||
CREATE TABLE IF NOT EXISTS task_events (
|
||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||
task_id TEXT NOT NULL,
|
||
run_id INTEGER,
|
||
kind TEXT NOT NULL,
|
||
payload TEXT,
|
||
created_at INTEGER NOT NULL
|
||
);
|
||
|
||
-- Historical attempt record. Each time the dispatcher claims a task, a
|
||
-- new row is created here; claim state, PID, heartbeat, runtime cap,
|
||
-- and structured summary all live on the run, not the task. Multiple
|
||
-- rows per task id when the task was retried after crash/timeout/block.
|
||
-- v2 of the kanban schema will use ``step_key`` to drive per-stage
|
||
-- workflow routing; in v1 the column is nullable and unused (kernel
|
||
-- ignores it).
|
||
CREATE TABLE IF NOT EXISTS task_runs (
|
||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||
task_id TEXT NOT NULL,
|
||
profile TEXT,
|
||
step_key TEXT,
|
||
status TEXT NOT NULL,
|
||
-- status: running | done | blocked | crashed | timed_out | failed | released
|
||
claim_lock TEXT,
|
||
claim_expires INTEGER,
|
||
worker_pid INTEGER,
|
||
max_runtime_seconds INTEGER,
|
||
last_heartbeat_at INTEGER,
|
||
started_at INTEGER NOT NULL,
|
||
ended_at INTEGER,
|
||
outcome TEXT,
|
||
-- outcome: completed | blocked | crashed | timed_out | spawn_failed |
|
||
-- gave_up | reclaimed | (null while still running)
|
||
summary TEXT,
|
||
metadata TEXT,
|
||
error TEXT
|
||
);
|
||
|
||
-- Files attached to a task (PDFs, images, source documents). The blob
|
||
-- lives on disk under ``attachments_root(board)/<task_id>/<stored_name>``;
|
||
-- this row carries metadata + the absolute ``stored_path`` so the
|
||
-- dashboard can list/download and ``build_worker_context`` can surface
|
||
-- the absolute path to the worker (which has full file-tool access). See
|
||
-- #35338.
|
||
CREATE TABLE IF NOT EXISTS task_attachments (
|
||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||
task_id TEXT NOT NULL,
|
||
filename TEXT NOT NULL,
|
||
stored_path TEXT NOT NULL,
|
||
content_type TEXT,
|
||
size INTEGER NOT NULL DEFAULT 0,
|
||
uploaded_by TEXT,
|
||
created_at INTEGER NOT NULL
|
||
);
|
||
|
||
-- Subscription from a gateway source (platform + chat + thread) to a
|
||
-- task. The gateway's kanban-notifier watcher tails task_events and
|
||
-- pushes ``completed`` / ``blocked`` / ``spawn_auto_blocked`` events to
|
||
-- the original requester so human-in-the-loop workflows close the loop.
|
||
CREATE TABLE IF NOT EXISTS kanban_notify_subs (
|
||
task_id TEXT NOT NULL,
|
||
platform TEXT NOT NULL,
|
||
chat_id TEXT NOT NULL,
|
||
thread_id TEXT NOT NULL DEFAULT '',
|
||
user_id TEXT,
|
||
user_id_alt TEXT,
|
||
chat_type TEXT,
|
||
notifier_profile TEXT,
|
||
delivery_mode TEXT NOT NULL DEFAULT 'notify',
|
||
delivery_metadata TEXT,
|
||
created_at INTEGER NOT NULL,
|
||
last_event_id INTEGER NOT NULL DEFAULT 0,
|
||
PRIMARY KEY (task_id, platform, chat_id, thread_id)
|
||
);
|
||
|
||
CREATE INDEX IF NOT EXISTS idx_tasks_assignee_status ON tasks(assignee, status);
|
||
CREATE INDEX IF NOT EXISTS idx_tasks_status ON tasks(status);
|
||
CREATE INDEX IF NOT EXISTS idx_links_child ON task_links(child_id);
|
||
CREATE INDEX IF NOT EXISTS idx_links_parent ON task_links(parent_id);
|
||
CREATE INDEX IF NOT EXISTS idx_comments_task ON task_comments(task_id, created_at);
|
||
CREATE INDEX IF NOT EXISTS idx_events_task ON task_events(task_id, created_at);
|
||
CREATE INDEX IF NOT EXISTS idx_runs_task ON task_runs(task_id, started_at);
|
||
CREATE INDEX IF NOT EXISTS idx_runs_status ON task_runs(status);
|
||
CREATE INDEX IF NOT EXISTS idx_attachments_task ON task_attachments(task_id, created_at);
|
||
CREATE INDEX IF NOT EXISTS idx_notify_task ON kanban_notify_subs(task_id);
|
||
"""
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# ID generation
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _new_task_id() -> str:
|
||
"""Short URL-safe id: 4 hex bytes (~4.3B; collision ~1.2e-5 at 10k tasks,
|
||
~1.2e-3 at 100k — 2 bytes hit the birthday paradox at ~50% by 10k).
|
||
Idempotency belongs to ``create_task(idempotency_key=...)``, not to id
|
||
uniqueness.
|
||
"""
|
||
return "t_" + secrets.token_hex(4)
|
||
|
||
|
||
def _claimer_id() -> str:
|
||
"""Return a ``host:pid`` string that identifies this claimer."""
|
||
import socket
|
||
try:
|
||
host = socket.gethostname() or "unknown"
|
||
except Exception:
|
||
host = "unknown"
|
||
return f"{host}:{os.getpid()}"
|
||
|
||
|
||
def _host_prefix() -> str:
|
||
"""``"<host>:"`` prefix shared by every claim lock issued from this host."""
|
||
return f"{_claimer_id().split(':', 1)[0]}:"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Task creation / mutation
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _canonical_assignee(assignee: Optional[str]) -> Optional[str]:
|
||
"""Lowercase-assignee normalization for Kanban rows (dashboard/CLI parity)."""
|
||
if assignee is None:
|
||
return None
|
||
from hermes_cli.profiles import normalize_profile_name
|
||
|
||
return normalize_profile_name(assignee)
|
||
|
||
|
||
def _resolve_project_link(
|
||
conn: sqlite3.Connection,
|
||
project_id: Optional[str],
|
||
project_source_task_id: Optional[str],
|
||
workspace_kind: str,
|
||
workspace_path: Optional[str],
|
||
) -> tuple[Optional[str], Any, Optional[str], str]:
|
||
"""Resolve the optional first-class Project link for ``create_task``.
|
||
|
||
Returns ``(project_id, project_obj, project_repo, workspace_kind)``. A
|
||
project-linked task is anchored to the project's primary repo as a git
|
||
worktree so its branch can be named deterministically (project slug + task
|
||
id) instead of the random ``wt/<task-id>`` worker fallback. Projects live in
|
||
the creator's per-profile projects.db; the repo path is absolute and the
|
||
branch name pure, so the cross-profile dispatcher needs no projects.db
|
||
access at dispatch time. ``project_repo`` is the primary repo of a
|
||
project-linked worktree task whose path still has to be derived once the
|
||
task id exists. An unresolvable id/slug drops the link (never a dangling
|
||
reference, never a crash).
|
||
"""
|
||
project_obj = None
|
||
project_repo: Optional[str] = None
|
||
if project_id is not None:
|
||
project_id = str(project_id).strip() or None
|
||
if project_id:
|
||
from hermes_cli import projects_db as _pdb
|
||
|
||
try:
|
||
with _pdb.connect_closing() as _pconn:
|
||
project_obj = _pdb.get_project(_pconn, project_id)
|
||
except Exception:
|
||
project_obj = None
|
||
if project_obj is None and project_source_task_id:
|
||
# Worker profiles have their own projects.db, while the Kanban DB is
|
||
# intentionally shared. Recover routing only from a canonical
|
||
# project-linked source task in this same board. This carries the
|
||
# repo + project branch convention forward without copying or
|
||
# opening the creator profile's project store, and without reusing
|
||
# the source task's literal worktree path.
|
||
source_task = get_task(conn, str(project_source_task_id))
|
||
if (
|
||
source_task is not None
|
||
and source_task.project_id == project_id
|
||
and source_task.workspace_kind == "worktree"
|
||
and source_task.workspace_path
|
||
):
|
||
source_path = Path(source_task.workspace_path)
|
||
if (
|
||
source_path.is_absolute()
|
||
and source_path.name == source_task.id
|
||
and source_path.parent.name == ".worktrees"
|
||
):
|
||
project_slug = None
|
||
if source_task.branch_name:
|
||
prefix, separator, leaf = source_task.branch_name.partition("/")
|
||
if separator and (
|
||
leaf == source_task.id
|
||
or leaf.startswith(f"{source_task.id}-")
|
||
):
|
||
try:
|
||
project_slug = _pdb.normalize_slug(prefix)
|
||
except ValueError:
|
||
project_slug = None
|
||
if project_slug is None:
|
||
try:
|
||
project_slug = _pdb.normalize_slug(project_id)
|
||
except ValueError:
|
||
project_slug = None
|
||
if project_slug:
|
||
project_repo = str(source_path.parent.parent)
|
||
project_obj = _pdb.Project(
|
||
id=project_id,
|
||
slug=project_slug,
|
||
name=project_slug,
|
||
created_at=0,
|
||
primary_path=project_repo,
|
||
)
|
||
if workspace_kind == "scratch":
|
||
workspace_kind = "worktree"
|
||
|
||
if project_obj is None:
|
||
# A project id/slug that doesn't resolve must not crash task
|
||
# creation or persist a dangling reference — drop the link and
|
||
# create the task as an ordinary (scratch) task.
|
||
project_id = None
|
||
else:
|
||
# Canonicalise (a slug may have been passed) and anchor the
|
||
# worktree under the project's primary repo.
|
||
project_id = project_obj.id
|
||
if workspace_kind == "scratch" and project_obj.primary_path:
|
||
workspace_kind = "worktree"
|
||
if (
|
||
workspace_kind == "worktree"
|
||
and workspace_path is None
|
||
and project_obj.primary_path
|
||
):
|
||
# Defer the concrete path to the insert loop: it's a fresh
|
||
# ``<repo>/.worktrees/<task-id>`` dir keyed on the new task id.
|
||
project_repo = str(project_obj.primary_path)
|
||
return project_id, project_obj, project_repo, workspace_kind
|
||
|
||
|
||
def _normalize_task_skills(skills: Optional[Iterable[str]]) -> Optional[list[str]]:
|
||
"""Strip, drop empties, dedupe (order-preserving) a per-task skills list.
|
||
|
||
Refuses commas inside a single name so a comma-joined string is never
|
||
splattered into one argv slot (the ``hermes --skills X,Y`` comma syntax is
|
||
handled in the dispatcher, not here). Toolset names are rejected all at
|
||
once — agents that confuse skills with toolsets usually pass several
|
||
(``["web", "browser", "terminal"]``) and serial-correcting wastes tokens.
|
||
"""
|
||
if skills is None:
|
||
return None
|
||
cleaned: list[str] = []
|
||
seen: set[str] = set()
|
||
toolset_typos: list[str] = []
|
||
for s in skills:
|
||
if not s:
|
||
continue
|
||
name = str(s).strip()
|
||
if not name:
|
||
continue
|
||
if "," in name:
|
||
raise ValueError(
|
||
f"skill name cannot contain comma: {name!r} "
|
||
f"(pass a list of separate names instead of a comma-joined string)"
|
||
)
|
||
if name.casefold() in KNOWN_TOOLSET_NAMES:
|
||
toolset_typos.append(name)
|
||
continue
|
||
if name in seen:
|
||
continue
|
||
seen.add(name)
|
||
cleaned.append(name)
|
||
if toolset_typos:
|
||
quoted = ", ".join(repr(n) for n in toolset_typos)
|
||
noun = "is a toolset name" if len(toolset_typos) == 1 else "are toolset names"
|
||
raise ValueError(
|
||
f"{quoted} {noun}, not skill name(s). "
|
||
"Put toolsets in the assignee profile's `toolsets:` config "
|
||
"instead of per-task skills. Skills are named skill bundles "
|
||
"(e.g. `blogwatcher`, `github-code-review`); toolsets are runtime "
|
||
"capabilities (e.g. `web`, `browser`, `terminal`)."
|
||
)
|
||
return cleaned
|
||
|
||
|
||
def create_task(
|
||
conn: sqlite3.Connection,
|
||
*,
|
||
title: str,
|
||
body: Optional[str] = None,
|
||
assignee: Optional[str] = None,
|
||
created_by: Optional[str] = None,
|
||
workspace_kind: str = "scratch",
|
||
workspace_path: Optional[str] = None,
|
||
branch_name: Optional[str] = None,
|
||
tenant: Optional[str] = None,
|
||
priority: int = 0,
|
||
parents: Iterable[str] = (),
|
||
triage: bool = False,
|
||
idempotency_key: Optional[str] = None,
|
||
max_runtime_seconds: Optional[int] = None,
|
||
skills: Optional[Iterable[str]] = None,
|
||
max_retries: Optional[int] = None,
|
||
model_override: Optional[str] = None,
|
||
provider_override: Optional[str] = None,
|
||
reasoning_effort: Optional[str] = None,
|
||
goal_mode: bool = False,
|
||
goal_max_turns: Optional[int] = None,
|
||
initial_status: str = "running",
|
||
session_id: Optional[str] = None,
|
||
board: Optional[str] = None,
|
||
project_id: Optional[str] = None,
|
||
project_source_task_id: Optional[str] = None,
|
||
) -> str:
|
||
"""Create a new task and optionally link it under parent tasks; returns its id.
|
||
|
||
Status is ``ready`` with no parents (or all parents ``done``), else ``todo``;
|
||
``triage=True`` forces ``triage`` regardless of parents (a specifier promotes
|
||
it later); ``initial_status="blocked"`` parks it for human-ops review.
|
||
|
||
``idempotency_key``: if a non-archived task with the same key exists its id
|
||
is returned instead of creating a duplicate (retried webhooks/automation).
|
||
``max_runtime_seconds``: cap before the dispatcher SIGTERMs (then SIGKILLs
|
||
after a grace window) and re-queues; ``None`` = no cap.
|
||
``skills``: skill names force-loaded into the worker (``hermes --skills``);
|
||
see ``_normalize_task_skills``. ``model_override``/``provider_override`` pin
|
||
the worker model (``-m <model> [--provider <name>]``); provider requires
|
||
model. ``reasoning_effort`` pins thinking depth (``--reasoning <level>``),
|
||
independent of the model override.
|
||
``project_source_task_id``: internal cross-profile fallback for a
|
||
worker-created child — when the active profile cannot resolve
|
||
``project_id`` in its own projects.db, a canonical project-linked task in
|
||
this board supplies the repo and branch convention (its literal worktree is
|
||
never reused). See ``_resolve_project_link``.
|
||
"""
|
||
model_override = (model_override or "").strip() or None
|
||
provider_override = (provider_override or "").strip() or None
|
||
reasoning_effort = normalize_reasoning_effort(reasoning_effort)
|
||
if provider_override and not model_override:
|
||
raise ValueError("provider_override requires a model_override")
|
||
assignee = _canonical_assignee(assignee)
|
||
if not title or not title.strip():
|
||
raise ValueError("title is required")
|
||
if initial_status not in VALID_INITIAL_STATUSES:
|
||
raise ValueError(
|
||
f"initial_status must be one of {sorted(VALID_INITIAL_STATUSES)}"
|
||
)
|
||
if workspace_kind not in VALID_WORKSPACE_KINDS:
|
||
raise ValueError(
|
||
f"workspace_kind must be one of {sorted(VALID_WORKSPACE_KINDS)}, "
|
||
f"got {workspace_kind!r}"
|
||
)
|
||
if branch_name is not None:
|
||
branch_name = str(branch_name).strip() or None
|
||
if branch_name and workspace_kind != "worktree":
|
||
raise ValueError("branch_name is only valid for worktree workspaces")
|
||
|
||
# Inherit the board's scoped project when the caller didn't name one, so a
|
||
# project-scoped board anchors every new task to that project's repo
|
||
# (deterministic worktree + branch) without each surface repeating it.
|
||
if project_id is None:
|
||
try:
|
||
_bmeta = read_board_metadata(board if board else get_current_board())
|
||
_board_project = (_bmeta.get("project_id") or "").strip()
|
||
if _board_project:
|
||
project_id = _board_project
|
||
except Exception:
|
||
pass
|
||
|
||
project_id, project_obj, project_repo, workspace_kind = _resolve_project_link(
|
||
conn, project_id, project_source_task_id, workspace_kind, workspace_path
|
||
)
|
||
|
||
parents = tuple(p for p in parents if p)
|
||
|
||
skills_list = _normalize_task_skills(skills)
|
||
|
||
# Idempotency check — return the existing task instead of creating a
|
||
# duplicate. Done BEFORE entering write_txn to keep the fast path fast
|
||
# and to avoid holding a write lock during the lookup. Race is
|
||
# acceptable: two concurrent creators with the same key might both
|
||
# insert, at which point both rows exist but the next lookup stabilises.
|
||
if idempotency_key:
|
||
row = conn.execute(
|
||
"SELECT id FROM tasks WHERE idempotency_key = ? "
|
||
"AND status != 'archived' "
|
||
"ORDER BY created_at DESC LIMIT 1",
|
||
(idempotency_key,),
|
||
).fetchone()
|
||
if row:
|
||
return row["id"]
|
||
|
||
now = int(time.time())
|
||
|
||
# Resolve workspace_path from board-level default_workdir when the
|
||
# caller did not specify one explicitly. Board defaults represent
|
||
# persistent project checkouts, so only persistent workspace kinds may
|
||
# inherit them. Scratch workspaces are auto-deleted on completion and
|
||
# must stay under the per-board scratch root created by
|
||
# ``resolve_workspace``; inheriting ``default_workdir`` for a scratch
|
||
# task would point cleanup at the user's source tree (#28818). The
|
||
# containment guard in ``_cleanup_workspace`` is the safety rail, but
|
||
# we also stop the bad state from being created in the first place.
|
||
if (
|
||
workspace_path is None
|
||
and project_repo is None
|
||
and workspace_kind in {"dir", "worktree"}
|
||
):
|
||
board_slug = board if board else get_current_board()
|
||
board_meta = read_board_metadata(board_slug)
|
||
board_default = board_meta.get("default_workdir")
|
||
if board_default:
|
||
workspace_path = str(board_default)
|
||
|
||
# Retry once on the extremely unlikely id collision.
|
||
for attempt in range(2):
|
||
task_id = _new_task_id()
|
||
try:
|
||
# ``allow_nested=True``: graph builders (kanban_swarm.create_swarm)
|
||
# compose create_task calls under one outer commit so the
|
||
# dispatcher can never observe a partially constructed graph.
|
||
with write_txn(conn, allow_nested=True):
|
||
# Parent ids are validated in every mode (even triage) so the
|
||
# eventual link rows don't dangle.
|
||
if parents:
|
||
missing = _find_missing_parents(conn, parents)
|
||
if missing:
|
||
raise ValueError(f"unknown parent task(s): {', '.join(missing)}")
|
||
# Determine task status from parent status, unless the caller
|
||
# parks it directly in blocked for human-ops review or in
|
||
# triage for a specifier.
|
||
if initial_status == "blocked":
|
||
task_status = "blocked"
|
||
elif triage:
|
||
task_status = "triage"
|
||
else:
|
||
task_status = "ready"
|
||
if parents:
|
||
# If any parent is not yet done, we're todo.
|
||
rows = conn.execute(
|
||
"SELECT status FROM tasks WHERE id IN "
|
||
"(" + ",".join("?" * len(parents)) + ")",
|
||
parents,
|
||
).fetchall()
|
||
if any(r["status"] != "done" for r in rows):
|
||
task_status = "todo"
|
||
|
||
# Project-linked worktree: a fresh worktree dir under the repo
|
||
# plus a deterministic branch (project slug + task id). Together
|
||
# these kill the random ``wt/<task-id>`` worker fallback and the
|
||
# unanchored ``.worktrees/<id>`` under the dispatcher's cwd.
|
||
if project_obj is not None and workspace_kind == "worktree":
|
||
if project_repo and not workspace_path:
|
||
workspace_path = os.path.join(
|
||
project_repo, ".worktrees", task_id
|
||
)
|
||
if not branch_name:
|
||
from hermes_cli import projects_db as _pdb
|
||
|
||
try:
|
||
branch_name = _pdb.branch_name_for(
|
||
project_obj, task_id, title=title or ""
|
||
)
|
||
except Exception:
|
||
branch_name = None
|
||
|
||
conn.execute(
|
||
"""
|
||
INSERT INTO tasks (
|
||
id, title, body, assignee, status, priority,
|
||
created_by, created_at, workspace_kind, workspace_path,
|
||
branch_name, project_id, tenant, idempotency_key,
|
||
max_runtime_seconds,
|
||
skills, max_retries, model_override, provider_override,
|
||
reasoning_effort,
|
||
goal_mode, goal_max_turns, session_id
|
||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||
""",
|
||
(
|
||
task_id,
|
||
title.strip(),
|
||
body,
|
||
assignee,
|
||
task_status,
|
||
priority,
|
||
created_by,
|
||
now,
|
||
workspace_kind,
|
||
workspace_path,
|
||
branch_name,
|
||
project_id,
|
||
tenant,
|
||
idempotency_key,
|
||
_opt_int(max_runtime_seconds),
|
||
json.dumps(skills_list) if skills_list is not None else None,
|
||
_opt_int(max_retries),
|
||
model_override,
|
||
provider_override,
|
||
reasoning_effort,
|
||
1 if goal_mode else 0,
|
||
_opt_int(goal_max_turns),
|
||
session_id,
|
||
),
|
||
)
|
||
for pid in parents:
|
||
_link(conn, pid, task_id)
|
||
# Notify-sub inheritance (ACK-edge: the originating channel
|
||
# still hears about a child that BLOCKs, not just the final
|
||
# fan-in) is handled by the single-owner helper below —
|
||
# _inherit_notify_subs copies every routing/delivery column.
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"created",
|
||
{
|
||
"assignee": assignee,
|
||
"status": task_status,
|
||
"parents": list(parents),
|
||
"tenant": tenant,
|
||
"workspace_kind": workspace_kind,
|
||
"workspace_path": workspace_path,
|
||
"branch_name": branch_name,
|
||
"project_id": project_id,
|
||
"skills": list(skills_list) if skills_list else None,
|
||
"goal_mode": bool(goal_mode) or None,
|
||
"model_override": model_override,
|
||
"provider_override": provider_override,
|
||
},
|
||
)
|
||
_inherit_notify_subs(conn, task_id, parents, created_at=now)
|
||
return task_id
|
||
except sqlite3.IntegrityError:
|
||
if attempt == 1:
|
||
raise
|
||
# Retry with a fresh id.
|
||
continue
|
||
raise RuntimeError("unreachable")
|
||
|
||
|
||
def _link(conn: sqlite3.Connection, parent_id: str, child_id: str) -> None:
|
||
conn.execute(
|
||
"INSERT OR IGNORE INTO task_links (parent_id, child_id) VALUES (?, ?)",
|
||
(parent_id, child_id),
|
||
)
|
||
|
||
|
||
def _find_missing_parents(conn: sqlite3.Connection, parents: Iterable[str]) -> list[str]:
|
||
parents = list(parents)
|
||
if not parents:
|
||
return []
|
||
placeholders = ",".join("?" * len(parents))
|
||
rows = conn.execute(
|
||
f"SELECT id FROM tasks WHERE id IN ({placeholders})",
|
||
parents,
|
||
).fetchall()
|
||
present = {r["id"] for r in rows}
|
||
return [p for p in parents if p not in present]
|
||
|
||
|
||
def _inherit_notify_subs(
|
||
conn: sqlite3.Connection,
|
||
child_id: str,
|
||
parents: Iterable[str],
|
||
*,
|
||
created_at: Optional[int] = None,
|
||
) -> None:
|
||
"""Copy gateway notification subscriptions from parent tasks to a child.
|
||
|
||
The inherited subscription starts caught up to the child's current event
|
||
cursor. This makes manual `link_tasks(parent, existing_child)` safe: the
|
||
parent chat receives future child terminal events without replaying the
|
||
child's pre-link history.
|
||
|
||
Copies EVERY routing/delivery column (chat_type, user_id_alt,
|
||
delivery_mode, delivery_metadata included) — this helper is the single
|
||
owner of subscription inheritance for create_task, link_tasks, and triage
|
||
decomposition. Omitting columns here silently degrades routing: a
|
||
DM-originated child completion falls back to chat_type='group' and wakes
|
||
a fresh group-scoped session instead of the originating DM (issue #73030).
|
||
"""
|
||
parent_ids = tuple(dict.fromkeys(p for p in parents if p))
|
||
if not parent_ids:
|
||
return
|
||
row = conn.execute(
|
||
"SELECT COALESCE(MAX(id), 0) AS cursor FROM task_events WHERE task_id = ?",
|
||
(child_id,),
|
||
).fetchone()
|
||
cursor = int(row["cursor"] if row is not None else 0)
|
||
placeholders = ",".join("?" * len(parent_ids))
|
||
conn.execute(
|
||
f"""
|
||
INSERT OR IGNORE INTO kanban_notify_subs
|
||
(task_id, platform, chat_id, thread_id, user_id, user_id_alt,
|
||
chat_type, notifier_profile, delivery_mode, delivery_metadata,
|
||
created_at, last_event_id)
|
||
SELECT ?, platform, chat_id, thread_id, user_id, user_id_alt,
|
||
COALESCE(chat_type, 'dm'), notifier_profile,
|
||
COALESCE(delivery_mode, 'notify'), delivery_metadata, ?, ?
|
||
FROM kanban_notify_subs
|
||
WHERE task_id IN ({placeholders})
|
||
""",
|
||
(
|
||
child_id,
|
||
int(created_at if created_at is not None else time.time()),
|
||
cursor,
|
||
*parent_ids,
|
||
),
|
||
)
|
||
|
||
|
||
def get_task(conn: sqlite3.Connection, task_id: str) -> Optional[Task]:
|
||
row = conn.execute("SELECT * FROM tasks WHERE id = ?", (task_id,)).fetchone()
|
||
return Task.from_row(row) if row else None
|
||
|
||
|
||
# Canonical sort-order mappings for ``hermes kanban list --sort``.
|
||
# Each value is a raw SQL fragment appended after ``ORDER BY``.
|
||
VALID_SORT_ORDERS: dict[str, str] = {
|
||
"created": "created_at ASC, id ASC",
|
||
"created-desc": "created_at DESC, id DESC",
|
||
"priority": "priority DESC, created_at ASC",
|
||
"priority-desc": "priority ASC, created_at ASC",
|
||
"status": "status ASC, created_at ASC",
|
||
"assignee": "assignee ASC, created_at ASC",
|
||
"title": "title ASC, id ASC",
|
||
"updated": "started_at DESC NULLS LAST, created_at DESC",
|
||
}
|
||
|
||
|
||
def list_tasks(
|
||
conn: sqlite3.Connection,
|
||
*,
|
||
assignee: Optional[str] = None,
|
||
status: Optional[str] = None,
|
||
tenant: Optional[str] = None,
|
||
session_id: Optional[str] = None,
|
||
include_archived: bool = False,
|
||
limit: Optional[int] = None,
|
||
order_by: Optional[str] = None,
|
||
workflow_template_id: Optional[str] = None,
|
||
current_step_key: Optional[str] = None,
|
||
) -> list[Task]:
|
||
query = "SELECT * FROM tasks WHERE 1=1"
|
||
params: list[Any] = []
|
||
if assignee is not None:
|
||
query += " AND assignee = ?"
|
||
params.append(_canonical_assignee(assignee))
|
||
if status is not None:
|
||
if status not in VALID_STATUSES:
|
||
raise ValueError(f"status must be one of {sorted(VALID_STATUSES)}")
|
||
query += " AND status = ?"
|
||
params.append(status)
|
||
if tenant is not None:
|
||
query += " AND tenant = ?"
|
||
params.append(tenant)
|
||
if session_id is not None:
|
||
query += " AND session_id = ?"
|
||
params.append(session_id)
|
||
if workflow_template_id is not None:
|
||
query += " AND workflow_template_id = ?"
|
||
params.append(workflow_template_id)
|
||
if current_step_key is not None:
|
||
query += " AND current_step_key = ?"
|
||
params.append(current_step_key)
|
||
if not include_archived and status != "archived":
|
||
query += " AND status != 'archived'"
|
||
if order_by is not None:
|
||
order_by = order_by.strip().lower()
|
||
if order_by not in VALID_SORT_ORDERS:
|
||
raise ValueError(
|
||
f"order_by must be one of {sorted(VALID_SORT_ORDERS.keys())}"
|
||
)
|
||
query += f" ORDER BY {VALID_SORT_ORDERS[order_by]}"
|
||
else:
|
||
query += " ORDER BY priority DESC, created_at ASC"
|
||
if limit:
|
||
query += f" LIMIT {int(limit)}"
|
||
rows = conn.execute(query, params).fetchall()
|
||
return [Task.from_row(r) for r in rows]
|
||
|
||
|
||
def assign_task(conn: sqlite3.Connection, task_id: str, profile: Optional[str]) -> bool:
|
||
"""Assign or reassign a task. Returns True on success.
|
||
|
||
Refuses to reassign a task that's currently running (claim_lock set).
|
||
Reassign after the current run completes if needed.
|
||
"""
|
||
profile = _canonical_assignee(profile)
|
||
with write_txn(conn):
|
||
row = conn.execute(
|
||
"SELECT status, claim_lock, assignee FROM tasks WHERE id = ?", (task_id,)
|
||
).fetchone()
|
||
if not row:
|
||
return False
|
||
if row["claim_lock"] is not None and row["status"] == "running":
|
||
raise RuntimeError(
|
||
f"cannot reassign {task_id}: currently running (claimed). "
|
||
"Wait for completion or reclaim the stale lock first."
|
||
)
|
||
if row["assignee"] != profile:
|
||
# The retry guard is scoped to the task/profile combination. A
|
||
# human reassigning the task is an explicit recovery action, so the
|
||
# new profile should not inherit the previous profile's streak.
|
||
conn.execute(
|
||
"UPDATE tasks SET assignee = ?, consecutive_failures = 0, "
|
||
"last_failure_error = NULL WHERE id = ?",
|
||
(profile, task_id),
|
||
)
|
||
else:
|
||
conn.execute("UPDATE tasks SET assignee = ? WHERE id = ?", (profile, task_id))
|
||
_append_event(conn, task_id, "assigned", {"assignee": profile})
|
||
# Task-mutation observer (RFC #58548), fired AFTER the assignment txn
|
||
# has committed so subscribers always observe durable board state.
|
||
notify_task_updated(conn, task_id, ("assignee",))
|
||
return True
|
||
|
||
|
||
def set_model_override(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
model: Optional[str],
|
||
provider: Optional[str] = None,
|
||
) -> bool:
|
||
"""Set (or clear) the per-task model/provider override.
|
||
|
||
``model=None`` (or empty) clears BOTH overrides — the worker falls back
|
||
to its profile's configured model. ``provider`` without ``model`` is
|
||
rejected: a bare provider switch has no defined meaning for the worker
|
||
spawn (``--provider`` alone would re-resolve the profile's model name
|
||
against a different backend, which is exactly the mismatch class this
|
||
feature exists to kill).
|
||
|
||
Allowed on any non-archived task, including ``running`` ones — the
|
||
override only takes effect on the NEXT dispatch, so setting it on a
|
||
running task that's about to be reclaimed/retried is the primary
|
||
rate-limit-recovery flow. Returns True on success.
|
||
"""
|
||
model = (model or "").strip() or None
|
||
provider = (provider or "").strip() or None
|
||
if provider and not model:
|
||
raise ValueError("provider_override requires a model_override")
|
||
if not model:
|
||
provider = None
|
||
with write_txn(conn):
|
||
status = _task_status(conn, task_id)
|
||
if status is None:
|
||
return False
|
||
if status == "archived":
|
||
raise RuntimeError(f"cannot set model override on archived task {task_id}")
|
||
conn.execute(
|
||
"UPDATE tasks SET model_override = ?, provider_override = ? WHERE id = ?",
|
||
(model, provider, task_id),
|
||
)
|
||
_append_event(
|
||
conn, task_id, "model_override_set",
|
||
{"model": model, "provider": provider},
|
||
)
|
||
# Task-mutation observer (RFC #58548), fired AFTER the txn commits.
|
||
notify_task_updated(conn, task_id, ("model_override", "provider_override"))
|
||
return True
|
||
|
||
|
||
def set_reasoning_effort(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
effort: Optional[str],
|
||
) -> bool:
|
||
"""Set (or clear) the per-task reasoning effort.
|
||
|
||
``effort=None`` (or empty) clears the override — the worker falls back to
|
||
its profile's own ``agent.reasoning_effort``. ``"none"`` is a real value,
|
||
not a clear: it pins thinking OFF for this task.
|
||
|
||
Deliberately independent of :func:`set_model_override`: a task may run the
|
||
profile's own model at a different depth, and clearing a model override
|
||
must not silently reset the depth the operator chose. Like the model
|
||
override, it takes effect on the NEXT dispatch, so it is settable on a
|
||
running task. Returns True on success.
|
||
"""
|
||
effort = normalize_reasoning_effort(effort)
|
||
with write_txn(conn):
|
||
status = _task_status(conn, task_id)
|
||
if status is None:
|
||
return False
|
||
if status == "archived":
|
||
raise RuntimeError(
|
||
f"cannot set reasoning effort on archived task {task_id}"
|
||
)
|
||
conn.execute(
|
||
"UPDATE tasks SET reasoning_effort = ? WHERE id = ?",
|
||
(effort, task_id),
|
||
)
|
||
_append_event(
|
||
conn, task_id, "reasoning_effort_set", {"reasoning_effort": effort}
|
||
)
|
||
# Task-mutation observer (RFC #58548), fired AFTER the txn commits.
|
||
notify_task_updated(conn, task_id, ("reasoning_effort",))
|
||
return True
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Links
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def link_tasks(conn: sqlite3.Connection, parent_id: str, child_id: str) -> None:
|
||
if parent_id == child_id:
|
||
raise ValueError("a task cannot depend on itself")
|
||
with write_txn(conn):
|
||
missing = _find_missing_parents(conn, [parent_id, child_id])
|
||
if missing:
|
||
raise ValueError(f"unknown task(s): {', '.join(missing)}")
|
||
if _would_cycle(conn, parent_id, child_id):
|
||
raise ValueError(
|
||
f"linking {parent_id} -> {child_id} would create a cycle"
|
||
)
|
||
_link(conn, parent_id, child_id)
|
||
# If child was ready but parent is not yet done, demote child to todo.
|
||
if _task_status(conn, parent_id) != "done":
|
||
conn.execute(
|
||
"UPDATE tasks SET status = 'todo' WHERE id = ? AND status = 'ready'",
|
||
(child_id,),
|
||
)
|
||
_append_event(
|
||
conn, child_id, "linked",
|
||
{"parent": parent_id, "child": child_id},
|
||
)
|
||
_inherit_notify_subs(conn, child_id, (parent_id,))
|
||
|
||
|
||
def _would_cycle(conn: sqlite3.Connection, parent_id: str, child_id: str) -> bool:
|
||
"""Return True if adding parent->child creates a cycle.
|
||
|
||
A cycle exists iff ``parent_id`` is already a descendant of
|
||
``child_id`` via existing parent->child links. We walk downward
|
||
from ``child_id`` and check whether we reach ``parent_id``.
|
||
"""
|
||
seen = set()
|
||
stack = [child_id]
|
||
while stack:
|
||
node = stack.pop()
|
||
if node == parent_id:
|
||
return True
|
||
if node in seen:
|
||
continue
|
||
seen.add(node)
|
||
rows = conn.execute(
|
||
"SELECT child_id FROM task_links WHERE parent_id = ?", (node,)
|
||
).fetchall()
|
||
stack.extend(r["child_id"] for r in rows)
|
||
return False
|
||
|
||
|
||
def unlink_tasks(conn: sqlite3.Connection, parent_id: str, child_id: str) -> bool:
|
||
with write_txn(conn):
|
||
cur = conn.execute(
|
||
"DELETE FROM task_links WHERE parent_id = ? AND child_id = ?",
|
||
(parent_id, child_id),
|
||
)
|
||
if cur.rowcount:
|
||
_append_event(
|
||
conn, child_id, "unlinked",
|
||
{"parent": parent_id, "child": child_id},
|
||
)
|
||
removed = cur.rowcount > 0
|
||
if removed:
|
||
# Dependency edge removed — re-evaluate promotion eligibility for the
|
||
# child immediately. Matches the contract of complete_task and
|
||
# unblock_task; without this the child stays stuck in todo until the
|
||
# next dispatcher tick or a manual `hermes kanban recompute` (issue #22459).
|
||
recompute_ready(conn)
|
||
return removed
|
||
|
||
|
||
def _linked_ids(conn: sqlite3.Connection, want: str, where: str, task_id: str) -> list[str]:
|
||
rows = conn.execute(
|
||
f"SELECT {want} FROM task_links WHERE {where} = ? ORDER BY {want}", (task_id,)
|
||
).fetchall()
|
||
return [r[want] for r in rows]
|
||
|
||
|
||
def parent_ids(conn: sqlite3.Connection, task_id: str) -> list[str]:
|
||
return _linked_ids(conn, "parent_id", "child_id", task_id)
|
||
|
||
|
||
def child_ids(conn: sqlite3.Connection, task_id: str) -> list[str]:
|
||
return _linked_ids(conn, "child_id", "parent_id", task_id)
|
||
|
||
|
||
def task_graph_contexts(
|
||
conn: sqlite3.Connection, task_ids: Iterable[str]
|
||
) -> dict[str, dict]:
|
||
"""Bulk-load compact direct graph state for graph-aware diagnostics."""
|
||
ordered_ids = list(dict.fromkeys(str(task_id) for task_id in task_ids if task_id))
|
||
contexts = {
|
||
task_id: {"parents": [], "children": []}
|
||
for task_id in ordered_ids
|
||
}
|
||
if not ordered_ids:
|
||
return contexts
|
||
|
||
placeholders = ",".join("?" for _ in ordered_ids)
|
||
for row in conn.execute(
|
||
"SELECT l.child_id AS owner_id, t.id, t.title, t.status "
|
||
"FROM task_links l JOIN tasks t ON t.id = l.parent_id "
|
||
f"WHERE l.child_id IN ({placeholders}) ORDER BY l.child_id, t.id",
|
||
tuple(ordered_ids),
|
||
).fetchall():
|
||
contexts[row["owner_id"]]["parents"].append({
|
||
"id": row["id"],
|
||
"title": row["title"],
|
||
"status": row["status"],
|
||
})
|
||
for row in conn.execute(
|
||
"SELECT l.parent_id AS owner_id, t.id, t.title, t.status "
|
||
"FROM task_links l JOIN tasks t ON t.id = l.child_id "
|
||
f"WHERE l.parent_id IN ({placeholders}) ORDER BY l.parent_id, t.id",
|
||
tuple(ordered_ids),
|
||
).fetchall():
|
||
contexts[row["owner_id"]]["children"].append({
|
||
"id": row["id"],
|
||
"title": row["title"],
|
||
"status": row["status"],
|
||
})
|
||
return contexts
|
||
|
||
|
||
def task_graph_context(conn: sqlite3.Connection, task_id: str) -> dict:
|
||
"""Return compact direct parent/child state for one task."""
|
||
return task_graph_contexts(conn, [task_id])[task_id]
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Comments & events
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def add_comment(
|
||
conn: sqlite3.Connection, task_id: str, author: str, body: str
|
||
) -> int:
|
||
if not body or not body.strip():
|
||
raise ValueError("comment body is required")
|
||
if not author or not author.strip():
|
||
raise ValueError("comment author is required")
|
||
now = int(time.time())
|
||
# ``allow_nested=True``: graph builders (kanban_swarm blackboard seeding)
|
||
# compose comment writes under one outer commit.
|
||
with write_txn(conn, allow_nested=True):
|
||
if not conn.execute(
|
||
"SELECT 1 FROM tasks WHERE id = ?", (task_id,)
|
||
).fetchone():
|
||
raise ValueError(f"unknown task {task_id}")
|
||
cur = conn.execute(
|
||
"INSERT INTO task_comments (task_id, author, body, created_at) "
|
||
"VALUES (?, ?, ?, ?)",
|
||
(task_id, author.strip(), body.strip(), now),
|
||
)
|
||
_append_event(conn, task_id, "commented", {"author": author, "len": len(body)})
|
||
return int(cur.lastrowid or 0)
|
||
|
||
|
||
def _task_rows(conn: sqlite3.Connection, table: str, task_id: str, order: str) -> list[sqlite3.Row]:
|
||
return conn.execute(
|
||
f"SELECT * FROM {table} WHERE task_id = ? ORDER BY {order}", (task_id,)
|
||
).fetchall()
|
||
|
||
|
||
def list_comments(conn: sqlite3.Connection, task_id: str) -> list[Comment]:
|
||
return [Comment.from_row(r) for r in _task_rows(conn, "task_comments", task_id, "created_at ASC")]
|
||
|
||
|
||
def list_comments_after(
|
||
conn: sqlite3.Connection, task_id: str, *, after_id: int = 0
|
||
) -> list[Comment]:
|
||
"""Return comments on ``task_id`` with ``id > after_id`` (ascending).
|
||
|
||
Keyed on the monotonic rowid rather than ``created_at`` so a same-second
|
||
burst can't be skipped. Used by the live worker bridge to fold new
|
||
operator notes into a running task without a restart (see
|
||
``tools.kanban_tools.inject_new_comments_from_env``).
|
||
"""
|
||
rows = conn.execute(
|
||
"SELECT id, task_id, author, body, created_at FROM task_comments "
|
||
"WHERE task_id = ? AND id > ? ORDER BY id ASC",
|
||
(task_id, int(after_id)),
|
||
).fetchall()
|
||
return [Comment.from_row(r) for r in rows]
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Attachments
|
||
# ---------------------------------------------------------------------------
|
||
|
||
# The attachment size cap is the module-level ``KANBAN_ATTACHMENT_MAX_BYTES``
|
||
# (defined near the top of this file) — one constant shared by the dashboard
|
||
# HTTP endpoint, the agent toolset, and the CLI so the limit cannot drift
|
||
# between surfaces.
|
||
|
||
|
||
class AttachmentTooLarge(ValueError):
|
||
"""Raised when an attachment exceeds the configured size cap.
|
||
|
||
Subclasses :class:`ValueError` so generic ``except ValueError`` handlers
|
||
(e.g. the dashboard's 400 fallback) still catch it, while callers that
|
||
want a distinct user-facing message (the tool/CLI 413-equivalent) can
|
||
catch it specifically.
|
||
"""
|
||
|
||
|
||
def _safe_attachment_name(raw: str) -> str:
|
||
"""Reduce a client-supplied filename to a safe basename.
|
||
|
||
Strips any directory components (both separators) so a malicious
|
||
``../../etc/passwd`` or ``C:\\x`` collapses to its leaf. Drops control
|
||
chars and leading dots so we never write a dotfile or a name with
|
||
embedded NULs/newlines. Rejects empty / dotfile-only names. The result
|
||
is only ever joined under the per-task attachments dir, never used
|
||
verbatim as a path from the client.
|
||
|
||
Raises :class:`ValueError` on an unusable name; HTTP callers map that
|
||
to a 400.
|
||
"""
|
||
name = (raw or "").replace("\\", "/").split("/")[-1].strip()
|
||
name = "".join(ch for ch in name if ch.isprintable() and ch not in "\x00").strip()
|
||
name = name.lstrip(".").strip()
|
||
if not name:
|
||
raise ValueError("invalid attachment filename")
|
||
return name[:200]
|
||
|
||
|
||
def _collision_free_path(dest_dir: Path, safe_name: str) -> Path:
|
||
"""Return a path under ``dest_dir`` that doesn't clobber an existing file.
|
||
|
||
``foo.pdf`` → ``foo.pdf``, then ``foo (1).pdf``, ``foo (2).pdf``, …
|
||
``safe_name`` must already be sanitised via :func:`_safe_attachment_name`.
|
||
"""
|
||
stem, dot, ext = safe_name.partition(".")
|
||
candidate = safe_name
|
||
n = 1
|
||
while (dest_dir / candidate).exists():
|
||
candidate = f"{stem} ({n}){dot}{ext}"
|
||
n += 1
|
||
return dest_dir / candidate
|
||
|
||
|
||
def store_attachment_bytes(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
filename: str,
|
||
data: bytes,
|
||
*,
|
||
content_type: Optional[str] = None,
|
||
uploaded_by: Optional[str] = None,
|
||
board: Optional[str] = None,
|
||
max_bytes: Optional[int] = None,
|
||
) -> int:
|
||
"""Validate, size-check, persist a blob, and record its metadata row.
|
||
|
||
This is the single write path shared by the dashboard endpoint, the
|
||
agent toolset (``kanban_attach`` / ``kanban_attach_url``), and the CLI
|
||
(``hermes kanban attach``) so name-sanitisation, the size cap, and the
|
||
collision-resolution all behave identically everywhere.
|
||
|
||
Steps: enforce ``max_bytes``, sanitise ``filename`` to a safe basename,
|
||
write the bytes under :func:`task_attachments_dir` with a
|
||
collision-free name, then insert the ``task_attachments`` row via
|
||
:func:`add_attachment`. Returns the new attachment id.
|
||
|
||
Raises :class:`AttachmentTooLarge` when ``data`` exceeds ``max_bytes``,
|
||
or :class:`ValueError` for a bad filename / unknown task. On any failure
|
||
after the blob is written (e.g. the task disappeared) the orphaned blob
|
||
is removed before re-raising.
|
||
"""
|
||
if max_bytes is None:
|
||
max_bytes = KANBAN_ATTACHMENT_MAX_BYTES
|
||
if len(data) > max_bytes:
|
||
raise AttachmentTooLarge(
|
||
f"attachment exceeds {max_bytes // (1024 * 1024)} MB limit"
|
||
)
|
||
safe_name = _safe_attachment_name(filename)
|
||
dest_dir = task_attachments_dir(task_id, board=board)
|
||
dest_dir.mkdir(parents=True, exist_ok=True)
|
||
dest_path = _collision_free_path(dest_dir, safe_name)
|
||
dest_path.write_bytes(data)
|
||
try:
|
||
return add_attachment(
|
||
conn,
|
||
task_id,
|
||
filename=dest_path.name,
|
||
stored_path=str(dest_path.resolve()),
|
||
content_type=content_type,
|
||
size=len(data),
|
||
uploaded_by=uploaded_by,
|
||
)
|
||
except Exception:
|
||
# Don't leave an orphan blob if the metadata insert fails (most
|
||
# commonly: the task id doesn't exist).
|
||
with contextlib.suppress(OSError):
|
||
dest_path.unlink(missing_ok=True)
|
||
raise
|
||
|
||
|
||
def add_attachment(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
filename: str,
|
||
stored_path: str,
|
||
content_type: Optional[str] = None,
|
||
size: int = 0,
|
||
uploaded_by: Optional[str] = None,
|
||
) -> int:
|
||
"""Record a file attachment for a task. Returns the new attachment id.
|
||
|
||
The caller is responsible for writing the blob to ``stored_path``
|
||
first (under :func:`task_attachments_dir`); this only persists the
|
||
metadata row and appends an ``attached`` event.
|
||
"""
|
||
if not filename or not filename.strip():
|
||
raise ValueError("attachment filename is required")
|
||
if not stored_path or not stored_path.strip():
|
||
raise ValueError("attachment stored_path is required")
|
||
now = int(time.time())
|
||
with write_txn(conn):
|
||
if not conn.execute(
|
||
"SELECT 1 FROM tasks WHERE id = ?", (task_id,)
|
||
).fetchone():
|
||
raise ValueError(f"unknown task {task_id}")
|
||
cur = conn.execute(
|
||
"INSERT INTO task_attachments "
|
||
"(task_id, filename, stored_path, content_type, size, uploaded_by, created_at) "
|
||
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||
(
|
||
task_id,
|
||
filename.strip(),
|
||
stored_path,
|
||
content_type,
|
||
int(size),
|
||
uploaded_by,
|
||
now,
|
||
),
|
||
)
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"attached",
|
||
{"filename": filename.strip(), "size": int(size), "by": uploaded_by},
|
||
)
|
||
return int(cur.lastrowid or 0)
|
||
|
||
|
||
def list_attachments(conn: sqlite3.Connection, task_id: str) -> list[Attachment]:
|
||
return [Attachment.from_row(r) for r in _task_rows(conn, "task_attachments", task_id, "created_at ASC, id ASC")]
|
||
|
||
|
||
def get_attachment(conn: sqlite3.Connection, attachment_id: int) -> Optional[Attachment]:
|
||
r = conn.execute(
|
||
"SELECT * FROM task_attachments WHERE id = ?", (attachment_id,)
|
||
).fetchone()
|
||
return None if r is None else Attachment.from_row(r)
|
||
|
||
|
||
def delete_attachment(conn: sqlite3.Connection, attachment_id: int) -> Optional[Attachment]:
|
||
"""Delete an attachment row and its on-disk blob. Returns the removed row.
|
||
|
||
Returns ``None`` when no row matched. The blob is removed best-effort
|
||
(a missing file is not an error); the metadata row is the source of
|
||
truth for whether an attachment "exists".
|
||
"""
|
||
with write_txn(conn):
|
||
att = get_attachment(conn, attachment_id)
|
||
if att is None:
|
||
return None
|
||
conn.execute("DELETE FROM task_attachments WHERE id = ?", (attachment_id,))
|
||
_append_event(
|
||
conn, att.task_id, "attachment_removed", {"filename": att.filename}
|
||
)
|
||
try:
|
||
p = Path(att.stored_path)
|
||
if p.is_file():
|
||
p.unlink()
|
||
except OSError:
|
||
pass
|
||
return att
|
||
|
||
|
||
def list_events(conn: sqlite3.Connection, task_id: str) -> list[Event]:
|
||
return [Event.from_row(r) for r in _task_rows(conn, "task_events", task_id, "created_at ASC, id ASC")]
|
||
|
||
|
||
def _insert_comment(
|
||
conn: sqlite3.Connection, task_id: str, author: str, body: str, created_at: int,
|
||
) -> None:
|
||
"""Raw ``task_comments`` INSERT for callers already inside a write txn.
|
||
|
||
``add_comment`` opens its own ``write_txn`` (raises on nesting) and emits
|
||
a ``commented`` event; transitions that record their own event use this.
|
||
"""
|
||
conn.execute(
|
||
"INSERT INTO task_comments (task_id, author, body, created_at) "
|
||
"VALUES (?, ?, ?, ?)",
|
||
(task_id, author, body, created_at),
|
||
)
|
||
|
||
|
||
def _append_event(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
kind: str,
|
||
payload: Optional[dict] = None,
|
||
*,
|
||
run_id: Optional[int] = None,
|
||
) -> None:
|
||
"""Record an event row. Called from within an already-open txn.
|
||
|
||
``run_id`` is optional: pass the current run id so UIs can group
|
||
events by attempt. For events that aren't scoped to a single run
|
||
(task created/edited/archived, dependency promotion) leave it None
|
||
and the row carries NULL.
|
||
"""
|
||
now = int(time.time())
|
||
pl = json.dumps(payload, ensure_ascii=False) if payload else None
|
||
conn.execute(
|
||
"INSERT INTO task_events (task_id, run_id, kind, payload, created_at) "
|
||
"VALUES (?, ?, ?, ?, ?)",
|
||
(task_id, run_id, kind, pl, now),
|
||
)
|
||
|
||
|
||
def _end_run(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
outcome: str,
|
||
summary: Optional[str] = None,
|
||
error: Optional[str] = None,
|
||
metadata: Optional[dict] = None,
|
||
status: Optional[str] = None,
|
||
) -> Optional[int]:
|
||
"""Close the currently-active run for ``task_id`` and clear the pointer.
|
||
|
||
``outcome`` is the semantic result (completed / blocked / crashed /
|
||
timed_out / spawn_failed / gave_up / reclaimed). ``status`` is the
|
||
run-row status (usually just ``outcome``, but callers can pass it
|
||
explicitly). Returns the closed run_id or ``None`` if no active run
|
||
existed (e.g. a CLI user calling ``hermes kanban complete`` on a
|
||
task that was never claimed).
|
||
"""
|
||
now = int(time.time())
|
||
run_id = _current_run_id(conn, task_id)
|
||
if run_id is None:
|
||
return None
|
||
conn.execute(
|
||
"""
|
||
UPDATE task_runs
|
||
SET status = ?,
|
||
outcome = ?,
|
||
summary = ?,
|
||
error = ?,
|
||
metadata = ?,
|
||
ended_at = ?,
|
||
claim_lock = NULL,
|
||
claim_expires = NULL,
|
||
worker_pid = NULL
|
||
WHERE id = ?
|
||
AND ended_at IS NULL
|
||
""",
|
||
(
|
||
status or outcome,
|
||
outcome,
|
||
summary,
|
||
error,
|
||
json.dumps(metadata, ensure_ascii=False) if metadata else None,
|
||
now,
|
||
run_id,
|
||
),
|
||
)
|
||
conn.execute(
|
||
"UPDATE tasks SET current_run_id = NULL WHERE id = ?", (task_id,),
|
||
)
|
||
return run_id
|
||
|
||
|
||
def _opt_int(value: Any) -> Optional[int]:
|
||
"""``int(value)`` or ``None`` when ``value`` is ``None`` (NULL column passthrough)."""
|
||
return int(value) if value is not None else None
|
||
|
||
|
||
def _task_status(conn: sqlite3.Connection, task_id: str) -> Optional[str]:
|
||
"""Current ``tasks.status`` for ``task_id``, or ``None`` when no such row."""
|
||
row = conn.execute("SELECT status FROM tasks WHERE id = ?", (task_id,)).fetchone()
|
||
return row["status"] if row else None
|
||
|
||
|
||
def _current_run_id(conn: sqlite3.Connection, task_id: str) -> Optional[int]:
|
||
row = conn.execute(
|
||
"SELECT current_run_id FROM tasks WHERE id = ?", (task_id,),
|
||
).fetchone()
|
||
return int(row["current_run_id"]) if row and row["current_run_id"] else None
|
||
|
||
|
||
def _synthesize_ended_run(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
outcome: str,
|
||
summary: Optional[str] = None,
|
||
error: Optional[str] = None,
|
||
metadata: Optional[dict] = None,
|
||
) -> int:
|
||
"""Insert a zero-duration, already-closed run row.
|
||
|
||
Used when a terminal transition happens on a task that was never
|
||
claimed (CLI user calling ``hermes kanban complete <ready-task>
|
||
--summary X``, or dashboard "mark done" on a ready task). Without
|
||
this, the handoff fields (summary / metadata / error) would be
|
||
silently dropped: ``_end_run`` is a no-op because there's no
|
||
current run.
|
||
|
||
The synthetic run has ``started_at == ended_at == now`` so it
|
||
shows up in attempt history as "instant" and doesn't skew elapsed
|
||
stats. Caller is responsible for leaving ``current_run_id`` NULL
|
||
(or for clearing it elsewhere in the same txn) since this
|
||
function does NOT touch the tasks row.
|
||
"""
|
||
now = int(time.time())
|
||
trow = conn.execute(
|
||
"SELECT assignee, current_step_key FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
profile = trow["assignee"] if trow else None
|
||
step_key = trow["current_step_key"] if trow else None
|
||
cur = conn.execute(
|
||
"""
|
||
INSERT INTO task_runs (
|
||
task_id, profile, step_key,
|
||
status, outcome,
|
||
summary, error, metadata,
|
||
started_at, ended_at
|
||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||
""",
|
||
(
|
||
task_id, profile, step_key,
|
||
outcome, outcome,
|
||
summary, error,
|
||
json.dumps(metadata, ensure_ascii=False) if metadata else None,
|
||
now, now,
|
||
),
|
||
)
|
||
return int(cur.lastrowid or 0)
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Dependency resolution (todo -> ready)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _has_sticky_block(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Return True when ``task_id`` is sticky-blocked by an explicit
|
||
worker/operator ``kanban_block`` call.
|
||
|
||
A ``blocked`` status has two sources: a deliberate worker/operator
|
||
``kanban_block`` handoff (emits a ``"blocked"`` event; must stay blocked
|
||
until an operator unblocks it) and a circuit-breaker trip in
|
||
``_record_task_failure`` (emits ``"gave_up"``, NOT ``"blocked"``; meant to
|
||
recover once conditions change). The cheapest discriminator is the most
|
||
recent ``"blocked"``/``"unblocked"`` event: if it is ``"blocked"`` the task
|
||
is sticky and ``recompute_ready`` must not auto-promote it. No such event
|
||
at all (breaker trip, direct DB edit) returns ``False`` — the legacy
|
||
auto-recover path.
|
||
"""
|
||
row = conn.execute(
|
||
"SELECT kind FROM task_events "
|
||
"WHERE task_id = ? AND kind IN ('blocked', 'unblocked') "
|
||
"ORDER BY id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
return bool(row) and row["kind"] == "blocked"
|
||
|
||
|
||
def _latest_event(
|
||
conn: sqlite3.Connection, task_id: str, kind: str, run_id: Optional[int] = None,
|
||
) -> Optional[sqlite3.Row]:
|
||
"""Newest ``task_events`` row of ``kind`` (optionally scoped to one run)."""
|
||
sql = "SELECT payload FROM task_events WHERE task_id = ? AND kind = ?"
|
||
params: tuple[Any, ...] = (task_id, kind)
|
||
if run_id is not None:
|
||
sql += " AND run_id = ?"
|
||
params = (*params, int(run_id))
|
||
return conn.execute(sql + " ORDER BY id DESC LIMIT 1", params).fetchone()
|
||
|
||
|
||
def _resume_status_from_events(conn: sqlite3.Connection, task_id: str) -> str:
|
||
"""Return the durable phase a blocked/dependency-wait task should resume.
|
||
|
||
Events written by review workers carry ``source_status``/``retry_status``;
|
||
an explicit unblock that must wait for parents carries ``resume_status``.
|
||
Legacy events omit these fields and therefore retain the historical
|
||
``ready`` behavior.
|
||
"""
|
||
row = conn.execute(
|
||
"SELECT payload FROM task_events "
|
||
"WHERE task_id = ? AND kind IN ("
|
||
"'blocked', 'block_loop_detected', 'dependency_wait', 'gave_up', "
|
||
"'unblocked', 'changes_requested', 'review_reopened', 'status', 'reclaimed', "
|
||
"'stale', 'timed_out', 'crashed', 'spawn_failed', 'rate_limited'"
|
||
") ORDER BY id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
payload = _json_dict(_row_get(row, "payload"))
|
||
for key in ("resume_status", "retry_status", "source_status"):
|
||
if payload.get(key) == "review":
|
||
return "review"
|
||
return "ready"
|
||
|
||
|
||
def recompute_ready(
|
||
conn: sqlite3.Connection, failure_limit: int = None,
|
||
) -> int:
|
||
"""Promote ``todo`` tasks to ``ready`` when all parents are ``done`` or ``archived``.
|
||
|
||
Returns the number of tasks promoted. Opens its own IMMEDIATE txn, so it
|
||
MUST be called OUTSIDE any open write transaction (plain ``write_txn``
|
||
raises on nesting); call it after the enclosing txn commits.
|
||
|
||
``blocked`` tasks are also considered (a task blocked purely by a parent
|
||
dependency unblocks itself when the parent completes), *except* when the
|
||
most recent block event was a worker-initiated ``kanban_block`` (stays
|
||
blocked until explicit ``kanban_unblock``) or ``consecutive_failures`` has
|
||
reached the effective limit (otherwise the counter would reset on every
|
||
recovery cycle and the breaker could never trip).
|
||
|
||
The effective limit resolves in the same order as ``_record_task_failure``
|
||
so the two never disagree: per-task ``max_retries``, then the caller's
|
||
``failure_limit`` (``kanban.failure_limit`` via ``dispatch_once``), then
|
||
``DEFAULT_FAILURE_LIMIT``.
|
||
"""
|
||
if failure_limit is None:
|
||
failure_limit = DEFAULT_FAILURE_LIMIT
|
||
promoted = 0
|
||
with write_txn(conn):
|
||
todo_rows = conn.execute(
|
||
"SELECT id, status, consecutive_failures, max_retries "
|
||
"FROM tasks WHERE status IN ('todo', 'blocked')"
|
||
).fetchall()
|
||
for row in todo_rows:
|
||
task_id = row["id"]
|
||
cur_status = row["status"]
|
||
if cur_status == "blocked" and _has_sticky_block(conn, task_id):
|
||
# Worker / operator asked for explicit human intervention — do not
|
||
# silently auto-recover. ``unblock_task`` is the only
|
||
# legitimate exit (it emits ``"unblocked"`` which flips
|
||
# this predicate back).
|
||
continue
|
||
parents = conn.execute(
|
||
"SELECT t.status FROM tasks t "
|
||
"JOIN task_links l ON l.parent_id = t.id "
|
||
"WHERE l.child_id = ?",
|
||
(task_id,),
|
||
).fetchall()
|
||
if all(p["status"] in ("done", "archived") for p in parents):
|
||
resume_status = _resume_status_from_events(conn, task_id)
|
||
if cur_status == "blocked":
|
||
# Don't auto-recover tasks that have hit the
|
||
# circuit-breaker failure limit. Without this
|
||
# guard, a task that repeatedly exhausts its
|
||
# iteration budget would cycle forever:
|
||
# block → auto-recover → respawn → budget
|
||
# exhausted → block → … The counter must also
|
||
# be preserved so the breaker can accumulate
|
||
# across recovery cycles.
|
||
failures = int(row["consecutive_failures"] or 0)
|
||
task_limit = row["max_retries"]
|
||
effective_limit = (
|
||
int(task_limit) if task_limit is not None
|
||
else int(failure_limit)
|
||
)
|
||
if failures >= effective_limit:
|
||
continue
|
||
conn.execute(
|
||
"UPDATE tasks SET status = ? "
|
||
"WHERE id = ? AND status = 'blocked'",
|
||
(resume_status, task_id),
|
||
)
|
||
else:
|
||
conn.execute(
|
||
"UPDATE tasks SET status = ? WHERE id = ? AND status = 'todo'",
|
||
(resume_status, task_id),
|
||
)
|
||
_append_event(
|
||
conn, task_id, "promoted",
|
||
{"status": resume_status} if resume_status != "ready" else None,
|
||
)
|
||
promoted += 1
|
||
return promoted
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Claim / complete / block
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def _parents_satisfied(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Return whether every direct parent is terminal for dependency gating."""
|
||
return conn.execute(
|
||
"SELECT 1 FROM task_links l "
|
||
"JOIN tasks p ON p.id = l.parent_id "
|
||
"WHERE l.child_id = ? "
|
||
"AND p.status NOT IN ('done', 'archived') LIMIT 1",
|
||
(task_id,),
|
||
).fetchone() is None
|
||
|
||
|
||
def _claim_and_open_run(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
source_status: str,
|
||
lock: str,
|
||
expires: int,
|
||
now: int,
|
||
*,
|
||
event_extra: Optional[dict] = None,
|
||
) -> Optional[int]:
|
||
"""CAS ``source_status -> running``, open a ``task_runs`` row and emit ``claimed``.
|
||
|
||
Caller holds the write transaction. Returns the new run id, or ``None``
|
||
when the CAS lost (task already claimed / not in ``source_status``).
|
||
"""
|
||
cur = conn.execute(
|
||
f"""
|
||
UPDATE tasks
|
||
SET status = 'running',
|
||
claim_lock = ?,
|
||
claim_expires = ?,
|
||
started_at = COALESCE(started_at, ?)
|
||
WHERE id = ?
|
||
AND status = '{source_status}'
|
||
AND claim_lock IS NULL
|
||
""",
|
||
(lock, expires, now, task_id),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return None
|
||
# Populate the run with the task's assignee / step / runtime cap.
|
||
trow = conn.execute(
|
||
"SELECT assignee, max_runtime_seconds, current_step_key "
|
||
"FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
run_cur = conn.execute(
|
||
"""
|
||
INSERT INTO task_runs (
|
||
task_id, profile, step_key, status,
|
||
claim_lock, claim_expires, max_runtime_seconds,
|
||
started_at
|
||
) VALUES (?, ?, ?, 'running', ?, ?, ?, ?)
|
||
""",
|
||
(
|
||
task_id,
|
||
trow["assignee"] if trow else None,
|
||
trow["current_step_key"] if trow else None,
|
||
lock,
|
||
expires,
|
||
trow["max_runtime_seconds"] if trow else None,
|
||
now,
|
||
),
|
||
)
|
||
run_id = run_cur.lastrowid
|
||
conn.execute(
|
||
"UPDATE tasks SET current_run_id = ? WHERE id = ?",
|
||
(run_id, task_id),
|
||
)
|
||
_append_event(
|
||
conn, task_id, "claimed",
|
||
{"lock": lock, "expires": expires, "run_id": run_id, **(event_extra or {})},
|
||
run_id=run_id,
|
||
)
|
||
return run_id
|
||
|
||
|
||
def claim_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
ttl_seconds: Optional[int] = None,
|
||
claimer: Optional[str] = None,
|
||
) -> Optional[Task]:
|
||
"""Atomically transition ``ready -> running``.
|
||
|
||
Returns the claimed ``Task`` on success, ``None`` if the task was
|
||
already claimed (or is not in ``ready`` status).
|
||
"""
|
||
now = int(time.time())
|
||
lock = claimer or _claimer_id()
|
||
expires = now + _resolve_claim_ttl_seconds(ttl_seconds)
|
||
with write_txn(conn):
|
||
# Structural invariant: never transition ready -> running while any
|
||
# parent is not yet 'done'. This is the single enforcement point
|
||
# regardless of which writer (create_task, link_tasks, unblock_task,
|
||
# release_stale_claims, manual SQL) set status='ready'. If a racy
|
||
# writer promoted a task with undone parents, demote it back to
|
||
# 'todo' here — recompute_ready will re-promote when the parents
|
||
# actually finish. See RCA at
|
||
# kanban/boards/cookai/workspaces/t_a6acd07d/root-cause.md.
|
||
if not _parents_satisfied(conn, task_id):
|
||
conn.execute(
|
||
"UPDATE tasks SET status = 'todo' "
|
||
"WHERE id = ? AND status = 'ready'",
|
||
(task_id,),
|
||
)
|
||
_append_event(
|
||
conn, task_id, "claim_rejected",
|
||
{"reason": "parents_not_done"},
|
||
)
|
||
return None
|
||
# Defensive: close a leaked prior run as 'reclaimed' so the CAS below
|
||
# doesn't strand it. No-op when the runs invariant holds.
|
||
_reclaim_dangling_run(
|
||
conn, task_id, statuses=("ready",), now=now,
|
||
note="invariant recovery on re-claim",
|
||
)
|
||
run_id = _claim_and_open_run(conn, task_id, "ready", lock, expires, now)
|
||
if run_id is None:
|
||
return None
|
||
claimed = get_task(conn, task_id)
|
||
_fire_kanban_lifecycle_hook(
|
||
"kanban_task_claimed",
|
||
task_id,
|
||
board=get_current_board(),
|
||
assignee=claimed.assignee if claimed else None,
|
||
run_id=run_id,
|
||
)
|
||
return claimed
|
||
|
||
|
||
def claim_review_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
ttl_seconds: Optional[int] = None,
|
||
claimer: Optional[str] = None,
|
||
) -> Optional[Task]:
|
||
"""Atomically ``review -> running``; the claimed ``Task`` or None when
|
||
already claimed / not in review. Parents are re-checked (one may have
|
||
been reopened while the task waited) and a NEW run is opened so the
|
||
review agent's lifecycle is tracked apart from the implementer's run.
|
||
"""
|
||
now = int(time.time())
|
||
lock = claimer or _claimer_id()
|
||
expires = now + _resolve_claim_ttl_seconds(ttl_seconds)
|
||
with write_txn(conn):
|
||
if not _parents_satisfied(conn, task_id):
|
||
demoted = conn.execute(
|
||
"UPDATE tasks SET status = 'todo' "
|
||
"WHERE id = ? AND status = 'review' AND claim_lock IS NULL",
|
||
(task_id,),
|
||
)
|
||
if demoted.rowcount == 1:
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"dependency_wait",
|
||
{
|
||
"reason": "parent_reopened",
|
||
"source_status": "review",
|
||
},
|
||
)
|
||
return None
|
||
run_id = _claim_and_open_run(
|
||
conn, task_id, "review", lock, expires, now,
|
||
event_extra={"source_status": "review"},
|
||
)
|
||
if run_id is None:
|
||
return None
|
||
return get_task(conn, task_id)
|
||
|
||
|
||
def _retry_status_for_run(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
run_id: Optional[int] = None,
|
||
) -> str:
|
||
"""Return the non-running phase an interrupted run must resume from.
|
||
|
||
Review claims record ``source_status=review`` on their claimed event. All
|
||
other and legacy runs retry from ``ready``. Keeping this decision in one
|
||
place prevents crash/timeout/reclaim paths from silently converting a
|
||
reviewer run into an implementation run.
|
||
"""
|
||
if run_id is None:
|
||
run_id = _current_run_id(conn, task_id)
|
||
if run_id is None:
|
||
return "ready"
|
||
event = _latest_event(conn, task_id, "claimed", run_id)
|
||
payload = _json_dict(_row_get(event, "payload"))
|
||
return "review" if payload.get("source_status") == "review" else "ready"
|
||
|
||
|
||
def goal_run_status(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
expected_run_id: Optional[int] = None,
|
||
) -> Optional[str]:
|
||
"""Resolve lifecycle status from the perspective of one worker run.
|
||
|
||
A successor may claim the task immediately after this run hands it off.
|
||
Returning the task's live ``running`` status in that case lets the old goal
|
||
loop mutate the successor. Bind terminal handoffs to the original run and
|
||
report any other ownership loss as ``superseded``.
|
||
"""
|
||
task = get_task(conn, task_id)
|
||
if task is None:
|
||
return None
|
||
if expected_run_id is not None:
|
||
row = conn.execute(
|
||
"SELECT outcome FROM task_runs WHERE id = ? AND task_id = ?",
|
||
(int(expected_run_id), task_id),
|
||
).fetchone()
|
||
outcome = (
|
||
str(row["outcome"])
|
||
if row and row["outcome"] is not None
|
||
else None
|
||
)
|
||
terminal_status = (
|
||
{
|
||
"completed": "done",
|
||
"review_requested": "review",
|
||
"changes_requested": "changes_requested",
|
||
"blocked": "blocked",
|
||
"dependency_wait": "blocked",
|
||
}.get(outcome)
|
||
if outcome is not None
|
||
else None
|
||
)
|
||
if terminal_status is not None:
|
||
return terminal_status
|
||
if outcome is not None or task.current_run_id != int(expected_run_id):
|
||
return "superseded"
|
||
if task.status in {"ready", "todo"}:
|
||
event = conn.execute(
|
||
"SELECT kind FROM task_events WHERE task_id = ? "
|
||
"ORDER BY id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
if event and event["kind"] == "changes_requested":
|
||
return "changes_requested"
|
||
return task.status
|
||
|
||
|
||
def heartbeat_claim(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
ttl_seconds: Optional[int] = None,
|
||
claimer: Optional[str] = None,
|
||
) -> bool:
|
||
"""Extend a running claim. Returns True if we still own it.
|
||
|
||
Workers that know they'll exceed 15 minutes should call this every
|
||
few minutes to keep ownership.
|
||
"""
|
||
expires = int(time.time()) + _resolve_claim_ttl_seconds(ttl_seconds)
|
||
lock = claimer or _claimer_id()
|
||
with write_txn(conn):
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET claim_expires = ? "
|
||
"WHERE id = ? AND status = 'running' AND claim_lock = ?",
|
||
(expires, task_id, lock),
|
||
)
|
||
if cur.rowcount == 1:
|
||
run_id = _current_run_id(conn, task_id)
|
||
if run_id is not None:
|
||
conn.execute(
|
||
"UPDATE task_runs SET claim_expires = ? WHERE id = ?",
|
||
(expires, run_id),
|
||
)
|
||
return True
|
||
return False
|
||
|
||
|
||
def release_stale_claims(
|
||
conn: sqlite3.Connection,
|
||
*,
|
||
signal_fn=None,
|
||
) -> int:
|
||
"""Reset any ``running`` task whose claim has expired.
|
||
|
||
A stale-by-TTL claim whose host-local worker PID is still alive is
|
||
*extended* (``claim_extended`` event) instead of reclaimed: reclaiming a
|
||
live worker mid-flight causes a spawn-then-reclaim loop on slow models
|
||
that spend longer than ``DEFAULT_CLAIM_TTL_SECONDS`` inside one tool-free
|
||
LLM call (no tool calls means no ``kanban_heartbeat``).
|
||
|
||
Backstop: a live PID whose ``last_heartbeat_at`` is older than
|
||
``DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS`` is reclaimed anyway — the
|
||
wedged-in-a-logic-loop case. ``_touch_activity`` (run_agent.py) bridges
|
||
chunk-level liveness into ``last_heartbeat_at``, so any genuinely active
|
||
worker stays fresh via normal API traffic. ``enforce_max_runtime`` and
|
||
``detect_crashed_workers`` remain the upper bounds for wedged/dead workers.
|
||
|
||
Returns the number of stale claims actually reclaimed (live-pid
|
||
extensions don't count). Safe to call often.
|
||
"""
|
||
now = int(time.time())
|
||
reclaimed = 0
|
||
host_prefix = _host_prefix()
|
||
stale = conn.execute(
|
||
"SELECT id, claim_lock, worker_pid, claim_expires, last_heartbeat_at, "
|
||
" assignee "
|
||
"FROM tasks "
|
||
"WHERE status = 'running' AND claim_expires IS NOT NULL "
|
||
" AND claim_expires < ?",
|
||
(now,),
|
||
).fetchall()
|
||
for row in stale:
|
||
lock = row["claim_lock"] or ""
|
||
host_local = lock.startswith(host_prefix)
|
||
hb = row["last_heartbeat_at"]
|
||
# Heartbeat staleness backstop: if we have a heartbeat at all
|
||
# and it's older than the max-stale threshold, the worker is
|
||
# not making observable progress. Reclaim instead of extending,
|
||
# even if the PID is still alive (it's likely in a logic loop).
|
||
heartbeat_stale = (
|
||
hb is not None
|
||
and (now - int(hb)) > DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS
|
||
)
|
||
if (
|
||
host_local
|
||
and row["worker_pid"]
|
||
and _pid_alive(row["worker_pid"])
|
||
and not heartbeat_stale
|
||
):
|
||
new_expires = now + _resolve_claim_ttl_seconds()
|
||
with write_txn(conn):
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET claim_expires = ? "
|
||
"WHERE id = ? AND status = 'running' "
|
||
" AND claim_lock IS ? "
|
||
" AND claim_expires IS NOT NULL "
|
||
" AND claim_expires < ?",
|
||
(new_expires, row["id"], row["claim_lock"], now),
|
||
)
|
||
if cur.rowcount != 1:
|
||
continue
|
||
run_id = _current_run_id(conn, row["id"])
|
||
if run_id is not None:
|
||
conn.execute(
|
||
"UPDATE task_runs SET claim_expires = ? WHERE id = ?",
|
||
(new_expires, run_id),
|
||
)
|
||
_append_event(
|
||
conn, row["id"], "claim_extended",
|
||
{
|
||
"reason": "pid_alive",
|
||
"worker_pid": int(row["worker_pid"]),
|
||
"claim_lock": row["claim_lock"],
|
||
"claim_expires_was": int(row["claim_expires"]),
|
||
"claim_expires_now": new_expires,
|
||
"last_heartbeat_at": _opt_int(row["last_heartbeat_at"]),
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
continue
|
||
|
||
termination = _terminate_reclaimed_worker(
|
||
row["worker_pid"], row["claim_lock"], signal_fn=signal_fn,
|
||
)
|
||
# Never release a claim while our own worker is still alive: that would
|
||
# spawn a duplicate beside it. Hold the claim and retry next tick.
|
||
if _worker_survived_termination(termination):
|
||
_defer_reclaim_for_live_worker(
|
||
conn, row["id"], row["claim_lock"], now, termination,
|
||
reason="ttl_expired_worker_alive",
|
||
)
|
||
continue
|
||
with write_txn(conn):
|
||
retry_status = _retry_status_for_run(conn, row["id"])
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET status = ?, claim_lock = NULL, "
|
||
"claim_expires = NULL, worker_pid = NULL "
|
||
"WHERE id = ? AND status = 'running' AND claim_lock IS ? "
|
||
"AND claim_expires IS NOT NULL AND claim_expires < ?",
|
||
(retry_status, row["id"], row["claim_lock"], now),
|
||
)
|
||
if cur.rowcount != 1:
|
||
continue
|
||
run_id = _end_run(
|
||
conn, row["id"],
|
||
outcome="reclaimed", status="reclaimed",
|
||
error=f"stale_lock={row['claim_lock']}",
|
||
metadata=termination,
|
||
)
|
||
payload = {
|
||
"stale_lock": row["claim_lock"],
|
||
"worker_pid": _opt_int(row["worker_pid"]),
|
||
"claim_expires": int(row["claim_expires"]),
|
||
"last_heartbeat_at": _opt_int(row["last_heartbeat_at"]),
|
||
"now": now,
|
||
"host_local": host_local,
|
||
"heartbeat_stale": bool(heartbeat_stale),
|
||
"retry_status": retry_status,
|
||
}
|
||
payload.update(termination)
|
||
_append_event(
|
||
conn, row["id"], "reclaimed",
|
||
payload,
|
||
run_id=run_id,
|
||
)
|
||
reclaimed += 1
|
||
# Worker-lifecycle observer (RFC #58548): the reclaim txn above has
|
||
# committed. The ``continue`` branches (rowcount mismatch, claim
|
||
# extension, deferred reclaim) never reach this point, so only a
|
||
# genuinely reclaimed stale claim fires.
|
||
if _kanban_observer_consumed("on_kanban_worker_stale_claim"):
|
||
_fire_kanban_lifecycle_hook(
|
||
"on_kanban_worker_stale_claim",
|
||
row["id"],
|
||
board=get_current_board(),
|
||
assignee=row["assignee"],
|
||
run_id=run_id,
|
||
worker_pid=_opt_int(row["worker_pid"]),
|
||
heartbeat_stale=bool(heartbeat_stale),
|
||
retry_status=retry_status,
|
||
)
|
||
return reclaimed
|
||
|
||
|
||
def reclaim_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
reason: Optional[str] = None,
|
||
signal_fn=None,
|
||
) -> bool:
|
||
"""Operator-driven reclaim regardless of TTL (unlike
|
||
:func:`release_stale_claims`): release the claim and restore the source
|
||
phase so a running worker can be aborted on demand. False when the task
|
||
is missing or not running.
|
||
"""
|
||
row = conn.execute(
|
||
"SELECT status, claim_lock, worker_pid FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
if not row:
|
||
return False
|
||
if row["status"] != "running" and row["claim_lock"] is None:
|
||
# Nothing to reclaim — already ready / blocked / done.
|
||
return False
|
||
prev_lock = row["claim_lock"]
|
||
termination = _terminate_reclaimed_worker(
|
||
row["worker_pid"], prev_lock, signal_fn=signal_fn,
|
||
)
|
||
with write_txn(conn):
|
||
retry_status = _retry_status_for_run(conn, task_id)
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET status = ?, claim_lock = NULL, "
|
||
"claim_expires = NULL, worker_pid = NULL "
|
||
"WHERE id = ? AND status IN ('running', 'ready', 'blocked') "
|
||
"AND claim_lock IS ?",
|
||
(retry_status, task_id, prev_lock),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
run_id = _end_run(
|
||
conn, task_id,
|
||
outcome="reclaimed", status="reclaimed",
|
||
error=(
|
||
f"manual_reclaim: {reason}" if reason
|
||
else f"manual_reclaim lock={prev_lock}"
|
||
),
|
||
metadata=termination,
|
||
)
|
||
payload = {
|
||
"manual": True,
|
||
"reason": reason,
|
||
"prev_lock": prev_lock,
|
||
"retry_status": retry_status,
|
||
}
|
||
payload.update(termination)
|
||
_append_event(
|
||
conn, task_id, "reclaimed",
|
||
payload,
|
||
run_id=run_id,
|
||
)
|
||
# Operator intervention — they've looked at the task, so the
|
||
# consecutive-failures counter is now stale. Give the next retry
|
||
# a fresh budget. (_clear_failure_counter opens its own write_txn,
|
||
# so it runs after the enclosing one commits.)
|
||
_clear_failure_counter(conn, task_id)
|
||
return True
|
||
|
||
|
||
def reassign_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
profile: Optional[str],
|
||
*,
|
||
reclaim_first: bool = False,
|
||
reason: Optional[str] = None,
|
||
) -> bool:
|
||
"""Reassign (``profile=None`` unassigns). A running task is refused
|
||
(False) unless ``reclaim_first`` releases its claim via
|
||
:func:`reclaim_task` first — the "this profile's model is broken, try
|
||
another" recovery path.
|
||
"""
|
||
if reclaim_first:
|
||
# Safe to call even if nothing to reclaim.
|
||
reclaim_task(conn, task_id, reason=reason or "reassign")
|
||
# assign_task handles its own txn + the still-running guard.
|
||
try:
|
||
return assign_task(conn, task_id, profile)
|
||
except RuntimeError:
|
||
# Task is still running and reclaim_first was False; caller
|
||
# needs to decide whether to retry with reclaim.
|
||
return False
|
||
|
||
|
||
def _verify_created_cards(
|
||
conn: sqlite3.Connection,
|
||
completing_task_id: str,
|
||
claimed_ids: Iterable[str],
|
||
) -> tuple[list[str], list[str]]:
|
||
"""Partition ``claimed_ids`` into (verified, phantom).
|
||
|
||
A card is "verified" iff a row exists in ``tasks`` AND at least one
|
||
of the following holds:
|
||
|
||
* ``created_by`` matches the completing task's ``assignee`` profile
|
||
(the common case: worker A spawns a card via ``kanban_create``,
|
||
which stamps ``created_by=A``).
|
||
* ``created_by`` matches the completing task's id (edge case where
|
||
a worker passed its own task id as the ``created_by`` value).
|
||
* The card is linked as a ``task_links.child`` of the completing
|
||
task — i.e. the worker explicitly called ``kanban_create`` with
|
||
``parents=[<current_task>]``. This accepts cards created through
|
||
the dashboard/CLI by a different principal but then attached to
|
||
the completing task by the worker.
|
||
|
||
``phantom`` returns ids that either don't exist at all, or exist
|
||
but don't satisfy any of the three trust conditions. The caller
|
||
decides what to do with each bucket; this helper never mutates.
|
||
"""
|
||
claimed = [str(x).strip() for x in (claimed_ids or []) if str(x).strip()]
|
||
if not claimed:
|
||
return [], []
|
||
# Dedupe while preserving order.
|
||
seen: set[str] = set()
|
||
ordered: list[str] = []
|
||
for cid in claimed:
|
||
if cid not in seen:
|
||
seen.add(cid)
|
||
ordered.append(cid)
|
||
|
||
row = conn.execute(
|
||
"SELECT assignee FROM tasks WHERE id = ?", (completing_task_id,),
|
||
).fetchone()
|
||
if row is None:
|
||
# Completing task not found — nothing resolves.
|
||
return [], ordered
|
||
completing_assignee = row["assignee"]
|
||
|
||
# Batch-fetch existence + created_by in one query.
|
||
placeholders = ",".join(["?"] * len(ordered))
|
||
rows = conn.execute(
|
||
f"SELECT id, created_by FROM tasks WHERE id IN ({placeholders})",
|
||
tuple(ordered),
|
||
).fetchall()
|
||
found = {r["id"]: r["created_by"] for r in rows}
|
||
|
||
# Pull the set of cards linked as children of the completing task.
|
||
# Cheap: one query, indexed on parent_id.
|
||
linked_children: set[str] = set(child_ids(conn, completing_task_id))
|
||
|
||
verified: list[str] = []
|
||
phantom: list[str] = []
|
||
for cid in ordered:
|
||
created_by = found.get(cid)
|
||
if created_by is None:
|
||
phantom.append(cid)
|
||
continue
|
||
# Accept if any of the three trust conditions holds.
|
||
if (
|
||
(completing_assignee and created_by == completing_assignee)
|
||
or created_by == completing_task_id
|
||
or cid in linked_children
|
||
):
|
||
verified.append(cid)
|
||
else:
|
||
phantom.append(cid)
|
||
return verified, phantom
|
||
|
||
|
||
# Task-id pattern used both by ``kanban_create`` (``t_<12 hex>``) and
|
||
# ``_new_task_id`` below. Kept permissive on length for forward compat:
|
||
# accept 8+ hex chars after the ``t_`` prefix.
|
||
_TASK_ID_PROSE_RE = re.compile(r"\bt_[a-f0-9]{8,}\b")
|
||
|
||
|
||
def _scan_prose_for_phantom_ids(
|
||
conn: sqlite3.Connection,
|
||
text: str,
|
||
) -> list[str]:
|
||
"""Regex-scan free-form text for ``t_<hex>`` references; return the
|
||
ones that don't exist in ``tasks``.
|
||
|
||
Used as a non-blocking advisory check on completion summaries. An
|
||
empty return means "no suspicious references found" — either the
|
||
text had no IDs at all, or every ID it mentioned resolves to a real
|
||
task. Duplicates are deduped.
|
||
"""
|
||
if not text:
|
||
return []
|
||
matches = _TASK_ID_PROSE_RE.findall(text)
|
||
if not matches:
|
||
return []
|
||
# Dedupe preserving order.
|
||
seen: set[str] = set()
|
||
unique: list[str] = []
|
||
for m in matches:
|
||
if m not in seen:
|
||
seen.add(m)
|
||
unique.append(m)
|
||
placeholders = ",".join(["?"] * len(unique))
|
||
rows = conn.execute(
|
||
f"SELECT id FROM tasks WHERE id IN ({placeholders})",
|
||
tuple(unique),
|
||
).fetchall()
|
||
existing = {r["id"] for r in rows}
|
||
return [m for m in unique if m not in existing]
|
||
|
||
|
||
class HallucinatedCardsError(ValueError):
|
||
"""Raised by ``complete_task`` when ``created_cards`` contains ids
|
||
that don't exist or weren't created by the completing worker.
|
||
|
||
The phantom list is attached as ``.phantom`` for callers that want
|
||
structured access. Kept as ``ValueError`` subclass so existing
|
||
tool-error handlers treat it as a recoverable user error.
|
||
"""
|
||
|
||
def __init__(self, phantom: list[str], completing_task_id: str):
|
||
self.phantom = list(phantom)
|
||
self.completing_task_id = completing_task_id
|
||
super().__init__(
|
||
f"completion blocked: claimed created_cards that do not exist "
|
||
f"or were not created by this worker: {', '.join(phantom)}"
|
||
)
|
||
|
||
|
||
class ArtifactPreservationError(RuntimeError):
|
||
"""Raised when a declared scratch deliverable cannot be preserved."""
|
||
|
||
|
||
def complete_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
result: Optional[str] = None,
|
||
summary: Optional[str] = None,
|
||
metadata: Optional[dict] = None,
|
||
created_cards: Optional[Iterable[str]] = None,
|
||
expected_run_id: Optional[int] = None,
|
||
fire_lifecycle_hook: bool = True,
|
||
) -> bool:
|
||
"""Transition ``running|ready|blocked|review -> done`` and record ``result``.
|
||
|
||
Accepts a task that is merely ``ready`` too, so a manual CLI
|
||
completion (``hermes kanban complete <id>``) works without requiring
|
||
a claim/start/complete sequence. ``review`` is accepted so a human
|
||
(or reviewer) can approve a task parked in the review lane by
|
||
:func:`request_review` — even when it has no active run
|
||
(``current_run_id IS NULL``), the handoff fields are preserved via
|
||
:func:`_synthesize_ended_run`.
|
||
|
||
``summary`` and ``metadata`` are stored on the closing run (if any)
|
||
and surfaced to downstream children via :func:`build_worker_context`.
|
||
When ``summary`` is omitted we fall back to ``result`` so single-run
|
||
callers do not have to pass both. ``metadata`` is a free-form dict
|
||
(e.g. ``{"changed_files": [...], "tests_run": [...]}``) — workers
|
||
are encouraged to use it for structured handoff facts.
|
||
|
||
``created_cards`` is an optional list of task ids the completing
|
||
worker claims to have created. Each id is verified against
|
||
``tasks.created_by``. If any id is phantom (does not exist or was
|
||
not created by this worker's assignee profile), completion is blocked
|
||
with a ``HallucinatedCardsError`` and a
|
||
``completion_blocked_hallucination`` event is emitted so the rejected
|
||
attempt is auditable. When all ids verify, they are recorded on the
|
||
``completed`` event payload.
|
||
|
||
After a successful completion, ``summary`` and ``result`` are scanned
|
||
for prose references like ``t_deadbeefcafe`` that do not resolve.
|
||
Any suspected phantom references are recorded as a
|
||
``suspected_hallucinated_references`` event. This pass is advisory
|
||
and never blocks.
|
||
"""
|
||
now = int(time.time())
|
||
# Fail before validating cards or staging artifacts; re-check inside the
|
||
# final write transaction below to close the parent-reopen race.
|
||
if not _parents_satisfied(conn, task_id):
|
||
return False
|
||
|
||
# Gate: verify created_cards BEFORE the main write txn. A rejected
|
||
# completion still needs an auditable event, so we emit it in a
|
||
# tiny dedicated txn, then raise. The caller is responsible for
|
||
# surfacing HallucinatedCardsError to the worker; this function
|
||
# never mutates task state on a phantom-card rejection.
|
||
if created_cards:
|
||
verified_cards, phantom_cards = _verify_created_cards(
|
||
conn, task_id, created_cards
|
||
)
|
||
if phantom_cards:
|
||
with write_txn(conn):
|
||
_append_event(
|
||
conn, task_id, "completion_blocked_hallucination",
|
||
{
|
||
"phantom_cards": phantom_cards,
|
||
"verified_cards": verified_cards,
|
||
"summary_preview": (
|
||
(summary or result or "").strip().splitlines()[0][:200]
|
||
if (summary or result)
|
||
else None
|
||
),
|
||
},
|
||
)
|
||
raise HallucinatedCardsError(phantom_cards, task_id)
|
||
else:
|
||
verified_cards = []
|
||
|
||
metadata = _merge_completion_prose_artifacts(
|
||
conn, task_id, metadata, summary=summary, result=result,
|
||
)
|
||
with write_txn(conn):
|
||
# Parent completion is a hard invariant even for direct human review
|
||
# approval. A parent may have been reopened after this task entered
|
||
# ``review`` or ``running``.
|
||
if not _parents_satisfied(conn, task_id):
|
||
return False
|
||
prior_status = _task_status(conn, task_id)
|
||
sql = """
|
||
UPDATE tasks
|
||
SET status = 'done',
|
||
result = ?,
|
||
completed_at = ?,
|
||
claim_lock = NULL,
|
||
claim_expires= NULL,
|
||
worker_pid = NULL,
|
||
block_kind = NULL,
|
||
block_recurrences = 0
|
||
WHERE id = ?
|
||
AND status IN ('running', 'ready', 'blocked', 'review')
|
||
"""
|
||
params: tuple = (result, now, task_id)
|
||
if expected_run_id is not None:
|
||
sql += " AND current_run_id = ?"
|
||
params = (*params, int(expected_run_id))
|
||
cur = conn.execute(sql, params)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
if isinstance(metadata, dict):
|
||
_persist_scratch_completion_artifacts(conn, task_id, metadata)
|
||
for stored_path in metadata.pop("_staged_artifacts", []):
|
||
path = Path(stored_path)
|
||
_insert_completion_attachment(
|
||
conn,
|
||
task_id,
|
||
filename=path.name,
|
||
stored_path=str(path),
|
||
size=path.stat().st_size,
|
||
created_at=now,
|
||
)
|
||
run_id = _end_run(
|
||
conn, task_id,
|
||
outcome="completed", status="done",
|
||
summary=summary if summary is not None else result,
|
||
metadata=metadata,
|
||
)
|
||
# If complete_task was called on a never-claimed task (ready or
|
||
# blocked → done with no run in flight), synthesize a
|
||
# zero-duration run so the handoff fields are persisted in
|
||
# attempt history instead of silently lost.
|
||
if run_id is None and (
|
||
summary or metadata or result or prior_status == "review"
|
||
):
|
||
synth_summary = summary if summary is not None else result
|
||
synth_metadata = metadata
|
||
if prior_status == "review" and not synth_summary and not synth_metadata:
|
||
synth_summary = "Review approved without additional evidence."
|
||
synth_metadata = {
|
||
"source_status": "review",
|
||
"approval": "manual",
|
||
}
|
||
run_id = _synthesize_ended_run(
|
||
conn, task_id,
|
||
outcome="completed",
|
||
summary=synth_summary,
|
||
metadata=synth_metadata,
|
||
)
|
||
# Carry the handoff summary in the event payload so gateway
|
||
# notifiers and dashboard WS consumers can render it without a
|
||
# second SQL round-trip. First line only, 400 char cap — the
|
||
# full summary stays on the run row.
|
||
event_summary = summary if summary is not None else result
|
||
if prior_status == "review" and not event_summary:
|
||
event_summary = "Review approved without additional evidence."
|
||
_ev_lines = (event_summary or "").strip().splitlines()
|
||
ev_summary = _ev_lines[0][:400] if _ev_lines else ""
|
||
completed_payload: dict = {
|
||
"result_len": len(result) if result else 0,
|
||
"summary": ev_summary or None,
|
||
}
|
||
if verified_cards:
|
||
completed_payload["verified_cards"] = verified_cards
|
||
# Carry artifact paths in the event payload so the gateway
|
||
# notifier can upload them as native attachments alongside the
|
||
# completion message. Workers pass these via
|
||
# ``kanban_complete(artifacts=[...])`` which stashes the list in
|
||
# ``metadata["artifacts"]`` — we promote it onto the event so
|
||
# consumers don't have to fetch the run row to find it.
|
||
if isinstance(metadata, dict):
|
||
md_artifacts = metadata.get("artifacts")
|
||
if isinstance(md_artifacts, (list, tuple)):
|
||
cleaned_artifacts = [
|
||
str(p).strip() for p in md_artifacts if isinstance(p, str) and str(p).strip()
|
||
]
|
||
if cleaned_artifacts:
|
||
completed_payload["artifacts"] = cleaned_artifacts
|
||
_append_event(
|
||
conn, task_id, "completed",
|
||
completed_payload,
|
||
run_id=run_id,
|
||
)
|
||
# Prose-scan the summary + result for t_<hex> references that do
|
||
# not resolve. Advisory — does not block the completion. Runs in
|
||
# its own txn so the completion itself is already durable by the
|
||
# time we emit the warning.
|
||
scan_text = " ".join(filter(None, [summary, result]))
|
||
if scan_text:
|
||
phantom_refs = _scan_prose_for_phantom_ids(conn, scan_text)
|
||
# Drop any phantom refs that were already flagged as verified
|
||
# above (shouldn't happen — verified means they exist — but
|
||
# belt-and-suspenders).
|
||
phantom_refs = [p for p in phantom_refs if p not in set(verified_cards)]
|
||
if phantom_refs:
|
||
with write_txn(conn):
|
||
_append_event(
|
||
conn, task_id, "suspected_hallucinated_references",
|
||
{
|
||
"phantom_refs": phantom_refs,
|
||
"source": "completion_summary",
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
# Successful completion — wipe the consecutive-failures counter.
|
||
# Failure history stays on the event log for audit; the counter
|
||
# just tracks "is there a current pathology the breaker should
|
||
# care about", and a success resets that question.
|
||
_clear_failure_counter(conn, task_id)
|
||
# Recompute ready status for dependents (separate txn so children see done).
|
||
recompute_ready(conn)
|
||
# Clean up the scratch workspace and any stale tmux session for the worker.
|
||
_cleanup_workspace(conn, task_id)
|
||
_done_task = get_task(conn, task_id)
|
||
if fire_lifecycle_hook:
|
||
_fire_kanban_lifecycle_hook(
|
||
"kanban_task_completed",
|
||
task_id,
|
||
board=get_current_board(),
|
||
assignee=_done_task.assignee if _done_task else None,
|
||
run_id=run_id,
|
||
summary=(summary if summary is not None else result),
|
||
)
|
||
return True
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Workspace / tmux cleanup
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def _merge_completion_prose_artifacts(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
metadata: Optional[dict],
|
||
*,
|
||
summary: Optional[str],
|
||
result: Optional[str],
|
||
) -> Optional[dict]:
|
||
"""Promote existing scratch files named in legacy completion prose.
|
||
|
||
``artifacts=[...]`` is preferred. Older workers only wrote an absolute
|
||
deliverable path in ``summary``/``result``; discover it while scratch still
|
||
exists so cleanup cannot erase the file the user was promised.
|
||
"""
|
||
workspace = _scratch_workspace(conn, task_id)
|
||
if workspace is None:
|
||
return metadata
|
||
if not _is_managed_scratch_path(workspace):
|
||
return metadata
|
||
text = "\n".join(part for part in (summary, result) if part)
|
||
if not text:
|
||
return metadata
|
||
prefix = re.escape(str(workspace))
|
||
discovered: list[str] = []
|
||
for match in re.finditer(prefix + r"(?:[/\\][^\s`\"'<>]+)", text):
|
||
raw = match.group(0).rstrip(".,;:!?)]}")
|
||
candidate = Path(raw)
|
||
if candidate.is_file():
|
||
discovered.append(str(candidate))
|
||
if not discovered:
|
||
return metadata
|
||
updated = dict(metadata) if isinstance(metadata, dict) else {}
|
||
existing = updated.get("artifacts")
|
||
merged = list(existing) if isinstance(existing, (list, tuple)) else []
|
||
seen = {str(path) for path in merged}
|
||
for path in discovered:
|
||
if path not in seen:
|
||
merged.append(path)
|
||
seen.add(path)
|
||
updated["artifacts"] = merged
|
||
return updated
|
||
|
||
|
||
def _persist_scratch_completion_artifacts(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
metadata: dict,
|
||
) -> None:
|
||
"""Copy scratch-workspace completion artifacts before cleanup removes them."""
|
||
raw_artifacts = metadata.get("artifacts")
|
||
if not isinstance(raw_artifacts, (list, tuple)):
|
||
return
|
||
|
||
workspace = _scratch_workspace(conn, task_id)
|
||
if workspace is None:
|
||
return
|
||
is_managed, board = _managed_scratch_path_info(workspace)
|
||
if not is_managed:
|
||
return
|
||
|
||
try:
|
||
workspace_root = workspace.resolve()
|
||
except OSError:
|
||
return
|
||
|
||
attachment_dir = task_attachments_dir(task_id, board=board)
|
||
persisted: list[str] = []
|
||
used_destinations: set[Path] = set()
|
||
changed = False
|
||
|
||
def _discard_copies() -> None:
|
||
for copied in used_destinations:
|
||
with contextlib.suppress(OSError):
|
||
copied.unlink(missing_ok=True)
|
||
with contextlib.suppress(OSError):
|
||
attachment_dir.rmdir()
|
||
|
||
for item in raw_artifacts:
|
||
artifact = str(item).strip() if isinstance(item, str) else ""
|
||
if not artifact:
|
||
continue
|
||
src = Path(artifact).expanduser()
|
||
try:
|
||
resolved_src = src.resolve()
|
||
except OSError:
|
||
persisted.append(artifact)
|
||
continue
|
||
|
||
if not resolved_src.is_relative_to(workspace_root):
|
||
persisted.append(artifact)
|
||
continue
|
||
|
||
if not src.is_file():
|
||
_discard_copies()
|
||
raise ArtifactPreservationError(
|
||
f"declared scratch artifact is unavailable or not a regular file: {artifact}"
|
||
)
|
||
|
||
size = resolved_src.stat().st_size
|
||
if size > KANBAN_ATTACHMENT_MAX_BYTES:
|
||
_discard_copies()
|
||
raise ArtifactPreservationError(
|
||
f"declared scratch artifact exceeds the "
|
||
f"{KANBAN_ATTACHMENT_MAX_BYTES}-byte limit: {artifact}"
|
||
)
|
||
|
||
dest: Optional[Path] = None
|
||
try:
|
||
attachment_dir.mkdir(parents=True, exist_ok=True)
|
||
dest = _unique_attachment_path(attachment_dir, resolved_src.name, used_destinations)
|
||
with resolved_src.open("rb") as source_file, dest.open("xb") as destination_file:
|
||
copied = 0
|
||
while chunk := source_file.read(1024 * 1024):
|
||
copied += len(chunk)
|
||
if copied > KANBAN_ATTACHMENT_MAX_BYTES:
|
||
raise ArtifactPreservationError(
|
||
f"declared scratch artifact grew beyond the size limit: {artifact}"
|
||
)
|
||
destination_file.write(chunk)
|
||
except Exception as exc:
|
||
if dest is not None:
|
||
with contextlib.suppress(OSError):
|
||
dest.unlink(missing_ok=True)
|
||
_discard_copies()
|
||
if isinstance(exc, ArtifactPreservationError):
|
||
raise
|
||
raise ArtifactPreservationError(
|
||
f"could not preserve declared scratch artifact {artifact}: {exc}"
|
||
) from exc
|
||
|
||
used_destinations.add(dest)
|
||
persisted.append(str(dest.resolve()))
|
||
changed = True
|
||
|
||
if changed:
|
||
metadata["artifacts"] = persisted
|
||
metadata["_staged_artifacts"] = [
|
||
path for path in persisted if path.startswith(str(attachment_dir.resolve()))
|
||
]
|
||
|
||
|
||
def _insert_completion_attachment(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
filename: str,
|
||
stored_path: str,
|
||
size: int,
|
||
created_at: int,
|
||
) -> None:
|
||
"""Record a worker-produced artifact in the existing attachment table."""
|
||
conn.execute(
|
||
"INSERT INTO task_attachments "
|
||
"(task_id, filename, stored_path, content_type, size, uploaded_by, created_at) "
|
||
"VALUES (?, ?, ?, NULL, ?, 'kanban_complete', ?)",
|
||
(task_id, filename, stored_path, size, created_at),
|
||
)
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"attached",
|
||
{"filename": filename, "size": size, "by": "kanban_complete"},
|
||
)
|
||
|
||
|
||
def _unique_attachment_path(directory: Path, filename: str, used: set[Path]) -> Path:
|
||
"""Return a non-conflicting path under ``directory`` for ``filename``."""
|
||
safe_name = Path(filename).name or "artifact"
|
||
candidate = directory / safe_name
|
||
if candidate not in used and not candidate.exists():
|
||
return candidate
|
||
|
||
stem = Path(safe_name).stem or "artifact"
|
||
suffix = Path(safe_name).suffix
|
||
idx = 1
|
||
while True:
|
||
candidate = directory / f"{stem}_{idx}{suffix}"
|
||
if candidate not in used and not candidate.exists():
|
||
return candidate
|
||
idx += 1
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# First-use tip for scratch workspaces
|
||
# ---------------------------------------------------------------------------
|
||
#
|
||
# Scratch workspaces are intentionally ephemeral (``_cleanup_workspace`` removes
|
||
# them on ``complete_task``); new users lose worker output without warning. On
|
||
# the FIRST scratch materialization per install: log a warning, append a
|
||
# ``tip_scratch_workspace`` event on the task, and touch a sentinel under
|
||
# ``kanban_home()`` so the tip never repeats. Per-install, not per-board.
|
||
|
||
|
||
def edit_completed_task_result(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
result: str,
|
||
summary: Optional[str] = None,
|
||
metadata: Optional[dict] = None,
|
||
) -> bool:
|
||
"""Backfill the user-visible result for an already completed task."""
|
||
handoff_summary = summary if summary is not None else result
|
||
with write_txn(conn):
|
||
if _task_status(conn, task_id) != "done":
|
||
return False
|
||
conn.execute(
|
||
"UPDATE tasks SET result = ? WHERE id = ?",
|
||
(result, task_id),
|
||
)
|
||
run = conn.execute(
|
||
"""
|
||
SELECT id FROM task_runs
|
||
WHERE task_id = ?
|
||
AND outcome = 'completed'
|
||
ORDER BY COALESCE(ended_at, started_at, 0) DESC, id DESC
|
||
LIMIT 1
|
||
""",
|
||
(task_id,),
|
||
).fetchone()
|
||
run_id = int(run["id"]) if run else None
|
||
if run_id is None:
|
||
run_id = _synthesize_ended_run(
|
||
conn, task_id,
|
||
outcome="completed",
|
||
summary=handoff_summary,
|
||
metadata=metadata,
|
||
)
|
||
else:
|
||
conn.execute(
|
||
"UPDATE task_runs SET summary = ? WHERE id = ?",
|
||
(handoff_summary, run_id),
|
||
)
|
||
if metadata is not None:
|
||
conn.execute(
|
||
"UPDATE task_runs SET metadata = ? WHERE id = ?",
|
||
(json.dumps(metadata, ensure_ascii=False), run_id),
|
||
)
|
||
_ev_lines = (handoff_summary or "").strip().splitlines()
|
||
ev_summary = _ev_lines[0][:400] if _ev_lines else ""
|
||
_append_event(
|
||
conn, task_id, "edited",
|
||
{
|
||
"fields": (
|
||
["result", "summary"]
|
||
+ (["metadata"] if metadata is not None else [])
|
||
),
|
||
"result_len": len(result) if result else 0,
|
||
"summary": ev_summary or None,
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
return True
|
||
|
||
|
||
def block_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
reason: Optional[str] = None,
|
||
kind: Optional[str] = None,
|
||
expected_run_id: Optional[int] = None,
|
||
) -> bool:
|
||
"""Transition ``running``/``ready`` → ``blocked`` (or route elsewhere).
|
||
|
||
``kind`` (:data:`VALID_BLOCK_KINDS` or ``None`` = legacy un-typed) routes:
|
||
``dependency`` -> ``todo`` (parent gating / ``recompute_ready`` promotes
|
||
it; never parked where a cron would keep "unblocking" it); everything
|
||
else -> ``blocked`` for a human, EXCEPT that a re-block for the SAME kind
|
||
after an unblock bumps ``block_recurrences`` and at
|
||
:data:`BLOCK_RECURRENCE_LIMIT` routes to ``triage`` instead, breaking the
|
||
cron-unblock ↔ worker-re-block loop. ``transient`` signals "may clear on
|
||
its own" but still counts toward the loop breaker so a forever-flaky task
|
||
escalates. Returns True on any transition, False when not blockable.
|
||
"""
|
||
if kind is not None and kind not in VALID_BLOCK_KINDS:
|
||
raise ValueError(
|
||
f"block kind must be one of {sorted(VALID_BLOCK_KINDS)} or None"
|
||
)
|
||
with write_txn(conn):
|
||
cur_row = conn.execute(
|
||
"SELECT status, block_kind, block_recurrences FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
if cur_row is None:
|
||
return False
|
||
source_status = (
|
||
_retry_status_for_run(conn, task_id)
|
||
if cur_row["status"] == "running"
|
||
else "ready"
|
||
)
|
||
prev_kind = _row_get(cur_row, "block_kind")
|
||
prev_recurrences = _row_get(cur_row, "block_recurrences")
|
||
prev_recurrences = int(prev_recurrences) if prev_recurrences is not None else 0
|
||
|
||
# Dependency blocks never enter the human ``blocked`` bucket — they
|
||
# wait in ``todo`` and let ``recompute_ready`` gate on parents. Routing
|
||
# here (rather than ``blocked``) is what keeps a cron from ever seeing
|
||
# a dependency-wait as something to "unblock".
|
||
if kind == "dependency":
|
||
new_status, event_kind = "todo", "dependency_wait"
|
||
set_sql, params = "block_kind = ?", (kind,)
|
||
payload = {"reason": reason, "kind": kind, "source_status": source_status}
|
||
else:
|
||
# Truly-blocked kinds. Increment the unblock-loop counter when this
|
||
# is a re-block for the SAME reason after a prior unblock: block_task
|
||
# only fires from running/ready (i.e. AFTER an unblock returned the
|
||
# task to the work pool), so a stored block_kind matching the
|
||
# incoming kind means blocked → unblocked → re-block, same cause.
|
||
# An un-typed (None) block compares as "same" to a prior un-typed one.
|
||
recurrences = prev_recurrences + 1 if prev_kind == kind else 1
|
||
set_sql = "block_kind = ?,\n block_recurrences = ?"
|
||
params = (kind, recurrences)
|
||
payload = {
|
||
"reason": reason,
|
||
"kind": kind,
|
||
"recurrences": recurrences,
|
||
"source_status": source_status,
|
||
}
|
||
if recurrences >= BLOCK_RECURRENCE_LIMIT:
|
||
# Loop detected — route to triage for a human-in-the-loop
|
||
# decision instead of letting the unblocker spin this task.
|
||
new_status, event_kind = "triage", "block_loop_detected"
|
||
payload["limit"] = BLOCK_RECURRENCE_LIMIT
|
||
else:
|
||
new_status, event_kind = "blocked", "blocked"
|
||
sql = f"""
|
||
UPDATE tasks
|
||
SET status = '{new_status}',
|
||
claim_lock = NULL,
|
||
claim_expires = NULL,
|
||
worker_pid = NULL,
|
||
{set_sql}
|
||
WHERE id = ?
|
||
AND status IN ('running', 'ready')
|
||
"""
|
||
params = (*params, task_id)
|
||
if expected_run_id is not None:
|
||
sql += " AND current_run_id = ?"
|
||
params = (*params, int(expected_run_id))
|
||
cur = conn.execute(sql, params)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
run_id = _end_run(
|
||
conn, task_id,
|
||
outcome="blocked", status="blocked",
|
||
summary=reason,
|
||
)
|
||
# Synthesize a run when blocking a never-claimed task so the reason
|
||
# is preserved in attempt history.
|
||
if run_id is None and reason:
|
||
run_id = _synthesize_ended_run(
|
||
conn, task_id, outcome="blocked", summary=reason,
|
||
)
|
||
_append_event(conn, task_id, event_kind, payload, run_id=run_id)
|
||
_blocked_task = get_task(conn, task_id)
|
||
|
||
def _fire_blocked_hook() -> None:
|
||
_fire_kanban_lifecycle_hook(
|
||
"kanban_task_blocked",
|
||
task_id,
|
||
board=get_current_board(),
|
||
assignee=_blocked_task.assignee if _blocked_task else None,
|
||
run_id=run_id,
|
||
reason=reason,
|
||
)
|
||
|
||
if kind == "dependency":
|
||
# Historical ordering: the dependency lane fires inside the txn.
|
||
_fire_blocked_hook()
|
||
return True
|
||
_fire_blocked_hook()
|
||
return True
|
||
|
||
|
||
def redact_review_value(value: Any) -> Any:
|
||
"""Redact secrets at the domain boundary for durable review handoffs."""
|
||
if isinstance(value, str):
|
||
from agent.redact import redact_sensitive_text
|
||
|
||
return redact_sensitive_text(value, force=True)
|
||
if isinstance(value, dict):
|
||
return {key: redact_review_value(item) for key, item in value.items()}
|
||
if isinstance(value, list):
|
||
return [redact_review_value(item) for item in value]
|
||
if isinstance(value, tuple):
|
||
return tuple(redact_review_value(item) for item in value)
|
||
return value
|
||
|
||
|
||
def request_review(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
summary: Optional[str] = None,
|
||
metadata: Optional[dict] = None,
|
||
reviewer: Optional[str] = None,
|
||
expected_run_id: Optional[int] = None,
|
||
force: bool = False,
|
||
with_reason: bool = False,
|
||
):
|
||
"""Transition implementation work into the first-class review phase.
|
||
|
||
Unlike :func:`block_task`, this transition never touches block recurrence
|
||
accounting. The current implementer and resolved reviewer are recorded on
|
||
the event so an autonomous reviewer can route requested changes back to the
|
||
right profile. Supplying ``reviewer`` reassigns the task before it is
|
||
exposed to the review dispatcher. On re-review, omitting it reuses the
|
||
reviewer provenance persisted by the latest ``changes_requested`` event.
|
||
|
||
When the task is ``running`` under a live claim, a caller that supplies no
|
||
``expected_run_id`` must pass ``force=True`` (explicit human/CLI override)
|
||
— otherwise the request is refused instead of silently clearing the live
|
||
worker's ``claim_lock``/``worker_pid``. Workers prove ownership by passing
|
||
their own run id as ``expected_run_id`` (unchanged).
|
||
|
||
Returns ``bool`` by default. With ``with_reason=True`` returns
|
||
``(ok, reason)`` mirroring :func:`request_changes` — ``reason`` is a
|
||
diagnostic string on failure, ``None`` on success.
|
||
"""
|
||
|
||
def _ret(ok: bool, reason: Optional[str] = None):
|
||
return (ok, reason) if with_reason else ok
|
||
|
||
summary = redact_review_value(summary)
|
||
metadata = redact_review_value(metadata)
|
||
with write_txn(conn):
|
||
if not _parents_satisfied(conn, task_id):
|
||
return _ret(False, "parent dependencies are not satisfied")
|
||
trow = conn.execute(
|
||
"SELECT assignee, status, claim_lock, current_run_id "
|
||
"FROM tasks WHERE id = ?", (task_id,),
|
||
).fetchone()
|
||
if trow is None:
|
||
return _ret(False, "task not found")
|
||
# Refuse to clear a live worker's claim without proof of ownership
|
||
# (expected_run_id) or an explicit human override (force=True).
|
||
if (
|
||
expected_run_id is None
|
||
and not force
|
||
and trow["status"] == "running"
|
||
and trow["claim_lock"] is not None
|
||
):
|
||
return _ret(
|
||
False,
|
||
"task is running under a live claim; pass expected_run_id "
|
||
"(worker ownership) or force=True (explicit operator "
|
||
"override) instead of clearing the live run's claim",
|
||
)
|
||
implementer = trow["assignee"]
|
||
if reviewer is None:
|
||
changes_run = conn.execute(
|
||
"SELECT id FROM task_runs "
|
||
"WHERE task_id = ? AND outcome = 'changes_requested' "
|
||
"ORDER BY id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
changes_event = None
|
||
if changes_run is not None:
|
||
changes_event = _latest_event(
|
||
conn, task_id, "changes_requested", changes_run["id"],
|
||
)
|
||
prior_reviewer = _json_dict(_row_get(changes_event, "payload")).get("reviewer")
|
||
if changes_run is not None:
|
||
if not isinstance(prior_reviewer, str) or not prior_reviewer.strip():
|
||
return _ret(
|
||
False,
|
||
"re-review has no durable reviewer provenance (the "
|
||
"latest changes_requested event is missing or "
|
||
"malformed); pass reviewer= explicitly",
|
||
)
|
||
reviewer = prior_reviewer
|
||
reviewer = _canonical_assignee(reviewer) if reviewer is not None else None
|
||
assignee_sql = ", assignee = ?" if reviewer is not None else ""
|
||
params: tuple[Any, ...]
|
||
if expected_run_id is None:
|
||
params = (reviewer, task_id) if reviewer is not None else (task_id,)
|
||
run_guard = ""
|
||
else:
|
||
params = (
|
||
(reviewer, task_id, int(expected_run_id))
|
||
if reviewer is not None
|
||
else (task_id, int(expected_run_id))
|
||
)
|
||
run_guard = " AND current_run_id = ?"
|
||
cur = conn.execute(
|
||
"""
|
||
UPDATE tasks
|
||
SET status = 'review',
|
||
claim_lock = NULL,
|
||
claim_expires = NULL,
|
||
worker_pid = NULL
|
||
""" + assignee_sql + """
|
||
WHERE id = ?
|
||
AND status IN ('running', 'ready')
|
||
""" + run_guard,
|
||
params,
|
||
)
|
||
if cur.rowcount != 1:
|
||
return _ret(
|
||
False,
|
||
"task is not in running/ready (or expected_run_id did not "
|
||
"match the current run)",
|
||
)
|
||
run_id = _end_run(
|
||
conn,
|
||
task_id,
|
||
outcome="review_requested",
|
||
status="review",
|
||
summary=summary,
|
||
metadata=metadata,
|
||
)
|
||
if run_id is None and (summary or metadata):
|
||
run_id = _synthesize_ended_run(
|
||
conn,
|
||
task_id,
|
||
outcome="review_requested",
|
||
summary=summary,
|
||
metadata=metadata,
|
||
)
|
||
lines = (summary or "").strip().splitlines()
|
||
event_summary = lines[0][:400] if lines else ""
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"review_requested",
|
||
{
|
||
"summary": event_summary or None,
|
||
"implementer": implementer,
|
||
"reviewer": reviewer,
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
return _ret(True)
|
||
|
||
|
||
def request_changes(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
reason: str,
|
||
expected_run_id: Optional[int] = None,
|
||
) -> tuple[bool, Optional[str]]:
|
||
"""Finish an active review run and route the task back for rework.
|
||
|
||
The transition is valid only for a run claimed from ``review``. It closes
|
||
that reviewer run, restores the implementer recorded by the latest
|
||
``review_requested`` event, reapplies parent gating, and emits an auditable
|
||
``changes_requested`` event. The second tuple item is the implementer on
|
||
success or a diagnostic reason on failure.
|
||
"""
|
||
reason = str(redact_review_value(reason or "")).strip()
|
||
if not reason:
|
||
return False, "reason is required"
|
||
|
||
with write_txn(conn):
|
||
task_row = conn.execute(
|
||
"SELECT status, assignee, current_run_id FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
if task_row is None:
|
||
return False, "task not found"
|
||
current_run_id = task_row["current_run_id"]
|
||
if task_row["status"] != "running" or current_run_id is None:
|
||
return False, "task is not in an active review run"
|
||
if expected_run_id is not None and int(current_run_id) != int(expected_run_id):
|
||
return False, "run_id mismatch"
|
||
|
||
claimed_event = _latest_event(conn, task_id, "claimed", current_run_id)
|
||
claimed_payload = _json_dict(_row_get(claimed_event, "payload"))
|
||
if claimed_payload.get("source_status") != "review":
|
||
return False, "active run was not claimed from review"
|
||
|
||
requested_event = _latest_event(conn, task_id, "review_requested")
|
||
if requested_event is None:
|
||
return False, "no prior review_requested event"
|
||
implementer = _json_dict(requested_event["payload"]).get("implementer")
|
||
if not isinstance(implementer, str) or not implementer.strip():
|
||
return False, "review handoff has no valid implementer provenance"
|
||
reviewer = task_row["assignee"]
|
||
if isinstance(reviewer, str) and reviewer.strip():
|
||
reviewer = _canonical_assignee(reviewer)
|
||
else:
|
||
reviewer = None
|
||
|
||
new_status = _landing_status_after_parents(conn, task_id)
|
||
# NOTE: consecutive_failures is deliberately PRESERVED (neither
|
||
# reset nor incremented). Review transitions are not evidence the
|
||
# pathology cleared — only complete_task's success path resets the
|
||
# breaker counter (mirrors unblock_task, #35072).
|
||
cur = conn.execute(
|
||
"""
|
||
UPDATE tasks
|
||
SET status = ?,
|
||
assignee = COALESCE(?, assignee),
|
||
claim_lock = NULL,
|
||
claim_expires = NULL,
|
||
worker_pid = NULL
|
||
WHERE id = ? AND status = 'running' AND current_run_id = ?
|
||
""",
|
||
(new_status, implementer, task_id, int(current_run_id)),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False, "task changed during review handoff"
|
||
run_id = _end_run(
|
||
conn,
|
||
task_id,
|
||
outcome="changes_requested",
|
||
status=new_status,
|
||
summary=reason,
|
||
)
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"changes_requested",
|
||
{
|
||
"reason": reason,
|
||
"implementer": implementer,
|
||
"reviewer": reviewer,
|
||
"status": new_status,
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
return True, implementer
|
||
|
||
|
||
def promote_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
actor: str,
|
||
reason: Optional[str] = None,
|
||
force: bool = False,
|
||
dry_run: bool = False,
|
||
) -> tuple[bool, Optional[str]]:
|
||
"""Manually promote a `todo` or `blocked` task to `ready`.
|
||
|
||
Mirrors the automatic promotion done by ``recompute_ready`` but
|
||
drives it from a deliberate operator action with an audit-trail
|
||
entry. Refuses to promote if any parent dep is not in a terminal
|
||
state (`done`/`archived`) unless ``force=True``. Does NOT change
|
||
assignee or claim state. Returns ``(True, None)`` on success and
|
||
``(False, reason)`` if refused. ``dry_run=True`` validates the
|
||
promotion would succeed without mutating state.
|
||
"""
|
||
cur_status = _task_status(conn, task_id)
|
||
if cur_status is None:
|
||
return False, f"task {task_id} not found"
|
||
|
||
if cur_status not in ("todo", "blocked"):
|
||
return False, (
|
||
f"task {task_id} is {cur_status!r}; promote only applies to "
|
||
f"'todo' or 'blocked'"
|
||
)
|
||
|
||
if not force:
|
||
parents = conn.execute(
|
||
"SELECT t.id, t.status FROM tasks t "
|
||
"JOIN task_links l ON l.parent_id = t.id "
|
||
"WHERE l.child_id = ?",
|
||
(task_id,),
|
||
).fetchall()
|
||
unsatisfied = [
|
||
p["id"] for p in parents
|
||
if p["status"] not in ("done", "archived")
|
||
]
|
||
if unsatisfied:
|
||
return False, (
|
||
f"unsatisfied parent dependencies: "
|
||
f"{', '.join(unsatisfied)} (use --force to override)"
|
||
)
|
||
|
||
if dry_run:
|
||
return True, None
|
||
|
||
with write_txn(conn):
|
||
upd = conn.execute(
|
||
"UPDATE tasks SET status = 'ready' "
|
||
"WHERE id = ? AND status IN ('todo', 'blocked')",
|
||
(task_id,),
|
||
)
|
||
if upd.rowcount != 1:
|
||
return False, f"task {task_id} status changed during promotion"
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"promoted_manual",
|
||
{"actor": actor, "reason": reason, "forced": force},
|
||
)
|
||
|
||
return True, None
|
||
|
||
|
||
def _reclaim_dangling_run(
|
||
conn: sqlite3.Connection, task_id: str, *, statuses, now: int, note: str,
|
||
) -> None:
|
||
"""Close a leaked ``current_run_id`` (run row still open) before a status
|
||
flip, preserving the runs invariant (``current_run_id IS NULL`` ⇔ run row
|
||
terminal). No-op in the common path where the prior transition already
|
||
closed the run. Shared by :func:`unblock_task` and
|
||
:func:`reopen_review_task` so the recovery can't drift.
|
||
"""
|
||
placeholders = ", ".join("?" for _ in statuses)
|
||
stale = conn.execute(
|
||
f"SELECT current_run_id FROM tasks WHERE id = ? AND status IN ({placeholders})",
|
||
(task_id, *statuses),
|
||
).fetchone()
|
||
if stale and stale["current_run_id"]:
|
||
conn.execute(
|
||
"""
|
||
UPDATE task_runs
|
||
SET status = 'reclaimed', outcome = 'reclaimed',
|
||
summary = COALESCE(summary, ?),
|
||
ended_at = ?,
|
||
claim_lock = NULL, claim_expires = NULL, worker_pid = NULL
|
||
WHERE id = ? AND ended_at IS NULL
|
||
""",
|
||
(note, now, int(stale["current_run_id"])),
|
||
)
|
||
|
||
|
||
def _landing_status_after_parents(conn: sqlite3.Connection, task_id: str) -> str:
|
||
"""Return ``'todo'`` if any parent isn't ``done`` yet, else ``'ready'``.
|
||
|
||
The parent-completion re-gate shared by :func:`unblock_task` and
|
||
:func:`reopen_review_task`: flipping straight to ``ready`` would bypass the
|
||
parent-completion invariant the dispatcher trusts (it would spawn a child
|
||
whose upstream work isn't finished). If parents are still in progress the
|
||
task waits in ``todo`` until ``recompute_ready`` picks it up. RCA: Bug 2 at
|
||
kanban/boards/cookai/workspaces/t_a6acd07d/root-cause.md. Kept in one place
|
||
so the two transitions can't drift.
|
||
"""
|
||
return "ready" if _parents_satisfied(conn, task_id) else "todo"
|
||
|
||
|
||
def unblock_task(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Transition ``blocked``/``scheduled`` to its safe resumable phase.
|
||
|
||
Defensively closes any stale ``current_run_id`` pointer before flipping
|
||
status. In the common path (``block_task`` closed the run already) this
|
||
is a no-op. If a future or external write left the pointer dangling,
|
||
the leaked run is closed as ``reclaimed`` inside the same txn so the
|
||
runs invariant (``current_run_id IS NULL`` ⇔ run row in terminal
|
||
state) holds for the rest of this function's lifetime.
|
||
"""
|
||
now = int(time.time())
|
||
with write_txn(conn):
|
||
resume_status = (
|
||
_resume_status_from_events(conn, task_id)
|
||
if _task_status(conn, task_id) == "blocked"
|
||
else "ready"
|
||
)
|
||
_reclaim_dangling_run(
|
||
conn, task_id, statuses=("blocked", "scheduled"), now=now,
|
||
note="invariant recovery on unblock",
|
||
)
|
||
# Re-gate on parent completion before restoring the source phase.
|
||
landing_status = _landing_status_after_parents(conn, task_id)
|
||
new_status = (
|
||
"review"
|
||
if landing_status == "ready" and resume_status == "review"
|
||
else landing_status
|
||
)
|
||
# NOTE: deliberately does NOT touch ``block_recurrences`` or
|
||
# ``block_kind``. Resetting the recurrence counter on unblock is exactly
|
||
# the amnesia that let a cron unblock → worker re-block loop run
|
||
# unbounded (Dale's report). The counter survives the unblock so that a
|
||
# subsequent same-cause ``block_task`` can detect the loop and route to
|
||
# triage at ``BLOCK_RECURRENCE_LIMIT``. It is reset to 0 only on a
|
||
# successful completion (see ``complete_task``). ``consecutive_failures``
|
||
# (the *dispatcher* spawn/crash/timeout counter — a different signal) is
|
||
# still reset here, which is correct: a deliberate unblock is a fresh
|
||
# start for the dispatcher's retry budget.
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET status = ?, current_run_id = NULL, "
|
||
"consecutive_failures = 0, last_failure_error = NULL "
|
||
"WHERE id = ? AND status IN ('blocked', 'scheduled')",
|
||
(new_status, task_id),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
_append_event(
|
||
conn, task_id, "unblocked",
|
||
(
|
||
{"status": new_status, "resume_status": resume_status}
|
||
if new_status != "ready" or resume_status != "ready"
|
||
else None
|
||
),
|
||
)
|
||
return True
|
||
|
||
|
||
def reopen_review_task(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Transition ``review`` -> ready (or todo) so the implementer re-runs.
|
||
|
||
The "changes requested" counterpart of :func:`request_review`: sends the
|
||
task back out of the review lane so the dispatcher re-runs the implementer
|
||
on the new comments. Mirrors :func:`unblock_task` (parent re-gating,
|
||
defensive stale-run close, ``consecutive_failures`` preserved) and emits a
|
||
``review_reopened`` event.
|
||
|
||
Deliberately does NOT touch ``block_recurrences``/``block_kind``: review is
|
||
not a block, so there is no loop counter to reset. (A stale counter from a
|
||
genuine block *before* review is left intact — only :func:`complete_task`
|
||
clears it.) Returns False when the task is missing or not in ``review``.
|
||
"""
|
||
now = int(time.time())
|
||
with write_txn(conn):
|
||
_reclaim_dangling_run(
|
||
conn, task_id, statuses=("review",), now=now,
|
||
note="invariant recovery on review reopen",
|
||
)
|
||
new_status = _landing_status_after_parents(conn, task_id)
|
||
review_event = _latest_event(conn, task_id, "review_requested")
|
||
handoff = _json_dict(_row_get(review_event, "payload"))
|
||
implementer = handoff.get("implementer")
|
||
if not isinstance(implementer, str) or not implementer.strip():
|
||
implementer = None
|
||
assignee_sql = ", assignee = ?" if implementer else ""
|
||
params: tuple[Any, ...] = (
|
||
(new_status, implementer, task_id)
|
||
if implementer
|
||
else (new_status, task_id)
|
||
)
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET status = ?, current_run_id = NULL, "
|
||
"claim_lock = NULL, claim_expires = NULL, worker_pid = NULL "
|
||
# consecutive_failures deliberately PRESERVED: review reopen is
|
||
# not a success signal; only complete_task resets the breaker
|
||
# counter (mirrors unblock_task, #35072).
|
||
+ assignee_sql
|
||
+ " WHERE id = ? AND status = 'review'",
|
||
params,
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
payload: dict[str, Any] = {"status": new_status}
|
||
if implementer:
|
||
payload["implementer"] = implementer
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"review_reopened",
|
||
payload if payload != {"status": "ready"} else None,
|
||
)
|
||
return True
|
||
|
||
|
||
def invalidate_descendants_for_parent_reopen(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
author: str,
|
||
) -> dict[str, Any]:
|
||
"""Retract every dispatchable/completed descendant of a reopened ancestor.
|
||
|
||
THE single implementation of done-reopen descendant invalidation. When a
|
||
``done``/``archived`` ancestor is reopened, every descendant whose state
|
||
assumed its result — ``ready``, ``review``, ``running`` or ``done`` — is
|
||
demoted to ``todo`` and re-gated on the graph. The CLI has no done-reopen
|
||
verb (``reopen-review`` is the review-phase transition), so every surface
|
||
that reopens a done task (dashboard drag-drop / PATCH via
|
||
``_set_status_direct``) must route through here.
|
||
|
||
Transactionality: composes under the caller's open transaction via
|
||
``write_txn(conn, allow_nested=True)`` so the ancestor's status flip and
|
||
the retractions commit atomically; standalone it opens its own. All SQL
|
||
is inline (no txn-opening helpers).
|
||
|
||
Non-silent contract — every invalidated descendant gets a
|
||
``descendant_invalidated`` event (``ancestor, prior_status, new_status,
|
||
resume_status``), the legacy ``status`` event
|
||
(``reason=ancestor_reopened``) the live feed renders, and a comment
|
||
naming the ancestor so operators see WHY a card moved.
|
||
|
||
Live ``running`` descendants are wasted spend: their run is closed
|
||
``reclaimed`` and the worker killed via :func:`_terminate_reclaimed_worker`
|
||
strictly post-commit, so the audit trail exists BEFORE the worker dies.
|
||
When composing under a caller's transaction the caller MUST drain the
|
||
returned ``terminations`` after its own commit.
|
||
|
||
``consecutive_failures`` resets to 0: ancestor reopen is a deliberate
|
||
operator action, so demoted work gets a fresh breaker budget. This is the
|
||
OPPOSITE of :func:`reopen_review_task` (preserves the counter) because the
|
||
autonomous review loop must not launder its own failure streak.
|
||
|
||
Returns ``{"invalidated": [...], "terminations": [...]}`` where each
|
||
invalidated entry is ``{id, prior_status, new_status, resume_status}``
|
||
and each termination is a ``(worker_pid, claim_lock)`` tuple.
|
||
"""
|
||
caller_owns_txn = bool(getattr(conn, "in_transaction", False))
|
||
now = int(time.time())
|
||
invalidated: list[dict[str, Any]] = []
|
||
terminations: list[tuple[Optional[int], Optional[str]]] = []
|
||
with write_txn(conn, allow_nested=True):
|
||
rows = conn.execute(
|
||
"""
|
||
WITH RECURSIVE descendants(id) AS (
|
||
SELECT child_id FROM task_links WHERE parent_id = ?
|
||
UNION
|
||
SELECT l.child_id
|
||
FROM task_links l
|
||
JOIN descendants d ON d.id = l.parent_id
|
||
)
|
||
SELECT t.id, t.status, t.current_run_id, t.worker_pid, t.claim_lock
|
||
FROM descendants d
|
||
JOIN tasks t ON t.id = d.id
|
||
ORDER BY t.id
|
||
""",
|
||
(task_id,),
|
||
).fetchall()
|
||
for row in rows:
|
||
previous_status = row["status"]
|
||
if previous_status not in {"ready", "review", "running", "done"}:
|
||
continue
|
||
resume_status = "ready"
|
||
run_id = None
|
||
if previous_status == "review":
|
||
resume_status = "review"
|
||
elif previous_status == "running":
|
||
resume_status = _retry_status_for_run(
|
||
conn, row["id"], row["current_run_id"]
|
||
)
|
||
terminations.append((row["worker_pid"], row["claim_lock"]))
|
||
run_id = _end_run(
|
||
conn,
|
||
row["id"],
|
||
outcome="reclaimed",
|
||
status="todo",
|
||
summary=f"ancestor {task_id} reopened",
|
||
)
|
||
# consecutive_failures = 0: deliberate operator reset — see
|
||
# docstring for why this diverges from reopen_review_task.
|
||
conn.execute(
|
||
"UPDATE tasks SET status = 'todo', completed_at = NULL, "
|
||
"claim_lock = NULL, claim_expires = NULL, worker_pid = NULL, "
|
||
"current_run_id = NULL, consecutive_failures = 0 WHERE id = ?",
|
||
(row["id"],),
|
||
)
|
||
_append_event(
|
||
conn,
|
||
row["id"],
|
||
"descendant_invalidated",
|
||
{
|
||
"ancestor": task_id,
|
||
"prior_status": previous_status,
|
||
"new_status": "todo",
|
||
"resume_status": resume_status,
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
# Legacy 'status' event kept so existing live-feed consumers
|
||
# still see the move without learning the new event kind.
|
||
_append_event(
|
||
conn,
|
||
row["id"],
|
||
"status",
|
||
{
|
||
"status": "todo",
|
||
"reason": "ancestor_reopened",
|
||
"parent": task_id,
|
||
"previous_status": previous_status,
|
||
"resume_status": resume_status,
|
||
},
|
||
run_id=run_id,
|
||
)
|
||
_insert_comment(
|
||
conn, row["id"], author,
|
||
f"Invalidated: ancestor {task_id} was reopened; "
|
||
f"retracted from '{previous_status}' to 'todo' "
|
||
f"(will resume via '{resume_status}').",
|
||
now,
|
||
)
|
||
invalidated.append(
|
||
{
|
||
"id": row["id"],
|
||
"prior_status": previous_status,
|
||
"new_status": "todo",
|
||
"resume_status": resume_status,
|
||
}
|
||
)
|
||
if not caller_owns_txn:
|
||
# Standalone call: we committed above, so the audit trail is durable
|
||
# — safe to kill workers now. Composed calls leave this to the
|
||
# caller (post-commit), preserving events-before-termination.
|
||
for pid, claim_lock in terminations:
|
||
_terminate_reclaimed_worker(pid, claim_lock)
|
||
return {"invalidated": invalidated, "terminations": terminations}
|
||
|
||
|
||
def specify_triage_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
title: Optional[str] = None,
|
||
body: Optional[str] = None,
|
||
assignee: Optional[str] = None,
|
||
author: Optional[str] = None,
|
||
) -> bool:
|
||
"""Flesh out a triage task and promote it to ``todo``.
|
||
|
||
Atomically updates ``title`` / ``body`` / ``assignee`` (when provided)
|
||
and transitions ``status: triage -> todo`` in a single write txn. Returns
|
||
False when the task is missing or not in the ``triage`` column — callers
|
||
should surface that as "nothing to specify" rather than an error.
|
||
|
||
``todo`` (not ``ready``) is the correct landing column: ``recompute_ready``
|
||
promotes parent-free / parent-done todos to ``ready`` on the next
|
||
dispatcher tick, which keeps the normal parent-gating behaviour intact
|
||
for specified tasks that happen to have open parents.
|
||
|
||
``author`` is recorded on an audit comment only when at least one of
|
||
``title`` / ``body`` / ``assignee`` actually changed — avoids noisy
|
||
comment spam for status-only promotions.
|
||
"""
|
||
if title is not None and not title.strip():
|
||
raise ValueError("title cannot be blank")
|
||
assignee = _canonical_assignee(assignee)
|
||
with write_txn(conn):
|
||
existing = conn.execute(
|
||
"SELECT title, body, assignee FROM tasks WHERE id = ? AND status = 'triage'",
|
||
(task_id,),
|
||
).fetchone()
|
||
if existing is None:
|
||
return False
|
||
sets: list[str] = ["status = 'todo'"]
|
||
params: list[Any] = []
|
||
changed_fields: list[str] = []
|
||
if title is not None and title.strip() != (existing["title"] or ""):
|
||
sets.append("title = ?")
|
||
params.append(title.strip())
|
||
changed_fields.append("title")
|
||
if body is not None and (body or "") != (existing["body"] or ""):
|
||
sets.append("body = ?")
|
||
params.append(body)
|
||
changed_fields.append("body")
|
||
if assignee is not None and assignee != (existing["assignee"] or None):
|
||
sets.append("assignee = ?")
|
||
params.append(assignee)
|
||
changed_fields.append("assignee")
|
||
params.append(task_id)
|
||
cur = conn.execute(
|
||
f"UPDATE tasks SET {', '.join(sets)} "
|
||
f"WHERE id = ? AND status = 'triage'",
|
||
tuple(params),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
if changed_fields and author and author.strip():
|
||
# Inline INSERT (rather than ``add_comment``) because we're
|
||
# already inside this function's write_txn — nested BEGIN
|
||
# IMMEDIATE would raise OperationalError. We also skip the
|
||
# 'commented' event that ``add_comment`` emits, since the
|
||
# 'specified' event below already records the change.
|
||
_insert_comment(
|
||
conn, task_id, author.strip(),
|
||
"Specified — updated " + ", ".join(changed_fields) + " and promoted to todo.",
|
||
int(time.time()),
|
||
)
|
||
_append_event(
|
||
conn,
|
||
task_id,
|
||
"specified",
|
||
{"changed_fields": changed_fields} if changed_fields else None,
|
||
)
|
||
# Outside the write_txn above, so we don't nest BEGIN IMMEDIATE — the
|
||
# ready-promotion pass opens its own IMMEDIATE txn. This runs the same
|
||
# logic the dispatcher would on its next tick, so a specified task
|
||
# with no open parents flips straight to 'ready' here instead of
|
||
# idling in 'todo' until the next sweep.
|
||
recompute_ready(conn)
|
||
return True
|
||
|
||
|
||
def _validate_children_graph(children: list) -> None:
|
||
"""Shape-check ``decompose_triage_task`` children and reject cycles.
|
||
|
||
Cheap, DB-free, so bad input aborts before the txn. The sibling parent
|
||
graph is checked whole (Kahn's sort) rather than edge-by-edge like
|
||
``link_tasks``/``_would_cycle``: a cycle silently deadlocks every involved
|
||
child in ``todo`` because ``recompute_ready`` can never promote them.
|
||
"""
|
||
for idx, child in enumerate(children):
|
||
if not isinstance(child, dict):
|
||
raise ValueError(f"child[{idx}] is not a dict")
|
||
title = child.get("title")
|
||
if not isinstance(title, str) or not title.strip():
|
||
raise ValueError(f"child[{idx}].title is required")
|
||
parents_idx = child.get("parents") or []
|
||
if not isinstance(parents_idx, list):
|
||
raise ValueError(f"child[{idx}].parents must be a list")
|
||
for p in parents_idx:
|
||
if not isinstance(p, int) or p < 0 or p >= len(children):
|
||
raise ValueError(
|
||
f"child[{idx}].parents[{p}] is not a valid index into children"
|
||
)
|
||
if p == idx:
|
||
raise ValueError(f"child[{idx}] cannot list itself as a parent")
|
||
|
||
_in_deg = [0] * len(children)
|
||
_adj: list[list[int]] = [[] for _ in range(len(children))]
|
||
for _i, _c in enumerate(children):
|
||
for _p in (_c.get("parents") or []):
|
||
_adj[_p].append(_i)
|
||
_in_deg[_i] += 1
|
||
_queue = [_i for _i in range(len(children)) if _in_deg[_i] == 0]
|
||
_seen = 0
|
||
while _queue:
|
||
_node = _queue.pop()
|
||
_seen += 1
|
||
for _nb in _adj[_node]:
|
||
_in_deg[_nb] -= 1
|
||
if _in_deg[_nb] == 0:
|
||
_queue.append(_nb)
|
||
if _seen != len(children):
|
||
raise ValueError("cyclic dependency detected in decomposed children list")
|
||
|
||
|
||
def decompose_triage_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
root_assignee: Optional[str],
|
||
children: list[dict],
|
||
author: Optional[str] = None,
|
||
auto_promote: bool = True,
|
||
) -> Optional[list[str]]:
|
||
"""Fan a triage task out into child tasks and promote the root to ``todo``.
|
||
|
||
The root task stays alive and becomes the parent of every child —
|
||
when all children reach ``done``, the root promotes to ``ready`` and
|
||
its assignee (typically the orchestrator profile) wakes back up to
|
||
judge completion or spawn more work.
|
||
|
||
``children``: dicts of ``title`` (required), ``body``, ``assignee`` (None
|
||
-> default fallback) and ``parents`` (indices into this same list).
|
||
Returns the created child ids in input order, or ``None`` when the root
|
||
is missing / not in ``triage`` / the graph has a cycle. Title/assignee
|
||
validation runs inside the same write_txn as the inserts so a malformed
|
||
entry aborts the whole decomposition (no orphan children).
|
||
"""
|
||
if not children:
|
||
return None
|
||
if root_assignee is not None:
|
||
root_assignee = _canonical_assignee(root_assignee)
|
||
|
||
_validate_children_graph(children)
|
||
|
||
# We do the full decomposition in a SINGLE write_txn so it's
|
||
# atomic: either every child is created AND the root flips to
|
||
# ``todo``, or nothing changes. We deliberately do NOT call any
|
||
# kb helper that opens its own write_txn (create_task, link_tasks,
|
||
# add_comment) from inside this block — see architecture.md
|
||
# write_txn pitfalls. Instead we inline the INSERTs and
|
||
# _append_event calls.
|
||
now = int(time.time())
|
||
child_ids: list[str] = []
|
||
with write_txn(conn):
|
||
root_row = conn.execute(
|
||
"SELECT id, status, tenant, workspace_kind, workspace_path "
|
||
"FROM tasks WHERE id = ?",
|
||
(task_id,),
|
||
).fetchone()
|
||
if root_row is None:
|
||
return None
|
||
if root_row["status"] != "triage":
|
||
return None
|
||
tenant = root_row["tenant"]
|
||
# Children inherit the root's workspace by default so a fan-out
|
||
# of a code-gen task lands in the parent's project dir/worktree
|
||
# rather than throwaway scratch tmp dirs. A child dict can still
|
||
# override with its own 'workspace_kind' / 'workspace_path'.
|
||
root_ws_kind = root_row["workspace_kind"] or "scratch"
|
||
root_ws_path = root_row["workspace_path"]
|
||
|
||
# Create children. Status is 'todo' regardless of parents — we
|
||
# link them under the root AFTER creation so the dispatcher
|
||
# sees a coherent state, and recompute_ready() at the end
|
||
# promotes parent-free children to 'ready'.
|
||
for idx, child in enumerate(children):
|
||
new_id = _new_task_id()
|
||
title = child["title"].strip()
|
||
body = child.get("body")
|
||
assignee = _canonical_assignee(child.get("assignee"))
|
||
# Per-child override wins; otherwise inherit the root's
|
||
# workspace. A child that sets workspace_kind without a path
|
||
# falls back to the root path only when kinds match (so a
|
||
# child can't accidentally point a 'dir' at the root's
|
||
# worktree path or vice versa).
|
||
child_ws_kind = child.get("workspace_kind") or root_ws_kind
|
||
if child.get("workspace_path"):
|
||
child_ws_path = child.get("workspace_path")
|
||
elif child_ws_kind == "worktree":
|
||
# Never share one worktree checkout between siblings: the
|
||
# root's literal path would put every child in the same
|
||
# directory on the first-dispatched sibling's branch, with
|
||
# no lock — siblings can be promoted and dispatched
|
||
# concurrently. Leave the path unset so dispatch
|
||
# materializes a fresh <repo>/.worktrees/<child-id> per
|
||
# child from the board anchor.
|
||
child_ws_path = None
|
||
elif child_ws_kind == root_ws_kind:
|
||
child_ws_path = root_ws_path
|
||
else:
|
||
child_ws_path = None
|
||
conn.execute(
|
||
"INSERT INTO tasks "
|
||
"(id, title, body, assignee, status, workspace_kind, "
|
||
" workspace_path, tenant, created_at, created_by) "
|
||
"VALUES (?, ?, ?, ?, 'todo', ?, ?, ?, ?, ?)",
|
||
(
|
||
new_id,
|
||
title,
|
||
body if isinstance(body, str) else None,
|
||
assignee,
|
||
child_ws_kind,
|
||
child_ws_path,
|
||
tenant,
|
||
now,
|
||
(author or "decomposer"),
|
||
),
|
||
)
|
||
_append_event(
|
||
conn, new_id, "created",
|
||
{"by": author or "decomposer", "from_decompose_of": task_id},
|
||
)
|
||
_inherit_notify_subs(conn, new_id, (task_id,), created_at=now)
|
||
child_ids.append(new_id)
|
||
|
||
# Link children to their sibling parents (within the decomposed graph).
|
||
for idx, child in enumerate(children):
|
||
for p_idx in child.get("parents") or []:
|
||
parent_id = child_ids[p_idx]
|
||
child_id = child_ids[idx]
|
||
_link(conn, parent_id, child_id)
|
||
_append_event(
|
||
conn, child_id, "linked",
|
||
{"parent": parent_id, "child": child_id},
|
||
)
|
||
|
||
# Link the ROOT task as a child of every leaf child — i.e. the
|
||
# root waits for the whole graph. Simpler than computing leaves:
|
||
# link root under every child. Cycle-free because the root is
|
||
# only ever a child here, never a parent of children.
|
||
for cid in child_ids:
|
||
_link(conn, cid, task_id)
|
||
|
||
# Flip the root: triage -> todo, set assignee to the orchestrator.
|
||
sets = ["status = 'todo'"]
|
||
params: list[Any] = []
|
||
if root_assignee is not None:
|
||
sets.append("assignee = ?")
|
||
params.append(root_assignee)
|
||
params.append(task_id)
|
||
conn.execute(
|
||
f"UPDATE tasks SET {', '.join(sets)} WHERE id = ?",
|
||
tuple(params),
|
||
)
|
||
|
||
# Audit comment + event on the root so the timeline shows the fan-out.
|
||
if author and author.strip():
|
||
_insert_comment(
|
||
conn, task_id, author.strip(),
|
||
"Decomposed into " + ", ".join(child_ids)
|
||
+ ". Root will wake when all children complete.",
|
||
now,
|
||
)
|
||
_append_event(
|
||
conn, task_id, "decomposed",
|
||
{
|
||
"child_ids": child_ids,
|
||
"root_assignee": root_assignee,
|
||
},
|
||
)
|
||
|
||
# Outside the write_txn: promote parent-free children to 'ready'
|
||
# so the dispatcher picks them up on its next tick. Same pattern
|
||
# specify_triage_task uses. When auto_promote is False children
|
||
# stay in 'todo' until the user manually promotes them — useful
|
||
# for manual-review-first workflows.
|
||
if auto_promote:
|
||
recompute_ready(conn)
|
||
return child_ids
|
||
|
||
|
||
def archive_task(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
with write_txn(conn):
|
||
cur = conn.execute(
|
||
"UPDATE tasks SET status = 'archived', "
|
||
" claim_lock = NULL, claim_expires = NULL, worker_pid = NULL "
|
||
"WHERE id = ? AND status != 'archived'",
|
||
(task_id,),
|
||
)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
# If archive happened while a run was still in flight (e.g. user
|
||
# archived a running task from the dashboard), close that run with
|
||
# outcome='reclaimed' so attempt history isn't orphaned.
|
||
run_id = _end_run(
|
||
conn, task_id,
|
||
outcome="reclaimed", status="reclaimed",
|
||
summary="task archived with run still active",
|
||
)
|
||
_append_event(conn, task_id, "archived", None, run_id=run_id)
|
||
# ``archived`` parents no longer block children, same as ``done``.
|
||
# Promote newly-unblocked dependents immediately instead of waiting
|
||
# for a later dispatcher tick.
|
||
recompute_ready(conn)
|
||
# Reap the workspace on archive too — tasks archived without ever
|
||
# completing previously kept their scratch dir / worktree forever.
|
||
_cleanup_workspace(conn, task_id)
|
||
return True
|
||
|
||
|
||
def _delete_task_relations(conn: sqlite3.Connection, task_id: str) -> None:
|
||
"""Delete every row referencing ``task_id`` (schema has no ON DELETE CASCADE)."""
|
||
conn.execute(
|
||
"DELETE FROM task_links WHERE parent_id = ? OR child_id = ?", (task_id, task_id),
|
||
)
|
||
for table in ("task_comments", "task_events", "task_runs", "kanban_notify_subs"):
|
||
conn.execute(f"DELETE FROM {table} WHERE task_id = ?", (task_id,))
|
||
|
||
|
||
def delete_archived_task(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Permanently remove an already-archived task and its related rows.
|
||
|
||
Safety guard: only archived tasks can be deleted. Active / blocked / done
|
||
tasks must be explicitly archived first so accidental data loss requires a
|
||
second deliberate action.
|
||
"""
|
||
with write_txn(conn):
|
||
if _task_status(conn, task_id) != "archived":
|
||
return False
|
||
_delete_task_relations(conn, task_id)
|
||
cur = conn.execute("DELETE FROM tasks WHERE id = ?", (task_id,))
|
||
return cur.rowcount == 1
|
||
|
||
|
||
def delete_task(conn: sqlite3.Connection, task_id: str) -> bool:
|
||
"""Hard-delete a task and cascade to all related rows.
|
||
|
||
Because the schema does not use ``ON DELETE CASCADE`` foreign keys,
|
||
we explicitly delete from child tables first, then the task row.
|
||
This keeps the operation atomic (single ``write_txn``).
|
||
|
||
Returns ``True`` if the task existed and was deleted, ``False``
|
||
if the task was not found.
|
||
"""
|
||
with write_txn(conn):
|
||
cur = conn.execute("DELETE FROM tasks WHERE id = ?", (task_id,))
|
||
if cur.rowcount != 1:
|
||
return False
|
||
_delete_task_relations(conn, task_id)
|
||
recompute_ready(conn)
|
||
return True
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
def schedule_task(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
reason: Optional[str] = None,
|
||
expected_run_id: Optional[int] = None,
|
||
) -> bool:
|
||
"""Park a task in ``scheduled`` so it is waiting on time, not human input.
|
||
|
||
``scheduled`` tasks are intentionally not dispatchable; an external cron,
|
||
human action, or automation can later call ``unblock_task`` to re-gate them
|
||
to ``ready`` (or ``todo`` if parents are still incomplete).
|
||
"""
|
||
with write_txn(conn):
|
||
params: list[Any] = [task_id]
|
||
sql = """
|
||
UPDATE tasks
|
||
SET status = 'scheduled',
|
||
claim_lock = NULL,
|
||
claim_expires= NULL,
|
||
worker_pid = NULL
|
||
WHERE id = ?
|
||
AND status IN ('todo', 'ready', 'running', 'blocked')
|
||
"""
|
||
if expected_run_id is not None:
|
||
sql += " AND current_run_id = ?"
|
||
params.append(int(expected_run_id))
|
||
cur = conn.execute(sql, params)
|
||
if cur.rowcount != 1:
|
||
return False
|
||
run_id = _end_run(
|
||
conn, task_id,
|
||
outcome="scheduled", status="scheduled",
|
||
summary=reason,
|
||
)
|
||
if run_id is None and reason:
|
||
run_id = _synthesize_ended_run(
|
||
conn, task_id,
|
||
outcome="scheduled",
|
||
summary=reason,
|
||
)
|
||
_append_event(conn, task_id, "scheduled", {"reason": reason}, run_id=run_id)
|
||
return True
|
||
|
||
|
||
# Dispatcher (one-shot pass)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Worker context builder (what a spawned worker sees)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def build_worker_context(conn: sqlite3.Connection, task_id: str) -> str:
|
||
"""Return the full text a worker should read to understand its task.
|
||
|
||
Sections in order: header, body, attachments, prior attempts on this task,
|
||
done-parent handoffs (``run.summary``/``metadata``, falling back to
|
||
``task.result`` for pre-runs data), the assignee's recent completed runs
|
||
on other tasks, comment thread. Every list is tail-capped (``_CTX_MAX_*``)
|
||
with the omitted head summarised, and every field is char-capped, so the
|
||
prompt stays bounded on pathological boards (retry storms, comment
|
||
storms, a single 1 MB summary).
|
||
"""
|
||
task = get_task(conn, task_id)
|
||
if not task:
|
||
raise ValueError(f"unknown task {task_id}")
|
||
|
||
# Single clock reading shared by every relative-age stamp below, so all
|
||
# ages in one rendering are consistent ("3h ago" / "3h ago", not drifting
|
||
# by the seconds it takes to build the block).
|
||
_now = int(time.time())
|
||
|
||
def _cap(s: Optional[str], limit: int = _CTX_MAX_FIELD_BYTES) -> str:
|
||
"""Truncate a string to `limit` chars with a visible ellipsis."""
|
||
if not s:
|
||
return ""
|
||
s = s.strip()
|
||
if len(s) <= limit:
|
||
return s
|
||
return s[:limit] + f"… [truncated, {len(s) - limit} chars omitted]"
|
||
|
||
def _stamp(ts: int) -> str:
|
||
"""``YYYY-MM-DD HH:MM`` plus a relative age when one is available."""
|
||
disp = time.strftime("%Y-%m-%d %H:%M", time.localtime(ts))
|
||
age = _relative_age(ts, _now)
|
||
return f"{disp}, {age}" if age else disp
|
||
|
||
def _metadata_line(metadata: Any) -> Optional[str]:
|
||
if not metadata:
|
||
return None
|
||
try:
|
||
return f"_metadata_: `{_cap(json.dumps(metadata, ensure_ascii=False, sort_keys=True))}`"
|
||
except Exception:
|
||
return None
|
||
|
||
def _tail(items: list, cap: int, noun: str) -> tuple[list, Optional[str]]:
|
||
"""Keep the newest ``cap`` items; describe the omitted head, if any."""
|
||
omitted = max(0, len(items) - cap)
|
||
if not omitted:
|
||
return items, None
|
||
return items[-cap:], (
|
||
f"_({omitted} earlier {noun}{'s' if omitted != 1 else ''} "
|
||
f"omitted; showing most recent {cap})_"
|
||
)
|
||
|
||
lines: list[str] = []
|
||
lines.append(f"# Kanban task {task.id}: {task.title}")
|
||
lines.append("")
|
||
lines.append(f"Assignee: {task.assignee or '(unassigned)'}")
|
||
lines.append(f"Status: {task.status}")
|
||
if task.tenant:
|
||
lines.append(f"Tenant: {task.tenant}")
|
||
lines.append(f"Workspace: {task.workspace_kind} @ {task.workspace_path or '(unresolved)'}")
|
||
if task.max_runtime_seconds is not None:
|
||
terminal_timeout = _worker_terminal_timeout_env(
|
||
task.max_runtime_seconds,
|
||
os.environ.get("TERMINAL_TIMEOUT"),
|
||
)
|
||
effective_terminal_timeout = terminal_timeout or os.environ.get("TERMINAL_TIMEOUT")
|
||
lines.append(f"Max runtime: {task.max_runtime_seconds}s")
|
||
if effective_terminal_timeout:
|
||
lines.append(f"Terminal timeout: {effective_terminal_timeout}s")
|
||
if task.branch_name:
|
||
lines.append(f"Branch: {task.branch_name}")
|
||
lines.append("")
|
||
|
||
if task.body and task.body.strip():
|
||
lines.append("## Body")
|
||
lines.append(_cap(task.body, _CTX_MAX_BODY_BYTES))
|
||
lines.append("")
|
||
|
||
# Attachments — files uploaded to this task (PDFs, source docs,
|
||
# images). Surface the absolute on-disk path so the worker, which has
|
||
# full file-tool access, can read them directly (read_file, terminal
|
||
# `pdftotext`, etc.). On the local terminal backend the path resolves
|
||
# as-is; remote backends need the kanban attachments dir mounted.
|
||
attachments = list_attachments(conn, task_id)
|
||
if attachments:
|
||
lines.append("## Attachments")
|
||
lines.append(
|
||
"Files attached to this task. Read them with the file/terminal "
|
||
"tools at the absolute paths below:"
|
||
)
|
||
for att in attachments:
|
||
size_kb = max(1, (att.size + 1023) // 1024) if att.size else 0
|
||
size_str = f", {size_kb} KB" if size_kb else ""
|
||
ctype = f", {att.content_type}" if att.content_type else ""
|
||
lines.append(f"- `{att.filename}`{ctype}{size_str} → `{att.stored_path}`")
|
||
lines.append("")
|
||
|
||
# Prior attempts — show closed runs so a retrying worker sees the
|
||
# history. Skip the currently-active run (that's this worker).
|
||
# Cap at _CTX_MAX_PRIOR_ATTEMPTS most-recent closed runs; older
|
||
# attempts get collapsed into a one-line marker so the worker knows
|
||
# more exist without bloating the prompt.
|
||
all_prior = [r for r in list_runs(conn, task_id) if r.ended_at is not None]
|
||
# list_runs returns ascending by started_at; "most recent" = last N
|
||
shown, omitted_note = _tail(all_prior, _CTX_MAX_PRIOR_ATTEMPTS, "attempt")
|
||
first_shown_idx = len(all_prior) - len(shown) + 1
|
||
if shown:
|
||
lines.append("## Prior attempts on this task")
|
||
if omitted_note:
|
||
lines.append(omitted_note)
|
||
for offset, run in enumerate(shown):
|
||
idx = first_shown_idx + offset
|
||
profile = run.profile or "(unknown)"
|
||
outcome = run.outcome or run.status
|
||
lines.append(f"### Attempt {idx} — {outcome} ({profile}, {_stamp(run.started_at)})")
|
||
if run.summary and run.summary.strip():
|
||
lines.append(_cap(run.summary))
|
||
if run.error and run.error.strip():
|
||
lines.append(f"_error_: {_cap(run.error)}")
|
||
meta_line = _metadata_line(run.metadata)
|
||
if meta_line:
|
||
lines.append(meta_line)
|
||
lines.append("")
|
||
|
||
# Parents: prefer the most-recent 'completed' run's summary + metadata,
|
||
# fall back to ``task.result`` when no run rows exist (legacy DBs,
|
||
# or tasks completed before the runs table landed).
|
||
parent_rows = conn.execute(
|
||
"SELECT parent_id FROM task_links WHERE child_id = ? ORDER BY parent_id",
|
||
(task_id,),
|
||
).fetchall()
|
||
parent_ids = [r["parent_id"] for r in parent_rows]
|
||
|
||
if parent_ids:
|
||
wrote_header = False
|
||
for pid in parent_ids:
|
||
pt = get_task(conn, pid)
|
||
if not pt or pt.status != "done":
|
||
continue
|
||
runs = [r for r in list_runs(conn, pid) if r.outcome == "completed"]
|
||
runs.sort(key=lambda r: r.started_at, reverse=True)
|
||
run = runs[0] if runs else None
|
||
|
||
if not wrote_header:
|
||
lines.append("## Parent task results")
|
||
lines.append(
|
||
"_Handoffs from upstream tasks, captured when each parent "
|
||
"completed (see age below). These are point-in-time "
|
||
"snapshots, not live state — if a result drives your "
|
||
"current work and it's not recent, re-verify against the "
|
||
"source before acting on it as current._"
|
||
)
|
||
wrote_header = True
|
||
|
||
# When did this parent's result get produced? Prefer the
|
||
# completed run's end time; fall back to the task's completed_at.
|
||
done_ts = None
|
||
if run is not None and getattr(run, "ended_at", None):
|
||
done_ts = run.ended_at
|
||
elif pt.completed_at:
|
||
done_ts = pt.completed_at
|
||
age = _relative_age(done_ts, _now)
|
||
lines.append(f"### {pid}" + (f" (completed {age})" if age else ""))
|
||
|
||
body_lines: list[str] = []
|
||
if run is not None and run.summary and run.summary.strip():
|
||
body_lines.append(_cap(run.summary))
|
||
elif pt.result:
|
||
body_lines.append(_cap(pt.result))
|
||
else:
|
||
body_lines.append("(no result recorded)")
|
||
|
||
meta_line = _metadata_line(run.metadata) if run is not None else None
|
||
if meta_line:
|
||
body_lines.append(meta_line)
|
||
lines.extend(body_lines)
|
||
lines.append("")
|
||
|
||
# Cross-task role history: what else has THIS assignee completed
|
||
# recently? Gives the worker implicit continuity — "I'm the reviewer
|
||
# and my last three reviews focused on security" — without forcing
|
||
# the user to wire anything into SOUL.md / MEMORY.md. Bounded to the
|
||
# most recent 5 completed runs, excluding this task so the retry
|
||
# section above isn't duplicated. Safe on assignee=None (skipped).
|
||
if task.assignee:
|
||
role_rows = conn.execute(
|
||
"SELECT t.id, t.title, r.summary, r.ended_at "
|
||
"FROM task_runs r JOIN tasks t ON r.task_id = t.id "
|
||
"WHERE r.profile = ? AND r.task_id != ? "
|
||
" AND r.outcome = 'completed' "
|
||
"ORDER BY r.ended_at DESC LIMIT 5",
|
||
(task.assignee, task_id),
|
||
).fetchall()
|
||
if role_rows:
|
||
lines.append(f"## Recent work by @{task.assignee}")
|
||
for row in role_rows:
|
||
s = (row["summary"] or "").strip().splitlines()
|
||
first = s[0][:200] if s else "(no summary)"
|
||
lines.append(
|
||
f"- {row['id']} — {row['title']} ({_stamp(int(row['ended_at']))}): {first}"
|
||
)
|
||
lines.append("")
|
||
|
||
# Comments: cap at the most-recent _CTX_MAX_COMMENTS so
|
||
# comment-storm tasks don't blow out the worker's prompt. Older
|
||
# comments summarised in a one-line marker like prior attempts.
|
||
shown_c, omitted_note = _tail(list_comments(conn, task_id), _CTX_MAX_COMMENTS, "comment")
|
||
if shown_c:
|
||
lines.append("## Comment thread")
|
||
if omitted_note:
|
||
lines.append(omitted_note)
|
||
for c in shown_c:
|
||
# Explicit "comment from worker" framing so operator-controlled
|
||
# HERMES_PROFILE values like "hermes-system" or "operator" can't be
|
||
# misread by the next worker as a system directive above the
|
||
# (attacker-influenceable) comment body. Defense-in-depth on top of
|
||
# the closed LLM-controlled author-forgery surface.
|
||
safe_author = (c.author or "").replace("`", "")
|
||
lines.append(f"comment from worker `{safe_author}` at {_stamp(c.created_at)}:")
|
||
lines.append(_cap(c.body, _CTX_MAX_COMMENT_BYTES))
|
||
lines.append("")
|
||
|
||
return "\n".join(lines).rstrip() + "\n"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Stats + SLA helpers
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def board_stats(conn: sqlite3.Connection) -> dict:
|
||
"""Per-status + per-assignee counts, plus the oldest ``ready`` age in
|
||
seconds (the clearest staleness signal for a router or HUD).
|
||
"""
|
||
by_status: dict[str, int] = {}
|
||
for row in conn.execute(
|
||
"SELECT status, COUNT(*) AS n FROM tasks "
|
||
"WHERE status != 'archived' GROUP BY status"
|
||
):
|
||
by_status[row["status"]] = int(row["n"])
|
||
|
||
by_assignee: dict[str, dict[str, int]] = {}
|
||
for row in conn.execute(
|
||
"SELECT assignee, status, COUNT(*) AS n FROM tasks "
|
||
"WHERE status != 'archived' AND assignee IS NOT NULL "
|
||
"GROUP BY assignee, status"
|
||
):
|
||
by_assignee.setdefault(row["assignee"], {})[row["status"]] = int(row["n"])
|
||
|
||
oldest_row = conn.execute(
|
||
"SELECT MIN(created_at) AS ts FROM tasks WHERE status = 'ready'"
|
||
).fetchone()
|
||
now = int(time.time())
|
||
oldest_ready_age = (
|
||
(now - int(oldest_row["ts"]))
|
||
if oldest_row and oldest_row["ts"] is not None else None
|
||
)
|
||
|
||
return {
|
||
"by_status": by_status,
|
||
"by_assignee": by_assignee,
|
||
"oldest_ready_age_seconds": oldest_ready_age,
|
||
"now": now,
|
||
}
|
||
|
||
|
||
def _to_epoch(val) -> Optional[int]:
|
||
"""Normalise a timestamp to unix epoch seconds.
|
||
|
||
Accepts ints (pass-through), numeric strings, and ISO-8601 strings.
|
||
Returns ``None`` for ``None`` / empty values.
|
||
"""
|
||
if val is None:
|
||
return None
|
||
if isinstance(val, int):
|
||
return val
|
||
if isinstance(val, float):
|
||
return int(val)
|
||
s = str(val).strip()
|
||
if not s:
|
||
return None
|
||
try:
|
||
return int(s)
|
||
except ValueError:
|
||
pass
|
||
# ISO-8601 fallback (e.g. '2026-05-10T15:00:00Z')
|
||
try:
|
||
from datetime import datetime
|
||
dt = datetime.fromisoformat(s.replace("Z", "+00:00"))
|
||
return int(dt.timestamp())
|
||
except (ValueError, OSError):
|
||
return None
|
||
|
||
|
||
def task_age(task: Task) -> dict:
|
||
"""Return age metrics for a single task. All values are seconds or None."""
|
||
now = int(time.time())
|
||
_c = _to_epoch(task.created_at)
|
||
_s = _to_epoch(task.started_at)
|
||
_co = _to_epoch(task.completed_at)
|
||
age_since_created = now - _c if _c is not None else None
|
||
age_since_started = now - _s if _s is not None else None
|
||
time_to_complete = (
|
||
_co - (_s or _c) if _co is not None else None
|
||
)
|
||
return {
|
||
"created_age_seconds": age_since_created,
|
||
"started_age_seconds": age_since_started,
|
||
"time_to_complete_seconds": time_to_complete,
|
||
}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Retention + garbage collection
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def gc_events(
|
||
conn: sqlite3.Connection, *, older_than_seconds: int = 30 * 24 * 3600,
|
||
) -> int:
|
||
"""Delete task_events rows older than ``older_than_seconds`` for tasks
|
||
in a terminal state (``done`` or ``archived``). Returns the number of
|
||
rows deleted. Running / ready / blocked tasks keep their full event
|
||
history."""
|
||
cutoff = int(time.time()) - int(older_than_seconds)
|
||
with write_txn(conn):
|
||
cur = conn.execute(
|
||
"DELETE FROM task_events WHERE created_at < ? AND task_id IN "
|
||
"(SELECT id FROM tasks WHERE status IN ('done', 'archived'))",
|
||
(cutoff,),
|
||
)
|
||
return int(cur.rowcount or 0)
|
||
|
||
|
||
def gc_worker_logs(
|
||
*, older_than_seconds: int = 30 * 24 * 3600,
|
||
board: Optional[str] = None,
|
||
) -> int:
|
||
"""Delete worker log files older than ``older_than_seconds``. Returns
|
||
the number of files removed. Kept separate from ``gc_events`` because
|
||
log files live on disk, not in SQLite. Scoped to ``board`` (defaults
|
||
to the active board) — per-board isolation means deleting logs from
|
||
board A cannot touch board B's logs."""
|
||
log_dir = worker_logs_dir(board=board)
|
||
if not log_dir.exists():
|
||
return 0
|
||
cutoff = time.time() - older_than_seconds
|
||
removed = 0
|
||
for p in log_dir.iterdir():
|
||
try:
|
||
if p.is_file() and p.stat().st_mtime < cutoff:
|
||
p.unlink()
|
||
removed += 1
|
||
except OSError:
|
||
continue
|
||
return removed
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Worker log accessor
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def worker_log_path(task_id: str, *, board: Optional[str] = None) -> Path:
|
||
"""Return the path to a worker's log file. The file may not exist
|
||
(task never spawned, or log already GC'd).
|
||
|
||
When ``board`` is None, resolves via the active board (env var →
|
||
current-board file → default). The dispatcher always passes the
|
||
board explicitly to avoid any resolution ambiguity when multiple
|
||
boards exist."""
|
||
return worker_logs_dir(board=board) / f"{task_id}.log"
|
||
|
||
|
||
def read_worker_log(
|
||
task_id: str, *, tail_bytes: Optional[int] = None,
|
||
board: Optional[str] = None,
|
||
) -> Optional[str]:
|
||
"""Read the worker log for ``task_id``. Returns None if the file
|
||
doesn't exist. If ``tail_bytes`` is set, only the last N bytes are
|
||
returned (useful for the dashboard drawer which shouldn't page megabytes)."""
|
||
path = worker_log_path(task_id, board=board)
|
||
if not path.exists():
|
||
return None
|
||
try:
|
||
if tail_bytes is None:
|
||
return path.read_text(encoding="utf-8", errors="replace")
|
||
size = path.stat().st_size
|
||
with open(path, "rb") as f:
|
||
if size > tail_bytes:
|
||
f.seek(size - tail_bytes)
|
||
# Skip a partial line if we tailed mid-line. But if the
|
||
# window has no newline at all (one giant log line),
|
||
# readline() would eat everything — in that case don't
|
||
# skip and return the raw tail.
|
||
probe = f.tell()
|
||
partial = f.readline()
|
||
if not partial.endswith(b"\n") and f.tell() >= size:
|
||
f.seek(probe)
|
||
data = f.read()
|
||
return data.decode("utf-8", errors="replace")
|
||
except OSError:
|
||
return None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Assignee enumeration (known profiles + per-profile board stats)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def list_profiles_on_disk() -> list[str]:
|
||
"""Profile names on disk: ``<default-root>/profiles/<name>/config.yaml``
|
||
plus the implicit ``default`` when the root exists. Reads paths directly
|
||
to avoid importing ``hermes_cli.profiles`` (a large chunk of CLI startup).
|
||
"""
|
||
try:
|
||
from hermes_constants import get_default_hermes_root
|
||
default_root = get_default_hermes_root()
|
||
profiles_dir = default_root / "profiles"
|
||
except Exception:
|
||
return []
|
||
|
||
names: set[str] = set()
|
||
if default_root.exists():
|
||
names.add("default")
|
||
|
||
if profiles_dir.is_dir():
|
||
try:
|
||
for entry in sorted(profiles_dir.iterdir()):
|
||
if not entry.is_dir():
|
||
continue
|
||
if (entry / "config.yaml").is_file():
|
||
names.add(entry.name)
|
||
except OSError:
|
||
pass
|
||
|
||
return sorted(names)
|
||
|
||
|
||
def known_assignees(conn: sqlite3.Connection) -> list[dict]:
|
||
"""Every assignee that is a profile on disk OR assigned to a non-archived
|
||
task, as ``{"name", "on_disk", "counts": {status: n}}`` — so a fresh
|
||
profile appears in pickers before it has any task.
|
||
"""
|
||
on_disk = set(list_profiles_on_disk())
|
||
|
||
# Count tasks per (assignee, status), excluding archived.
|
||
counts: dict[str, dict[str, int]] = {}
|
||
for row in conn.execute(
|
||
"SELECT assignee, status, COUNT(*) AS n FROM tasks "
|
||
"WHERE status != 'archived' AND assignee IS NOT NULL "
|
||
"GROUP BY assignee, status"
|
||
):
|
||
counts.setdefault(row["assignee"], {})[row["status"]] = int(row["n"])
|
||
|
||
names = sorted(on_disk | set(counts.keys()))
|
||
return [
|
||
{
|
||
"name": name,
|
||
"on_disk": name in on_disk,
|
||
"counts": counts.get(name, {}),
|
||
}
|
||
for name in names
|
||
]
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Runs (attempt history on a task)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
def list_runs(
|
||
conn: sqlite3.Connection,
|
||
task_id: str,
|
||
*,
|
||
include_active: bool = True,
|
||
state_type: Optional[str] = None,
|
||
state_name: Optional[str] = None,
|
||
) -> list[Run]:
|
||
"""All runs for ``task_id`` in start order. ``include_active=False`` returns
|
||
only closed runs; ``state_type`` (``status``/``outcome``) + ``state_name``
|
||
filter on that column and must be passed together.
|
||
"""
|
||
if (state_type is None) ^ (state_name is None):
|
||
raise ValueError("state_type and state_name must both be set or both omitted")
|
||
if state_type is not None and state_type not in ("status", "outcome"):
|
||
raise ValueError("state_type must be 'status' or 'outcome'")
|
||
q = "SELECT * FROM task_runs WHERE task_id = ?"
|
||
params: list[Any] = [task_id]
|
||
if not include_active:
|
||
q += " AND ended_at IS NOT NULL"
|
||
if state_type is not None:
|
||
q += f" AND {state_type} = ?"
|
||
params.append(state_name)
|
||
q += " ORDER BY started_at ASC, id ASC"
|
||
rows = conn.execute(q, params).fetchall()
|
||
return [Run.from_row(r) for r in rows]
|
||
|
||
|
||
def get_run(conn: sqlite3.Connection, run_id: int) -> Optional[Run]:
|
||
row = conn.execute(
|
||
"SELECT * FROM task_runs WHERE id = ?", (int(run_id),),
|
||
).fetchone()
|
||
return Run.from_row(row) if row else None
|
||
|
||
|
||
def latest_run(conn: sqlite3.Connection, task_id: str) -> Optional[Run]:
|
||
"""Return the most recent run regardless of outcome (active or closed)."""
|
||
row = conn.execute(
|
||
"SELECT * FROM task_runs WHERE task_id = ? "
|
||
"ORDER BY started_at DESC, id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
return Run.from_row(row) if row else None
|
||
|
||
|
||
def latest_summary(conn: sqlite3.Connection, task_id: str) -> Optional[str]:
|
||
"""Latest non-null ``task_runs.summary`` (newest ``ended_at``, ``id`` for
|
||
ties), or None. Workers hand off via ``complete_task(summary=...)`` and
|
||
leave ``tasks.result`` NULL, so show/dashboard views need this or a
|
||
completed task looks like a no-op.
|
||
"""
|
||
row = conn.execute(
|
||
"SELECT summary FROM task_runs "
|
||
"WHERE task_id = ? AND summary IS NOT NULL AND summary != '' "
|
||
"ORDER BY COALESCE(ended_at, started_at) DESC, id DESC LIMIT 1",
|
||
(task_id,),
|
||
).fetchone()
|
||
return row["summary"] if row else None
|
||
|
||
|
||
def latest_summaries(
|
||
conn: sqlite3.Connection, task_ids: Iterable[str]
|
||
) -> dict[str, str]:
|
||
"""``{task_id: latest non-null run summary}`` for many tasks in one query
|
||
(dashboard board endpoint; avoids N+1 :func:`latest_summary` calls).
|
||
Window function -> needs SQLite >= 3.25 (default on every supported
|
||
platform). Tasks without a summary are omitted.
|
||
"""
|
||
ids = list(task_ids)
|
||
if not ids:
|
||
return {}
|
||
placeholders = ",".join("?" for _ in ids)
|
||
rows = conn.execute(
|
||
f"""
|
||
SELECT task_id, summary FROM (
|
||
SELECT task_id, summary,
|
||
ROW_NUMBER() OVER (
|
||
PARTITION BY task_id
|
||
ORDER BY COALESCE(ended_at, started_at) DESC, id DESC
|
||
) AS rn
|
||
FROM task_runs
|
||
WHERE task_id IN ({placeholders})
|
||
AND summary IS NOT NULL AND summary != ''
|
||
) WHERE rn = 1
|
||
""",
|
||
ids,
|
||
).fetchall()
|
||
return {r["task_id"]: r["summary"] for r in rows}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Split modules — re-exported so ``kanban_db.<name>`` keeps resolving (and
|
||
# stays the single monkeypatch target).
|
||
# ---------------------------------------------------------------------------
|
||
from hermes_cli.kanban_db_connect import ( # noqa: E402,F401
|
||
DEFAULT_BUSY_TIMEOUT_MS,
|
||
KanbanDbCorruptError,
|
||
RepairResult,
|
||
_BUSY_MAX_RETRIES,
|
||
_BUSY_RETRY_MAX_S,
|
||
_BUSY_RETRY_MIN_S,
|
||
_CORRUPT_BACKUP_RETENTION,
|
||
_EARLY_TASK_COLUMNS,
|
||
_INITIALIZED_PATHS,
|
||
_INIT_LOCK,
|
||
_INIT_LOCK_POLL_SECONDS,
|
||
_INIT_LOCK_TIMEOUT_SECONDS,
|
||
_LAST_WAL_CHECKPOINT,
|
||
_LATER_TASK_COLUMNS,
|
||
_REBUILD_SPECS,
|
||
_RENAMED_TASK_COLUMNS,
|
||
_REPAIRABLE_INDEX_ERROR_PATTERNS,
|
||
_SQLITE_HEADER,
|
||
_WAL_CHECKPOINT_INTERVAL_SECONDS,
|
||
_WAL_CHECKPOINT_LOCK,
|
||
_attempt_index_reindex_repair,
|
||
_backup_corrupt_db,
|
||
_check_file_length_invariant,
|
||
_cross_process_init_lock,
|
||
_dispatch_tick_lock,
|
||
_execute_boundary_with_retry,
|
||
_guard_existing_db_is_healthy,
|
||
_integrity_messages_ok,
|
||
_is_busy_error,
|
||
_looks_like_tls_record_at,
|
||
_maybe_checkpoint_wal,
|
||
_migrate_add_optional_columns,
|
||
_open_configured,
|
||
_probe_integrity,
|
||
_prune_corrupt_backups,
|
||
_rebuild_drifted_tables,
|
||
_repairable_index_names,
|
||
_resolve_busy_timeout_ms,
|
||
_run_integrity_check,
|
||
_schema_is_present,
|
||
_sqlite_connect,
|
||
_table_has_drifted,
|
||
_try_lock_nb,
|
||
_unlock,
|
||
_validate_sqlite_header,
|
||
connect,
|
||
connect_closing,
|
||
init_db,
|
||
repair_db,
|
||
write_txn,
|
||
)
|
||
from hermes_cli.kanban_db_workspace import ( # noqa: E402,F401
|
||
_SCRATCH_TIP_MESSAGE,
|
||
_SCRATCH_TIP_SENTINEL_NAME,
|
||
_cleanup_worker_tmux,
|
||
_cleanup_workspace,
|
||
_cleanup_worktree_workspace,
|
||
_ensure_git_worktree,
|
||
_git_branch_exists,
|
||
_git_common_dir,
|
||
_git_current_branch,
|
||
_git_dir,
|
||
_git_toplevel,
|
||
_is_linked_worktree_checkout,
|
||
_is_managed_scratch_path,
|
||
_managed_scratch_path_info,
|
||
_mark_scratch_tip_shown,
|
||
_maybe_emit_scratch_tip,
|
||
_nearest_existing_path,
|
||
_repo_root_for_worktree_target,
|
||
_resolve_worktree_workspace,
|
||
_scratch_tip_sentinel_path,
|
||
_scratch_tip_shown,
|
||
_scratch_workspace,
|
||
_try_cleanup_parent_workspaces,
|
||
resolve_workspace,
|
||
set_branch_name,
|
||
set_workspace_path,
|
||
)
|
||
from hermes_cli.kanban_db_dispatch import ( # noqa: E402,F401
|
||
DEFAULT_FAILURE_LIMIT,
|
||
DEFAULT_LOG_BACKUP_COUNT,
|
||
DEFAULT_LOG_ROTATE_BYTES,
|
||
DEFAULT_RATE_LIMIT_COOLDOWN_SECONDS,
|
||
DEFAULT_SPAWN_FAILURE_LIMIT,
|
||
DERIVED_MAX_IN_PROGRESS_CEILING,
|
||
DERIVED_MAX_IN_PROGRESS_FLOOR,
|
||
DispatchResult,
|
||
KANBAN_TERMINAL_TIMEOUT_GRACE_SECONDS,
|
||
MEMORY_GUARD_MB_PER_WORKER,
|
||
_PROTOCOL_VIOLATION_FAILURE_LIMIT,
|
||
_PROTOCOL_VIOLATION_SCAN_LIMIT,
|
||
_RECENT_WORKER_EXITS_MAX,
|
||
_RECENT_WORKER_EXIT_TTL_SECONDS,
|
||
_RESPAWN_BLOCKER_RE,
|
||
_RESPAWN_GUARD_PR_URL_RE,
|
||
_RESPAWN_GUARD_PR_WINDOW,
|
||
_RESPAWN_GUARD_SUCCESS_WINDOW,
|
||
_STALE_HEARTBEAT_GAP_SECONDS,
|
||
_absolute_hermes_path,
|
||
_apply_default_assignee,
|
||
_classify_worker_exit,
|
||
_clear_failure_counter,
|
||
_default_spawn,
|
||
_defer_reclaim_for_live_worker,
|
||
_dispatch_lane_task,
|
||
_dispatch_once_locked,
|
||
_error_fingerprint,
|
||
_has_spawnable,
|
||
_hermes_path_argv,
|
||
_is_windows_batch_shim,
|
||
_looks_like_path,
|
||
_memory_pressure_level,
|
||
_module_hermes_argv,
|
||
_path_search_names,
|
||
_pid_alive,
|
||
_positive_int,
|
||
_protocol_violation_streak,
|
||
_recent_worker_exits,
|
||
_record_spawn_failure,
|
||
_record_task_failure,
|
||
_record_worker_exit,
|
||
_resolve_hermes_argv,
|
||
_resolve_worker_cli_toolsets,
|
||
_retag_legacy_worker_sessions,
|
||
_retagged_workspace_roots,
|
||
_rotate_worker_log,
|
||
_rotated_log_path,
|
||
_safe_which_no_cwd,
|
||
_set_worker_pid,
|
||
_system_memory_sample,
|
||
_terminate_reclaimed_worker,
|
||
_worker_survived_termination,
|
||
_worker_terminal_timeout_env,
|
||
check_respawn_guard,
|
||
configured_max_in_progress,
|
||
count_running_tasks,
|
||
count_running_tasks_other_boards,
|
||
derive_default_max_in_progress,
|
||
detect_crashed_workers,
|
||
detect_stale_running,
|
||
dispatch_once,
|
||
enforce_max_runtime,
|
||
has_spawnable_ready,
|
||
has_spawnable_review,
|
||
heartbeat_worker,
|
||
reap_worker_zombies,
|
||
reconcile_orphaned_running,
|
||
resolve_max_in_progress,
|
||
review_dispatch_enabled,
|
||
run_daemon,
|
||
worker_log_rotation_config,
|
||
)
|
||
from hermes_cli.kanban_db_notify import ( # noqa: E402,F401
|
||
_NOTIFY_DELIVERY_MODES,
|
||
_decode_notify_delivery_metadata,
|
||
_encode_notify_delivery_metadata,
|
||
_notify_cursor,
|
||
_notify_profile_filter,
|
||
add_notify_sub,
|
||
advance_notify_cursor,
|
||
claim_unseen_events_for_sub,
|
||
count_notify_subs,
|
||
list_notify_subs,
|
||
purge_stale_done_notify_subs,
|
||
remove_notify_sub,
|
||
rewind_notify_cursor,
|
||
unseen_events_for_sub,
|
||
)
|