Files
hermes-agent/agent/curator.py
Teknium c408601937 refactor(agent/review): simplify curator, background_review, verify, insights, title and learning modules (-22% LOC)
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.

- Dead code: 27 private helpers with zero references removed
  (_auto_title_session, _resolve_review_model, _parse_make_targets,
  _filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
  _merge_runs, learning_graph_render bucket/period/node helpers,
  _memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
  _retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
  _ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
  _write_file/_write_json (4 curator report writers), _msg_text
  (background_review <- side_question), _report_failure/_notify_title
  (title_generator instant/auto paths), _is_under (verification_evidence),
  _scoped SQL pair builder + _query (insights), _optional_lock
  (background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
  curator_backup summary bits, learning_graph_render buckets, insights
  section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
  comments collapsed; every non-obvious WHY/invariant kept in compact form.

Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
2026-09-02 13:30:25 -07:00

1421 lines
63 KiB
Python

"""Curator — background skill maintenance orchestrator.
Inactivity-triggered (no cron daemon): when the agent is idle and the last run
is older than ``interval_hours``, ``maybe_run_curator()`` auto-transitions
lifecycle states from activity timestamps, optionally forks an AIAgent that may
pin/archive/consolidate/patch skills via skill_manage, and persists scheduler
state in ``.curator_state``.
Invariants: only curator-managed skills are touched; never delete, only archive
(recoverable); pinned skills bypass all auto-transitions; the fork uses the
auxiliary client and never touches the main session's prompt cache.
"""
from __future__ import annotations
import json
import logging
import os
import re
import threading
from collections import Counter
from datetime import datetime, timedelta, timezone
from pathlib import Path
from typing import Any, Callable, Dict, List, NamedTuple, Optional, Set
from hermes_constants import get_hermes_home
from tools import skill_usage
from utils import atomic_json_write
logger = logging.getLogger(__name__)
DEFAULT_INTERVAL_HOURS = 24 * 7 # 7 days
DEFAULT_MIN_IDLE_HOURS = 2
DEFAULT_STALE_AFTER_DAYS = 30
DEFAULT_ARCHIVE_AFTER_DAYS = 90
# The LLM consolidation fork is opt-in; the deterministic inactivity prune
# (apply_automatic_transitions) always runs when the curator is enabled.
DEFAULT_CONSOLIDATE = False
# --- .curator_state — persistent scheduler + status ---
def _state_file() -> Path:
return get_hermes_home() / "skills" / ".curator_state"
def _default_state() -> Dict[str, Any]:
return {
"last_run_at": None, "last_run_duration_seconds": None, "last_run_summary": None,
"last_run_summary_shown_at": None, "last_report_path": None, "paused": False, "run_count": 0,
}
def load_state() -> Dict[str, Any]:
path = _state_file()
if not path.exists():
return _default_state()
try:
data = json.loads(path.read_text(encoding="utf-8"))
if isinstance(data, dict):
base = _default_state()
base.update({k: v for k, v in data.items() if k in base or k.startswith("_")})
return base
except (OSError, json.JSONDecodeError) as e:
logger.debug("Failed to read curator state: %s", e)
return _default_state()
def save_state(data: Dict[str, Any]) -> None:
try:
atomic_json_write(_state_file(), data, indent=2, sort_keys=True)
except Exception as e:
logger.debug("Failed to save curator state: %s", e, exc_info=True)
def set_paused(paused: bool) -> None:
state = load_state()
state["paused"] = bool(paused)
save_state(state)
def is_paused() -> bool:
return bool(load_state().get("paused"))
# --- Config access ---
def _subdict(node: Any, *keys: str) -> Dict[str, Any]:
"""Walk nested dict keys; {} if any level is missing or not a dict."""
for key in keys:
node = node.get(key) if isinstance(node, dict) else None
return node if isinstance(node, dict) else {}
def _read_config_section(*path: str, label: str, log: logging.Logger = logger) -> Dict[str, Any]:
"""Read a nested section of ~/.hermes/config.yaml. Tolerates missing file."""
try:
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
log.debug("Failed to load config for %s: %s", label, e)
return {}
return _subdict(cfg, *path)
def _load_config() -> Dict[str, Any]:
"""Read curator.* config."""
return _read_config_section("curator", label="curator")
def _config_number(key: str, default, cast):
try:
return cast(_load_config().get(key, default))
except (TypeError, ValueError):
return default
def is_enabled() -> bool:
"""Default ON when no config says otherwise."""
return bool(_load_config().get("enabled", True))
def get_interval_hours() -> int:
return _config_number("interval_hours", DEFAULT_INTERVAL_HOURS, int)
def get_min_idle_hours() -> float:
return _config_number("min_idle_hours", DEFAULT_MIN_IDLE_HOURS, float)
def get_stale_after_days() -> int:
return _config_number("stale_after_days", DEFAULT_STALE_AFTER_DAYS, int)
def get_archive_after_days() -> int:
return _config_number("archive_after_days", DEFAULT_ARCHIVE_AFTER_DAYS, int)
def get_prune_builtins() -> bool:
"""Bundled built-ins are curation candidates (ON by default); they age out like
agent-created skills and a suppression list keeps them archived across
`hermes update` re-seeds. Hub-installed skills are never pruned."""
return bool(_load_config().get("prune_builtins", True))
def get_consolidate() -> bool:
"""Whether a run includes the LLM consolidation pass. OFF by default: only the
deterministic prune runs, no aux-model fork. ``hermes curator run
--consolidate`` overrides per invocation."""
return bool(_load_config().get("consolidate", DEFAULT_CONSOLIDATE))
# --- Idle / interval check ---
def _parse_iso(ts: Optional[str]) -> Optional[datetime]:
if not ts:
return None
try:
return datetime.fromisoformat(ts)
except (TypeError, ValueError):
return None
def should_run_now(now: Optional[datetime] = None) -> bool:
"""Gates: curator.enabled, not paused, ``last_run_at`` present AND older than
interval_hours. First observation seeds ``last_run_at`` to now and defers one
interval, so a fresh install/update never mutates the library on its first
tick. ``hermes curator run`` bypasses this; the idle check is the caller's."""
if not is_enabled() or is_paused():
return False
state = load_state()
last = _parse_iso(state.get("last_run_at"))
if now is None:
now = datetime.now(timezone.utc)
if last is None:
try:
state["last_run_at"] = now.isoformat()
state["last_run_summary"] = (
"deferred first run — curator seeded, will run after one "
"interval; use `hermes curator run --dry-run` to preview now"
)
save_state(state)
except Exception as e: # pragma: no cover — best-effort persistence
logger.debug("Failed to seed curator last_run_at: %s", e)
return False
if last.tzinfo is None:
last = last.replace(tzinfo=timezone.utc)
return (now - last) >= timedelta(hours=get_interval_hours())
# --- Automatic state transitions (pure function, no LLM) ---
def _cron_referenced_skills() -> Set[str]:
"""Skill names referenced by any cron job (incl. paused/disabled).
Best-effort: a cron import error or corrupt jobs store must never break the
curator, so any failure yields an empty set (no protection, but no crash).
"""
try:
from cron.jobs import referenced_skill_names as _refs
return _refs()
except Exception as e:
logger.debug("Curator could not read cron skill references: %s", e, exc_info=True)
return set()
def _archive_as_curator(_u, name: str) -> bool:
"""Archive via skill_usage with the ledger actor tagged 'curator', so the
ledger entry reads as an autonomous transition, not a foreground call."""
try:
from tools.skill_ledger import reset_ledger_actor, set_ledger_actor
_tok = set_ledger_actor("curator")
except Exception:
_tok = reset_ledger_actor = None # type: ignore[assignment]
try:
ok, _msg = _u.archive_skill(name)
finally:
if _tok is not None:
try:
reset_ledger_actor(_tok)
except Exception:
pass
return ok
def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int]:
"""Move every curator-managed skill between active/stale/archived based on
its latest real activity; pinned skills are never touched. Built-ins are
seeded with a baseline record on first sight so their inactivity clock
starts NOW, not at epoch. Returns a counter dict."""
from tools import skill_usage as _u
if now is None:
now = datetime.now(timezone.utc)
stale_cutoff = now - timedelta(days=get_stale_after_days())
archive_cutoff = now - timedelta(days=get_archive_after_days())
cron_referenced = _cron_referenced_skills()
counts = {"marked_stale": 0, "archived": 0, "reactivated": 0, "checked": 0, "seeded": 0}
for row in _u.curated_report():
counts["checked"] += 1
name = row["name"]
if row.get("pinned"):
continue
# Cron-referenced skills are in use by definition (usage only bumps when
# a job fires, so paused/rare jobs would age them out). Treat as pinned.
if name in cron_referenced:
continue
# First sight with no persisted record: anchor its clock to now and defer.
if not row.get("_persisted", True):
_u.seed_record_if_missing(name)
counts["seeded"] += 1
continue
# Never-active skills anchor on created_at so they don't self-archive.
last_activity = _parse_iso(row.get("last_activity_at"))
anchor = last_activity or _parse_iso(row.get("created_at")) or now
if anchor.tzinfo is None:
anchor = anchor.replace(tzinfo=timezone.utc)
current = row.get("state", _u.STATE_ACTIVE)
# use_count == 0 is absence of evidence, not staleness: never archive a
# never-used skill younger than stale_after_days.
never_used = int(row.get("use_count", 0) or 0) == 0
if never_used and anchor > stale_cutoff:
if current == _u.STATE_STALE:
_u.set_state(name, _u.STATE_ACTIVE)
counts["reactivated"] += 1
continue
if anchor <= archive_cutoff and current != _u.STATE_ARCHIVED:
if _archive_as_curator(_u, name):
counts["archived"] += 1
elif anchor <= stale_cutoff and current == _u.STATE_ACTIVE:
_u.set_state(name, _u.STATE_STALE)
counts["marked_stale"] += 1
elif anchor > stale_cutoff and current == _u.STATE_STALE:
# Used again after being marked stale — reactivate.
_u.set_state(name, _u.STATE_ACTIVE)
counts["reactivated"] += 1
return counts
# --- Review prompt for the forked agent ---
CURATOR_DRY_RUN_BANNER = (
"═══════════════════════════════════════════════════════════════\n"
"DRY-RUN — REPORT ONLY. DO NOT MUTATE THE SKILL LIBRARY.\n"
"═══════════════════════════════════════════════════════════════\n"
"\n"
"This is a PREVIEW pass. Follow every instruction below EXCEPT:\n"
"\n"
" • DO NOT call skill_manage with action=patch, create, delete, "
"write_file, or remove_file.\n"
" • skills_list and skill_view are FINE — read as much as you need.\n"
"\n"
"Your output IS the deliverable. Produce the exact same "
"human-readable summary and structured YAML block you would "
"produce on a live run — but describe the actions you WOULD take, "
"not actions you took. A downstream reviewer will read the report "
"and decide whether to approve a live run with "
"`hermes curator run` (no flag).\n"
"\n"
"If you accidentally take a mutating action, say so explicitly in "
"the summary so the reviewer can revert it.\n"
"═══════════════════════════════════════════════════════════════"
)
CURATOR_REVIEW_PROMPT = (
"You are running as Hermes' background skill CURATOR. This is an "
"UMBRELLA-BUILDING consolidation pass, not a passive audit and not a "
"duplicate-finder.\n\n"
"The goal of the skill collection is a LIBRARY OF CLASS-LEVEL "
"INSTRUCTIONS AND EXPERIENTIAL KNOWLEDGE. A collection of hundreds of "
"narrow skills where each one captures one session's specific bug is "
"a FAILURE of the library — not a feature. An agent searching skills "
"matches on descriptions, not on exact names (note: long descriptions "
"are truncated to 57 chars in the system prompt skill index — keep the "
"trigger class in that window). One broad umbrella "
"skill with labeled subsections beats five narrow siblings for "
"discoverability, not the other way around.\n\n"
"The right target shape is CLASS-LEVEL skills with rich SKILL.md "
"bodies + `references/`, `templates/`, and `scripts/` subfiles for "
"session-specific detail — not one-session-one-skill micro-entries.\n\n"
"Hard rules — do not violate:\n"
"1. DO NOT touch bundled, hub-installed, or external-dir skills "
"(`skills.external_dirs`). The candidate list below is already filtered "
"to local curator-managed skills only; external skills are externally "
"owned and read-only to this background curator.\n"
"2. DO NOT delete any skill. Archiving (moving the skill's directory "
"into ~/.hermes/skills/.archive/) is the maximum destructive action. "
"Archives are recoverable; deletion is not.\n"
"3. DO NOT touch skills shown as pinned=yes. Skip them entirely.\n"
"3b. DO NOT archive, delete, consolidate, move, or otherwise modify any "
"skill named in the protected built-ins list (currently: plan). These "
"back load-bearing UX (slash-command entry points referenced in docs and "
"tips) and are filtered out of the candidate list below — never resurrect "
"one as an archive or absorb target.\n"
"3c. DO NOT archive or prune any skill marked `cron=yes` in the candidate "
"list. A cron job depends on it and will fail to load it on its next "
"run. You MAY still consolidate it into an umbrella — but only because "
"the curator rewrites cron job skill references to follow consolidations; "
"never simply prune it.\n"
"4. DO NOT use usage counters as a reason to skip consolidation. The "
"counters are new and often mostly zero. Judge overlap on CONTENT, "
"not on use_count. 'use=0' is not evidence a skill is valuable; it's "
"absence of evidence either way. Corollary: 'use=0' is ALSO not a "
"reason to PRUNE a skill. Never archive a never-used skill (use=0) "
"unless it is at least 30 days old (check last_activity / created date) "
"AND its content is genuinely obsolete or fully absorbed elsewhere — a "
"recently-created skill simply may not have had its trigger come up yet.\n"
"5. DO NOT reject consolidation on the grounds that 'each skill has "
"a distinct trigger'. Pairwise distinctness is the wrong bar. The "
"right bar is: 'would a human maintainer write this as N separate "
"skills, or as one skill with N labeled subsections?' When the "
"answer is the latter, merge.\n\n"
"How to work — not optional:\n"
"1. Scan the full candidate list. Identify PREFIX CLUSTERS (skills "
"sharing a first word or domain keyword). Examples you are likely "
"to find: hermes-config-*, hermes-dashboard-*, gateway-*, codex-*, "
"ollama-*, anthropic-*, gemini-*, mcp-*, salvage-*, pr-*, "
"competitor-*, python-*, security-*, etc. Expect 10-25 clusters.\n"
"2. For each cluster with 2+ members, do NOT ask 'are these pairs "
"overlapping?' — ask 'what is the UMBRELLA CLASS these skills all "
"serve? Would a maintainer name that class and write one skill for "
"it?' If yes, pick (or create) the umbrella and absorb the siblings "
"into it.\n"
"3. Three ways to consolidate — use the right one per cluster:\n"
" a. MERGE INTO EXISTING UMBRELLA — one skill in the cluster is "
"already broad enough to be the umbrella (example: `pr-triage-"
"salvage` for the PR review cluster). Patch it to add a labeled "
"section for each sibling's unique insight, then archive the "
"siblings.\n"
" b. CREATE A NEW UMBRELLA SKILL.md — no existing member is broad "
"enough. Use skill_manage action=create to write a new class-level "
"skill whose SKILL.md covers the shared workflow and has short "
"labeled subsections. Archive the now-absorbed narrow siblings.\n"
" c. DEMOTE TO REFERENCES/TEMPLATES/SCRIPTS — a sibling has "
"narrow-but-valuable session-specific content. Move it into the "
"umbrella's appropriate support directory:\n"
" • `references/<topic>.md` for session-specific detail OR "
"condensed knowledge banks (quoted research, API docs excerpts, "
"domain notes, provider quirks, reproduction recipes)\n"
" • `templates/<name>.<ext>` for starter files meant to be "
"copied and modified\n"
" • `scripts/<name>.<ext>` for statically re-runnable actions "
"(verification scripts, fixture generators, probes)\n"
" Then archive the old sibling. Re-home the content through the "
"LEDGERED tool surface: `skill_manage action=write_file` on the umbrella "
"to place the file (subdirectories are created for you), then "
"`skill_manage action=remove_file` on the source to drop the original, "
"then `skill_manage action=delete` on the source. Never a terminal move "
"— a shell mv/cp writes the same bytes with no ledger entry, so the "
"archive that follows snapshots an already-stripped package and "
"`hermes curator rollback` restores a hollow skill (issue #96962).\n\n"
"Package integrity — not optional:\n"
"Before demoting or archiving a skill, inspect it as a COMPLETE "
"directory package, not just SKILL.md. A skill root may include "
"`references/`, `templates/`, `scripts/`, and `assets/`; `skill_view` "
"discovers those relative to the skill root. A reference markdown file "
"inside another skill is NOT a new skill root and does not get its own "
"linked-file discovery.\n"
"If the source skill has support files OR SKILL.md contains relative "
"links such as `references/...`, `templates/...`, `scripts/...`, or "
"`assets/...`, DO NOT flatten only SKILL.md into "
"`<umbrella>/references/<old>.md`. Choose one safe path instead:\n"
" • keep it as a standalone skill, OR\n"
" • fully merge it by re-homing every needed support file into the "
"umbrella's canonical `references/`, `templates/`, `scripts/`, or "
"`assets/` directories AND rewrite the destination instructions to "
"the new paths, OR\n"
" • archive the entire original skill package unchanged.\n"
"Never leave archived/demoted instructions pointing at files that were "
"left behind under the old skill directory.\n"
"4. Also flag skills whose NAME is too narrow (contains a PR number, "
"a feature codename, a specific error string, an 'audit' / "
"'diagnosis' / 'salvage' session artifact). These almost always "
"belong as a subsection or support file under a class-level umbrella.\n"
"5. Iterate. After one consolidation round, scan the remaining set "
"and look for the NEXT umbrella opportunity. Don't stop after 3 "
"merges.\n\n"
"Your toolset:\n"
" - skills_list, skill_view — read the current landscape\n"
" READ BEFORE WRITE — enforced, not advisory. Before skill_manage "
"action=patch, action=edit, action=write_file on a file that already "
"exists, or action=remove_file, call skill_view on that SAME target in "
"this review turn — skill_view(name) for SKILL.md, "
"skill_view(name, file_path=...) for a supporting file — and build the "
"write from the content it just returned. A write without that read is "
"REFUSED and nothing is saved.\n"
" - skill_manage action=patch — add sections to the umbrella\n"
" - skill_manage action=create — create a new umbrella SKILL.md\n"
" - skill_manage action=write_file — add a references/, templates/, "
"or scripts/ file under an existing skill (the skill must already "
"exist)\n"
" - skill_manage action=delete — archive a skill. MUST pass "
"`absorbed_into=<umbrella>` when you've merged its content into another "
"skill, or `absorbed_into=\"\"` when you're truly pruning with no "
"forwarding target. This drives cron-job skill-reference migration — "
"guessing from your YAML summary after the fact is fragile.\n"
" You have NO terminal access in this pass — every filesystem mutation "
"goes through skill_manage above so it is ledgered and rollback-able "
"(issue #96962). Reading files works through skill_view (including "
"skill_view(name, file_path=...) for support files).\n\n"
"'keep' is a legitimate decision ONLY when the skill is already a "
"class-level umbrella and none of the proposed merges would improve "
"discoverability. 'This is narrow but distinct from its siblings' "
"is NOT a reason to keep — it's a reason to move it under an "
"umbrella as a subsection or support file.\n\n"
"Expected output: real umbrella-ification. Process every obvious "
"cluster. If you end the pass with fewer than 10 archives, you "
"stopped too early — go back and look at the clusters you left "
"alone.\n\n"
"When done, write a human summary AND a structured machine-readable "
"block so downstream tooling can distinguish consolidation from "
"pruning. Format EXACTLY:\n\n"
"## Structured summary (required)\n"
"```yaml\n"
"consolidations:\n"
" - from: <old-skill-name>\n"
" into: <umbrella-skill-name>\n"
" reason: <one short sentence — why merged, not just 'similar'>\n"
"prunings:\n"
" - name: <skill-name>\n"
" reason: <one short sentence — why archived with no merge target>\n"
"```\n\n"
"Every skill you moved to .archive/ MUST appear in exactly one of the "
"two lists. If you consolidated X into umbrella Y (patched Y, wrote "
"a references file to Y, or created Y with X's content absorbed), X "
"goes under `consolidations` with `into: Y`. If you archived X with "
"no absorption — truly stale, irrelevant, or obsolete — X goes under "
"`prunings`. Leave a list empty (`consolidations: []`) if none. Do "
"not omit the block. The block comes AFTER your human-readable "
"summary of clusters processed, patches made, and decisions left alone."
)
CURATOR_PRUNE_BUILTINS_NOTE = (
"\n\nPRUNE-BUILTINS MODE IS ON: bundled built-in skills "
"ARE included in the candidate list below and MAY be "
"archived for staleness/irrelevance, overriding hard "
"rule #1 for bundled skills ONLY. Hub-installed skills "
"remain strictly off-limits. Treat a stale built-in the "
"same as a stale agent-created skill: archive it (never "
"delete). It will be restored on `hermes update` only if "
"the user explicitly restores it."
)
# --- Per-run reports — {YYYYMMDD-HHMMSS}/run.json + REPORT.md under logs/curator/ ---
def _reports_root() -> Path:
"""``~/.hermes/logs/curator/`` (telemetry next to agent.log, not under skills/).
mkdir'd here too so gateway-only / bare-library entry paths work."""
root = get_hermes_home() / "logs" / "curator"
try:
root.mkdir(parents=True, exist_ok=True)
except OSError as e:
logger.debug("Curator reports dir create failed: %s", e)
return root
def _needle_in_path_component(needle: str, path: str) -> bool:
"""True if *needle* equals a complete filename stem or directory name in
*path* — so "api" does not match "references/api-design.md". Hyphens and
underscores are normalised ("open-webui-setup" matches "open_webui_setup.md")."""
norm_needle = needle.replace("-", "_")
return any(
part and part.rsplit(".", 1)[0].replace("-", "_") == norm_needle
for part in path.replace("\\", "/").split("/")
)
def _skill_manage_args(tc: Any, *, raw_fallback: bool) -> Optional[Dict[str, Any]]:
"""Parsed arguments of a ``skill_manage`` tool call (JSON string or dict), or
None to skip. With *raw_fallback*, a malformed string yields ``{"_raw": raw}``
so substring matching still catches the common case."""
if not isinstance(tc, dict) or tc.get("name") != "skill_manage":
return None
raw = tc.get("arguments") or ""
if isinstance(raw, dict):
return raw
if not isinstance(raw, str):
return None
try:
args = json.loads(raw)
except Exception:
return {"_raw": raw} if raw_fallback else None
return args if isinstance(args, dict) else None
_REFERENCE_FIELDS = ("file_path", "file_content", "content", "new_string", "_raw")
def _find_reference(args: Dict[str, Any], needles: Set[str]) -> Optional[str]:
"""First argument value (in ``_REFERENCE_FIELDS`` order) that references
one of *needles*. ``file_path`` must match a whole path component; content
fields match on word boundaries so "test" does not match "latest"."""
for key in _REFERENCE_FIELDS:
hay = args.get(key)
if not isinstance(hay, str):
continue
for needle in needles:
if not needle:
continue
if (_needle_in_path_component(needle, hay) if key == "file_path"
else re.search(rf'\b{re.escape(needle)}\b', hay)):
return hay
return None
def _classify_removed_skills(
removed: List[str],
added: List[str],
after_names: Set[str],
tool_calls: List[Dict[str, Any]],
) -> Dict[str, List[Dict[str, Any]]]:
"""Split ``removed`` into consolidated vs pruned. Heuristic: a ``skill_manage``
call on a DIFFERENT, surviving-or-new skill whose file_path/content arguments
reference the removed name is the "absorbed" signal; earliest match wins.
Returns ``{"consolidated": [{name, into, evidence}], "pruned": [{name}]}``."""
consolidated: List[Dict[str, Any]] = []
pruned: List[Dict[str, Any]] = []
parsed_calls = [
args for args in (_skill_manage_args(tc, raw_fallback=True) for tc in tool_calls or [])
if args is not None
]
destinations = set(after_names) | set(added or [])
for name in removed:
if not name:
continue
needles = {name, name.replace("-", "_"), name.replace("_", "-")}
for args in parsed_calls:
target = args.get("name")
# Calls on the removed skill itself, or on a skill that no longer
# exists, are not consolidation evidence.
if not isinstance(target, str) or not target or target == name or target not in destinations:
continue
hay = _find_reference(args, needles)
if hay is not None:
consolidated.append({"name": name, "into": target, "evidence": (
f"skill_manage action={args.get('action', '?')} on '{target}' referenced '{name}' in {hay[:80]}"
)})
break
else:
pruned.append({"name": name})
return {"consolidated": consolidated, "pruned": pruned}
def _clean_str(value: Any) -> str:
return value.strip() if isinstance(value, str) else ""
def _parse_structured_summary(
llm_final: str,
) -> Dict[str, List[Dict[str, str]]]:
"""Extract the required fenced ```yaml block (``consolidations:`` /
``prunings:`` lists) from the curator's final response. Tolerant: missing
block or malformed YAML → empty lists (caller falls back to the tool-call
heuristic); a partial block returns what parsed.
Returns ``{"consolidations": [{from, into, reason}], "prunings": [{name, reason}]}``."""
out: Dict[str, List[Dict[str, str]]] = {"consolidations": [], "prunings": []}
if not llm_final or not isinstance(llm_final, str):
return out
# Match ```yaml specifically so a code sample the model quoted elsewhere is
# never mistaken for the summary.
match = re.search(r"```ya?ml\s*\n(.*?)\n```", llm_final, re.DOTALL | re.IGNORECASE)
if not match:
return out
try:
import yaml # type: ignore
data = yaml.safe_load(match.group(1))
except Exception:
return out
if not isinstance(data, dict):
return out
def _entries(key: str) -> List[Dict[str, Any]]:
raw = data.get(key) or []
return [e for e in raw if isinstance(e, dict)] if isinstance(raw, list) else []
for entry in _entries("consolidations"):
frm, into = _clean_str(entry.get("from")), _clean_str(entry.get("into"))
if frm and into:
out["consolidations"].append({"from": frm, "into": into, "reason": _clean_str(entry.get("reason"))})
for entry in _entries("prunings"):
name = _clean_str(entry.get("name"))
if name:
out["prunings"].append({"name": name, "reason": _clean_str(entry.get("reason"))})
return out
def _extract_absorbed_into_declarations(
tool_calls: List[Dict[str, Any]],
) -> Dict[str, Dict[str, Any]]:
"""Model-declared absorption targets from ``skill_manage(action='delete')``
calls — the authoritative classification signal (beats YAML parsing and
substring heuristics). Returns ``{name: {"into": umbrella | "", "declared": True}}``;
``into == ""`` is an explicit prune. Deletes omitting ``absorbed_into`` are
absent so the caller falls back to heuristic/YAML (older runs)."""
out: Dict[str, Dict[str, Any]] = {}
for tc in tool_calls or []:
args = _skill_manage_args(tc, raw_fallback=False)
if args is None or args.get("action") != "delete":
continue
name = args.get("name")
# absorbed_into must be present (empty string is meaningful).
target = args.get("absorbed_into")
if isinstance(name, str) and name.strip() and isinstance(target, str):
out[name.strip()] = {"into": target.strip(), "declared": True}
return out
def _reconcile_classification(
removed: List[str],
heuristic: Dict[str, List[Dict[str, Any]]],
model_block: Dict[str, List[Dict[str, str]]],
destinations: Set[str],
absorbed_declarations: Optional[Dict[str, Dict[str, Any]]] = None,
) -> Dict[str, List[Dict[str, Any]]]:
"""Merge heuristic (tool-call evidence) with the model's structured block.
First match wins; every removed skill lands in exactly one bucket:
- ``absorbed_into`` declared at delete is authoritative: existing target →
consolidated; ``""`` → pruned; missing target → fall through.
- Model-declared consolidation wins when its ``into`` is in ``destinations``.
- Model named a missing umbrella → heuristic's finding, else pruned.
- Heuristic-only consolidation kept, marked ``source="tool-call audit"``.
- Otherwise pruned (model-declared, or no-evidence fallback).
"""
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
declared = absorbed_declarations or {}
consolidated: List[Dict[str, Any]] = []
pruned: List[Dict[str, Any]] = []
for name in removed:
mc = model_cons.get(name)
mp = model_pruned.get(name)
hc = heur_cons.get(name)
dec = declared.get(name)
def _cons(into: str, source: str, reason: str = "", *, with_hc: bool = False, **extra: Any) -> None:
entry: Dict[str, Any] = {"name": name, "into": into, "source": source, "reason": reason}
if with_hc and hc and hc.get("evidence"):
entry["evidence"] = hc["evidence"]
entry.update(extra)
consolidated.append(entry)
def _prune(source: str, reason: str = "") -> None:
pruned.append({"name": name, "source": source, "reason": reason})
if dec is not None:
into_claim = dec.get("into", "")
if into_claim and into_claim in destinations:
_cons(into_claim, "absorbed_into (model-declared at delete)",
(mc.get("reason") or "") if mc else "", with_hc=True)
continue
if into_claim == "":
_prune("absorbed_into=\"\" (model-declared prune)", (mp.get("reason") or "") if mp else "")
continue
if mc and mc.get("into") in destinations:
_cons(mc["into"], "model" + ("+audit" if hc else ""), mc.get("reason") or "", with_hc=True)
elif mc: # model named a missing umbrella
if hc:
_cons(hc["into"], "tool-call audit (model named missing umbrella)",
evidence=hc.get("evidence", ""), model_claimed_into=mc["into"])
else:
_prune("fallback (model named missing umbrella, no tool-call evidence)")
elif hc:
_cons(hc["into"], "tool-call audit (model omitted from structured block)",
evidence=hc.get("evidence", ""))
else:
_prune("model" if mp else "no-evidence fallback", mp.get("reason", "") if mp else "")
return {"consolidated": consolidated, "pruned": pruned}
class _RunDiff(NamedTuple):
after_names: Set[str]
removed: List[str]
added: List[str]
consolidated: List[Dict[str, Any]]
pruned: List[Dict[str, Any]]
def _diff_and_classify(
before_names: Set[str],
after_names: Set[str],
tool_calls: List[Dict[str, Any]],
model_final: str,
) -> _RunDiff:
"""Diff the before/after skill sets and classify every removal: the model's
YAML block carries intent + rationale, the tool-call heuristic audits for
hallucinated umbrellas/omissions, per-delete ``absorbed_into`` beats both."""
removed = sorted(before_names - after_names)
added = sorted(after_names - before_names)
heuristic = _classify_removed_skills(
removed=removed, added=added, after_names=after_names, tool_calls=tool_calls,
)
classification = _reconcile_classification(
removed=removed,
heuristic=heuristic,
model_block=_parse_structured_summary(model_final),
destinations=set(after_names) | set(added),
absorbed_declarations=_extract_absorbed_into_declarations(tool_calls),
)
return _RunDiff(after_names, removed, added, classification["consolidated"], classification["pruned"])
def _build_rename_summary(
*,
before_names: Set[str],
after_report: List[Dict[str, Any]],
tool_calls: List[Dict[str, Any]],
model_final: str,
) -> str:
"""The "where did my skills go?" lines appended to the user-visible
``final_summary``; "" when nothing was archived. Capped at 10 entries so a
big consolidation doesn't flood agent.log (full list is in REPORT.md); the
pin hint appears only when a consolidation produced an umbrella."""
after_names = {r.get("name") for r in after_report if isinstance(r, dict)}
if not before_names - after_names:
return ""
diff = _diff_and_classify(before_names, after_names, tool_calls, model_final)
SHOW = 10
total = len(diff.consolidated) + len(diff.pruned)
entries = [f" • {e.get('name', '?')} → {e.get('into', '?')}" for e in diff.consolidated]
entries += [
f" • {e.get('name', '?') if isinstance(e, dict) else e} — pruned (stale)"
for e in diff.pruned
]
lines = [f"archived {total} skill(s):"] + entries[:SHOW]
if total > SHOW:
lines.append(f" … and {total - SHOW} more")
lines.append("full report: hermes curator status")
umbrellas = sorted({e.get("into") for e in diff.consolidated if e.get("into")})
if umbrellas:
lines.append(f"keep an umbrella stable: hermes curator pin {umbrellas[0]}")
return "\n".join(lines)
def _rewrite_cron_refs(consolidated: List[Dict[str, Any]], pruned: List[Dict[str, Any]]) -> Dict[str, Any]:
"""Point cron jobs at the umbrella when the curator consolidated a skill they
list — otherwise the scheduler fails to load it and the job runs without its
instructions. Best-effort: a cron-module issue never breaks the curator."""
try:
consolidated_map = {
e["name"]: e["into"] for e in consolidated if isinstance(e, dict) and e.get("name") and e.get("into")
}
pruned_names = [e["name"] for e in pruned if isinstance(e, dict) and e.get("name")]
if consolidated_map or pruned_names:
from cron.jobs import rewrite_skill_refs
return rewrite_skill_refs(consolidated=consolidated_map, pruned=pruned_names)
return {"rewrites": [], "jobs_updated": 0, "jobs_scanned": 0}
except Exception as e:
logger.debug("Curator cron skill rewrite failed: %s", e, exc_info=True)
return {"rewrites": [], "jobs_updated": 0, "jobs_scanned": 0, "error": str(e)}
def _write_file(path: Path, label: str, render: Callable[[], str]) -> None:
"""Best-effort write; *render* runs inside the guard so a serialisation
error is logged, not raised."""
try:
path.write_text(render(), encoding="utf-8")
except Exception as e:
logger.debug("Curator %s write failed: %s", label, e)
def _write_json(path: Path, payload: Any, label: str) -> None:
_write_file(path, label, lambda: json.dumps(payload, indent=2, ensure_ascii=False) + "\n")
def _write_run_report(
*,
started_at: datetime,
elapsed_seconds: float,
auto_counts: Dict[str, int],
auto_summary: str,
before_report: List[Dict[str, Any]],
before_names: Set[str],
after_report: List[Dict[str, Any]],
llm_meta: Dict[str, Any],
) -> Optional[Path]:
"""Write run.json + REPORT.md under logs/curator/{YYYYMMDD-HHMMSS}/. Returns
the report dir, or None if it couldn't be created (reporting is best-effort)."""
root = _reports_root()
try:
root.mkdir(parents=True, exist_ok=True)
except Exception as e:
logger.debug("Curator report dir create failed: %s", e)
return None
stamp = started_at.strftime("%Y%m%d-%H%M%S")
run_dir, suffix = root / stamp, 1 # crash-rerun within the same second gets a disambiguator
while run_dir.exists():
suffix += 1
run_dir = root / f"{stamp}-{suffix}"
try:
run_dir.mkdir(parents=True, exist_ok=False)
except Exception as e:
logger.debug("Curator run dir create failed: %s", e)
return None
tool_calls = llm_meta.get("tool_calls", []) or []
after_by_name = {r.get("name"): r for r in after_report if isinstance(r, dict)}
before_by_name = {r.get("name"): r for r in before_report if isinstance(r, dict)}
diff = _diff_and_classify(
before_names, set(after_by_name), tool_calls, llm_meta.get("final", "") or ""
)
transitions: List[Dict[str, str]] = []
for name in sorted(diff.after_names & before_names):
s_before = (before_by_name.get(name) or {}).get("state")
s_after = (after_by_name.get(name) or {}).get("state")
if s_before and s_after and s_before != s_after:
transitions.append({"name": name, "from": s_before, "to": s_after})
tc_counts: Dict[str, int] = dict(Counter(tc.get("name", "unknown") for tc in tool_calls))
cron_rewrites = _rewrite_cron_refs(diff.consolidated, diff.pruned)
payload = {
"started_at": started_at.isoformat(),
"duration_seconds": round(elapsed_seconds, 2),
"model": llm_meta.get("model", ""),
"provider": llm_meta.get("provider", ""),
"auto_transitions": auto_counts,
"counts": {
"before": len(before_names),
"after": len(diff.after_names),
"delta": len(diff.after_names) - len(before_names),
"archived_this_run": len(diff.removed),
"added_this_run": len(diff.added),
"consolidated_this_run": len(diff.consolidated),
"pruned_this_run": len(diff.pruned),
"state_transitions": len(transitions),
"cron_jobs_rewritten": int(cron_rewrites.get("jobs_updated", 0)),
"tool_calls_total": sum(tc_counts.values()),
},
"tool_call_counts": tc_counts,
"archived": diff.removed,
"consolidated": diff.consolidated,
"pruned": diff.pruned,
"pruned_names": [p["name"] for p in diff.pruned],
"added": diff.added,
"state_transitions": transitions,
"cron_rewrites": cron_rewrites,
"llm_final": llm_meta.get("final", ""),
"llm_summary": llm_meta.get("summary", ""),
"llm_error": llm_meta.get("error"),
"tool_calls": llm_meta.get("tool_calls", []),
}
_write_json(run_dir / "run.json", payload, "run.json")
_write_file(run_dir / "REPORT.md", "REPORT.md", lambda: _render_report_markdown(payload))
# Only when a job was touched, to keep no-op run dirs uncluttered.
if int(cron_rewrites.get("jobs_updated", 0)) > 0:
_write_json(run_dir / "cron_rewrites.json", cron_rewrites, "cron_rewrites.json")
return run_dir
def _render_report_markdown(p: Dict[str, Any]) -> str:
"""Render the human-readable REPORT.md."""
lines: List[str] = []
duration = p.get("duration_seconds", 0) or 0
mins, secs = divmod(int(duration), 60)
dur_label = f"{mins}m {secs}s" if mins else f"{secs}s"
counts = p.get("counts") or {}
auto = p.get("auto_transitions") or {}
tc_counts = p.get("tool_call_counts") or {}
error = p.get("llm_error")
lines += [
f"# Curator run — {p.get('started_at', '')}\n",
f"Model: `{p.get('model') or '(not resolved)'}` via `{p.get('provider') or '(not resolved)'}` · "
f"Duration: {dur_label} · "
f"Agent-created skills: {counts.get('before', 0)} → {counts.get('after', 0)} "
f"({counts.get('delta', 0):+d})\n",
]
if error:
lines.append(f"> ⚠ LLM pass error: `{error}`\n")
lines += [
"## Auto-transitions (pure, no LLM)\n",
f"- checked: {auto.get('checked', 0)}",
f"- marked stale: {auto.get('marked_stale', 0)}",
f"- archived (no LLM, pure time-based staleness): {auto.get('archived', 0)}",
f"- reactivated: {auto.get('reactivated', 0)}",
"",
"## LLM consolidation pass\n",
f"- tool calls: **{counts.get('tool_calls_total', 0)}** "
f"(by name: {', '.join(f'{k}={v}' for k, v in sorted(tc_counts.items())) or 'none'})",
f"- consolidated into umbrellas: **{counts.get('consolidated_this_run', 0)}**",
f"- pruned (archived for staleness): **{counts.get('pruned_this_run', 0)}**",
f"- new skills this run: **{counts.get('added_this_run', 0)}**",
f"- state transitions (active ↔ stale ↔ archived): "
f"**{counts.get('state_transitions', 0)}**",
"",
]
def _overflow(items: list, show: int, hint: str) -> None:
if len(items) > show:
lines.append(f"- … and {len(items) - show} more ({hint})")
lines.append("")
def _reason(entry: Dict[str, Any]) -> str:
reason = (entry.get("reason") or "").strip()
return f" — {reason}" if reason else ""
# Consolidated — the directory is archived (recoverable by design) but the
# live content continues inside the destination umbrella.
consolidated = p.get("consolidated") or []
if consolidated:
lines += [
f"### Consolidated into umbrella skills ({len(consolidated)})\n",
"_These skills were **absorbed into another skill** during this run — "
"their content still lives, just under a different name. "
"The original directory was moved to `~/.hermes/skills/.archive/` for "
"safety and can be restored via `hermes curator restore <name>` if the "
"consolidation was wrong._\n",
]
for entry in consolidated[:50]:
line = f"- `{entry.get('name', '?')}` → merged into `{entry.get('into', '?')}`" + _reason(entry)
source = entry.get("source", "")
if source and source.startswith("tool-call audit"):
# The model didn't enumerate this one — explains the missing rationale.
line += f" _(detected via {source})_"
lines.append(line)
if entry.get("model_claimed_into"):
lines.append(
f" ⚠ The curator's summary named `{entry['model_claimed_into']}` "
"as the umbrella but that skill doesn't exist post-run; "
"showing the tool-call audit's finding instead."
)
_overflow(consolidated, 50, "see `run.json`")
pruned = p.get("pruned") or []
if pruned:
lines += [
f"### Pruned — archived for staleness ({len(pruned)})\n",
"_These skills were archived without being merged into an umbrella "
"(e.g. stale, unused, or judged irrelevant). "
"Directories live under `~/.hermes/skills/.archive/`. "
"Restore any via `hermes curator restore <name>`._\n",
]
for entry in pruned[:50]:
# Reconciler entries are dicts {name, source, reason}; tolerate bare strings (older format).
if isinstance(entry, dict):
lines.append(f"- `{entry.get('name', '?')}`" + _reason(entry))
else:
lines.append(f"- `{entry}`")
_overflow(pruned, 50, "see `run.json`")
added = p.get("added") or []
if added:
lines += [
f"### New skills this run ({len(added)})\n",
"_Usually these are new class-level umbrellas created via `skill_manage action=create`._\n",
]
lines += [f"- `{n}`" for n in added]
lines.append("")
trans = p.get("state_transitions") or []
if trans:
lines.append(f"### State transitions ({len(trans)})\n")
lines += [f"- `{t.get('name')}`: {t.get('from')} → {t.get('to')}" for t in trans]
lines.append("")
# Cron rewrites — lets users audit that the auto-rewrite did the right thing.
cron_rewrites_list = (p.get("cron_rewrites") or {}).get("rewrites") or []
if cron_rewrites_list:
lines += [
f"### Cron job skill references rewritten ({len(cron_rewrites_list)})\n",
"_Cron jobs that referenced a consolidated or pruned skill were "
"updated in-place so they keep loading the right instructions "
"on their next run. See `cron_rewrites.json` for the full record._\n",
]
for entry in cron_rewrites_list[:25]:
job_name = entry.get("job_name") or entry.get("job_id") or "?"
before, after = entry.get("before") or [], entry.get("after") or []
lines.append(f"- `{job_name}`: `{', '.join(before)}` → `{', '.join(after) or '(none)'}`")
lines += [f" - `{old}` → `{new}` (consolidated)" for old, new in (entry.get("mapped") or {}).items()]
lines += [f" - `{name}` dropped (pruned)" for name in (entry.get("dropped") or [])]
_overflow(cron_rewrites_list, 25, "see `cron_rewrites.json`")
final = (p.get("llm_final") or "").strip()
if final:
lines += ["## LLM final summary\n", final, ""]
elif not error and (p.get("llm_summary") or ""):
lines += ["## LLM summary\n", p.get("llm_summary"), ""]
lines += [
"## Recovery\n",
"- Restore an archived skill: `hermes curator restore <name>`",
"- All archives live under `~/.hermes/skills/.archive/` and are recoverable by `mv`",
"- See `run.json` in this directory for the full machine-readable record.",
"",
]
return "\n".join(lines)
# --- Orchestrator — spawn a forked AIAgent for the LLM review pass ---
def _render_candidate_list() -> str:
"""Human/agent-readable list of curator-managed skills with usage stats."""
rows = skill_usage.curated_report()
if not rows:
return "No curator-managed skills to review."
cron_referenced = _cron_referenced_skills()
lines = [f"Curator-managed skills ({len(rows)}):\n"] + [
f"- {r['name']} provenance={r.get('provenance', 'agent')} state={r['state']} "
f"pinned={'yes' if r.get('pinned') else 'no'} cron={'yes' if r['name'] in cron_referenced else 'no'} "
f"activity={r.get('activity_count', 0)} use={r.get('use_count', 0)} view={r.get('view_count', 0)} "
f"patches={r.get('patch_count', 0)} last_activity={r.get('last_activity_at') or 'never'}"
for r in rows
]
return "\n".join(lines)
def _llm_meta(summary: str, error: Optional[str] = None) -> Dict[str, Any]:
"""Structured result of an LLM pass that did not run (skipped or failed)."""
return {"final": "", "summary": summary, "model": "", "provider": "", "tool_calls": [], "error": error}
def _notify(on_summary: Optional[Callable[[str], None]], message: str) -> None:
if on_summary:
try:
on_summary(message)
except Exception:
pass
def _safe_curated_report() -> List[Dict[str, Any]]:
try:
return skill_usage.curated_report()
except Exception:
return []
def run_curator_review(
on_summary: Optional[Callable[[str], None]] = None,
synchronous: bool = False,
dry_run: bool = False,
consolidate: Optional[bool] = None,
) -> Dict[str, Any]:
"""Execute a single curator review pass: (1) automatic state transitions (no
LLM); (2) if *consolidate* and there are candidates, fork an AIAgent on the
review prompt; (3) update .curator_state; (4) call *on_summary*.
*synchronous* runs the LLM review in the calling thread (default: daemon
thread). *consolidate* ``None`` reads ``curator.consolidate`` (OFF by
default); when off only the deterministic prune runs — no fork, no aux cost.
*dry_run* SKIPS the stale/archive transitions and instructs the fork to
report only; REPORT.md is still written and recorded in
``state.last_report_path`` so users can read what WOULD have happened.
"""
if consolidate is None:
consolidate = get_consolidate()
start = datetime.now(timezone.utc)
if dry_run:
# Count candidates without mutating state.
counts = {"checked": len(_safe_curated_report()), "marked_stale": 0, "archived": 0, "reactivated": 0}
else:
# Pre-mutation snapshot — best-effort, never blocks the run: a transient
# disk issue must not silently disable the curator forever.
try:
from agent import curator_backup
snap = curator_backup.snapshot_skills(reason="pre-curator-run")
if snap is not None:
_notify(on_summary, f"curator: snapshot created ({snap.name})")
except Exception as e:
logger.debug("Curator pre-run snapshot failed: %s", e, exc_info=True)
counts = apply_automatic_transitions(now=start)
auto_summary = ", ".join(
f"{counts[key]} {label}"
for key, label in (("marked_stale", "marked stale"), ("archived", "archived"), ("reactivated", "reactivated"))
if counts[key]
) or "no changes"
# Persist before the LLM pass so a crash mid-review still records the run.
# Dry-run does NOT bump last_run_at/run_count (a preview must not push the
# next real pass out) but still records a summary for `hermes curator status`.
state = load_state()
if not dry_run:
state["last_run_at"] = start.isoformat()
state["run_count"] = int(state.get("run_count", 0)) + 1
prefix = "dry-run auto: " if dry_run else "auto: "
state["last_run_summary"] = f"{prefix}{auto_summary}"
save_state(state)
def _llm_pass():
# Snapshot skill state BEFORE the LLM pass so the report can diff.
before_report = _safe_curated_report()
before_names = {r.get("name") for r in before_report if isinstance(r, dict)}
if not consolidate:
# Prune-only run: record it and write a report, but never fork.
final_summary = f"{prefix}{auto_summary}; llm: skipped (consolidation off)"
llm_meta = _llm_meta("skipped (consolidation off)")
else:
try:
candidate_list = _render_candidate_list()
if "No agent-created skills" in candidate_list:
final_summary = f"{prefix}{auto_summary}; llm: skipped (no candidates)"
llm_meta = _llm_meta("skipped (no candidates)")
else:
# With prune-builtins on, bundled skills are candidates too:
# relax hard rule #1 for them (archive only; hub stays off-limits).
builtins_note = CURATOR_PRUNE_BUILTINS_NOTE if get_prune_builtins() else ""
prompt = f"{CURATOR_REVIEW_PROMPT}{builtins_note}\n\n{candidate_list}"
if dry_run:
prompt = f"{CURATOR_DRY_RUN_BANNER}\n\n{prompt}"
llm_meta = _run_llm_review(prompt)
final_summary = (
f"{prefix}{auto_summary}; llm: {llm_meta.get('summary', 'no change')}"
)
except Exception as e:
logger.debug("Curator LLM pass failed: %s", e, exc_info=True)
final_summary = f"{prefix}{auto_summary}; llm: error ({e})"
llm_meta = _llm_meta(f"error ({e})", str(e))
# Append the rename map (`old-name → umbrella`) so users needn't dig
# into REPORT.md. Best-effort: never block the run on formatting.
try:
rename_lines = _build_rename_summary(
before_names=before_names,
after_report=skill_usage.curated_report(),
tool_calls=llm_meta.get("tool_calls", []) or [],
model_final=llm_meta.get("final", "") or "",
)
if rename_lines:
final_summary = f"{final_summary}\n{rename_lines}"
except Exception as e:
logger.debug("Curator rename summary build failed: %s", e, exc_info=True)
elapsed = (datetime.now(timezone.utc) - start).total_seconds()
state2 = load_state()
state2["last_run_duration_seconds"] = elapsed
state2["last_run_summary"] = final_summary
# Per-run report, best-effort; path recorded for `hermes curator status`.
after_report = _safe_curated_report()
try:
report_path = _write_run_report(
started_at=start,
elapsed_seconds=elapsed,
auto_counts=counts,
auto_summary=auto_summary,
before_report=before_report,
before_names=before_names,
after_report=after_report,
llm_meta=llm_meta,
)
if report_path is not None:
state2["last_report_path"] = str(report_path)
except Exception as e:
logger.debug("Curator report write failed: %s", e, exc_info=True)
save_state(state2)
_notify(on_summary, f"curator: {final_summary}")
if synchronous:
_llm_pass()
else:
threading.Thread(target=_llm_pass, daemon=True, name="curator-review").start()
return {"started_at": start.isoformat(), "auto_transitions": counts, "summary_so_far": auto_summary}
# --- Provider/model resolution for the review fork ---
class _ReviewRuntimeBinding(NamedTuple):
"""Provider/model for the curator review fork plus per-slot overrides."""
provider: str
model: str
explicit_api_key: Optional[str]
explicit_base_url: Optional[str]
request_overrides: Dict[str, Any]
def _strip_aux_credential(value: Any) -> Optional[str]:
return (str(value).strip() or None) if value is not None else None
def _merge_request_overrides(runtime_overrides: Any, slot_extra_body: Any) -> Dict[str, Any]:
"""Merge resolver metadata with task-local request body fields."""
merged = dict(runtime_overrides or {})
if isinstance(slot_extra_body, dict) and slot_extra_body:
merged["extra_body"] = {**(merged.get("extra_body") or {}), **slot_extra_body}
return merged
def _slot_binding(provider: str, model: str, slot: Dict[str, Any]) -> _ReviewRuntimeBinding:
return _ReviewRuntimeBinding(
provider, model, _strip_aux_credential(slot.get("api_key")),
_strip_aux_credential(slot.get("base_url")), _merge_request_overrides({}, slot.get("extra_body")),
)
def _resolve_review_runtime(cfg: Dict[str, Any]) -> _ReviewRuntimeBinding:
"""Curator is a regular auxiliary task slot (``auxiliary.curator.*``), so it
rides the canonical aux-model plumbing. Precedence:
1. ``auxiliary.curator.{provider,model}`` when both are set non-auto
2. Legacy ``curator.auxiliary.{provider,model}`` (deprecated) when both set
3. Main ``model.{provider,default/model}`` pair ("auto" + "" = main chat model)
Non-empty slot ``api_key``/``base_url`` are returned as explicit overrides so
``resolve_runtime_provider`` doesn't reuse the main chat credential chain.
"""
_cur_task = _subdict(cfg, "auxiliary", "curator")
_task_provider = (_cur_task.get("provider") or "").strip() or None
_task_model = (_cur_task.get("model") or "").strip() or None
if _task_provider and _task_provider != "auto" and _task_model:
return _slot_binding(_task_provider, _task_model, _cur_task)
_legacy = _subdict(cfg, "curator", "auxiliary")
_legacy_provider = _legacy.get("provider") or None
_legacy_model = _legacy.get("model") or None
if _legacy_provider and _legacy_model:
logger.info(
"curator: using deprecated curator.auxiliary.{provider,model} "
"config — please migrate to auxiliary.curator.{provider,model}"
)
return _slot_binding(str(_legacy_provider), str(_legacy_model), _legacy)
_main = _subdict(cfg, "model")
return _ReviewRuntimeBinding(
_main.get("provider") or "auto", _main.get("default") or _main.get("model") or "", None, None, {},
)
def _run_llm_review(prompt: str) -> Dict[str, Any]:
"""Spawn an AIAgent fork on the review prompt. Returns ``final`` (untruncated
response), ``summary`` (240-char cap), ``model``/``provider`` (what ran),
``tool_calls`` ([{name, arguments}], truncated) and ``error``. Never raises."""
import contextlib
result_meta: Dict[str, Any] = _llm_meta("")
try:
from run_agent import AIAgent
except Exception as e:
result_meta["error"] = result_meta["summary"] = f"AIAgent import failed: {e}"
return result_meta
# Resolve provider + model the same way the CLI does: AIAgent() without
# explicit provider/model hits an auto-resolution path that fails for
# OAuth-only providers and pooled credentials (HTTP 400 "No models provided").
_rp: Dict[str, Any] = {}
_request_overrides: Dict[str, Any] = {}
_resolved_provider, _model_name = None, ""
try:
from hermes_cli.config import load_config_readonly
from hermes_cli.runtime_provider import resolve_runtime_provider
_binding = _resolve_review_runtime(load_config_readonly())
_model_name = _binding.model
_rp = resolve_runtime_provider(
requested=_binding.provider,
target_model=_binding.model,
explicit_api_key=_binding.explicit_api_key,
explicit_base_url=_binding.explicit_base_url,
)
_resolved_provider = _rp.get("provider") or _binding.provider
_request_overrides = _merge_request_overrides(
_rp.get("request_overrides"),
_binding.request_overrides.get("extra_body"),
)
if isinstance(_rp.get("model"), str) and _rp["model"].strip():
_model_name = _rp["model"].strip()
except Exception as e:
logger.debug("Curator provider resolution failed: %s", e, exc_info=True)
result_meta["model"] = _model_name
result_meta["provider"] = _resolved_provider or ""
review_agent = None
try:
_agent_kwargs: Dict[str, Any] = {}
if isinstance(_rp.get("max_output_tokens"), int):
_agent_kwargs["max_tokens"] = _rp["max_output_tokens"]
_acp_command = _rp.get("command")
if isinstance(_acp_command, str) and _acp_command:
_agent_kwargs["acp_command"] = _acp_command
_agent_kwargs["acp_args"] = list(_rp.get("args") or [])
review_agent = AIAgent(
model=_model_name,
provider=_resolved_provider,
api_key=_rp.get("api_key"),
base_url=_rp.get("base_url"),
api_mode=_rp.get("api_mode"),
credential_pool=_rp.get("credential_pool"),
request_overrides=_request_overrides,
**_agent_kwargs,
# No ``terminal``: a shell mv/cp/rm under the skills tree writes bytes
# with NO ledger entry, so rollback would restore a hollow skill. Every
# mutation goes through ledgered skill_manage; dropping the toolset
# closes the hole by construction (no command heuristic can).
enabled_toolsets=["skills"],
# Umbrella-building over hundreds of skills takes 50-100 API calls.
max_iterations=9999,
quiet_mode=True,
platform="curator",
skip_context_files=True,
skip_memory=True,
)
# Disable recursive nudges — the curator must never spawn its own review.
review_agent._memory_nudge_interval = 0
review_agent._skill_nudge_interval = 0
# Tag as autonomous background curation so skill_manage's background-review
# write guards (external/bundled/hub) fire; turn_context binds this onto
# the write-origin ContextVar at turn start.
review_agent._memory_write_origin = "background_review"
# Silence the fork's tool-call chatter (CLI synchronous foreground runs).
with open(os.devnull, "w", encoding="utf-8") as _devnull, \
contextlib.redirect_stdout(_devnull), \
contextlib.redirect_stderr(_devnull):
conv_result = review_agent.run_conversation(user_message=prompt)
final = str(conv_result.get("final_response") or "").strip() if isinstance(conv_result, dict) else ""
result_meta["final"] = final
result_meta["summary"] = (final[:240] + "…") if len(final) > 240 else (final or "no change")
# Collect tool calls for the report; truncate arguments so a giant
# skill_manage create doesn't blow up the report.
_calls: List[Dict[str, Any]] = []
for msg in getattr(review_agent, "_session_messages", []) or []:
for tc in (msg.get("tool_calls") or []) if isinstance(msg, dict) else []:
if not isinstance(tc, dict):
continue
fn = tc.get("function") or {}
args_raw = fn.get("arguments") or ""
if isinstance(args_raw, str) and len(args_raw) > 400:
args_raw = args_raw[:400] + "…"
_calls.append({"name": fn.get("name") or "", "arguments": args_raw})
result_meta["tool_calls"] = _calls
except Exception as e:
result_meta["error"] = result_meta["summary"] = f"error: {e}"
finally:
if review_agent is not None:
try:
review_agent.close()
except Exception:
pass
return result_meta
# --- Public entrypoint for the session-start hook ---
def maybe_run_curator(
*,
idle_for_seconds: Optional[float] = None,
on_summary: Optional[Callable[[str], None]] = None,
) -> Optional[Dict[str, Any]]:
"""Best-effort: run a curator pass if all gates pass. Returns the result
dict if a pass was started, else None. Never raises."""
try:
# Idle gating: only enforce when the caller provided a measurement.
if not should_run_now() or (
idle_for_seconds is not None and idle_for_seconds < get_min_idle_hours() * 3600.0
):
return None
return run_curator_review(on_summary=on_summary)
except Exception as e:
logger.debug("maybe_run_curator failed: %s", e, exc_info=True)
return None