Files
hermes-agent/agent/review_engine.py
Teknium c408601937 refactor(agent/review): simplify curator, background_review, verify, insights, title and learning modules (-22% LOC)
Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.

- Dead code: 27 private helpers with zero references removed
  (_auto_title_session, _resolve_review_model, _parse_make_targets,
  _filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
  _merge_runs, learning_graph_render bucket/period/node helpers,
  _memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
  _retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
  _ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
  _write_file/_write_json (4 curator report writers), _msg_text
  (background_review <- side_question), _report_failure/_notify_title
  (title_generator instant/auto paths), _is_under (verification_evidence),
  _scoped SQL pair builder + _query (insights), _optional_lock
  (background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
  curator_backup summary bits, learning_graph_render buckets, insights
  section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
  comments collapsed; every non-obvious WHY/invariant kept in compact form.

Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
2026-09-02 13:30:25 -07:00

251 lines
9.9 KiB
Python

"""Shared engine for the /review command — every surface calls this.
/review spawns an independent, full-privilege background subagent (the same
async rail as ``delegate_task(background=true)``) to thoroughly review whatever
the recent conversation presented (PR, diff, code, docs). Its result re-enters
the spawning session as a normal async-delegation completion.
Model routing: ``auxiliary.review`` (provider/model/base_url/api_key/api_mode)
when configured, else the parent agent's credentials (main-model-first). It is
passed as ``credentials_cfg`` to ``delegate_task`` so native-SDK providers,
api_mode detection and credential pools behave identically to
``delegation.provider`` pins.
Surfaces (CLI/gateway ``/review``, TUI/Desktop) are thin adapters: snapshot the
conversation, call :func:`start_review`, print the dispatch note.
"""
from __future__ import annotations
import json
import logging
import re
from typing import Any, Dict, List, Optional
logger = logging.getLogger(__name__)
# How many recent chat messages (user + assistant turns) the reviewer gets.
DEFAULT_CONTEXT_MESSAGES = 10
# Per-message excerpt cap: generous (a PR summary/diff excerpt is exactly what
# the reviewer needs) but bounded against a pathological turn.
_MESSAGE_CHAR_CAP = 12_000
def _message_text(message: Dict[str, Any]) -> str:
"""Display text of a message; multimodal parts are joined, non-text parts noted."""
content = message.get("content")
if isinstance(content, str):
return content
if isinstance(content, list):
parts = [
str(part.get("text") or "") if part.get("type") == "text"
else f"[{part.get('type', 'attachment')}]"
for part in content
if isinstance(part, dict)
]
return "\n".join(p for p in parts if p)
return ""
def snapshot_recent_messages(
messages: List[Dict[str, Any]],
limit: int = DEFAULT_CONTEXT_MESSAGES,
) -> List[Dict[str, str]]:
"""Last ``limit`` user/assistant messages as {role, text} dicts, oldest first.
System messages, tool results and empty-text messages (pure tool-call
assistant stubs) are excluded.
"""
out: List[Dict[str, str]] = []
for message in reversed(list(messages or [])):
if not isinstance(message, dict):
continue
role = str(message.get("role") or "")
text = _message_text(message).strip() if role in ("user", "assistant") else ""
if not text:
continue
if len(text) > _MESSAGE_CHAR_CAP:
text = text[:_MESSAGE_CHAR_CAP] + "\n[... truncated ...]"
out.append({"role": role, "text": text})
if len(out) >= limit:
break
out.reverse()
return out
def collect_parent_loaded_skills(
parent_agent,
messages: List[Dict[str, Any]],
limit: int = 8,
) -> List[str]:
"""Names of skills the parent agent was operating under.
Launch-preloaded skills come from the stable marker in the parent's
``ephemeral_system_prompt`` (``build_preloaded_skills_prompt``); mid-session
loads from ``skill_view`` tool calls in the history. Preloaded first, then
history loads, deduped, capped at ``limit`` (a reviewer told to load 30
skills would burn its budget before working).
"""
names: List[str] = []
prompt = str(getattr(parent_agent, "ephemeral_system_prompt", "") or "")
candidates = [m.group(1) for m in re.finditer(r'with the "([^"]+)" skill\s+preloaded', prompt)]
for message in messages or []:
if not isinstance(message, dict) or message.get("role") != "assistant":
continue
for tool_call in message.get("tool_calls") or []:
fn = tool_call.get("function") or {} if isinstance(tool_call, dict) else {}
if fn.get("name") != "skill_view":
continue
try:
args = json.loads(fn.get("arguments") or "{}")
except Exception:
continue
# Only whole-skill loads seed the reviewer; a reference-file read
# is a detail of the parent's task covered by loading the SKILL.md.
if isinstance(args, dict) and not args.get("file_path"):
candidates.append(str(args.get("name") or ""))
for name in candidates:
cleaned = name.strip()
if cleaned and cleaned not in names:
names.append(cleaned)
return names[:limit]
def build_review_task(
snapshot: List[Dict[str, str]],
user_prompt: str = "",
loaded_skills: Optional[List[str]] = None,
) -> tuple:
"""Compose the reviewer subagent's (goal, context) pair."""
goal = (
"Act as an independent senior reviewer. Thoroughly review the work "
"presented in the conversation excerpt provided in your context: "
"investigate any code, pull request, branch, commit, documentation, "
"design, or other artifact it references (open the PR, read the "
"diff, run the code or tests where feasible) rather than judging "
"from the excerpt alone. Produce a full, structured review: what "
"the work does, whether it is correct and complete, concrete "
"defects or risks found (with file/line references where possible), "
"what was verified vs. only read, and a clear final verdict with "
"recommended next steps."
)
lines = [
"You were spawned by the /review command. The following is an "
"excerpt of the most recent conversation between the user and "
"their primary agent. It is your starting evidence — the work to "
"review is referenced in it.",
"",
"--- Recent conversation (oldest first) ---",
]
for message in snapshot:
label = "USER" if message["role"] == "user" else "PRIMARY AGENT"
lines += [f"[{label}]", message["text"], ""]
lines.append("--- End of conversation excerpt ---")
if loaded_skills:
skill_list = ", ".join(loaded_skills)
lines += [
"",
"The primary agent was operating under these loaded skills: "
f"{skill_list}. Before reviewing, load each with "
"skill_view(name=...) and treat their conventions, invariants, "
"and review standards as binding for your assessment — the work "
"was produced under them and must be judged against them.",
]
if user_prompt.strip():
lines += ["", "Additional review instructions from the user:", user_prompt.strip()]
lines += [
"",
"Your review is delivered back into that conversation, addressed to "
"the primary agent and its user. Be direct and specific; do not "
"soften findings.",
]
return goal, "\n".join(lines)
def _load_review_credentials_cfg() -> Optional[Dict[str, Any]]:
"""Read ``auxiliary.review`` into a delegation-credentials-shaped dict.
None when nothing is configured (provider=auto/empty and no model/base_url):
the reviewer then inherits the parent agent's credentials.
"""
try:
from hermes_cli.config import load_config_readonly
review = (load_config_readonly().get("auxiliary") or {}).get("review") or {}
if not isinstance(review, dict):
return None
except Exception:
return None
cfg = {k: str(review.get(k) or "").strip() for k in ("provider", "model", "base_url", "api_key", "api_mode")}
if cfg["provider"].lower() == "auto":
cfg["provider"] = ""
if not (cfg["provider"] or cfg["model"] or cfg["base_url"]):
return None
return cfg
def start_review(
parent_agent,
messages: List[Dict[str, Any]],
user_prompt: str = "",
) -> Dict[str, Any]:
"""Dispatch the reviewer subagent in the background.
Returns the parsed ``delegate_task`` dispatch dict (``status: "dispatched"``
with a ``delegation_id``, or the synchronous result dict on channels that
cannot route async completions). Raises ValueError when there is nothing
to review or the dispatch is rejected/errored.
"""
if parent_agent is None:
raise ValueError("No active agent — send a message first.")
snapshot = snapshot_recent_messages(messages)
if not snapshot:
raise ValueError("Nothing to review yet — the conversation is empty.")
loaded_skills = collect_parent_loaded_skills(parent_agent, messages)
goal, context = build_review_task(snapshot, user_prompt, loaded_skills)
credentials_cfg = _load_review_credentials_cfg()
from tools.delegate_tool import delegate_task
raw = delegate_task(
goal=goal,
context=context,
background=True,
parent_agent=parent_agent,
credentials_cfg=credentials_cfg,
)
try:
result = json.loads(raw)
except Exception:
raise ValueError(f"Review dispatch failed: {raw!r}")
if isinstance(result, dict) and result.get("error"):
raise ValueError(str(result["error"]))
if not isinstance(result, dict):
raise ValueError(f"Review dispatch failed: {raw!r}")
result.setdefault("review_model", (credentials_cfg or {}).get("model") or "")
return result
def format_dispatch_note(result: Dict[str, Any], user_prompt: str = "") -> str:
"""Human-facing one-liner for a successful dispatch. Shared by surfaces."""
model = str(result.get("review_model") or "").strip()
model_note = f" on {model}" if model else ""
focus_note = f" (focus: {user_prompt.strip()})" if user_prompt.strip() else ""
if result.get("status") == "dispatched":
return (
f"⚖ Review subagent dispatched{model_note}{focus_note} — it is "
f"investigating the last {DEFAULT_CONTEXT_MESSAGES} messages in "
f"the background and its full review will re-enter this "
f"conversation when it finishes."
)
# Synchronous fallback (channels that cannot route async completions).
return (
f"⚖ Review completed synchronously{model_note}{focus_note} — "
f"results:\n{json.dumps(result.get('results', result), ensure_ascii=False)[:4000]}"
)