Cluster: agent/{curator,curator_backup,background_review,review_engine,
review_idle_queue,insights,learning_graph,learning_graph_render,
learning_mutations,learn_prompt,verification_evidence,verification_stop,
verify_hooks,side_question,title_generator,turn_summary,
manual_compression_feedback,trajectory,moa_trace,trace_upload,verify/*}.
13662 -> 10693 LOC (-2969, -21.7%), behavior-neutral.
- Dead code: 27 private helpers with zero references removed
(_auto_title_session, _resolve_review_model, _parse_make_targets,
_filter_verifiable_paths, _find_subsequence, _is_under_root/_temp_dir,
_merge_runs, learning_graph_render bucket/period/node helpers,
_memories_dir/_memory_local_index/_node_detail, _cron_jobs_file,
_retention_cutoff, _scope_for_args, _clean_token, _count_diff_lines,
_ordered_verbs, _hermes_meta, _iter_skill_files).
- Unified helpers: _read_config_section (curator + curator_backup),
_write_file/_write_json (4 curator report writers), _msg_text
(background_review <- side_question), _report_failure/_notify_title
(title_generator instant/auto paths), _is_under (verification_evidence),
_scoped SQL pair builder + _query (insights), _optional_lock
(background_review), verify.recipes table-driven detection.
- if/elif routing -> dict dispatch: side_question role labels,
curator_backup summary bits, learning_graph_render buckets, insights
section rendering, verify recipe pickers.
- Redundant defensive layers, single-use wrappers and verbose narrative
comments collapsed; every non-obvious WHY/invariant kept in compact form.
Verification: parity.py (all REMOVED symbols zero-ref), import smoke for
every module + cli/run_agent/gateway.run/hermes_cli.main/
agent.conversation_loop/tui_gateway.server, old-vs-new fuzz parity on all
shared pure functions, SQL trace parity for insights and
verification_evidence, cluster tests 1354 passed / 0 failed (46 files).
231 lines
8.7 KiB
Python
231 lines
8.7 KiB
Python
"""Context-aware side questions (``/btw``).
|
|
|
|
Answers a quick question ABOUT the current conversation without touching it (no
|
|
synthetic turns, no role-alternation risk, no prompt-cache invalidation). Two
|
|
paths, picked automatically:
|
|
|
|
* **Cache-parity fork (preferred).** With a live parent ``AIAgent``, a detached
|
|
fork from :func:`agent.background_review.build_cache_parity_fork` replays the
|
|
parent's snapshot verbatim against the warm prefix cache. Tool calls are denied
|
|
at dispatch, persistence is detached, usage goes to the parent.
|
|
* **One-shot digest (fallback).** With no live parent (e.g. the gateway evicted
|
|
the cached agent), a rendered transcript goes through :func:`agent.oneshot.run_oneshot`.
|
|
|
|
``auxiliary.side_question.provider`` / ``.model`` route the fork to another model
|
|
and replay a compact digest (cold cache on a different model).
|
|
"""
|
|
|
|
import logging
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
from agent.background_review import _msg_text
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Free-form auxiliary task name (auxiliary.side_question.*), main-model-first.
|
|
SIDE_QUESTION_TASK = "side_question"
|
|
|
|
# Fork path: the model may waste an iteration on a (denied) tool call first.
|
|
_FORK_MAX_ITERATIONS = 3
|
|
|
|
# Fallback one-shot path: per-message and total character budgets.
|
|
_PER_MESSAGE_CHAR_CAP = 2000
|
|
_TRANSCRIPT_CHAR_BUDGET = 24000
|
|
|
|
_FORK_PROMPT = (
|
|
"The user asked a quick SIDE question with /btw while the main work "
|
|
"continues in the original session.\n"
|
|
"Rules:\n"
|
|
"- Answer ONLY the side question, using the conversation above as "
|
|
"context. Do not continue, redo, or critique the main task.\n"
|
|
"- Do NOT call any tools — they are disabled for this side question. "
|
|
"Answer directly in text.\n"
|
|
"- If the conversation does not contain enough information to answer, "
|
|
"say so plainly instead of guessing.\n"
|
|
"- Be concise and direct."
|
|
)
|
|
|
|
_ONESHOT_INSTRUCTIONS = (
|
|
"You are the same AI assistant that is currently working inside the "
|
|
"conversation transcribed below. The user has asked a quick SIDE question "
|
|
"with /btw while the main work continues.\n"
|
|
"Rules:\n"
|
|
"- Answer ONLY the side question. Do not continue, redo, or critique the "
|
|
"main task.\n"
|
|
"- Use the transcript as your primary context; it is a snapshot and may "
|
|
"not include the very latest activity.\n"
|
|
"- If the transcript does not contain enough information to answer, say "
|
|
"so plainly instead of guessing.\n"
|
|
"- Be concise and direct."
|
|
)
|
|
|
|
_ROLE_LABELS = {"user": "USER", "assistant": "ASSISTANT", "tool": "TOOL RESULT"}
|
|
|
|
|
|
def trim_snapshot_for_fork(history: Optional[List[Dict[str, Any]]]) -> List[Dict[str, Any]]:
|
|
"""Drop trailing messages until the snapshot ends with a completed assistant text.
|
|
|
|
A mid-turn snapshot can end on unresolved ``tool_calls``, a tool result, or
|
|
the in-flight user message; appending the side question after any of those
|
|
breaks role alternation on strict providers. Trimming only the TAIL keeps
|
|
the warm prefix-cache property.
|
|
"""
|
|
msgs = list(history or [])
|
|
while msgs:
|
|
last = msgs[-1]
|
|
if isinstance(last, dict) and last.get("role") == "assistant" and not last.get("tool_calls"):
|
|
break
|
|
msgs.pop()
|
|
return msgs
|
|
|
|
|
|
def render_history_for_side_question(
|
|
history: Optional[List[Dict[str, Any]]],
|
|
char_budget: int = _TRANSCRIPT_CHAR_BUDGET,
|
|
) -> str:
|
|
"""Render a snapshot as a plain-text transcript (fallback path only).
|
|
|
|
Newest-biased fit to ``char_budget``. Tool calls are summarized by name, tool
|
|
results included truncated (so "what did that output" stays answerable), the
|
|
system prompt skipped.
|
|
"""
|
|
lines: List[str] = []
|
|
for msg in history or []:
|
|
if not isinstance(msg, dict):
|
|
continue
|
|
role = msg.get("role")
|
|
text = _msg_text(msg)
|
|
if role == "assistant" and msg.get("tool_calls"):
|
|
names = [(tc.get("function") or {}).get("name", "?") for tc in msg["tool_calls"] if isinstance(tc, dict)]
|
|
lines.append(f"ASSISTANT [called tools: {', '.join(names)}]")
|
|
label = _ROLE_LABELS.get(role)
|
|
if label and text:
|
|
lines.append(f"{label}: {text[:_PER_MESSAGE_CHAR_CAP]}")
|
|
|
|
kept: List[str] = []
|
|
used = 0
|
|
for line in reversed(lines):
|
|
cost = len(line) + 1
|
|
if used + cost > char_budget and kept:
|
|
break
|
|
kept.append(line)
|
|
used += cost
|
|
kept.reverse()
|
|
|
|
if not kept:
|
|
return "(no prior conversation)"
|
|
prefix = "[...older conversation omitted...]\n" if len(kept) < len(lines) else ""
|
|
return prefix + "\n".join(kept)
|
|
|
|
|
|
def _side_question_task_config() -> Dict[str, Any]:
|
|
"""Return ``auxiliary.side_question`` from config (or ``{}``)."""
|
|
try:
|
|
from hermes_cli.config import load_config_readonly
|
|
|
|
cfg = load_config_readonly()
|
|
except Exception:
|
|
return {}
|
|
aux = cfg.get("auxiliary", {}) if isinstance(cfg.get("auxiliary"), dict) else {}
|
|
task = aux.get(SIDE_QUESTION_TASK, {})
|
|
return task if isinstance(task, dict) else {}
|
|
|
|
|
|
def _answer_via_fork(parent_agent: Any, question: str, history: Optional[List[Dict[str, Any]]]) -> str:
|
|
"""Answer via a cache-parity fork of ``parent_agent`` on the calling thread.
|
|
|
|
The thread-scoped tool whitelist is emptied so any tool call is denied at
|
|
dispatch: ``tools[]`` stays byte-identical for cache parity, but the side
|
|
question can never mutate anything.
|
|
"""
|
|
from agent.background_review import (
|
|
_digest_history,
|
|
_record_review_usage_to_parent,
|
|
_snapshot_review_usage,
|
|
build_cache_parity_fork,
|
|
)
|
|
from hermes_cli.plugins import clear_thread_tool_whitelist, set_thread_tool_whitelist
|
|
|
|
fork, _rt, routed = build_cache_parity_fork(
|
|
parent_agent, _side_question_task_config(),
|
|
max_iterations=_FORK_MAX_ITERATIONS, write_origin="side_question",
|
|
)
|
|
try:
|
|
set_thread_tool_whitelist(
|
|
set(),
|
|
deny_msg_fmt=(
|
|
"Side question (/btw) denied tool call: {tool_name}. "
|
|
"Tools are disabled here — answer directly from the "
|
|
"conversation context."
|
|
),
|
|
)
|
|
snapshot = trim_snapshot_for_fork(history)
|
|
result = fork.run_conversation(
|
|
user_message=f"{_FORK_PROMPT}\n\nSide question: {question}",
|
|
conversation_history=_digest_history(snapshot) if routed else snapshot,
|
|
)
|
|
answer = (result or {}).get("final_response", "") or ""
|
|
if not answer and result and result.get("error"):
|
|
raise RuntimeError(str(result["error"]))
|
|
return answer.strip()
|
|
finally:
|
|
clear_thread_tool_whitelist()
|
|
# Attribute the fork's usage to the parent session; teardown never raises.
|
|
for step in (
|
|
lambda: _record_review_usage_to_parent(parent_agent, _snapshot_review_usage(fork)),
|
|
fork.shutdown_memory_provider,
|
|
fork.close,
|
|
):
|
|
try:
|
|
step()
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _answer_via_oneshot(question: str, history: Optional[List[Dict[str, Any]]], **run_kwargs: Any) -> str:
|
|
"""Fallback: answer from a rendered transcript digest in one aux call."""
|
|
from agent.oneshot import run_oneshot
|
|
|
|
user_input = (
|
|
"Conversation transcript (snapshot):\n"
|
|
"-----\n"
|
|
f"{render_history_for_side_question(history)}\n"
|
|
"-----\n\n"
|
|
f"Side question: {question}"
|
|
)
|
|
return run_oneshot(
|
|
instructions=_ONESHOT_INSTRUCTIONS, user_input=user_input, task=SIDE_QUESTION_TASK, **run_kwargs
|
|
)
|
|
|
|
|
|
def answer_side_question(
|
|
question: str,
|
|
history: Optional[List[Dict[str, Any]]],
|
|
*,
|
|
parent_agent: Any = None,
|
|
main_runtime: Optional[Dict[str, Any]] = None,
|
|
max_tokens: int = 2048,
|
|
temperature: Optional[float] = 0.3,
|
|
timeout: float = 180.0,
|
|
) -> str:
|
|
"""Answer ``question`` against a snapshot of ``history``: cache-parity fork when
|
|
``parent_agent`` is live, else (or on empty answer / failure) the one-shot digest.
|
|
Raises on failure — callers surface the error on their own UI."""
|
|
question = (question or "").strip()
|
|
if not question:
|
|
raise ValueError("answer_side_question requires a non-empty question")
|
|
|
|
if parent_agent is not None:
|
|
try:
|
|
answer = _answer_via_fork(parent_agent, question, history)
|
|
if answer:
|
|
return answer
|
|
logger.warning("/btw fork returned an empty answer; falling back to one-shot")
|
|
except Exception:
|
|
logger.warning("/btw cache-parity fork failed; falling back to one-shot", exc_info=True)
|
|
|
|
return _answer_via_oneshot(
|
|
question, history,
|
|
main_runtime=main_runtime, max_tokens=max_tokens, temperature=temperature, timeout=timeout,
|
|
)
|