refactor(hermes_cli): group D — collapse doctor/edit/notepad/sweep branches, debug subcommand dispatch, wrap new long lines

This commit is contained in:
Teknium
2026-09-02 21:21:22 -07:00
parent 8f047aced9
commit ffe8b15a3e
5 changed files with 73 additions and 86 deletions

View File

@@ -275,7 +275,7 @@ def _cleanup_stale_runtime_files(profile_dir: Path) -> None:
def _read_prior_exit_label(profile_dir: Path) -> str:
"""Exception-free ``lifecycle_ledger.read_prior_exit_label``: forensics never block cont-init."""
"""Exception-free ``lifecycle_ledger.read_prior_exit_label`` — forensics never block boot."""
try:
from gateway.lifecycle_ledger import read_prior_exit_label
return read_prior_exit_label(profile_dir)

View File

@@ -21,13 +21,10 @@ from cron.lifecycle_guard import ( # noqa: F401 (re-exported for terminal_tool
def _normalize_skills(single_skill=None, skills: Optional[Iterable[str]] = None) -> Optional[List[str]]:
if skills is None:
if single_skill is None:
return None
raw_items = [single_skill]
else:
raw_items = list(skills)
"""Deduped, stripped skill names; None when neither argument was given."""
if skills is None and single_skill is None:
return None
raw_items = list(skills) if skills is not None else [single_skill]
normalized: List[str] = []
for item in raw_items:
text = str(item or "").strip()
@@ -78,7 +75,9 @@ def _builtin_gateway_liveness() -> Optional[bool]:
return True
except Exception:
pass # a crashing lock probe is "unknown", not "dead" — let the pid scan decide
from hermes_cli.gateway import find_gateway_pids, named_profile_served_by_running_multiplexer
from hermes_cli.gateway import (
find_gateway_pids, named_profile_served_by_running_multiplexer,
)
if find_gateway_pids():
return True
@@ -161,7 +160,9 @@ def _print_banner(title: str) -> None:
def _unverified_targets(unverified) -> str:
return ", ".join(str(t) for t in unverified) if isinstance(unverified, list) else str(unverified)
if isinstance(unverified, list):
return ", ".join(str(t) for t in unverified)
return str(unverified)
_STATE_BADGES = {"paused": ("[paused]", Colors.YELLOW), "completed": ("[completed]", Colors.BLUE)}
@@ -217,7 +218,8 @@ def cron_list(show_all: bool = False):
if mon_state.get("last_changed_at"):
rows.append(("Changed", mon_state["last_changed_at"]))
if job.get("no_agent"):
rows.append(("Mode", f"{color('no-agent', Colors.DIM)} (script stdout delivered directly)"))
mode = color("no-agent", Colors.DIM) + " (script stdout delivered directly)"
rows.append(("Mode", mode))
if job.get("workdir"):
rows.append(("Workdir", job["workdir"]))
@@ -410,7 +412,7 @@ def _print_ticker_health(pids: list) -> None:
)
print(" Cron jobs may NOT be firing. Restart: hermes gateway restart")
elif ok_age is not None and ok_age > STALE_AFTER:
# Loop alive (fresh heartbeat) but no tick SUCCEEDED in a long time → failing every iteration.
# Loop alive (fresh heartbeat) but no tick SUCCEEDED in a long time → every tick fails.
_warn(
"⚠ Gateway and cron ticker are running, but no tick has "
f"succeeded in {int(ok_age)}s — ticks may be failing."
@@ -503,7 +505,8 @@ def _print_active_jobs_summary(jobs) -> None:
]
if late:
print()
print(color(f" ⚠ {len(late)} job(s) last fired late (missed-fire catch-up):", Colors.YELLOW))
print(color(f" ⚠ {len(late)} job(s) last fired late (missed-fire catch-up):",
Colors.YELLOW))
for j in late:
d = j["last_dispatch"]
print(
@@ -584,26 +587,20 @@ def _cron_doctor_issues_for_job(job: Dict[str, Any]) -> List[str]:
unverified = job.get("last_delivery_unverified")
if unverified:
issues.append(
f"last delivery unverified (adapter acked without evidence): {_unverified_targets(unverified)}"
)
issues.append("last delivery unverified (adapter acked without evidence): "
+ _unverified_targets(unverified))
if job.get("enabled", True) and job.get("state") not in {"paused", "completed"}:
next_run = str(job.get("next_run_at") or "").strip()
if not next_run:
issues.append("active job has no next_run_at")
else:
overdue = _next_run_overdue_issue(next_run)
if overdue:
issues.append(overdue)
issue = _next_run_overdue_issue(next_run) if next_run else "active job has no next_run_at"
if issue:
issues.append(issue)
script = str(job.get("script") or "").strip()
if job.get("no_agent") and not script:
issues.append("no-agent job has no script")
if script:
script_issue = _script_health_issue(script)
if script_issue:
issues.append(script_issue)
if script and (script_issue := _script_health_issue(script)):
issues.append(script_issue)
workdir = str(job.get("workdir") or "").strip()
if workdir and not Path(workdir).expanduser().exists():
@@ -617,11 +614,7 @@ def cron_doctor() -> int:
from cron.jobs import list_jobs
jobs = list_jobs(include_disabled=False)
findings: List[tuple[Dict[str, Any], List[str]]] = []
for job in jobs:
issues = _cron_doctor_issues_for_job(job)
if issues:
findings.append((job, issues))
findings = [(job, issues) for job in jobs if (issues := _cron_doctor_issues_for_job(job))]
if not findings:
print(color("✓ Cron doctor found no issues", Colors.GREEN))
@@ -635,9 +628,7 @@ def cron_doctor() -> int:
print(color(f"Cron doctor found {issue_count} issue(s) across {len(findings)} job(s):", Colors.YELLOW))
print()
for job, issues in findings:
job_id = job.get("id", "?")
name = job.get("name", "(unnamed)")
print(f" {color(job_id, Colors.YELLOW)} {name}")
print(f" {color(job.get('id', '?'), Colors.YELLOW)} {job.get('name', '(unnamed)')}")
for issue in issues:
print(f" - {issue}")
print()
@@ -724,9 +715,7 @@ def cron_edit(args):
final_skills = replacement_skills
elif add_skills or remove_skills:
final_skills = [skill for skill in existing_skills if skill not in remove_skills]
for skill in add_skills:
if skill not in final_skills:
final_skills.append(skill)
final_skills += [skill for skill in add_skills if skill not in final_skills]
result = _cron_api(action="update", job_id=args.job_id,
schedule=getattr(args, "schedule", None),
@@ -782,14 +771,12 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int:
# (execution_mode="background" and/or a delegation_id) and keeps running AFTER this
# CLI exits — a terminal success/failure verdict would be a lie, so report the dispatch.
delegation_id = job.get("delegation_id")
if job.get("execution_mode") == "background" or delegation_id:
if delegation_id:
print(f" Running in background (delegation {delegation_id}).")
else:
print(" Running in background.")
if delegation_id:
print(f" Running in background (delegation {delegation_id}).")
elif job.get("execution_mode") == "background":
print(" Running in background.")
elif job.get("executed"):
outcome = "succeeded" if job.get("execution_success") else "failed"
print(f" Ran now: {outcome}.")
print(f" Ran now: {'succeeded' if job.get('execution_success') else 'failed'}.")
elif job.get("execution_skipped"):
print(f" {job['execution_skipped']}")
else:
@@ -862,7 +849,6 @@ def cron_notepad(args) -> int:
print(color(f"No notepad key '{key}' for job {job_id}.", Colors.YELLOW))
return 1
# list (default)
notes = notepad.list_notes(job_id)
if not notes:
print(color(f"Notepad for job {job_id} is empty.", Colors.DIM))

View File

@@ -102,7 +102,9 @@ def _scan_dashboard_processes(*, exclude_pids: set[int] | None = None) -> list[t
seen = {pid for pid, _ in found} | skip
for entry in ledger_entries():
pid = entry.get("pid")
if entry.get("purpose") in ("serve", "dashboard") and isinstance(pid, int) and pid not in seen:
if entry.get("purpose") not in ("serve", "dashboard") or not isinstance(pid, int):
continue
if pid not in seen:
found.append((pid, str(entry.get("argv") or "")))
except Exception:
pass # ledger unavailable → scan-only behavior
@@ -540,7 +542,7 @@ def _is_desktop_local_serve_cmdline(command: str) -> bool:
def _process_ppid(pid: int) -> int | None:
"""Best-effort parent pid; None on failure (always on Windows: desktop tree-kill reaps there)."""
"""Best-effort parent pid; None on failure (always None on Windows: desktop tree-kill reaps)."""
try:
if sys.platform == "win32":
return None

View File

@@ -99,7 +99,8 @@ def _register_self_hosted_client(
raise RuntimeError(
detail or "Your account is not permitted to register a self-hosted dashboard."
) from exc
raise RuntimeError(f"Portal returned HTTP {exc.code}" + (f": {detail}" if detail else "")) from exc
suffix = f": {detail}" if detail else ""
raise RuntimeError(f"Portal returned HTTP {exc.code}{suffix}") from exc
except urllib.error.URLError as exc:
raise RuntimeError(f"Could not reach Nous Portal at {portal_base_url}: {exc.reason}") from exc

View File

@@ -63,7 +63,7 @@ def _load_pending() -> list[dict]:
return []
if not isinstance(data, list):
return []
return [e for e in data if isinstance(e, dict) and "url" in e and "expire_at" in e]
return [e for e in data if isinstance(e, dict) and {"url", "expire_at"} <= e.keys()]
def _save_pending(entries: list[dict]) -> None:
@@ -99,18 +99,14 @@ def _sweep_expired_pastes(now: Optional[float] = None) -> tuple[int, int]:
if expire_at > current:
remaining.append(entry)
continue
try:
if delete_paste(entry.get("url", "")):
deleted += 1
continue
gone = delete_paste(entry.get("url", ""))
except Exception:
pass # network hiccup, 404 (already gone), ...
if expire_at + 86400 > current:
remaining.append(entry)
gone = False # network hiccup, 404 (already gone), ...
if gone or expire_at + 86400 <= current:
deleted += 1 # deleted, or given up on → count as reaped
else:
deleted += 1 # count as reaped
remaining.append(entry)
if deleted:
_save_pending(remaining)
@@ -299,7 +295,9 @@ def _redact_log_text(text: str) -> str:
return _EMAIL_ADDRESS_RE.sub("[REDACTED_EMAIL]", text)
def _read_tail_bytes(log_path: Path, size: int, max_bytes: int, tail_lines: int) -> tuple[bytes, bool]:
def _read_tail_bytes(
log_path: Path, size: int, max_bytes: int, tail_lines: int,
) -> tuple[bytes, bool]:
"""Read the whole file, or enough of its tail for both views → (raw, truncated).
For oversized files, read backwards until we have ``max_bytes`` for the standalone upload
@@ -313,7 +311,8 @@ def _read_tail_bytes(log_path: Path, size: int, max_bytes: int, tail_lines: int)
chunks: list[bytes] = []
total = 0
newline_count = 0
while pos > 0 and (total < max_bytes or newline_count <= tail_lines + 1) and total < max_bytes * 2:
while (pos > 0 and total < max_bytes * 2
and (total < max_bytes or newline_count <= tail_lines + 1)):
read_size = min(chunk_size, pos)
pos -= read_size
f.seek(pos)
@@ -384,7 +383,9 @@ def _tail_budget(name: str, log_lines: int) -> int:
return log_lines if name == "agent" else min(log_lines, 100)
def _capture_default_log_snapshots(log_lines: int, *, redact: bool = True) -> dict[str, LogSnapshot]:
def _capture_default_log_snapshots(
log_lines: int, *, redact: bool = True,
) -> dict[str, LogSnapshot]:
"""Capture all logs used by debug-share exactly once."""
return {
name: _capture_log_snapshot(name, tail_lines=_tail_budget(name, log_lines), redact=redact)
@@ -414,7 +415,8 @@ def _capture_dump() -> str:
def collect_debug_report(
*, log_lines: int = 200, dump_text: str = "", log_snapshots: Optional[dict[str, LogSnapshot]] = None,
*, log_lines: int = 200, dump_text: str = "",
log_snapshots: Optional[dict[str, LogSnapshot]] = None,
) -> str:
"""Build the summary debug report (system dump + log tails) as upload-ready text.
@@ -471,7 +473,8 @@ def collect_share_bundle(log_lines: int = 200, redact: bool = True) -> dict[str,
dump_text = _capture_dump()
log_snapshots = _capture_default_log_snapshots(log_lines, redact=redact)
report = collect_debug_report(log_lines=log_lines, dump_text=dump_text, log_snapshots=log_snapshots)
report = collect_debug_report(log_lines=log_lines, dump_text=dump_text,
log_snapshots=log_snapshots)
banner = _REDACTION_BANNER if redact else ""
bundle: dict[str, str] = {"report": banner + report}
for name in _FULL_LOGS:
@@ -506,7 +509,9 @@ class DebugShareResult:
report: str = "" # the summary report text (kept for local fallback)
def build_debug_share(*, log_lines: int = 200, expiry: int = 7, redact: bool = True) -> DebugShareResult:
def build_debug_share(
*, log_lines: int = 200, expiry: int = 7, redact: bool = True,
) -> DebugShareResult:
"""Collect the debug report + full logs, upload each, return the URLs.
Shared core behind ``hermes debug share`` and the dashboard ``POST /api/ops/debug-share``.
@@ -517,7 +522,9 @@ def build_debug_share(*, log_lines: int = 200, expiry: int = 7, redact: bool = T
bundle = collect_share_bundle(log_lines=log_lines, redact=redact)
if redact:
logger.info("hermes debug share: applied force-mode redaction to log snapshots before upload")
logger.info(
"hermes debug share: applied force-mode redaction to log snapshots before upload"
)
report = bundle["report"]
@@ -607,12 +614,9 @@ def run_debug_share(args):
print("\nDebug report uploaded:")
for label, url in result.urls.items():
print(f" {label:<{label_width}} {url}")
if result.failures:
print(f"\n (failed to upload: {', '.join(result.failures)})")
hours = result.auto_delete_seconds // 3600
print(f"\n⏱ Pastes will auto-delete in {hours} hours.")
print(f"\n⏱ Pastes will auto-delete in {result.auto_delete_seconds // 3600} hours.")
print("To delete now: hermes debug delete <url>")
print("\nShare these links with the Hermes team for support.")
@@ -633,7 +637,7 @@ _NOUS_PRIVACY_NOTICE = """\
def _run_debug_share_nous(args, *, log_lines: int, redact: bool) -> None:
"""``hermes debug share --nous``: gzip the same bundle into the Nous envelope, upload to Nous-S3."""
"""``hermes debug share --nous``: gzip the same bundle into the Nous envelope → Nous-S3."""
from hermes_cli.diagnostics_upload import share_to_nous
print(_NOUS_PRIVACY_NOTICE)
@@ -664,17 +668,12 @@ def _run_debug_share_nous(args, *, log_lines: int, redact: bool) -> None:
sys.exit(1)
view_url = res.get("viewUrl") or res.get("view_url")
print("\nDebug bundle uploaded to Nous (private):")
if view_url:
print(f" View URL {view_url}")
else:
print(f" (no view URL returned; upload id: {res.get('id', '?')})")
expires_at = res.get("expiresAt") or res.get("expires_at")
if expires_at:
print(f"\n⏱ Auto-deletes at {expires_at} (14-day retention).")
else:
print("\n⏱ Auto-deletes after 14 days.")
print("\nDebug bundle uploaded to Nous (private):")
print(f" View URL {view_url}" if view_url
else f" (no view URL returned; upload id: {res.get('id', '?')})")
print(f"\n⏱ Auto-deletes at {expires_at} (14-day retention)." if expires_at
else "\n⏱ Auto-deletes after 14 days.")
print(
"\nShare this private link with the Nous team — only Nous staff "
@@ -713,13 +712,12 @@ def run_debug(args):
# Opportunistic sweep of expired pastes on every ``hermes debug`` call.
_best_effort_sweep_expired_pastes()
subcmd = getattr(args, "debug_command", None)
if subcmd == "share":
run_debug_share(args)
elif subcmd == "delete":
run_debug_delete(args)
else:
handlers = {"share": run_debug_share, "delete": run_debug_delete}
handler = handlers.get(getattr(args, "debug_command", None))
if handler is None:
print(_DEBUG_USAGE)
else:
handler(args)
_DEBUG_USAGE = """\