From 5de687bc4707c80e1c89f4c0e45f6fe7845eb2ce Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 20:56:52 -0700 Subject: [PATCH] refactor(hermes_cli): unify status-bar width tiers, dedupe session mixin rewind/resume paths, compact docs --- hermes_cli/checkpoints.py | 61 +- hermes_cli/cli_session_mixin.py | 1392 ++++++++++------------------ hermes_cli/cli_status_bar_mixin.py | 1183 ++++++++--------------- 3 files changed, 894 insertions(+), 1742 deletions(-) diff --git a/hermes_cli/checkpoints.py b/hermes_cli/checkpoints.py index 9796b20d6f..f0545a5673 100644 --- a/hermes_cli/checkpoints.py +++ b/hermes_cli/checkpoints.py @@ -37,22 +37,16 @@ def cmd_status(args: argparse.Namespace) -> int: print(f" legacy-* {_fmt_bytes(info['legacy_size_bytes'])}") print(f"Projects: {info['project_count']}") - projects = sorted( - info["projects"], - key=lambda p: (p.get("last_touch") or 0), - reverse=True, - ) + projects = sorted(info["projects"], key=lambda p: (p.get("last_touch") or 0), reverse=True) if projects: print() print(f" {'WORKDIR':<60} {'COMMITS':>7} {'LAST TOUCH':>12} STATE") - for p in projects[: args.limit if hasattr(args, "limit") and args.limit else 20]: + for p in projects[: getattr(args, "limit", None) or 20]: wd = p.get("workdir") or "(unknown)" if len(wd) > 60: wd = "…" + wd[-59:] state = "live" if p.get("exists") else "orphan" - commits = p.get("commits", 0) - last = _fmt_age(p.get("last_touch")) - print(f" {wd:<60} {commits:>7} {last:>12} {state}") + print(f" {wd:<60} {p.get('commits', 0):>7} {_fmt_age(p.get('last_touch')):>12} {state}") legacy = info.get("legacy_archives", []) if legacy: @@ -76,22 +70,15 @@ def cmd_prune(args: argparse.Namespace) -> int: max_size_mb = args.max_size_mb delete_orphans = not args.keep_orphans - # When set, restricts orphan deletion to exactly the identities shown in - # the confirmation preview below (v2 project hashes / pre-v2 shadow repo - # paths). `None` means "no restriction" — used for --force, where there - # is no preview to bind to. + # Restricts orphan deletion to exactly the identities shown in the confirmation preview + # (v2 project hashes / pre-v2 shadow repo paths). `None` = no restriction (--force: no + # preview to bind to). orphan_allowlist: Optional[set] = None if delete_orphans and not args.force: info = store_status() - orphans = [ - p for p in info.get("projects", []) - if not p.get("exists") - ] - pre_v2_orphans = [ - p for p in info.get("pre_v2_projects", []) - if not p.get("exists") - ] + orphans = [p for p in info.get("projects", []) if not p.get("exists")] + pre_v2_orphans = [p for p in info.get("pre_v2_projects", []) if not p.get("exists")] if orphans or pre_v2_orphans: print(f"This will permanently delete {len(orphans) + len(pre_v2_orphans)} " "orphan checkpoint project(s) whose workdir is not currently reachable:") @@ -107,13 +94,10 @@ def cmd_prune(args: argparse.Namespace) -> int: if not _confirm("Delete these orphan projects?"): print("Aborted.") return 1 - # Bind the deletion to exactly what was just displayed (and, when - # non-empty, confirmed) — a project that becomes orphaned only - # *after* this preview (e.g. its workdir disappears while waiting on - # input()) must not be swept up under this same run. This is set - # unconditionally for every non-force run: an EMPTY preview binds to - # an EMPTY allowlist, so a zero-orphan preview can never authorize - # deletion of orphans discovered by the later rescan. + # Bind deletion to exactly what was displayed/confirmed: a project that goes orphan + # only *after* the preview (workdir vanishes while waiting on input()) must not be + # swept up. Set unconditionally for every non-force run — an EMPTY preview binds an + # EMPTY allowlist, so it can never authorize orphans found by the later rescan. orphan_allowlist = {p["hash"] for p in orphans} orphan_allowlist.update(p["path"] for p in pre_v2_orphans) @@ -208,25 +192,15 @@ def register_cli(parser: argparse.ArgumentParser) -> None: parser.set_defaults(func=cmd_status) # bare `hermes checkpoints` → status subs = parser.add_subparsers(dest="checkpoints_command", metavar="COMMAND") - p_status = subs.add_parser( - "status", - help="Show total size, project count, and per-project breakdown", - ) - p_status.add_argument("--limit", type=int, default=20, - help="Max projects to list (default 20)") + p_status = subs.add_parser("status", help="Show total size, project count, and per-project breakdown") + p_status.add_argument("--limit", type=int, default=20, help="Max projects to list (default 20)") p_status.set_defaults(func=cmd_status) - p_list = subs.add_parser( - "list", - help="Alias for 'status'", - ) + p_list = subs.add_parser("list", help="Alias for 'status'") p_list.add_argument("--limit", type=int, default=20) p_list.set_defaults(func=cmd_status) - p_prune = subs.add_parser( - "prune", - help="Delete orphan/stale checkpoints and GC the store", - ) + p_prune = subs.add_parser("prune", help="Delete orphan/stale checkpoints and GC the store") p_prune.add_argument("--retention-days", type=int, default=7, help="Drop projects whose last_touch is older than N days (default 7)") p_prune.add_argument("--max-size-mb", type=int, default=500, @@ -243,6 +217,5 @@ def register_cli(parser: argparse.ArgumentParser) -> None: ("clear-legacy", "Delete only the legacy-/ archives from v1 migration", cmd_clear_legacy), ): p_clear = subs.add_parser(name, help=help_text) - p_clear.add_argument("-f", "--force", action="store_true", - help="Skip confirmation prompt") + p_clear.add_argument("-f", "--force", action="store_true", help="Skip confirmation prompt") p_clear.set_defaults(func=func) diff --git a/hermes_cli/cli_session_mixin.py b/hermes_cli/cli_session_mixin.py index 4fd1acf21b..bc070c0e7d 100644 --- a/hermes_cli/cli_session_mixin.py +++ b/hermes_cli/cli_session_mixin.py @@ -1,8 +1,8 @@ -"""Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo persistence, manual compression, and exit summary +"""Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo +persistence, manual compression, and exit summary. -Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal -symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin -never imports ``cli`` at module load time (import cycle). +Mixin bound onto ``HermesCLI`` via the MRO. cli.py-internal symbols are imported LAZILY +inside each method (``from cli import ...``) — never at module load time (import cycle). """ from __future__ import annotations @@ -19,23 +19,126 @@ from rich.markup import escape as _escape from typing import Any, Dict, List, Optional +def _user_turn_indices(history: list) -> list[int]: + """Indices of *real* user turns: excludes ephemeral scaffolding, display_kind timeline + rows and compaction handoffs — the same predicate as resume turn counting.""" + from agent.context_compressor import user_originated_turn_view + from run_agent import _is_ephemeral_scaffolding + + return [ + i for i, m in enumerate(history) + if not _is_ephemeral_scaffolding(m) and user_originated_turn_view(m) is not None + ] + + +def _timestamp_or(value, default): + """``datetime.fromtimestamp(value)`` or *default* when the value is missing/unparseable.""" + from cli import datetime + + if not value: + return default + try: + return datetime.fromtimestamp(float(value)) + except Exception: + return default + + +def _dim_notice(cli, msg: str, quiet: bool) -> None: + """Print a dim notice — raw to stderr on quiet (pre-TUI) paths. Module-level so tests can + call the resume helpers unbound against a minimal stand-in.""" + if quiet: + print(msg, file=sys.stderr) + else: + cli._console_print(f"[dim]{_escape(msg)}[/dim]") + + +def _reset_model_to_config_default(cli, silent: bool) -> None: + """/new is a full boundary: re-derive model/provider from config.yaml so a + session-only ``/model --session`` switch never leaks into the next session. + Best-effort — an unreachable default must never block /new. Module-level helper (like + ``_apply_new_session_title``): tests drive ``new_session`` unbound on a SimpleNamespace.""" + from cli import CLI_CONFIG, _cprint, _split_model_config_default, logger + _model_config = CLI_CONFIG.get("model", {}) + if isinstance(_model_config, dict): + _raw_default = _model_config.get("default") or _model_config.get("model") or "" + _config_provider = _model_config.get("provider", "") + else: + _raw_default, _config_provider = (_model_config or ""), "" + _config_model, _ = _split_model_config_default(_raw_default) + if not _config_model or _config_model == getattr(cli, "model", None): + return + try: + from hermes_cli.model_switch import switch_model as _switch_model + + r = _switch_model( + raw_input=_config_model, + current_provider=cli.provider or "", + current_model=cli.model or "", + current_base_url=cli.base_url or "", + current_api_key=cli.api_key or "", + is_global=False, + explicit_provider=_config_provider or "", + ) + if not r.success: + return + if cli.agent: + cli.agent.switch_model( + new_model=r.new_model, new_provider=r.target_provider, api_key=r.api_key, + base_url=r.base_url, api_mode=r.api_mode, + capabilities=getattr(r, "runtime_capabilities", None), + ) + cli.model = r.new_model + cli.provider = r.target_provider + cli.requested_provider = r.target_provider + cli._explicit_api_key = r.api_key + cli._explicit_base_url = r.base_url + if r.api_key: + cli.api_key = r.api_key + if r.base_url: + cli.base_url = r.base_url + if r.api_mode: + cli.api_mode = r.api_mode + if not silent: + _cprint(f" (model reset to config default: {r.new_model})") + except Exception: + logger.debug("/new model reset to config default failed", exc_info=True) + + +def _apply_new_session_title(cli, title: str) -> Optional[str]: + """Sanitize + persist a /new title; returns the stored title or None (untitled).""" + from cli import _cprint + from hermes_state import SessionDB + try: + sanitized = SessionDB.sanitize_title(title) + except ValueError as e: + _cprint(f" Title rejected: {e}") + return None + if not sanitized: + _cprint(" Title is empty after cleanup — session started untitled.") + return None + try: + cli._session_db.set_session_title(cli.session_id, sanitized) + except ValueError as e: + _cprint(f" {e} — session started untitled.") + return None + except Exception: + return None + cli._pending_title = None + cli._status_bar_title_checked_at = 0.0 + return sanitized + + class CLISessionMixin: - """Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo persistence, manual compression, and exit summary""" + """Session lifecycle for the interactive CLI: new/resume/save, undo/retry rewinds, yolo + persistence, manual compression, and exit summary.""" def _restore_session_cwd(self, session_meta: dict, *, quiet: bool = False) -> None: """Relaunch a resumed session in the directory it was started from. - Idempotent and safe to call from every resume path. When the stored - ``cwd`` differs from the current process directory, we both - ``os.chdir()`` (so the process and any ``os.getcwd()`` fallback agree) - and retarget ``TERMINAL_CWD`` (so the terminal tool, code-exec tool, - and relative-path resolution all land in the same place — the local - terminal backend snapshots cwd on first use, which happens after this). - - No-ops when: the session recorded no cwd (gateway/remote/older - sessions), the directory no longer exists, or we're already there. - A missing directory degrades to a single dim warning rather than a - crash — repos get moved and deleted. + Idempotent; called from every resume path. Both ``os.chdir()`` and ``TERMINAL_CWD`` + are retargeted so the process and the terminal/code-exec tools agree (the local + terminal backend snapshots cwd on first use, after this). No-op when no cwd was + recorded, the directory is gone (dim warning, never a crash), or we're already there. """ recorded = (session_meta or {}).get("cwd") if not recorded: @@ -46,72 +149,44 @@ class CLISessionMixin: except OSError: current = None if current and os.path.realpath(recorded) == os.path.realpath(current): - return # Already where the session lived — nothing to announce. - - if not os.path.isdir(recorded): - msg = f"⚠ Session's working directory is gone: {recorded} — staying in {current or '.'}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") return - + if not os.path.isdir(recorded): + _dim_notice(self, + f"⚠ Session's working directory is gone: {recorded} — staying in {current or '.'}", + quiet, + ) + return try: os.chdir(recorded) except OSError as e: - msg = f"⚠ Could not enter session's working directory {recorded}: {e}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") + _dim_notice( + self, f"⚠ Could not enter session's working directory {recorded}: {e}", quiet + ) return - - # Retarget the terminal/code-exec tools to match the process cwd. os.environ["TERMINAL_CWD"] = recorded - - msg = f"↻ Working directory: {recorded}" - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") + _dim_notice(self, f"↻ Working directory: {recorded}", quiet) def _restore_session_yolo(self, session_meta: dict, *, quiet: bool = False) -> None: - """Re-enable YOLO bypass on resume when the session had it on. - - Companion to ``_restore_session_cwd`` — called from every resume path - (startup ``--resume``/``-c`` and mid-chat ``/resume``). The persisted - flag lives in the session row's ``model_config.yolo_mode`` (written by - ``/yolo`` toggles and ``--yolo`` launches); without this restore the - in-memory ``tools.approval._session_yolo`` set starts empty in a fresh - process and the user's bypass silently reverts. - - No-op when the flag is absent/false, when YOLO is already active for - this session (idempotent across repeated resume paths), or when the - process was itself launched with ``--yolo`` (frozen bypass already - covers everything). - """ + """Re-enable YOLO bypass on resume when the session row's ``model_config.yolo_mode`` + says so — the in-memory ``tools.approval._session_yolo`` set starts empty in a fresh + process. No-op when already active or when the process was launched with ``--yolo``.""" try: from hermes_state import SessionDB from tools.approval import ( - _YOLO_MODE_FROZEN, - enable_session_yolo, - is_session_yolo_enabled, + _YOLO_MODE_FROZEN, enable_session_yolo, is_session_yolo_enabled, ) except Exception: return - if _YOLO_MODE_FROZEN: - return - if not SessionDB.session_yolo_enabled(session_meta): + if _YOLO_MODE_FROZEN or not SessionDB.session_yolo_enabled(session_meta): return session_key = self.session_id or "default" if is_session_yolo_enabled(session_key): return enable_session_yolo(session_key) - msg = "⚡ YOLO mode restored from session — all commands auto-approved. /yolo to turn off." - if quiet: - print(msg, file=sys.stderr) - else: - self._console_print(f"[dim]{_escape(msg)}[/dim]") + _dim_notice(self, + "⚡ YOLO mode restored from session — all commands auto-approved. /yolo to turn off.", + quiet, + ) def _render_resume_history_panel_lines(self, panel) -> list[str]: """Render the resume panel at the current terminal width for resize replay.""" @@ -119,30 +194,24 @@ class CLISessionMixin: from io import StringIO buf = StringIO() - width = shutil.get_terminal_size((80, 24)).columns console = Console( - file=buf, - force_terminal=True, - color_system="truecolor", - highlight=False, - width=width, + file=buf, force_terminal=True, color_system="truecolor", highlight=False, + width=shutil.get_terminal_size((80, 24)).columns, ) with _suspend_output_history(): console.print(panel) return buf.getvalue().rstrip("\n").splitlines() def _resolve_checkpoint_ref(self, ref: str, checkpoints: list) -> str | None: - """Resolve a checkpoint number or hash to a full commit hash.""" + """Resolve a 1-indexed checkpoint number (or pass through a git hash).""" try: - idx = int(ref) - 1 # 1-indexed for user - if 0 <= idx < len(checkpoints): - return checkpoints[idx]["hash"] - else: - print(f" Invalid checkpoint number. Use 1-{len(checkpoints)}.") - return None + idx = int(ref) - 1 except ValueError: - # Treat as a git hash return ref + if 0 <= idx < len(checkpoints): + return checkpoints[idx]["hash"] + print(f" Invalid checkpoint number. Use 1-{len(checkpoints)}.") + return None def _show_status(self): """Show compact startup status line.""" @@ -152,18 +221,12 @@ class CLISessionMixin: tool_status = "tools deferred" else: tools = get_tool_definitions(enabled_toolsets=self.enabled_toolsets, quiet_mode=True) - tool_count = len(tools) if tools else 0 - tool_status = f"{tool_count} tools" + tool_status = f"{len(tools) if tools else 0} tools" - # Format model name (shorten if needed) model_short = self.model.split("/")[-1] if "/" in self.model else self.model if len(model_short) > 30: model_short = model_short[:27] + "..." - - # Get API status indicator api_indicator = "[green bold]●[/]" if self.api_key else "[red bold]●[/]" - - # Build status line with proper markup — skin-aware colors try: from hermes_cli.skin_engine import get_active_skin skin = get_active_skin() @@ -172,23 +235,21 @@ class CLISessionMixin: label_color = skin.get_color("ui_label", "#DAA520") except Exception: separator_color, accent_color, label_color = "#B8860B", "#FFBF00", "cyan" + sep = f" [dim {separator_color}]·[/] " toolsets_info = "" if self.enabled_toolsets and "all" not in self.enabled_toolsets: - toolsets_info = f" [dim {separator_color}]·[/] [{label_color}]toolsets: {', '.join(self.enabled_toolsets)}[/]" - - provider_info = f" [dim {separator_color}]·[/] [dim]provider: {self.provider}[/]" + toolsets_info = f"{sep}[{label_color}]toolsets: {', '.join(self.enabled_toolsets)}[/]" + provider_info = f"{sep}[dim]provider: {self.provider}[/]" if self._provider_source: - provider_info += f" [dim {separator_color}]·[/] [dim]auth: {self._provider_source}[/]" - + provider_info += f"{sep}[dim]auth: {self._provider_source}[/]" self._console_print( - f" {api_indicator} [{accent_color}]{model_short}[/] " - f"[dim {separator_color}]·[/] [bold {label_color}]{tool_status}[/]" - f"{toolsets_info}{provider_info}" + f" {api_indicator} [{accent_color}]{model_short}[/]{sep}" + f"[bold {label_color}]{tool_status}[/]{toolsets_info}{provider_info}" ) def _show_session_status(self): """Show gateway-style status for the current CLI session.""" - from cli import datetime, display_hermes_home + from cli import display_hermes_home session_meta = {} if self._session_db: try: @@ -197,25 +258,13 @@ class CLISessionMixin: session_meta = {} title = (session_meta.get("title") or "").strip() - - created_at = self.session_start - started_at = session_meta.get("started_at") - if started_at: - try: - created_at = datetime.fromtimestamp(float(started_at)) - except Exception: - created_at = self.session_start - + created_at = _timestamp_or(session_meta.get("started_at"), self.session_start) updated_at = created_at for field in ("updated_at", "last_updated_at", "last_activity_at"): - value = session_meta.get(field) - if not value: - continue - try: - updated_at = datetime.fromtimestamp(float(value)) + candidate = _timestamp_or(session_meta.get(field), None) + if candidate is not None: + updated_at = candidate break - except Exception: - pass agent = getattr(self, "agent", None) total_tokens = getattr(agent, "session_total_tokens", 0) or 0 @@ -223,7 +272,6 @@ class CLISessionMixin: model = getattr(self, "model", None) or "(unknown)" is_running = bool(getattr(self, "_agent_running", False)) - # Reasoning level (C-02): resolve the effective effort for display. reasoning_label = None try: rc = getattr(agent, "reasoning_config", None) or getattr(self, "reasoning_config", None) @@ -233,12 +281,11 @@ class CLISessionMixin: elif rc.get("effort"): reasoning_label = str(rc.get("effort")) show_r = getattr(self, "show_reasoning", None) - if reasoning_label: - reasoning_label += f" (display: {'on' if show_r else 'off'})" if show_r is not None else "" + if reasoning_label and show_r is not None: + reasoning_label += f" (display: {'on' if show_r else 'off'})" except Exception: reasoning_label = None - # Approval mode (C-02). approval_label = None try: from tools.approval import _get_approval_mode, is_approval_bypass_active_for_session @@ -251,37 +298,30 @@ class CLISessionMixin: except Exception: approval_label = None - # Context window usage (C-02): reuse the status-bar snapshot which - # already computes tokens / max / percent. + # Context window usage: reuse the status-bar snapshot (tokens / max / percent). ctx_label = None try: snap = self._get_status_bar_snapshot() - ctx_tokens = snap.get("context_tokens") or 0 ctx_max = snap.get("context_length") ctx_pct = snap.get("context_percent") if ctx_max: left = "" if isinstance(ctx_pct, (int, float)): left = f"{max(0, 100 - int(ctx_pct))}% left · " - ctx_label = f"{left}{ctx_tokens:,} / {ctx_max:,} tokens used" + ctx_label = f"{left}{snap.get('context_tokens') or 0:,} / {ctx_max:,} tokens used" except Exception: ctx_label = None lines = [ - "Hermes CLI Status", - "", - f"Session ID: {self.session_id}", - f"Path: {display_hermes_home()}", + "Hermes CLI Status", "", f"Session ID: {self.session_id}", f"Path: {display_hermes_home()}", ] if title: lines.append(f"Title: {title}") lines.append(f"Model: {model} ({provider})") - if reasoning_label: - lines.append(f"Reasoning: {reasoning_label}") - if approval_label: - lines.append(f"Approvals: {approval_label}") - if ctx_label: - lines.append(f"Context: {ctx_label}") + optional = (("Reasoning", reasoning_label), ("Approvals", approval_label), ("Context", ctx_label)) + for label, value in optional: + if value: + lines.append(f"{label}: {value}") lines.extend([ f"Created: {created_at.strftime('%Y-%m-%d %H:%M')}", f"Last Activity: {updated_at.strftime('%Y-%m-%d %H:%M')}", @@ -298,12 +338,8 @@ class CLISessionMixin: from hermes_cli.session_listing import query_session_listing return query_session_listing( - self._session_db, - source="cli", - current_session_id=self.session_id, - include_all_sources=False, - include_unnamed=True, - limit=limit, + self._session_db, source="cli", current_session_id=self.session_id, + include_all_sources=False, include_unnamed=True, limit=limit, exclude_sources=["kanban", "tool"], ) except Exception: @@ -354,12 +390,9 @@ class CLISessionMixin: show_ts = bool(getattr(self, "show_timestamps", False)) def _ts_suffix(message: dict) -> str: - # Messages restored from SessionDB carry a unix `timestamp`; live - # unsaved turns may not. Only annotate when both the toggle is on - # and the turn actually has a stored time — never fabricate one. - if not show_ts: - return "" - ts = message.get("timestamp") + # Only annotate when the toggle is on AND the turn has a stored unix + # `timestamp` (SessionDB-restored rows do; live turns may not) — never fabricate. + ts = message.get("timestamp") if show_ts else None if not ts: return "" try: @@ -372,7 +405,6 @@ class CLISessionMixin: nonlocal hidden_tool_messages if not hidden_tool_messages: return - noun = "message" if hidden_tool_messages == 1 else "messages" _cli_visible_print("\n [Tools]") _cli_visible_print(f" ({hidden_tool_messages} tool {noun} hidden)") @@ -385,62 +417,47 @@ class CLISessionMixin: for msg in self.conversation_history: role = msg.get("role", "unknown") - if role == "tool": hidden_tool_messages += 1 continue - if role not in {"user", "assistant"}: continue - flush_tool_summary() visible_index += 1 content = msg.get("content") content_text = "" if content is None else str(content) - + preview = content_text[:preview_limit] + suffix = "..." if len(content_text) > preview_limit else "" if role == "user": _cli_visible_print(f"\n [You #{visible_index}]{_ts_suffix(msg)}") - _cli_visible_print( - f" {content_text[:preview_limit]}{'...' if len(content_text) > preview_limit else ''}" - ) + _cli_visible_print(f" {preview}{suffix}") continue _cli_visible_print(f"\n [Hermes #{visible_index}]{_ts_suffix(msg)}") tool_calls = msg.get("tool_calls") or [] - if content_text: - preview = content_text[:preview_limit] - suffix = "..." if len(content_text) > preview_limit else "" - elif tool_calls: - tool_count = len(tool_calls) - noun = "call" if tool_count == 1 else "calls" - preview = f"(requested {tool_count} tool {noun})" - suffix = "" - else: - preview = "(no text response)" + if not content_text: suffix = "" + if tool_calls: + noun = "call" if len(tool_calls) == 1 else "calls" + preview = f"(requested {len(tool_calls)} tool {noun})" + else: + preview = "(no text response)" _cli_visible_print(f" {preview}{suffix}") flush_tool_summary() _cli_visible_print() def _notify_session_boundary(self, event_type: str) -> None: - """Fire a session-boundary plugin hook (on_session_finalize or on_session_reset). - - Non-blocking — errors are caught and logged. Safe to call from any - lifecycle point (shutdown, /new, /reset). - """ + """Fire a session-boundary plugin hook (on_session_finalize / on_session_reset). + Non-blocking; errors swallowed. Safe from shutdown, /new, /reset.""" try: from hermes_cli.lifecycle import finalize_session, invoke_hook context = { "session_id": self.agent.session_id if self.agent else None, "platform": getattr(self, "platform", None) or "cli", - "reason": ( - "new_session" - if event_type == "on_session_reset" - else "session_boundary" - ), + "reason": "new_session" if event_type == "on_session_reset" else "session_boundary", } if event_type == "on_session_finalize": finalize_session(**context) @@ -450,22 +467,14 @@ class CLISessionMixin: pass def _discard_session_if_empty(self, session_id: Optional[str]) -> bool: - """Drop a just-ended session row when it never gained content. - - Starting the CLI and immediately quitting (or rotating with /new, - /clear) used to leave an empty untitled row behind that clutters - ``/resume`` and ``hermes sessions list``. Delegates the - check-and-delete to ``SessionDB.delete_session_if_empty``, which - only removes rows with no messages, no title, and no child - sessions. Ported from google-gemini/gemini-cli#27770. - """ + """Drop a just-ended session row that never gained content (quit-immediately, /new, + /clear) so it doesn't clutter ``/resume``. ``SessionDB.delete_session_if_empty`` only + removes rows with no messages, no title and no children (gemini-cli#27770 port).""" from cli import logger if not self._session_db or not session_id: return False - # In-memory transcript is authoritative: if this CLI object holds - # conversation messages (flushed to the DB or not), the session is - # not empty. Protects against pruning a real conversation whose DB - # flush failed or hasn't happened yet. + # In-memory transcript is authoritative: a real conversation whose DB flush failed + # or hasn't happened yet must never be pruned. if getattr(self, "conversation_history", None): return False try: @@ -474,53 +483,34 @@ class CLISessionMixin: session_id, sessions_dir=_ghh() / "sessions" ) except Exception: - logger.debug( - "Could not prune empty session %s", session_id, exc_info=True - ) + logger.debug("Could not prune empty session %s", session_id, exc_info=True) return False def _launch_session_boundary_memory_flush( - self, - history_snapshot: list, - *, - session_id: Optional[str] = None, + self, history_snapshot: list, *, session_id: Optional[str] = None, ) -> Optional[list]: """Stage old-session memory extraction so /new stays responsive. - The context-engine ``on_session_end`` boundary is delivered - synchronously here: it is cheap (local state clear, no LLM call) and - ordering-sensitive — it must land before ``reset_session_state()`` - rebinds the engine to the new session. + The context-engine ``on_session_end`` is delivered synchronously here: cheap (no LLM) + and ordering-sensitive — it must land before ``reset_session_state()`` rebinds the + engine. The memory-provider half (LLM-bound, seconds) is NOT run here: the returned + snapshot goes to ``MemoryManager.commit_session_boundary_async`` as one end→switch + task on the serialized worker, so a late ``on_session_end`` can never run after + ``on_session_switch`` and misattribute the old transcript to the new session. - The memory-provider half (LLM-bound extraction, seconds) is NOT run - here. The returned snapshot is handed by ``new_session()`` to - ``MemoryManager.commit_session_boundary_async`` as a single - end→switch task on the manager's serialized background worker, so - extraction can never race the provider rebinding (providers key off - internal ``_session_id`` state — a late ``on_session_end`` after - ``on_session_switch`` would misattribute the old transcript to the - new session). - - Returns the history snapshot to queue, or ``None`` when there is - nothing to extract (no agent / empty history / no memory manager). + Returns the snapshot to queue, or ``None`` when there is nothing to extract. """ from cli import logger agent = getattr(self, "agent", None) if not agent or not history_snapshot: return None - engine = getattr(agent, "context_compressor", None) if engine is not None and hasattr(engine, "on_session_end"): try: engine.on_session_end(session_id or "", history_snapshot) except Exception: - logger.debug( - "Context engine on_session_end failed at /new boundary", - exc_info=True, - ) - - # No provider extraction to queue when no memory manager is - # configured — new_session() falls back to the inline switch path. + logger.debug("Context engine on_session_end failed at /new boundary", exc_info=True) + # No memory manager → new_session() falls back to the inline switch path. if getattr(agent, "_memory_manager", None) is None: return None return history_snapshot @@ -528,130 +518,52 @@ class CLISessionMixin: def new_session(self, silent=False, title=None): """Start a fresh session with a new session ID and cleared agent state.""" from cli import ( - CLI_CONFIG, - _cprint, - _parse_reasoning_config, - _parse_service_tier_config, - _split_model_config_default, - _sync_process_session_id, - datetime, - logger, + CLI_CONFIG, _parse_reasoning_config, _parse_service_tier_config, + _sync_process_session_id, datetime, ) old_session_id = self.session_id _boundary_snapshot = None - if self.agent and self.conversation_history: - # Deliver the context-engine boundary synchronously and get back - # the history snapshot for the deferred provider extraction — - # queued below (after rotation) so /new never blocks on the - # LLM-bound extraction call. - _boundary_snapshot = self._launch_session_boundary_memory_flush( - list(self.conversation_history), - session_id=old_session_id, - ) - self._notify_session_boundary("on_session_finalize") - elif self.agent: - # First session or empty history — still finalize the old session + if self.agent: + if self.conversation_history: + # Context-engine boundary now; provider extraction is queued below (after + # rotation) so /new never blocks on the LLM-bound call. + _boundary_snapshot = self._launch_session_boundary_memory_flush( + list(self.conversation_history), session_id=old_session_id, + ) self._notify_session_boundary("on_session_finalize") if self._session_db and old_session_id: - # Flush any un-persisted messages from the current turn to the - # old session *before* rotating. /new can be called mid-turn - # when _flush_messages_to_session_db() has not yet run — without - # this, messages generated during the current turn are silently - # lost on session rotation (#47202). + # /new can arrive mid-turn before _flush_messages_to_session_db() ran — flush + # the current turn to the OLD session before rotating or it is silently lost. if self.agent: try: self.agent._flush_messages_to_session_db( - self.conversation_history, - conversation_history=self.conversation_history, + self.conversation_history, conversation_history=self.conversation_history, ) except Exception: - pass # best-effort + pass try: self._session_db.end_session(old_session_id, "new_session") except Exception: pass - # Don't let immediately-rotated empty sessions pile up in - # /resume and `hermes sessions list` (gemini-cli#27770 port). self._discard_session_if_empty(old_session_id) self.session_start = datetime.now() - timestamp_str = self.session_start.strftime("%Y%m%d_%H%M%S") - short_uuid = uuid.uuid4().hex[:6] - self.session_id = f"{timestamp_str}_{short_uuid}" + self.session_id = f"{self.session_start.strftime('%Y%m%d_%H%M%S')}_{uuid.uuid4().hex[:6]}" + # getattr: tests drive new_session unbound against a SimpleNamespace stand-in. getattr(self, "_write_terminal_breadcrumb", lambda: None)() self.conversation_history = [] self._pending_title = None self._resumed = False - # /new clears the -m / --model override flag: an explicit CLI model - # was for the previous session only, not for every session spawned - # afterwards. + # An explicit -m/--model was for the previous session only. self._explicit_model_override = False self.reasoning_config = _parse_reasoning_config( CLI_CONFIG["agent"].get("reasoning_effort", "") ) - # /new is a full conversation boundary: session-scoped runtime - # overrides (/model --session, /fast, one-turn restores) do not carry - # forward. Re-derive model/provider and service tier from config.yaml - # so a session-only switch never leaks into the next session (#48055, - # #23131). + # Session-scoped overrides (/model --session, /fast, one-turn restores) don't carry over. self._pending_one_turn_model_restore = None - self.service_tier = _parse_service_tier_config( - CLI_CONFIG["agent"].get("service_tier", "") - ) - _model_config = CLI_CONFIG.get("model", {}) - _raw_default2 = (_model_config.get("default") or _model_config.get("model") or "") if isinstance(_model_config, dict) else (_model_config or "") - _config_model, _ = _split_model_config_default(_raw_default2) - if _config_model and _config_model != getattr(self, "model", None): - _config_provider = ( - _model_config.get("provider", "") - if isinstance(_model_config, dict) - else "" - ) - try: - from hermes_cli.model_switch import switch_model as _switch_model - - _reset_result = _switch_model( - raw_input=_config_model, - current_provider=self.provider or "", - current_model=self.model or "", - current_base_url=self.base_url or "", - current_api_key=self.api_key or "", - is_global=False, - explicit_provider=_config_provider or "", - ) - if _reset_result.success: - if self.agent: - self.agent.switch_model( - new_model=_reset_result.new_model, - new_provider=_reset_result.target_provider, - api_key=_reset_result.api_key, - base_url=_reset_result.base_url, - api_mode=_reset_result.api_mode, - capabilities=getattr( - _reset_result, "runtime_capabilities", None - ), - ) - self.model = _reset_result.new_model - self.provider = _reset_result.target_provider - self.requested_provider = _reset_result.target_provider - self._explicit_api_key = _reset_result.api_key - self._explicit_base_url = _reset_result.base_url - if _reset_result.api_key: - self.api_key = _reset_result.api_key - if _reset_result.base_url: - self.base_url = _reset_result.base_url - if _reset_result.api_mode: - self.api_mode = _reset_result.api_mode - if not silent: - _cprint( - f" (model reset to config default: " - f"{_reset_result.new_model})" - ) - except Exception: - # Best-effort: an unreachable config default must never block - # /new. The session keeps the current working model. - logger.debug("/new model reset to config default failed", exc_info=True) + self.service_tier = _parse_service_tier_config(CLI_CONFIG["agent"].get("service_tier", "")) + _reset_model_to_config_default(self, silent) _sync_process_session_id(self.session_id) if self.agent: @@ -678,63 +590,30 @@ class CLISessionMixin: source=os.environ.get("HERMES_SESSION_SOURCE", "cli"), model=self.model, model_config={ - "max_iterations": self.max_turns, - "reasoning_config": self.reasoning_config, + "max_iterations": self.max_turns, "reasoning_config": self.reasoning_config, }, ) self.agent._session_db_created = True except Exception: pass - if title and self._session_db: - from hermes_state import SessionDB - try: - sanitized = SessionDB.sanitize_title(title) - except ValueError as e: - _cprint(f" Title rejected: {e}") - sanitized = None - title = None - if sanitized: - try: - self._session_db.set_session_title(self.session_id, sanitized) - self._pending_title = None - self._status_bar_title_checked_at = 0.0 - title = sanitized - except ValueError as e: - _cprint(f" {e} — session started untitled.") - title = None - except Exception: - title = None - elif title is not None: - # sanitize_title returned empty (whitespace-only / unprintable) - _cprint(" Title is empty after cleanup — session started untitled.") - title = None - # Notify memory providers that session_id rotated to a fresh - # conversation. reset=True signals providers to flush accumulated - # per-session state (_session_turns, _turn_counter, _document_id). - # Fires BEFORE the plugin on_session_reset hook (shell hooks only - # see the new id; Python providers see the transition). See #6672. - # - # When the old session has history, end-of-session extraction - # (LLM-bound, seconds) and this switch are queued as ONE task on - # the memory manager's serialized worker — end strictly before - # switch, without blocking /new (#16454). With no history there - # is nothing to extract; switch inline as before. + if title: + title = _apply_new_session_title(self, title) + # Tell memory providers the session_id rotated (reset=True flushes per-session + # state) BEFORE the plugin on_session_reset hook. With old history, end-of-session + # extraction and this switch are queued as ONE task on the serialized worker — + # end strictly before switch, without blocking /new. No history → switch inline. try: _mm = getattr(self.agent, "_memory_manager", None) if _mm is not None: if _boundary_snapshot: _mm.commit_session_boundary_async( - _boundary_snapshot, - new_session_id=self.session_id, - parent_session_id=old_session_id or "", - reason="new_session", + _boundary_snapshot, new_session_id=self.session_id, + parent_session_id=old_session_id or "", reason="new_session", ) else: _mm.on_session_switch( - self.session_id, - parent_session_id=old_session_id or "", - reset=True, - reason="new_session", + self.session_id, parent_session_id=old_session_id or "", + reset=True, reason="new_session", ) except Exception: pass @@ -747,75 +626,51 @@ class CLISessionMixin: print("(^_^)v New session started!") def _consume_pending_resume_selection(self, text: str) -> bool: - """Resolve a bare numeric reply that follows a bare ``/resume`` prompt. + """Resolve a bare numeric reply following a bare ``/resume`` prompt. - After ``/resume`` (no args) prints the recent-sessions list it arms - ``self._pending_resume_sessions``. The next submitted input is given - one chance to be a bare session number (``3``); if so we resume that - session here. Anything else (another command, free text, blank) simply - disarms the prompt and is handled normally by the caller. - - Returns True if the input was consumed as a resume selection (caller - must not treat it as chat); False otherwise. The pending state is - always one-shot: it is cleared on the first submitted input regardless - of outcome. See #34584. + ``/resume`` (no args) arms ``self._pending_resume_sessions``; the next input gets one + chance to be a bare session number. The pending state is one-shot — cleared on the + first input regardless of outcome, so a stray later number is never hijacked. + Returns True if the input was consumed (caller must not treat it as chat). """ from cli import _cprint pending = self._pending_resume_sessions if not pending: return False - # One-shot: disarm now so a non-matching input can't leave the prompt - # armed and hijack a later number the user meant as chat. self._pending_resume_sessions = None - if not isinstance(text, str): return False stripped = text.strip() - # Only a pure number selects; let "/resume 3", titles, or any other - # text fall through to normal handling. + # Only a pure number selects; "/resume 3", titles etc. fall through. if not stripped.isdigit(): return False - index = int(stripped) if index < 1 or index > len(pending): _cprint(f" Resume index {index} is out of range.") _cprint(" Use /resume with no arguments to see available sessions.") return True - self._handle_resume_command(f"/resume {index}") return True def save_conversation(self, cmd: str = "/save"): - """Handle /save — export the current session to json, md, or html. + """Handle ``/save [json|md|html] [filename] [redact]``. - Usage: ``/save [json|md|html] [filename] [redact]`` - - The snapshot is a convenience export for sharing or off-line - inspection; every message is already persisted incrementally to the - SQLite session DB, so the live session remains resumable via - ``hermes --resume `` regardless of whether the user ever runs - ``/save``. ``redact`` runs the export through the force-mode secret + A convenience export only — every message is already persisted to the session DB, so + the live session stays resumable regardless. ``redact`` runs the force-mode secret redaction pass before writing. """ from cli import datetime from hermes_cli.session_export import ( - SAVE_USAGE, - normalize_save_format, - render_session_for_save, + SAVE_USAGE, normalize_save_format, render_session_for_save, ) parts = cmd.split()[1:] + redact = bool(parts) and parts[-1].lower() in ("redact", "--redact") + if redact: + parts = parts[:-1] if not parts: print(SAVE_USAGE) return - redact = False - if parts[-1].lower() in ("redact", "--redact"): - redact = True - parts = parts[:-1] - if not parts: - print(SAVE_USAGE) - return - try: fmt = normalize_save_format(parts[0]) except ValueError as e: @@ -824,10 +679,8 @@ class CLISessionMixin: return filename = parts[1] if len(parts) > 1 else None - # Prefer the durable DB row (has metadata + tool calls); fall back to - # the in-memory history for sessions that never touched the DB. - # getattr: test doubles (SimpleNamespace / object.__new__) may not - # carry _session_db or session_id. + # Prefer the durable DB row (metadata + tool calls); fall back to in-memory history. + # getattr: test doubles may not carry _session_db / session_id. session_data = None _db = getattr(self, "_session_db", None) _sid = getattr(self, "session_id", None) @@ -841,18 +694,14 @@ class CLISessionMixin: print("(;_;) No conversation to save.") return session_data = { - "id": self.session_id, - "model": self.model, - "started_at": self.session_start.timestamp(), - "messages": self.conversation_history, + "id": self.session_id, "model": self.model, + "started_at": self.session_start.timestamp(), "messages": self.conversation_history, } - if redact: from hermes_cli.session_export_md import redact_session_data session_data = redact_session_data(session_data) - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") saved_dir = get_hermes_home() / "sessions" / "saved" try: saved_dir.mkdir(parents=True, exist_ok=True) @@ -864,6 +713,7 @@ class CLISessionMixin: if not path.is_absolute(): path = Path.cwd() / path else: + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") path = saved_dir / f"hermes_conversation_{timestamp}.{fmt}" try: @@ -894,10 +744,7 @@ class CLISessionMixin: user_originated_turn_view, ) from agent.memory_manager import sanitize_context - from agent.tool_dispatch_helpers import ( - _is_multimodal_tool_result, - _multimodal_text_summary, - ) + from agent.tool_dispatch_helpers import _is_multimodal_tool_result, _multimodal_text_summary from run_agent import _is_ephemeral_scaffolding def _persistence_content(content: Any) -> Any: @@ -910,9 +757,7 @@ class CLISessionMixin: if isinstance(part, dict) and part.get("type") == "text": text_parts.append(str(part.get("text", ""))) elif isinstance(part, dict) and part.get("type") in { - "image", - "image_url", - "input_image", + "image", "image_url", "input_image", }: text_parts.append("[screenshot]") return "\n".join(text_parts) if text_parts else None @@ -920,38 +765,23 @@ class CLISessionMixin: def _comparison_content(message: Dict[str, Any]) -> Any: content = _persistence_content(message.get("content")) - if message.get("role") in {"user", "assistant"} and isinstance( - content, str - ): + if message.get("role") in {"user", "assistant"} and isinstance(content, str): return sanitize_context(content).strip() return content - expected_active_ids = self._session_db.get_active_message_ids( - self.session_id - ) + def _user_indices(messages): + return [i for i, m in enumerate(messages) if user_originated_turn_view(m) is not None] + + changed = RuntimeError("session history changed before the rewind could be persisted") + expected_active_ids = self._session_db.get_active_message_ids(self.session_id) durable = self._session_db.get_messages_as_conversation( - self.session_id, - include_row_ids=True, + self.session_id, include_row_ids=True ) - warm_persistence_history = [ - message - for message in warm_history - if not _is_ephemeral_scaffolding(message) - ] - warm_user_indices = [ - index - for index, message in enumerate(warm_persistence_history) - if user_originated_turn_view(message) is not None - ] - durable_user_indices = [ - index - for index, message in enumerate(durable) - if user_originated_turn_view(message) is not None - ] + warm_persistence_history = [m for m in warm_history if not _is_ephemeral_scaffolding(m)] + warm_user_indices = _user_indices(warm_persistence_history) + durable_user_indices = _user_indices(durable) if len(durable_user_indices) != len(warm_user_indices): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) + raise changed if user_ordinal < 0 or user_ordinal >= len(durable_user_indices): raise RuntimeError("persisted rewind target is no longer available") @@ -963,19 +793,14 @@ class CLISessionMixin: durable_prefix, durable_live_view = history_before_user_originated_turn( durable, durable_target_index ) - if _comparison_content(durable_live_view) != _comparison_content( - warm_live_view - ): - raise RuntimeError( - "session history changed before the rewind could be persisted" - ) + if _comparison_content(durable_live_view) != _comparison_content(warm_live_view): + raise changed target_row_id = durable_target.get("_row_id") if not isinstance(target_row_id, int): raise RuntimeError("persisted rewind target has no row identity") scaffold, _ = split_user_originated_turn(durable_target) result = self._session_db.rewind_to_message( - self.session_id, - target_row_id, + self.session_id, target_row_id, preserve_compaction_handoff=scaffold is not None, expected_active_ids=expected_active_ids, expected_target_content=durable_live_view.get("content"), @@ -989,52 +814,53 @@ class CLISessionMixin: warm_prefix[-1] = durable_prefix[-1] return warm_prefix, durable_live_view, result + def _publish_truncated_history(self, truncated: list, *, invalidate_prompt: bool) -> None: + """Install a rewound history and mirror it onto the agent (flush index reset so the + next turn re-flushes from the truncated head).""" + self.conversation_history = truncated + agent = self.agent + if agent is None: + return + if invalidate_prompt and hasattr(agent, "_invalidate_system_prompt"): + try: + agent._invalidate_system_prompt() + except Exception: + pass + if hasattr(agent, "_last_flushed_db_idx"): + try: + agent._last_flushed_db_idx = len(self.conversation_history) + except Exception: + pass + if hasattr(agent, "_session_messages"): + agent._session_messages = self.conversation_history + if hasattr(agent, "_db_flush_scan_prefix"): + agent._db_flush_scan_prefix = self.conversation_history[:] + def retry_last(self): - """Retry the last user message by removing the last exchange and re-sending. - - Removes the last assistant response (and any tool-call messages) and - the last user message, then re-sends that user message to the agent. - Returns the message to re-send, or None if there's nothing to retry. - """ + """Retry the last user message: drop the last exchange and return the text to re-send + (None when there is nothing to retry).""" if not self.conversation_history: print("(._.) No messages to retry.") return None - - # Walk backwards to the last *real* user message. Timeline bookkeeping - # rows (display_kind set) are role=user but are not user turns — match - # CLI resume counting and user_originated_turn_view. Compaction - # handoffs are excluded too (durable role=user, sometimes without - # display_kind on legacy sessions; #80622). + from agent.context_compressor import ( - history_before_user_originated_turn, - retryable_user_text, - user_originated_turn_view, + history_before_user_originated_turn, retryable_user_text, ) from agent.memory_manager import sanitize_context - from run_agent import _is_ephemeral_scaffolding warm_history = list(self.conversation_history) - - user_indices = [ - index - for index, message in enumerate(warm_history) - if not _is_ephemeral_scaffolding(message) - and user_originated_turn_view(message) is not None - ] - + user_indices = _user_turn_indices(warm_history) if not user_indices: print("(._.) No user message found to retry.") return None - last_user_idx = user_indices[-1] - - # Resolve a lossless live payload before touching either persistence or - # memory. A force-user-leading compaction row is one physical carrier: - # its historical handoff remains in the prefix while only the embedded - # human ask is retried. Media cannot be replayed by /retry, so fail - # closed before archiving anything. + + # Resolve a lossless live payload before touching persistence or memory. A + # force-user-leading compaction row is one physical carrier: its handoff stays in + # the prefix while only the embedded human ask is retried. Media cannot be replayed + # by /retry, so fail closed before archiving anything. try: truncated, live_view = history_before_user_originated_turn( - warm_history, last_user_idx + warm_history, user_indices[-1] ) live_content = live_view.get("content") if isinstance(live_content, str): @@ -1044,10 +870,8 @@ class CLISessionMixin: print(f"(._.) Cannot retry that message safely: {exc}") return None - # Persist the rewind before publishing the shorter in-memory view. - # The DB owns the physical carrier split so the archived original and - # retained scaffold are committed atomically. A plain user row keeps - # the legacy rewind shape (no replacement scaffold). + # Persist the rewind before publishing the shorter in-memory view: the DB owns the + # physical carrier split so archived original + retained scaffold commit atomically. if self._session_db is not None and self.session_id: try: truncated, _, _ = self._rewind_persisted_user_turn( @@ -1059,68 +883,29 @@ class CLISessionMixin: print(f"(x_x) Retry rewind failed; history was not changed: {exc}") return None - self.conversation_history = truncated - if self.agent is not None: - if hasattr(self.agent, "_session_messages"): - self.agent._session_messages = self.conversation_history - if hasattr(self.agent, "_last_flushed_db_idx"): - self.agent._last_flushed_db_idx = len(self.conversation_history) - if hasattr(self.agent, "_db_flush_scan_prefix"): - self.agent._db_flush_scan_prefix = self.conversation_history[:] - + self._publish_truncated_history(truncated, invalidate_prompt=False) print(f"(^_^)b Retrying: \"{last_message[:60]}{'...' if len(last_message) > 60 else ''}\"") return last_message def undo_last(self, n: int = 1, prefill: bool = True): - """Back up N user turns: truncate history, soft-delete on disk, prefill. + """Back up N user turns: truncate history, soft-delete on disk, prefill the composer. - Walks backwards N user messages and discards everything from the - Nth-from-last user message onward (its assistant response, tool - calls, etc.). ``n`` defaults to 1 (the last exchange); ``/undo 3`` - backs up three user turns. If ``n`` exceeds the number of user - turns, it backs up to the oldest one. - - Beyond the in-memory ``conversation_history`` slice, this also: - • soft-deletes the truncated rows in SessionDB (``active=0``) so - they're hidden from re-prompts and search but kept for audit; - • notifies memory providers via ``on_session_switch(rewound=True)``; - • mirrors /branch's agent surgery (system-prompt invalidation + - flush-index reset); - • when ``prefill`` is set and an input buffer is available, - pre-fills the composer with the backed-up message text so it - can be edited and resubmitted. - - ``prefill=False`` is used by callers that drive the undo - programmatically (e.g. checkpoint rollback) and don't want to - touch the user's input buffer. + Discards everything from the Nth-from-last user message onward (clamped to the oldest + turn). Rows are soft-deleted in SessionDB (``active=0``, kept for audit), memory + providers get ``on_session_switch(rewound=True)``, and the agent is patched like + /branch does. ``prefill=False`` is for programmatic callers (checkpoint rollback) + that must not touch the input buffer. """ from cli import logger if not self.conversation_history: print("(._.) No messages to undo.") return + n = max(n, 1) - if n < 1: - n = 1 - - # Walk backwards collecting the indices of the last N *real* user - # messages (exclude display_kind timeline rows and compaction - # handoffs — same predicate as user_originated_turn_view, resume - # turn counting, and /retry; #80622). - from agent.context_compressor import ( - history_before_user_originated_turn, - user_originated_turn_view, - ) - from run_agent import _is_ephemeral_scaffolding + from agent.context_compressor import history_before_user_originated_turn warm_history = list(self.conversation_history) - - user_indices = [ - index - for index, message in enumerate(warm_history) - if not _is_ephemeral_scaffolding(message) - and user_originated_turn_view(message) is not None - ] - + user_indices = _user_turn_indices(warm_history) if not user_indices: print("(._.) No user message found to undo.") return @@ -1128,30 +913,20 @@ class CLISessionMixin: turns_undone = min(n, len(user_indices)) target_ordinal = len(user_indices) - turns_undone cut_idx = user_indices[target_ordinal] - removed_count = len(warm_history) - cut_idx - truncated, live_view = history_before_user_originated_turn( - warm_history, cut_idx - ) + truncated, live_view = history_before_user_originated_turn(warm_history, cut_idx) removed_text = self._undo_content_to_text(live_view.get("content")) - # Soft-delete the truncated rows on disk so re-prompts and search - # see the clean transcript while the rows survive for audit. rewound_rows = 0 if self._session_db is not None and self.session_id: try: - truncated, durable_live_view, result = ( - self._rewind_persisted_user_turn( - warm_history=warm_history, - user_ordinal=target_ordinal, - warm_live_view=live_view, - ) - ) - # Canonicalize the editable prefill before mutation. The raw - # physical carrier contains the reference summary wrapper. - durable_text = self._undo_content_to_text( - durable_live_view.get("content") + truncated, durable_live_view, result = self._rewind_persisted_user_turn( + warm_history=warm_history, + user_ordinal=target_ordinal, + warm_live_view=live_view, ) + # Canonical editable prefill: the raw carrier holds the reference-summary wrapper. + durable_text = self._undo_content_to_text(durable_live_view.get("content")) if durable_text: removed_text = durable_text rewound_rows = result.get("rewound_count", 0) @@ -1161,50 +936,25 @@ class CLISessionMixin: return # Publish only after the durable rewind succeeds (or no store exists). - self.conversation_history = truncated - - # Agent surgery: invalidate the system-prompt cache and reset the - # flush index so the next turn re-flushes from the truncated head. + self._publish_truncated_history(truncated, invalidate_prompt=True) if self.agent is not None: - if hasattr(self.agent, "_invalidate_system_prompt"): - try: - self.agent._invalidate_system_prompt() - except Exception: - pass - if hasattr(self.agent, "_last_flushed_db_idx"): - try: - self.agent._last_flushed_db_idx = len(self.conversation_history) - except Exception: - pass - if hasattr(self.agent, "_session_messages"): - self.agent._session_messages = self.conversation_history - if hasattr(self.agent, "_db_flush_scan_prefix"): - self.agent._db_flush_scan_prefix = self.conversation_history[:] - # Notify memory providers — same hook /branch fires, with the - # rewound flag so per-turn document caches invalidate (#6672, #21910). + # Same hook /branch fires; rewound=True invalidates per-turn document caches. try: _mm = getattr(self.agent, "_memory_manager", None) if _mm is not None and self.session_id: _mm.on_session_switch( - self.session_id, - parent_session_id="", - reset=False, - rewound=True, + self.session_id, parent_session_id="", reset=False, rewound=True ) except Exception: pass turn_word = "turn" if turns_undone == 1 else "turns" - msg_count = rewound_rows or removed_count print( - f"(^_^)b Undid {turns_undone} {turn_word} ({msg_count} message(s)). " + f"(^_^)b Undid {turns_undone} {turn_word} ({rewound_rows or removed_count} message(s)). " f"Backed up to: \"{removed_text[:60]}{'...' if len(removed_text) > 60 else ''}\"" ) - remaining = len(self.conversation_history) - print(f" {remaining} message(s) remaining in history.") - - # Pre-fill the composer with the backed-up message so the user can - # edit and resubmit (Claude-Code-style). Editable, not auto-sent. + print(f" {len(self.conversation_history)} message(s) remaining in history.") + # Editable, not auto-sent (Claude-Code-style). if prefill and removed_text: self._prefill_input_buffer(removed_text) @@ -1215,22 +965,15 @@ class CLISessionMixin: return content if isinstance(content, list): parts = [ - p.get("text", "") - for p in content - if isinstance(p, dict) and p.get("type") == "text" + p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text" ] return "\n".join(t for t in parts if t) return "" def _write_terminal_breadcrumb(self) -> None: - """Record this terminal's live session for bare ``hermes -c``. - - Called at session start and whenever ``self.session_id`` is - reassigned mid-run (/new, /branch, auto-compression rotation) so a - later bare ``-c`` in THIS terminal resumes THIS conversation's live - tip. Best-effort — never raises, no-op without a terminal identity - or when session.terminal_continue is false. - """ + """Record this terminal's live session for bare ``hermes -c``. Called whenever + ``self.session_id`` is (re)assigned so a later bare ``-c`` in THIS terminal resumes + this conversation's live tip. Best-effort; no-op without a terminal identity.""" try: from hermes_cli.terminal_breadcrumbs import write_breadcrumb @@ -1239,97 +982,56 @@ class CLISessionMixin: pass def _transfer_session_yolo(self, old_session_id: str, new_session_id: str) -> None: - """Move YOLO bypass state from an old session key to a new one. - - Called whenever ``self.session_id`` is reassigned mid-run — ``/branch`` - forks into a new session, and auto-compression rotates the agent's - session id into a fresh continuation session. Without this transfer - the user's ``/yolo ON`` toggle would silently revert on the very next - turn (the same UX failure mode that motivated this entire fix), since - ``_session_yolo`` is keyed by session id. - - Mirrors ``tui_gateway/server.py`` (~line 1297-1305) which performs the - same transfer for the TUI's session-rename path. No-op when YOLO - wasn't enabled or when the ids match. - """ + """Move YOLO bypass state to a new session key when ``self.session_id`` is reassigned + mid-run (/branch, auto-compression rotation) — ``_session_yolo`` is keyed by id, so + without this the toggle silently reverts. Mirrors tui_gateway's rename path.""" if not old_session_id or not new_session_id or old_session_id == new_session_id: return try: from tools.approval import ( - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, + disable_session_yolo, enable_session_yolo, is_session_yolo_enabled, ) except Exception: return if is_session_yolo_enabled(old_session_id): enable_session_yolo(new_session_id) disable_session_yolo(old_session_id) - # Carry the persisted flag onto the continuation row so a later - # `hermes --resume ` restores the bypass too. getattr - # guard: tests call this unbound against a minimal stand-in. + # Carry the persisted flag onto the continuation row so a later --resume restores + # it too. getattr: tests call this unbound against a minimal stand-in. _persist = getattr(self, "_persist_session_yolo", None) if _persist: _persist(new_session_id, True) def _is_session_yolo_active(self) -> bool: - """Whether YOLO bypass is currently enabled for this CLI session. - - Reads from ``tools.approval._session_yolo`` (the same set that - ``enable_session_yolo`` / ``disable_session_yolo`` write to) so the - status bar reflects the actual bypass state instead of a stale env - var. Also honors the process-start ``--yolo`` flag, which freezes - ``HERMES_YOLO_MODE`` into ``_YOLO_MODE_FROZEN`` before tool imports - happen. - """ + """Whether YOLO bypass is on for this session: reads ``tools.approval._session_yolo`` + (not a stale env var) and honors the frozen process-start ``--yolo`` flag.""" try: - from tools.approval import ( - _YOLO_MODE_FROZEN, - is_session_yolo_enabled, - ) + from tools.approval import _YOLO_MODE_FROZEN, is_session_yolo_enabled except Exception: return False if _YOLO_MODE_FROZEN: return True - # Use ``getattr`` so test fixtures that build a CLI via ``__new__`` - # (skipping ``__init__``) don't trip an AttributeError here; the - # status-bar builders swallow exceptions silently but lose every - # field after the failure. - session_key = getattr(self, "session_id", None) or "default" - return is_session_yolo_enabled(session_key) + # getattr: __new__-built test fixtures skip __init__; the status-bar builders + # swallow exceptions but would lose every field after the failure. + return is_session_yolo_enabled(getattr(self, "session_id", None) or "default") def _toggle_yolo(self): - """Toggle YOLO mode — skip all dangerous command approval prompts. + """Toggle per-session YOLO mode (skip dangerous-command approvals). - Per-session toggle that mirrors the gateway and TUI ``/yolo`` handlers - (see ``gateway/run.py:_handle_yolo_command`` and - ``tui_gateway/server.py`` key=="yolo"). We deliberately do NOT mutate - ``HERMES_YOLO_MODE`` here — that env var is read once at module import - time into ``tools.approval._YOLO_MODE_FROZEN`` to keep prompt-injected - skills from flipping the bypass mid-session, so setting it after CLI - startup is a silent no-op. Routing through ``enable_session_yolo`` / - ``disable_session_yolo`` gives the same auditable, per-session bypass - the other surfaces have. ``run_conversation`` binds - ``self.session_id`` as the active approval session key via - ``set_current_session_key`` so the bypass takes effect on the very - next dangerous command in this run. + Mirrors the gateway/TUI ``/yolo`` handlers. Deliberately does NOT touch + ``HERMES_YOLO_MODE``: that env var is frozen into ``tools.approval._YOLO_MODE_FROZEN`` + at import (so prompt-injected skills can't flip the bypass), making a later set a + silent no-op. ``run_conversation`` binds ``self.session_id`` as the active approval + key, so the bypass applies to the very next dangerous command. """ from cli import _cprint from hermes_cli.colors import Colors as _Colors from tools.approval import ( - _YOLO_MODE_FROZEN, - disable_session_yolo, - enable_session_yolo, - is_session_yolo_enabled, + _YOLO_MODE_FROZEN, disable_session_yolo, enable_session_yolo, is_session_yolo_enabled, ) - # Process-level YOLO (--yolo flag / HERMES_YOLO_MODE at startup) is - # frozen into tools.approval at import time and cannot be disabled by - # the session toggle. Before this guard, /yolo printed "YOLO mode OFF — - # dangerous commands will require approval" while every command kept - # auto-approving (the frozen flag short-circuits the approval gate - # ahead of the session check) — a false safety claim. Say the truth - # instead of toggling a bypass that has no effect. + # A frozen process-level bypass short-circuits the approval gate ahead of the session + # check — toggling "OFF" would be a false safety claim. Say so instead. if _YOLO_MODE_FROZEN: _cprint( f" ⚡ YOLO is {_Colors.BOLD}{_Colors.RED}locked ON{_Colors.RESET}" @@ -1340,9 +1042,7 @@ class CLISessionMixin: return session_key = self.session_id or "default" - # ``getattr`` guard: tests exercise this method unbound against a - # minimal stand-in object (see tests/cli/test_cli_yolo_toggle.py); - # persistence is best-effort either way. + # getattr: tests call this unbound against a minimal stand-in; persistence is best-effort. _persist = getattr(self, "_persist_session_yolo", None) if is_session_yolo_enabled(session_key): disable_session_yolo(session_key) @@ -1362,15 +1062,9 @@ class CLISessionMixin: ) def _persist_session_yolo(self, session_key: str, enabled: bool) -> None: - """Persist the YOLO flag to the session row so --resume restores it. - - Best-effort: the in-memory toggle is authoritative for this process; - persistence only affects a future ``hermes --resume``. Skipped when the - session store is unavailable or the row doesn't exist yet (the row is - created lazily on the first turn — ``_toggle_yolo`` before any chat - writes nothing, and the launch-time ``--yolo`` flag is carried into the - creation-time model_config instead). - """ + """Persist the YOLO flag to the session row so --resume restores it. Best-effort; the + in-memory toggle is authoritative. Skipped without a store or before the row exists + (rows are created lazily on the first turn).""" db = getattr(self, "_session_db", None) if db is None or not session_key or session_key == "default": return @@ -1380,86 +1074,57 @@ class CLISessionMixin: pass def _manual_compress(self, cmd_original: str = ""): - """Manually trigger context compression on the current conversation. + """Manually trigger context compression. - Two modes: - - * ``/compress []`` — compress the *whole* history. An - optional focus topic guides the summariser to preserve - information related to *focus* while being more aggressive - about discarding everything else. Inspired by Claude Code's - ``/compact `` feature. - * ``/compress here [N]`` — boundary-aware compression. Summarize - everything *except* the most recent ``N`` exchanges (default - 2), which are preserved verbatim. Inspired by Claude Code's - Rewind "Summarize up to here" action (v2.1.139, May 2026, - https://code.claude.com/docs/en/whats-new/2026-w20). Lets the - user pick the compression boundary instead of leaving it to - the automatic token-budget heuristic. + * ``/compress []`` — compress the whole history; an optional focus topic tells + the summariser what to preserve while discarding the rest more aggressively. + * ``/compress here [N]`` — boundary-aware: summarize everything except the most recent + ``N`` exchanges (default 2), kept verbatim. + No ``compression_enabled`` gate: that flag disables *automatic* compaction only, and + the context-overflow error path directs users here when it is off. """ if not self.conversation_history or len(self.conversation_history) < 4: print("(._.) Not enough conversation to compress (need at least 4 messages).") return - if not self.agent: print("(._.) No active agent -- send a message first.") return - # No compression_enabled gate here: the config flag disables - # *automatic* compaction only. Manual /compress is an explicit user - # action — the context-overflow error path (conversation_loop.py) - # directs users here when auto-compaction is off, and the gateway's - # /compress handler has never gated on the flag. - from hermes_cli.partial_compress import ( - extract_compress_flags, - parse_partial_compress_args, - rejoin_compressed_head_and_tail, - split_history_for_partial_compress, - summarize_compress_preview, - ) - from agent.conversation_compression import ( - finalize_context_engine_compression_notification, + extract_compress_flags, parse_partial_compress_args, rejoin_compressed_head_and_tail, + split_history_for_partial_compress, summarize_compress_preview, ) + from agent.conversation_compression import finalize_context_engine_compression_notification + from agent.model_metadata import estimate_request_tokens_rough - # Args after the command word (e.g. "/compress here 3" -> "here 3"). raw_args = "" if cmd_original: _parts = cmd_original.strip().split(None, 1) if len(_parts) > 1: raw_args = _parts[1].strip() - - # Strip --preview/--dry-run/--aggressive before positional parsing - # so the flags coexist with 'here [N]' / focus-topic forms. + # Strip --preview/--dry-run/--aggressive before positional parsing. raw_args, preview, aggressive = extract_compress_flags(raw_args) partial, keep_last, focus_topic = parse_partial_compress_args(raw_args) focus_topic = focus_topic or "" if aggressive: - # LLM-free hard truncation is not supported: it would need its - # own transcript-persistence path outside the guarded - # _compress_context rotation machinery. Surface that instead of - # silently mis-parsing the flag as a focus topic. + # LLM-free hard truncation would need its own persistence path outside the + # guarded _compress_context rotation; surface that instead of mis-parsing. print("(._.) --aggressive is not supported; use '/compress here [N]' " "to keep only recent exchanges, or /undo to drop turns.") if not preview: return + # Include system prompt + tool schemas in estimates — a transcript-only number + # understates real request pressure and can even appear to grow after compression. + _estimate_kw = { + "system_prompt": getattr(self.agent, "_cached_system_prompt", "") or "", + "tools": getattr(self.agent, "tools", None) or None, + } if preview: - from agent.model_metadata import estimate_request_tokens_rough - _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" - _tools = getattr(self.agent, "tools", None) or None - approx_tokens = estimate_request_tokens_rough( - self.conversation_history, - system_prompt=_sys_prompt, - tools=_tools, - ) + approx_tokens = estimate_request_tokens_rough(self.conversation_history, **_estimate_kw) report = summarize_compress_preview( - self.conversation_history, - partial, - keep_last, - focus_topic or None, - approx_tokens, + self.conversation_history, partial, keep_last, focus_topic or None, approx_tokens, ) for line in report["lines"]: print(f"🗜️ {line}") @@ -1468,36 +1133,20 @@ class CLISessionMixin: original_count = len(self.conversation_history) with self._busy_command("Compressing context...", blocks_input=False): try: - from agent.model_metadata import estimate_request_tokens_rough from agent.manual_compression_feedback import summarize_manual_compression original_history = list(self.conversation_history) - # Boundary-aware split: only the head is summarized; the - # most recent `keep_last` exchanges ride along verbatim. + # Boundary-aware split: only the head is summarized. A degenerate split + # (nothing to keep / no head) falls back to full compression. tail: list = [] head = original_history if partial: - head, tail = split_history_for_partial_compress( - original_history, keep_last - ) + head, tail = split_history_for_partial_compress(original_history, keep_last) if not tail: - # Split degenerated (everything would be kept, or - # no head left to compress). Fall back to full - # compression so the user still gets an action. partial = False head = original_history - # Include system prompt + tool schemas in the estimate — - # a transcript-only number understates real request pressure - # and can even appear to grow after compression because a - # dense handoff summary replaces many short turns (#6217). - _sys_prompt = getattr(self.agent, "_cached_system_prompt", "") or "" - _tools = getattr(self.agent, "tools", None) or None - approx_tokens = estimate_request_tokens_rough( - original_history, - system_prompt=_sys_prompt, - tools=_tools, - ) + approx_tokens = estimate_request_tokens_rough(original_history, **_estimate_kw) if partial: print(f"🗜️ Summarizing up to here: compressing {len(head)} of " f"{original_count} messages (~{approx_tokens:,} tokens), " @@ -1508,90 +1157,47 @@ class CLISessionMixin: else: print(f"🗜️ Compressing {original_count} messages (~{approx_tokens:,} tokens)...") - # Pass None as system_message so _compress_context rebuilds - # the system prompt from scratch via _build_system_prompt(None). - # Passing _cached_system_prompt caused duplication because - # _build_system_prompt appends system_message to prompt_parts - # which already contain the agent identity — resulting in the - # identity block appearing twice (issue #15281). + # system_message=None so _compress_context rebuilds the prompt from scratch; + # passing _cached_system_prompt duplicated the identity block. compressed, _ = self.agent._compress_context( - head, - None, - approx_tokens=approx_tokens, - focus_topic=focus_topic or None, - force=True, - defer_context_engine_notification=True, + head, None, approx_tokens=approx_tokens, focus_topic=focus_topic or None, + force=True, defer_context_engine_notification=True, ) - # If _compress_context returned unchanged because a - # concurrent compression lock is held, tell the user - # clearly instead of showing the misleading - # "No changes from compression" no-op text. The wording - # distinguishes a confirmed holder from an unconfirmed - # acquisition failure (describe_compression_lock_skip). - # Type-pinned check (is True / str): the flag's only real - # values are None/True/holder-string, and a bare getattr - # truthiness test is fooled by MagicMock auto-attributes on - # test-double agents (skill pitfall: MagicMock vs hasattr). - _lock_skip_signal = getattr( - self.agent, "_compression_skipped_due_to_lock", None - ) + # Unchanged because a concurrent compression lock is held: say so instead of + # the misleading "No changes" no-op text. Type-pinned check (is True / str) — + # a bare truthiness test is fooled by MagicMock auto-attributes on test doubles. + _lock_skip_signal = getattr(self.agent, "_compression_skipped_due_to_lock", None) if _lock_skip_signal is True or isinstance(_lock_skip_signal, str): - from agent.manual_compression_feedback import ( - describe_compression_lock_skip, - ) + from agent.manual_compression_feedback import describe_compression_lock_skip print( - " " - + describe_compression_lock_skip( - self.agent._compression_skipped_due_to_lock - ) + " " + describe_compression_lock_skip(self.agent._compression_skipped_due_to_lock) ) self.agent._compression_skipped_due_to_lock = None - # No boundary was committed on a lock-skip; discard the - # deferred context-engine notification (exactly-once). - finalize_context_engine_compression_notification( - self.agent, - committed=False, - ) + # No boundary committed → discard the deferred notification (exactly-once). + finalize_context_engine_compression_notification(self.agent, committed=False) return if partial and tail: compressed = rejoin_compressed_head_and_tail(compressed, tail) self.conversation_history = compressed - # _compress_context ends the old session and creates a new child - # session on the agent (run_agent.py::_compress_context). Sync the - # CLI's session_id so /status, /resume, exit summary, and title - # generation all point at the live continuation session, not the - # ended parent. Without this, subsequent end_session() calls target - # the already-closed parent and the child is orphaned. - if ( - getattr(self.agent, "session_id", None) - and self.agent.session_id != self.session_id - ): + # _compress_context ends the old session and creates a child session on the + # agent. Sync the CLI's session_id so /status, /resume, exit summary and title + # generation point at the live continuation, not the ended parent. + agent_sid = getattr(self.agent, "session_id", None) + if agent_sid and agent_sid != self.session_id: self.session_id = self.agent.session_id - getattr(self, "_write_terminal_breadcrumb", lambda: None)() + self._write_terminal_breadcrumb() self._pending_title = None - # Manual /compress replaces conversation_history with a new - # compressed handoff for the child session. Persist it from - # offset 0 so resume can recover the continuation after exit. + # Persist the new handoff from offset 0 so resume can recover it after exit. self.agent._flush_messages_to_session_db(self.conversation_history, None) - finalize_context_engine_compression_notification( - self.agent, - committed=True, - ) + finalize_context_engine_compression_notification(self.agent, committed=True) new_tokens = estimate_request_tokens_rough( - self.conversation_history, - system_prompt=_sys_prompt, - tools=_tools, + self.conversation_history, **_estimate_kw ) summary = summarize_manual_compression( - original_history, - self.conversation_history, - approx_tokens, - new_tokens, - compression_state=getattr( - self.agent, "context_compressor", None - ), + original_history, self.conversation_history, approx_tokens, new_tokens, + compression_state=getattr(self.agent, "context_compressor", None), ) if ( summary.get("aborted") @@ -1605,49 +1211,25 @@ class CLISessionMixin: print(f" {summary['token_line']}") if summary["note"]: print(f" {summary['note']}") - except Exception as e: - finalize_context_engine_compression_notification( - self.agent, - committed=False, - ) + finalize_context_engine_compression_notification(self.agent, committed=False) print(f" ❌ Compression failed: {e}") def _persist_prompt_summary(self, icon: str, label: str, detail: str, outcome: str) -> None: - """Print a one-line scrollback summary of a resolved modal prompt. - - Modal panels (approval / clarify) live in the prompt_toolkit layout and - vanish on the next repaint, so the question and the decision leave no - trace in the terminal scrollback. When display.persist_prompts is on - (default), emit a dim single line after the prompt resolves so the - decision survives in chat history. - """ + """Print a one-line scrollback summary of a resolved modal prompt (approval/clarify + panels vanish on repaint); gated by ``display.persist_prompts``.""" from cli import CLI_CONFIG, _DIM, _RST, _cprint if not CLI_CONFIG.get("display", {}).get("persist_prompts", True): return - detail = " ".join(detail.split()) - if len(detail) > 120: - detail = detail[:119] + "…" - outcome = " ".join(outcome.split()) - if len(outcome) > 120: - outcome = outcome[:119] + "…" + detail, outcome = (" ".join(s.split()) for s in (detail, outcome)) + detail = detail[:119] + "…" if len(detail) > 120 else detail + outcome = outcome[:119] + "…" if len(outcome) > 120 else outcome _cprint(f"\n{_DIM}{icon} {label}: {detail} → {outcome}{_RST}") def _clear_terminal_on_exit(self): - """Clear screen + scrollback so nothing is stranded above the exit summary. - - Called from ``_print_exit_summary`` after ``app.run()`` has returned and - prompt_toolkit has torn down its renderer + restored terminal modes — - so a direct write to the real stdout fd is safe (the StdoutProxy / - patch_stdout layer is gone by now). - - Sequence: ``ESC[3J`` (erase scrollback) + ``ESC[2J`` (erase visible - screen) + ``ESC[H`` (cursor home). Modern terminals on Linux, macOS and - Windows (Terminal / conhost with VT processing, which prompt_toolkit - already enables) all honor these. Best-effort: skip silently when - stdout isn't a real console, and fall back to the platform ``clear`` / - ``cls`` command if the escape write fails. - """ + """Clear screen + scrollback (``ESC[3J ESC[2J ESC[H``) so nothing is stranded above + the exit summary. Only safe after ``app.run()`` returned and prompt_toolkit restored + terminal modes. Skips when stdout isn't a console; falls back to ``clear``/``cls``.""" try: stream = sys.stdout if stream is None or not stream.isatty(): @@ -1660,22 +1242,15 @@ class CLISessionMixin: return except Exception: pass - # Fallback: shell clear command (rarely needed — escapes work on every - # VT-capable terminal, but this covers exotic stdout wrappers). try: os.system("cls" if os.name == "nt" else "clear") except Exception: pass def _persist_active_session_before_close(self): - """Best-effort SQLite/JSON flush before the CLI marks a session closed. - - ``run_conversation()`` normally persists at turn boundaries, but a - terminal close/SIGHUP/SIGTERM can unwind the prompt_toolkit app while - the agent thread still holds the current turn only in memory. Flush the - agent's live ``_session_messages`` before ``end_session()`` so resume, - session_search, and state.db do not lose the interrupted turn. - """ + """Best-effort flush of the agent's live ``_session_messages`` before ``end_session()`` + — a terminal close/SIGHUP can unwind the app while the agent thread still holds the + current turn only in memory.""" from cli import logger agent = getattr(self, "agent", None) if not agent or not hasattr(agent, "_persist_session"): @@ -1684,10 +1259,9 @@ class CLISessionMixin: persist_lock = getattr(agent, "_session_persist_lock", None) def _snapshot_and_persist() -> None: - # This snapshot must share the staging lock with ``chat()``. Without - # it, close can retain a mutable history baseline just before chat - # appends its pending dict; the later flush then mistakes that dict - # for durable history and stamps it without writing a row (#63766). + # Must share the staging lock with ``chat()``: otherwise close can retain a + # history baseline just before chat appends its pending dict, and the later flush + # stamps that dict durable without writing a row. messages = getattr(agent, "_session_messages", None) pending_cli_message = getattr(agent, "_pending_cli_user_message", None) if not isinstance(messages, list): @@ -1695,39 +1269,30 @@ class CLISessionMixin: if not isinstance(messages, list): return if isinstance(pending_cli_message, dict) and not any( - message is pending_cli_message for message in messages + m is pending_cli_message for m in messages ): - # The UI has accepted a new input but the worker still exposes its - # prior snapshot. Include only that staged dict; the baseline below - # keeps any durable resumed prefix from being re-appended. + # The UI accepted a new input but the worker still exposes its prior snapshot. messages = [*messages, pending_cli_message] if not messages: return - # A normal turn builds a new list that reuses the resumed-history dicts. - # Keep that CLI history as the baseline so a signal between assigning - # ``_session_messages`` and the turn's DB flush cannot append its durable - # prefix a second time. Once the CLI takes the turn result, however, both - # names can point at the same live list; passing that alias would mark an - # unflushed tail durable without writing it. Marker-only persistence is - # correct only in that alias case. + # Baseline: the CLI history a normal turn built its new list from, so a signal + # between assigning ``_session_messages`` and the DB flush cannot append the + # durable prefix twice. When both names alias the same live list, marker-only + # persistence would mark an unflushed tail durable — pass None instead. conversation_history = getattr(self, "conversation_history", None) - pending_cli_message = getattr(agent, "_pending_cli_user_message", None) if ( isinstance(conversation_history, list) and conversation_history and conversation_history[-1] is pending_cli_message ): - # The UI accepted this user message before the agent finished its - # early persistence. Its dict can already be in ``messages`` but is - # not durable yet, so exclude it from the resumed-history baseline. + # Accepted but not yet durable: exclude it from the resumed-history baseline. conversation_history = conversation_history[:-1] elif not isinstance(conversation_history, list) or conversation_history is messages: conversation_history = None - # A first-turn close can arrive before the worker builds its cached - # prompt. Build or restore it before the DB row is created so the - # durable transcript never leaves a NULL system_prompt cache entry. + # A first-turn close can precede the cached prompt; build it so the durable + # transcript never gets a NULL system_prompt cache entry. if getattr(agent, "_cached_system_prompt", None) is None: try: from agent.conversation_loop import _restore_or_build_system_prompt @@ -1743,7 +1308,7 @@ class CLISessionMixin: agent._persist_session(messages, conversation_history) if getattr(agent, "session_id", None): self.session_id = agent.session_id - getattr(self, "_write_terminal_breadcrumb", lambda: None)() + self._write_terminal_breadcrumb() try: if persist_lock is None: @@ -1755,77 +1320,58 @@ class CLISessionMixin: logger.debug("Could not persist active CLI session before close: %s", e) def _print_exit_summary(self, clear_screen: bool = True): - """Print session resume info on exit, similar to Claude Code. - - Args: - clear_screen: When True (default), clear the terminal screen and - scrollback before printing the summary. This is appropriate for - interactive TUI teardown (#38252). Single-query (-q) mode should - pass False to preserve the printed answer (#53009). - """ + """Print session resume info on exit. ``clear_screen`` (interactive TUI teardown) + wipes screen + scrollback first; single-query mode passes False to keep the answer.""" from cli import datetime if clear_screen: - # Clear the screen + scrollback before printing the summary so the - # live bottom chrome (status bar, input box, separator rules) and the - # rest of the session transcript don't get stranded above the exit - # summary (#38252). By this point app.run() has returned and - # prompt_toolkit has restored terminal modes, so writing raw escapes - # to stdout is safe. ESC[3J clears scrollback, ESC[2J clears the - # visible screen, ESC[H homes the cursor — so the summary prints at a - # clean top-left. Falls back to the platform clear command if stdout - # isn't a TTY-capable stream. Honors NO_COLOR/dumb terminals by - # skipping silently when there's no real console. self._clear_terminal_on_exit() print() msg_count = len(self.conversation_history) - if msg_count > 0: - user_msgs = len([m for m in self.conversation_history if m.get("role") == "user"]) - tool_calls = len([m for m in self.conversation_history if m.get("role") == "tool" or m.get("tool_calls")]) - elapsed = datetime.now() - self.session_start - hours, remainder = divmod(int(elapsed.total_seconds()), 3600) - minutes, seconds = divmod(remainder, 60) - if hours > 0: - duration_str = f"{hours}h {minutes}m {seconds}s" - elif minutes > 0: - duration_str = f"{minutes}m {seconds}s" - else: - duration_str = f"{seconds}s" - - # Look up session title for resume-by-name hint - session_title = None - if self._session_db: - try: - session_title = self._session_db.get_session_title(self.session_id) - except Exception: - pass - - print("Resume this session with:") - # Session IDs are profile-constrained, so the resume hint must - # include `-p ` for non-default profiles. Without this, - # copying the hint from a non-default profile fails to find the - # session on the next invocation. The "default" and "custom" - # profile names use the standard HERMES_HOME, so no -p needed. - try: - from hermes_cli.profiles import get_active_profile_name - _active_profile = get_active_profile_name() - except Exception: - _active_profile = "default" - profile_flag = ( - "" if _active_profile in ("default", "custom") else f" -p {_active_profile}" - ) - print(f" hermes --resume {self.session_id}{profile_flag}") - if session_title: - print(f" hermes -c \"{session_title}\"{profile_flag}") - print() - print(f"Session: {self.session_id}") - if session_title: - print(f"Title: {session_title}") - print(f"Duration: {duration_str}") - print(f"Messages: {msg_count} ({user_msgs} user, {tool_calls} tool calls)") - else: + if not msg_count: try: from hermes_cli.skin_engine import get_active_goodbye goodbye = get_active_goodbye("Goodbye! ⚕") except Exception: goodbye = "Goodbye! ⚕" print(goodbye) + return + + user_msgs = len([m for m in self.conversation_history if m.get("role") == "user"]) + tool_calls = len([ + m for m in self.conversation_history if m.get("role") == "tool" or m.get("tool_calls") + ]) + elapsed = datetime.now() - self.session_start + hours, remainder = divmod(int(elapsed.total_seconds()), 3600) + minutes, seconds = divmod(remainder, 60) + if hours > 0: + duration_str = f"{hours}h {minutes}m {seconds}s" + elif minutes > 0: + duration_str = f"{minutes}m {seconds}s" + else: + duration_str = f"{seconds}s" + + session_title = None + if self._session_db: + try: + session_title = self._session_db.get_session_title(self.session_id) + except Exception: + pass + + print("Resume this session with:") + # Session IDs are profile-constrained: non-default profiles need `-p ` in + # the hint ("default"/"custom" use the standard HERMES_HOME). + try: + from hermes_cli.profiles import get_active_profile_name + _active_profile = get_active_profile_name() + except Exception: + _active_profile = "default" + profile_flag = "" if _active_profile in ("default", "custom") else f" -p {_active_profile}" + print(f" hermes --resume {self.session_id}{profile_flag}") + if session_title: + print(f" hermes -c \"{session_title}\"{profile_flag}") + print() + print(f"Session: {self.session_id}") + if session_title: + print(f"Title: {session_title}") + print(f"Duration: {duration_str}") + print(f"Messages: {msg_count} ({user_msgs} user, {tool_calls} tool calls)") diff --git a/hermes_cli/cli_status_bar_mixin.py b/hermes_cli/cli_status_bar_mixin.py index f14124fe6a..0c7e6fd709 100644 --- a/hermes_cli/cli_status_bar_mixin.py +++ b/hermes_cli/cli_status_bar_mixin.py @@ -1,8 +1,8 @@ -"""Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the interactive CLI +"""Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the +interactive CLI. -Mixin split out of ``cli.py``; bound onto ``HermesCLI`` via the MRO. cli.py-internal -symbols are imported LAZILY inside each method (``from cli import ...``) — the mixin -never imports ``cli`` at module load time (import cycle). +Mixin bound onto ``HermesCLI`` via the MRO. cli.py-internal symbols are imported LAZILY +inside each method (``from cli import ...``) — never at module load time (import cycle). """ from __future__ import annotations @@ -16,13 +16,23 @@ from agent.pet import render as pet_render from hermes_cli.banner import _format_context_length from typing import Any, Dict, Optional +_SB = "class:status-bar" +_DIM = "class:status-bar-dim" +_STRONG = "class:status-bar-strong" + + +def _finite(v): + """Drop NaN / negative / absurd provider timings (e.g. -0.8s seen in logs).""" + return None if v is None or v != v or v < 0 or v > 1e6 else v + class CLIStatusBarMixin: - """Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the interactive CLI""" + """Status bar, spinner, turn-summary, pet pane, and prompt-stash rendering for the + interactive CLI.""" def _status_bar_context_style(self, percent_used: Optional[int]) -> str: if percent_used is None: - return "class:status-bar-dim" + return _DIM if percent_used >= 95: return "class:status-bar-critical" if percent_used > 80: @@ -32,15 +42,9 @@ class CLIStatusBarMixin: return "class:status-bar-good" def _cache_hit_rate(self, snapshot: dict, precision: int = 1) -> "tuple[float, str] | None": - """Return (cache_pct, formatted_label) or None if no cache data. - - Centralises the cache-hit-rate computation so both the plain-text - status bar and the prompt-toolkit fragment path share one formula. - Prefers the baseline-delta percentage computed in - ``_get_status_bar_snapshot`` (resets on model switch / compression, - so it reflects the *current* cache regime); falls back to the - session-lifetime ratio when no delta is available. - """ + """Return (cache_pct, label) or None without cache data. Prefers the baseline-delta pct + from ``_get_status_bar_snapshot`` (resets on model switch / compression, so it reflects + the *current* cache regime); falls back to the session-lifetime ratio.""" delta_pct = snapshot.get("cache_hit_pct") if delta_pct is not None: return float(delta_pct), f"◎ {float(delta_pct):.{precision}f}%" @@ -52,7 +56,7 @@ class CLIStatusBarMixin: return None def _cache_hit_rate_style(self, cache_pct: float) -> str: - """Style for cache hit rate — higher is better (opposite of context %).""" + """Higher is better (opposite of context %).""" if cache_pct >= 70: return "class:status-bar-good" if cache_pct >= 40: @@ -61,21 +65,16 @@ class CLIStatusBarMixin: @staticmethod def _battery_status_style(category: str) -> str: - """Map a battery colour category to a status-bar style class.""" return { "good": "class:status-bar-good", "warn": "class:status-bar-warn", "bad": "class:status-bar-bad", "critical": "class:status-bar-critical", - }.get(category, "class:status-bar-dim") + }.get(category, _DIM) def _handle_battery_command(self, cmd_original: str) -> None: - """Toggle the status-bar battery read-out. - - ``/battery`` toggles, ``/battery on|off`` sets explicitly, and - ``/battery status`` reports the current setting plus a live reading. - The choice is persisted to ``display.battery`` so it survives restarts. - """ + """``/battery`` toggles, ``/battery on|off`` sets, ``/battery status`` reports the + setting plus a live reading. Persisted to ``display.battery``.""" from cli import save_config_value parts = (cmd_original or "").split() arg = parts[1].strip().lower() if len(parts) > 1 else "" @@ -112,29 +111,24 @@ class CLIStatusBarMixin: self._battery_visible = target save_config_value("display.battery", target) - - if target: - if reading is not None and not reading.available: - self._console_print( - " Battery indicator on — no battery detected, so nothing will show here" - ) - elif reading is not None and reading.available: - self._console_print( - f" Battery indicator on — {format_battery(reading)}" - ) - else: - self._console_print(" Battery indicator on") - else: + if not target: self._console_print(" Battery indicator off") + elif reading is not None and not reading.available: + self._console_print( + " Battery indicator on — no battery detected, so nothing will show here" + ) + elif reading is not None: + self._console_print(f" Battery indicator on — {format_battery(reading)}") + else: + self._console_print(" Battery indicator on") @staticmethod def _compression_count_style(count: int) -> str: - """Return a style class reflecting context compression pressure.""" if count >= 10: return "class:status-bar-bad" if count >= 5: return "class:status-bar-warn" - return "class:status-bar-dim" + return _DIM def _build_context_bar(self, percent_used: Optional[int], width: int = 10) -> str: safe_percent = max(0, min(100, percent_used or 0)) @@ -142,31 +136,22 @@ class CLIStatusBarMixin: return f"[{('█' * filled) + ('░' * max(0, width - filled))}]" @staticmethod - def _format_prompt_elapsed(prompt_start_time: Optional[float], prompt_duration: float, live: bool = False) -> str: - """Format per-prompt elapsed time for the status bar. - - Always returns a string — shows 0s on fresh start before first turn. - Keeps seconds visible at all scales so it increments smoothly: - 59s → 1m → 1m 1s → ... → 1m 59s → 2m → 2m 1s → ... - 59m 59s → 1h → 1h 0m 1s → ... - 23h 59m 59s → 1d → 1d 0h 1m → ... - - Emoji prefix: ⏱ when turn is live, ⏲ when frozen or fresh start. - Uses width-1 (no variation selector) glyphs so the status bar stays - aligned in monospace terminals. - """ + def _format_prompt_elapsed( + prompt_start_time: Optional[float], prompt_duration: float, live: bool = False + ) -> str: + """Per-prompt elapsed time. Always a string (``⏲ 0s`` on fresh start); seconds stay + visible at every scale so it increments smoothly (``1m 59s → 2m → 2m 1s``). ⏱ while + live, ⏲ frozen — width-1 glyphs (no variation selector) keep the bar aligned.""" if prompt_start_time is None and prompt_duration == 0.0: return "⏲ 0s" - elapsed = time.time() - prompt_start_time if prompt_start_time is not None else prompt_duration - elapsed = max(0.0, elapsed) - - days = int(elapsed // 86400) - remaining = elapsed % 86400 - hours = int(remaining // 3600) - remaining = remaining % 3600 - minutes = int(remaining // 60) - seconds = int(remaining % 60) - + if prompt_start_time is not None: + elapsed = max(0.0, time.time() - prompt_start_time) + else: + elapsed = max(0.0, prompt_duration) + days, remaining = divmod(elapsed, 86400) + hours, remaining = divmod(remaining, 3600) + minutes, seconds = int(remaining // 60), int(remaining % 60) + days, hours = int(days), int(hours) if days > 0: time_str = f"{days}d {hours}h {minutes}m" elif hours > 0: @@ -175,41 +160,28 @@ class CLIStatusBarMixin: time_str = f"{minutes}m {seconds}s" if seconds else f"{minutes}m" else: time_str = f"{int(elapsed)}s" - - emoji = "⏱" if live else "⏲" - return f"{emoji} {time_str}" + return f"{'⏱' if live else '⏲'} {time_str}" @staticmethod def _format_idle_since(last_finished_at: Optional[float], turn_live: bool) -> str: - """Format time since the last final agent response for the status bar. - - Returns an empty string while a turn is live (the per-prompt elapsed - timer covers that case) or before the first turn has completed. - Compact read-out: ``✓ 42s`` / ``✓ 3m`` / ``✓ 1h 12m``. - """ + """``✓ 42s`` since the last final response; empty while a turn is live or before the + first turn completes.""" from cli import format_duration_compact if turn_live or last_finished_at is None: return "" - idle = max(0.0, time.time() - last_finished_at) - return f"✓ {format_duration_compact(idle)}" + return f"✓ {format_duration_compact(max(0.0, time.time() - last_finished_at))}" def _get_status_bar_snapshot(self) -> Dict[str, Any]: - # Prefer the agent's model name — it updates on fallback. - # self.model reflects the originally configured model and never - # changes mid-session, so the TUI would show a stale name after - # _try_activate_fallback() switches provider/model. from cli import _reverse_alias_for_display, datetime, format_duration_compact agent = getattr(self, "agent", None) + # Prefer the agent's model name — it updates on fallback; self.model never changes. model_name = (getattr(agent, "model", None) or self.model or "unknown") - # Friendly display: prefer reverse-alias from config.yaml ``model_aliases:`` - # before slash/length truncation. This turns long Palantir RIDs like - # ``ri.language-model-service..language-model.anthropic-claude-4-7-opus`` - # into the user's chosen short name (e.g. ``opus-4.7``) in the status bar. + # Friendly display: reverse-alias from config ``model_aliases:`` first (turns long + # Palantir RIDs into the user's short name), else slash/length truncation. model_short = _reverse_alias_for_display(model_name) if model_short == model_name: model_short = model_name.split("/")[-1] if "/" in model_name else model_name - # Strip Palantir RID prefixes via the shared display formatter so - # this site and ``ModelSwitchResult`` confirmation can't drift. + # Shared RID-prefix stripper so this and ModelSwitchResult can't drift. from hermes_cli.model_switch import format_model_for_display model_short = format_model_for_display(model_short) if model_short.endswith(".gguf"): @@ -217,6 +189,8 @@ class CLIStatusBarMixin: if len(model_short) > 26: model_short = f"{model_short[:23]}..." + prompt_start = getattr(self, "_prompt_start_time", None) + turn_live = prompt_start is not None elapsed_seconds = max(0.0, (datetime.now() - self.session_start).total_seconds()) snapshot = { "model_name": model_name, @@ -224,13 +198,10 @@ class CLIStatusBarMixin: "duration": format_duration_compact(elapsed_seconds), "session_title": self._get_status_bar_session_title(), "prompt_elapsed": self._format_prompt_elapsed( - getattr(self, "_prompt_start_time", None), - getattr(self, "_prompt_duration", 0.0), - live=getattr(self, "_prompt_start_time", None) is not None, + prompt_start, getattr(self, "_prompt_duration", 0.0), live=turn_live, ), "idle_since": self._format_idle_since( - getattr(self, "_last_turn_finished_at", None), - turn_live=getattr(self, "_prompt_start_time", None) is not None, + getattr(self, "_last_turn_finished_at", None), turn_live=turn_live, ), "context_tokens": 0, "context_length": None, @@ -249,9 +220,10 @@ class CLIStatusBarMixin: "active_background_subagents": 0, "battery_label": "", "battery_category": "dim", - # Focus view badge (/focus). Persistent indicator so the reduced - # output mode is never invisible. Display-only. - "focus_label": "", + "focus_label": "", # /focus badge: the reduced-output mode is never invisible. + "goal_active": False, + "goal_turns_used": 0, + "goal_max_turns": 0, } try: @@ -263,16 +235,10 @@ class CLIStatusBarMixin: except Exception: pass - # Battery read-out (first status-bar element when enabled). Reads are - # memoised for a few seconds inside agent.battery, so polling it on - # every status-bar repaint is cheap. + # Battery reads are memoised inside agent.battery, so per-repaint polling is cheap. if getattr(self, "_battery_visible", False): try: - from agent.battery import ( - battery_category, - format_battery, - read_battery, - ) + from agent.battery import battery_category, format_battery, read_battery _batt = read_battery() snapshot["battery_label"] = format_battery(_batt) @@ -280,41 +246,27 @@ class CLIStatusBarMixin: except Exception: pass - # Count live /bg tasks. The dict entry is removed in the - # task thread's finally block, so len() reflects truly-running tasks. - # len() on a CPython dict is atomic; safe to read without a lock. + # Live /bg tasks: entries are removed in the task thread's finally block; dict len() + # is atomic in CPython, no lock needed. try: bg_tasks = getattr(self, "_background_tasks", None) if bg_tasks: snapshot["active_background_tasks"] = len(bg_tasks) except Exception: pass - - # Count live background terminal processes (terminal tool background - # sessions tracked by tools.process_registry). Cheap O(1) read. try: from tools.process_registry import process_registry snapshot["active_background_processes"] = process_registry.count_running() except Exception: pass - - # Count live background/async subagents (delegate_task batches and - # background single delegations tracked by tools.async_delegation). - # active_count() iterates an in-memory records dict under a lock — - # cheap and only counts records still in the "running" state. try: from tools.async_delegation import active_count as _async_active_count snapshot["active_background_subagents"] = _async_active_count() except Exception: pass - # Standing /goal state (Ralph loop). GoalManager is cached on self and - # keeps its state in memory, so this is a cheap attribute read — no DB - # hit per repaint. Only an *active* goal earns a segment; paused/done - # goals stay out of the bar (matching the desktop's active-first row). - snapshot["goal_active"] = False - snapshot["goal_turns_used"] = 0 - snapshot["goal_max_turns"] = 0 + # Standing /goal (Ralph loop): GoalManager is cached on self — no DB hit per repaint. + # Only an *active* goal earns a segment (paused/done stay out, like the desktop). try: goal_mgr = self._get_goal_manager() if goal_mgr is not None and goal_mgr.is_active(): @@ -325,39 +277,26 @@ class CLIStatusBarMixin: except Exception: pass - if not agent: return snapshot - snapshot["session_input_tokens"] = getattr(agent, "session_input_tokens", 0) or 0 - snapshot["session_output_tokens"] = getattr(agent, "session_output_tokens", 0) or 0 - snapshot["session_cache_read_tokens"] = getattr(agent, "session_cache_read_tokens", 0) or 0 - snapshot["session_cache_write_tokens"] = getattr(agent, "session_cache_write_tokens", 0) or 0 - snapshot["session_prompt_tokens"] = getattr(agent, "session_prompt_tokens", 0) or 0 - snapshot["session_completion_tokens"] = getattr(agent, "session_completion_tokens", 0) or 0 - snapshot["session_total_tokens"] = getattr(agent, "session_total_tokens", 0) or 0 - snapshot["session_api_calls"] = getattr(agent, "session_api_calls", 0) or 0 + for key in ( + "session_input_tokens", "session_output_tokens", "session_cache_read_tokens", + "session_cache_write_tokens", "session_prompt_tokens", "session_completion_tokens", + "session_total_tokens", "session_api_calls", + ): + snapshot[key] = getattr(agent, key, 0) or 0 compressor = getattr(agent, "context_compressor", None) if compressor: - # last_prompt_tokens is parked at the -1 sentinel right after a - # compression, until the next real API call reports a prompt count - # (awaiting_real_usage_after_compression). The status bar must not - # render that sentinel verbatim — it produced "-1/200K" / "-1%". - # Clamp it to 0 so the one transitional turn reads as empty context. - context_tokens = getattr(compressor, "last_prompt_tokens", 0) or 0 - if context_tokens < 0: - context_tokens = 0 - # Durable-transcript view: on reasoning models a long tool loop - # replays the current turn's thinking + scaffolding on every - # request, so the LAST request's prompt_tokens can exceed the - # durable transcript by hundreds of K — all of which evaporates - # at the turn boundary. Rendering that raw figure makes the bar - # sawtooth (e.g. 850K mid-turn -> 600K next turn) and reads as a - # broken compaction. Anchor the display on the turn's FIRST - # response (minimal replay) plus a delta estimate of messages - # appended since, excluding stale thinking. Display-only: the - # compression trigger keeps using real last-request usage. + # last_prompt_tokens parks at the -1 sentinel right after a compression until the + # next real API call; clamp so the bar never renders "-1/200K". + context_tokens = max(0, getattr(compressor, "last_prompt_tokens", 0) or 0) + # Display-only anchoring: on reasoning models a long tool loop replays the turn's + # thinking on every request, so the LAST request's prompt_tokens can exceed the + # durable transcript by hundreds of K and the bar sawtooths at the turn boundary. + # Anchor on the turn's FIRST response plus a delta estimate of appended messages. + # The compression trigger keeps using real last-request usage. try: from agent.model_metadata import anchored_context_tokens @@ -371,127 +310,86 @@ class CLIStatusBarMixin: context_tokens = _anchored except Exception: pass - context_length = getattr(compressor, "context_length", 0) or 0 - if context_length < 0: - context_length = 0 + context_length = max(0, getattr(compressor, "context_length", 0) or 0) snapshot["context_tokens"] = context_tokens snapshot["context_length"] = context_length or None snapshot["compressions"] = getattr(compressor, "compression_count", 0) or 0 if context_length: - snapshot["context_percent"] = max(0, min(100, round((context_tokens / context_length) * 100))) + pct = round((context_tokens / context_length) * 100) + snapshot["context_percent"] = max(0, min(100, pct)) - # -- Cache-hit ratio (delta since last reset) -- - # Reset baseline on model switch and on compression — both invalidate - # the prompt cache. Formula verified against live logs: - # hit = cache_read / prompt_tokens (prompt = input+cache_read+cache_write) - # see agent/conversation_loop.py:4314 cache=read/prompt (87%) - # and CanonicalUsage.prompt_tokens = input+read+write + # Cache-hit ratio since the last baseline reset (model switch and compression both + # invalidate the prompt cache). hit = cache_read / prompt_tokens, where + # prompt = input + cache_read + cache_write (CanonicalUsage). + pct = None try: base_model = getattr(self, "_cache_hit_baseline_model", None) base_prompt = int(getattr(self, "_cache_hit_baseline_prompt", 0) or 0) base_read = int(getattr(self, "_cache_hit_baseline_read", 0) or 0) base_comps = int(getattr(self, "_cache_hit_baseline_compressions", 0) or 0) - cur_model = snapshot.get("model_name") or model_name - cur_comps = int(snapshot.get("compressions", 0) or 0) - cur_prompt = int(snapshot.get("session_prompt_tokens", 0) or 0) - cur_read = int(snapshot.get("session_cache_read_tokens", 0) or 0) + cur_comps = int(snapshot["compressions"] or 0) + cur_prompt = int(snapshot["session_prompt_tokens"] or 0) + cur_read = int(snapshot["session_cache_read_tokens"] or 0) + + def _rebase(*, tokens: bool) -> None: + nonlocal base_prompt, base_read, base_comps + self._cache_hit_baseline_model = model_name + self._cache_hit_baseline_compressions = base_comps = cur_comps + if tokens: + self._cache_hit_baseline_prompt = base_prompt = cur_prompt + self._cache_hit_baseline_read = base_read = cur_read + if base_model is None: - self._cache_hit_baseline_model = cur_model - self._cache_hit_baseline_compressions = cur_comps - base_model = cur_model - base_comps = cur_comps - if cur_model != base_model: - self._cache_hit_baseline_model = cur_model - self._cache_hit_baseline_prompt = cur_prompt - self._cache_hit_baseline_read = cur_read - self._cache_hit_baseline_compressions = cur_comps - base_prompt = cur_prompt - base_read = cur_read - base_comps = cur_comps + _rebase(tokens=False) + elif model_name != base_model: + _rebase(tokens=True) if cur_comps != base_comps: - self._cache_hit_baseline_compressions = cur_comps - self._cache_hit_baseline_prompt = cur_prompt - self._cache_hit_baseline_read = cur_read - base_prompt = cur_prompt - base_read = cur_read + _rebase(tokens=True) delta_prompt = cur_prompt - base_prompt delta_read = cur_read - base_read - # A zero-read regime hides the segment entirely (no cache data - # is not the same as a 0% hit worth alarming about), and the pct - # stays a float so renderers control their own precision. + # A zero-read regime hides the segment (no data ≠ an alarming 0%); pct stays a + # float so renderers choose their own precision. if delta_prompt > 0 and delta_read > 0: pct = max(0.0, min(100.0, (delta_read / delta_prompt) * 100)) - snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct:.0f}%" - elif cur_prompt > 0 and cur_read > 0 and base_prompt == 0 and base_read == 0: - pct = max(0.0, min(100.0, (cur_read / cur_prompt) * 100)) - snapshot["cache_hit_pct"] = pct - snapshot["cache_hit_label"] = f"{pct:.0f}%" - else: - snapshot["cache_hit_pct"] = None - snapshot["cache_hit_label"] = "" except Exception: - snapshot["cache_hit_pct"] = None - snapshot["cache_hit_label"] = "" + pct = None + snapshot["cache_hit_pct"] = pct + snapshot["cache_hit_label"] = f"{pct:.0f}%" if pct is not None else "" - # -- Rolling avg latency / velocity (last 10 calls) -- - # Reads the deque maintained in agent/conversation_loop.py (and - # agent_init). Codex app-server has no latency, so it stays hidden there. + # Rolling avg latency / velocity over the deques kept by agent/conversation_loop.py + # (hidden on Codex app-server, which reports no latency). + avg_lat = avg_vel = None try: - agent_obj = getattr(self, "agent", None) - lhist = list(getattr(agent_obj, "_api_latency_history", []) or []) if agent_obj else [] - ohist = list(getattr(agent_obj, "_api_output_history", []) or []) if agent_obj else [] - # Keep the two histories aligned (they are appended together). - n = min(len(lhist), len(ohist)) + lhist = list(getattr(agent, "_api_latency_history", []) or []) + ohist = list(getattr(agent, "_api_output_history", []) or []) + n = min(len(lhist), len(ohist)) # appended together; keep aligned if n: - lhist = lhist[-n:] - ohist = ohist[-n:] - # Simple mean for latency; sum/sum for velocity (true throughput, not mean of ratios). - avg_lat = sum(lhist) / len(lhist) if lhist else None - total_out = sum(ohist) + lhist, ohist = lhist[-n:], ohist[-n:] total_lat = sum(lhist) - avg_vel = (total_out / total_lat) if total_lat > 0 else None - # Guard against NaN / inf from weird provider timings (e.g. -0.8s in logs). - if avg_lat is not None and (avg_lat != avg_lat or avg_lat < 0 or avg_lat > 1e6): - avg_lat = None - if avg_vel is not None and (avg_vel != avg_vel or avg_vel < 0 or avg_vel > 1e6): - avg_vel = None - snapshot["avg_latency"] = float(avg_lat) if avg_lat is not None else None - snapshot["avg_latency_label"] = f"{avg_lat:.1f}s" if avg_lat is not None else "" - snapshot["avg_velocity"] = float(avg_vel) if avg_vel is not None else None - snapshot["avg_velocity_label"] = f"{avg_vel:.0f} t/s" if avg_vel is not None else "" - else: - snapshot["avg_latency"] = None - snapshot["avg_latency_label"] = "" - snapshot["avg_velocity"] = None - snapshot["avg_velocity_label"] = "" + # Mean for latency; sum/sum for velocity (true throughput, not mean of ratios). + avg_lat = _finite(total_lat / n) + avg_vel = _finite(sum(ohist) / total_lat if total_lat > 0 else None) except Exception: - snapshot["avg_latency"] = None - snapshot["avg_latency_label"] = "" - snapshot["avg_velocity"] = None - snapshot["avg_velocity_label"] = "" - + avg_lat = avg_vel = None + snapshot["avg_latency"] = float(avg_lat) if avg_lat is not None else None + snapshot["avg_latency_label"] = f"{avg_lat:.1f}s" if avg_lat is not None else "" + snapshot["avg_velocity"] = float(avg_vel) if avg_vel is not None else None + snapshot["avg_velocity_label"] = f"{avg_vel:.0f} t/s" if avg_vel is not None else "" return snapshot def _get_status_bar_session_title(self) -> str: - """Return the current title without polling state.db on every repaint.""" + """Current title, polling state.db at most every 1.5s (not on every repaint).""" pending = str(getattr(self, "_pending_title", None) or "").strip() session_id = str(getattr(self, "session_id", "") or "") - if pending: - self._status_bar_title_session_id = session_id - self._status_bar_title_cache = pending - self._status_bar_title_checked_at = time.monotonic() - return pending - now = time.monotonic() - cached_session_id = getattr(self, "_status_bar_title_session_id", None) - checked_at = float(getattr(self, "_status_bar_title_checked_at", 0.0) or 0.0) - if cached_session_id == session_id and now - checked_at < 1.5: - return str(getattr(self, "_status_bar_title_cache", "") or "") - - title = "" + if not pending: + cached_session_id = getattr(self, "_status_bar_title_session_id", None) + checked_at = float(getattr(self, "_status_bar_title_checked_at", 0.0) or 0.0) + if cached_session_id == session_id and now - checked_at < 1.5: + return str(getattr(self, "_status_bar_title_cache", "") or "") + title = pending db = getattr(self, "_session_db", None) - if db is not None and session_id: + if not pending and db is not None and session_id: try: title = str(db.get_session_title(session_id) or "").strip() except Exception: @@ -503,13 +401,8 @@ class CLIStatusBarMixin: @staticmethod def _status_bar_display_width(text: str) -> int: - """Return terminal cell width for status-bar text. - - len() is not enough for prompt_toolkit layout decisions because some - glyphs can render wider than one Python codepoint. Keeping the status - bar within the real display width prevents it from wrapping onto a - second line and leaving behind duplicate rows. - """ + """Terminal cell width (some glyphs render wider than one codepoint); keeps the bar + from wrapping onto a second line and leaving duplicate rows.""" try: from prompt_toolkit.utils import get_cwidth return get_cwidth(text or "") @@ -521,23 +414,17 @@ class CLIStatusBarMixin: """Trim status-bar text to a single terminal row.""" if max_width <= 0: return "" - try: - from prompt_toolkit.utils import get_cwidth - except Exception: - get_cwidth = None - - if cls._status_bar_display_width(text) <= max_width: + cw = cls._status_bar_display_width + if cw(text) <= max_width: return text - ellipsis = "..." - ellipsis_width = cls._status_bar_display_width(ellipsis) + ellipsis_width = cw(ellipsis) if max_width <= ellipsis_width: return ellipsis[:max_width] - out = [] width = 0 for ch in text: - ch_width = get_cwidth(ch) if get_cwidth else len(ch) + ch_width = cw(ch) if width + ch_width + ellipsis_width > max_width: break out.append(ch) @@ -545,31 +432,35 @@ class CLIStatusBarMixin: return "".join(out).rstrip() + ellipsis @classmethod - def _right_align_status_title(cls, text: str, title: str, width: int) -> str: - """Pin a bounded session-title badge to the far-right status-bar edge.""" + def _status_title_badge(cls, title: str, width: int) -> "tuple[str, int] | None": + """(badge, left_width) for the far-right session-title badge, or None when it + doesn't fit (no title / bar narrower than 24 cells).""" title = str(title or "").strip() if not title or width < 24: - return cls._trim_status_bar_text(text, width) - + return None title_width = max(6, min(30, width // 3)) badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " - suffix = f" ─{badge}" - left_width = max(0, width - cls._status_bar_display_width(suffix)) + suffix_width = cls._status_bar_display_width(" ─") + cls._status_bar_display_width(badge) + return badge, max(0, width - suffix_width) + + @classmethod + def _right_align_status_title(cls, text: str, title: str, width: int) -> str: + """Pin a bounded session-title badge to the far-right status-bar edge.""" + placed = cls._status_title_badge(title, width) + if placed is None: + return cls._trim_status_bar_text(text, width) + badge, left_width = placed left = cls._trim_status_bar_text(text.rstrip(), left_width) padding = " " * max(0, left_width - cls._status_bar_display_width(left)) - return f"{left}{padding}{suffix}" + return f"{left}{padding} ─{badge}" @classmethod def _right_align_status_title_fragments(cls, frags, title: str, width: int): """Styled counterpart to :meth:`_right_align_status_title`.""" - title = str(title or "").strip() - if not title or width < 24: + placed = cls._status_title_badge(title, width) + if placed is None: return frags - - title_width = max(6, min(30, width // 3)) - badge = f" {cls._trim_status_bar_text(title, title_width - 2)} " - suffix_width = cls._status_bar_display_width(" ─") + cls._status_bar_display_width(badge) - left_width = max(0, width - suffix_width) + badge, left_width = placed trimmed = [] used = 0 for style, value in frags: @@ -586,23 +477,15 @@ class CLIStatusBarMixin: trimmed.append((style, clipped)) used += cls._status_bar_display_width(clipped) break - if used < left_width: - trimmed.append(("class:status-bar-dim", " " * (left_width - used))) - trimmed.extend([ - ("class:status-bar-dim", " ─"), - ("class:status-bar-session-title", badge), - ]) + trimmed.append((_DIM, " " * (left_width - used))) + trimmed.extend([(_DIM, " ─"), ("class:status-bar-session-title", badge)]) return trimmed @staticmethod def _get_tui_terminal_width(default: tuple[int, int] = (80, 24)) -> int: - """Return the live prompt_toolkit width, falling back to ``shutil``. - - The TUI layout can be narrower than ``shutil.get_terminal_size()`` reports, - especially on Termux/mobile shells, so prefer prompt_toolkit's width whenever - an app is active. - """ + """Live prompt_toolkit width (can be narrower than shutil's, esp. Termux), falling + back to ``shutil``.""" try: from prompt_toolkit.application import get_app return get_app().output.get_size().columns @@ -617,21 +500,10 @@ class CLIStatusBarMixin: @staticmethod def _scrollback_box_width(width: Optional[int] = None) -> int: - """Return the full viewport width for printed scrollback box rules. - - Previously this clamped to ``max(32, min(width, 56))`` as a defense - against terminal-emulator reflow on column-shrink (#25975, salvaging - #24403). That clamp made response/reasoning borders look stubby on - any modern wide terminal. We now trust the prompt_toolkit - ``_output_screen_diff`` monkey-patch landed in #26137 (salvaging - #25981) to keep chrome out of scrollback in the first place, and - accept that an aggressive column-shrink may visually reflow already - printed Panel borders — that's a cosmetic artifact of stamped - scrollback history, not a live-render bug. - - A small floor (32 cols) is kept so the box still renders on tiny - terminals without negative ``'─' * (w - 2)`` math. - """ + """Full viewport width for printed scrollback box rules, floored at 32 cols so tiny + terminals never hit negative ``'─' * (w - 2)`` math. (The old 56-col clamp against + reflow-on-shrink is gone: the ``_output_screen_diff`` patch keeps chrome out of + scrollback, and reflow of already-printed borders is a cosmetic artifact.)""" if width is None: try: width = shutil.get_terminal_size((80, 24)).columns @@ -640,27 +512,24 @@ class CLIStatusBarMixin: return max(32, int(width or 80)) def _agent_spacer_height(self, width: Optional[int] = None) -> int: - """Return the spacer height shown above the status bar while the agent runs.""" + """Spacer height above the status bar while the agent runs.""" if not getattr(self, "_agent_running", False): return 0 return 0 if self._use_minimal_tui_chrome(width=width) else 1 def _spinner_widget_height(self, width: Optional[int] = None) -> int: - """Return the visible height for the spinner/status text line above the status bar.""" + """Visible height of the spinner/status line above the status bar.""" spinner_line = self._render_spinner_text() - if not spinner_line: - return 0 - if self._use_minimal_tui_chrome(width=width): + if not spinner_line or self._use_minimal_tui_chrome(width=width): return 0 width = width or self._get_tui_terminal_width() if width and width > 10: import math - text_width = self._status_bar_display_width(spinner_line) - return max(1, math.ceil(text_width / width)) + return max(1, math.ceil(self._status_bar_display_width(spinner_line) / width)) return 1 def _render_spinner_text(self) -> str: - """Return the live spinner/status text exactly as rendered in the TUI.""" + """The live spinner/status text exactly as rendered in the TUI.""" txt = getattr(self, "_spinner_text", "") if not txt: return "" @@ -668,20 +537,13 @@ class CLIStatusBarMixin: t0 = getattr(self, "_tool_start_time", 0) or 0 if t0 > 0: elapsed = time.monotonic() - t0 + # Fixed-width timers (01m05s / " 5.2s") avoid status-line wrap jitter on repaint. if elapsed >= 60: - _m, _s = int(elapsed // 60), int(elapsed % 60) - # Fixed-width timer to avoid status-line wrap jitter while - # scrolling/repainting (e.g. 01m05s, 12m09s). - elapsed_str = f"{_m:02d}m{_s:02d}s" + elapsed_str = f"{int(elapsed // 60):02d}m{int(elapsed % 60):02d}s" else: - # Keep width stable before the 60s rollover as well. elapsed_str = f"{elapsed:5.1f}s" - if flow: - return f" {txt} ({elapsed_str} · {flow})" - return f" {txt} ({elapsed_str})" - if flow: - return f" {txt} ({flow})" - return f" {txt}" + return f" {txt} ({elapsed_str} · {flow})" if flow else f" {txt} ({elapsed_str})" + return f" {txt} ({flow})" if flow else f" {txt}" def _spinner_token_flow(self) -> str: """Cumulative output tokens for the running turn, for the spinner.""" @@ -703,12 +565,8 @@ class CLIStatusBarMixin: return "" def _turn_summary_is_active(self) -> bool: - """Whether the per-turn summary line should render for this surface. - - Gated off for: the config key, quiet/tool-progress-off mode, and any - non-interactive path (single query, ``-Q``, gateway/messaging) — those - surfaces either want machine-readable output or carry their own footer. - """ + """Whether the per-turn summary line renders here: off for the config key, quiet / + tool-progress-off mode, and every non-interactive path (-q, -Q, gateway).""" if not getattr(self, "_turn_summary_enabled", False): return False if getattr(self, "tool_progress_mode", "all") == "off": @@ -748,19 +606,20 @@ class CLIStatusBarMixin: def _turn_summary_emit(self) -> None: """Print the post-turn accounting line, when enabled for this surface.""" - from cli import _DIM, _RST, _cprint, logger + from cli import _DIM as _D, _RST, _cprint, logger collector = getattr(self, "_turn_summary_collector", None) if collector is None or not self._turn_summary_is_active(): return try: started = getattr(self, "_turn_summary_start", 0.0) or 0.0 - elapsed = max(0.0, time.monotonic() - started) if started else 0.0 - line = collector.render(elapsed) + line = collector.render(max(0.0, time.monotonic() - started) if started else 0.0) if line: - _cprint(f" {_DIM}{line}{_RST}") + _cprint(f" {_D}{line}{_RST}") except Exception: logger.debug("Turn summary render failed", exc_info=True) + # ── pet pane ────────────────────────────────────────────────────────────── + def _pet_clear_runtime(self) -> None: """Drop renderer + queued Kitty state. Caller holds ``_pet_lock``.""" self._pet_enabled = False @@ -771,38 +630,31 @@ class CLIStatusBarMixin: self._pet_kitty_image_id = 0 def _pet_resolve_config(self) -> None: - """(Re)resolve the active pet from config — picks up live enable/disable/ - - switch made via ``/pet`` or ``hermes pets`` without a restart, mirroring - the TUI's steady poll. Cheap and fail-open: any problem disables the pet. - """ + """(Re)resolve the active pet from config so ``/pet`` / ``hermes pets`` changes apply + without a restart (mirrors the TUI's steady poll). Fail-open: any problem disables.""" try: from agent.pet import constants, store from hermes_cli.config import load_config + from utils import is_truthy_value cfg = load_config() display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} - - from utils import is_truthy_value - enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) slug = str(pet_cfg.get("slug", "") or "") scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) cols = constants.resolve_cols(scale, pet_cfg.get("unicode_cols", 0)) configured_mode = str(pet_cfg.get("render_mode", "auto") or "auto").lower() - # Placeholders only on kitty/Ghostty. WezTerm speaks kitty APC but - # not U+10EEEE — detect_terminal_graphics() still returns kitty - # there, which is why this gate is narrower. - use_kitty = configured_mode in ("", "auto", "kitty") and pet_render.supports_kitty_placeholders() + # Placeholders only on kitty/Ghostty: WezTerm speaks kitty APC but not U+10EEEE + # while detect_terminal_graphics() still says kitty, hence the narrower gate. + use_kitty = ( + configured_mode in ("", "auto", "kitty") and pet_render.supports_kitty_placeholders() + ) renderer_mode = "kitty" if use_kitty else "unicode" - if not enabled or configured_mode == "off": - with self._pet_lock: - self._pet_clear_runtime() - return - - pet = store.resolve_active_pet(slug) + pet = None + if enabled and configured_mode != "off": + pet = store.resolve_active_pet(slug) if pet is None or not pet.exists: with self._pet_lock: self._pet_clear_runtime() @@ -839,13 +691,13 @@ class CLIStatusBarMixin: self._pet_event_until = time.monotonic() + secs def _on_reaction(self, kind: str) -> None: - """User affection (ily / <3 / good bot), core-detected — the pet's share - of the vibe signal that plays hearts on the TUI/desktop. Flash a celebrate.""" + """Core-detected user affection (ily / <3 / good bot): the pet's share of the vibe + signal that plays hearts on the TUI/desktop.""" if kind == "vibe": self._pet_flash("jump") def _pet_react_turn_end(self) -> None: - """Flash the end-of-turn beat: failed on error, jump on a finished plan, else wave.""" + """End-of-turn beat: failed on error, jump on a finished plan, else wave.""" if not self._pet_enabled: return from agent.pet.state import todos_all_done @@ -861,21 +713,15 @@ class CLIStatusBarMixin: self._pet_flash("jump" if done else "wave") def _derive_pet_state(self) -> str: - """Map current CLI activity to a pet animation state. - - A transient reaction beat (wave/jump/failed) wins while it's live; - otherwise the steady state comes from the shared - :func:`agent.pet.state.derive_pet_state` so the CLI can't drift from the - TUI/desktop priority order. - """ + """A live transient beat wins; otherwise the shared ``agent.pet.state.derive_pet_state`` + priority order so the CLI can't drift from the TUI/desktop.""" if self._pet_event and time.monotonic() < self._pet_event_until: return self._pet_event self._pet_event = "" from agent.pet.state import derive_pet_state - # A live blocking modal (approval / clarify / sudo / secret / slash - # confirm) means the agent is paused on the user → the `waiting` pose, - # which outranks the in-flight signals in derive_pet_state. + # Any blocking modal (approval / clarify / sudo / secret / slash confirm) means the + # agent is paused on the user → `waiting`, which outranks the in-flight signals. awaiting_input = bool( self._approval_state or self._clarify_state @@ -883,7 +729,6 @@ class CLIStatusBarMixin: or self._secret_state or getattr(self, "_slash_confirm_state", None) ) - return derive_pet_state( awaiting_input=awaiting_input, busy=getattr(self, "_agent_running", False), @@ -917,8 +762,7 @@ class CLIStatusBarMixin: if renderer is None or renderer.mode != "kitty": return None try: - # PNG encoding is outside _pet_lock: first visit of a state must - # not stall the prompt under the lock. + # PNG encoding outside _pet_lock: first visit of a state must not stall the prompt. payload = renderer.kitty_payload(state, image_id=image_id) except Exception: payload = None @@ -930,11 +774,8 @@ class CLIStatusBarMixin: return payload def _pet_queue_kitty_frame(self, state: str | None = None) -> None: - """Queue one virtual Kitty frame for the next prompt_toolkit render. - - No-op when the pet pane was never initialized (``__new__`` fixtures - and ``_force_full_redraw`` / resize recovery on a pet-less CLI). - """ + """Queue one virtual Kitty frame for the next prompt_toolkit render. No-op when the + pet pane was never initialized (``__new__`` fixtures, redraw on a pet-less CLI).""" if not getattr(self, "_pet_enabled", False): return if state is None: @@ -944,7 +785,8 @@ class CLIStatusBarMixin: return with self._pet_lock: if self._pet_renderer is not None and self._pet_renderer.mode == "kitty": - self._pet_kitty_pending = payload["frames"][self._pet_frame_idx % len(payload["frames"])] + frames = payload["frames"] + self._pet_kitty_pending = frames[self._pet_frame_idx % len(frames)] def _pet_flush_kitty_frame(self, app) -> None: """Write a queued APC after prompt_toolkit has finished its screen diff.""" @@ -960,19 +802,25 @@ class CLIStatusBarMixin: except (OSError, ValueError): pass - def _pet_fragments(self): - """Return prompt_toolkit FormattedText for the current pet frame, or [].""" + def _pet_view(self) -> "tuple[str, bool] | None": + """(state, is_kitty) for the current frame, or None when no pet shows.""" with self._pet_lock: if not self._pet_enabled or self._pet_renderer is None: - return [] - state = self._derive_pet_state() - kitty = self._pet_renderer.mode == "kitty" + return None + return self._derive_pet_state(), self._pet_renderer.mode == "kitty" + + def _pet_fragments(self): + """prompt_toolkit FormattedText for the current pet frame, or [].""" + view = self._pet_view() + if view is None: + return [] + state, kitty = view + frags = [] if kitty: payload = self._pet_kitty_payload_for(state) if not payload: return [] color = pet_render.kitty_color_hex(payload["image_id"]) - frags = [] for y, row in enumerate(payload["placeholder"]): if y: frags.append(("", "\n")) @@ -984,42 +832,38 @@ class CLIStatusBarMixin: return [] grid = grids[self._pet_frame_idx % len(grids)] - frags = [] + def _hex(r, g, b): + return f"#{r:02x}{g:02x}{b:02x}" + for y, row in enumerate(grid): if y: frags.append(("", "\n")) - for top, bottom in row: - tr, tg, tb, ta = top - br, bg, bb, ba = bottom - top_op = ta >= 32 - bot_op = ba >= 32 + for (tr, tg, tb, ta), (br, bg, bb, ba) in row: + top_op, bot_op = ta >= 32, ba >= 32 if not top_op and not bot_op: frags.append(("", " ")) elif top_op and bot_op: - frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x} bg:#{br:02x}{bg:02x}{bb:02x}", "▀")) + frags.append((f"fg:{_hex(tr, tg, tb)} bg:{_hex(br, bg, bb)}", "▀")) elif top_op: - # Upper half only — leave the lower half the terminal's bg - # instead of painting it black (cleaner on light themes). - frags.append((f"fg:#{tr:02x}{tg:02x}{tb:02x}", "▀")) + # Upper half only — leave the lower half the terminal's bg (cleaner on light + # themes). + frags.append((f"fg:{_hex(tr, tg, tb)}", "▀")) else: - frags.append((f"fg:#{br:02x}{bg:02x}{bb:02x}", "▄")) + frags.append((f"fg:{_hex(br, bg, bb)}", "▄")) return frags def _pet_widget_height(self) -> int: """Visible rows for the pet window — 0 collapses it when no pet shows.""" - with self._pet_lock: - if not self._pet_enabled or self._pet_renderer is None: - return 0 - state = self._derive_pet_state() - kitty = self._pet_renderer.mode == "kitty" + view = self._pet_view() + if view is None: + return 0 + state, kitty = view if kitty: payload = self._pet_kitty_payload_for(state) return int(payload.get("rows", 0)) if payload else 0 with self._pet_lock: grids = self._pet_frames_for(state) - if not grids or not grids[0]: - return 0 - return len(grids[0]) + return len(grids[0]) if grids and grids[0] else 0 def _pet_anim_loop(self) -> None: """Advance the frame + invalidate on a timer while a pet is enabled.""" @@ -1055,7 +899,11 @@ class CLIStatusBarMixin: return self._pet_resolve_config() with self._pet_lock: - kitty = self._pet_enabled and self._pet_renderer is not None and self._pet_renderer.mode == "kitty" + kitty = ( + self._pet_enabled + and self._pet_renderer is not None + and self._pet_renderer.mode == "kitty" + ) if kitty: self._pet_queue_kitty_frame() self._pet_anim_running = True @@ -1069,35 +917,18 @@ class CLIStatusBarMixin: thread.join(timeout=0.3) self._pet_anim_thread = None + # ── voice ───────────────────────────────────────────────────────────────── + def _voice_record_key_label(self) -> str: - """Return the configured voice push-to-talk key formatted for UI. - - Shared helper so every voice-facing status line / placeholder / - recording hint advertises the SAME label as the registered - prompt_toolkit binding. - - Cached at startup (see ``set_voice_record_key_cache``) rather - than re-read per render. Two reasons (Copilot round-13 on - #19835): - - * The prompt_toolkit binding is registered once at session - start via ``@kb.add(_voice_key)``; re-reading config per - render meant the status bar could advertise a new shortcut - after a config edit while the actual binding was still the - startup chord — exactly the display/binding drift this PR - is trying to eliminate. - * The label is on the hot render path (status bar + composer - placeholder invalidated every 150ms during recording), so - reading config on every call added avoidable UI overhead. - """ + """The push-to-talk key label every voice-facing hint advertises. Cached at startup + (``set_voice_record_key_cache``) because the prompt_toolkit binding is registered once — + re-reading config per render could advertise a chord that isn't bound — and this sits on + the hot render path.""" return getattr(self, "_voice_record_key_display_cache", None) or "Ctrl+B" def set_voice_record_key_cache(self, raw_key: object) -> None: - """Populate the voice label cache from a raw ``voice.record_key``. - - Called at CLI startup after the prompt_toolkit binding is - registered so the cached label always matches the live binding. - """ + """Populate the voice label cache from a raw ``voice.record_key``; called after the + prompt_toolkit binding is registered so the label matches the live binding.""" try: from hermes_cli.voice import format_voice_record_key_for_status self._voice_record_key_display_cache = format_voice_record_key_for_status(raw_key) @@ -1105,7 +936,7 @@ class CLIStatusBarMixin: self._voice_record_key_display_cache = "Ctrl+B" def _get_voice_status_fragments(self, width: Optional[int] = None): - """Return the voice status bar fragments for the interactive TUI.""" + """Voice status bar fragments for the interactive TUI.""" width = width or self._get_tui_terminal_width() compact = self._use_minimal_tui_chrome(width=width) label = self._voice_record_key_label() @@ -1114,45 +945,33 @@ class CLIStatusBarMixin: return [("class:voice-status-recording", " ● REC ")] return [("class:voice-status-recording", f" ● REC {label} to stop ")] if self._voice_processing: - if compact: - return [("class:voice-status", " ◉ STT ")] - return [("class:voice-status", " ◉ Transcribing... ")] + return [("class:voice-status", " ◉ STT " if compact else " ◉ Transcribing... ")] if compact: return [("class:voice-status", f" 🎤 {label} ")] tts = " | TTS on" if self._voice_tts else "" cont = " | Continuous" if self._voice_continuous else "" return [("class:voice-status", f" 🎤 Voice mode{tts}{cont} — {label} to record ")] + # ── status bar rendering ────────────────────────────────────────────────── + @staticmethod def _status_bar_goal_segment(snapshot: Dict[str, Any]) -> str: - """Return the ``⊙ goal 3/20`` segment, or ``""`` when no goal is active. - - Active-goal-only by design: paused/done goals don't occupy status-bar - real estate (they already print their own glyph lines in the thread). - """ + """``⊙ goal 3/20`` while a goal is active, else ``""`` (paused/done goals already + print their own glyph lines in the thread).""" if not snapshot.get("goal_active"): return "" used = snapshot.get("goal_turns_used") or 0 max_turns = snapshot.get("goal_max_turns") or 0 - if max_turns: - return f"⊙ goal {used}/{max_turns}" - return "⊙ goal" + return f"⊙ goal {used}/{max_turns}" if max_turns else "⊙ goal" def _get_status_bar_field_set(self) -> Optional[frozenset]: - """Return the set of visible status-bar fields from config. + """Visible status-bar fields from ``display.status_bar.fields`` (module-level + ``CLI_CONFIG``; no per-render YAML parse). ``None`` = not customized, show everything. - Reads ``display.status_bar.fields`` from the module-level - ``CLI_CONFIG`` (no per-render YAML parse — the status bar repaints - every frame). Returns ``None`` when the user has not customized the - bar (use built-in defaults, i.e. show everything), or a - ``frozenset`` of field names when the list is non-empty. - - Available fields: model, context_detail, context_pct, cache_hit, - latency, tps, compressions, bg_tasks, bg_processes, bg_subagents, - goal, duration, prompt_elapsed, idle_since, focus, yolo, stash, - battery, title, total_tokens. - ``total_tokens`` is opt-in only (never shown by default). - The field order is fixed; the config controls visibility only. + Fields: model, context_detail, context_pct, cache_hit, latency, tps, compressions, + bg_tasks, bg_processes, bg_subagents, goal, duration, prompt_elapsed, idle_since, + focus, yolo, stash, battery, title, total_tokens (opt-in only). Order is fixed; the + config controls visibility only. """ from cli import CLI_CONFIG if hasattr(self, "_status_bar_field_set_cache"): @@ -1169,378 +988,199 @@ class CLIStatusBarMixin: self._status_bar_field_set_cache = result return result - def _build_status_bar_text(self, width: Optional[int] = None) -> str: - """Return a compact one-line session status string for the TUI footer.""" + def _status_bar_segments( + self, snapshot, width: int, field_set, yolo_active: bool, *, styled: bool + ) -> list: + """Ordered status-bar segments for one width tier (<52 / <76 / wide), each a list of + ``(style, text)`` fragments. Shared by the plain-text and prompt_toolkit renderers so + the two can never drift; ``styled`` selects the graphical context bar.""" from cli import format_token_count_compact - try: - snapshot = self._get_status_bar_snapshot() - if width is None: - width = self._get_tui_terminal_width() - percent = snapshot["context_percent"] + model_short = snapshot["model_short"] + duration_label = snapshot["duration"] + goal_segment = self._status_bar_goal_segment(snapshot) + focus_label = snapshot.get("focus_label") or "" + + def _ok(name: str) -> bool: + return field_set is None or name in field_set + + segs: list = [] + + def add(name, style, text): + if _ok(name): + segs.append([(style, text)]) + + def add_count(name, key, glyph, style=_STRONG): + count = snapshot.get(key, 0) + if count: + add(name, style(count) if callable(style) else style, f"{glyph} {count}") + + percent = snapshot["context_percent"] + if _ok("model"): + if styled: + segs.append([(_SB, " ⚕ "), (_STRONG, model_short)]) + else: + segs.append([("", f"⚕ {model_short}")]) + narrow, wide = width < 52, width >= 76 + if narrow: + # Narrow bars put duration ahead of the goal segment; the other tiers reverse it. + add("duration", _DIM, duration_label) + else: percent_label = f"{percent}%" if percent is not None else "--" - duration_label = snapshot["duration"] - battery_label = snapshot.get("battery_label") or "" - battery_prefix = f"{battery_label} │ " if battery_label else "" - focus_label = snapshot.get("focus_label") or "" - session_title = snapshot.get("session_title") or "" - - yolo_active = self._is_session_yolo_active() - goal_segment = self._status_bar_goal_segment(snapshot) - field_set = self._get_status_bar_field_set() - - def _ok(name: str) -> bool: - return field_set is None or name in field_set - - if not _ok("title"): - session_title = "" - - if not _ok("goal"): - goal_segment = "" - if not _ok("focus"): - focus_label = "" - if width < 52: - segs = [] - if _ok("model"): - segs.append(f"⚕ {snapshot['model_short']}") - if _ok("duration"): - segs.append(duration_label) - if goal_segment: - segs.append(goal_segment) - if focus_label: - segs.append(focus_label) - if yolo_active and _ok("yolo"): - segs.append("⚠ YOLO") - text = battery_prefix + " · ".join(segs) if segs else f"{battery_prefix}⚕ {snapshot['model_short']}" - return self._right_align_status_title(text, session_title, width) - if width < 76: - parts = [] - if _ok("model"): - parts.append(f"⚕ {snapshot['model_short']}") - if _ok("context_pct"): - parts.append(percent_label) - cache = self._cache_hit_rate(snapshot, precision=0) - if cache and _ok("cache_hit"): - parts.append(cache[1]) - if battery_label: - parts.insert(0, battery_label) - compressions = snapshot.get("compressions", 0) - if compressions and _ok("compressions"): - parts.append(f"🗜️ {compressions}") - bg_count = snapshot.get("active_background_tasks", 0) - if bg_count and _ok("bg_tasks"): - parts.append(f"▶ {bg_count}") - bg_proc_count = snapshot.get("active_background_processes", 0) - if bg_proc_count and _ok("bg_processes"): - parts.append(f"⚙ {bg_proc_count}") - bg_subagent_count = snapshot.get("active_background_subagents", 0) - if bg_subagent_count and _ok("bg_subagents"): - parts.append(f"⛓ {bg_subagent_count}") - if goal_segment: - parts.append(goal_segment) - if _ok("duration"): - parts.append(duration_label) - if focus_label: - parts.append(focus_label) - if yolo_active and _ok("yolo"): - parts.append("⚠ YOLO") - if not parts: - parts = [f"⚕ {snapshot['model_short']}"] - return self._right_align_status_title(" · ".join(parts), session_title, width) - - parts = [] - if _ok("model"): - parts.append(f"⚕ {snapshot['model_short']}") - if _ok("context_detail"): + if wide and _ok("context_detail"): if snapshot["context_length"]: ctx_total = _format_context_length(snapshot["context_length"]) ctx_used = format_token_count_compact(snapshot["context_tokens"]) context_label = f"{ctx_used}/{ctx_total}" else: context_label = "ctx --" - parts.append(context_label) + segs.append([(_DIM, context_label)]) if _ok("context_pct"): - parts.append(percent_label) - if battery_label: - parts.insert(0, battery_label) - compressions = snapshot.get("compressions", 0) - cache = self._cache_hit_rate(snapshot) - if cache and _ok("cache_hit"): - parts.append(cache[1]) - _avg_lat = snapshot.get("avg_latency_label") or "" - if _avg_lat and _ok("latency"): - parts.append(f"◷ {_avg_lat}") - _avg_vel = snapshot.get("avg_velocity_label") or "" - if _avg_vel and _ok("tps"): - parts.append(f"↑ {_avg_vel}") - if compressions and _ok("compressions"): - parts.append(f"🗜️ {compressions}") - bg_count = snapshot.get("active_background_tasks", 0) - if bg_count and _ok("bg_tasks"): - parts.append(f"▶ {bg_count}") - bg_proc_count = snapshot.get("active_background_processes", 0) - if bg_proc_count and _ok("bg_processes"): - parts.append(f"⚙ {bg_proc_count}") - bg_subagent_count = snapshot.get("active_background_subagents", 0) - if bg_subagent_count and _ok("bg_subagents"): - parts.append(f"⛓ {bg_subagent_count}") - if goal_segment: - parts.append(goal_segment) - if _ok("duration"): - parts.append(duration_label) - prompt_elapsed = snapshot.get("prompt_elapsed") - if prompt_elapsed and _ok("prompt_elapsed"): - parts.append(prompt_elapsed) - idle_since = snapshot.get("idle_since") - if idle_since and _ok("idle_since"): - parts.append(idle_since) - if focus_label: - parts.append(focus_label) - if yolo_active and _ok("yolo"): - parts.append("⚠ YOLO") - # Session token total (Σ) — opt-in only via an explicit fields - # list, so default bars never widen. + bar_style = self._status_bar_context_style(percent) + if wide and styled: + segs.append([ + (bar_style, self._build_context_bar(percent)), (_DIM, " "), (bar_style, percent_label), + ]) + else: + segs.append([(bar_style, percent_label)]) + cache = self._cache_hit_rate(snapshot, precision=1 if wide else 0) + if cache: + add("cache_hit", self._cache_hit_rate_style(cache[0]), cache[1]) + if wide: + for name, key, glyph in ( + ("latency", "avg_latency_label", "◷"), ("tps", "avg_velocity_label", "↑"), + ): + label = snapshot.get(key) or "" + if label: + add(name, _DIM, f"{glyph} {label}") + add_count("compressions", "compressions", "🗜️", self._compression_count_style) + add_count("bg_tasks", "active_background_tasks", "▶") + add_count("bg_processes", "active_background_processes", "⚙") + add_count("bg_subagents", "active_background_subagents", "⛓") + if goal_segment: + add("goal", _STRONG, goal_segment) + if not narrow: + add("duration", _DIM, duration_label) + if wide: + for name in ("prompt_elapsed", "idle_since"): + label = snapshot.get(name) + if label: + add(name, _DIM, label) + if focus_label: + add("focus", _STRONG, focus_label) + if yolo_active: + add("yolo", "class:status-bar-yolo", "⚠ YOLO") + if wide: + # Session token total (Σ) — opt-in only via an explicit fields list. total_tokens = snapshot.get("session_total_tokens", 0) if total_tokens and field_set is not None and "total_tokens" in field_set: - parts.append(f"Σ{format_token_count_compact(total_tokens)}") + segs.append([(_DIM, f"Σ{format_token_count_compact(total_tokens)}")]) + return segs + + def _build_status_bar_text(self, width: Optional[int] = None) -> str: + """Compact one-line session status string for the TUI footer.""" + try: + snapshot = self._get_status_bar_snapshot() + if width is None: + width = self._get_tui_terminal_width() + model_short = snapshot["model_short"] + battery_label = snapshot.get("battery_label") or "" + field_set = self._get_status_bar_field_set() + show_title = field_set is None or "title" in field_set + session_title = (snapshot.get("session_title") or "") if show_title else "" + segs = self._status_bar_segments( + snapshot, width, field_set, self._is_session_yolo_active(), styled=False + ) + parts = ["".join(t for _, t in seg) for seg in segs] + if width < 52: + battery_prefix = f"{battery_label} │ " if battery_label else "" + text = f"{battery_prefix}⚕ {model_short}" + if parts: + text = battery_prefix + " · ".join(parts) + return self._right_align_status_title(text, session_title, width) + if battery_label: + parts.insert(0, battery_label) if not parts: - parts = [f"⚕ {snapshot['model_short']}"] - return self._right_align_status_title(" │ ".join(parts), session_title, width) + parts = [f"⚕ {model_short}"] + sep = " · " if width < 76 else " │ " + return self._right_align_status_title(sep.join(parts), session_title, width) except Exception: return f"⚕ {self.model if getattr(self, 'model', None) else 'Hermes'}" def _get_status_bar_fragments(self): - from cli import format_token_count_compact - if not self._status_bar_visible or getattr(self, '_model_picker_state', None) or getattr(self, '_command_palette_state', None): + if ( + not self._status_bar_visible + or getattr(self, "_model_picker_state", None) + or getattr(self, "_command_palette_state", None) + ): return [] try: snapshot = self._get_status_bar_snapshot() - # Use prompt_toolkit's own terminal width when running inside the - # TUI — shutil.get_terminal_size() can return stale or fallback - # values (especially on SSH) that differ from what prompt_toolkit - # actually renders, causing the fragments to overflow to a second - # line and produce duplicated status bar rows over long sessions. + # prompt_toolkit's own width: shutil's can be stale (esp. over SSH) and an overflow + # produces duplicated status-bar rows over long sessions. width = self._get_tui_terminal_width() - duration_label = snapshot["duration"] - yolo_active = self._is_session_yolo_active() - goal_segment = self._status_bar_goal_segment(snapshot) - battery_label = snapshot.get("battery_label") or "" - battery_style = self._battery_status_style(snapshot.get("battery_category", "dim")) - focus_label = snapshot.get("focus_label") or "" - session_title = snapshot.get("session_title") or "" field_set = self._get_status_bar_field_set() def _ok(name: str) -> bool: return field_set is None or name in field_set - if not _ok("title"): - session_title = "" + session_title = (snapshot.get("session_title") or "") if _ok("title") else "" + segs = self._status_bar_segments( + snapshot, width, field_set, self._is_session_yolo_active(), styled=True + ) + sep = " · " if width < 76 else " │ " + frags: list = [] + for seg in segs: + if frags: + frags.append((_DIM, sep)) + frags.extend(seg) + if not frags: + frags = [(_SB, " ⚕ "), (_STRONG, snapshot["model_short"])] + frags.append((_SB, " ")) - if not _ok("goal"): - goal_segment = "" - if not _ok("focus"): - focus_label = "" - - def _append(frag_list, sep, *pieces): - if frag_list: - frag_list.append(("class:status-bar-dim", sep)) - frag_list.extend(pieces) - - if width < 52: - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("duration"): - _append(frags, " · ", ("class:status-bar-dim", duration_label)) - if goal_segment: - _append(frags, " · ", ("class:status-bar-strong", goal_segment)) - if focus_label: - _append(frags, " · ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - else: - percent = snapshot["context_percent"] - percent_label = f"{percent}%" if percent is not None else "--" - if width < 76: - compressions = snapshot.get("compressions", 0) - bg_count = snapshot.get("active_background_tasks", 0) - bg_proc_count = snapshot.get("active_background_processes", 0) - bg_subagent_count = snapshot.get("active_background_subagents", 0) - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("context_pct"): - _append(frags, " · ", (self._status_bar_context_style(percent), percent_label)) - cache = self._cache_hit_rate(snapshot, precision=0) - if cache and _ok("cache_hit"): - _append(frags, " · ", (self._cache_hit_rate_style(cache[0]), cache[1])) - if compressions and _ok("compressions"): - _append(frags, " · ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) - if bg_count and _ok("bg_tasks"): - _append(frags, " · ", ("class:status-bar-strong", f"▶ {bg_count}")) - if bg_proc_count and _ok("bg_processes"): - _append(frags, " · ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) - if bg_subagent_count and _ok("bg_subagents"): - _append(frags, " · ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) - if goal_segment: - _append(frags, " · ", ("class:status-bar-strong", goal_segment)) - if _ok("duration"): - _append(frags, " · ", ("class:status-bar-dim", duration_label)) - if focus_label: - _append(frags, " · ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " · ", ("class:status-bar-yolo", "⚠ YOLO")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - else: - bar_style = self._status_bar_context_style(percent) - compressions = snapshot.get("compressions", 0) - bg_count = snapshot.get("active_background_tasks", 0) - bg_proc_count = snapshot.get("active_background_processes", 0) - bg_subagent_count = snapshot.get("active_background_subagents", 0) - frags = [] - if _ok("model"): - frags.append(("class:status-bar", " ⚕ ")) - frags.append(("class:status-bar-strong", snapshot["model_short"])) - if _ok("context_detail"): - if snapshot["context_length"]: - ctx_total = _format_context_length(snapshot["context_length"]) - ctx_used = format_token_count_compact(snapshot["context_tokens"]) - context_label = f"{ctx_used}/{ctx_total}" - else: - context_label = "ctx --" - _append(frags, " │ ", ("class:status-bar-dim", context_label)) - if _ok("context_pct"): - _append( - frags, - " │ ", - (bar_style, self._build_context_bar(percent)), - ("class:status-bar-dim", " "), - (bar_style, percent_label), - ) - cache = self._cache_hit_rate(snapshot) - if cache and _ok("cache_hit"): - _append(frags, " │ ", (self._cache_hit_rate_style(cache[0]), cache[1])) - _avg_lat = snapshot.get("avg_latency_label") or "" - if _avg_lat and _ok("latency"): - _append(frags, " │ ", ("class:status-bar-dim", f"◷ {_avg_lat}")) - _avg_vel = snapshot.get("avg_velocity_label") or "" - if _avg_vel and _ok("tps"): - _append(frags, " │ ", ("class:status-bar-dim", f"↑ {_avg_vel}")) - if compressions and _ok("compressions"): - _append(frags, " │ ", (self._compression_count_style(compressions), f"🗜️ {compressions}")) - if bg_count and _ok("bg_tasks"): - _append(frags, " │ ", ("class:status-bar-strong", f"▶ {bg_count}")) - if bg_proc_count and _ok("bg_processes"): - _append(frags, " │ ", ("class:status-bar-strong", f"⚙ {bg_proc_count}")) - if bg_subagent_count and _ok("bg_subagents"): - _append(frags, " │ ", ("class:status-bar-strong", f"⛓ {bg_subagent_count}")) - if goal_segment: - _append(frags, " │ ", ("class:status-bar-strong", goal_segment)) - if _ok("duration"): - _append(frags, " │ ", ("class:status-bar-dim", duration_label)) - # Position 7: per-prompt elapsed timer (live or frozen) - prompt_elapsed = snapshot.get("prompt_elapsed") - if prompt_elapsed and _ok("prompt_elapsed"): - _append(frags, " │ ", ("class:status-bar-dim", prompt_elapsed)) - # Position 8: idle time since the last final agent response - idle_since = snapshot.get("idle_since") - if idle_since and _ok("idle_since"): - _append(frags, " │ ", ("class:status-bar-dim", idle_since)) - # Persistent focus-view badge — so the reduced-output mode - # is never invisible (mirrors the YOLO badge convention). - if focus_label: - _append(frags, " │ ", ("class:status-bar-strong", focus_label)) - if yolo_active and _ok("yolo"): - _append(frags, " │ ", ("class:status-bar-yolo", "⚠ YOLO")) - # Session token total (Σ) — opt-in only via an explicit - # fields list, so default bars never widen. - total_tokens = snapshot.get("session_total_tokens", 0) - if total_tokens and field_set is not None and "total_tokens" in field_set: - _append(frags, " │ ", ("class:status-bar-dim", f"Σ{format_token_count_compact(total_tokens)}")) - if not frags: - frags = [ - ("class:status-bar", " ⚕ "), - ("class:status-bar-strong", snapshot["model_short"]), - ] - frags.append(("class:status-bar", " ")) - - # Stash indicator (📌 N) — appended after all width tiers so the - # user always knows a parked draft exists, even on narrow - # terminals. Placed before the battery prepend so it stays at the - # right edge, and it is the first thing the width trim below drops - # if the bar genuinely cannot fit. + # Stash indicator (📌 N) after every width tier so a parked draft is never + # invisible; before the battery prepend, and the first thing the trim drops. try: stash_indicator = self._prompt_stash.indicator() except Exception: stash_indicator = "" if stash_indicator and _ok("stash"): - # Insert before the trailing pad fragment so the bar keeps its - # one-cell right margin. - if frags and frags[-1] == ("class:status-bar", " "): - frags[-1:-1] = [ - ("class:status-bar-dim", " · "), - ("class:status-bar-strong", stash_indicator), - ] + pieces = [(_DIM, " · "), (_STRONG, stash_indicator)] + if frags and frags[-1] == (_SB, " "): + frags[-1:-1] = pieces # keep the one-cell right margin else: - frags.append(("class:status-bar-dim", " · ")) - frags.append(("class:status-bar-strong", stash_indicator)) + frags.extend(pieces) - # Battery is the first status-bar element when enabled: prepend it - # ahead of the leading ⚕ marker in whichever width tier ran above. + # Battery is the first element when enabled: prepend ahead of the ⚕ marker. + battery_label = snapshot.get("battery_label") or "" if battery_label and _ok("battery"): - frags[0:0] = [ - ("class:status-bar", " "), - (battery_style, battery_label), - ("class:status-bar-dim", " │"), - ] + battery_style = self._battery_status_style(snapshot.get("battery_category", "dim")) + frags[0:0] = [(_SB, " "), (battery_style, battery_label), (_DIM, " │")] frags = self._right_align_status_title_fragments(frags, session_title, width) - total_width = sum(self._status_bar_display_width(text) for _, text in frags) if total_width > width: plain_text = "".join(text for _, text in frags) - trimmed = self._trim_status_bar_text(plain_text, width) - return [("class:status-bar", trimmed)] + return [(_SB, self._trim_status_bar_text(plain_text, width))] return frags except Exception: - return [("class:status-bar", f" {self._build_status_bar_text()} ")] + return [(_SB, f" {self._build_status_bar_text()} ")] + + # ── prompt stash panel ──────────────────────────────────────────────────── @staticmethod def _fmt_stash_age(stashed_at: float) -> str: - """Return human-readable age string for a stash entry.""" - import time as _t - secs = int(_t.monotonic() - stashed_at) + secs = int(time.monotonic() - stashed_at) if secs < 10: return "just now" if secs < 90: return f"{secs}s ago" mins = secs // 60 - if mins < 60: - return f"{mins} min ago" - return f"{mins // 60}h ago" + return f"{mins} min ago" if mins < 60 else f"{mins // 60}h ago" def _render_stash_panel(self, stash_list: list, cursor: int, width: int) -> list: - """Return prompt_toolkit formatted_text fragments for the stash panel box. - - Every horizontal measurement goes through ``_status_bar_display_width`` - (prompt_toolkit's ``get_cwidth``) rather than ``len()``. The header - contains 📌, which is one Python codepoint but two terminal cells; the - original PR chased that off-by-one through three successive - "subtract 1 from len()" commits. Measuring in display cells fixes it - for real and keeps CJK previews from bleeding past the right border. - """ + """prompt_toolkit fragments for the stash panel box. Every horizontal measurement uses + ``_status_bar_display_width`` (display cells), not ``len()`` — 📌 is one codepoint but + two cells, and CJK previews would otherwise bleed past the right border.""" cw = self._status_bar_display_width W = max(12, min(width - 4, 80)) @@ -1550,9 +1190,8 @@ class CLIStatusBarMixin: FTR_PREFIX = "╰" FTR_SUFFIX = " ↑↓ Enter=restore D=delete Esc ─╯" - # On narrow terminals the full hint text is wider than the box itself. - # Drop to compact affordances rather than letting the frame bleed past - # the right edge (which is what made the panel look broken). + # On narrow terminals the full hint text is wider than the box: drop to compact + # affordances rather than letting the frame bleed past the right edge. if cw(hdr_prefix_str) + cw(HDR_SUFFIX) > W: hdr_prefix_str = f"╭─ 📌 {n} " HDR_SUFFIX = "─╮" @@ -1563,35 +1202,29 @@ class CLIStatusBarMixin: hdr_dashes = max(0, W - cw(hdr_prefix_str) - cw(HDR_SUFFIX)) ftr_dashes = max(0, W - cw(FTR_PREFIX) - cw(FTR_SUFFIX)) - - # Row inner width: W minus the two '│' border cells. - INNER = W - 2 - + INNER = W - 2 # minus the two '│' border cells frags: list = [] def line(text: str, style: str = "") -> None: - # Final guard: never emit a line wider than the box, whatever the - # label lengths worked out to. + # Final guard: never emit a line wider than the box. frags.append((style, self._trim_status_bar_text(text, W) + "\n")) line(f"{hdr_prefix_str}{'─' * hdr_dashes}{HDR_SUFFIX}", "class:subagent-border") - for i, item in enumerate(stash_list): age = self._fmt_stash_age(item["stashed_at"]) - # Row: " ► [N] {age:<10} {preview} " - prefix = f" {'►' if i == cursor else ' '} [{i + 1}] {age:<10} " + marker = "►" if i == cursor else " " + prefix = f" {marker} [{i + 1}] {age:<10} " if cw(prefix) > INNER - 2: - prefix = f" {'►' if i == cursor else ' '} [{i + 1}] " + prefix = f" {marker} [{i + 1}] " avail = max(0, INNER - cw(prefix) - 1) preview = self._trim_status_bar_text(item.get("preview") or "", avail) preview = preview + " " * max(0, avail - cw(preview)) - row = self._trim_status_bar_text(f"│{prefix}{preview} │", W) if i == cursor: + row = self._trim_status_bar_text(f"│{prefix}{preview} │", W) frags.append(("class:subagent-selected", row + "\n")) else: frags.append(("class:subagent-border", "│")) frags.append(("class:subagent-sub", f"{prefix}{preview} ")) frags.append(("class:subagent-border", "│\n")) - line(f"{FTR_PREFIX}{'─' * ftr_dashes}{FTR_SUFFIX}", "class:subagent-border") return frags