From be5c6a2fd8e6bbadf51d8023d303cc129d1f5b45 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:28:22 -0700 Subject: [PATCH] refactor(agent/prompt): remove dead code, unify duplicated helpers, compact docstrings across prompt/skill/redaction modules Dead (zero refs): coding_system_blocks, get_friendly_tool_labels, get_scan_ordered_skills_dirs, _project_quarantine_cache_clear, clear_stable_prefixes, _redact_http_request_target_query_params, _has_http_method_substring, PromptCachePlan.marker_count, display _diff_* colour thunks (-> _diff_ansi), pass-through RedactingFormatter.__init__. Unified: _slugify -> slugify_skill_name; reload diff -> diff_command_snapshots; _is_summary_item -> is_compaction_summary_message alias; sanitizer walkers -> _sanitize_messages/_sanitize_structure; assignment redaction passes -> _redact_assignments/_should_redact_assignment; quiet-mode tool lines -> _CUTE_LINES table. --- agent/coding_context.py | 450 +++----- agent/context_references.py | 259 ++--- agent/display.py | 950 ++++++----------- agent/i18n.py | 152 +-- agent/markdown_tables.py | 227 ++-- agent/message_sanitization.py | 805 ++++---------- agent/native_compaction.py | 302 ++---- agent/onboarding.py | 184 ++-- agent/plan_prompt.py | 37 +- agent/prompt_cache_boundary.py | 82 +- agent/prompt_cache_scope.py | 242 ++--- agent/prompt_caching.py | 300 ++---- agent/redact.py | 1181 ++++++++------------- agent/replay_cleanup.py | 200 ++-- agent/runtime_cwd.py | 86 +- agent/skill_bundles.py | 246 +---- agent/skill_commands.py | 762 ++++++------- agent/skill_utils.py | 787 +++++--------- agent/subdirectory_hints.py | 193 +--- agent/think_scrubber.py | 318 ++---- tests/agent/test_project_skills.py | 16 +- tests/agent/test_prompt_cache_boundary.py | 8 +- tests/agent/test_prompt_caching.py | 8 +- 23 files changed, 2570 insertions(+), 5225 deletions(-) diff --git a/agent/coding_context.py b/agent/coding_context.py index 0eb41c85b0..ce8f6c8cd3 100644 --- a/agent/coding_context.py +++ b/agent/coding_context.py @@ -1,52 +1,29 @@ """Coding-context awareness — base Hermes, every interactive surface. -When the user runs Hermes inside a code workspace (CLI, TUI, desktop app, or an -editor over ACP), Hermes shifts into a **coding posture**. This module is the -single place that decides whether we're in that posture and what it implies, -so the rest of the codebase never re-derives "are we coding?" on its own. +When Hermes runs inside a code workspace (CLI, TUI, desktop, ACP editor) it +shifts into a **coding posture**. This module is the single place that decides +whether we're in that posture and what it implies, so nothing else re-derives +"are we coding?". The posture is a frozen :class:`RuntimeMode` selected from a +small :class:`ContextProfile` registry (``coding`` / ``general``); a profile is +*data* (toolset, operating brief, skill-index hints) that every domain reads: -Architecture — one seam, many consumers ----------------------------------------- -The posture is modelled as a frozen :class:`RuntimeMode` selected from a small -:class:`ContextProfile` registry (today: ``coding`` and ``general``). A profile -is *data* — it declares the toolset to collapse to, the operating brief to -inject, and hints for other domains (model routing, memory, subagents). Every -domain reads the same resolved object instead of probing git/config itself: + * System prompt — ``RuntimeMode.system_prompt_parts()`` → operating brief + + live git/workspace snapshot (``agent/system_prompt.py``). + * Toolset — ``RuntimeMode.toolset_selection()`` → ``coding`` toolset + enabled + MCP servers, ONLY under the opt-in ``focus`` mode. The default posture is + prompt-only and never strips a toolset the user explicitly enabled. + * Delegation — subagents inherit the toolset and prompt builder, so the + posture propagates for free. - * **System prompt** — ``RuntimeMode.system_blocks()`` → the operating brief + - a live git/workspace snapshot (``agent/system_prompt.py``). - * **Toolset** — ``RuntimeMode.toolset_selection()`` → the ``coding`` toolset - plus the user's enabled MCP servers (``cli.py`` / ``tui_gateway``). Only - under the opt-in ``focus`` mode: the default posture is prompt-only and - never touches the user's configured toolsets (toolsets like messaging / - smart-home / music are off-by-default anyway, and someone who explicitly - enabled image-gen or Spotify shouldn't lose it for being in a git repo). - * **Delegation** — subagents inherit the parent's toolset and run through the - same prompt builder, so the coding posture propagates to children for free. - * **Model / memory / compression** — declared on the profile - (``model_hint``, ``memory_policy``) as the extension seam; consumers read - ``mode.profile`` rather than re-deciding. +Cache safety: the mode is resolved once and immutable; the workspace snapshot +is built once at prompt-build time and never re-probed per turn (the brief +tells the model to re-check with ``git``). A ``/coding`` flip takes effect next +session. -Cache safety ------------- -The mode is resolved **once** and is immutable. The workspace snapshot is built -once at prompt-build time and baked into the *stable* system-prompt tier — never -re-probed per turn (that would shatter the prompt cache). Branch and dirty state -drift mid-session, so the brief tells the model to re-check with ``git`` before -acting on the snapshot. A ``/coding`` flip therefore only takes effect next -session (deferred), the same contract as ``/skills install`` vs ``--now``. - -Activation (config ``agent.coding_context``): - - * ``auto`` (default) — posture (brief + snapshot) on an interactive coding - surface sitting in a code workspace (git repo or recognised project root). - Prompt-only; toolsets and the skill index untouched. - * ``focus`` — like ``auto``, but additionally collapses the toolset to the - ``coding`` set + enabled MCP servers and demotes non-coding skill - categories to names-only in the prompt's skill index (no skill is ever - hidden). Explicit opt-in for a lean schema. - * ``on`` — force the posture anywhere (incl. non-workspaces). Prompt-only. - * ``off`` — disable entirely. +Activation (config ``agent.coding_context``): ``auto`` (default) — posture on an +interactive surface in a code workspace, prompt-only; ``focus`` — also collapse +the toolset and demote non-coding skill categories to names-only (never +hidden); ``on`` — force the posture anywhere; ``off`` — disable. """ from __future__ import annotations @@ -67,8 +44,7 @@ logger = logging.getLogger("hermes.coding_context") CODING_TOOLSET = "coding" # Surfaces where a coding posture makes sense under ``auto``. Messaging -# platforms (telegram, discord, slack, …) are intentionally absent — a chat bot -# in a group is not pair-programming. +# platforms are intentionally absent — a chat bot in a group is not pairing. INTERACTIVE_CODING_PLATFORMS = {"cli", "tui", "acp", "desktop", ""} # Project-root signals that mark a directory as a code workspace even when it @@ -85,10 +61,8 @@ _PROJECT_MARKERS = ( # Agent-instruction files surfaced separately from manifests in the snapshot. _CONTEXT_FILES = ("AGENTS.md", "CLAUDE.md", ".cursorrules") -# Source-file extensions that make a git repo a *code* workspace even with no -# manifest. Without this, `git init` on a notes/writing/research folder (a huge -# non-coding use case) would flip the whole session into the coding posture just -# for having a `.git`. A manifest still wins on its own (see `_PROJECT_MARKERS`). +# Source extensions that make a manifest-less git repo a *code* workspace, so +# `git init` on a notes/writing folder does not flip the session into coding. _CODE_EXTENSIONS = frozenset({ ".py", ".pyi", ".ipynb", ".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs", ".go", ".rs", ".java", ".kt", ".kts", ".scala", ".rb", ".php", ".c", ".h", @@ -97,24 +71,16 @@ _CODE_EXTENSIONS = frozenset({ ".hs", ".clj", ".erl", ".pl", }) -# Dirs never worth scanning for the code check (deps/build/vcs/venv noise). _CODE_SCAN_SKIP_DIRS = frozenset({ ".git", "node_modules", "venv", ".venv", "__pycache__", "dist", "build", "target", ".next", ".turbo", "vendor", }) - # Bounded sweep: a code workspace reveals itself in the first handful of entries. _CODE_SCAN_MAX_ENTRIES = 500 def _has_code_files(root: Path) -> bool: - """Cheap, bounded check for source files in a repo's top two levels. - - Lets a git repo of loose scripts (no manifest) still read as a code - workspace while a bare notes/writing repo does not. Scans the root and its - immediate subdirectories only, capped at ``_CODE_SCAN_MAX_ENTRIES`` stats — - a handful of readdirs at session start, not a full walk. - """ + """Bounded check for source files in the root and its immediate subdirs.""" seen = 0 stack = [(root, True)] while stack: @@ -138,6 +104,7 @@ def _has_code_files(root: Path) -> bool: continue return False + # Lockfile → package manager, checked in priority order. _PY_LOCKFILES = (("uv.lock", "uv"), ("poetry.lock", "poetry"), ("Pipfile.lock", "pipenv")) _JS_LOCKFILES = ( @@ -153,21 +120,13 @@ _MAX_FACT_FILE_BYTES = 256 * 1024 _GIT_TIMEOUT = 2.5 -# Per-model edit-format steering. Matching the edit tool format to how a model -# was trained reduces mistakes and wasted reasoning (OpenAI/Codex handle -# patch-style diffs best; Anthropic models — and most open-weight coding -# models, whose RL scaffolds use str_replace-style editors — do best with -# string-replacement). Our `patch` tool exposes both: mode="patch" (V4A -# multi-file) and mode="replace" (find-and-swap). We nudge each family toward -# its native format. Unknown families get nothing (the brief's neutral wording -# stands). Substrings match the model id; aligned with TOOL_USE_ENFORCEMENT_MODELS. -# -# GPT/Codex get V4A for ALL edits, single-file included: in codex-rs, -# apply_patch (V4A — apply_patch.lark) is the ONLY file editor, no -# str_replace-style tool exists, and the shipped model prompts say to use -# apply_patch even "for single file edits" — so a replace-mode nudge would -# steer those models toward a format their first-party harness never taught -# them. +# Per-model edit-format steering: nudge each family toward the `patch` mode it +# was trained on (unknown families get nothing). GPT/Codex get V4A for ALL +# edits incl. single-file — codex-rs ships apply_patch as its ONLY editor and +# its prompts say to use it even for single files, so a replace-mode nudge +# would steer them toward a format their first-party harness never taught. +# Anthropic and most open-weight coding models were RL'd on str_replace-style +# editors. Substrings match the model id; aligned with TOOL_USE_ENFORCEMENT_MODELS. _EDIT_FORMAT_GUIDANCE: dict[str, tuple[tuple[str, ...], str]] = { "patch": ( ("gpt", "codex"), @@ -188,12 +147,7 @@ _EDIT_FORMAT_GUIDANCE: dict[str, tuple[tuple[str, ...], str]] = { def _model_family(model: Optional[str]) -> Optional[str]: - """Classify a model id into an edit-format family key, or ``None``. - - Used to steer the coding posture toward the edit tool format a model was - trained on. Family-agnostic by design: an unrecognised model gets ``None`` - and the operating brief's neutral edit wording applies. - """ + """Edit-format family key for a model id, or ``None`` (neutral wording applies).""" if not model: return None lowered = model.lower() @@ -206,14 +160,11 @@ def _model_family(model: Optional[str]) -> Optional[str]: def _edit_format_line(model: Optional[str]) -> str: """The edit-format guidance line for this model's family (``""`` if none).""" family = _model_family(model) - if family is None: - return "" - return _EDIT_FORMAT_GUIDANCE[family][1] + return "" if family is None else _EDIT_FORMAT_GUIDANCE[family][1] -# Operating brief for the coding posture. Tool names referenced here (read_file, -# search_files, patch, write_file, terminal, todo) are in the coding toolset and -# in _HERMES_CORE_TOOLS, so they're present on every surface this fires on. +# Operating brief for the coding posture. Tool names referenced here are in the +# coding toolset and in _HERMES_CORE_TOOLS, so they exist on every surface this fires on. CODING_AGENT_GUIDANCE = ( "You are a coding agent pairing with the user inside their codebase. " "Operate like a careful senior engineer.\n" @@ -264,6 +215,12 @@ CODING_AGENT_GUIDANCE = ( "answer, not a preamble." ) +_TODO_SENTENCE = ( + "- Track multi-step work with `todo_list`. Reference code as " + "`path:line` instead of pasting whole files." +) +_NO_TODO_SENTENCE = "- Reference code as `path:line` instead of pasting whole files." + # ── Context profiles (declarative posture definitions) ────────────────────── @@ -272,35 +229,22 @@ CODING_AGENT_GUIDANCE = ( class ContextProfile: """A named operating posture. Pure data — consumers read these fields. - ``toolset`` — collapse to this toolset (+ enabled MCP) when no explicit - selection is pinned; ``None`` keeps the platform default. - ``guidance`` — operating brief injected into the stable system prompt; - ``""`` injects nothing. - ``model_hint`` — routing preference key for smart model routing - (extension seam; not yet consumed by the router). - ``memory_policy``— memory namespace/weighting hint (extension seam). - ``compact_skill_categories`` — skill categories DEMOTED to names-only in - the system-prompt skill index under the opt-in ``focus`` - mode. Never hidden: every skill name stays visible - (so memory-anchored recall keeps working) — only the - descriptions are dropped to cut index noise. Deny-list - semantics so unknown/custom categories keep full - entries. + ``toolset``: collapse to this toolset (+ enabled MCP) under ``focus``; + ``None`` keeps the platform default. ``guidance``: operating brief for the + stable system prompt. ``model_hint``: routing preference (extension seam). + ``compact_skill_categories``: categories DEMOTED to names-only in the skill + index under ``focus`` — deny-list, never hidden, so recall keeps working. """ name: str toolset: Optional[str] = None guidance: str = "" model_hint: Optional[str] = None - memory_policy: str = "default" compact_skill_categories: tuple[str, ...] = () -# Skill categories that are clearly not part of a coding workflow. Demoted to -# names-only in the prompt's skill index under the opt-in ``focus`` mode only -# (deny-list — anything not listed here, incl. custom user categories, keeps -# full entries). Coding-adjacent categories (devops, github, mcp, -# data-science, diagramming, research, security, …) are intentionally absent. +# Clearly non-coding skill categories (deny-list: custom categories keep full +# entries). Coding-adjacent ones (devops, github, mcp, research, …) are absent. _NON_CODING_SKILL_CATEGORIES = ( "apple", "communication", "cooking", "creative", "email", "finance", "gaming", "gifs", "health", "media", "music", "note-taking", @@ -315,7 +259,6 @@ CODING_PROFILE = ContextProfile( toolset=CODING_TOOLSET, guidance=CODING_AGENT_GUIDANCE, model_hint="coding", - memory_policy="project", compact_skill_categories=_NON_CODING_SKILL_CATEGORIES, ) @@ -332,45 +275,38 @@ def get_profile(name: str) -> ContextProfile: # ── Helpers ───────────────────────────────────────────────────────────────── +_MODE_ALIASES = { + **dict.fromkeys(("focus", "strict", "lean"), "focus"), + **dict.fromkeys(("on", "true", "yes", "1", "always"), "on"), + **dict.fromkeys(("off", "false", "no", "0", "never"), "off"), +} -def _coding_mode(config: Optional[dict[str, Any]]) -> str: - """Return the normalized ``agent.coding_context`` mode (auto/focus/on/off).""" + +def _agent_config_value(config: Optional[dict[str, Any]], key: str, default: Any, *, readonly: bool) -> Any: + """``config["agent"][key]``, loading config when none was passed.""" if config is None: try: - from hermes_cli.config import load_config_readonly + from hermes_cli.config import load_config, load_config_readonly - config = load_config_readonly() + config = load_config_readonly() if readonly else load_config() except Exception: config = {} - raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto") - mode = str(raw).strip().lower() - if mode in {"focus", "strict", "lean"}: - return "focus" - if mode in {"on", "true", "yes", "1", "always"}: - return "on" - if mode in {"off", "false", "no", "0", "never"}: - return "off" - return "auto" + return ((config or {}).get("agent", {}) or {}).get(key, default) + + +def _coding_mode(config: Optional[dict[str, Any]]) -> str: + """Normalized ``agent.coding_context`` mode (auto/focus/on/off).""" + raw = _agent_config_value(config, "coding_context", "auto", readonly=True) + return _MODE_ALIASES.get(str(raw).strip().lower(), "auto") def _coding_instructions(config: Optional[dict[str, Any]]) -> str: - """Standing operator instructions for the coding posture (config). + """Standing operator instructions (``agent.coding_instructions``: str or list). - ``agent.coding_instructions`` — a string or list of strings appended to the - coding brief as an extra stable system block, so a user can pin project-wide - coding-workflow rules (e.g. "for UI work don't run tsc/lint until I approve; - clean the diff before committing") without editing the shipped brief. - Cache-safe: resolved once per session into the stable system-prompt tier, - like the rest of the posture. + Appended to the brief as an extra stable block so a user can pin + project-wide workflow rules without editing the shipped brief. """ - if config is None: - try: - from hermes_cli.config import load_config - - config = load_config() - except Exception: - config = {} - raw = ((config or {}).get("agent", {}) or {}).get("coding_instructions", "") + raw = _agent_config_value(config, "coding_instructions", "", readonly=False) if isinstance(raw, (list, tuple)): return "\n".join(str(item).strip() for item in raw if str(item).strip()) return str(raw or "").strip() @@ -403,19 +339,14 @@ def _home() -> Optional[Path]: def _marker_root(cwd: Path) -> Optional[Path]: - """Nearest ancestor that looks like a project root, or ``None``. + """Nearest ancestor (≤6 levels) that looks like a project root, or ``None``. - Walks up at most a few levels so a manifest in the workspace root counts - even when the user is in a subdirectory. ``$HOME`` itself is skipped — a - Makefile or AGENTS.md sitting in the home directory is global user config, - not a project-root signal. + ``$HOME`` and the shared temp root are skipped: a Makefile/AGENTS.md in the + home dir is global user config, and a stray manifest in /tmp must not flip + every session whose cwd lives under it into the coding posture. """ current = cwd.resolve() home = _home() - # Shared world-writable temp roots are never project roots: a stray - # manifest in /tmp (left by any process) must not flip every session - # whose cwd lives under the temp dir into the coding posture. Same - # reasoning as the $HOME skip below. try: temp_root = Path(tempfile.gettempdir()).resolve() except Exception: @@ -435,17 +366,10 @@ def _detect_profile_name(mode: str, platform: str, cwd_str: str) -> str: """Resolve which profile applies. ``auto``/``focus``: coding when the surface is interactive AND the cwd is a - code workspace (a git repo or a recognised project root). ``on``: always - coding. ``off``: always general. - - A git repo rooted at ``$HOME`` (the dotfiles pattern) is NOT a workspace - signal — without the guard, every session anywhere under a dotfiles-managed - home directory would silently flip to the coding posture. - - Detection is intentionally not memoized: it's a handful of ``stat`` calls, - and callers resolve the mode once per session anyway. Caching here would - risk a stale posture if a long-lived process (gateway/TUI) serves sessions - from different working directories. + code workspace (project root, or a git repo that actually holds code). + ``on``: always coding. ``off``: always general. A git repo rooted at + ``$HOME`` (dotfiles) is NOT a workspace signal. Deliberately not memoized: + a long-lived gateway/TUI process serves sessions from different cwds. """ if mode == "off": return GENERAL_PROFILE.name @@ -454,16 +378,10 @@ def _detect_profile_name(mode: str, platform: str, cwd_str: str) -> str: if platform and platform.strip().lower() not in INTERACTIVE_CODING_PLATFORMS: return GENERAL_PROFILE.name cwd = Path(cwd_str) - # A recognized project root (manifest / AGENTS.md / .cursorrules) is a code - # workspace on its own — cheap stat checks, no scan. if _marker_root(cwd) is not None: return CODING_PROFILE.name git_root = _git_root(cwd) - if git_root is not None and git_root == _home(): - git_root = None # dotfiles repo at $HOME — not a code workspace - # A bare git repo only counts when it actually holds code, so `git init` on a - # notes/writing/research folder stays in the general posture. - if git_root is not None and _has_code_files(git_root): + if git_root is not None and git_root != _home() and _has_code_files(git_root): return CODING_PROFILE.name return GENERAL_PROFILE.name @@ -475,23 +393,18 @@ def _detect_profile_name(mode: str, platform: str, cwd_str: str) -> str: class RuntimeMode: """The resolved operating posture for a session. Immutable by construction. - Built once via :func:`resolve_runtime_mode` and consumed by every domain - that cares about the coding/general distinction. Never mutate or re-resolve - mid-session — that would break the prompt cache. + Built once via :func:`resolve_runtime_mode`; never re-resolved mid-session + (that would break the prompt cache). """ profile: ContextProfile surface: str cwd: Path - # The normalized ``agent.coding_context`` mode this posture was resolved - # under (auto/focus/on/off). Toolset collapse is gated on ``focus``. + # Normalized ``agent.coding_context`` mode; toolset collapse is gated on ``focus``. config_mode: str = "auto" - # The model id this session runs (e.g. "anthropic/claude-opus-4.8"). Used - # only to steer edit-format guidance toward the model's family — see - # ``_edit_format_line``. Fixed for the session, so cache-safe. + # Model id, used only to steer edit-format guidance (fixed per session). model: Optional[str] = None - # Standing operator instructions (``agent.coding_instructions``), appended - # as an extra stable system block. Empty unless the user configures it. + # ``agent.coding_instructions``, appended as an extra stable block. instructions: str = "" @property @@ -505,94 +418,58 @@ class RuntimeMode: def toolset_selection(self, config: Optional[dict[str, Any]] = None) -> Optional[list[str]]: """Toolset list for this posture, or ``None`` to keep the platform default. - Non-``None`` only under the opt-in ``focus`` mode. The default posture - is prompt-only: most strippable toolsets are off-by-default anyway, and - a user who explicitly enabled one (image-gen for frontend/game assets, - messaging for build notifications, …) keeps it while coding. - - Callers apply this only when the user hasn't pinned an explicit - selection (``--toolsets``, ``HERMES_TUI_TOOLSETS``, …); they never - override a pin. Returns the profile's toolset plus enabled MCP servers. + Non-``None`` only under ``focus``. Callers apply it only when the user + hasn't pinned an explicit selection (``--toolsets``, ``HERMES_TUI_TOOLSETS``). """ - if self.config_mode != "focus": - return None - if self.profile.toolset is None: + if self.config_mode != "focus" or self.profile.toolset is None: return None return [self.profile.toolset, *_enabled_mcp_servers(config)] def system_prompt_parts( self, valid_tool_names=None ) -> tuple[list[str], list[str], list[str]]: - """Return prefix, workspace, and trailing posture blocks separately. + """Return (prefix, workspace, trailing) posture blocks. - The operating brief carries a model-family edit-format nudge appended - to it (one cached string, not a separate block) so the model is steered - toward the `patch` mode it handles best — see ``_edit_format_line``. - - ``valid_tool_names`` (when provided) tailors the brief to the session's - toolset: the ``todo`` tracking sentence is dropped when the todo tool - isn't loaded (e.g. Blank Slate), so the brief never references a tool - the model can't call. The toolset is fixed at session construction, - so the rendered brief is deterministic per session — cache-safe. - - The three lists preserve the historical flat prompt order: the brief, - the live workspace snapshot, then configured operator instructions. - Prompt assembly can therefore put a cache boundary before the snapshot - without changing the persisted system-prompt bytes. + The brief carries the model-family edit-format nudge appended to it + (one cached string). ``valid_tool_names`` drops the ``todo_list`` + sentence when that tool isn't loaded (e.g. Blank Slate). The three + lists preserve the historical flat order — brief, workspace snapshot, + operator instructions — so prompt assembly can put a cache boundary + before the snapshot without changing the persisted bytes. """ if not self.is_coding: return [], [], [] prefix: list[str] = [] - workspace_parts: list[str] = [] - trailing: list[str] = [] if self.profile.guidance: brief = self.profile.guidance if valid_tool_names is not None and "todo_list" not in valid_tool_names: - brief = brief.replace( - "- Track multi-step work with `todo_list`. Reference code as " - "`path:line` instead of pasting whole files.", - "- Reference code as `path:line` instead of pasting " - "whole files.", - ) + brief = brief.replace(_TODO_SENTENCE, _NO_TODO_SENTENCE) edit_line = _edit_format_line(self.model) if edit_line: brief = f"{brief}\n{edit_line}" prefix.append(brief) workspace = build_coding_workspace_block(self.cwd) - if workspace: - workspace_parts.append(workspace) - # Operator instructions ride their own block so the brief (block 0) stays - # byte-stable and cache-keyed independently of user config. - if self.instructions: - trailing.append(f"Operator instructions (from config):\n{self.instructions}") + workspace_parts = [workspace] if workspace else [] + # Operator instructions ride their own block so the brief stays + # byte-stable independently of user config. + trailing = ( + [f"Operator instructions (from config):\n{self.instructions}"] + if self.instructions else [] + ) return prefix, workspace_parts, trailing def system_blocks(self) -> list[str]: - """Return posture blocks in their historical display order. - - ``system_prompt_parts`` is the cache-aware API. This compatibility - helper retains the public flat list for callers outside prompt assembly. - """ + """Posture blocks as one flat list in historical order (compat helper).""" prefix, workspace, trailing = self.system_prompt_parts() return [*prefix, *workspace, *trailing] def compact_skill_categories(self) -> frozenset[str]: - """Skill categories to demote to names-only in the prompt's skill index. + """Skill categories to demote to names-only in the skill index. - Gated on the opt-in ``focus`` mode, like the toolset collapse: the - default posture leaves the skill index untouched. Users who didn't ask - for a lean prompt keep full entries for every category — index changes - under ``auto`` proved too surprising in practice, even names-only ones - (a demoted description is information the model no longer weighs when - deciding what to load). - - Demoted — never hidden — even under ``focus``. An earlier revision - fully pruned these categories from the index, which caused silent - capability loss in a real workflow: agent-created skills are the - model's accumulated project memory (server-ops runbooks, learned - pitfalls, …), and models do not reliably reach for ``skills_list`` to - rediscover what the index stopped showing them. Names-only keeps every - skill loadable on recall while still cutting the description noise. + Gated on ``focus`` like the toolset collapse — index changes under + ``auto`` proved too surprising. Demoted, never hidden: fully pruning + them caused silent capability loss (agent-created skills are the + model's project memory and models don't reliably re-run ``skills_list``). """ if not self.is_coding or self.config_mode != "focus": return frozenset() @@ -606,14 +483,10 @@ def resolve_runtime_mode( config: Optional[dict[str, Any]] = None, model: Optional[str] = None, ) -> RuntimeMode: - """Resolve the operating posture once. Cheap — a handful of ``stat`` calls. + """Resolve the operating posture once (a handful of ``stat`` calls). - This is the single entry point every domain should call. The returned - object is immutable and safe to cache for the session. Detection itself is - intentionally *not* memoized (see ``_detect_profile_name``) so a long-lived - process can't pin a stale posture; callers resolve once per session and - hold the result. ``model`` is recorded only to steer edit-format guidance; - it never affects detection. + The single entry point every domain should call; the result is immutable + and safe to hold for the session. ``model`` only steers edit-format guidance. """ resolved_cwd = _resolve_cwd(cwd) mode = _coding_mode(config) @@ -649,32 +522,12 @@ def coding_selection( cwd: Optional[str | Path] = None, config: Optional[dict[str, Any]] = None, ) -> Optional[list[str]]: - """Toolset selection for the coding posture. - - ``None`` unless the user opted into ``focus`` mode AND the posture is - active — the default coding posture never overrides configured toolsets. - """ + """Toolset selection for the coding posture (``None`` unless ``focus`` and active).""" return resolve_runtime_mode( platform=platform, cwd=cwd, config=config ).toolset_selection(config) -def coding_system_blocks( - *, - platform: Optional[str] = None, - cwd: Optional[str | Path] = None, - config: Optional[dict[str, Any]] = None, - model: Optional[str] = None, -) -> list[str]: - """Stable system-prompt blocks for the current posture (empty when general). - - ``model`` steers the brief's edit-format nudge toward the model's family. - """ - return resolve_runtime_mode( - platform=platform, cwd=cwd, config=config, model=model - ).system_blocks() - - def coding_system_prompt_parts( *, platform: Optional[str] = None, @@ -695,25 +548,14 @@ def coding_compact_skill_categories( cwd: Optional[str | Path] = None, config: Optional[dict[str, Any]] = None, ) -> frozenset[str]: - """Skill categories the active posture demotes to names-only in the index. - - Empty outside the coding posture and outside the opt-in ``focus`` mode — - the default posture never touches the skill index. Under ``focus``, - demoted — never hidden: every skill name stays in the index and remains - loadable via ``skill_view`` / ``skills_list``; only descriptions are - dropped. - """ + """Skill categories the active posture demotes to names-only (empty outside ``focus``).""" return resolve_runtime_mode( platform=platform, cwd=cwd, config=config ).compact_skill_categories() def _enabled_mcp_servers(config: Optional[dict[str, Any]]) -> list[str]: - """Names of MCP servers the user has enabled — kept in the coding posture. - - MCP servers (figma, browser, tophat, …) are explicitly configured and part - of the coding workflow, not noise to strip. - """ + """Names of MCP servers the user has enabled — kept in the coding posture.""" try: from hermes_cli.config import read_raw_config from hermes_cli.tools_config import _parse_enabled_flag @@ -735,10 +577,9 @@ def _enabled_mcp_servers(config: Optional[dict[str, Any]]) -> list[str]: def _git(cwd: Path, *args: str) -> str: """``git -C `` → stripped stdout, or ``""`` on any failure. - Uses the shared :func:`bounded_git_probe` so the post-kill cleanup is bounded - on Windows — a plain ``subprocess.run(timeout=...)`` here deadlocked the agent - turn inside ``build_coding_workspace_block`` when a killed git left a suspended - descendant holding the pipe handles (issue #66037). + :func:`bounded_git_probe` bounds the post-kill cleanup on Windows — a plain + ``subprocess.run(timeout=...)`` deadlocked when a killed git left a + suspended descendant holding the pipe handles. """ return bounded_git_probe(["git", "-C", str(cwd), *args], timeout=_GIT_TIMEOUT) @@ -780,12 +621,7 @@ def _read_small(path: Path) -> str: @dataclass(frozen=True) class ProjectFacts: - """Structured project facts — the model's verify loop, detected once. - - The same data that feeds the workspace snapshot, exposed structurally so - non-prompt consumers (e.g. the desktop verify UI) read it instead of - re-detecting and drifting from the prompt. - """ + """Structured project facts — exposed so non-prompt consumers (desktop verify UI) don't re-detect.""" manifests: list[str] package_managers: list[str] @@ -796,9 +632,8 @@ class ProjectFacts: def detect_project_facts(root: Path) -> ProjectFacts: """Detect manifests, package manager(s), verify commands, and context files. - Cheap: stat calls plus reads of a couple of small files. The single source - of truth for both the prompt snapshot (:func:`_project_facts`) and the - gateway's ``project.facts`` — so the UI never re-sniffs verify commands. + Single source of truth for the prompt snapshot and the gateway's + ``project.facts``. Cheap: stat calls plus a couple of small file reads. """ manifests = [m for m in _PROJECT_MARKERS if m not in _CONTEXT_FILES and (root / m).is_file()] package_managers = list( @@ -833,16 +668,9 @@ def detect_project_facts(root: Path) -> ProjectFacts: def _project_facts(root: Path) -> list[str]: - """Render :func:`detect_project_facts` as workspace-snapshot lines. - - Hands the model its *verify loop* up front — which manifest, which package - manager, and the exact test/lint/build commands — instead of making it - rediscover them every session. Built once at prompt-build time; the string - output must stay byte-stable to preserve the prompt cache. - """ + """Render :func:`detect_project_facts` as workspace-snapshot lines (byte-stable).""" f = detect_project_facts(root) facts: list[str] = [] - if f.manifests: line = f"- Project: {', '.join(f.manifests[:6])}" if f.package_managers: @@ -852,22 +680,25 @@ def _project_facts(root: Path) -> list[str]: facts.append(f"- Verify: {'; '.join(f.verify_commands)}") if f.context_files: facts.append(f"- Context files: {', '.join(f.context_files)}") - return facts +def _workspace_roots(cwd: Optional[str | Path]) -> tuple[Optional[Path], Optional[Path]]: + """(git_root, workspace_root) for *cwd*; workspace root is git root else marker root.""" + resolved = _resolve_cwd(cwd) + git_root = _git_root(resolved) + return git_root, git_root or _marker_root(resolved) + + def project_facts_for(cwd: Optional[str | Path] = None) -> Optional[dict[str, Any]]: """Structured project facts for ``cwd`` — ``None`` outside a workspace. - Same detection the system-prompt snapshot uses (git root, else marker root), - exposed for non-prompt consumers (the desktop verify UI) so they never - re-derive "are we coding?" or duplicate the verify-command sniffing. + Same detection the system-prompt snapshot uses, exposed for non-prompt + consumers (the desktop verify UI). """ - resolved = _resolve_cwd(cwd) - root = _git_root(resolved) or _marker_root(resolved) + _, root = _workspace_roots(cwd) if root is None: return None - f = detect_project_facts(root) return { "root": str(root), @@ -881,13 +712,10 @@ def project_facts_for(cwd: Optional[str | Path] = None) -> Optional[dict[str, An def build_coding_workspace_block(cwd: Optional[str | Path] = None) -> str: """Workspace snapshot for the system prompt (empty outside a workspace). - Git state (branch/status/commits) when the cwd is in a repo, plus detected - project facts (manifest, package manager, verify commands, context files) - — so marker-only (non-git) projects still get a snapshot. + Git state when the cwd is in a repo, plus detected project facts — so + marker-only (non-git) projects still get a snapshot. """ - resolved = _resolve_cwd(cwd) - git_root = _git_root(resolved) - root = git_root or _marker_root(resolved) + git_root, root = _workspace_roots(cwd) if root is None: return "" @@ -908,11 +736,9 @@ def build_coding_workspace_block(cwd: Optional[str | Path] = None) -> str: elif head == "(detached)": lines.append("- Branch: (detached HEAD)") - # Linked worktree: the per-worktree git dir differs from the shared common dir. - # We surface the fact that it's a worktree (so the model knows branches/stashes - # are shared state) but deliberately do NOT expose the primary tree path — - # giving the model a second absolute path causes it to sometimes run commands - # in the wrong directory. + # Linked worktree: say so (branches/stashes are shared state) but do + # NOT expose the primary tree path — a second absolute path makes the + # model run commands in the wrong directory. git_dir, common_dir = _git(root, "rev-parse", "--git-dir"), _git(root, "rev-parse", "--git-common-dir") if git_dir and common_dir and Path(git_dir).resolve() != Path(common_dir).resolve(): lines.append("- Worktree: linked (git state shared with primary tree)") diff --git a/agent/context_references.py b/agent/context_references.py index cadb044163..90552285ab 100644 --- a/agent/context_references.py +++ b/agent/context_references.py @@ -1,3 +1,5 @@ +"""@-reference expansion (``@file:``, ``@folder:``, ``@diff``, ``@git:``, ``@url:`` + plugin prefixes).""" + from __future__ import annotations import asyncio @@ -7,6 +9,7 @@ import mimetypes import os import re import subprocess +from abc import ABC, abstractmethod from dataclasses import dataclass, field from pathlib import Path from typing import Awaitable, Callable @@ -20,10 +23,8 @@ from hermes_cli._subprocess_compat import ( ) from hermes_cli.sizefmt import format_bytes -from abc import ABC, abstractmethod - # --------------------------------------------------------------------------- -# Plugin context-reference provider API (Issue #26193) +# Plugin context-reference provider API # --------------------------------------------------------------------------- BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"}) @@ -43,11 +44,7 @@ class ContextCompletionItem: class ContextReferenceProvider(ABC): - """Base class for plugin-registered @-prefix context reference providers. - - Plugins subclass this and register via - ``PluginContext.register_context_reference()``. - """ + """Base class for plugin @-prefix providers, registered via ``PluginContext.register_context_reference()``.""" prefix: str = "" # e.g. "issue", "channel", "doc" description: str = "" # shown in autocomplete meta column @@ -86,8 +83,7 @@ _QUOTED_REFERENCE_VALUE = r'(?:`[^`\n]+`|"[^"\n]+"|\'[^\'\n]+\')' REFERENCE_PATTERN = re.compile( rf"(?diff|staged)\b|(?Pfile|folder|git|url):(?P{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+))" ) -# Plugin fallback pattern – catches any @: not handled by the -# built-in regex so that plugin-registered prefixes can be resolved. +# Plugin fallback: any @: the built-in regex did not claim. _PLUGIN_REFERENCE_PATTERN = re.compile( rf"(?[a-zA-Z][a-zA-Z0-9_-]*):(?P{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+)" ) @@ -138,9 +134,8 @@ class ContextReferenceResult: def format_reference_value(value: str) -> str: """Quote a reference value so ``REFERENCE_PATTERN`` reads it back whole. - The unquoted alternative in the pattern is ``\\S+``, so a path containing a - space parses as a truncated ref with the tail left behind as loose text. - Mirrors ``formatRefValue`` in the desktop's directive-text.tsx. + The unquoted alternative is ``\\S+``, so a path with a space would parse as a + truncated ref. Mirrors ``formatRefValue`` in the desktop's directive-text.tsx. """ if not _NEEDS_QUOTING.search(value): return value @@ -158,26 +153,14 @@ def parse_context_references(message: str) -> list[ContextReference]: for match in REFERENCE_PATTERN.finditer(message): simple = match.group("simple") if simple: - refs.append( - ContextReference( - raw=match.group(0), - kind=simple, - target="", - start=match.start(), - end=match.end(), - ) - ) + refs.append(ContextReference(raw=match.group(0), kind=simple, target="", start=match.start(), end=match.end())) continue - kind = match.group("kind") value = _strip_trailing_punctuation(match.group("value") or "") - line_start = None - line_end = None - target = _strip_reference_wrappers(value) - if kind == "file": target, line_start, line_end = _parse_file_reference_value(value) - + else: + target, line_start, line_end = _strip_reference_wrappers(value), None, None refs.append( ContextReference( raw=match.group(0), @@ -190,26 +173,24 @@ def parse_context_references(message: str) -> list[ContextReference]: ) ) - # Second pass: resolve plugin-registered prefixes the built-in pattern missed + # Second pass: plugin-registered prefixes the built-in pattern missed. if _context_reference_providers: for match in _PLUGIN_REFERENCE_PATTERN.finditer(message): kind = match.group("kind") - if kind in BUILTIN_PREFIXES: + if kind in BUILTIN_PREFIXES or kind not in _context_reference_providers: continue - # Skip if already captured by the built-in pattern if any(r.kind == kind and r.start == match.start() for r in refs): continue - if kind in _context_reference_providers: - value = _strip_trailing_punctuation(match.group("value") or "") - refs.append( - ContextReference( - raw=match.group(0), - kind=kind, - target=_strip_reference_wrappers(value), - start=match.start(), - end=match.end(), - ) + value = _strip_trailing_punctuation(match.group("value") or "") + refs.append( + ContextReference( + raw=match.group(0), + kind=kind, + target=_strip_reference_wrappers(value), + start=match.start(), + end=match.end(), ) + ) return refs @@ -222,6 +203,7 @@ def preprocess_context_references( url_fetcher: Callable[[str], str | Awaitable[str]] | None = None, allowed_root: str | Path | None = None, ) -> ContextReferenceResult: + """Sync wrapper; safe both without a loop (CLI) and inside a running loop (gateway).""" coro = preprocess_context_references_async( message, cwd=cwd, @@ -229,7 +211,6 @@ def preprocess_context_references( url_fetcher=url_fetcher, allowed_root=allowed_root, ) - # Safe for both CLI (no loop) and gateway (loop already running). try: loop = asyncio.get_running_loop() except RuntimeError: @@ -254,31 +235,17 @@ async def preprocess_context_references_async( return ContextReferenceResult(message=message, original_message=message) cwd_path = Path(cwd).expanduser().resolve() - # Default to the current working directory so @ references cannot escape - # the active workspace unless a caller explicitly widens the root. - allowed_root_path = ( - Path(allowed_root).expanduser().resolve() if allowed_root is not None else cwd_path - ) + # Default root = cwd so @ references cannot escape the workspace unless a caller widens it. + allowed_root_path = Path(allowed_root).expanduser().resolve() if allowed_root is not None else cwd_path warnings: list[str] = [] blocks: list[str] = [] injected_tokens = 0 - # Expand all references concurrently. Each _expand_reference is independent - # (no shared state during expansion) — a message with several @url: refs - # would otherwise pay one full web_extract round-trip per ref in series. - # gather preserves positional order, so we reassemble warnings/blocks in the - # original ref order exactly as the prior serial loop did; the token-budget - # check below is unchanged (it runs once, after all refs are expanded). + # Expand concurrently (each ref is independent; several @url: refs would otherwise + # serialize web_extract round-trips). gather preserves order, so warnings/blocks + # are assembled in ref order; the token-budget check runs once afterwards. expanded = await asyncio.gather( - *( - _expand_reference( - ref, - cwd_path, - url_fetcher=url_fetcher, - allowed_root=allowed_root_path, - ) - for ref in refs - ) + *(_expand_reference(ref, cwd_path, url_fetcher=url_fetcher, allowed_root=allowed_root_path) for ref in refs) ) for warning, block in expanded: if warning: @@ -302,17 +269,14 @@ async def preprocess_context_references_async( expanded=False, blocked=True, ) - if injected_tokens > soft_limit: warnings.append( f"@ context injection warning: {injected_tokens} tokens exceeds the 25% soft limit ({soft_limit})." ) - # Leave the `@file:`/`@folder:` tokens where the user typed them. The token - # IS the reference, not scaffolding around it: clients render each one as an - # inline chip, so stripping them left a sentence with a hole in it ("review - # and ship") and made the desktop re-derive the refs from the attached block - # to show them as a detached list above the prose. + # The `@file:`/`@folder:` tokens stay where the user typed them: the token IS the + # reference (clients render it as an inline chip); stripping it left a hole in the + # sentence and forced the desktop to re-derive refs from the attached block. final = message if warnings: final = f"{final}\n\n--- Context Warnings ---\n" + "\n".join(f"- {warning}" for warning in warnings) @@ -337,6 +301,7 @@ async def _expand_reference( url_fetcher: Callable[[str], str | Awaitable[str]] | None = None, allowed_root: Path | None = None, ) -> tuple[str | None, str | None]: + """Return ``(warning, block)`` for one reference; exactly one side is set.""" try: if ref.kind == "file": return _expand_file_reference(ref, cwd, allowed_root=allowed_root) @@ -357,7 +322,6 @@ async def _expand_reference( except Exception as exc: return f"{ref.raw}: {exc}", None - # Plugin-provided context references provider = _context_reference_providers.get(ref.kind) if provider is not None: try: @@ -383,13 +347,8 @@ def _expand_file_reference( if not path.is_file(): return f"{ref.raw}: path is not a file", None if _is_binary_file(path): - # A binary file can't be inlined as text, but it IS on disk (the agent's - # tools run where this resolves — the local cwd, or the staged copy in a - # remote session workspace). Returning a bare "not supported" warning - # with no content was a dead end: the model saw a failure and gave up - # (told the user the file type wasn't supported). Instead, hand it an - # actionable block — the path, type, size, and a nudge to use its tools — - # so it can read/convert/view the file itself. + # A bare "not supported" warning was a dead end (the model gave up); the file IS + # on disk where the agent's tools run, so hand it an actionable block instead. return None, _binary_reference_block(ref, path) text = path.read_text(encoding="utf-8") @@ -400,8 +359,7 @@ def _expand_file_reference( text = "\n".join(lines[start_idx:end_idx]) lang = _code_fence_language(path) - label = ref.raw - return None, f"📄 {label} ({estimate_tokens_rough(text)} tokens)\n```{lang}\n{text}\n```" + return None, f"📄 {ref.raw} ({estimate_tokens_rough(text)} tokens)\n```{lang}\n{text}\n```" def _expand_folder_reference( @@ -416,37 +374,45 @@ def _expand_folder_reference( return f"{ref.raw}: folder not found", None if not path.is_dir(): return f"{ref.raw}: path is not a folder", None - listing = _build_folder_listing(path, cwd) return None, f"📁 {ref.raw} ({estimate_tokens_rough(listing)} tokens)\n{listing}" +def _run_quiet( + cmd: list[str], cwd: Path, timeout: int, env: dict | None = None +) -> subprocess.CompletedProcess: + """subprocess.run with captured text output, no stdin, and no console flash on Windows.""" + popen_kwargs: dict = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {} + if env is not None: + popen_kwargs["env"] = env + return subprocess.run( + cmd, + cwd=cwd, + capture_output=True, + text=True, encoding='utf-8', errors='replace', + timeout=timeout, + stdin=subprocess.DEVNULL, + **popen_kwargs, + ) + + def _expand_git_reference( ref: ContextReference, cwd: Path, args: list[str], label: str, ) -> tuple[str | None, str | None]: - _popen_kwargs = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {} try: - result = subprocess.run( - ["git", *harden_git_argv(args)], - cwd=cwd, - capture_output=True, - text=True, encoding='utf-8', errors='replace', - timeout=30, - stdin=subprocess.DEVNULL, - env=noninteractive_git_env(), - **_popen_kwargs, + # Repo-supplied config/attributes must never execute code (GHSA-7x36-8jrh-v4pw). + result = _run_quiet( + ["git", *harden_git_argv(args)], cwd, 30, env=noninteractive_git_env() ) except subprocess.TimeoutExpired: return f"{ref.raw}: git command timed out (30s)", None if result.returncode != 0: stderr = (result.stderr or "").strip() or "git command failed" return f"{ref.raw}: {stderr}", None - content = result.stdout.strip() - if not content: - content = "(no output)" + content = result.stdout.strip() or "(no output)" return None, f"🧾 {label} ({estimate_tokens_rough(content)} tokens)\n```diff\n{content}\n```" @@ -466,8 +432,7 @@ async def _default_url_fetcher(url: str) -> str: from tools.web_tools import web_extract_tool raw = await web_extract_tool([url], format="markdown") - payload = json.loads(raw) - docs = payload.get("results", []) + docs = json.loads(raw).get("results", []) if not docs: return "" doc = docs[0] @@ -488,6 +453,7 @@ def _resolve_path(cwd: Path, target: str, *, allowed_root: Path | None = None) - def _ensure_reference_path_allowed(path: Path) -> None: + """Refuse credential/internal paths. Fails CLOSED: the gateway feeds untrusted remote text here.""" from hermes_constants import get_hermes_home home = Path(os.path.expanduser("~")).resolve() hermes_home = get_hermes_home().resolve() @@ -499,7 +465,6 @@ def _ensure_reference_path_allowed(path: Path) -> None: if path in blocked_exact: raise ValueError("path is a sensitive credential file and cannot be attached") - for blocked_dir in blocked_dirs: try: path.relative_to(blocked_dir) @@ -507,16 +472,9 @@ def _ensure_reference_path_allowed(path: Path) -> None: continue raise ValueError("path is a sensitive credential or internal Hermes path and cannot be attached") - # Anchor to the canonical read deny-list (agent/file_safety.get_read_block_error), - # the single source of truth used by the file/terminal read path. The narrow - # list above predates that guard and never caught the real credential stores: - # provider keys (auth.json), Anthropic OAuth tokens (.anthropic_oauth.json), - # MCP OAuth material (mcp-tokens/), webhook HMAC secrets, and project-local - # .env files. That gap matters because the gateway feeds UNTRUSTED remote - # message text into reference expansion, so `@file:~/.hermes/auth.json` from a - # chat peer would otherwise read the operator's keys straight into context. - # Routing through the canonical guard closes the gap today and keeps this path - # protected automatically whenever that deny-list grows. + # Anchor to the canonical read deny-list (agent/file_safety.get_read_block_error): the + # narrow list above never caught auth.json, .anthropic_oauth.json, mcp-tokens/, webhook + # secrets or project .env files, and it grows automatically with that deny-list. try: from agent.file_safety import get_read_block_error @@ -527,13 +485,8 @@ def _ensure_reference_path_allowed(path: Path) -> None: except ValueError: raise except Exception: - # Fail CLOSED on the security path. This guard exists specifically to - # cover credential stores the narrow list above misses (auth.json, - # .anthropic_oauth.json, mcp-tokens/, ...). If the canonical lookup - # ever fails, silently falling through would re-open that exact hole — - # the gateway feeds untrusted remote text here, so a probe could then - # attach the operator's keys. Refuse instead: a spurious block on a - # legitimate file is a recoverable annoyance; a leaked credential is not. + # If the canonical lookup fails, falling through would re-open the exact hole this + # guard closes; a spurious block is recoverable, a leaked credential is not. raise ValueError( "path could not be verified against the credential deny-list and cannot be attached" ) @@ -583,27 +536,26 @@ def _parse_file_reference_value(value: str) -> tuple[str, int | None, int | None return _strip_reference_wrappers(value), None, None +_TEXT_EXTENSIONS = (".py", ".md", ".txt", ".json", ".yaml", ".yml", ".toml", ".js", ".ts") + + def _is_binary_file(path: Path) -> bool: mime, _ = mimetypes.guess_type(path.name) - if mime and not mime.startswith("text/") and not any( - path.name.endswith(ext) for ext in (".py", ".md", ".txt", ".json", ".yaml", ".yml", ".toml", ".js", ".ts") - ): + if mime and not mime.startswith("text/") and not path.name.endswith(_TEXT_EXTENSIONS): return True - chunk = path.read_bytes()[:4096] - return b"\x00" in chunk + return b"\x00" in path.read_bytes()[:4096] def _build_folder_listing(path: Path, cwd: Path, limit: int = 200) -> str: lines = [f"{path.relative_to(cwd)}/"] entries = _iter_visible_entries(path, cwd, limit=limit) + base_depth = len(path.relative_to(cwd).parts) for entry in entries: - rel = entry.relative_to(cwd) - indent = " " * max(len(rel.parts) - len(path.relative_to(cwd).parts) - 1, 0) + indent = " " * max(len(entry.relative_to(cwd).parts) - base_depth - 1, 0) if entry.is_dir(): lines.append(f"{indent}- {entry.name}/") else: - meta = _file_metadata(entry) - lines.append(f"{indent}- {entry.name} ({meta})") + lines.append(f"{indent}- {entry.name} ({_file_metadata(entry)})") if len(entries) >= limit: lines.append("- ...") return "\n".join(lines) @@ -629,29 +581,16 @@ def _iter_visible_entries(path: Path, cwd: Path, limit: int) -> list[Path]: dirs[:] = sorted(d for d in dirs if not d.startswith(".") and d != "__pycache__") files = sorted(f for f in files if not f.startswith(".")) root_path = Path(root) - for d in dirs: - output.append(root_path / d) - if len(output) >= limit: - return output - for f in files: - output.append(root_path / f) + for name in dirs + files: + output.append(root_path / name) if len(output) >= limit: return output return output def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None: - _popen_kwargs = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {} try: - result = subprocess.run( - ["rg", "--files", str(path.relative_to(cwd))], - cwd=cwd, - capture_output=True, - text=True, encoding='utf-8', errors='replace', - timeout=10, - stdin=subprocess.DEVNULL, - **_popen_kwargs, - ) + result = _run_quiet(["rg", "--files", str(path.relative_to(cwd))], cwd, 10) except (FileNotFoundError, OSError, subprocess.TimeoutExpired): return None if result.returncode != 0: @@ -661,19 +600,15 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None: def _agent_visible_path(path: Path) -> str: - """Map a host path to the path the agent's tools can read in the active backend. + """Map a host path to what the agent's tools can read in the active backend. - Under a container backend (docker) the gateway host path dangles inside the - sandbox — the container has its own filesystem and the host path is not - mounted. Files staged into an auto-mounted cache dir (``images/``, - ``attachments/``, ...) are translated to their in-container path via the - existing ``tools.credential_files`` machinery (#76577). Falls back to the - host path when the backend is local or translation is unavailable. + Under a container backend the host path dangles inside the sandbox; files staged + into an auto-mounted cache dir are translated via ``tools.credential_files``. + Falls back to the host path when the backend is local or translation fails. """ try: - # Desktop/in-process gateways may not have bridged ``terminal.*`` - # config into ``TERMINAL_ENV`` at startup; run the idempotent bridge so - # the credential_files translation gate sees the active backend. + # In-process gateways may not have bridged terminal.* config into TERMINAL_ENV + # yet; run the idempotent bridge so the translation gate sees the active backend. from tools.terminal_tool import _ensure_terminal_env_bridged _ensure_terminal_env_bridged() @@ -709,18 +644,20 @@ def _file_metadata(path: Path) -> str: return f"{line_count} lines" +_FENCE_LANGUAGES = { + ".py": "python", + ".js": "javascript", + ".ts": "typescript", + ".tsx": "tsx", + ".jsx": "jsx", + ".json": "json", + ".md": "markdown", + ".sh": "bash", + ".yml": "yaml", + ".yaml": "yaml", + ".toml": "toml", +} + + def _code_fence_language(path: Path) -> str: - mapping = { - ".py": "python", - ".js": "javascript", - ".ts": "typescript", - ".tsx": "tsx", - ".jsx": "jsx", - ".json": "json", - ".md": "markdown", - ".sh": "bash", - ".yml": "yaml", - ".yaml": "yaml", - ".toml": "toml", - } - return mapping.get(path.suffix.lower(), "") + return _FENCE_LANGUAGES.get(path.suffix.lower(), "") diff --git a/agent/display.py b/agent/display.py index 7a5dfc47a4..109f3f205f 100644 --- a/agent/display.py +++ b/agent/display.py @@ -1,7 +1,6 @@ """CLI presentation -- spinner, kawaii faces, tool preview formatting. -Pure display functions and classes with no AIAgent dependency. -Used by AIAgent._execute_tool_calls for CLI feedback. +Pure display functions with no AIAgent dependency; used for CLI feedback. """ import logging @@ -20,13 +19,11 @@ from utils import safe_json_loads from agent.redact import redact_sensitive_text from agent.tool_result_classification import file_mutation_result_landed -# ANSI escape codes for coloring tool failure indicators -_RED = "\033[31m" -_RESET = "\033[0m" - logger = logging.getLogger(__name__) _ANSI_RESET = "\033[0m" +_MAX_INLINE_DIFF_FILES = 6 +_MAX_INLINE_DIFF_LINES = 80 def _display_url(value: Any) -> str: @@ -36,9 +33,12 @@ def _display_url(value: Any) -> str: return value.strip() if isinstance(value, str) else "" -# Diff colors — resolved lazily from the skin engine so they adapt -# to light/dark themes. Falls back to sensible defaults on import -# failure. We cache after first resolution for performance. +def _hex_rgb(h: str) -> tuple[int, int, int]: + return int(h[1:3], 16), int(h[3:5], 16), int(h[5:7], 16) + + +# Diff colors resolve lazily from the skin engine (light/dark aware) and are +# cached after the first resolution. _diff_colors_cached: dict[str, str] | None = None @@ -47,69 +47,47 @@ def _diff_ansi() -> dict[str, str]: global _diff_colors_cached if _diff_colors_cached is not None: return _diff_colors_cached - # Defaults that work on dark terminals dim = "\033[38;2;150;150;150m" file_c = "\033[38;2;180;160;255m" hunk = "\033[38;2;120;120;140m" minus = "\033[38;2;255;255;255;48;2;120;20;20m" plus = "\033[38;2;255;255;255;48;2;20;90;20m" - try: from hermes_cli.skin_engine import get_active_skin skin = get_active_skin() def _hex_fg(key: str, fallback_rgb: tuple[int, int, int]) -> str: h = skin.get_color(key, "") - if h and len(h) == 7 and h[0] == "#": - r, g, b = int(h[1:3], 16), int(h[3:5], 16), int(h[5:7], 16) - return f"\033[38;2;{r};{g};{b}m" - r, g, b = fallback_rgb + r, g, b = _hex_rgb(h) if h and len(h) == 7 and h[0] == "#" else fallback_rgb return f"\033[38;2;{r};{g};{b}m" dim = _hex_fg("banner_dim", (150, 150, 150)) file_c = _hex_fg("session_label", (180, 160, 255)) hunk = _hex_fg("session_border", (120, 120, 140)) - # minus/plus use background colors — derive from ui_error/ui_ok + # minus/plus use dark-tinted backgrounds derived from ui_error/ui_ok err_h = skin.get_color("ui_error", "#ef5350") ok_h = skin.get_color("ui_ok", "#4caf50") if err_h and len(err_h) == 7: - er, eg, eb = int(err_h[1:3], 16), int(err_h[3:5], 16), int(err_h[5:7], 16) - # Use a dark tinted version as background + er, eg, eb = _hex_rgb(err_h) minus = f"\033[38;2;255;255;255;48;2;{max(er//2,20)};{max(eg//4,10)};{max(eb//4,10)}m" if ok_h and len(ok_h) == 7: - or_, og, ob = int(ok_h[1:3], 16), int(ok_h[3:5], 16), int(ok_h[5:7], 16) + or_, og, ob = _hex_rgb(ok_h) plus = f"\033[38;2;255;255;255;48;2;{max(or_//4,10)};{max(og//2,20)};{max(ob//4,10)}m" except Exception: pass - - _diff_colors_cached = { - "dim": dim, "file": file_c, "hunk": hunk, - "minus": minus, "plus": plus, - } + _diff_colors_cached = {"dim": dim, "file": file_c, "hunk": hunk, "minus": minus, "plus": plus} return _diff_colors_cached -# Module-level helpers — each call resolves from the active skin lazily. -def _diff_dim(): return _diff_ansi()["dim"] -def _diff_file(): return _diff_ansi()["file"] -def _diff_hunk(): return _diff_ansi()["hunk"] -def _diff_minus(): return _diff_ansi()["minus"] -def _diff_plus(): return _diff_ansi()["plus"] -_MAX_INLINE_DIFF_FILES = 6 -_MAX_INLINE_DIFF_LINES = 80 - - @dataclass class LocalEditSnapshot: """Pre-tool filesystem snapshot used to render diffs locally after writes.""" paths: list[Path] = field(default_factory=list) before: dict[str, str | None] = field(default_factory=dict) -# ========================================================================= -# Configurable tool preview length (0 = no limit) -# Set once at startup by CLI or gateway from display.tool_preview_length config. -# ========================================================================= + +# Configurable tool preview length; set once at startup from display.tool_preview_length. _tool_preview_max_len: int = 0 # 0 = unlimited @@ -124,12 +102,8 @@ def get_tool_preview_max_len() -> int: return _tool_preview_max_len -# ========================================================================= -# Skin-aware helpers (lazy import to avoid circular deps) -# ========================================================================= - def _get_skin(): - """Get the active skin config, or None if not available.""" + """Get the active skin config, or None if not available (lazy import avoids cycles).""" try: from hermes_cli.skin_engine import get_active_skin return get_active_skin() @@ -140,26 +114,16 @@ def _get_skin(): def get_skin_tool_prefix() -> str: """Get tool output prefix character from active skin.""" skin = _get_skin() - if skin: - return skin.tool_prefix - return "┊" + return skin.tool_prefix if skin else "┊" def get_tool_emoji(tool_name: str, default: str = "⚡") -> str: - """Get the display emoji for a tool. - - Resolution order: - 1. Active skin's ``tool_emojis`` overrides (if a skin is loaded) - 2. Tool registry's per-tool ``emoji`` field - 3. *default* fallback - """ - # 1. Skin override + """Display emoji for a tool: skin ``tool_emojis`` override, then registry, then *default*.""" skin = _get_skin() if skin and skin.tool_emojis: override = skin.tool_emojis.get(tool_name) if override: return override - # 2. Registry default try: from tools.registry import registry emoji = registry.get_emoji(tool_name, default="") @@ -167,7 +131,6 @@ def get_tool_emoji(tool_name: str, default: str = "⚡") -> str: return emoji except Exception: pass - # 3. Hardcoded fallback return default @@ -209,42 +172,34 @@ def _split_shell_words(segment: str) -> list[str]: words: list[str] = [] buf: list[str] = [] quote: str | None = None - for i, ch in enumerate(segment): if quote: buf.append(ch) if ch == quote and (i == 0 or segment[i - 1] != "\\"): quote = None continue - if ch in {"'", '"'}: quote = ch buf.append(ch) continue - if ch.isspace(): if buf: words.append("".join(buf)) buf = [] continue - buf.append(ch) - if buf: words.append("".join(buf)) - return words def _strip_shell_pipe_tail(segment: str) -> str: words = _split_shell_words(segment) out: list[str] = [] - for i, word in enumerate(words): if word == "|" and _shell_basename(words[i + 1] if i + 1 < len(words) else "") in _SHELL_PIPE_TAIL_HEADS: break out.append(word) - return " ".join(out).strip() @@ -254,38 +209,33 @@ def _split_shell_compound(command: str) -> list[str]: quote: str | None = None i = 0 + def _flush() -> None: + segment = _strip_shell_pipe_tail("".join(buf).strip()) + if segment: + segments.append(segment) + while i < len(command): ch = command[i] - if quote: buf.append(ch) if ch == quote and (i == 0 or command[i - 1] != "\\"): quote = None i += 1 continue - if ch in {"'", '"'}: quote = ch buf.append(ch) i += 1 continue - op_len = 2 if command.startswith("&&", i) or command.startswith("||", i) else 1 if ch in {";", "\n"} else 0 if op_len: - segment = _strip_shell_pipe_tail("".join(buf).strip()) - if segment: - segments.append(segment) + _flush() buf = [] i += op_len continue - buf.append(ch) i += 1 - - segment = _strip_shell_pipe_tail("".join(buf).strip()) - if segment: - segments.append(segment) - + _flush() return segments @@ -327,23 +277,19 @@ def summarize_shell_command(command: str) -> str: original = _oneline(command) if not original: return "" - segments = _split_shell_compound(original) if len(segments) <= 1: return _clean_shell_segment(segments[0] if segments else original) or original - core: list[str] = [] for segment in segments: cleaned = _clean_shell_segment(segment) head = _shell_head_word(cleaned) if cleaned and head not in _SHELL_SILENT_HEADS and not _is_shell_boundary_echo(cleaned): core.append(cleaned) - if not core: return original if len(core) == 1: return core[0] - count = len(core) - 1 return f"{core[0]} + {count} {'command' if count == 1 else 'commands'}" @@ -359,20 +305,13 @@ def _read_file_line_label(args: dict) -> str: def redact_browser_typed_text_for_display(value: Any, typed_text: Any) -> Any: - """Apply secret redaction to browser_type text in display-facing payloads. + """Replace every occurrence of a secret-looking browser_type value with its redacted form. - Backends sometimes echo the attempted input in error strings or fallback - metadata. When the raw typed value contains a recognizable secret (API - key, token, JWT, etc.) the redacted form differs from the raw value, so we - replace every occurrence of the raw value with its redacted form before a - browser_type result reaches logs, callbacks, the model, or chat history. - - Normal typed text (search queries, addresses, form fields) matches no - secret pattern, so it passes through unchanged and stays readable. - - Redaction is forced here regardless of the global ``security.redact_secrets`` - preference: a typed credential leaking into chat history is a security - boundary, not mere log hygiene. + Backends echo the attempted input in error strings/fallback metadata, so the raw + value is swapped for its redacted form before the result reaches logs, callbacks, + the model, or chat history. Normal typed text matches no secret pattern and passes + through unchanged. Redaction is forced regardless of ``security.redact_secrets``: + a typed credential leaking into chat history is a security boundary, not log hygiene. """ if typed_text is None: return value @@ -381,15 +320,11 @@ def redact_browser_typed_text_for_display(value: Any, typed_text: Any) -> Any: return value redacted = redact_sensitive_text(needle, force=True) if redacted == needle: - # Nothing secret-looking in the typed text; leave payload untouched. return value if isinstance(value, str): return value.replace(needle, redacted) if isinstance(value, dict): - return { - key: redact_browser_typed_text_for_display(item, typed_text) - for key, item in value.items() - } + return {key: redact_browser_typed_text_for_display(item, typed_text) for key, item in value.items()} if isinstance(value, list): return [redact_browser_typed_text_for_display(item, typed_text) for item in value] if isinstance(value, tuple): @@ -398,13 +333,7 @@ def redact_browser_typed_text_for_display(value: Any, typed_text: Any) -> Any: def redact_tool_args_for_display(tool_name: str, args: dict | None) -> dict | None: - """Return a copy of tool args safe for logs/progress UI. - - For ``browser_type`` the ``text`` argument is run through the same - secret-pattern redactor used for logs. Recognizable credentials (API - keys, tokens) are masked before the value reaches tool progress - notifications; normal typed text is left intact for debuggability. - """ + """Return a copy of tool args safe for logs/progress UI (masks ``browser_type`` secrets).""" if not isinstance(args, dict): return args if tool_name == "browser_type" and isinstance(args.get("text"), str): @@ -443,58 +372,62 @@ def _browser_exec_step_label(args: dict, max_chars: int = 80) -> str | None: return label +_PRIMARY_ARGS = { + "terminal": "command", "web_search": "query", "web_extract": "urls", + "read_file": "path", "write_file": "path", "patch": "path", + "search_files": "pattern", "browser_navigate": "url", + "browser_click": "ref", "browser_type": "text", + "image_generate": "prompt", "text_to_speech": "text", + "vision_analyze": "question", + "skill_view": "name", "skills_list": "category", + "cronjob_manage": "action", + "execute_code": "code", "browser_exec": "code", "delegate_task": "goal", + "clarify": "question", "skill_manage": "name", +} +_FALLBACK_PREVIEW_KEYS = ("query", "text", "command", "path", "name", "prompt", "code", "goal") + + +def _delegate_action_preview(args: dict) -> str | None: + """Shared ``list/steer/stop `` preview for delegate_task, or None for spawn calls.""" + action = str(args.get("action") or "").strip().lower() + if action in ("list", "steer", "stop"): + return f"{action} {str(args.get('subagent_id') or '').strip()}".strip() + return None + + def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -> str | None: """Build a short preview of a tool call's primary argument for display. - *max_len* controls truncation. ``None`` (default) defers to the global - ``_tool_preview_max_len`` set via config; ``0`` means unlimited. + *max_len* ``None`` defers to the global ``_tool_preview_max_len``; ``0`` means unlimited. """ if max_len is None: max_len = _tool_preview_max_len if not args: return None args = redact_tool_args_for_display(tool_name, args) or args - primary_args = { - "terminal": "command", "web_search": "query", "web_extract": "urls", - "read_file": "path", "write_file": "path", "patch": "path", - "search_files": "pattern", "browser_navigate": "url", - "browser_click": "ref", "browser_type": "text", - "image_generate": "prompt", "text_to_speech": "text", - "vision_analyze": "question", - "skill_view": "name", "skills_list": "category", - "cronjob_manage": "action", - "execute_code": "code", "browser_exec": "code", "delegate_task": "goal", - "clarify": "question", "skill_manage": "name", - } - # browser_exec: prefer the leading `# …` comment as a friendly step label + def _done(preview: str) -> str | None: + return _truncate_preview(preview, max_len) if preview else None + if tool_name == "browser_exec": label = _browser_exec_step_label(args) if label is not None: return _truncate_preview(label, max_len) - preview = _oneline(str(args.get("code", "") or "")) - return _truncate_preview(preview, max_len) if preview else None + return _done(_oneline(str(args.get("code", "") or ""))) - # delegate_task: show goal (single) or individual task goals (batch) if tool_name == "delegate_task": - action = str(args.get("action") or "").strip().lower() - if action in ("list", "steer", "stop"): - sid = str(args.get("subagent_id") or "").strip() - preview = f"{action} {sid}".strip() - return _truncate_preview(preview, max_len) + action_preview = _delegate_action_preview(args) + if action_preview is not None: + return _truncate_preview(action_preview, max_len) tasks = args.get("tasks") if tasks and isinstance(tasks, list): task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=40) - preview = ( - f"{task_count} tasks: " + " | ".join(goals) - if goals else f"{len(tasks)} parallel tasks" - ) + preview = f"{task_count} tasks: " + " | ".join(goals) if goals else f"{len(tasks)} parallel tasks" return _truncate_preview(preview, max_len) goal = args.get("goal", "") if goal is None: return None - preview = _oneline(str(goal)) - return _truncate_preview(preview, max_len) if preview else None + return _done(_oneline(str(goal))) if tool_name == "process_manage": action = args.get("action", "") @@ -513,30 +446,24 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - if tool_name == "todo_list": todos_arg = args.get("todos") - merge = args.get("merge", False) if todos_arg is None: return "reading task list" - elif merge: + if args.get("merge", False): return f"updating {len(todos_arg)} task(s)" - else: - return f"planning {len(todos_arg)} task(s)" + return f"planning {len(todos_arg)} task(s)" if tool_name in {"terminal", "execute_code"}: - key = "code" if tool_name == "execute_code" else "command" - command = args.get(key) + command = args.get("code" if tool_name == "execute_code" else "command") if command is None: return None - preview = summarize_shell_command(str(command)) - return _truncate_preview(preview, max_len) if preview else None + return _done(summarize_shell_command(str(command))) if tool_name == "read_file": path = args.get("path") or args.get("file") or args.get("filepath") if path is None: return None label = Path(str(path).replace("\\", "/")).name or str(path) - line_label = _read_file_line_label(args) - preview = f"{label} {line_label}".strip() - return _truncate_preview(preview, max_len) if preview else None + return _done(f"{label} {_read_file_line_label(args)}".strip()) if tool_name == "session_search": query = _oneline(args.get("query", "")) @@ -548,12 +475,9 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - if action == "add": content = _oneline(args.get("content", "")) return f"+{target}: \"{content[:25]}{'...' if len(content) > 25 else ''}\"" - elif action == "replace": + if action in ("replace", "remove"): old = _oneline(args.get("old_text") or "") or "" - return f"~{target}: \"{old[:20]}\"" - elif action == "remove": - old = _oneline(args.get("old_text") or "") or "" - return f"-{target}: \"{old[:20]}\"" + return f"{'~' if action == 'replace' else '-'}{target}: \"{old[:20]}\"" return action if tool_name == "send_message": @@ -568,25 +492,15 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) - file_path = args.get("file_path") if file_path: file_path = _oneline(str(file_path)) - preview = f"{name} → {file_path}" if name else file_path - else: - preview = name - return _truncate_preview(preview, max_len) if preview else None - - key = primary_args.get(tool_name) - if not key: - for fallback_key in ("query", "text", "command", "path", "name", "prompt", "code", "goal"): - if fallback_key in args: - key = fallback_key - break + return _done(f"{name} → {file_path}" if name else file_path) + return _done(name) + key = _PRIMARY_ARGS.get(tool_name) or next((k for k in _FALLBACK_PREVIEW_KEYS if k in args), None) if not key or key not in args: return None - value = args[key] if isinstance(value, list): value = value[0] if value else "" - preview = _oneline(str(value)) if not preview: return None @@ -602,12 +516,11 @@ def prepare_tool_preview( fallback: str, max_len: int, ) -> ToolPreview: - """Build one canonical compact preview before platform formatting. + """Build one canonical compact preview plus explicit truncation/URL metadata. - The uncapped preview is rebuilt from the tool arguments when possible so - an upstream display cap cannot discard its link target. Platforms then - receive explicit truncation and URL metadata instead of inferring either - fact from the rendered text. + The uncapped preview is rebuilt from the arguments so an upstream display cap + cannot discard its link target; platforms get truncation/URL facts explicitly + instead of inferring them from the rendered text. """ full_text = build_tool_preview(tool_name, args, max_len=0) or fallback text = _truncate_preview(full_text, max_len) @@ -625,17 +538,11 @@ def prepare_tool_preview( # ========================================================================= -# Friendly tool labels (human-phrased verbs for built-in tools) -# -# Turns "web_search " into "Searching the web for " — the -# ChatGPT-style "Searching…/Reading…" surface. Curated and built-in only: -# we know each core tool's semantics, so the verb is fixed, not computed. -# Custom/plugin/MCP tools have no entry and fall back to the raw preview. +# Friendly tool labels: "web_search " -> "Searching the web for ". +# Curated built-ins only — we know each core tool's semantics so the verb is fixed, +# not computed; custom/plugin/MCP tools have no entry and fall back to the raw preview. # ========================================================================= -# Each entry maps a built-in tool name to its present-participle verb phrase. -# A trailing space-then-preview is appended by build_tool_label() when the -# tool's argument preview is available (e.g. "Reading docs/api.md"). _TOOL_VERBS: dict[str, str] = { "web_search": "Searching the web", "web_extract": "Reading", @@ -662,19 +569,10 @@ _TOOL_VERBS: dict[str, str] = { "memory": "Updating memory", "todo_list": "Updating tasks", } - -# Verbs that read better without the raw argument preview appended. -_TOOL_VERBS_NO_PREVIEW: frozenset[str] = frozenset({ - "skills_list", - "session_search", -}) - -# Verbs that take a "for" connector before the preview (search-style phrasing): -# "Searching the web for " reads better than "Searching the web ". -_TOOL_VERBS_FOR_CONNECTOR: frozenset[str] = frozenset({ - "web_search", - "search_files", -}) +# Verbs that read better without the argument preview appended. +_TOOL_VERBS_NO_PREVIEW: frozenset[str] = frozenset({"skills_list", "session_search"}) +# Verbs joined to the preview with " for " (search-style phrasing). +_TOOL_VERBS_FOR_CONNECTOR: frozenset[str] = frozenset({"web_search", "search_files"}) _friendly_tool_labels: bool = True @@ -685,18 +583,11 @@ def set_friendly_tool_labels(enabled: bool) -> None: _friendly_tool_labels = bool(enabled) -def get_friendly_tool_labels() -> bool: - """Return whether friendly tool labels are enabled.""" - return _friendly_tool_labels - - def get_tool_verb(tool_name: str) -> str | None: - """Return the friendly verb for a built-in tool, or None. + """Friendly verb for a built-in tool, or None (labels disabled / no curated verb). - Returns None when friendly labels are disabled or the tool has no curated - verb (custom/plugin/MCP tools). Callers that already hold a computed - argument preview can compose ``f"{verb} {preview}"`` themselves; use - :func:`tool_verb_connector` to pick the right joiner. + Callers holding a computed preview compose ``f"{verb}{connector}{preview}"`` + themselves via :func:`tool_verb_connector`. """ if not _friendly_tool_labels: return None @@ -714,76 +605,45 @@ def verb_drops_preview(tool_name: str) -> bool: def build_status_phrase(tool_name: str, args: dict | None, max_len: int = 49) -> str | None: - """Build a short present-tense status phrase for platform status surfaces. + """Lowercase "is …" phrase for platform status lines (e.g. Slack setStatus). - Used by text-rendering "typing" indicators (Slack's - ``assistant.threads.setStatus`` line) to show what the agent is doing - right now: ``is running scripts/run_tests.sh…`` instead of a static - ``is thinking...``. The phrase is phrased to follow the bot's display - name ("Hermes is running …"), so it starts lowercase with "is". - - Pass ``args=None`` for a verb-only phrase (``is running…``) — used when - ``display.live_status`` is ``verb`` to keep argument previews out of - shared channels. - - Returns None for the ``_thinking`` pseudo-tool and when friendly labels - are disabled (callers fall back to their static default). ``max_len`` - caps the total phrase length; Slack truncates its status line around 50 - characters, so the default stays just under that. + Phrased to follow the bot's display name ("Hermes is running …"). ``args=None`` + gives a verb-only phrase (``display.live_status: verb`` keeps argument previews + out of shared channels). Returns None for ``_thinking`` or when friendly labels + are disabled so callers fall back to their static default. Default ``max_len`` + stays under Slack's ~50-char status truncation. """ - if not tool_name or tool_name == "_thinking": + if not tool_name or tool_name == "_thinking" or not _friendly_tool_labels: return None - if not _friendly_tool_labels: - return None - verb = _TOOL_VERBS.get(tool_name) - if verb: - head = f"is {verb[0].lower()}{verb[1:]}" - else: - # Custom / plugin / MCP tools: generic but still informative. - head = f"is using {tool_name}" - + head = f"is {verb[0].lower()}{verb[1:]}" if verb else f"is using {tool_name}" phrase = head if args and verb and tool_name not in _TOOL_VERBS_NO_PREVIEW: preview = build_tool_preview(tool_name, args, max_len=None) if preview: - # Previews can contain newlines (terminal commands); keep the - # status to the first line. + # Previews can contain newlines (terminal commands); keep the first line. preview = preview.splitlines()[0].strip() phrase = f"{head}{tool_verb_connector(tool_name)}{preview}" - if len(phrase) > max_len - 1: - phrase = phrase[: max_len - 2].rstrip() + "…" - else: - phrase = phrase + "…" - return phrase + return phrase[: max_len - 2].rstrip() + "…" + return phrase + "…" def build_tool_label(tool_name: str, args: dict, max_len: int | None = None) -> str | None: - """Build a human-phrased status label for a tool call. + """Human-phrased label ("Searching the web for ...") for curated built-ins. - For built-in tools with a known verb (``web_search`` -> "Searching the - web for ..."), returns the verb optionally followed by the argument - preview. For everything else (custom/plugin/MCP tools, or when friendly - labels are disabled) returns the raw preview, so callers can use this as a + Custom/plugin/MCP tools (or labels disabled) get the raw preview, so this is a drop-in replacement for :func:`build_tool_preview`. """ - if not _friendly_tool_labels: - return build_tool_preview(tool_name, args, max_len=max_len) - - verb = _TOOL_VERBS.get(tool_name) + verb = _TOOL_VERBS.get(tool_name) if _friendly_tool_labels else None if not verb: return build_tool_preview(tool_name, args, max_len=max_len) - if tool_name in _TOOL_VERBS_NO_PREVIEW: return verb - preview = build_tool_preview(tool_name, args, max_len=max_len) if not preview: return verb - if tool_name in _TOOL_VERBS_FOR_CONNECTOR: - return f"{verb} for {preview}" - return f"{verb} {preview}" + return f"{verb}{tool_verb_connector(tool_name)}{preview}" # ========================================================================= @@ -793,9 +653,7 @@ def build_tool_label(tool_name: str, args: dict, max_len: int | None = None) -> def _resolved_path(path: str) -> Path: """Resolve a possibly-relative filesystem path against the current cwd.""" candidate = Path(os.path.expanduser(path)) - if candidate.is_absolute(): - return candidate - return Path.cwd() / candidate + return candidate if candidate.is_absolute() else Path.cwd() / candidate def _snapshot_text(path: Path) -> str | None: @@ -820,27 +678,21 @@ def _resolve_skill_manage_paths(args: dict) -> list[Path]: name = args.get("name") if not action or not name: return [] - from tools.skill_manager_tool import _find_skill, _resolve_skill_dir if action == "create": - skill_dir = _resolve_skill_dir(name, args.get("category")) - return [skill_dir / "SKILL.md"] - + return [_resolve_skill_dir(name, args.get("category")) / "SKILL.md"] existing = _find_skill(name) if not existing: return [] - skill_dir = Path(existing["path"]) + file_path = args.get("file_path") if action in {"edit", "patch"}: - file_path = args.get("file_path") return [skill_dir / file_path] if file_path else [skill_dir / "SKILL.md"] if action in {"write_file", "remove_file"}: - file_path = args.get("file_path") return [skill_dir / file_path] if file_path else [] if action == "delete": - files = [path for path in sorted(skill_dir.rglob("*")) if path.is_file()] - return files + return [path for path in sorted(skill_dir.rglob("*")) if path.is_file()] return [] @@ -848,18 +700,11 @@ def _resolve_local_edit_paths(tool_name: str, function_args: dict | None) -> lis """Resolve local filesystem targets for write-capable tools.""" if not isinstance(function_args, dict): return [] - - if tool_name == "write_file": + if tool_name in {"write_file", "patch"}: path = function_args.get("path") return [_resolved_path(path)] if path else [] - - if tool_name == "patch": - path = function_args.get("path") - return [_resolved_path(path)] if path else [] - if tool_name == "skill_manage": return _resolve_skill_manage_paths(function_args) - return [] @@ -868,11 +713,7 @@ def capture_local_edit_snapshot(tool_name: str, function_args: dict | None) -> L paths = _resolve_local_edit_paths(tool_name, function_args) if not paths: return None - - snapshot = LocalEditSnapshot(paths=paths) - for path in paths: - snapshot.before[str(path)] = _snapshot_text(path) - return snapshot + return LocalEditSnapshot(paths=paths, before={str(path): _snapshot_text(path) for path in paths}) def _result_succeeded(result: str | None) -> bool: @@ -880,11 +721,7 @@ def _result_succeeded(result: str | None) -> bool: if not result: return False data = safe_json_loads(result) - if data is None: - return False - if not isinstance(data, dict): - return False - if data.get("error"): + if not isinstance(data, dict) or data.get("error"): return False if "success" in data: return bool(data.get("success")) @@ -895,14 +732,12 @@ def _diff_from_snapshot(snapshot: LocalEditSnapshot | None) -> str | None: """Generate unified diff text from a stored before-state and current files.""" if not snapshot: return None - chunks: list[str] = [] for path in snapshot.paths: before = snapshot.before.get(str(path)) after = _snapshot_text(path) if before == after: continue - display_path = _display_diff_path(path) diff = "".join( unified_diff( @@ -914,7 +749,6 @@ def _diff_from_snapshot(snapshot: LocalEditSnapshot | None) -> str | None: ) if diff: chunks.append(diff) - if not chunks: return None return "".join(chunk if chunk.endswith("\n") else chunk + "\n" for chunk in chunks) @@ -934,10 +768,7 @@ def extract_edit_diff( diff = data.get("diff") if isinstance(diff, str) and diff.strip(): return diff - - if tool_name not in {"write_file", "patch", "skill_manage"}: - return None - if not _result_succeeded(result): + if tool_name not in {"write_file", "patch", "skill_manage"} or not _result_succeeded(result): return None return _diff_from_snapshot(snapshot) @@ -955,12 +786,15 @@ def _emit_inline_diff(diff_text: str, print_fn) -> bool: return False +# Unified-diff line prefix -> diff color key (checked in order; "--- "/"+++ " handled first). +_DIFF_LINE_COLORS = (("@@", "hunk"), ("-", "minus"), ("+", "plus"), (" ", "dim")) + + def _render_inline_unified_diff(diff: str) -> list[str]: """Render unified diff lines in Hermes' inline transcript style.""" rendered: list[str] = [] from_file = None to_file = None - for raw_line in diff.splitlines(): if raw_line.startswith("--- "): from_file = raw_line[4:].strip() @@ -968,23 +802,15 @@ def _render_inline_unified_diff(diff: str) -> list[str]: if raw_line.startswith("+++ "): to_file = raw_line[4:].strip() if from_file or to_file: - rendered.append(f"{_diff_file()}{from_file or 'a/?'} → {to_file or 'b/?'}{_ANSI_RESET}") + rendered.append(f"{_diff_ansi()['file']}{from_file or 'a/?'} → {to_file or 'b/?'}{_ANSI_RESET}") continue - if raw_line.startswith("@@"): - rendered.append(f"{_diff_hunk()}{raw_line}{_ANSI_RESET}") - continue - if raw_line.startswith("-"): - rendered.append(f"{_diff_minus()}{raw_line}{_ANSI_RESET}") - continue - if raw_line.startswith("+"): - rendered.append(f"{_diff_plus()}{raw_line}{_ANSI_RESET}") - continue - if raw_line.startswith(" "): - rendered.append(f"{_diff_dim()}{raw_line}{_ANSI_RESET}") - continue - if raw_line: - rendered.append(raw_line) - + for prefix, color in _DIFF_LINE_COLORS: + if raw_line.startswith(prefix): + rendered.append(f"{_diff_ansi()[color]}{raw_line}{_ANSI_RESET}") + break + else: + if raw_line: + rendered.append(raw_line) return rendered @@ -992,17 +818,14 @@ def _split_unified_diff_sections(diff: str) -> list[str]: """Split a unified diff into per-file sections.""" sections: list[list[str]] = [] current: list[str] = [] - for line in diff.splitlines(): if line.startswith("--- ") and current: sections.append(current) current = [line] continue current.append(line) - if current: sections.append(current) - return ["\n".join(section) for section in sections if section] @@ -1017,37 +840,27 @@ def _summarize_rendered_diff_sections( rendered: list[str] = [] omitted_files = 0 omitted_lines = 0 - for idx, section in enumerate(sections): - if idx >= max_files: - omitted_files += 1 - omitted_lines += len(_render_inline_unified_diff(section)) - continue - section_lines = _render_inline_unified_diff(section) remaining_budget = max_lines - len(rendered) - if remaining_budget <= 0: - omitted_lines += len(section_lines) + if idx >= max_files or remaining_budget <= 0: omitted_files += 1 + omitted_lines += len(section_lines) continue - if len(section_lines) <= remaining_budget: rendered.extend(section_lines) continue - rendered.extend(section_lines[:remaining_budget]) omitted_lines += len(section_lines) - remaining_budget omitted_files += 1 + max(0, len(sections) - idx - 1) for leftover in sections[idx + 1:]: omitted_lines += len(_render_inline_unified_diff(leftover)) break - if omitted_files or omitted_lines: summary = f"… omitted {omitted_lines} diff line(s)" if omitted_files: summary += f" across {omitted_files} additional file(s)/section(s)" - rendered.append(f"{_diff_hunk()}{summary}{_ANSI_RESET}") - + rendered.append(f"{_diff_ansi()['hunk']}{summary}{_ANSI_RESET}") return rendered @@ -1060,12 +873,7 @@ def render_edit_diff_with_delta( print_fn=None, ) -> bool: """Render an edit diff inline without taking over the terminal UI.""" - diff = extract_edit_diff( - tool_name, - result, - function_args=function_args, - snapshot=snapshot, - ) + diff = extract_edit_diff(tool_name, result, function_args=function_args, snapshot=snapshot) if not diff: return False try: @@ -1112,44 +920,30 @@ class KawaiiSpinner: "analyzing", "computing", "synthesizing", "formulating", "brainstorming", ] - @classmethod - def get_waiting_faces(cls) -> list: - """Return waiting faces from the active skin, falling back to KAWAII_WAITING.""" + @staticmethod + def _skin_spinner_list(key: str, fallback: list) -> list: + """Return the active skin's ``spinner[key]`` list, or *fallback* when absent/empty.""" try: skin = _get_skin() if skin: - faces = skin.spinner.get("waiting_faces", []) - if faces: - return faces + values = skin.spinner.get(key, []) + if values: + return values except Exception: pass - return cls.KAWAII_WAITING + return fallback + + @classmethod + def get_waiting_faces(cls) -> list: + return cls._skin_spinner_list("waiting_faces", cls.KAWAII_WAITING) @classmethod def get_thinking_faces(cls) -> list: - """Return thinking faces from the active skin, falling back to KAWAII_THINKING.""" - try: - skin = _get_skin() - if skin: - faces = skin.spinner.get("thinking_faces", []) - if faces: - return faces - except Exception: - pass - return cls.KAWAII_THINKING + return cls._skin_spinner_list("thinking_faces", cls.KAWAII_THINKING) @classmethod def get_thinking_verbs(cls) -> list: - """Return thinking verbs from the active skin, falling back to THINKING_VERBS.""" - try: - skin = _get_skin() - if skin: - verbs = skin.spinner.get("thinking_verbs", []) - if verbs: - return verbs - except Exception: - pass - return cls.THINKING_VERBS + return cls._skin_spinner_list("thinking_verbs", cls.THINKING_VERBS) def __init__(self, message: str = "", spinner_type: str = 'dots', print_fn=None): self.message = message @@ -1159,20 +953,13 @@ class KawaiiSpinner: self.frame_idx = 0 self.start_time = None self.last_line_len = 0 - # Optional callable to route all output through (e.g. a no-op for silent - # background agents). When set, bypasses self._out entirely so that - # agents with _print_fn overridden remain fully silent. + # When set, all output bypasses self._out so silenced agents stay silent. self._print_fn = print_fn - # Capture stdout NOW, before any redirect_stdout(devnull) from - # child agents can replace sys.stdout with a black hole. + # Capture stdout NOW, before any child redirect_stdout(devnull) replaces it. self._out = sys.stdout def _write(self, text: str, end: str = '\n', flush: bool = False): - """Write to the stdout captured at spinner creation time. - - If a print_fn was supplied at construction, all output is routed through - it instead — allowing callers to silence the spinner with a no-op lambda. - """ + """Write via print_fn when supplied, else to the stdout captured at creation.""" if self._print_fn is not None: try: self._print_fn(text) @@ -1195,14 +982,11 @@ class KawaiiSpinner: return False def _is_patch_stdout_proxy(self) -> bool: - """Return True when stdout is prompt_toolkit's StdoutProxy. + """True when stdout is prompt_toolkit's StdoutProxy. - patch_stdout wraps sys.stdout in a StdoutProxy that queues writes and - injects newlines around each flush(). The \\r overwrite never lands on - the correct line — each spinner frame ends up on its own line. - - The CLI already drives a TUI widget (_spinner_text) for spinner display, - so KawaiiSpinner's \\r-based animation is redundant under StdoutProxy. + StdoutProxy queues writes and injects newlines around each flush, so the + \\r overwrite never lands: each frame would land on its own line. The CLI + drives its own TUI spinner widget in that mode, so we stay silent. """ try: from prompt_toolkit.patch_stdout import StdoutProxy @@ -1211,29 +995,19 @@ class KawaiiSpinner: return False def _animate(self): - # When stdout is not a real terminal (e.g. Docker, systemd, pipe), - # skip the animation entirely — it creates massive log bloat. - # Just log the start once and let stop() log the completion. + # Non-TTY (Docker, systemd, pipe): log once instead of spamming frames. if not self._is_tty: self._write(f" [tool] {self.message}", flush=True) while self.running: time.sleep(0.5) return - - # When running inside prompt_toolkit's patch_stdout context the CLI - # renders spinner state via a dedicated TUI widget (_spinner_text). - # Driving a \r-based animation here too causes visual overdraw: the - # StdoutProxy injects newlines around each flush, so every frame lands - # on a new line and overwrites the status bar. + # Under patch_stdout the \r animation would overdraw the TUI status bar. if self._is_patch_stdout_proxy(): while self.running: time.sleep(0.1) return - - # Cache skin wings at start (avoid per-frame imports) skin = _get_skin() wings = skin.get_spinner_wings() if skin else [] - while self.running: if os.getenv("HERMES_SPINNER_PAUSE"): time.sleep(0.1) @@ -1262,35 +1036,28 @@ class KawaiiSpinner: def update_text(self, new_message: str): self.message = new_message - def print_above(self, text: str): - """Print a line above the spinner without disrupting animation. + def _clear_line_blanks(self) -> str: + # Clear with spaces (not \033[K) to avoid garbled escapes under patch_stdout. + return ' ' * max(self.last_line_len + 5, 40) - Clears the current spinner line, prints the text, and lets the - next animation tick redraw the spinner on the line below. - Thread-safe: uses the captured stdout reference (self._out). - Works inside redirect_stdout(devnull) because _write bypasses - sys.stdout and writes to the stdout captured at spinner creation. + def print_above(self, text: str): + """Print a line above the spinner; the next tick redraws the spinner below it. + + Works inside redirect_stdout(devnull) because _write targets the stdout + captured at spinner creation, not the current sys.stdout. """ if not self.running: self._write(f" {text}", flush=True) return - # Clear spinner line with spaces (not \033[K) to avoid garbled escape - # codes when prompt_toolkit's patch_stdout is active — same approach - # as stop(). Then print text; spinner redraws on next tick. - blanks = ' ' * max(self.last_line_len + 5, 40) - self._write(f"\r{blanks}\r {text}", flush=True) + self._write(f"\r{self._clear_line_blanks()}\r {text}", flush=True) def stop(self, final_message: str = None): self.running = False if self.thread: self.thread.join(timeout=0.5) - is_tty = self._is_tty if is_tty: - # Clear the spinner line with spaces instead of \033[K to avoid - # garbled escape codes when prompt_toolkit's patch_stdout is active. - blanks = ' ' * max(self.last_line_len + 5, 40) - self._write(f"\r{blanks}\r", end='', flush=True) + self._write(f"\r{self._clear_line_blanks()}\r", end='', flush=True) if final_message: elapsed = f" ({time.time() - self.start_time:.1f}s)" if self.start_time else "" if is_tty: @@ -1302,7 +1069,7 @@ class KawaiiSpinner: self.start() return self - def __exit__(self, exc_type, exc_val, exc_tb): + def __exit__(self, *exc): self.stop() return False @@ -1315,13 +1082,8 @@ _ERROR_SUFFIX_MAX_LEN = 48 def _trim_error(msg: str) -> str: - """Shrink an error message for inline display in a tool status line. - - Strips overly long absolute paths down to just the filename so the - suffix stays readable on narrow terminals. - """ + """Shrink an error message for inline display (long 'File not found' paths -> filename).""" msg = msg.strip() - # Common case: "File not found: /very/long/absolute/path/foo.py" if "File not found:" in msg: _, _, tail = msg.partition("File not found:") tail = tail.strip() @@ -1333,18 +1095,9 @@ def _trim_error(msg: str) -> str: def _detect_tool_failure(tool_name: str, result: str | None) -> tuple[bool, str]: - """Inspect a tool result string for signs of failure. - - Returns ``(is_failure, suffix)`` where *suffix* is a short informational - tag like ``" [exit 1]"`` for terminal failures, ``" [full]"`` for memory - overflow, or a trimmed error message (``" [File not found: foo.py]"``). - On success returns ``(False, "")``. - """ - if result is None: + """Return ``(is_failure, suffix)`` for a tool result, e.g. ``(True, " [exit 1]")``.""" + if result is None or file_mutation_result_landed(tool_name, result): return False, "" - if file_mutation_result_landed(tool_name, result): - return False, "" - data = safe_json_loads(result) # Terminal: non-zero exit code is the canonical failure signal. @@ -1358,217 +1111,203 @@ def _detect_tool_failure(tool_name: str, result: str | None) -> tuple[bool, str] return True, f" [exit {exit_code}]" return False, "" - # Memory: distinguish "store full" from real errors. - if tool_name == "memory": - if isinstance(data, dict): - if data.get("success") is False and "exceed the limit" in data.get("error", ""): - return True, " [full]" - - # Structured error in JSON result (any tool that surfaces {"error": ...}). if isinstance(data, dict): + # Memory: distinguish "store full" from real errors. + if ( + tool_name == "memory" + and data.get("success") is False + and "exceed the limit" in data.get("error", "") + ): + return True, " [full]" err = data.get("error") or data.get("message") if err and (data.get("success") is False or "error" in data): return True, f" [{_trim_error(str(err))}]" - # Generic heuristic for non-terminal tools - # Multimodal tool results (dicts with _multimodal=True) are not strings — - # treat them as successes since failures would be JSON-encoded strings. + # Multimodal results (dicts) are successes; failures arrive as JSON-encoded strings. if not isinstance(result, str): return False, "" lower = result[:500].lower() if '"error"' in lower or '"failed"' in lower or result.startswith("Error"): return True, " [error]" - return False, "" +def _domain(url: str) -> str: + return url.replace("https://", "").replace("http://", "").split("/")[0] + + +def _cute_trunc(s) -> str: + """Tail-truncate to the configured preview cap (0 = unlimited).""" + s = str(s) + limit = _tool_preview_max_len + if limit == 0: + return s + return (s[:limit-3] + "...") if len(s) > limit else s + + +def _cute_path(p) -> str: + """Head-truncate a path to the configured preview cap, keeping the filename end.""" + p = str(p) + limit = _tool_preview_max_len + if limit == 0: + return p + return ("..." + p[-(limit-3):]) if len(p) > limit else p + + +def _cute_web_extract(a: dict, _r) -> str: + urls = a.get("urls", []) + if urls: + url = _display_url(urls[0] if isinstance(urls, list) else urls) + if url: + extra = f" +{len(urls)-1}" if isinstance(urls, list) and len(urls) > 1 else "" + return f"┊ 📄 fetch {_cute_trunc(_domain(url))}{extra}" + return "┊ 📄 fetch pages" + + +def _cute_todo_list(a: dict, result) -> str: + todos_arg = a.get("todos") + total = done = 0 + if result: + try: + data = safe_json_loads(result) + if data: + s = data.get("summary", {}) + total = s.get("total", 0) + done = s.get("completed", 0) + except Exception: + pass + if todos_arg is None: + detail = f"{done}/{total} task(s)" if total > 0 else "reading tasks" + elif a.get("merge", False): + detail = f"update {done}/{total} ✓" if total > 0 and done > 0 else f"update {len(todos_arg)} task(s)" + else: + detail = f"{done}/{total} task(s)" if total > 0 and done > 0 else f"{len(todos_arg)} task(s)" + return f"┊ 📋 plan {detail}" + + +def _cute_memory(a: dict, _r) -> str: + action = a.get("action", "?") + target = a.get("target", "") + if action == "add": + return f"┊ 🧠 memory +{target}: \"{_cute_trunc(a.get('content', ''))}\"" + if action in ("replace", "remove"): + old = a.get("old_text") or "" + return f"┊ 🧠 memory {'~' if action == 'replace' else '-'}{target}: \"{_cute_trunc(old)}\"" + return f"┊ 🧠 memory {action}" + + +def _cute_skill_view(a: dict, _r) -> str: + label = a.get("name", "") + file_path = a.get("file_path") + if file_path: + label = f"{label} → {file_path}" if label else str(file_path) + return f"┊ 📚 skill {_cute_trunc(label)}" + + +def _cute_cronjob(a: dict, _r) -> str: + action = a.get("action", "?") + if action == "create": + skills = a.get("skills") or ([] if not a.get("skill") else [a.get("skill")]) + label = a.get("name") or (skills[0] if skills else None) or a.get("prompt", "task") + return f"┊ ⏰ cron create {_cute_trunc(label)}" + if action == "list": + return "┊ ⏰ cron listing" + return f"┊ ⏰ cron {action} {a.get('job_id', '')}" + + +def _cute_execute_code(a: dict, _r) -> str: + code = a.get("code", "") + first_line = code.strip().split("\n")[0] if code.strip() else "" + return f"┊ 🐍 exec {_cute_trunc(first_line)}" + + +def _cute_browser_exec(a: dict, _r) -> str: + # Leading `# …` comment becomes the step label; code stays collapsed behind the preview cap. + label = _browser_exec_step_label(a) + if label is not None: + return f"┊ 🌐 browser {label}" + return f"┊ 🌐 browser {_cute_trunc(' '.join(str(a.get('code', '') or '').split()))}" + + +def _cute_delegate(a: dict, _r) -> str: + action_preview = _delegate_action_preview(a) + if action_preview is not None: + return f"┊ 🔀 delegate {_cute_trunc(action_preview)}" + tasks = a.get("tasks") + if tasks and isinstance(tasks, list): + task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=30) + detail = " | ".join(goals) if goals else "parallel" + return f"┊ 🔀 delegate {task_count or len(tasks)}x: {_cute_trunc(detail)}" + return f"┊ 🔀 delegate {_cute_trunc(a.get('goal', ''))}" + + +def _cute_process_manage(a: dict, _r) -> str: + action = a.get("action", "?") + sid = a.get("session_id", "")[:12] + return f"┊ ⚙️ proc {'ls processes' if action == 'list' else f'{action} {sid}'}" + + +_SCROLL_ARROWS = {"down": "↓", "up": "↑", "right": "→", "left": "←"} + +# Completion-line renderers: tool -> f(args, result) -> "┊ {emoji} {verb:9} {detail}" (duration appended by caller). +_CUTE_LINES = { + "web_search": lambda a, r: f"┊ 🔍 search {_cute_trunc(a.get('query', ''))}", + "web_extract": _cute_web_extract, + "terminal": lambda a, r: f"┊ 💻 $ {_cute_trunc(build_tool_preview('terminal', a) or a.get('command', ''))}", + "process_manage": _cute_process_manage, + "read_file": lambda a, r: f"┊ 📖 read {_cute_trunc(build_tool_preview('read_file', a) or a.get('path', ''))}", + "write_file": lambda a, r: f"┊ ✍️ write {_cute_path(a.get('path', ''))}", + "patch": lambda a, r: f"┊ 🔧 patch {_cute_path(a.get('path', ''))}", + "search_files": lambda a, r: ( + f"┊ 🔎 {'find' if a.get('target', 'content') == 'files' else 'grep':9} {_cute_trunc(a.get('pattern', ''))}" + ), + "browser_navigate": lambda a, r: f"┊ 🌐 navigate {_cute_trunc(_domain(a.get('url', '')))}", + "browser_snapshot": lambda a, r: f"┊ 📸 snapshot {'full' if a.get('full') else 'compact'}", + "browser_click": lambda a, r: f"┊ 👆 click {a.get('ref', '?')}", + "browser_type": lambda a, r: f"┊ ⌨️ type \"{_cute_trunc(a.get('text', ''))}\"", + "browser_scroll": lambda a, r: ( + f"┊ {_SCROLL_ARROWS.get(a.get('direction', 'down'), '↓')} scroll {a.get('direction', 'down')}" + ), + "browser_back": lambda a, r: "┊ ◀️ back ", + "browser_press": lambda a, r: f"┊ ⌨️ press {a.get('key', '?')}", + "browser_get_images": lambda a, r: "┊ 🖼️ images extracting", + "browser_vision": lambda a, r: "┊ 👁️ vision analyzing page", + "todo_list": _cute_todo_list, + "session_search": lambda a, r: f"┊ 🔍 recall \"{_cute_trunc(a.get('query', ''))}\"", + "memory": _cute_memory, + "skills_list": lambda a, r: f"┊ 📚 skills list {a.get('category', 'all')}", + "skill_view": _cute_skill_view, + "image_generate": lambda a, r: f"┊ 🎨 create {_cute_trunc(a.get('prompt', ''))}", + "text_to_speech": lambda a, r: f"┊ 🔊 speak {_cute_trunc(a.get('text', ''))}", + "vision_analyze": lambda a, r: f"┊ 👁️ vision {_cute_trunc(a.get('question', ''))}", + "send_message": lambda a, r: f"┊ 📨 send {a.get('target', '?')}: \"{_cute_trunc(a.get('message', ''))}\"", + "cronjob_manage": _cute_cronjob, + "execute_code": _cute_execute_code, + "browser_exec": _cute_browser_exec, + "delegate_task": _cute_delegate, +} + + def _get_cute_tool_message( tool_name: str, args: dict, duration: float, result: str | None = None, ) -> str: - """Generate a formatted tool completion line for CLI quiet mode. + """Formatted tool completion line for CLI quiet mode: ``| {emoji} {verb:9} {detail} {duration}``. - Format: ``| {emoji} {verb:9} {detail} {duration}`` - - When *result* is provided the line is checked for failure indicators. - Failed tool calls get a red prefix and an informational suffix. + Failed tool calls get an informational suffix from :func:`_detect_tool_failure`; + the leading ``┊`` is swapped for the active skin's tool prefix. """ args = redact_tool_args_for_display(tool_name, args) or args - dur = f"{duration:.1f}s" is_failure, failure_suffix = _detect_tool_failure(tool_name, result) skin_prefix = get_skin_tool_prefix() - - def _trunc(s, n=40): - s = str(s) - if _tool_preview_max_len == 0: - return s # no limit - limit = _tool_preview_max_len - return (s[:limit-3] + "...") if len(s) > limit else s - - def _path(p, n=35): - p = str(p) - if _tool_preview_max_len == 0: - return p # no limit - limit = _tool_preview_max_len - return ("..." + p[-(limit-3):]) if len(p) > limit else p - - def _wrap(line: str) -> str: - """Apply skin tool prefix and failure suffix.""" - if skin_prefix != "┊": - line = line.replace("┊", skin_prefix, 1) - if not is_failure: - return line - return f"{line}{failure_suffix}" - - if tool_name == "web_search": - return _wrap(f"┊ 🔍 search {_trunc(args.get('query', ''), 42)} {dur}") - if tool_name == "web_extract": - urls = args.get("urls", []) - if urls: - url = _display_url(urls[0] if isinstance(urls, list) else urls) - if not url: - return _wrap(f"┊ 📄 fetch pages {dur}") - domain = url.replace("https://", "").replace("http://", "").split("/")[0] - extra = f" +{len(urls)-1}" if isinstance(urls, list) and len(urls) > 1 else "" - return _wrap(f"┊ 📄 fetch {_trunc(domain, 35)}{extra} {dur}") - return _wrap(f"┊ 📄 fetch pages {dur}") - if tool_name == "terminal": - return _wrap(f"┊ 💻 $ {_trunc(build_tool_preview(tool_name, args) or args.get('command', ''), 42)} {dur}") - if tool_name == "process_manage": - action = args.get("action", "?") - sid = args.get("session_id", "")[:12] - labels = {"list": "ls processes", "poll": f"poll {sid}", "log": f"log {sid}", - "wait": f"wait {sid}", "kill": f"kill {sid}", "write": f"write {sid}", "submit": f"submit {sid}"} - return _wrap(f"┊ ⚙️ proc {labels.get(action, f'{action} {sid}')} {dur}") - if tool_name == "read_file": - return _wrap(f"┊ 📖 read {_trunc(build_tool_preview(tool_name, args) or args.get('path', ''), 42)} {dur}") - if tool_name == "write_file": - return _wrap(f"┊ ✍️ write {_path(args.get('path', ''))} {dur}") - if tool_name == "patch": - return _wrap(f"┊ 🔧 patch {_path(args.get('path', ''))} {dur}") - if tool_name == "search_files": - pattern = _trunc(args.get("pattern", ""), 35) - target = args.get("target", "content") - verb = "find" if target == "files" else "grep" - return _wrap(f"┊ 🔎 {verb:9} {pattern} {dur}") - if tool_name == "browser_navigate": - url = args.get("url", "") - domain = url.replace("https://", "").replace("http://", "").split("/")[0] - return _wrap(f"┊ 🌐 navigate {_trunc(domain, 35)} {dur}") - if tool_name == "browser_snapshot": - mode = "full" if args.get("full") else "compact" - return _wrap(f"┊ 📸 snapshot {mode} {dur}") - if tool_name == "browser_click": - return _wrap(f"┊ 👆 click {args.get('ref', '?')} {dur}") - if tool_name == "browser_type": - return _wrap(f"┊ ⌨️ type \"{_trunc(args.get('text', ''), 30)}\" {dur}") - if tool_name == "browser_scroll": - d = args.get("direction", "down") - arrow = {"down": "↓", "up": "↑", "right": "→", "left": "←"}.get(d, "↓") - return _wrap(f"┊ {arrow} scroll {d} {dur}") - if tool_name == "browser_back": - return _wrap(f"┊ ◀️ back {dur}") - if tool_name == "browser_press": - return _wrap(f"┊ ⌨️ press {args.get('key', '?')} {dur}") - if tool_name == "browser_get_images": - return _wrap(f"┊ 🖼️ images extracting {dur}") - if tool_name == "browser_vision": - return _wrap(f"┊ 👁️ vision analyzing page {dur}") - if tool_name == "todo_list": - todos_arg = args.get("todos") - merge = args.get("merge", False) - # Parse result for completion progress - total = 0 - done = 0 - if result: - try: - data = safe_json_loads(result) - if data: - s = data.get("summary", {}) - total = s.get("total", 0) - done = s.get("completed", 0) - except Exception: - pass - if todos_arg is None: - if total > 0: - return _wrap(f"┊ 📋 plan {done}/{total} task(s) {dur}") - return _wrap(f"┊ 📋 plan reading tasks {dur}") - elif merge: - if total > 0 and done > 0: - return _wrap(f"┊ 📋 plan update {done}/{total} ✓ {dur}") - return _wrap(f"┊ 📋 plan update {len(todos_arg)} task(s) {dur}") - else: - if total > 0 and done > 0: - return _wrap(f"┊ 📋 plan {done}/{total} task(s) {dur}") - return _wrap(f"┊ 📋 plan {len(todos_arg)} task(s) {dur}") - if tool_name == "session_search": - return _wrap(f"┊ 🔍 recall \"{_trunc(args.get('query', ''), 35)}\" {dur}") - if tool_name == "memory": - action = args.get("action", "?") - target = args.get("target", "") - if action == "add": - return _wrap(f"┊ 🧠 memory +{target}: \"{_trunc(args.get('content', ''), 30)}\" {dur}") - elif action == "replace": - old = args.get("old_text") or "" - old = old if old else "" - return _wrap(f"┊ 🧠 memory ~{target}: \"{_trunc(old, 20)}\" {dur}") - elif action == "remove": - old = args.get("old_text") or "" - old = old if old else "" - return _wrap(f"┊ 🧠 memory -{target}: \"{_trunc(old, 20)}\" {dur}") - return _wrap(f"┊ 🧠 memory {action} {dur}") - if tool_name == "skills_list": - return _wrap(f"┊ 📚 skills list {args.get('category', 'all')} {dur}") - if tool_name == "skill_view": - label = args.get("name", "") - file_path = args.get("file_path") - if file_path: - label = f"{label} → {file_path}" if label else str(file_path) - return _wrap(f"┊ 📚 skill {_trunc(label, 44)} {dur}") - if tool_name == "image_generate": - return _wrap(f"┊ 🎨 create {_trunc(args.get('prompt', ''), 35)} {dur}") - if tool_name == "text_to_speech": - return _wrap(f"┊ 🔊 speak {_trunc(args.get('text', ''), 30)} {dur}") - if tool_name == "vision_analyze": - return _wrap(f"┊ 👁️ vision {_trunc(args.get('question', ''), 30)} {dur}") - if tool_name == "send_message": - return _wrap(f"┊ 📨 send {args.get('target', '?')}: \"{_trunc(args.get('message', ''), 25)}\" {dur}") - if tool_name == "cronjob_manage": - action = args.get("action", "?") - if action == "create": - skills = args.get("skills") or ([] if not args.get("skill") else [args.get("skill")]) - label = args.get("name") or (skills[0] if skills else None) or args.get("prompt", "task") - return _wrap(f"┊ ⏰ cron create {_trunc(label, 24)} {dur}") - if action == "list": - return _wrap(f"┊ ⏰ cron listing {dur}") - return _wrap(f"┊ ⏰ cron {action} {args.get('job_id', '')} {dur}") - if tool_name == "execute_code": - code = args.get("code", "") - first_line = code.strip().split("\n")[0] if code.strip() else "" - return _wrap(f"┊ 🐍 exec {_trunc(first_line, 35)} {dur}") - if tool_name == "browser_exec": - label = _browser_exec_step_label(args) - if label is not None: - # Leading `# …` comment (the tool description asks for one): - # surface it as the user-facing step label; the code itself stays - # collapsed behind display.tool_preview_length. - return _wrap(f"┊ 🌐 browser {label} {dur}") - code = " ".join(str(args.get("code", "") or "").split()) - return _wrap(f"┊ 🌐 browser {_trunc(code, 35)} {dur}") - if tool_name == "delegate_task": - _action = str(args.get("action") or "").strip().lower() - if _action in ("list", "steer", "stop"): - _sid = str(args.get("subagent_id") or "").strip() - return _wrap(f"┊ 🔀 delegate {_trunc(f'{_action} {_sid}'.strip(), 35)} {dur}") - tasks = args.get("tasks") - if tasks and isinstance(tasks, list): - task_count, goals = _delegate_task_goal_parts(tasks, per_goal_len=30) - detail = " | ".join(goals) if goals else "parallel" - count_label = task_count or len(tasks) - return _wrap(f"┊ 🔀 delegate {count_label}x: {_trunc(detail, 35)} {dur}") - return _wrap(f"┊ 🔀 delegate {_trunc(args.get('goal', ''), 35)} {dur}") - - preview = build_tool_preview(tool_name, args) or "" - return _wrap(f"┊ ⚡ {tool_name[:9]:9} {_trunc(preview, 35)} {dur}") + render = _CUTE_LINES.get(tool_name) + if render is not None: + body = render(args, result) + else: + body = f"┊ ⚡ {tool_name[:9]:9} {_cute_trunc(build_tool_preview(tool_name, args) or '')}" + line = f"{body} {duration:.1f}s" + if skin_prefix != "┊": + line = line.replace("┊", skin_prefix, 1) + return line if not is_failure else f"{line}{failure_suffix}" def get_cute_tool_message( @@ -1582,8 +1321,3 @@ def get_cute_tool_message( safe_name = tool_name[:9] if isinstance(tool_name, str) and tool_name else "tool" safe_duration = f"{duration:.1f}s" if isinstance(duration, (int, float)) else "done" return f"┊ ⚡ {safe_name:9} completed {safe_duration}" - - -# ========================================================================= -# Honcho session line (one-liner with clickable OSC 8 hyperlink) -# ========================================================================= diff --git a/agent/i18n.py b/agent/i18n.py index 7c0dcf5c87..8402df3064 100644 --- a/agent/i18n.py +++ b/agent/i18n.py @@ -1,32 +1,8 @@ -"""Lightweight internationalization (i18n) for Hermes static user-facing messages. +"""Lightweight i18n for Hermes' static user-facing strings (approval prompts, a few gateway replies). -Scope (thin slice, by design): only the highest-impact static strings shown -to the user by Hermes itself -- approval prompts, a handful of gateway slash -command replies, restart-drain notices. Agent-generated output, log lines, -error tracebacks, tool outputs, and slash-command descriptions all stay in -English. - -Catalog files live under ``locales/.yaml`` at the repo root. Each -catalog is a flat dict keyed by dotted paths (e.g. ``approval.choose`` or -``gateway.approval_expired``). Missing keys fall back to English; if English -is missing too, the key path itself is returned so a broken catalog never -crashes the agent. - -Usage:: - - from agent.i18n import t - print(t("approval.choose_long")) # current lang - print(t("gateway.draining", count=3)) # {count} formatted - print(t("approval.choose_long", lang="zh")) # explicit override - -Language resolution order: - 1. Explicit ``lang=`` argument passed to :func:`t` - 2. ``HERMES_LANGUAGE`` environment variable (for tests / quick override) - 3. ``display.language`` from config.yaml - 4. ``"en"`` (baseline) - -Supported languages: en, zh, zh-hant, ja, de, es, fr, tr, uk, af, ko, it, ga, -pt, ru, hu, ar. Unknown values fall back to en. +Catalogs are ``locales/.yaml`` flattened to dotted keys. Missing keys +fall back to English, then to the key itself, so a broken catalog never crashes. +Language resolution: explicit ``lang=`` > ``HERMES_LANGUAGE`` > ``display.language`` > ``en``. """ from __future__ import annotations @@ -46,15 +22,13 @@ SUPPORTED_LANGUAGES: tuple[str, ...] = ( ) DEFAULT_LANGUAGE = "en" -# Accept a few natural aliases so users who type "chinese" / "zh-CN" / "jp" -# get the right catalog instead of silently falling back to English. +# Natural aliases so "chinese" / "zh-CN" / "jp" hit the right catalog instead of +# silently falling back to English. Bare "chinese" defaults to Simplified; +# Taiwan/HK/Macau tags route to the distinct Traditional catalog. pt-br shares +# the pt catalog (no separate br one). _LANGUAGE_ALIASES: dict[str, str] = { "english": "en", "en-us": "en", "en-gb": "en", - # Simplified Chinese — explicit codes route here; bare "chinese" / "mandarin" - # also default to Simplified since that's the larger user base. "chinese": "zh", "mandarin": "zh", "zh-cn": "zh", "zh-hans": "zh", "zh-sg": "zh", - # Traditional Chinese — distinct catalog. Cover Taiwan / Hong Kong / Macau - # locale tags plus the common "traditional" alias. "traditional-chinese": "zh-hant", "traditional_chinese": "zh-hant", "zh-tw": "zh-hant", "zh-hk": "zh-hant", "zh-mo": "zh-hant", "japanese": "ja", "jp": "ja", "ja-jp": "ja", @@ -63,23 +37,14 @@ _LANGUAGE_ALIASES: dict[str, str] = { "french": "fr", "français": "fr", "france": "fr", "fr-fr": "fr", "fr-be": "fr", "fr-ca": "fr", "fr-ch": "fr", "ukrainian": "uk", "ukrainisch": "uk", "українська": "uk", "uk-ua": "uk", "ua": "uk", "turkish": "tr", "türkçe": "tr", "tr-tr": "tr", - # Afrikaans — South African Dutch-derived language; "af-ZA" is the common BCP-47 tag. "afrikaans": "af", "af-za": "af", - # Korean "korean": "ko", "한국어": "ko", "ko-kr": "ko", - # Italian "italian": "it", "italiano": "it", "it-it": "it", "it-ch": "it", - # Irish (Gaeilge) — ga is the BCP-47 code "irish": "ga", "gaeilge": "ga", "ga-ie": "ga", - # Portuguese — bare "portuguese" routes to European Portuguese; pt-br - # is in the same family but rendered identically here (no separate br catalog). "portuguese": "pt", "português": "pt", "portugues": "pt", "pt-pt": "pt", "pt-br": "pt", "brazilian": "pt", "brasileiro": "pt", - # Russian "russian": "ru", "русский": "ru", "ru-ru": "ru", - # Hungarian "hungarian": "hu", "magyar": "hu", "hu-hu": "hu", - # Arabic — bare "arabic"/endonym plus the common regional BCP-47 tags. "arabic": "ar", "العربية": "ar", "ar-sa": "ar", "ar-eg": "ar", "ar-ae": "ar", "ar-ma": "ar", "ar-dz": "ar", } @@ -89,18 +54,10 @@ _catalog_lock = threading.Lock() def _locales_dir() -> Path: - """Return the directory containing locale YAML files. + """Locale dir: ``HERMES_BUNDLED_LOCALES`` (sealed packaging, e.g. Nix) if it exists, else ``/locales``. - Resolution order, first existing wins: - - 1. ``HERMES_BUNDLED_LOCALES`` env var -- set by the Nix wrapper (or any - sealed-packaging system) to point at the installed catalog directory. - 2. ``/locales`` -- source checkouts and editable installs, - where the working tree sits next to ``agent/``. - - Falling through to the source-style path (even when missing) keeps - ``_load_catalog`` error messages informative -- it logs the path it - looked at -- rather than raising. + The source path is returned even when missing so ``_load_catalog`` can log + the path it looked at rather than raise. """ override = os.getenv("HERMES_BUNDLED_LOCALES", "").strip() if override: @@ -112,19 +69,11 @@ def _locales_dir() -> Path: "falling back to bundled/source locale resolution", override, ) - - # agent/i18n.py -> agent/ -> repo root (source checkout, editable install) - source_dir = Path(__file__).resolve().parent.parent / "locales" - return source_dir + return Path(__file__).resolve().parent.parent / "locales" def _normalize_lang(value: Any) -> str: - """Normalize a user-supplied language value to a supported code. - - Accepts supported codes directly, common aliases (``chinese`` -> ``zh``), - and case-insensitive regional tags (``zh-CN`` -> ``zh``). Returns the - default language for unknown values. - """ + """Map a user-supplied value (code, alias, or regional tag like ``zh-CN``) to a supported code, else default.""" if not isinstance(value, str): return DEFAULT_LANGUAGE key = value.strip().lower() @@ -134,20 +83,20 @@ def _normalize_lang(value: Any) -> str: return key if key in _LANGUAGE_ALIASES: return _LANGUAGE_ALIASES[key] - # Try stripping a region suffix (e.g. "pt-br" -> "pt" won't be supported, - # but "zh-CN" -> "zh" will). - base = key.split("-", 1)[0] + base = key.split("-", 1)[0] # strip region suffix if base in SUPPORTED_LANGUAGES: return base return DEFAULT_LANGUAGE -def _load_catalog(lang: str) -> dict[str, str]: - """Load and flatten one locale YAML file into a dotted-key dict. +def _cache_catalog(lang: str, flat: dict[str, str]) -> dict[str, str]: + with _catalog_lock: + _catalog_cache[lang] = flat + return flat - YAML files can be nested for human readability; this produces the flat - key space :func:`t` expects. Cached per-language for the process. - """ + +def _load_catalog(lang: str) -> dict[str, str]: + """Load one locale YAML flattened to dotted keys; cached per language (empty dict on any failure).""" with _catalog_lock: cached = _catalog_cache.get(lang) if cached is not None: @@ -156,46 +105,34 @@ def _load_catalog(lang: str) -> dict[str, str]: path = _locales_dir() / f"{lang}.yaml" if not path.is_file(): logger.debug("i18n catalog missing for %s at %s", lang, path) - with _catalog_lock: - _catalog_cache[lang] = {} - return {} + return _cache_catalog(lang, {}) try: - import yaml # PyYAML is already a hermes dependency + import yaml with path.open("r", encoding="utf-8") as f: raw = yaml.safe_load(f) or {} except Exception as exc: logger.warning("Failed to load i18n catalog %s: %s", path, exc) - with _catalog_lock: - _catalog_cache[lang] = {} - return {} + return _cache_catalog(lang, {}) flat: dict[str, str] = {} _flatten_into(raw, "", flat) - with _catalog_lock: - _catalog_cache[lang] = flat - return flat + return _cache_catalog(lang, flat) def _flatten_into(node: Any, prefix: str, out: dict[str, str]) -> None: + # Non-string, non-dict leaves are ignored -- catalogs are text-only. if isinstance(node, dict): for key, value in node.items(): child_key = f"{prefix}.{key}" if prefix else str(key) _flatten_into(value, child_key, out) elif isinstance(node, str): out[prefix] = node - # Non-string, non-dict leaves are ignored -- catalogs are text-only. @lru_cache(maxsize=1) def _config_language_cached() -> str | None: - """Read ``display.language`` from config.yaml once per process. - - Cached because ``t()`` is called in hot paths (every approval prompt, - every gateway reply) and re-reading YAML each call would be wasteful. - ``reset_language_cache()`` clears this when config changes at runtime - (e.g. after the setup wizard). - """ + """``display.language`` from config.yaml, read once per process (``t()`` is a hot path).""" try: from hermes_cli.config import load_config_readonly cfg = load_config_readonly() @@ -208,11 +145,7 @@ def _config_language_cached() -> str | None: def reset_language_cache() -> None: - """Invalidate cached language resolution and catalogs. - - Call after :func:`hermes_cli.config.save_config` if a running process - needs to pick up a changed ``display.language`` without restart. - """ + """Invalidate cached language resolution and catalogs (call after ``save_config`` changes ``display.language``).""" _config_language_cached.cache_clear() with _catalog_lock: _catalog_cache.clear() @@ -223,41 +156,22 @@ def get_language() -> str: env_lang = os.environ.get("HERMES_LANGUAGE") if env_lang: return _normalize_lang(env_lang) - cfg_lang = _config_language_cached() - if cfg_lang: - return cfg_lang - return DEFAULT_LANGUAGE + return _config_language_cached() or DEFAULT_LANGUAGE def t(key: str, lang: str | None = None, **format_kwargs: Any) -> str: - """Translate a dotted key to the active language. + """Translate a dotted catalog key to the active (or explicit ``lang``) language. - Parameters - ---------- - key - Dotted path into the catalog, e.g. ``"approval.choose_long"``. - lang - Explicit language override. Takes precedence over env + config. - **format_kwargs - ``str.format`` substitution arguments (``t("gateway.drain", count=3)`` - expects a catalog entry with a ``{count}`` placeholder). - - Returns - ------- - The translated string, or the English fallback if the key is missing in - the target language, or the bare key if English is also missing. + ``format_kwargs`` are applied with ``str.format``. Falls back to English, + then to the bare key; a format failure returns the unformatted string. """ target = _normalize_lang(lang) if lang else get_language() - catalog = _load_catalog(target) - value = catalog.get(key) + value = _load_catalog(target).get(key) if value is None and target != DEFAULT_LANGUAGE: - # Fall through to English rather than showing a key path to the user. value = _load_catalog(DEFAULT_LANGUAGE).get(key) if value is None: - # Last-ditch: return the key itself. A broken catalog should not - # crash anything; it just looks ugly until someone fixes it. logger.debug("i18n miss: key=%r lang=%r", key, target) value = key diff --git a/agent/markdown_tables.py b/agent/markdown_tables.py index f37569cede..dbd0a5b467 100644 --- a/agent/markdown_tables.py +++ b/agent/markdown_tables.py @@ -1,30 +1,15 @@ """CJK/wide-character-aware re-alignment of model-emitted markdown tables. -Models pad markdown tables assuming each character occupies one terminal -cell. CJK glyphs and most emoji render as two cells, so the model's -spacing collapses into drift the moment a table reaches a real terminal — -header pipes line up, every body row drifts right by N cells per CJK -char. +Models pad tables assuming one cell per character; CJK glyphs and most emoji +take two, so body rows drift right on real terminals. This rebuilds padding +with ``wcwidth.wcswidth`` while preserving pipes/dashes so the table still reads +as plain text in ``strip``/unrendered modes (Rich already aligns CJK itself). -This module rebuilds row padding using ``wcwidth.wcswidth`` (display -columns), preserving the table's pipes and dashes so it still reads as a -plain-text table in ``strip`` / unrendered display modes. Standard Rich -markdown rendering already aligns CJK correctly inside a wide enough -panel; this helper is for the paths that print the model's text more or -less verbatim. - -The helper is deliberately conservative: - -* Only contiguous ``| ... |`` blocks with a divider line are rewritten. -* Anything that does not look like a table is passed through unchanged. -* Single-line / mid-stream fragments are left alone — callers buffer - table rows and flush them once the block is complete. - -There is a small, intentional caveat: ``wcwidth`` returns ``-1`` for some -emoji-with-variation-selector sequences (e.g. ``⚠️``); we clamp those to -0 so they do not corrupt the column width math. The 1-cell drift on -those specific glyphs is preferable to silently widening every table -that contains one. +Deliberately conservative: only contiguous ``| ... |`` blocks with a divider are +rewritten; everything else passes through; single-line/mid-stream fragments are +left alone (callers buffer rows and flush complete blocks). ``wcwidth`` returns +``-1`` for some emoji+variation-selector sequences (``⚠️``); those clamp to 0 — +a 1-cell drift on that glyph beats widening every table that contains one. """ from __future__ import annotations @@ -47,13 +32,7 @@ _MIN_COL_WIDTH = 3 # matches the divider's minimum dash run. def _disp_width(s: str) -> int: - """``wcswidth`` clamped to a non-negative integer. - - ``wcswidth`` returns ``-1`` when it encounters a control char or an - unknown sequence; treat those as zero-width rather than letting a - negative number flow into ``max`` and break the column-width math. - """ - + """``wcswidth`` clamped to >= 0 (it returns -1 for control/unknown sequences).""" w = wcswidth(s) return w if w > 0 else 0 @@ -64,7 +43,6 @@ def _pad_to_width(s: str, target: int) -> str: def split_table_row(row: str) -> List[str]: """Split ``| a | b | c |`` into ``["a", "b", "c"]`` with trims.""" - s = row.strip() if s.startswith("|"): s = s[1:] @@ -75,7 +53,6 @@ def split_table_row(row: str) -> List[str]: def is_table_divider(row: str) -> bool: """True when ``row`` is a markdown table separator line.""" - cells = split_table_row(row) return len(cells) > 1 and all(_DIVIDER_CELL_RE.match(c) for c in cells) @@ -83,76 +60,70 @@ def is_table_divider(row: str) -> bool: def looks_like_table_row(row: str) -> bool: """True when ``row`` could plausibly be a markdown table row. - Used by streaming callers to decide whether to buffer an in-flight - line. We are intentionally permissive here — the realigner itself - only rewrites blocks that are accompanied by a divider, so a false - positive here at most delays the print of one line. + Intentionally permissive for streaming callers deciding whether to buffer a + line: the realigner only rewrites divider-backed blocks, so a false positive + at most delays printing one line. A leading pipe is the strongest signal; + without it we accept >= 2 pipes so models that omit the leading pipe still match. """ - if "|" not in row: return False stripped = row.strip() if not stripped: return False - # A leading pipe is the strongest signal; without it we still allow - # rows with at least two pipes so models that omit the leading pipe - # don't slip past us. - if stripped.startswith("|"): - return True - return stripped.count("|") >= 2 + return stripped.startswith("|") or stripped.count("|") >= 2 def _render_block(rows: List[List[str]], available_width: int | None = None) -> List[str]: """Render ``rows`` (header + body, divider implied) at uniform widths. - If ``available_width`` is given and the rebuilt horizontal table - would exceed it, fall back to a vertical key-value rendering so - rows do not soft-wrap mid-cell — terminal soft-wrap destroys - column alignment visually even when the underlying bytes are - perfectly padded, which is exactly the "tables look broken" - user report this code path is meant to address. + When the horizontal table would exceed ``available_width`` fall back to a + vertical key-value rendering: terminal soft-wrap mid-cell destroys alignment + visually even when the bytes are perfectly padded. """ - ncols = max(len(r) for r in rows) rows = [r + [""] * (ncols - len(r)) for r in rows] + widths = [max(_MIN_COL_WIDTH, *(_disp_width(r[c]) for r in rows)) for c in range(ncols)] - widths = [ - max(_MIN_COL_WIDTH, *(_disp_width(r[c]) for r in rows)) - for c in range(ncols) - ] - - # Total horizontal width for the rendered row: - # `| ` + cell + ` ` for each column, plus the final closing `|`. + # `| ` + cell + ` ` per column, plus the closing `|`. horizontal_width = sum(widths) + 3 * ncols + 1 - if available_width is not None and horizontal_width > max(available_width, 20): return _render_vertical(rows, ncols, available_width) def _row(cells: List[str]) -> str: - return ( - "| " - + " | ".join(_pad_to_width(c, widths[k]) for k, c in enumerate(cells)) - + " |" - ) + return "| " + " | ".join(_pad_to_width(c, widths[k]) for k, c in enumerate(cells)) + " |" - out = [_row(rows[0])] - out.append("|" + "|".join("-" * (w + 2) for w in widths) + "|") - for r in rows[1:]: - out.append(_row(r)) + out = [_row(rows[0]), "|" + "|".join("-" * (w + 2) for w in widths) + "|"] + out.extend(_row(r) for r in rows[1:]) + return out + + +def _hard_break(word: str, w: int) -> List[str]: + """Split a single over-wide word into display-width-``w`` chunks.""" + out: List[str] = [] + buf = "" + bw = 0 + for ch in word: + cw = _disp_width(ch) or 1 + if bw + cw > w and buf: + out.append(buf) + buf = ch + bw = cw + else: + buf += ch + bw += cw + if buf: + out.append(buf) return out def _wrap_to_width(text: str, width: int) -> List[str]: - """Soft-wrap ``text`` at word boundaries to fit ``width`` display cells. + """Soft-wrap ``text`` at word boundaries to ``width`` display cells. - Falls back to hard-breaking the longest word if a single token is - wider than ``width``. Empty input yields a single empty string so - the caller's row count stays predictable. + Words wider than ``width`` are hard-broken. Empty input yields a single + empty string so the caller's row count stays predictable. """ - if width <= 0 or not text: return [text] - words = text.split() if not words: return [""] @@ -161,100 +132,61 @@ def _wrap_to_width(text: str, width: int) -> List[str]: current = "" current_w = 0 - def _hard_break(word: str, w: int) -> List[str]: - out: List[str] = [] - buf = "" - bw = 0 - for ch in word: - cw = _disp_width(ch) or 1 - if bw + cw > w and buf: - out.append(buf) - buf = ch - bw = cw - else: - buf += ch - bw += cw - if buf: - out.append(buf) - return out + def _start(word: str, ww: int) -> None: + nonlocal current, current_w + if ww <= width: + current, current_w = word, ww + else: + pieces = _hard_break(word, width) + lines.extend(pieces[:-1]) + current = pieces[-1] if pieces else "" + current_w = _disp_width(current) for word in words: ww = _disp_width(word) if not current: - if ww <= width: - current = word - current_w = ww - else: - pieces = _hard_break(word, width) - lines.extend(pieces[:-1]) - current = pieces[-1] if pieces else "" - current_w = _disp_width(current) - continue - if current_w + 1 + ww <= width: + _start(word, ww) + elif current_w + 1 + ww <= width: current += " " + word current_w += 1 + ww else: lines.append(current) - if ww <= width: - current = word - current_w = ww - else: - pieces = _hard_break(word, width) - lines.extend(pieces[:-1]) - current = pieces[-1] if pieces else "" - current_w = _disp_width(current) + _start(word, ww) if current: lines.append(current) return lines or [""] -def _render_vertical( - rows: List[List[str]], ncols: int, available_width: int -) -> List[str]: - """Render a too-wide table as vertical ``Header: value`` rows. +def _render_vertical(rows: List[List[str]], ncols: int, available_width: int) -> List[str]: + """Render a too-wide table as ``Header: value`` blocks (Claude Code's narrow fallback). - Mirrors Claude Code's narrow-terminal fallback in - ``MarkdownTable.tsx``: each body row becomes a small block of - ``Header: cell-value`` lines (continuation lines indented two - spaces) separated by a thin ``─`` divider between rows. Keeps - every line narrower than ``available_width`` so the terminal does - not soft-wrap mid-cell. + Each body row becomes one block with continuation lines indented two spaces, + blocks separated by a thin ``─`` rule; every line stays under ``available_width``. """ - if not rows: return [] - headers = rows[0] + [""] * (ncols - len(rows[0])) - body = rows[1:] - labels = [h or f"Column {i + 1}" for i, h in enumerate(headers)] - sep_width = max(20, min(40, available_width - 2)) if available_width else 30 separator = "─" * sep_width indent = " " - indent_w = _disp_width(indent) + cont_budget = max(10, available_width - _disp_width(indent)) out: List[str] = [] - for ri, row in enumerate(body): + for ri, row in enumerate(rows[1:]): if ri > 0: out.append(separator) for ci in range(ncols): label = labels[ci] value = row[ci] if ci < len(row) else "" - label_w = _disp_width(label) - first_budget = max(10, available_width - label_w - 2) - cont_budget = max(10, available_width - indent_w) if not value: out.append(f"{label}:") continue - wrapped = _wrap_to_width(value, first_budget) + wrapped = _wrap_to_width(value, max(10, available_width - _disp_width(label) - 2)) out.append(f"{label}: {wrapped[0]}") if len(wrapped) > 1: - # Re-flow continuation text at the wider continuation - # budget — words split across the narrower first-line - # budget should re-pack greedily for the rest. - cont_text = " ".join(wrapped[1:]) - for cl in _wrap_to_width(cont_text, cont_budget): + # Re-flow continuation text at the wider continuation budget. + for cl in _wrap_to_width(" ".join(wrapped[1:]), cont_budget): if cl.strip(): out.append(f"{indent}{cl}") return out @@ -263,16 +195,10 @@ def _render_vertical( def realign_markdown_tables(text: str, available_width: int | None = None) -> str: """Rewrite every ``| ... |`` + divider block with wcwidth-aware padding. - Lines that are not part of a recognised table are returned verbatim, - so this is safe to apply to arbitrary assistant prose. - - If ``available_width`` is given (terminal cells available for the - rendered table), tables wider than that are rendered as vertical - key-value pairs instead of a horizontal pipe-bordered grid. This - avoids the terminal soft-wrapping mid-cell, which destroys column - alignment visually even when the bytes are perfectly padded. + Non-table lines are returned verbatim, so this is safe on arbitrary prose. + With ``available_width`` (terminal cells), tables wider than that render as + vertical key-value pairs instead of soft-wrapping mid-cell. """ - if "|" not in text: return text @@ -280,30 +206,21 @@ def realign_markdown_tables(text: str, available_width: int | None = None) -> st out: List[str] = [] i = 0 n = len(lines) - while i < n: line = lines[i] # A table starts with a header row whose next line is a divider. - if ( - "|" in line - and i + 1 < n - and is_table_divider(lines[i + 1]) - ): + if "|" in line and i + 1 < n and is_table_divider(lines[i + 1]): header = split_table_row(line) body: List[List[str]] = [] j = i + 2 while j < n and "|" in lines[j] and lines[j].strip(): - if is_table_divider(lines[j]): - j += 1 - continue - body.append(split_table_row(lines[j])) + if not is_table_divider(lines[j]): + body.append(split_table_row(lines[j])) j += 1 - if any(c for c in header) or body: out.extend(_render_block([header] + body, available_width)) i = j continue out.append(line) i += 1 - return "\n".join(out) diff --git a/agent/message_sanitization.py b/agent/message_sanitization.py index d7b374a500..d4deab39c0 100644 --- a/agent/message_sanitization.py +++ b/agent/message_sanitization.py @@ -1,15 +1,9 @@ """Message and tool-payload sanitization helpers. -Pure functions extracted from ``run_agent.py`` so the AIAgent module can -stay focused on the conversation loop. These walk OpenAI-format message -lists and structured payloads, repairing or stripping problematic -characters that would otherwise crash ``json.dumps`` inside the OpenAI -SDK or be rejected by upstream APIs. - -All helpers are stateless and side-effect-free except for in-place -mutation of their input (where documented). Backward-compatible -re-exports from ``run_agent`` remain in place so existing imports -``from run_agent import _sanitize_surrogates`` keep working. +Pure functions (extracted from ``run_agent.py``) that walk OpenAI-format message +lists and structured payloads, repairing or stripping characters that would +crash ``json.dumps`` in the OpenAI SDK or be rejected upstream. Stateless except +for documented in-place mutation; ``run_agent`` re-exports them for old imports. """ from __future__ import annotations @@ -18,142 +12,137 @@ import hashlib import json import logging import re -from typing import Any +from typing import Any, Callable logger = logging.getLogger(__name__) -# Lone surrogate code points are invalid in UTF-8 and crash json.dumps -# inside the OpenAI SDK. Used by every surrogate-sanitization helper -# below as well as by run_agent and the CLI for paste-from-clipboard -# scrubbing. +# Lone surrogate code points are invalid in UTF-8 and crash json.dumps inside +# the OpenAI SDK. Also used by run_agent and the CLI for paste scrubbing. _SURROGATE_RE = re.compile(r'[\ud800-\udfff]') +# Message keys handled explicitly by _sanitize_messages; every OTHER key is +# swept generically (reasoning, reasoning_content, reasoning_details, ...). +_MESSAGE_CORE_KEYS = frozenset({"content", "name", "tool_calls", "role"}) + def _sanitize_surrogates(text: str) -> str: - """Replace lone surrogate code points with U+FFFD (replacement character). - - Surrogates are invalid in UTF-8 and will crash ``json.dumps()`` inside the - OpenAI SDK. This is a fast no-op when the text contains no surrogates. - """ + """Replace lone surrogate code points with U+FFFD; no-op when none present.""" if _SURROGATE_RE.search(text): return _SURROGATE_RE.sub('\ufffd', text) return text -def _sanitize_structure_surrogates(payload: Any) -> bool: - """Replace surrogate code points in nested dict/list payloads in-place. +def _strip_non_ascii(text: str) -> str: + """Drop non-ASCII characters — last resort for ASCII-only system encodings (LANG=C).""" + return text.encode('ascii', errors='ignore').decode('ascii') - Mirror of ``_sanitize_structure_non_ascii`` but for surrogate recovery. - Used to scrub nested structured fields (e.g. ``reasoning_details`` — an - array of dicts with ``summary``/``text`` strings) that flat per-field - checks don't reach. Returns True if any surrogates were replaced. - """ + +def _fix_str_field(container: Any, key: Any, fix: Callable[[str], str]) -> bool: + """Apply ``fix`` to ``container[key]`` if it is a str; True if it changed.""" + value = container.get(key) if isinstance(container, dict) else container[key] + if isinstance(value, str): + fixed = fix(value) + if fixed != value: + container[key] = fixed + return True + return False + + +def _sanitize_structure(payload: Any, fix: Callable[[str], str]) -> bool: + """Apply ``fix`` to every str inside nested dict/list ``payload`` in-place.""" found = False def _walk(node): nonlocal found if isinstance(node, dict): - for key, value in node.items(): - if isinstance(value, str): - if _SURROGATE_RE.search(value): - node[key] = _SURROGATE_RE.sub('\ufffd', value) - found = True - elif isinstance(value, (dict, list)): - _walk(value) + items = list(node.items()) elif isinstance(node, list): - for idx, value in enumerate(node): - if isinstance(value, str): - if _SURROGATE_RE.search(value): - node[idx] = _SURROGATE_RE.sub('\ufffd', value) - found = True - elif isinstance(value, (dict, list)): - _walk(value) + items = list(enumerate(node)) + else: + return + for key, value in items: + if isinstance(value, str): + found |= _fix_str_field(node, key, fix) + elif isinstance(value, (dict, list)): + _walk(value) _walk(payload) return found -def _sanitize_messages_surrogates(messages: list) -> bool: - """Sanitize surrogate characters from all string content in a messages list. +def _sanitize_messages(messages: list, fix: Callable[[str], str], *, deep: bool) -> bool: + """Apply ``fix`` to the string fields of every message dict in-place. - Walks message dicts in-place. Returns True if any surrogates were found - and replaced, False otherwise. Covers content/text, name, tool call - metadata/arguments, AND any additional string or nested structured fields - (``reasoning``, ``reasoning_content``, ``reasoning_details``, etc.) so - retries don't fail on a non-content field. Byte-level reasoning models - (xiaomi/mimo, kimi, glm) can emit lone surrogates in reasoning output - that flow through to ``api_messages["reasoning_content"]`` on the next - turn and crash json.dumps inside the OpenAI SDK. + Covers content / content-part text, name, tool_call function arguments, and + every non-core top-level str field (reasoning_content etc.) so retries don't + fail on a non-content field. ``deep=True`` additionally covers tool_call ids, + function names, and NESTED non-core fields (``reasoning_details`` arrays from + byte-level reasoning models such as xiaomi/mimo, kimi, glm). """ found = False for msg in messages: if not isinstance(msg, dict): continue content = msg.get("content") - if isinstance(content, str) and _SURROGATE_RE.search(content): - msg["content"] = _SURROGATE_RE.sub('\ufffd', content) - found = True + if isinstance(content, str): + found |= _fix_str_field(msg, "content", fix) elif isinstance(content, list): for part in content: if isinstance(part, dict): - text = part.get("text") - if isinstance(text, str) and _SURROGATE_RE.search(text): - part["text"] = _SURROGATE_RE.sub('\ufffd', text) - found = True - name = msg.get("name") - if isinstance(name, str) and _SURROGATE_RE.search(name): - msg["name"] = _SURROGATE_RE.sub('\ufffd', name) - found = True + found |= _fix_str_field(part, "text", fix) + found |= _fix_str_field(msg, "name", fix) tool_calls = msg.get("tool_calls") if isinstance(tool_calls, list): for tc in tool_calls: if not isinstance(tc, dict): continue - tc_id = tc.get("id") - if isinstance(tc_id, str) and _SURROGATE_RE.search(tc_id): - tc["id"] = _SURROGATE_RE.sub('\ufffd', tc_id) - found = True + if deep: + found |= _fix_str_field(tc, "id", fix) fn = tc.get("function") if isinstance(fn, dict): - fn_name = fn.get("name") - if isinstance(fn_name, str) and _SURROGATE_RE.search(fn_name): - fn["name"] = _SURROGATE_RE.sub('\ufffd', fn_name) - found = True - fn_args = fn.get("arguments") - if isinstance(fn_args, str) and _SURROGATE_RE.search(fn_args): - fn["arguments"] = _SURROGATE_RE.sub('\ufffd', fn_args) - found = True - # Walk any additional string / nested fields (reasoning, - # reasoning_content, reasoning_details, etc.) — surrogates from - # byte-level reasoning models (xiaomi/mimo, kimi, glm) can lurk - # in these fields and aren't covered by the per-field checks above. - # Matches _sanitize_messages_non_ascii's coverage (PR #10537). - for key, value in msg.items(): - if key in {"content", "name", "tool_calls", "role"}: + if deep: + found |= _fix_str_field(fn, "name", fix) + found |= _fix_str_field(fn, "arguments", fix) + for key, value in list(msg.items()): + if key in _MESSAGE_CORE_KEYS: continue if isinstance(value, str): - if _SURROGATE_RE.search(value): - msg[key] = _SURROGATE_RE.sub('\ufffd', value) - found = True - elif isinstance(value, (dict, list)): - if _sanitize_structure_surrogates(value): - found = True + found |= _fix_str_field(msg, key, fix) + elif deep and isinstance(value, (dict, list)): + found |= _sanitize_structure(value, fix) return found +def _sanitize_structure_surrogates(payload: Any) -> bool: + """Replace surrogates in nested dict/list payloads in-place; True if any replaced.""" + return _sanitize_structure(payload, _sanitize_surrogates) + + +def _sanitize_messages_surrogates(messages: list) -> bool: + """Replace surrogates in all string content of a messages list in-place; True if any found.""" + return _sanitize_messages(messages, _sanitize_surrogates, deep=True) + + +def _sanitize_structure_non_ascii(payload: Any) -> bool: + """Strip non-ASCII from nested dict/list payloads in-place; True if any stripped.""" + return _sanitize_structure(payload, _strip_non_ascii) + + +def _sanitize_messages_non_ascii(messages: list) -> bool: + """Strip non-ASCII from a messages list in-place (ASCII-only locales); True if any stripped.""" + return _sanitize_messages(messages, _strip_non_ascii, deep=False) + + +def _sanitize_tools_non_ascii(tools: list) -> bool: + """Strip non-ASCII characters from tool payloads in-place.""" + return _sanitize_structure_non_ascii(tools) + + def _escape_invalid_chars_in_json_strings(raw: str) -> str: - """Escape unescaped control chars inside JSON string values. + """Escape literal control chars (0x00-0x1F) inside JSON string values as ``\\uXXXX``. - Walks the raw JSON character-by-character, tracking whether we are - inside a double-quoted string. Inside strings, replaces literal - control characters (0x00-0x1F) that aren't already part of an escape - sequence with their ``\\uXXXX`` equivalents. Pass-through for everything - else. - - Ported from #12093 — complements the other repair passes in - ``_repair_tool_call_arguments`` when ``json.loads(strict=False)`` is - not enough (e.g. llama.cpp backends that emit literal apostrophes or - tabs alongside other malformations). + Complements ``json.loads(strict=False)`` in ``_repair_tool_call_arguments`` + for llama.cpp-style output that mixes control chars with other malformations. """ out: list[str] = [] in_string = False @@ -163,7 +152,6 @@ def _escape_invalid_chars_in_json_strings(raw: str) -> str: ch = raw[i] if in_string: if ch == "\\" and i + 1 < n: - # Already-escaped char — pass through as-is out.append(ch) out.append(raw[i + 1]) i += 2 @@ -183,41 +171,29 @@ def _escape_invalid_chars_in_json_strings(raw: str) -> str: return "".join(out) -# When a repair is about to destroy the only copy of a tool call's original -# argument bytes (rewriting them to "{}"), the WARNING log is the last -# surviving copy of content that can hold real user data (#80498). Bound the -# logged string at this size instead of a short preview so it stays -# recoverable from agent.log without letting a pathological payload flood -# the log. +# When a repair rewrites arguments to "{}", the WARNING log is the last surviving +# copy of content that can hold real user data (e.g. a truncated write_file's +# streamed file content). Bound it here rather than at a short preview. _FULL_ARGS_LOG_BOUND = 100_000 def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: - """Attempt to repair malformed tool_call argument JSON. - - Models like GLM-5.1 via Ollama can produce truncated JSON, trailing - commas, Python ``None``, etc. The API proxy rejects these with HTTP 400 - "invalid tool call arguments". This function applies common repairs; - if all fail it returns ``"{}"`` so the request succeeds (better than - crashing the session). All repairs are logged at WARNING level. + """Repair malformed tool_call argument JSON (truncation, trailing commas, + Python ``None``, literal control chars); returns ``"{}"`` if unrepairable so + the request succeeds instead of crashing the session. Repairs log at WARNING. """ raw_stripped = raw_args.strip() if isinstance(raw_args, str) else "" - # Fast-path: empty / whitespace-only -> empty object if not raw_stripped: logger.warning("Sanitized empty tool_call arguments for %s", tool_name) return "{}" - # Python-literal None -> normalise to {} if raw_stripped == "None": logger.warning("Sanitized Python-None tool_call arguments for %s", tool_name) return "{}" - # Repair pass 0: llama.cpp backends sometimes emit literal control - # characters (tabs, newlines) inside JSON string values. json.loads - # with strict=False accepts these and lets us re-serialise the - # result into wire-valid JSON without any string surgery. This is - # the most common local-model repair case (#12068). + # Pass 0: strict=False accepts literal control chars inside strings (the + # most common local-model case) and re-serialises to wire-valid JSON. try: parsed = json.loads(raw_stripped, strict=False) reserialised = json.dumps(parsed, separators=(",", ":")) @@ -230,18 +206,15 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: except (json.JSONDecodeError, TypeError, ValueError): pass - # Attempt common JSON repairs - fixed = raw_stripped - # 1. Strip trailing commas before } or ] - fixed = re.sub(r',\s*([}\]])', r'\1', fixed) - # 2. Close unclosed structures + # Passes 1-3: strip trailing commas, close unclosed structures, then trim + # excess closers (bounded). + fixed = re.sub(r',\s*([}\]])', r'\1', raw_stripped) open_curly = fixed.count('{') - fixed.count('}') open_bracket = fixed.count('[') - fixed.count(']') if open_curly > 0: fixed += '}' * open_curly if open_bracket > 0: fixed += ']' * open_bracket - # 3. Remove excess closing braces/brackets (bounded to 50 iterations) for _ in range(50): try: json.loads(fixed) @@ -264,9 +237,8 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: except json.JSONDecodeError: pass - # Repair pass 4: escape unescaped control chars inside JSON strings, - # then retry. Catches cases where strict=False alone fails because - # other malformations are present too. + # Pass 4: escape control chars inside strings (strict=False alone fails + # when other malformations are present too), then retry. try: escaped = _escape_invalid_chars_in_json_strings(fixed) if escaped != fixed: @@ -279,12 +251,6 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: except (json.JSONDecodeError, TypeError, ValueError): pass - # Last resort: replace with empty object so the API request doesn't - # crash the entire session. Log the FULL original string (bounded) — - # for callers that discard the original (e.g. the pre-send transcript - # sanitizer), this WARNING is the last surviving copy of bytes that can - # contain real user content (#80498: a truncated write_file call's - # streamed file content). logger.warning( "Unrepairable tool_call arguments for %s — " "replaced with empty object (was: %s)", @@ -296,21 +262,12 @@ def _repair_tool_call_arguments(raw_args: str, tool_name: str = "?") -> str: def close_interrupted_tool_sequence(messages: list, final_response: Any = None) -> bool: """Append a synthetic assistant turn when an interrupted tail is a tool result. - A turn cut short by ``/stop`` can leave the transcript ending on a raw - ``tool`` message (a tool finished, or its execution was cancelled, but the - model never streamed a closing assistant turn). Persisting that tail means - the next user message lands as ``… tool → user`` — a role-alternation - violation that strict providers (Gemini, Claude) react to by hallucinating - a continuation of the user's message and ignoring prior context, which - reads to the user as "lost context" (#48879). - - ``finalize_turn`` closes this on the happy interrupt path, but the - retry/backoff/error interrupt aborts in ``conversation_loop`` ``return`` - early and never reach it — this shared helper closes the sequence on all of - them. ``final_response`` is usually empty on an interrupt, so an explicit - placeholder is used rather than an empty-content assistant turn. - - Mutates ``messages`` in place. Returns True if a closing turn was appended. + A transcript ending on a raw ``tool`` message makes the next user message + land as ``tool → user`` — a role-alternation violation strict providers + (Gemini, Claude) answer by hallucinating a continuation and dropping prior + context. ``finalize_turn`` covers the happy interrupt path; the retry/backoff + early-returns in ``conversation_loop`` need this shared helper. Mutates in + place; returns True if a closing turn was appended. """ if not messages: return False @@ -327,99 +284,15 @@ def close_interrupted_tool_sequence(messages: list, final_response: Any = None) return True -def _strip_non_ascii(text: str) -> str: - """Remove non-ASCII characters, replacing with closest ASCII equivalent or removing. - - Used as a last resort when the system encoding is ASCII and can't handle - any non-ASCII characters (e.g. LANG=C on Chromebooks). - """ - return text.encode('ascii', errors='ignore').decode('ascii') - - -def _sanitize_messages_non_ascii(messages: list) -> bool: - """Strip non-ASCII characters from all string content in a messages list. - - This is a last-resort recovery for systems with ASCII-only encoding - (LANG=C, Chromebooks, minimal containers). Returns True if any - non-ASCII content was found and sanitized. - """ - found = False - for msg in messages: - if not isinstance(msg, dict): - continue - # Sanitize content (string) - content = msg.get("content") - if isinstance(content, str): - sanitized = _strip_non_ascii(content) - if sanitized != content: - msg["content"] = sanitized - found = True - elif isinstance(content, list): - for part in content: - if isinstance(part, dict): - text = part.get("text") - if isinstance(text, str): - sanitized = _strip_non_ascii(text) - if sanitized != text: - part["text"] = sanitized - found = True - # Sanitize name field (can contain non-ASCII in tool results) - name = msg.get("name") - if isinstance(name, str): - sanitized = _strip_non_ascii(name) - if sanitized != name: - msg["name"] = sanitized - found = True - # Sanitize tool_calls - tool_calls = msg.get("tool_calls") - if isinstance(tool_calls, list): - for tc in tool_calls: - if isinstance(tc, dict): - fn = tc.get("function", {}) - if isinstance(fn, dict): - fn_args = fn.get("arguments") - if isinstance(fn_args, str): - sanitized = _strip_non_ascii(fn_args) - if sanitized != fn_args: - fn["arguments"] = sanitized - found = True - # Sanitize any additional top-level string fields (e.g. reasoning_content) - for key, value in msg.items(): - if key in {"content", "name", "tool_calls", "role"}: - continue - if isinstance(value, str): - sanitized = _strip_non_ascii(value) - if sanitized != value: - msg[key] = sanitized - found = True - return found - - -def _sanitize_tools_non_ascii(tools: list) -> bool: - """Strip non-ASCII characters from tool payloads in-place.""" - return _sanitize_structure_non_ascii(tools) - - def serialized_messages_bytes(messages: list) -> int: - """Exact serialized size, in bytes, of the ``messages`` request payload. + """Exact serialized byte size of the ``messages`` payload (HTTP 413 recovery). - Recovery path for HTTP 413 (payload too large). A 413 is a *byte*-size - error, but Hermes' context estimator deliberately prices an image at a - flat per-image token cost so that a screenshot does not trigger premature - compaction (see ``estimate_messages_tokens_rough``). That makes the - token estimate structurally unable to *score* recovery from an - image-dominated 413: compaction can free megabytes of base64 while the - estimate barely moves, so a token-scored progress check reports - "no progress" and the turn dies permanently. - - This measures the thing the provider actually rejected — serialized - bytes — exactly and for free. It is a faithful proxy for the request - body's ``messages`` field (the only part recovery can shrink) and is - measured identically before and after each compression pass, so the - before/after ratio is exact. It is NOT an estimate. - - Non-serializable values fall back to ``str()`` so a malformed message - can never crash the 413 recovery path. + A 413 is a BYTE-size error, but the token estimator deliberately prices an + image at a flat per-image cost, so it cannot score recovery from an + image-dominated 413 (compaction frees megabytes while the estimate barely + moves → "no progress"). This measures what the provider actually rejected, + identically before and after each pass. Non-serializable values fall back to + ``str()`` so a malformed message can never crash recovery. """ if not isinstance(messages, list) or not messages: return 0 @@ -430,34 +303,19 @@ def serialized_messages_bytes(messages: list) -> int: ).encode("utf-8") ) except (TypeError, ValueError): - # Extremely defensive — ``default=str`` already covers exotic - # values. Never let byte accounting take down error recovery. return sum(len(str(m)) for m in messages) def _strip_images_from_messages(messages: list) -> bool: - """Remove image_url content parts from all messages in-place. + """Remove image content parts from all messages in-place (server rejected images). - Called when a server signals it does not support images (e.g. - "Only 'text' content type is supported."). Mutates messages so the - next API call sends text only. - - Preserves message alternation invariants: - * ``tool``-role messages whose content was entirely images are replaced - with a plaintext placeholder, NOT deleted — deleting them would leave - the paired ``tool_call_id`` on the prior assistant message unmatched, - which providers reject with HTTP 400. - * Assistant messages carrying ``tool_calls`` are likewise replaced, not - deleted — dropping them would orphan their tool responses. - * Other messages whose content becomes empty are dropped. In practice - this only hits synthetic image-only user messages appended for - attachment delivery; real user turns always include text. - - This runs on the persistent history as well as the per-call copy, so any - message it rewrites must also lose its ``api_content`` sidecar: the sidecar - carries the exact bytes previously sent — here, the images this strip - exists to remove — and the next turn substitutes it back into ``content``, - undoing the strip on the wire. + Preserves alternation invariants: ``tool`` messages and assistant messages + carrying ``tool_calls`` whose content was entirely images are replaced with a + placeholder, NOT deleted (deleting orphans the paired ``tool_call_id`` → + HTTP 400); other now-empty messages (synthetic image-only attachment turns) + are dropped. Any rewritten message also loses its ``api_content`` sidecar — + it carries the exact bytes previously sent, i.e. the images being removed, + and would be substituted back on the wire next turn. Returns True if any image parts were removed. """ @@ -481,22 +339,18 @@ def _strip_images_from_messages(messages: list) -> bool: if new_parts: msg["content"] = new_parts elif msg.get("role") == "tool" or msg.get("tool_calls"): - # Preserve message linkage — providers require every assistant - # tool_call to have a matching tool response, and an assistant - # message carrying tool_calls must survive even if its content - # was entirely images. msg["content"] = "[image content removed — server does not support images]" else: - # Synthetic image-only user/assistant message with no text and - # no tool_calls; safe to drop. to_delete.append(i) - # Content was rewritten — the pre-strip sidecar is now stale. drop_stale_api_content(msg) for i in reversed(to_delete): del messages[i] return found +# Provider error bodies (lowercased substring match) meaning "image/multimodal +# input unsupported" — the loop then strips images and retries text-only instead +# of cascading into compression / context-too-large recovery or wedging on retries. _IMAGE_REJECTION_PHRASES = ( "only 'text' content type is supported", "only text content type is supported", @@ -512,56 +366,22 @@ _IMAGE_REJECTION_PHRASES = ( "does not support multimodal", "does not support vision", "model does not support image", - # Some OpenAI-compatible endpoints (e.g. Alibaba/DashScope-style - # gateways) reject non-text content blocks with this generic body - # instead of naming image_url or vision support explicitly. - # (issue #57948) + # DashScope-style gateways reject non-text blocks with this generic body. "unexpected item type in content", - # ChatGPT-account Codex backend - # (https://chatgpt.com/backend-api/codex) rejects - # data:image/...base64 URLs in input_image fields - # with HTTP 400 "Invalid 'input[N].content[K].image_url'. - # Expected a valid URL, but got a value with an - # invalid format." The OpenAI Responses API on the - # public endpoint accepts data URLs, but the - # ChatGPT-account variant does not. Without this - # phrase the agent cascaded into compression / - # context-too-large recovery instead of just - # stripping the images. Match is narrow on - # purpose — keyed on the field-path apostrophe so - # we don't false-trip on other URL validation - # errors. (issue #23570) + # ChatGPT-account Codex backend rejects data:image URLs in input_image + # ("Invalid 'input[N].content[K].image_url'. Expected a valid URL ..."); + # keyed on the field-path apostrophe so other URL errors don't false-trip. "image_url'. expected", - # ChatGPT-account Codex can also reject corrupt/unsupported - # native image payloads with this wording. Treat it like a - # provider image rejection so the loop strips images and - # retries text-only instead of aborting the session. + # ChatGPT-account Codex wording for corrupt/unsupported native image payloads. "image data you provided does not represent a valid image", - # DeepSeek's OpenAI-compatible API reports text-only - # request-body variants as: - # "unknown variant `image_url`, expected `text`". + # DeepSeek's text-only request-body variant error. "unknown variant `image_url`, expected `text`", "unknown variant image_url, expected text", - # OpenRouter routes a request to upstream endpoints and, - # when none of the candidate endpoints for the model accept - # image input, returns HTTP 404 "No endpoints found that - # support image input". Without this phrase the agent never - # strips the images, the retry loop re-sends the same - # rejected request until exhaustion, and the gateway leaves - # every subsequent message queued behind the stuck turn — - # the P1 in issue #21160. The 404 passes the 4xx gate in the - # conversation loop. + # OpenRouter HTTP 404 when no upstream endpoint accepts image input (passes + # the 4xx gate; without this the gateway queue wedges behind the stuck turn). "no endpoints found that support image input", - # Kimi / Moonshot / other OpenAI-compatible Chinese - # providers reject truncated or corrupt image bytes with - # HTTP 400 "Invalid request: prepare image failed ... - # failed to decode image: invalid or unsupported image - # format". Like the Codex case above, the bad bytes are - # baked into immutable conversation history and re-sent on - # every retry, wedging the session. Strip the images so the - # turn recovers instead of exhausting retries. (issue - # #76884; complements the proactive full-decode validation - # in tools/vision_tools._normalize_to_supported_image) + # Kimi/Moonshot et al. reject truncated/corrupt image bytes baked into + # immutable history ("prepare image failed ... failed to decode image"). "failed to decode image", ) @@ -572,35 +392,6 @@ def _looks_like_image_content_rejection(error_body: str) -> bool: return any(phrase in body for phrase in _IMAGE_REJECTION_PHRASES) -def _sanitize_structure_non_ascii(payload: Any) -> bool: - """Strip non-ASCII characters from nested dict/list payloads in-place.""" - found = False - - def _walk(node): - nonlocal found - if isinstance(node, dict): - for key, value in node.items(): - if isinstance(value, str): - sanitized = _strip_non_ascii(value) - if sanitized != value: - node[key] = sanitized - found = True - elif isinstance(value, (dict, list)): - _walk(value) - elif isinstance(node, list): - for idx, value in enumerate(node): - if isinstance(value, str): - sanitized = _strip_non_ascii(value) - if sanitized != value: - node[idx] = sanitized - found = True - elif isinstance(value, (dict, list)): - _walk(value) - - _walk(payload) - return found - - __all__ = [ "_SURROGATE_RE", "close_interrupted_tool_sequence", @@ -614,13 +405,13 @@ __all__ = [ "_sanitize_tools_non_ascii", "_strip_images_from_messages", "_sanitize_structure_non_ascii", - # call_id policy owners (F4 consolidation) + # call_id policy owners "deterministic_call_id", "coalesce_tool_call_id", "tool_call_id_variants", "tool_result_id_variants", "uniquify_tool_call_ids", - # reasoning_content policy owners (F4 consolidation) + # reasoning_content policy owners "reasoning_echo_family", "matches_reasoning_echo_family", "needs_reasoning_echo", @@ -631,46 +422,36 @@ __all__ = [ # --------------------------------------------------------------------------- -# call_id policy — single owner (audit F4, incident chain I4) -# --------------------------------------------------------------------------- +# call_id policy — single owner for hash synthesis, ``call_id or id`` +# coalescing, and duplicate-id repair. # -# Three forked policy sites converged here: -# * agent/codex_responses_adapter.py `_deterministic_call_id` — hash -# synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2). -# * run_agent.AIAgent._get_tool_call_id_static — `call_id or id` -# coalescing for dicts and SDK objects. -# * run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with -# deterministic `_d` suffixes (#58327 loss class). +# NOT consolidated on purpose: agent/transports/codex_event_projector's +# _deterministic_call_id maps codex app-server ITEM ids (`codex__`), +# not chat tool-call content; merging would change ids and invalidate caches. # -# NOT consolidated (different scheme on purpose): -# agent/transports/codex_event_projector._deterministic_call_id maps codex -# app-server ITEM ids (`codex__`), not chat tool-call -# content; merging the two would change ids and invalidate prompt caches. -# -# HARD INVARIANT: everything here must stay deterministic (never uuid4) and +# HARD INVARIANT: everything here stays deterministic (never uuid4) and # byte-identical for existing inputs — these ids feed prompt-cache prefixes. +# --------------------------------------------------------------------------- + + +def _tc_field(tc: Any, key: str) -> Any: + """Read ``key`` from a tool-call entry that may be a dict or an SDK object.""" + return tc.get(key) if isinstance(tc, dict) else getattr(tc, key, None) def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str: - """Generate a deterministic call_id from tool call content. - - Used as a fallback when the API doesn't provide a call_id. - Deterministic IDs prevent cache invalidation — random UUIDs would - make every API call's prefix unique, breaking OpenAI's prompt cache. - """ + """Deterministic call_id fallback when the API omits one (random ids would + make every prefix unique and break prompt caching).""" seed = f"{fn_name}:{arguments}:{index}" digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12] return f"call_{digest}" def _expand_tool_id_variants(values: tuple[Any, ...]) -> frozenset[str]: - """Return every wire spelling of one or more tool-call identifiers. + """Every wire spelling of one tool-call identifier. - Responses bridges may expose the pairing id and response-item id - separately, or encode both as ``call_id|response_item_id``. The values - are aliases for one call, not distinct calls. Keeping the expansion in - the shared policy module prevents the repair and pre-send paths from - drifting apart again. + Responses bridges may expose the pairing id and response-item id separately + or encode both as ``call_id|response_item_id``; all are aliases for ONE call. """ variants: set[str] = set() for raw in values: @@ -690,19 +471,9 @@ def _expand_tool_id_variants(values: tuple[Any, ...]) -> frozenset[str]: def tool_call_id_variants(tc: Any) -> frozenset[str]: """Return all pairing-id variants carried by a tool-call entry.""" - if isinstance(tc, dict): - values = ( - tc.get("call_id"), - tc.get("id"), - tc.get("response_item_id"), - ) - else: - values = ( - getattr(tc, "call_id", None), - getattr(tc, "id", None), - getattr(tc, "response_item_id", None), - ) - return _expand_tool_id_variants(values) + return _expand_tool_id_variants( + (_tc_field(tc, "call_id"), _tc_field(tc, "id"), _tc_field(tc, "response_item_id")) + ) def tool_result_id_variants(tool_call_id: Any) -> frozenset[str]: @@ -711,18 +482,13 @@ def tool_result_id_variants(tool_call_id: Any) -> frozenset[str]: def coalesce_tool_call_id(tc: Any) -> str: - """Extract the effective call ID from a tool_call entry (dict or object). + """Effective call id of a tool_call entry (dict or object). - Single owner for the canonical pairing rule: Codex Responses tool calls - carry ``call_id`` (authoritative pairing key), Chat Completions ones carry - ``id`` only, and bridge ids may encode ``call_id|response_item_id``. - Returns ``""`` when neither pairing field is set. + Codex Responses calls carry ``call_id`` (authoritative pairing key), Chat + Completions carry ``id`` only, and bridge ids may be ``call_id|response_item_id``. + Returns ``""`` when neither is set. """ - if isinstance(tc, dict): - values = (tc.get("call_id"), tc.get("id")) - else: - values = (getattr(tc, "call_id", None), getattr(tc, "id", None)) - for raw in values: + for raw in (_tc_field(tc, "call_id"), _tc_field(tc, "id")): if not isinstance(raw, str): continue value = raw.strip() @@ -732,37 +498,25 @@ def coalesce_tool_call_id(tc: Any) -> str: def uniquify_tool_call_ids(tool_calls: list) -> list: - """Ensure every tool call in a single assistant turn has a distinct id. + """Ensure every tool call in one assistant turn has a distinct id. - Some models/providers reuse one call id across different calls in a - single batch (observed with native Kimi Responses replays, Ollama- - compatible endpoints, and degraded models at long context; same bug - class as openclaw/openclaw#110518 / #110956). Duplicate ids are lossy - downstream: the pre-API sanitizer keeps only the first call/result - pair per id (#58327), so the later call's result silently vanishes - from every replayed payload, and strict providers (Anthropic - tool_use, DeepSeek) reject duplicate ids outright. - - The first occurrence keeps its id; later collisions get a - deterministic ``_d`` suffix — never a random UUID, which would - break prompt-cache prefix stability across replays. Mutates the - entries in place (SDK models / SimpleNamespace / dicts) and returns - the same list. Blank/missing ids are left for the deterministic - fallback in ``build_assistant_message``. + Some providers reuse one id across calls in a batch; the pre-API sanitizer + then keeps only the first call/result pair per id (the later result silently + vanishes) and strict providers reject duplicates outright. First occurrence + keeps its id; later collisions get a deterministic ``_d`` suffix + (never uuid4 — cache-prefix stability). Mutates entries in place (SDK models + / SimpleNamespace / dicts) and returns the same list. Blank ids are left for + the deterministic fallback in ``build_assistant_message``. """ seen: set = set() for tc in tool_calls or []: - # Same coalescing rule as ``coalesce_tool_call_id`` but tolerant of - # non-string ids (degraded models can emit ints/None here). - if isinstance(tc, dict): - raw = tc.get("call_id") or tc.get("id") or "" - else: - raw = getattr(tc, "call_id", None) or getattr(tc, "id", None) or "" + # Same coalescing rule as coalesce_tool_call_id, tolerant of non-string ids. + raw = _tc_field(tc, "call_id") or _tc_field(tc, "id") or "" raw = raw.strip() if isinstance(raw, str) else "" if not raw: continue - # Composite Responses ids ("call_x|fc_y") collide on the call - # half — that's the pairing key providers enforce per turn. + # Composite Responses ids ("call_x|fc_y") collide on the call half — + # that's the pairing key providers enforce per turn. cid = raw.split("|", 1)[0] if not cid: continue @@ -777,8 +531,8 @@ def uniquify_tool_call_ids(tool_calls: list) -> list: seen.add(new_id) def _renamed(value): - # Preserve a composite id's response-item half so the - # provider's real fc_/item id survives the rename. + # Keep a composite id's response-item half so the provider's real + # fc_/item id survives the rename. if isinstance(value, str) and "|" in value: return f"{new_id}|{value.split('|', 1)[1]}" return new_id @@ -800,7 +554,7 @@ def uniquify_tool_call_ids(tool_calls: list) -> list: "Could not uniquify duplicate tool call id %s", cid ) continue - _fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None) + _fn = _tc_field(tc, "function") _fn_name = (_fn.get("name") if isinstance(_fn, dict) else getattr(_fn, "name", None)) or "?" logger.warning( "Model reused tool call id %s within one turn; renamed the " @@ -811,29 +565,21 @@ def uniquify_tool_call_ids(tool_calls: list) -> list: # --------------------------------------------------------------------------- -# reasoning_content policy — single owner (audit F4) -# --------------------------------------------------------------------------- +# reasoning_content policy — single owner. The POLICY (which provider direction +# gets strip vs re-pad) lives here as one rule table + apply functions; adapters +# keep only SYNTAX mapping (e.g. anthropic_adapter → thinking block). # -# The strip-vs-repad decision was previously forked across the wire files in -# separate incident commits (2b3a4f0af8 strip for strict providers, -# b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi pad). The -# POLICY — which provider direction gets which treatment — lives here as one -# rule table + apply functions; adapters keep only SYNTAX mapping (e.g. -# anthropic_adapter turning reasoning_content into a thinking block). -# -# Direction table: # require-side (echo-back enforced; replays 400 without the field): # kimi — provider kimi-coding/kimi-coding-cn, or host api.kimi.com / -# moonshot.ai / moonshot.cn. Host-driven on purpose: -# aggregators re-exporting kimi models reject the echo. +# moonshot.ai / moonshot.cn. Host-driven on purpose: aggregators +# re-exporting kimi models reject the echo. # deepseek — provider "deepseek", model contains "deepseek", or host -# api.deepseek.com (#15250; V4 rejects empty-string pads, -# hence the " " single-space pad, #17341). -# mimo — provider "xiaomi", model contains "mimo", or host -# *.xiaomimimo.com. -# strict side (field rejected with 400/422 "Extra inputs are not -# permitted"): everyone else — Mistral, Cerebras, Groq, SambaNova, … -# (#45655). Strip the key entirely, even a single-space pad. +# api.deepseek.com. V4 rejects empty-string pads → " " single space. +# mimo — provider "xiaomi", model contains "mimo", or host *.xiaomimimo.com. +# strict side (field rejected 400/422 "Extra inputs are not permitted"): +# everyone else — Mistral, Cerebras, Groq, SambaNova, … Strip the key +# entirely, even a single-space pad. +# --------------------------------------------------------------------------- _REASONING_ECHO_RULES: tuple = ( # (family, exact providers (raw), exact providers (lowered), @@ -845,13 +591,7 @@ _REASONING_ECHO_RULES: tuple = ( ("mimo", frozenset(), frozenset({"xiaomi"}), ("mimo",), ("api.xiaomimimo.com", "xiaomimimo.com")), ) - - -def _family_rule(family: str) -> tuple: - for rule in _REASONING_ECHO_RULES: - if rule[0] == family: - return rule - raise KeyError(family) +_REASONING_ECHO_RULE_BY_FAMILY = {rule[0]: rule for rule in _REASONING_ECHO_RULES} def matches_reasoning_echo_family( @@ -859,13 +599,12 @@ def matches_reasoning_echo_family( ) -> bool: """True when (provider, model, base_url) matches one echo-back family. - Families can overlap (e.g. a deepseek-named model pointed at a kimi - host); this membership test is independent per family so per-family - predicates keep their original semantics. + Families can overlap (a deepseek-named model on a kimi host); membership is + tested independently per family. Raises KeyError for an unknown family. """ from utils import base_url_host_matches - _, raw_providers, lowered_providers, model_subs, hosts = _family_rule(family) + _, raw_providers, lowered_providers, model_subs, hosts = _REASONING_ECHO_RULE_BY_FAMILY[family] provider_lower = (provider or "").lower() model_lower = (model or "").lower() if provider in raw_providers or provider_lower in lowered_providers: @@ -876,13 +615,8 @@ def matches_reasoning_echo_family( def reasoning_echo_family(provider: Any, model: Any, base_url: Any) -> "str | None": - """Classify the provider direction for the reasoning_content echo policy. - - Returns ``"kimi"``, ``"deepseek"``, or ``"mimo"`` (first match in table - order) when the target endpoint enforces reasoning_content echo-back on - assistant turns, else ``None`` (strict/indifferent side — the field must - be stripped). - """ + """``"kimi"`` / ``"deepseek"`` / ``"mimo"`` (first match in table order) when the + endpoint enforces reasoning_content echo-back, else ``None`` (strip side).""" for rule in _REASONING_ECHO_RULES: if matches_reasoning_echo_family(rule[0], provider, model, base_url): return rule[0] @@ -900,22 +634,16 @@ def stale_thinking_reaches_wire( """True when stale assistant ``reasoning``/``reasoning_content`` text is actually replayed on the wire for the active route. - This is the single wire-truth predicate the compaction TRIGGER estimator - and the tail-budget walks must share (#84371): when they disagree, a - reasoning-heavy session can simultaneously look over-threshold to - preflight and fully tail-protected to the walk — an infinite ineffective - compaction loop. - - * ``codex_responses``: the Responses input builder - (``_chat_messages_to_responses_input``) never reads the text keys — - reasoning continuity rides the encrypted ``codex_reasoning_items`` - sidecar, which both estimators already charge unconditionally. Stale - thinking TEXT never ships → ``False``. - * chat-completions echo-back families (DeepSeek/Kimi/MiMo thinking - mode): ``apply_reasoning_content_policy`` replays the stored - ``reasoning_content`` verbatim on EVERY assistant turn → ``True``. - * everything else: stripped or one-space-padded at send time (#73624) - → ``False``. + The single wire-truth predicate the compaction TRIGGER estimator and the + tail-budget walks must share: if they disagree, a reasoning-heavy session + can look over-threshold to preflight yet fully tail-protected to the walk — + an infinite ineffective compaction loop. + * ``codex_responses``: the Responses input builder never reads the text keys + (continuity rides the encrypted ``codex_reasoning_items`` sidecar, already + charged by both estimators) → False. + * echo-back families: ``apply_reasoning_content_policy`` replays stored + ``reasoning_content`` verbatim on every assistant turn → True. + * everything else: stripped or one-space-padded at send time → False. """ if (api_mode or "") == "codex_responses": return False @@ -927,31 +655,17 @@ def apply_reasoning_content_policy( ) -> None: """Copy provider-facing reasoning fields onto an API replay message. - ``needs_thinking_pad`` is the require-side flag (see - ``needs_reasoning_echo`` / the agent's cached - ``_needs_thinking_reasoning_pad``). Mutates ``api_msg`` in place. + ``needs_thinking_pad`` is the require-side flag (``needs_reasoning_echo``). + Mutates ``api_msg`` in place. """ if source_msg.get("role") != "assistant": return - # 1. Explicit reasoning_content already set. - # - # When the active provider enforces the thinking-mode echo-back - # (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their - # own space-placeholder written at creation time and any valid reasoning - # from the same provider. Sessions persisted BEFORE #17341 have - # empty-string placeholders pinned at creation time; DeepSeek V4 Pro - # rejects those with HTTP 400, so upgrade "" → " " on replay. - # - # When the active provider does NOT enforce echo-back, strip the field - # entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq, - # SambaNova, …) reject ANY reasoning_content key in input messages with - # HTTP 400/422 ("Extra inputs are not permitted"), even an empty string - # or a single-space pad. This is the cross-provider fallback case: a - # reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a - # fallback to a strict provider replays that pad and 422s. Stripping - # here covers the rebuild path; ``reapply_reasoning_echo`` covers the - # already-built api_messages path. Refs #45655. + # 1. Explicit reasoning_content set. Require-side: preserve verbatim, + # upgrading legacy "" placeholders to " " (DeepSeek V4 400s on ""). Strict + # side: strip entirely — a reasoning primary pads history with " ", then a + # fallback to Mistral/Cerebras/Groq replays the pad and 422s. This covers + # the rebuild path; reapply_reasoning_echo covers already-built api_messages. existing = source_msg.get("reasoning_content") if isinstance(existing, str): if not needs_thinking_pad: @@ -962,17 +676,10 @@ def apply_reasoning_content_policy( api_msg["reasoning_content"] = existing return - # 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi, - # if the source turn has tool_calls AND a 'reasoning' field but no - # 'reasoning_content' key, the 'reasoning' text was written by a - # prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message - # pins reasoning_content at creation time for tool-call turns, so the - # shape (reasoning set, reasoning_content absent, tool_calls present) - # is unreachable from same-provider DeepSeek history after this fix. - # Inject a single space to satisfy the API without leaking another - # provider's chain of thought to DeepSeek/Kimi. Space (not "") - # because DeepSeek V4 Pro rejects empty-string reasoning_content - # in thinking mode (refs #17341). + # 2. Cross-provider poisoned history: tool_calls + 'reasoning' but no + # 'reasoning_content' key means the reasoning text came from ANOTHER + # provider (DeepSeek's own build pins reasoning_content for tool-call + # turns). Pad with " " to satisfy the API without leaking foreign CoT. normalized_reasoning = source_msg.get("reasoning") if ( needs_thinking_pad @@ -983,12 +690,9 @@ def apply_reasoning_content_policy( api_msg["reasoning_content"] = " " return - # 3. Healthy session: promote 'reasoning' field to 'reasoning_content' - # for providers that use the internal 'reasoning' key. - # This must happen before the unconditional empty-string fallback so - # genuine reasoning content is not overwritten (#15812 regression in - # PR #15478). Only promote for providers that enforce echo-back — - # strict providers reject the field (refs #45655). + # 3. Healthy session: promote internal 'reasoning' → 'reasoning_content' + # (must precede the unconditional pad so real reasoning isn't overwritten), + # but only for echo-back providers — strict ones reject the field. if isinstance(normalized_reasoning, str) and normalized_reasoning: if needs_thinking_pad: api_msg["reasoning_content"] = normalized_reasoning @@ -996,51 +700,27 @@ def apply_reasoning_content_policy( api_msg.pop("reasoning_content", None) return - # 4. DeepSeek / Kimi thinking mode: all assistant messages need - # reasoning_content. Inject a single space to satisfy the provider's - # requirement when no explicit reasoning content is present. Covers - # both tool-call turns (already-poisoned history with no reasoning - # at all) and plain text turns. Space (not "") because DeepSeek V4 - # Pro tightened validation and rejects empty string with HTTP 400 - # ("The reasoning content in the thinking mode must be passed back - # to the API"). Refs #17341. + # 4. Require-side with no reasoning at all: every assistant turn needs the + # field; " " (not "") because DeepSeek V4 rejects empty string. if needs_thinking_pad: api_msg["reasoning_content"] = " " return - # 5. reasoning_content was present but not a string (e.g. None after - # context compaction). Don't pass null to the API. + # 5. reasoning_content present but not a string (e.g. None after + # compaction) — never pass null to the API. api_msg.pop("reasoning_content", None) def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int: - """Re-pad (or strip) assistant turns' reasoning_content for the active provider. + """Re-pad (or strip) assistant turns' reasoning_content for the ACTIVE provider. - ``api_messages`` is built once, before the retry loop, while the *primary* - provider is active. A mid-conversation fallback can then switch providers, - so the reasoning fields baked into ``api_messages`` are shaped for the - *prior* provider and must be reconciled against the *current* one: + ``api_messages`` is built once before the retry loop under the primary + provider; a mid-conversation fallback can switch providers, so the baked-in + reasoning fields must be reconciled: switching TO a require-side provider + needs the pad re-applied (else 400), switching TO a strict provider needs + the stale pad stripped (else 422). Idempotent; call every iteration. - * Switching TO a require-side provider (DeepSeek / Kimi / MiMo thinking - mode): assistant turns built when the prior provider did NOT need the - echo-back go out without ``reasoning_content`` and the new provider - rejects them with HTTP 400 ("The reasoning_content in the thinking mode - must be passed back"). Re-apply the pad. - - * Switching TO a strict provider that rejects the field (Mistral, - Cerebras, Groq, SambaNova, …): assistant turns built under a reasoning - primary carry a ``reasoning_content`` pad (often a single space ``" "``), - and the strict provider rejects it with HTTP 400/422 ("Extra inputs are - not permitted"). Strip the field. This is the exact cross-provider - fallback bug from #45655 — a DeepSeek primary pads history with ``" "``, - the request falls back to Mistral, and Mistral 422s on the stale pad. - - Calling this immediately before building the request kwargs reconciles the - fields against the *current* provider. It is idempotent and safe to call - every iteration; it covers every fallback path. - - Returns the number of assistant turns whose reasoning_content was added or - removed. + Returns the number of assistant turns whose reasoning_content changed. """ changed = 0 for api_msg in api_messages: @@ -1052,27 +732,14 @@ def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int: apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad) if api_msg.get("reasoning_content"): changed += 1 - else: - # Strict provider — strip any stale reasoning_content pad left - # over from a reasoning primary so the fallback request doesn't - # 400/422 on it. - if "reasoning_content" in api_msg: - api_msg.pop("reasoning_content", None) - changed += 1 + elif "reasoning_content" in api_msg: + api_msg.pop("reasoning_content", None) + changed += 1 return changed -# --------------------------------------------------------------------------- -# Image / multimodal parts — evaluated, NOT consolidated (verdict: syntax) -# --------------------------------------------------------------------------- -# -# The per-adapter image handling is format-specific SYNTAX, not shared policy: -# * anthropic_adapter (~1817): data-URL → Anthropic `source: {type: base64}` -# block mapping — Anthropic wire shape only. -# * codex_responses_adapter (~113/165/812): chat `image_url` parts → -# Responses `input_image` items and image counting for log summaries — -# Responses wire shape only. -# * transports/chat_completions: pass-through (native format). -# The one genuinely shared image POLICY — removing images when a server -# rejects them while preserving tool_call_id pairing — already has a single -# owner here: ``_strip_images_from_messages`` above. +# Image / multimodal parts are deliberately NOT consolidated here: per-adapter +# handling (anthropic base64 source blocks, Responses input_image items) is +# format-specific SYNTAX. The one shared image POLICY — removing images when a +# server rejects them while preserving tool_call_id pairing — is +# ``_strip_images_from_messages`` above. diff --git a/agent/native_compaction.py b/agent/native_compaction.py index 14ce28e932..30e6efdb99 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -1,43 +1,24 @@ """Native OpenAI Responses server-side compaction — gpt-5.6 on direct OpenAI routes only. -OpenAI's Responses API supports server-side compaction: include -``context_management=[{"type": "compaction", "compact_threshold": N}]`` in a -``/v1/responses`` request and, when the rendered input crosses N tokens, the -server summarizes older context into an opaque ``compaction`` output item -(``encrypted_content``, sealed to the issuing endpoint). Replaying that item -as an input item on later requests stands in for the pruned history, so the -model keeps long-horizon recall without the client ever seeing a summary. -Docs: https://developers.openai.com/api/docs/guides/compaction +Including ``context_management=[{"type": "compaction", "compact_threshold": N}]`` +in a ``/v1/responses`` request makes the server summarize older context into an +opaque ``compaction`` item (``encrypted_content``, sealed to the issuing +endpoint) once the input crosses N tokens; replaying that item stands in for +the pruned history. Docs: https://developers.openai.com/api/docs/guides/compaction -Hermes' support is deliberately narrow (live verification, Aug 2026): +Support is deliberately narrow (live-verified): +* gpt-5.6 family only — gpt-5.1/5.2 fail server-side (HTTP 500 blocking, a + permanent stall streaming) with no structured "unsupported" rejection, so an + explicit model-family check is the only safe gate. +* Direct OpenAI routes only (api.openai.com or the ChatGPT Codex backend) — + other Responses surfaces would 400 on the field and cannot mint/decrypt the blob. -* **gpt-5.6 family only.** gpt-5.6 and its variants compact correctly. - Sending the field to gpt-5.1 / gpt-5.2 reliably fails server-side — - HTTP 500 on the blocking path and a permanent stall on the streaming - path (90s watchdog x 3 retries = a dead turn). There is no structured - "unsupported" rejection to downgrade on, so the only safe gate is an - explicit model-family check. -* **Direct OpenAI routes only:** api.openai.com (API key) or the ChatGPT - Codex backend (subscription OAuth). Every other Responses surface - (xAI, GitHub/Copilot, relays, local servers) never sees the field — - most would 400 on the unknown parameter, and none can mint or decrypt - the compaction blob. - -Ownership model: Hermes' local compression stays fully armed as the -fallback owner. The native threshold is clamped safely below the local -compressor's trigger so the server compacts first; if it doesn't (native -disabled mid-session, provider hiccup, non-eligible route), the local -summarizer fires exactly as before. There is no new custody state — the -captured compaction items ride the existing ``codex_reasoning_items`` -sidecar, which already handles persistence (state.db), gateway session -replay, cross-issuer stamping, and the encrypted-replay kill switch. - -This module stays free of transport/adapter dependencies so the transport, -adapter, and conversation loop can share the gate without import cycles. The -two exceptions — ``agent.context_compressor`` and ``agent.message_content`` — -sit below this module in the dependency graph (neither imports -``native_compaction``), so importing their provenance/text primitives here -introduces no cycle. +Hermes' local compressor stays armed as fallback owner: the native threshold is +clamped below the local trigger so the server compacts first, and captured +compaction items ride the existing ``codex_reasoning_items`` sidecar (persistence, +replay, cross-issuer stamping, kill switch). This module stays free of +transport/adapter imports so transport, adapter, and loop share the gate +without cycles; ``context_compressor`` and ``message_content`` sit below it. """ from __future__ import annotations @@ -52,14 +33,11 @@ from agent.message_content import flatten_message_text logger = logging.getLogger(__name__) # Native compaction fires this many tokens below the local compressor's -# trigger so the server always gets the first shot at compaction. +# trigger so the server always gets the first shot. LOCAL_TRIGGER_SAFETY_MARGIN = 8_192 - -# Deterministic fallback when automatic mode cannot inspect a local trigger. +# Fallback when automatic mode has no local trigger to follow. DEFAULT_COMPACT_THRESHOLD = 200_000 - -# Model-family gate. Substring match on the lowercased model id so dated -# snapshots (gpt-5.6-2026-07-xx) and variants (gpt-5.6-mini) stay eligible. +# Substring match so dated snapshots and variants (gpt-5.6-mini) stay eligible. _ELIGIBLE_MODEL_MARKER = "gpt-5.6" @@ -77,11 +55,10 @@ def resolve_native_compaction_capabilities( ) -> Dict[str, bool]: """Resolve the native-compaction capability for a runtime destination. - The result is deliberately explicit: a resolved ``False`` is different - from an unresolved capability and must survive model switches unchanged. + A resolved ``False`` is distinct from "unresolved" and must survive model + switches unchanged. """ - normalized_provider = (provider or "").strip().lower() - direct_default = normalized_provider == "openai" and not base_url + direct_default = (provider or "").strip().lower() == "openai" and not base_url eligible = is_native_compaction_model(model) and ( direct_default or is_direct_openai_route(base_url, is_codex_backend=is_codex_backend) @@ -110,10 +87,10 @@ def resolve_compact_threshold( ) -> int: """Resolve automatic mode or clamp an explicit native threshold. - An omitted or invalid setting follows the resolved local compressor trigger. - An explicit positive integer remains absolute unless it must be clamped so - native compaction fires first. ``local_trigger_tokens`` is - ``ContextCompressor.threshold_tokens`` when a compressor is attached. + An omitted/invalid setting follows the local compressor trigger + (``ContextCompressor.threshold_tokens``) minus the safety margin. An + explicit positive integer is absolute unless it must be clamped so native + compaction fires first. Booleans are never thresholds. """ local = None try: @@ -139,7 +116,7 @@ def resolve_compact_threshold( ) except (TypeError, ValueError): configured = None - if isinstance(configured_threshold, bool) or configured is None or configured <= 0: + if configured is None or configured <= 0: return upper if upper is not None else DEFAULT_COMPACT_THRESHOLD if upper is None: return configured @@ -150,11 +127,7 @@ _checkpoint_suppression_logged = False def _warn_native_compaction_suppressed_by_checkpoint_gate() -> None: - """Log once per process that the checkpoint gate suppresses native compaction. - - The suppression itself is re-evaluated per request; only the log line is - deduplicated so a long session does not repeat it on every API call. - """ + """Log once per process; the suppression itself is re-evaluated per request.""" global _checkpoint_suppression_logged if _checkpoint_suppression_logged: return @@ -175,27 +148,22 @@ def native_compaction_context_management( ) -> Optional[List[Dict[str, Any]]]: """Return the ``context_management`` payload for this request, or None. - None means "do not send the field" — the request is byte-identical to - pre-feature behavior. All gates are re-checked per request so a - mid-session model switch or the in-session kill switch - (``agent.codex_responses_native_compaction = False``, set by the - conversation loop's rejection recovery) takes effect on the next call. + None means "do not send the field" (request byte-identical to pre-feature). + Every gate is re-checked per request so a mid-session model switch or the + in-session kill switch (``agent.codex_responses_native_compaction = False``, + set by rejection recovery) takes effect on the next call. """ capabilities = getattr(agent, "runtime_capabilities", None) - if isinstance(capabilities, dict): - if not bool(capabilities.get("native_compaction", False)): - return None - if not bool(getattr(agent, "codex_responses_native_compaction", False)): + if isinstance(capabilities, dict) and not capabilities.get("native_compaction", False): return None - # compression.enabled: false disables ALL automatic compaction, native - # included — mirrors the codex_app_server_auto contract. - if not bool(getattr(agent, "compression_enabled", True)): + if not getattr(agent, "codex_responses_native_compaction", False): return None - # compression.checkpoint_required: server-side compaction is a lossy - # boundary the provider owns — no pre-compress checkpoint can run before - # the server replaces older context. Keep the checkpoint-aware Hermes - # compressor authoritative instead of silently letting the server - # compact. Explicit-True check matches the compress_context() gate. + # compression.enabled: false disables ALL automatic compaction, native included. + if not getattr(agent, "compression_enabled", True): + return None + # Server-side compaction is a lossy boundary the provider owns — no + # pre-compress checkpoint can run first — so the checkpoint-aware Hermes + # compressor stays authoritative. Explicit-True matches compress_context(). if getattr(agent, "compression_checkpoint_required", False) is True: _warn_native_compaction_suppressed_by_checkpoint_gate() return None @@ -219,13 +187,10 @@ def native_compaction_context_management( return [{"type": "compaction", "compact_threshold": threshold}] -# Retention budget for plaintext user messages carried across a native -# compaction boundary (mirrors Codex CLI's RETAINED_MESSAGE_TOKEN_BUDGET). -# Live verification (Aug 2026, gpt-5.6 @ api.openai.com): the server renders +# Retention budgets for plaintext user messages / local compression summaries +# carried across a native compaction boundary (mirrors Codex CLI's +# RETAINED_MESSAGE_TOKEN_BUDGET; the summary budget prevents summary inflation). RETAINED_USER_MESSAGE_TOKEN_BUDGET = 64_000 - -# Retention budget for local compression summary messages carried across a native -# compaction boundary to prevent summary token inflation. RETAINED_SUMMARY_TOKEN_BUDGET = 32_000 @@ -235,12 +200,7 @@ def _approx_tokens(text: str) -> int: def _extract_item_text(item: Any) -> Optional[str]: - """Extract measurable text from message content and fallback fields. - - Returns None when the item carries no measurable text. Handles string - content, multipart lists (input_text/text/output_text), and nested - metadata text. - """ + """Measurable text from a Responses item (string/multipart/metadata), or None.""" if not isinstance(item, dict): return None @@ -272,12 +232,10 @@ def _extract_item_text(item: Any) -> Optional[str]: def _has_retainable_image_content(item: Any) -> bool: - """Return True for a converted Responses message with a valid image part. + """True for a converted Responses message with a valid ``input_image`` part. - The pruning boundary receives normalized Responses items, so only the - adapter-owned ``input_image`` shape is authority here. Unknown, malformed, - or empty multipart placeholders must not become durable history merely - because their list is non-empty. + Only the adapter-owned ``input_image`` shape counts: unknown or empty + multipart placeholders must not become durable history for being non-empty. """ if not isinstance(item, dict): return False @@ -295,26 +253,11 @@ def _has_retainable_image_content(item: Any) -> bool: return False -def _is_summary_item(item: Any) -> bool: - """True when *item* is a canonical Hermes compression-summary message. - - Delegates entirely to - ``agent.context_compressor.is_compaction_summary_message`` — the single - authoritative provenance check already used by every other summary - consumer (memory providers, frontends, the compactor itself). It prefers - the exact, truthy ``COMPRESSED_SUMMARY_METADATA_KEY`` marker and falls - back to the canonical prefix classifier (``SUMMARY_PREFIX`` / - ``LEGACY_SUMMARY_PREFIX`` / historical prefixes, including the - merge-into-tail shape) for the case where the underscore-prefixed key - was already stripped by a wire sanitizer. - - Deliberately NOT a second heuristic: no arbitrary underscore-key scan, no - inference from a falsy or unrelated metadata key, and no matching on - ad-hoc content headings like ``"## Summary"`` in ordinary text — any of - those can promote a normal user/assistant message (or adversarial - content) to durable retained history (#90975 review). - """ - return is_compaction_summary_message(item) +# Canonical provenance check (metadata marker, then canonical prefix classifier). +# Deliberately NOT a second heuristic: no underscore-key scan, no matching on +# ad-hoc headings — either could promote ordinary or adversarial content to +# durable retained history. +_is_summary_item = is_compaction_summary_message def prune_pre_checkpoint_items( @@ -326,48 +269,29 @@ def prune_pre_checkpoint_items( ) -> List[Dict[str, Any]]: """Restructure Responses input around the newest compaction checkpoint. - The server drops every input item that precedes a replayed ``compaction`` - item (live-verified Aug 2026), so sending pre-checkpoint history is dead - weight AND silently erases the user's plaintext asks — including any - local-compression summary the agent already produced, which previously - vanished here because it carries ``role="assistant"``, not ``"user"`` - (#90975). When a checkpoint is present, rebuild the wire as:: + The server drops every input item preceding a replayed ``compaction`` item, + which silently erases the user's plaintext asks and any local-compression + summary (``role="assistant"``). With a checkpoint present, rebuild as:: [checkpoint run] + [retained user & summary messages (newest-first budget)] + [post] - The NEWEST contiguous run of checkpoints wins. - - Retained user messages are kept verbatim within - ``retained_user_token_budget``; the boundary message is head-truncated - when it only partially fits (string content only) — goals are usually - stated up front, so the head is the valuable end. A recognized + - User messages are kept verbatim within ``retained_user_token_budget``; + the boundary message is head-truncated when it only partially fits + (string content only — goals are stated up front). A recognized image-only user message is retained whole at one-token cost. - - Compression summary messages (``_is_summary_item``, the canonical - ``agent.context_compressor`` provenance check) are retained whole - within ``retained_summary_token_budget``. A summary is never - byte/character-sliced: Hermes summaries carry structural framing - (handoff prefix, end marker, merge-into-tail delimiters) that a blind - slice can corrupt, so one that doesn't fit whole is dropped instead. - A summary already retained once (identical text) is never duplicated, - so repeated checkpoints stay idempotent. - - ``enable_summary_retention`` is a function-level override (used by - tests and callers that need the pre-#90975 behavior back); it is not - wired to a user-facing config surface. - - Original relative chronological order between user messages and - summaries is preserved. - - ``item_sources`` (optional, parallel to ``items``) is the raw chat - message each Responses item was converted from. By the time a summary - reaches this function as a converted ``item`` it can already be lossy: - a merge-into-tail tool-result carrier becomes a typed - ``function_call_output`` (no ``content``/``role`` survives the - conversion at all), and a merge-into-tail assistant carrier can be - shadowed by a stale exact ``codex_message_items`` replay captured - before the merge rewrote its content. When a source is provided and is - itself a canonical summary carrier (``is_compaction_summary_message``), - its content is read directly from the source — never from the - converted item — and it is retained as a synthesized - ``role="assistant"`` message regardless of what shape the original - item took. Without ``item_sources`` (default), retention only sees - what survived conversion, matching pre-#90976 behavior (#90976). + - Summaries are retained whole within ``retained_summary_token_budget`` and + never sliced (their structural framing would corrupt); one that doesn't + fit is dropped. Identical summary text is never retained twice. + - Relative order between user messages and summaries is preserved. + - ``item_sources`` (parallel to ``items``) is the raw chat message each item + was converted from. Conversion can be lossy for summaries (a + merge-into-tail carrier becomes a typed ``function_call_output``, or an + assistant carrier is shadowed by a stale exact replay), so when a source + is itself a canonical summary carrier its content is read from the + SOURCE and retained as a synthesized ``role="assistant"`` message. + - ``enable_summary_retention`` is a function-level override for tests, not + a config surface. """ if not isinstance(items, list) or not items: return items @@ -379,7 +303,6 @@ def prune_pre_checkpoint_items( if last_cp is None: return items - # Extend backwards over the contiguous run ending at last_cp. first_cp = last_cp while ( first_cp > 0 @@ -403,14 +326,12 @@ def prune_pre_checkpoint_items( seen_summary_texts: set = set() def _try_retain_summary(text: Optional[str]) -> Optional[Dict[str, Any]]: - """Check budget/dedup/cost for a summary; return cost info or None.""" + """Budget/dedup check for a summary; return its cost or None.""" if not text or summary_remaining <= 0 or text in seen_summary_texts: return None cost = _approx_tokens(text) if cost > summary_remaining: - # Never byte-slice a summary's structural framing — drop it - # whole rather than corrupt the handoff prefix / end marker. - return None + return None # never slice a summary's structural framing seen_summary_texts.add(text) return {"cost": cost} @@ -418,15 +339,10 @@ def prune_pre_checkpoint_items( if not isinstance(item, dict): continue - # Canonical source-based summary detection: reads the ORIGINAL chat - # message's own content, so it sees past a lossy conversion (a - # typed `function_call_output` wrapper, or a stale exact-replay - # message) that erased the summary from `item` itself (#90976). - # This is never a heuristic promotion of arbitrary item content — - # it only fires when the source message itself is a canonical, - # provenance-tagged summary carrier. + # Source-based detection sees past a lossy conversion; it only fires + # when the source itself is a provenance-tagged summary carrier. if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source): - text = flatten_message_text(source.get("content")) if isinstance(source, dict) else "" + text = flatten_message_text(source.get("content")) text = text if text.strip() else None result = _try_retain_summary(text) if result: @@ -438,9 +354,7 @@ def prune_pre_checkpoint_items( summary_remaining -= result["cost"] continue - # Skip typed non-message items (function_call_output etc. never - # carry role=user or a summary flag, but stay defensive about - # future shapes). + # Typed non-message items never carry role=user or a summary flag. if "type" in item and item.get("type") != "message": continue @@ -476,8 +390,7 @@ def prune_pre_checkpoint_items( retained_reversed.append(truncated) user_remaining = 0 - retained_ordered = list(reversed(retained_reversed)) - result = checkpoint_run + retained_ordered + post + result = checkpoint_run + list(reversed(retained_reversed)) + post logger.debug( "Pruned pre-checkpoint items: %d input -> %d retained (user_rem=%d, summary_rem=%d)", @@ -490,25 +403,21 @@ def prune_pre_checkpoint_items( return result +_REJECTION_MARKERS = ( + "unknown", "unsupported", "invalid", "unexpected", "not permitted", + "not allowed", "unrecognized", "extra field", "no such", "bad request", + "not supported", +) + + def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: - """True when a provider error is a STRUCTURED rejection of the - context_management field. + """True when a provider error is a STRUCTURED rejection of ``context_management``. - Used by the conversation loop's one-shot recovery: strip the field, - disable native compaction for the rest of the session, retry. Matching - is deliberately narrow — a transient 5xx/timeout whose body merely - ECHOES the request (and therefore contains the field name) must NOT - permanently downgrade native compaction for the session (#82777). - - Two conditions, both required when a status is known: - - * ``status_code`` is 400 (or unknown/None — some transports surface - only a message string; field-name matching alone is then the best - available signal, preserving pre-#82777 behavior for them), and - * the error text names ``context_management`` / ``compact_threshold`` - alongside rejection language ("unknown", "unsupported", "invalid", - "unexpected", "not permitted"...). A bare field-name echo without - rejection language does not match. + Drives the loop's one-shot recovery (strip the field, disable for the + session, retry), so matching is narrow: a transient 5xx whose body merely + ECHOES the request must not permanently downgrade native compaction. Requires + ``status_code`` 400 (or unknown — some transports surface only a message) + AND the field name alongside rejection language. """ text = str(error or "").lower() if "context_management" not in text and "compact_threshold" not in text: @@ -519,23 +428,15 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: return False except (TypeError, ValueError): pass - rejection_markers = ( - "unknown", "unsupported", "invalid", "unexpected", "not permitted", - "not allowed", "unrecognized", "extra field", "no such", "bad request", - "not supported", - ) - return any(marker in text for marker in rejection_markers) + return any(marker in text for marker in _REJECTION_MARKERS) def has_compaction_checkpoint(items: Any) -> bool: """Does this ``codex_reasoning_items`` sidecar carry a compaction checkpoint? - A ``type: "compaction"`` item is the server-side stand-in for history that - has already been pruned — cumulative context, not per-turn reasoning. It - rides the same sidecar as ordinary reasoning items, so anything that - rewrites or discards that sidecar (or the message carrying it) has to ask - this question first: the checkpoint exists in exactly one place, and the - request that loses it loses the compacted history with it. + A ``type: "compaction"`` item is cumulative context, not per-turn + reasoning, and exists in exactly one place: anything that rewrites or + discards the sidecar must ask this first or lose the compacted history. """ return any( isinstance(item, dict) and item.get("type") == "compaction" @@ -547,16 +448,11 @@ def merge_interim_reasoning_items( prior_items: Any, new_items: Any, ) -> List[Dict[str, Any]]: - """Merge ``codex_reasoning_items`` across Codex incomplete-continuation - dedup, preserving native compaction checkpoints. + """Merge ``codex_reasoning_items`` across Codex incomplete-continuation dedup. - The incomplete-retry path updates a visually-duplicate interim assistant - message in place with the newer response's replay payload. A checkpoint - captured on the EARLIER response is a cumulative context carrier the - continuation won't re-emit (the replayed checkpoint keeps the server - render under threshold), so a blind overwrite drops the only copy and the - next request balloons back to full history. Rule: newer items win, but - prior checkpoints are prepended unless the newer payload carries its own. + A checkpoint captured on the EARLIER response is not re-emitted by the + continuation, so a blind overwrite drops the only copy. Rule: newer items + win, but prior checkpoints are prepended unless the newer payload has its own. """ kept_checkpoints = [ item diff --git a/agent/onboarding.py b/agent/onboarding.py index 148fbcc9fd..214644e197 100644 --- a/agent/onboarding.py +++ b/agent/onboarding.py @@ -1,13 +1,9 @@ -""" -Contextual first-touch onboarding hints. +"""Contextual first-touch onboarding hints. -Instead of blocking first-run questionnaires, show a one-time hint the *first* -time a user hits a behavior fork — message-while-running, first long-running -tool, etc. Each hint is shown once per install (tracked in ``config.yaml`` under -``onboarding.seen.``) and then never again. - -Keep this module tiny and dependency-free so both the CLI and gateway can import -it without pulling in heavy modules. +Each hint is shown once per install the *first* time a user hits a behavior +fork (message-while-running, first long tool, ...), tracked in ``config.yaml`` +under ``onboarding.seen.``. Kept tiny and dependency-free so both the CLI +and gateway can import it. """ from __future__ import annotations @@ -19,79 +15,75 @@ from typing import Any, Mapping, Optional logger = logging.getLogger(__name__) -# ------------------------------------------------------------------------- # Flag names (stable — used as config.yaml keys under onboarding.seen) -# ------------------------------------------------------------------------- - BUSY_INPUT_FLAG = "busy_input_prompt" TOOL_PROGRESS_FLAG = "tool_progress_prompt" OPENCLAW_RESIDUE_FLAG = "openclaw_residue_cleanup" PROFILE_BUILD_FLAG = "profile_build_offered" -# ------------------------------------------------------------------------- -# Hint content -# ------------------------------------------------------------------------- +# ── Hint content ────────────────────────────────────────────────────────── +# Busy-input hints are keyed by the effective busy_input_mode that was just +# applied so the message matches reality; "interrupt" is the default branch. + +_BUSY_INPUT_HINTS_GATEWAY = { + "queue": ( + "💡 First-time tip — I queued your message instead of interrupting. " + "Send `/busy interrupt` to make new messages stop the current task " + "immediately, or `/busy status` to check. This notice won't appear again." + ), + "steer": ( + "💡 First-time tip — I steered your message into the current run; " + "it will arrive after the next tool call instead of interrupting. " + "Send `/busy interrupt` or `/busy queue` to change this, or " + "`/busy status` to check. This notice won't appear again." + ), + "redirect": ( + "💡 First-time tip — I redirected the current run using your message. " + "Completed work stays in context, and `/stop` still cancels the task. " + "Send `/busy queue` to wait for a separate turn, or `/busy status` " + "to check. This notice won't appear again." + ), +} +_BUSY_INPUT_HINT_GATEWAY_DEFAULT = ( + "💡 First-time tip — I just interrupted my current task to answer you. " + "Send `/busy queue` to queue follow-ups for after the current task instead, " + "`/busy steer` to inject them mid-run without interrupting, or " + "`/busy status` to check. This notice won't appear again." +) + +_BUSY_INPUT_HINTS_CLI = { + "queue": ( + "(tip) Your message was queued for the next turn. " + "Use /busy interrupt to make Enter stop the current run instead, " + "or /busy steer to inject mid-run. This tip only shows once." + ), + "steer": ( + "(tip) Your message was steered into the current run; it arrives " + "after the next tool call. Use /busy interrupt or /busy queue to " + "change this. This tip only shows once." + ), + "redirect": ( + "(tip) Your correction redirected the current run without discarding " + "completed work. Use /stop to cancel or /busy queue to wait for a " + "separate turn. This tip only shows once." + ), +} +_BUSY_INPUT_HINT_CLI_DEFAULT = ( + "(tip) Your message interrupted the current run. " + "Use /busy queue to queue messages for the next turn instead, " + "or /busy steer to inject mid-run. This tip only shows once." +) + def busy_input_hint_gateway(mode: str) -> str: - """Hint shown the first time a user messages while the agent is busy. - - ``mode`` is the effective busy_input_mode that was just applied, so the - message matches reality ("I just interrupted…" vs "I just queued…"). - """ - if mode == "queue": - return ( - "💡 First-time tip — I queued your message instead of interrupting. " - "Send `/busy interrupt` to make new messages stop the current task " - "immediately, or `/busy status` to check. This notice won't appear again." - ) - if mode == "steer": - return ( - "💡 First-time tip — I steered your message into the current run; " - "it will arrive after the next tool call instead of interrupting. " - "Send `/busy interrupt` or `/busy queue` to change this, or " - "`/busy status` to check. This notice won't appear again." - ) - if mode == "redirect": - return ( - "💡 First-time tip — I redirected the current run using your message. " - "Completed work stays in context, and `/stop` still cancels the task. " - "Send `/busy queue` to wait for a separate turn, or `/busy status` " - "to check. This notice won't appear again." - ) - return ( - "💡 First-time tip — I just interrupted my current task to answer you. " - "Send `/busy queue` to queue follow-ups for after the current task instead, " - "`/busy steer` to inject them mid-run without interrupting, or " - "`/busy status` to check. This notice won't appear again." - ) + """Hint shown the first time a user messages while the agent is busy (markdown).""" + return _BUSY_INPUT_HINTS_GATEWAY.get(mode, _BUSY_INPUT_HINT_GATEWAY_DEFAULT) def busy_input_hint_cli(mode: str) -> str: """CLI version of the busy-input hint (plain text, no markdown).""" - if mode == "queue": - return ( - "(tip) Your message was queued for the next turn. " - "Use /busy interrupt to make Enter stop the current run instead, " - "or /busy steer to inject mid-run. This tip only shows once." - ) - if mode == "steer": - return ( - "(tip) Your message was steered into the current run; it arrives " - "after the next tool call. Use /busy interrupt or /busy queue to " - "change this. This tip only shows once." - ) - if mode == "redirect": - return ( - "(tip) Your correction redirected the current run without discarding " - "completed work. Use /stop to cancel or /busy queue to wait for a " - "separate turn. This tip only shows once." - ) - return ( - "(tip) Your message interrupted the current run. " - "Use /busy queue to queue messages for the next turn instead, " - "or /busy steer to inject mid-run. This tip only shows once." - ) + return _BUSY_INPUT_HINTS_CLI.get(mode, _BUSY_INPUT_HINT_CLI_DEFAULT) def tool_progress_hint_gateway() -> str: @@ -110,13 +102,7 @@ def tool_progress_hint_cli() -> str: def openclaw_residue_hint_cli() -> str: - """Banner shown the first time Hermes starts and finds ``~/.openclaw/``. - - Points users at ``hermes claw migrate`` (non-destructive port of config, - memory, and skills) first. ``hermes claw cleanup`` is mentioned as the - follow-up step for users who have already migrated and want to archive - the old directory — with a warning that archiving breaks OpenClaw. - """ + """Banner shown the first time Hermes finds ``~/.openclaw/``: migrate first, cleanup (which breaks OpenClaw) after.""" return ( "A legacy OpenClaw directory was detected at ~/.openclaw/.\n" "To port your config, memory, and skills over to Hermes, run " @@ -129,10 +115,7 @@ def openclaw_residue_hint_cli() -> str: def detect_openclaw_residue(home: Optional[Path] = None) -> bool: - """Return True if an OpenClaw workspace directory is present in ``$HOME``. - - Pure filesystem check — no side effects. ``home`` override exists for tests. - """ + """True if ``$HOME/.openclaw`` is a directory (pure check; ``home`` override for tests).""" base = home or Path.home() try: return (base / ".openclaw").is_dir() @@ -140,25 +123,15 @@ def detect_openclaw_residue(home: Optional[Path] = None) -> bool: return False -# ------------------------------------------------------------------------- -# Onboarding profile-build path (opt-in, consent-gated) -# ------------------------------------------------------------------------- +# ── Onboarding profile-build path (opt-in, consent-gated) ───────────────── def profile_build_mode(config: Mapping[str, Any]) -> str: - """Resolve the onboarding profile-build mode from config. + """``config.onboarding.profile_build``: ``"off"`` never offers; anything else -> ``"ask"`` (offer on first contact). - Returns one of: - ``"ask"`` — on first contact, OFFER to build a profile (default). - ``"off"`` — never offer; the first-message note stays a plain intro. - - Read from ``config.onboarding.profile_build``. Unknown / missing values - fall back to ``"ask"`` so the default experience offers the flow. Any - network/account lookups inside the flow are separately consented to in - conversation — this setting only governs whether the offer is made. + This only governs whether the offer is made; lookups inside the flow are + consented to separately in conversation. """ - if not isinstance(config, Mapping): - return "ask" - onboarding = config.get("onboarding") + onboarding = config.get("onboarding") if isinstance(config, Mapping) else None if not isinstance(onboarding, Mapping): return "ask" mode = onboarding.get("profile_build") @@ -170,11 +143,9 @@ def profile_build_mode(config: Mapping[str, Any]) -> str: def profile_build_directive() -> str: """System-note directive appended to the very first message ever. - Instructs the agent to run a short, opt-in, consent-gated profile-build - flow and persist confirmed facts to the user-profile memory store - (``memory`` tool, ``target="user"``). Phrased so the agent ASKS before any - lookup and never silently reads connected accounts — directly addressing - the privacy concern that reading email/accounts unprompted feels invasive. + Runs a short opt-in profile-build flow persisting to the user-profile memory + store; phrased so the agent ASKS before any lookup and never silently reads + connected accounts. """ return ( "\n\n[System note: This is the user's very first message ever. " @@ -196,9 +167,7 @@ def profile_build_directive() -> str: ) -# ------------------------------------------------------------------------- -# State read / write -# ------------------------------------------------------------------------- +# ── State read / write ──────────────────────────────────────────────────── def _get_seen_dict(config: Mapping[str, Any]) -> Mapping[str, Any]: onboarding = config.get("onboarding") if isinstance(config, Mapping) else None @@ -214,12 +183,7 @@ def is_seen(config: Mapping[str, Any], flag: str) -> bool: def mark_seen(config_path: Path, flag: str) -> bool: - """Persist ``onboarding.seen. = True`` to ``config_path``. - - Uses the atomic YAML writer so a concurrent process can't observe a - partially-written file. Returns True on success, False on any error - (including the config file being absent — onboarding is best-effort). - """ + """Persist ``onboarding.seen. = True`` atomically; False on any error (best-effort).""" try: import yaml from hermes_cli.config import atomic_config_write @@ -239,7 +203,7 @@ def mark_seen(config_path: Path, flag: str) -> bool: seen = {} cfg["onboarding"]["seen"] = seen if seen.get(flag) is True: - return True # already marked — nothing to do + return True seen[flag] = True atomic_config_write(config_path, cfg) return True diff --git a/agent/plan_prompt.py b/agent/plan_prompt.py index 0678371d50..02e08f31a8 100644 --- a/agent/plan_prompt.py +++ b/agent/plan_prompt.py @@ -1,31 +1,16 @@ #!/usr/bin/env python3 -"""``/plan`` — build the plan-mode prompt that turns the user's request into a -saved markdown implementation plan, with no execution. +"""``/plan`` — build the plan-mode prompt: a saved markdown implementation plan, no execution. -``/plan`` used to be a bundled skill (``skills/software-development/plan``) -whose auto-generated slash command fell off the capped Telegram/Discord command -menus for most installs (skills are the only tier trimmed at the platform -caps, alphabetically — ``plan`` sat past the cutoff). It is now a first-class -built-in: this module builds ONE prompt that instructs the live agent to - - 1. Stay in planning mode for the turn — read-only inspection is allowed, - but no implementation, no mutating commands, no side effects. - 2. Write a concrete, bite-sized, TDD-shaped markdown plan under - ``.hermes/plans/`` in the active workspace via ``write_file``. - -There is no engine and no model-tool footprint: the agent does the work with -its existing toolset, so this works identically on local, Docker, and remote -terminal backends. Every surface (CLI ``/plan``, gateway ``/plan``, TUI -``/plan``) calls :func:`build_plan_prompt` and feeds the result to the agent -as a normal turn — same pattern as ``/learn`` and ``/init``, preserving -prompt-cache invariants (no system-prompt or history mutation). +A first-class built-in (the former bundled skill fell off capped Telegram/Discord +command menus). No engine, no model-tool footprint: every surface feeds +:func:`build_plan_prompt` to the agent as a normal turn, like ``/learn`` and +``/init``, so system prompt and history stay untouched (prompt-cache safe). """ from __future__ import annotations -# The plan-mode ground rules + authoring craft, distilled from the retired -# bundled skill (v2.0.0, writing-craft adapted from obra/superpowers). -# Embedded in the prompt so the agent plans the way a maintainer would. +# Plan-mode ground rules + authoring craft, distilled from the retired bundled +# skill (writing-craft adapted from obra/superpowers). _PLAN_MODE_RULES = """\ For this turn, you are in PLAN MODE — planning only. @@ -76,13 +61,7 @@ Interaction style: def build_plan_prompt(task: str = "") -> str: - """Build the plan-mode prompt for the live agent. - - Args: - task: What to plan. Empty → infer the task from the current - conversation context (mirrors the retired skill's behavior and - issue #36821's "plan from context" expectation). - """ + """Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context.""" task = (task or "").strip() if task: task_block = f"Task to plan:\n{task}\n" diff --git a/agent/prompt_cache_boundary.py b/agent/prompt_cache_boundary.py index 9f88277a88..1d751e61b0 100644 --- a/agent/prompt_cache_boundary.py +++ b/agent/prompt_cache_boundary.py @@ -1,52 +1,30 @@ -"""Builder-declared stable prefixes for Anthropic prompt caching (#81867). +"""Builder-declared stable prefixes for Anthropic prompt caching. -Skill, webhook, and cron builders concatenate a large static scaffold -(activation note + expanded skill body) with a small volatile invocation -tail (ticket payload, timestamps, run context) into one user-message -string. Only the builder knows the exact byte where the volatile tail -begins, so it registers the stable prefix here at construction time; the -cache planner consults the registry to place a cache breakpoint at that -boundary instead of caching the whole message as one atomic block. +Skill/webhook/cron builders concatenate a large static scaffold with a small +volatile invocation tail into one user-message string. Only the builder knows +where the tail begins, so it registers the stable prefix here and the cache +planner places a breakpoint at that boundary instead of caching the whole +message. Re-parsing marker strings out of the message at request time is +deliberately avoided: markers can legitimately appear inside skill bodies or +event payloads, and any delimiter heuristic then shrinks the cached prefix or +silently absorbs volatile bytes into it. -This deliberately avoids re-parsing scaffold marker strings out of the -message at request time: markers can legitimately appear inside skill -bodies or inside event payloads (e.g. a helpdesk ticket quoting an agent -transcript), and any delimiter-search heuristic then either shrinks the -cached prefix or — worse — silently absorbs volatile bytes into it, -reintroducing the per-invocation cache miss this exists to fix. - -The registry is process-local by design. A freshly fired webhook/cron -invocation is always built and sent by the same process, which is the -only window where the split pays off. Any miss (restart, eviction, -historic message) falls back to the pre-existing whole-message policy. - -Split-shape lifetime: the split is applied only while the skill message is -one of the plan's marked endpoints (the last few cacheable messages). Once -later turns rotate it out of that window it ships as a single string block -again, which changes the block boundary once and re-ingests the prefix from -that message onward exactly one time in a long-lived session. Webhook/cron -invocations — the workload this exists for — send the skill turn as the -newest message every time, so they always hit the split shape; the one-time -re-ingest only affects long interactive sessions and nets out far below the -per-invocation full rewrite this removes. +Process-local by design: a webhook/cron fire is built and sent by the same +process, and any miss (restart, eviction, historic message) falls back to the +whole-message policy. The split only applies while the message is one of the +plan's marked endpoints; once it rotates out it ships as one block again +(one-time re-ingest in long interactive sessions, never for webhook/cron). """ import threading from collections import OrderedDict from typing import Optional -# A couple dozen distinct active scaffolds (webhook routes x skills x cron -# jobs) is generous for one gateway process; beyond that, oldest entries -# fall back to whole-message caching rather than growing unboundedly. +# A couple dozen active scaffolds is generous for one gateway process. _MAX_ENTRIES = 32 - -# Entries hold whole expanded skill bodies, so an entry count alone does not -# bound memory — a handful of large skills can retain tens of MB in a -# long-lived gateway process. Evict by total retained characters too (a -# conservative proxy for bytes: actual memory is 1–4x depending on the -# string's widest code point), always keeping the newest entry so a single -# oversized scaffold still gets a boundary instead of silently disabling -# the split. +# Entries hold whole expanded skill bodies, so also bound total retained chars +# (1-4x bytes). The newest entry is always kept so one oversized scaffold still +# gets a boundary instead of silently disabling the split. _MAX_CHARS = 4 * 1024 * 1024 _lock = threading.Lock() @@ -67,29 +45,19 @@ def register_stable_prefix(prefix: str) -> None: def find_stable_prefix(content: str) -> Optional[str]: - """Longest registered prefix that is a *proper* prefix of ``content`` with non-whitespace tail. + """Longest registered *proper* prefix of ``content`` with a non-whitespace tail. - Proper with non-whitespace tail (``bool(content[len(prefix):].strip())``) so the - split never produces an empty or whitespace-only volatile text block, which - Anthropic rejects on the wire (HTTP 400). - - A hit refreshes the entry's LRU position: a scaffold fired every minute - by cron must not be evicted by a burst of one-off skill invocations, - which would silently drop it back to whole-message caching. + The tail must be non-whitespace so the split never yields an empty text + block (Anthropic rejects it with HTTP 400). A hit refreshes the entry's LRU + position so a scaffold fired every minute by cron is not evicted by a + burst of one-off skill invocations. """ with _lock: best: Optional[str] = None for prefix in _prefixes: - if content.startswith(prefix) and bool(content[len(prefix):].strip()): + if content.startswith(prefix) and content[len(prefix):].strip(): if best is None or len(prefix) > len(best): best = prefix if best is not None: - # After the scan so the OrderedDict is never mutated mid-iteration. - _prefixes.move_to_end(best) + _prefixes.move_to_end(best) # after the scan: never mutate mid-iteration return best - - -def clear_stable_prefixes() -> None: - """Test isolation helper.""" - with _lock: - _prefixes.clear() diff --git a/agent/prompt_cache_scope.py b/agent/prompt_cache_scope.py index f4b14b345e..03abf38c76 100644 --- a/agent/prompt_cache_scope.py +++ b/agent/prompt_cache_scope.py @@ -1,68 +1,30 @@ """Rotation-stable logical cache scope for prompt_cache_key derivation. -Context-compression rotation (legacy ``compression.in_place: false`` mode) -mints a new physical ``session_id`` mid-conversation to segment the -transcript. The prompt-cache scope introduced by #79161 was derived from that -physical id, so every rotation moved the conversation into a fresh cache -bucket even though it is logically the same conversation continuing -(issue #79017). +Legacy compression rotation (``compression.in_place: false``) mints a new +physical ``session_id`` mid-conversation, which moved the conversation into a +fresh cache bucket each time. ``resolve_prompt_cache_scope()`` instead maps +the physical id to the ROOT of its compression lineage via +``SessionDB.get_compression_lineage()`` — NOT ``get_conversation_root`` / +``_conversation_root_id`` (the Portal-attribution walk), which follows +``parent_session_id`` blindly and would collapse /branch children and delegate +trees into one id. The two resolvers are intentionally different. -``resolve_prompt_cache_scope()`` maps the physical session id to the ROOT of -its *compression lineage* — the pre-rotation session id — using -``SessionDB.get_compression_lineage()``, whose fork-aware semantics -(hardened in #79193) give exactly the scope boundaries the cache key needs. -NOT ``SessionDB.get_conversation_root`` / ``run_agent._conversation_root_id`` -(the Portal-attribution walk): that one follows ``parent_session_id`` blindly, -collapsing /branch children and whole delegate trees into one id, which would -violate the #79161 isolation this scope must preserve. The two resolvers are -intentionally different — do not "deduplicate" them. +Scope boundaries: rotation children walk back to the original segment; ``/new`` +starts a fresh scope; ``/branch`` children, delegate subagents, and tool-tagged +children are explicit fork children with their own isolated scope; cron fires +keep their physical id (the per-fire timestamp is stripped later). -- compression-rotation children walk back to the original segment - (rotation-stable scope — the fix); -- ``/new`` starts a lineage-less session (fresh scope); -- ``/branch`` children (``_branched_from``), delegate subagents - (``_delegate_from``), and tool-tagged children (``source="tool"``) are - explicit fork children and keep their own isolated scope, preserving the - sibling/subagent isolation #79161 established; -- cron fires keep their physical ``cron__`` id here — the per-fire - timestamp is stripped later by ``_cache_scope_from_session_id`` exactly as - before. +Hosts that mint one physical id per RESPONSE (Studio group chat, ``/v1/responses`` +with client-managed history) carry no lineage, so the walk returns the physical +id and the scope moves every reply. Hermes must not infer the conversation from +id SYNTAX (that collides client-supplied ids); the host declares it via +``gateway_session_key`` (``X-Hermes-Session-Key`` / ``build_session_key``), +consumed by ``declared_conversation_scope()``, which wins over the lineage walk. +The declared key is hashed to ``gwk_`` because it embeds +platform/chat/user identifiers and leaves the process as a provider routing key. -A host that mints one physical ``session_id`` per RESPONSE (Hermes Studio's -group chat, and ``POST /v1/responses`` with client-managed history, which -mints ``str(uuid4())`` per request) re-keys every conversation-affinity hint -Hermes sends — ``prompt_cache_key`` on both OpenAI-wire transports, plus the -OpenRouter/Nous sticky ``session_id`` and xAI's ``x-grok-conv-id`` through -``portal_tags`` (issue #96811). Those rows carry no lineage, so the walk -above correctly returns the physical id and the scope moves every reply. - -Hermes must not infer the logical conversation from the id's SYNTAX (that -rule collides independent client-supplied ids and merges Studio members -truncated past its 96-character boundary — the #79017 failure class). The -host has to declare it, and one carrier already means exactly that: -``gateway_session_key`` — the "stable per-chat key" (``agent:main:telegram: -dm:123``) built by ``gateway.session.build_session_key`` from the -``X-Hermes-Session-Key`` header, which branching deliberately does NOT key -off. ``declared_conversation_scope()`` consumes it, and it wins over the -lineage walk because it is stable across rotation AND across per-response -ids. Two boundaries it must not cross: - -- explicit fork children (``/branch``, delegate subagents, tool children) - share their parent's chat key but are separate conversations — the row's - fork markers keep them on their own scope (#79161); -- background-review forks run on a clone of the live runtime, so they are - excluded by ``_persist_disabled`` for the same reason. - -The declared key is hashed into ``gwk_`` before it becomes a -scope: unlike a session id it embeds platform/chat/user identifiers, and -this value leaves the process verbatim as OpenRouter's sticky ``session_id`` -and xAI's ``x-grok-conv-id``. - -The resolution is memoized per (agent, session_id): the lineage walk runs -once per transcript segment — NOT per API call — and re-runs only when -rotation actually changes ``agent.session_id`` (per the no-DB-on-the-hot-path -constraint recorded on #79017). Default installs compact in place and never -rotate, so they hit the memo forever and behave byte-identically to before. +Resolution is memoized per (agent, session_id, db-present): the lineage walk +runs once per transcript segment, never per API call. """ import hashlib @@ -72,16 +34,13 @@ from typing import Any, Optional logger = logging.getLogger(__name__) _MEMO_ATTR = "_prompt_cache_scope_memo" -# Namespace for a scope resolved from a host-declared conversation key. _DECLARED_SCOPE_PREFIX = "gwk_" def _lineage_root(session_id: str, session_db: Any) -> Optional[str]: - """Return the compression-lineage root of *session_id*, or None. + """Compression-lineage root of *session_id*, or None. - Defensive about the DB handle: test doubles and partially constructed - agents can hand back non-list results — anything that is not a non-empty - list/tuple whose first element is a non-empty string is ignored. + Tolerates non-list results from test doubles / partially built agents. """ if session_db is None: return None @@ -102,24 +61,13 @@ def _agent_source( ) -> str: """The ``sessions.source`` this agent's conversation is recorded under. - Read from the agent's own row when it exists, because that is the value - the peer queries below match on. - - ``row_source`` is that value when the caller already has it — the single - identity read in :func:`declared_conversation_scope` — where ``""`` means - "the row was read and carries no source". ``None`` means "not read yet" - and keeps the original lookup, which is the path a ``SessionDB`` without - :meth:`~hermes_state.SessionDB.declared_scope_identity` still takes. - - Before the row lands — this module resolves the first scope ahead of - ``_ensure_db_session`` — it uses the SAME resolver persistence will use, - ``run_agent._session_source_for_agent``, not ``agent.platform``. The two - diverge whenever ``HERMES_SESSION_SOURCE`` overrides the platform, and the - divergence is not a cosmetic one: the declared scope is non-``None`` - immediately, so ``resolve_prompt_cache_scope`` memoizes it for this session - id and never re-resolves once the authoritative row appears. Both sides of - a ``/new`` would then read the platform domain, miss the boundary recorded - under the override, and hash the same scope. + ``row_source`` is the row's value when the caller already read it (``""`` + = read, no source; ``None`` = not read yet, do the lookup). Before the row + lands, use the SAME resolver persistence uses + (``run_agent._session_source_for_agent``), not ``agent.platform``: they + diverge under ``HERMES_SESSION_SOURCE``, and the declared scope is memoized + immediately, so both sides of a ``/new`` would otherwise miss the boundary + recorded under the override and hash the same scope. """ if row_source is None and session_id and session_db is not None: try: @@ -132,8 +80,7 @@ def _agent_source( return row_source platform = getattr(agent, "platform", None) try: - # Imported lazily: run_agent imports this module, and this is the - # single owner of the source a session row is created with. + # Lazy: run_agent imports this module. from run_agent import _session_source_for_agent source = str(_session_source_for_agent(platform) or "").strip() @@ -145,23 +92,14 @@ def _agent_source( def _conversation_generation(session_key: str, source: str, session_db: Any) -> str: - """Return the durable generation for *session_key*'s current conversation. + """Durable generation for *session_key*'s current conversation (``""`` if none). - The declared key names a chat and deliberately survives `/new` and policy - resets. Hashing it alone would therefore reuse one affinity scope across - distinct conversations, violating the #79017/#86733 contract: warm across - compression, cold across a conversation boundary. - - ``SessionDB.latest_conversation_boundary`` reads the monotonic - ``conversation_generations`` counter for ``(source, session_key)``. The - counter advances in the same transaction that records an - ``_RESET_END_REASONS`` boundary. It is independent of prunable session rows - and wall-clock time, so deletion, bulk pruning, and clock rollback cannot - reissue an old generation. Compression continues the current conversation - and does not advance it. - - This lookup runs on the memoized resolution path, not once per API call. - Return ``""`` when the key has never reset or the DB exposes no generation. + The declared key names a chat and survives ``/new`` and policy resets, so + hashing it alone would reuse one scope across distinct conversations. The + ``conversation_generations`` counter advances in the same transaction that + records a reset boundary and is independent of prunable rows and + wall-clock, so pruning or clock rollback cannot reissue a generation. + Compression does not advance it. """ reader = getattr(session_db, "latest_conversation_boundary", None) if not callable(reader): @@ -173,30 +111,17 @@ def _conversation_generation(session_key: str, source: str, session_db: Any) -> def declared_conversation_scope(agent: Any) -> Optional[str]: - """Return the host-declared logical conversation scope, or None. + """Host-declared logical conversation scope (``gwk_``), or None. - Resolved from ``agent._gateway_session_key`` (the ``X-Hermes-Session-Key`` - /``build_session_key`` per-chat key) qualified by the conversation - generation currently live on it (:func:`_conversation_generation`), hashed - together into ``gwk_`` so no platform/chat/user identifier - reaches a provider on the wire and the value stays inside every caller's - length/charset budget. - - The key alone would outlive the conversation — it survives ``/new`` and the - idle/daily policy resets by design — so the generation is what makes this - carrier legal: stable across a host's per-response physical ids, and cold - on every conversation replacement. - - None — meaning "fall back to the physical-id scope" — when no key was - declared, when this agent is a background-review fork (``_persist_disabled``: - it clones the live runtime, including the key), when the session row is an - explicit fork child (``/branch``, delegate, tool), and on any DB error - during either lookup. + Hashes ``(source, gateway_session_key, generation)`` so no platform/chat/ + user identifier reaches a provider. None — fall back to the physical-id + scope — when no key is declared, when the agent is a background-review + fork (``_persist_disabled`` clones the live runtime incl. the key), when + the row is an explicit fork child, and on any DB error (fail closed rather + than merge a fork onto its parent's key). """ key = str(getattr(agent, "_gateway_session_key", "") or "").strip() - if not key: - return None - if getattr(agent, "_persist_disabled", False): + if not key or getattr(agent, "_persist_disabled", False): return None sid = str(getattr(agent, "session_id", None) or "") db = getattr(agent, "_session_db", None) @@ -204,13 +129,9 @@ def declared_conversation_scope(agent: Any) -> Optional[str]: row_source: Optional[str] = None if sid and db is not None: try: - # One read for both halves of the row's identity: the fork verdict - # and the source the peer queries match on live on the same - # ``sessions`` row, and asking for them separately read it twice - # per resolution (@teknium1 on #98811). A SessionDB without the - # combined view keeps the original call, so nothing that predates - # it — including the doubles that certify the fail-closed contract - # below — changes behaviour. + # One read for both halves of the row identity (fork verdict + + # source). A SessionDB without the combined view keeps the + # original call. identity = getattr(db, "declared_scope_identity", None) if callable(identity): is_fork, row_source = identity(sid) @@ -219,11 +140,6 @@ def declared_conversation_scope(agent: Any) -> Optional[str]: if is_fork: return None except Exception: - # Degrade to the physical-id scope rather than risk merging a - # fork onto its parent's key on a transient DB failure. The - # source read is inside this same guard for the same reason: it - # was always the second half of a read that had already failed - # closed here. logger.debug("declared-scope fork check failed", exc_info=True) return None source = _agent_source(agent, sid, db, row_source) @@ -231,65 +147,46 @@ def declared_conversation_scope(agent: Any) -> Optional[str]: try: generation = _conversation_generation(key, source, db) except Exception: - # Same fail-closed rule as the fork check: an unqualified key - # spans /new, so degrade to the physical-id scope instead. logger.debug("declared-scope generation read failed", exc_info=True) return None - # The carrier is the SAME identity tuple the peer queries use: two hosts - # may legally declare the same key string under different sources, and the - # scope leaves this process as a routing key, so it must not collapse them. + # Same identity tuple the peer queries use: two hosts may declare the + # same key under different sources and must not collapse. carrier = f"{source}|{key}|{generation}" digest = hashlib.sha256(carrier.encode("utf-8", errors="replace")).hexdigest()[:24] return f"{_DECLARED_SCOPE_PREFIX}{digest}" def resolve_prompt_cache_scope(agent: Any) -> str: - """Resolve the rotation-stable cache-scope id for *agent*'s conversation. + """Rotation-stable cache-scope id for *agent*'s conversation. - Returns the host-declared conversation scope when one applies - (``declared_conversation_scope``), else the compression-lineage ROOT of - ``agent.session_id`` (the physical id itself when the session has no - compression ancestry, no DB is attached, or the walk fails). The result is memoized on the agent - keyed by the current session id, so the DB walk happens once per - transcript segment rather than once per API call. + Declared scope when one applies, else the compression-lineage root of + ``agent.session_id`` (the physical id when there is no ancestry, no DB, or + the walk fails). Memoized on the agent keyed by session id. """ sid = str(getattr(agent, "session_id", None) or "") if not sid: return "" db = getattr(agent, "_session_db", None) - # Memo key includes DB presence: an agent that starts DB-less and gains a - # handle later (run_agent._get_session_db_for_recall lazily attaches one) + # DB presence is part of the key: an agent that gains a DB handle later # must re-resolve instead of staying pinned to the physical id. key = (sid, db is not None) memo = getattr(agent, _MEMO_ATTR, None) if isinstance(memo, tuple) and len(memo) == 2 and memo[0] == key: return memo[1] - # A declared conversation key outranks the lineage walk: it is stable - # across compression rotation AND across a host's per-response ids, which - # the walk cannot see (#96811). root = declared_conversation_scope(agent) or ( _lineage_root(sid, db) if db is not None else None ) scope = root or sid - # Memoize on a successful walk, or when there is no DB to consult at all, - # or when the agent will never persist a row (background-review forks set - # _persist_disabled but still hold a DB handle — without this, every API - # call would re-run the lineage query forever). - # A failed/empty walk on a persisting agent is NOT memoized: falling back - # to the physical id is the correct degraded answer right now (row not - # persisted yet, transient DB error), but pinning it for the whole segment - # would keep the scope wrong after the session row lands. - if ( - root is not None - or db is None - or getattr(agent, "_persist_disabled", False) - ): + # Memoize on success, with no DB, or when the agent never persists a row + # (background-review forks hold a DB handle but set _persist_disabled). + # A failed/empty walk on a persisting agent is NOT memoized: the physical + # id is right for now (row not yet persisted, transient error) but would + # stay wrong for the whole segment once the row lands. + if root is not None or db is None or getattr(agent, "_persist_disabled", False): try: setattr(agent, _MEMO_ATTR, (key, scope)) except Exception: - # Frozen/slotted test doubles — resolution still works, just - # unmemoized. - pass + pass # frozen/slotted doubles: resolution works, just unmemoized return scope @@ -303,14 +200,11 @@ def declared_conversation_scope_safe(agent: Any) -> Optional[str]: def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]: - """Never-raising variant of :func:`resolve_prompt_cache_scope`. + """Never-raising variant of :func:`resolve_prompt_cache_scope` (None on failure/empty). - Returns None on any failure (or when there is no scope). Consumers treat - None/empty as "fall back to the physical session_id", so a resolution - failure degrades to pre-#79017 behavior instead of blocking the caller — - important at turn_context's call site, where an exception raised inside - the ``set_runtime_main(...)`` argument list would otherwise skip the whole - runtime binding, not just the cache scope. + Consumers treat None as "use the physical session_id"; at turn_context's + call site an exception inside the ``set_runtime_main(...)`` argument list + would skip the whole runtime binding, not just the cache scope. """ try: return resolve_prompt_cache_scope(agent) or None diff --git a/agent/prompt_caching.py b/agent/prompt_caching.py index d3f4cdca3a..df857bb9eb 100644 --- a/agent/prompt_caching.py +++ b/agent/prompt_caching.py @@ -1,13 +1,10 @@ -"""Anthropic prompt caching strategy. +"""Anthropic prompt caching strategy — pure functions, no AIAgent dependency. -The default layout uses 4 cache_control breakpoints: the static system -prefix, the end of the system prompt, and the last 2 non-system messages. -When a static system prefix is unavailable, it falls back to one system -breakpoint plus the last 3 messages. All markers use the same TTL (5m or 1h). -This preserves intra-session caching while allowing new sessions to reuse the -stable system-prompt prefix. - -Pure functions -- no class state, no AIAgent dependency. +Default layout: 4 cache_control breakpoints — the static system prefix, the end +of the system prompt, and the last 2 non-system messages. Without a static +prefix: one system breakpoint plus the last 3 messages. All markers share one +TTL (5m or 1h). This keeps intra-session caching while letting new sessions +reuse the stable system-prompt prefix. """ import copy @@ -24,30 +21,18 @@ class PromptCachePlan: messages: List[Dict[str, Any]] tools: List[Dict[str, Any]] - @property - def marker_count(self) -> int: - """Wire-visible cache markers in this plan (computed on demand). - - Only tests consume this; keeping it lazy avoids walking every - message part and tool schema on the per-request hot path. - """ - return _count_cache_markers(self.messages, self.tools) - def envelope_tool_part_cache_markers_supported( provider: str | None, base_url: str | None ) -> bool: """Whether the envelope-layout route honors part-level markers on role:tool. - OpenRouter (and Nous Portal, which proxies to it) relocate a - ``cache_control`` sitting on a tool message's content part onto the - ``tool_result`` block during their OpenAI→Anthropic translation, so the - marker is honored there. LiteLLM-style OpenAI-wire proxies instead map - content parts verbatim: the part-level marker lands at - ``tool_result.content[0]``, which the Anthropic Messages schema forbids — - a non-retryable HTTP 400 that kills the whole turn (#89886). On those - routes tool messages must not carry part-level markers at all; the - breakpoint budget reallocates to the nearest eligible message instead. + OpenRouter (and Nous Portal, which proxies to it) relocate a part-level + ``cache_control`` onto the ``tool_result`` block during OpenAI→Anthropic + translation. LiteLLM-style proxies copy parts verbatim, so the marker lands + at ``tool_result.content[0]`` — forbidden by the Anthropic schema, a + non-retryable 400. On those routes tool messages carry no part markers and + the breakpoint budget reallocates to the nearest eligible message. """ from agent.agent_runtime_helpers import _is_litellm_route @@ -65,27 +50,19 @@ def _apply_cache_marker( content = msg.get("content") if role == "tool" and native_anthropic: - # Native Anthropic layout: top-level marker; the adapter moves it - # inside the tool_result block. + # Top-level marker; the native adapter moves it inside tool_result. msg["cache_control"] = cache_marker return - if role == "tool" and not tool_part_markers: - # Envelope route whose OpenAI→Anthropic translation copies content - # parts verbatim (LiteLLM et al.): a part-level marker becomes - # tool_result.content[0].cache_control → non-retryable 400 (#89886). + # LiteLLM-style envelope: a part marker becomes + # tool_result.content[0].cache_control → non-retryable 400. return if content is None or content == "": - if role == "tool" and not native_anthropic: - # OpenRouter rejects top-level cache_control on role:tool (silent - # hang) and an empty message has no content part to carry the - # marker — skip. Non-empty tool content falls through below and - # gets the marker on a content part, which OpenRouter honors. - return - if role == "assistant" and not native_anthropic: - # Empty assistant turns are pure tool_calls. A top-level marker - # here is ignored on the envelope layout, so skip. + # Envelope layout: OpenRouter rejects top-level cache_control on + # role:tool (silent hang), and ignores it on empty assistant turns + # (pure tool_calls) — neither has a content part to carry it. + if role in ("tool", "assistant") and not native_anthropic: return msg["cache_control"] = cache_marker return @@ -96,17 +73,13 @@ def _apply_cache_marker( if stable_prefix is not None: suffix = content[len(stable_prefix):] if suffix.strip(): - # Builder-declared boundary (#81867): the scaffold carries the - # breakpoint, the volatile invocation tail rides unmarked so a - # changed ticket ID or timestamp no longer invalidates the - # whole skill body. Request-local only — the canonical session - # message stays a plain string. + # Builder-declared boundary: the scaffold carries the + # breakpoint and the volatile tail rides unmarked, so a + # changed ticket ID/timestamp no longer invalidates the + # skill body. Request-local only — the stored message + # stays a plain string. msg["content"] = [ - { - "type": "text", - "text": stable_prefix, - "cache_control": cache_marker, - }, + {"type": "text", "text": stable_prefix, "cache_control": cache_marker}, {"type": "text", "text": suffix}, ] return @@ -126,17 +99,12 @@ def _can_carry_marker( ) -> bool: """True if a marker on this message is actually honored by the provider. - On the native Anthropic layout every message works (top-level markers are - relocated by the adapter). On the envelope layout (OpenRouter et al.) only - markers inside content parts are honored: empty-content messages (e.g. - assistant turns that are pure tool_calls) and empty tool messages would - receive a top-level marker the provider ignores — wasting one of the four - breakpoints. Skip those so the breakpoints land on messages that count. - - ``tool_part_markers=False`` (LiteLLM-style envelope routes, #89886) - additionally excludes ALL role:tool messages: their part-level marker - would be forwarded verbatim into ``tool_result.content[]`` and rejected - with a non-retryable 400, so the breakpoint must reallocate instead. + Native Anthropic honors every message (the adapter relocates top-level + markers). The envelope layout only honors markers inside content parts, so + empty-content messages would waste one of the four breakpoints; with + ``tool_part_markers=False`` (LiteLLM-style routes) every role:tool message + is excluded too, since its part marker would be rejected with a 400. + Must agree with :func:`_apply_cache_marker`, which marks only the LAST part. """ if native_anthropic: return True @@ -146,10 +114,6 @@ def _can_carry_marker( if content is None or content == "": return False if isinstance(content, list): - # _apply_cache_marker only marks the LAST content part, so the carrier - # predicate must agree: a list whose last element isn't a dict cannot - # actually receive a marker and would waste a breakpoint. Mirror the - # `content` truthiness + last-element-dict check in _apply_cache_marker. return bool(content) and isinstance(content[-1], dict) return isinstance(content, str) @@ -162,66 +126,31 @@ def _build_marker(ttl: str) -> Dict[str, str]: return marker -# Alibaba-family providers (Qwen routes). Their context cache documents a -# five-minute window (renewed on hit) and rejects the Anthropic 1h tier. -# Shared with agent_runtime_helpers.anthropic_prompt_cache_policy so the -# cache-policy opt-in and the TTL clamp can never desync (#84733). +# Alibaba-family providers (Qwen routes): documented five-minute context cache, +# Anthropic 1h tier rejected. Shared with +# agent_runtime_helpers.anthropic_prompt_cache_policy so the cache-policy +# opt-in and the TTL clamp never desync. Do NOT narrow this set to extend a +# TTL — it also drives the marker-layout opt-in, so narrowing DISABLES caching. ALIBABA_FAMILY_PROVIDERS = frozenset({ "opencode", - "opencode-zen", "opencode-go", + "opencode-zen", "alibaba", }) - -# --- 1h-tier membership: an ALLOW-list, deliberately minimal ---------------- -# -# #84733 clamped 1h -> 5m for the whole alibaba/opencode family, reasoning from -# Alibaba's PUBLISHED Qwen docs. Wire measurement on the opencode-go route -# contradicts the docs. Controlled run: identical request, only the ttl flag -# varying, read back after 11 minutes with no intervening call (a read renews -# the window and would mask expiry): -# -# qwen3.8-max ttl=1h -> cache_read 2122 SURVIVED -# qwen3.8-max ttl=- -> cache_read 0 EXPIRED <- control -# glm-5.2 ttl=1h -> cache_read 2092 SURVIVED -# minimax-m2.5 ttl=1h -> cache_read 0 EXPIRED -# -# Read the two non-qwen rows for what they are: evidence about the ROUTE, not -# about traffic Hermes sends today. anthropic_prompt_cache_policy currently -# opts opencode-go in only for qwen models, so glm-5.2 and minimax-m2.5 on -# that route receive no cache_control marker at all and never reach this -# clamp in production. They constrain the route-level rule; they are not -# live paths. -# -# Only opencode-go is listed: it is the only route measured. Other opencode -# routes stay clamped because they were NOT measured, not because they are -# known bad. opencode-zen returns cache_creation.ephemeral_1h_input_tokens for -# Claude models, so it is a candidate -- but qwen on zen is unmeasured, so -# adding the provider wholesale would outrun the evidence. -# -# WARNING: opencode-go labels EVERY write `ephemeral_5m_input_tokens` whatever -# ttl was requested. That label is NOT evidence of the retention window -- it -# is what made the original docs-based reasoning look confirmed. Verify only -# with a delayed read past 5 minutes and no intervening call. -# -# NOTE: kept separate from ALIBABA_FAMILY_PROVIDERS on purpose. That set also -# drives the cache-marker-layout OPT-IN in -# agent_runtime_helpers.anthropic_prompt_cache_policy; narrowing it would -# silently DISABLE caching for qwen on opencode-go rather than extend its TTL. +# 1h-tier ALLOW-list: only routes wire-measured to retain a 1h marker (delayed +# read past 5 minutes with no intervening call — an intervening read renews the +# window and masks expiry). Other opencode routes stay clamped because they are +# UNMEASURED, not known-bad. Note opencode-go labels every write +# `ephemeral_5m_input_tokens` regardless of requested ttl; that label is not +# evidence of the retention window. MEASURED_1H_PROVIDERS = frozenset({ "opencode-go", }) -# Models measured to ignore the 1h tier even on a 1h-capable route. -# -# SCOPE: consulted only for providers already in MEASURED_1H_PROVIDERS. The -# measurement was taken on the opencode-go route, so it says nothing about the -# same model reached some other way -- and MiniMax on its own -# Anthropic-compatible endpoint IS a separate, cache-eligible route -# (anthropic_prompt_cache_policy opts it in by provider id / host match). -# Checking this set globally would have silently regressed that unrelated -# route's configured 1h to 5m off the back of an opencode-go observation. +# Models measured to ignore the 1h tier on a MEASURED_1H_PROVIDERS route. +# Consulted only there: the same model on its own Anthropic-compatible endpoint +# is a separate cache-eligible route and must not inherit this clamp. NO_1H_TIER_MODELS = frozenset({ "minimax-m2.5", }) @@ -235,9 +164,8 @@ def _flat_model(model: str) -> str: def is_qwen_model(model: str) -> bool: """True when ``model`` names a Qwen-family model (case-insensitive). - Shared by the TTL clamp below and - ``agent_runtime_helpers.anthropic_prompt_cache_policy`` so the - cache-policy opt-in and the clamp can never desync (#84733). + Shared with ``agent_runtime_helpers.anthropic_prompt_cache_policy`` so the + cache-policy opt-in and the TTL clamp never desync. """ return "qwen" in (model or "").lower() @@ -250,30 +178,18 @@ def effective_cache_ttl( ) -> str: """Clamp a requested cache TTL to what the destination route supports. - Qwen/Alibaba context caching documents an explicit five-minute window - (renewed on hit); the Anthropic ``1h`` tier is ignored/rejected there, - so a configured ``1h`` regresses to ``5m`` instead of shipping a marker - the provider drops and creating a false 1h-cache expectation (#84733). - Exception: routes in ``MEASURED_1H_PROVIDERS`` were wire-measured to - honour the tier (delayed read past 5 minutes) and keep ``1h`` — minus - any model in ``NO_1H_TIER_MODELS`` measured to ignore it on that route. - All other caching routes keep the requested TTL. - - ``None`` (caching active with no explicit tier) resolves to ``5m``. + Qwen/Alibaba routes document a five-minute window and drop the ``1h`` + tier, so a configured ``1h`` regresses to ``5m`` there instead of creating + a false 1h-cache expectation — except on ``MEASURED_1H_PROVIDERS``, which + keep ``1h`` minus any ``NO_1H_TIER_MODELS`` model. The measured-route check + runs BEFORE the generic Qwen clamp, which would otherwise swallow every + Qwen model on it. ``None`` resolves to ``5m``. """ if ttl != "1h": return ttl or "5m" if (provider or "").lower() in MEASURED_1H_PROVIDERS: - # Route measured to honour the tier -- checked BEFORE the generic - # is_qwen_model clamp below, which would otherwise swallow every Qwen - # model on it. Within the route, a model measured to ignore the tier - # still wins; the denial stays nested here so an opencode-go - # observation cannot leak out and reclamp the same model on an - # unrelated route. return "5m" if _flat_model(model) in NO_1H_TIER_MODELS else "1h" - if is_qwen_model(model): - return "5m" - if (provider or "").lower() in ALIBABA_FAMILY_PROVIDERS: + if is_qwen_model(model) or (provider or "").lower() in ALIBABA_FAMILY_PROVIDERS: return "5m" return "1h" @@ -289,21 +205,13 @@ def _apply_system_cache_markers( ) -> int: """Mark the static system prefix (and optionally the full prompt). - The system prompt remains one stored string. Splitting it only in the - outgoing request keeps session persistence and non-Anthropic transports - unchanged while making the stable prefix independently cacheable. - - ``mark_suffix=False`` is the tool-cache-plan layout: only the static - prefix carries a marker, the volatile suffix rides unmarked (its - breakpoint budget is spent on the tools array instead). - - ``fallback_to_whole=False`` skips marking entirely when the prefix - split is not possible (no prefix, mismatched prefix, non-string - content) instead of marking the whole message. - - When the prompt IS exactly the static prefix (empty suffix), the whole - message is marked as a single block — never a two-part split with an - empty text block, which Anthropic rejects. + The system prompt stays one stored string; it is split only in the + outgoing request so persistence and non-Anthropic transports are + unchanged. ``mark_suffix=False`` is the tool-cache-plan layout (suffix + unmarked, its budget spent on the tools array). ``fallback_to_whole=False`` + marks nothing when the prefix split is impossible. When the prompt IS the + prefix (empty/whitespace suffix) the whole message is marked as one block — + never a split with an empty text block, which Anthropic rejects. Returns the number of markers applied (0, 1, or 2). """ @@ -320,17 +228,10 @@ def _apply_system_cache_markers( if mark_suffix: suffix_part["cache_control"] = cache_marker message["content"] = [ - { - "type": "text", - "text": static_system_prefix, - "cache_control": cache_marker, - }, + {"type": "text", "text": static_system_prefix, "cache_control": cache_marker}, suffix_part, ] return 2 if mark_suffix else 1 - # Empty/whitespace-only suffix: the stored prompt IS the static prefix. Mark it as - # one whole block — a [marked-prefix, ""] split would put an empty - # text block on the wire (HTTP 400 on native Anthropic). _apply_cache_marker(message, cache_marker, native_anthropic=native_anthropic) return 1 @@ -345,27 +246,17 @@ def strip_anthropic_cache_control( ) -> List[Dict[str, Any]]: """Remove ``cache_control`` markers and undo decoration-produced list shapes. - Used before re-applying decoration after a mid-turn provider failover so - the mutated, undecorated shape (image shrink / ASCII cleanup / etc.) is - preserved while markers match the *new* provider's cache policy (#72626). + Used before re-decorating after a mid-turn provider failover, so the + mutated undecorated shape is preserved while markers match the new + provider's policy. Flattening back to a plain string is restricted to the + exact shapes :func:`apply_anthropic_cache_control` produces from string + content — a single text part, the two-part ``[static, volatile]`` system + split, or the two-part skill split — so the ``""``-join is provably + byte-exact; organic multi-part text and parts with extra keys keep their + structure. Marker removal is copy-on-write on part dicts: parts can alias + caller-held lists and stripping must never rewrite the stored transcript. - Flattening back to a plain string is restricted to the exact shapes - :func:`apply_anthropic_cache_control` produces from string content — - a single ``{"type": "text"}`` part, the two-part ``[static, volatile]`` - system split, or the two-part builder-declared skill split (recognised - by its marker-on-the-first-part shape, so flattening never depends on - the prefix registry still holding the entry) — so the ``""``-join is - provably byte-exact. Organic - multi-part text (merged user turns, imported transcripts) and parts - carrying extra keys (``citations`` etc.) keep their structure; only - per-part markers are removed. Marker removal is copy-on-write on the - part dicts: content parts can alias caller-held message lists (the main - send path now hands structurally-cloned copies via - _clone_message_for_send, but other callers may pass shallow copies), - and stripping must never rewrite the stored transcript. - - Mutates the top-level message dicts of ``api_messages`` in place and - returns the same list. + Mutates the top-level message dicts in place and returns the same list. """ for msg in api_messages: if not isinstance(msg, dict): @@ -374,13 +265,11 @@ def strip_anthropic_cache_control( content = msg.get("content") if not isinstance(content, list): continue - # Two-part skill-invocation split (#81867). The builder-declared - # boundary is the only decoration that marks the *first* part of a - # user message: list content otherwise receives its marker on the - # last part, and the two-part [static, volatile] split is role-gated - # to system. So the shape alone identifies it, and flattening stays - # correct even when the prefix registry has since evicted the entry - # (failover re-decorates a request built many messages ago, #72626). + # The builder-declared skill split is the only decoration that marks + # the FIRST part of a user message (list content is otherwise marked + # on the last part; the [static, volatile] split is system-only), so + # the shape alone identifies it even after the prefix registry has + # evicted the entry. skill_split_shape = ( msg.get("role") == "user" and len(content) == 2 @@ -505,9 +394,9 @@ def build_prompt_cache_plan( ) -> PromptCachePlan: """Build isolated cache sections for one resolved request destination. - ``tool_part_markers=False`` (LiteLLM-style envelope routes, #89886) - keeps ``cache_control`` off role:tool content parts; breakpoints - reallocate to the nearest eligible non-tool message. + ``tool_part_markers=False`` (LiteLLM-style envelope routes) keeps + ``cache_control`` off role:tool content parts; breakpoints reallocate to + the nearest eligible non-tool message. """ messages = copy.deepcopy(api_messages or []) strip_anthropic_cache_control(messages) @@ -558,23 +447,13 @@ def apply_anthropic_cache_control( ) -> List[Dict[str, Any]]: """Apply Anthropic cache-control markers to API messages. - When ``static_system_prefix`` exactly matches the beginning of a string - system prompt, it receives an early marker and the full system prompt gets - a trailing marker. The remaining two markers target the latest cacheable - non-system messages. Without that prefix, the legacy system-and-3 layout - is retained. - - Idempotent: pre-existing ``cache_control`` markers are stripped from a - per-message copy before new ones are placed, so calling this twice (or - handing it messages a prior call already marked) can never accumulate - past 4 markers. Only messages that already carry a marker pay the copy - cost — a shallow top-level copy suffices because - :func:`strip_anthropic_cache_control` is copy-on-write on content parts — - and the rest of the copy-on-write contract is unchanged (#90971). - - ``tool_part_markers=False`` (LiteLLM-style envelope routes, #89886) - keeps markers off role:tool messages entirely; the breakpoint budget - reallocates to the nearest eligible non-tool message. + With a matching ``static_system_prefix`` the prefix gets an early marker + and the full system prompt a trailing one; the remaining two markers go to + the latest cacheable non-system messages. Without it, the legacy + system-and-3 layout applies. Idempotent: pre-existing markers are stripped + from a per-message copy first, so repeated calls never accumulate past 4 + markers; a shallow top-level copy suffices because + :func:`strip_anthropic_cache_control` is copy-on-write on content parts. Returns: Shallow copy of message list with selective deep copies of modified messages. @@ -594,9 +473,6 @@ def apply_anthropic_cache_control( and any(isinstance(part, dict) and "cache_control" in part for part in content) ) if has_marker: - # Shallow top-level copy is enough: strip pops the top-level key - # and rebuilds content lists/part dicts copy-on-write, so the - # caller's message (and any aliased parts) are never mutated. messages[i] = strip_anthropic_cache_control([dict(msg)])[0] breakpoints_used = 0 diff --git a/agent/redact.py b/agent/redact.py index f192b9e721..99297e1a0b 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -1,10 +1,8 @@ """Regex-based secret redaction for logs and tool output. -Applies pattern matching to mask API keys, tokens, and credentials -before they reach log files, verbose output, or gateway logs. - -Short tokens (< 18 chars) are fully masked. Longer tokens preserve -the first 6 and last 4 characters for debuggability. +Masks API keys, tokens, and credentials before they reach log files, verbose +output, or gateway logs. Short tokens (< 18 chars) are fully masked; longer +tokens keep the first 6 and last 4 characters for debuggability. """ import logging @@ -14,18 +12,15 @@ import shlex import threading from urllib.parse import unquote_plus -# Basenames treated as ``.env`` files by _command_reads_env_file. Imported -# from agent/file_safety (the read-block list) so the two defenses can't -# drift: if file_tools blocks a read and the agent falls back to ``cat``, -# the terminal redactor still catches it. file_safety matches -# case-insensitively (``resolved.name.lower()``); the lookup mirrors that. +# Shared with agent/file_safety's read-block list so the two defenses can't +# drift: if file_tools blocks a read and the agent falls back to ``cat``, the +# terminal redactor still catches it. Both compare ``basename.lower()``. from agent.file_safety import _BLOCKED_PROJECT_ENV_BASENAMES as _ENV_FILE_BASENAMES logger = logging.getLogger(__name__) -# Sensitive query-string parameter names (case-insensitive exact match). -# Ported from nearai/ironclaw#2529 — catches tokens whose values don't match -# any known vendor prefix regex (e.g. opaque tokens, short OAuth codes). +# Sensitive query-string parameter names (case-insensitive exact match). Catches +# tokens whose values match no vendor prefix regex (opaque tokens, OAuth codes). _SENSITIVE_QUERY_PARAMS = frozenset({ "access_token", "refresh_token", @@ -45,38 +40,15 @@ _SENSITIVE_QUERY_PARAMS = frozenset({ "x-amz-signature", }) -# Sensitive form-urlencoded / JSON body key names (case-insensitive exact match). -# Exact match, NOT substring — "token_count" and "session_id" must NOT match. -# Ported from nearai/ironclaw#2529. -_SENSITIVE_BODY_KEYS = frozenset({ - "access_token", - "refresh_token", - "id_token", - "token", - "api_key", - "apikey", - "client_secret", - "password", - "auth", - "jwt", - "secret", - "private_key", - "authorization", - "key", -}) - -# Snapshot at import time so runtime env mutations (e.g. LLM-generated -# `export HERMES_REDACT_SECRETS=false`) cannot disable redaction -# mid-session. ON by default — secure default per issue #17691. Users who -# need raw credential values in tool output (e.g. working on the redactor -# itself) can opt out via `security.redact_secrets: false` in config.yaml -# (bridged to this env var in hermes_cli/main.py, gateway/run.py, and -# cli.py) or `HERMES_REDACT_SECRETS=false` in ~/.hermes/.env. An opt-out -# warning is logged at gateway and CLI startup so operators see the -# downgrade — see `_log_redaction_status()` in gateway/run.py and cli.py. +# Snapshot at import time so runtime env mutations (e.g. an LLM-generated +# `export HERMES_REDACT_SECRETS=false`) cannot disable redaction mid-session. +# ON by default; opt out via `security.redact_secrets: false` (bridged to this +# env var by hermes_cli/main.py, gateway/run.py, cli.py — which log a warning). _REDACT_ENABLED = os.getenv("HERMES_REDACT_SECRETS", "true").lower() in {"1", "true", "yes", "on"} -# Known API key prefixes -- match the prefix + contiguous token chars +# Known API key prefixes -- match the prefix + contiguous token chars. +# Every pattern MUST start with a literal prefix: _PREFIX_SUBSTRINGS (the cheap +# pre-screen gate) is derived from these literals and must stay false-negative-free. _PREFIX_PATTERNS = [ r"sk-[A-Za-z0-9_-]{10,}", # OpenAI / OpenRouter / Anthropic (sk-ant-*) r"ghp_[A-Za-z0-9]{10,}", # GitHub PAT (classic) @@ -119,9 +91,7 @@ _PREFIX_PATTERNS = [ r"fw-[A-Za-z0-9]{30,}", # Fireworks AI API key r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key - # GitLab token families (each pattern keeps a full literal prefix so the - # _PREFIX_SUBSTRINGS pre-screen stays false-negative-free). Ported from - # openclaw/openclaw#112954; follow-up invited in #4541. + # GitLab token families (each keeps a full literal prefix for the pre-screen). r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token @@ -139,77 +109,53 @@ _PREFIX_PATTERNS = [ r"pk-lf-[A-Za-z0-9\-]{8,}", # Langfuse public key (sk-lf- already covered by sk- pattern) ] -# ENV assignment patterns: KEY=value where KEY contains a secret-like name. -# Uppercase keys tolerate spaces around "=" (e.g. ``FOO_SECRET = bar``) because -# an all-caps key is almost never prose/code. -# Bare ``KEY`` / ``PASS`` / ``PW`` suffixes are included (``FAL_KEY=…``, -# ``MYSQL_PASS=…``, ``DB_PW=…``) — issue #77484. The regex is IGNORECASE so -# lowercase env names (``openai_key=…``) are caught here too. The secret name -# must sit at a word boundary (``_``-delimited or whole-word) so generic -# prose words (``password=``, ``token=``, ``KEYBOARD=``, ``PASSAGE=``) do not -# match — those are handled by the config/form/URL paths, and a bare -# ``password=…`` in a form body must not be swallowed greedily by ``\S+``. +# ENV assignment: KEY=value where KEY carries a secret-like name. +# Uppercase keys tolerate spaces around "=" and allow the keyword embedded +# anywhere (``MYTOKEN=…``) — an all-caps key is almost never prose. Bare +# ``KEY``/``PASS``/``PW`` suffixes are included (``FAL_KEY=``, ``DB_PW=``); the +# post-match validator _key_has_secret_keyword rejects ``KEYBOARD=``/``PASSAGE=``. _SECRET_ENV_NAMES = r"(?:API_?KEY|KEY|TOKEN|SECRET|PASSWORD|PASSWD|PASS|PW|CREDENTIAL|AUTH)" -# Uppercase keys keep the legacy embedded match (``MYTOKEN=…``, ``FOO_SECRET``) -# — an all-caps key is almost never prose. _ENV_ASSIGN_RE = re.compile( rf"([A-Z0-9_]{{0,50}}{_SECRET_ENV_NAMES}[A-Z0-9_]{{0,50}})\s*=\s*(['\"]?)(\S+)\2", ) -# Lowercase env names: only underscore-boundary forms (``openai_key=…``, -# ``FAL_KEY=…``, ``db_pw=…``) — NOT bare ``password=``/``token=``/``secret=``, -# which appear in prose, URLs, and form bodies (issue #77484). -# Anchor each attempt to the start of an identifier run. Without the -# lookbehind, ``re.sub`` retries the greedy ``[a-z0-9_]+`` prefix at every byte -# of a long non-matching opaque payload, making strict compaction redaction -# quadratic while holding the GIL (#99255). +# Lowercase env names: only underscore-boundary forms (``openai_key=``, +# ``db_pw=``) — NOT bare ``password=``/``token=``, which appear in prose, URLs, +# and form bodies. The lookbehind anchors each attempt to the start of an +# identifier run; without it re.sub retries the greedy prefix at every byte of +# a long opaque payload (quadratic while holding the GIL). _ENV_ASSIGN_LOWER_RE = re.compile( rf"(? bool: return True if cur.isupper() and prev.islower(): return True # camelCase: clientSecret - # Acronym run ending: APIToken — the 'T' begins a new word when it is - # followed by lowercase while the preceding run is uppercase. - if cur.isupper() and prev.isupper() and i + 1 < len(s) and s[i + 1].islower(): - return True - return False + # Acronym run ending (APIToken): 'T' starts a word when followed by lowercase. + return cur.isupper() and prev.isupper() and i + 1 < len(s) and s[i + 1].islower() def _is_word_end(s: str, j: int, *, allow_plural: bool = True) -> bool: @@ -311,47 +238,35 @@ def _is_word_end(s: str, j: int, *, allow_plural: bool = True) -> bool: return False -def _key_has_secret_keyword(key: str) -> bool: - """True if ``key`` contains a secret keyword at a word boundary. +def _has_word_bounded_keyword(key: str, keyword_re: "re.Pattern[str]") -> bool: + """True if ``keyword_re`` matches ``key`` at a word boundary (see _KEY_KEYWORD_RE).""" + return any( + _is_word_start(key, m.start()) and _is_word_end(key, m.end()) + for m in keyword_re.finditer(key) + ) - Post-match validator for _CFG_DOTTED_RE / _CFG_ANCHORED_RE / - _YAML_ASSIGN_RE hits — rejects prose words that merely embed a keyword - (``secretary``, ``tokenizer``, ``authored``). Safe to call with the - _ENV_ASSIGN_RE key too: all-caps keys short-circuit to the legacy - embedded-match behavior. + +def _key_has_secret_keyword(key: str) -> bool: + """Post-match validator for the _CFG_*/_YAML_/_ENV_ASSIGN_RE key group. + + Rejects prose words that merely embed a keyword (``secretary``, ``tokenizer``, + ``authored``). All-caps keys get the same word-bounded test: ``API_KEY`` / + ``DB_PW`` count, ``KEYBOARD`` / ``PASSAGE`` do not. """ - letters = [c for c in key if c.isalpha()] - if letters and all(c.isupper() for c in letters): - # Legacy all-caps behavior (MYTOKEN=…): an all-caps key is almost - # never prose. Exception: a bare ``KEY``/``PASS``/``PW`` embedded in - # a longer all-caps word (``KEYBOARD``, ``PASSAGE``) is prose, not a - # credential — only a word-bounded compound (``API_KEY``, - # ``MYSQL_PASSWORD``, ``FAL_KEY``, ``DB_PW``) counts (issue #77484). - for m in _KEY_KEYWORD_RE.finditer(key): - if _is_word_start(key, m.start()) and _is_word_end(key, m.end()): - return True - return False - for m in _KEY_KEYWORD_RE.finditer(key): - if _is_word_start(key, m.start()) and _is_word_end(key, m.end()): - return True - return False + return _has_word_bounded_keyword(key, _KEY_KEYWORD_RE) def _key_has_strong_secret_keyword(key: str) -> bool: """Return whether ``key`` names an unambiguously credential-bearing field.""" - for match in _STRONG_KEY_KEYWORD_RE.finditer(key): - if _is_word_start(key, match.start()) and _is_word_end(key, match.end()): - return True - return False + return _has_word_bounded_keyword(key, _STRONG_KEY_KEYWORD_RE) def _looks_like_opaque_credential(value: str) -> bool: - """Return whether an ambiguous token/key value has credential-like shape. + """Credential-like shape test for ambiguous ``token``/``key`` values. - Known vendor prefixes and JWTs have dedicated redactors. This catches the - remaining opaque family without treating short technical scalars such as - ``CPU``, ``local``, or training captions as secrets merely because their - key contains ``token`` or ``key``. + Vendor prefixes and JWTs have dedicated redactors; this catches the remaining + opaque family without treating short technical scalars (``CPU``, ``local``) + as secrets merely because their key contains ``token`` or ``key``. """ if value == "***" or value.startswith("«redacted:"): return True @@ -372,6 +287,21 @@ def _assignment_value_requires_redaction(key: str, value: str) -> bool: """Apply value-aware gating to key-name-only assignment matches.""" return _key_has_strong_secret_keyword(key) or _looks_like_opaque_credential(value) + +def _should_redact_assignment(key: str, value: str, *, check_keyword: bool) -> bool: + """Shared gate for the ENV / JSON / YAML assignment passes. + + Skips programmatic env lookups used as values (code snippets, not secrets), + optionally requires a word-bounded secret keyword in the key, then applies + the value-shape gate. + """ + if _ENV_LOOKUP_VALUE_RE.match(value): + return False + if check_keyword and not _key_has_secret_keyword(key): + return False + return _assignment_value_requires_redaction(key, value) + + # JSON field patterns: "apiKey": "value", "token": "value", etc. _JSON_KEY_NAMES = r"(?:api_?[Kk]ey|token|secret|password|access_token|refresh_token|auth_token|bearer|secret_value|raw_secret|secret_input|key_material)" _JSON_FIELD_RE = re.compile( @@ -380,26 +310,19 @@ _JSON_FIELD_RE = re.compile( ) # Authorization headers — any scheme (Bearer, Basic, Token, Digest, …) plus the -# bare-credential form, and Proxy-Authorization. The credential token is masked -# while the header name and scheme word are preserved for debuggability. The -# previous rule only matched ``Bearer``, so ``Basic `` and -# ``token `` leaked verbatim into logs/transcripts. -# -# The credential class excludes quote characters (``"`` / ``'``): a token sitting -# flush against a closing quote (``"Authorization: Bearer sk-..."``) must not pull -# that quote into the match, or masking turns value corruption into *syntax* -# corruption — the closing quote vanishes and the command/string no longer parses -# (unterminated quote → shell EOF / Python SyntaxError). Real credentials never -# contain ``"`` or ``'``, so excluding them is safe. See #43083. +# bare-credential form, and Proxy-Authorization; header name and scheme word are +# preserved. The credential class excludes quotes: a token flush against a +# closing quote must not pull it into the mask, or value corruption becomes +# SYNTAX corruption (unterminated quote → shell EOF / SyntaxError). Real +# credentials never contain ``"`` or ``'``. _AUTH_HEADER_RE = re.compile( r"((?:Proxy-)?Authorization:\s*)([A-Za-z][\w.+-]*\s+)?([^\s\"']+)", re.IGNORECASE, ) -# API-key style auth headers carrying a single opaque value (no scheme word). -# Anthropic and many providers authenticate with ``x-api-key``; values without -# a known vendor prefix (custom/local backends) would otherwise leak when a -# request or curl command is logged or echoed into tool output / transcripts. +# API-key style auth headers carrying a single opaque value (no scheme word); +# values without a vendor prefix (custom/local backends) would otherwise leak +# when a request or curl command is echoed into tool output / transcripts. _SECRET_HEADER_NAMES = ( r"(?:x-api-key|x-goog-api-key|api-key|apikey|x-api-token|x-auth-token|x-access-token)" ) @@ -408,8 +331,7 @@ _SECRET_HEADER_RE = re.compile( re.IGNORECASE, ) -# Telegram bot tokens: bot: or :, -# where token part is restricted to [-A-Za-z0-9_] and length >= 30 +# Telegram bot tokens: bot: or :, token >= 30 chars. _TELEGRAM_RE = re.compile( r"(bot)?(\d{8,}):([-A-Za-z0-9_]{30,})", ) @@ -419,34 +341,24 @@ _PRIVATE_KEY_RE = re.compile( r"-----BEGIN[A-Z ]*PRIVATE KEY-----[\s\S]*?-----END[A-Z ]*PRIVATE KEY-----" ) -# Database connection strings: protocol://user:PASSWORD@host -# Catches postgres, mysql, mongodb, redis, amqp URLs and redacts the password. -# The userinfo and password groups forbid whitespace ([^:\s]+ / [^@\s]+) so the -# match can never span a line break. A real DSN password never contains -# whitespace; without this bound the greedy [^@]+ would scan past the end of a -# code line to the next stray "@" (e.g. a Python decorator), swallowing -# intervening lines and corrupting tool OUTPUT for any source containing a -# postgresql:// f-string template. See issue #33801. +# Database connection strings: protocol://user:PASSWORD@host. The userinfo and +# password groups forbid whitespace so a match can never span a line break — a +# greedy ``[^@]+`` scanned past a code line to the next stray ``@`` (e.g. a +# decorator) and corrupted tool output for any source with a DSN f-string. _DB_CONNSTR_RE = re.compile( r"((?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://[^:\s]+:)([^@\s]+)(@)", re.IGNORECASE, ) -# Bare-token credential in a web/transport URL: ``scheme://TOKEN@host``. -# This is the ``git remote set-url origin https://PASSWORD@github.com/...`` -# shape from issue #6396 — a single opaque credential in the userinfo position -# with NO ``user:pass`` colon. It is unambiguously a secret: legitimate -# round-trip URLs (OAuth callbacks, magic links, pre-signed shares — see the -# "Web-URL redaction is intentionally OFF" note in redact_sensitive_text) carry -# their tokens in the QUERY STRING, never in bare userinfo. The colon form -# ``user:pass@`` is deliberately left to pass through (commit "pass web URLs -# through unchanged", #34029) and is NOT matched here — the token class forbids -# ``:``. DB schemes are handled by _DB_CONNSTR_RE above and excluded here. -# -# Guards against false positives: -# - 8+ char floor skips short usernames (git, admin, root, deploy, ubuntu). -# - The token class ``[^\s:@/]`` cannot cross ``/``, so an ``@`` sitting in a -# path or query (e.g. ``?q=user@example.com``) is never treated as userinfo. +# Bare-token credential in a web/transport URL: ``scheme://TOKEN@host`` (the +# ``git remote set-url https://PASSWORD@github.com/...`` shape) — a single +# opaque credential in userinfo with NO ``user:pass`` colon. Unambiguously a +# secret: round-trip URLs (OAuth callbacks, magic links, pre-signed shares) +# carry tokens in the QUERY STRING, never bare userinfo. The ``user:pass@`` form +# deliberately passes through (token class forbids ``:``); DB schemes are +# handled by _DB_CONNSTR_RE. False-positive guards: 8+ char floor skips short +# usernames (git, admin, deploy); the class forbids ``/`` so an ``@`` in a path +# or query (``?q=user@example.com``) is never treated as userinfo. _URL_BARE_TOKEN_RE = re.compile( r"((?:https?|wss?|git|ssh|ftp|ftps|sftp)://)" # scheme r"([^\s:@/]{8,})" # bare token (no colon/slash/@), 8+ chars @@ -454,20 +366,18 @@ _URL_BARE_TOKEN_RE = re.compile( re.IGNORECASE, ) -# JWT tokens: header.payload[.signature] — always start with "eyJ" (base64 for "{") -# Matches 1-part (header only), 2-part (header.payload), and full 3-part JWTs. +# JWT tokens: header.payload[.signature] — always start with "eyJ" (base64 "{"). +# Matches 1-part (header only), 2-part, and full 3-part JWTs. _JWT_RE = re.compile( r"eyJ[A-Za-z0-9_-]{10,}" # Header (always starts with eyJ) r"(?:\.[A-Za-z0-9_=-]{4,}){0,2}" # Optional payload and/or signature ) -# E.164 phone numbers: +, 7-15 digits -# Negative lookahead prevents matching hex strings or identifiers +# E.164 phone numbers: +, 7-15 digits. +# Negative lookahead prevents matching hex strings or identifiers. _SIGNAL_PHONE_RE = re.compile(r"(\+[1-9]\d{6,14})(?![A-Za-z0-9])") -# URLs containing query strings — matches `scheme://...?...[# or end]`. -# Used to scan text for URLs whose query params may contain secrets. -# Ported from nearai/ironclaw#2529. +# URLs containing query strings — `scheme://...?...[# or end]` (CDP-URL path). _URL_WITH_QUERY_RE = re.compile( r"(https?|wss?|ftp)://" # scheme r"([^\s/?#]+)" # authority (may include userinfo) @@ -476,85 +386,67 @@ _URL_WITH_QUERY_RE = re.compile( r"(#\S*)?", # optional fragment ) -# URLs containing userinfo — `scheme://user:password@host` for ANY scheme -# (not just DB protocols already covered by _DB_CONNSTR_RE above). -# Catches things like `https://user:token@api.example.com/v1/foo`. +# URLs containing userinfo — `scheme://user:password@host` for ANY web scheme +# (DB protocols are covered by _DB_CONNSTR_RE). CDP-URL path. _URL_USERINFO_RE = re.compile( r"(https?|wss?|ftp)://([^/\s:@]+):([^/\s@]+)@", ) # Strict provider-egress URL redaction accepts more URL-reference forms than # the display/log helpers above. Parameter delimiters stay in capture groups so -# redaction preserves the original query/fragment layout byte-for-byte, while -# the key is decoded separately for classification. Values stop at query or -# fragment pair separators; both ``&`` and ``;`` are valid in deployed URLs. +# the original query/fragment layout is preserved byte-for-byte; the key is +# decoded separately for classification. Values stop at ``&``/``;`` (both valid). _STRICT_URL_PARAM_RE = re.compile( r"([?#&;])([A-Za-z0-9_.~+%\-]+)=([^#&;\s\"'<>]*)" ) -# Match userinfo in both absolute (``scheme://user:pass@host``) and -# network-path (``//user:pass@host``) references. The authority boundary stops -# at path/query/fragment delimiters so an ``@`` elsewhere in a URL is ignored. -# -# Anchored on the mandatory ``//`` rather than an optional scheme prefix: the -# scheme sits outside the match either way (replacement callbacks re-emit -# group(1), so ``https:`` stays untouched in the surrounding text), and the -# old optional-scheme prefix ``(?:[A-Za-z][A-Za-z0-9+.-]*:)?`` backtracked -# catastrophically (O(n²)) on long unbroken alphanumeric runs — a 320KB -# synthetic compaction payload spent ~55s inside this pattern per sub() call. -# Output-equivalence to the old pattern was fuzz-verified (20k random strings -# plus targeted URL forms). +# Userinfo in absolute (``scheme://user:pass@host``) and network-path +# (``//user:pass@host``) references; the authority stops at path/query/fragment +# delimiters so an ``@`` elsewhere is ignored. Anchored on the mandatory ``//`` +# rather than an optional scheme prefix: the scheme sits outside the match +# either way, and the old optional-scheme prefix backtracked O(n²) on long +# alphanumeric runs (~55s per sub() on a 320KB compaction payload). +# Output-equivalence was fuzz-verified. _STRICT_URL_USERINFO_RE = re.compile( r"(//)([^/\s?#@]+)@" ) -# HTTP access logs often use a relative request target rather than a full URL: -# `"POST /webhook?password=... HTTP/1.1"`. The full-URL redactor above only -# sees strings containing `://`, so handle request-target query strings too. -_HTTP_REQUEST_TARGET_QUERY_RE = re.compile( - r"\b((?:GET|POST|PUT|PATCH|DELETE|HEAD|OPTIONS|TRACE|CONNECT)\s+[^ \t\r\n\"']*?)" - r"\?([^ \t\r\n\"']+)", - re.IGNORECASE, -) - # Form-urlencoded body detection: conservative — only applies when the entire # text looks like a query string (k=v&k=v pattern with no newlines). _FORM_BODY_RE = re.compile( r"^[A-Za-z_][A-Za-z0-9_.-]*=[^&\s]*(?:&[A-Za-z_][A-Za-z0-9_.-]*=[^&\s]*)+$" ) -# Control / zero-width characters that can split a token body: a secret -# smuggled as ``sk-abc\x1bdef…`` or ``ghp_abc\n123…`` escapes the contiguous -# prefix regexes (issue #77484). Used by _mask_control_split_tokens. +# Control / zero-width characters that can split a token body (``sk-abc\x1bdef``, +# ``ghp_abc\n123``) and escape the contiguous prefix regexes. _CONTROL_CHARS_RE = re.compile( r"[\x00-\x1f\x7f\u200b-\u200f\u2028-\u202f\u2060\ufeff]" ) -# Union of every _PREFIX_PATTERNS body class — a control-stripped match may -# only span original chars that are token-body or control chars (see -# _mask_control_split_tokens). ``=`` is deliberately excluded: a KEY=value -# assignment separator must never let a match span across unrelated text. +# Union of every _PREFIX_PATTERNS body class — a control-stripped match may only +# span original chars that are token-body or control chars. ``=`` is deliberately +# excluded: a KEY=value separator must never let a match span unrelated text. _TOKEN_BODY_CHARS = frozenset( "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789_-." ) -# Compile known prefix patterns into one alternation -_PREFIX_RE = re.compile( - r"(? "re.Pattern[str]": + return re.compile( + r"(? str: """Mask tokens whose body is split by control/zero-width characters. - A credential like ``sk-abc\\x1bdef456…`` or ``ghp_abc\\n123def…`` has its - token body interrupted, so the contiguous _PREFIX_RE cannot match it and - the secret leaks verbatim (issue #77484). Strategy: build a copy with all - control chars removed (the token is contiguous again, matching even when - each fragment alone is too short), match on that, then mask the - corresponding span in the *original* — but only when the original span - contains solely token-body and control chars (a match that crosses into a - different line's unrelated text, e.g. ``EXA_API_KEY=*** is rejected). + Match on a control-stripped copy (the token is contiguous again, even when + each fragment alone is too short), then mask the corresponding span in the + ORIGINAL — only when that span holds solely token-body and control chars, so + a match can never cross into another line's unrelated text. """ stripped = _CONTROL_CHARS_RE.sub("", text) if stripped == text: @@ -566,27 +458,18 @@ def _mask_control_split_tokens(text: str, mask_fn) -> str: body = m.group(1) start_orig = orig_idx[m.start(1)] end_orig = orig_idx[m.end(1) - 1] + 1 - # If a fragment inside the span already matches _PREFIX_RE on its - # own AND the span crosses a LINE boundary (\n / \r), do NOT join. - # A complete token at end-of-line followed by a word line - # (``ghp_\nbutton [ref=e3]``) joins into one stripped-copy - # match and the mask eats ``button``. Line structure is legitimate; - # the self-matching fragment is handled by the ordinary prefix pass - # (any remainder past the newline is left unmasked — accepted - # residual to preserve line structure). - # For NON-newline controls (ESC, ZWSP, ...) the join proceeds even - # when a fragment self-matches: those bytes never legitimately sit - # between a token and adjacent prose, and skipping there let the - # non-matching remainder of a split token leak - # (``sk-\x1b`` masked only the head). span = text[start_orig:end_orig] + # A fragment that already matches on its own AND a span crossing a LINE + # boundary: do NOT join. A complete token at end-of-line followed by a + # word line (``ghp_\nbutton``) would otherwise mask ``button``; the + # self-matching fragment is handled by the ordinary prefix pass. For + # non-newline controls (ESC, ZWSP) the join proceeds even when a fragment + # self-matches — those bytes never legitimately sit between a token and + # prose, and skipping would leak the tail of ``sk-\x1b``. if ("\n" in span or "\r" in span) and _PREFIX_RE.search(span): continue - # Reject matches whose original span crosses a non-token char - # (e.g. ``sk_abc…\nTAVILY_API_KEY=…`` — the ``=`` is not part of a - # token body, so the regex matched across unrelated lines). Also - # reject when the match runs into a ``KEY=`` name: a real token value - # is followed by a newline/space/end, not ``=``. + # Reject spans containing a non-token char (``sk_abc…\nTAVILY_API_KEY=`` + # matched across lines) and matches running into a ``KEY=`` name. if (all(c in _TOKEN_BODY_CHARS or _CONTROL_CHARS_RE.match(c) for c in span) and (end_orig >= len(text) or text[end_orig] != "=")): @@ -596,10 +479,9 @@ def _mask_control_split_tokens(text: str, mask_fn) -> str: return "".join(out) -# Display-mask strip for mask_secret: EVERY control char incl. \n/\t, C1, -# DEL, and zero-width/format chars — a masked secret must never emit -# multiline, tabbed, or invisible bytes into config/status/dump display -# output (#55319, #55321). +# Display-mask strip for mask_secret: EVERY control char incl. \n/\t, C1, DEL, +# and zero-width/format chars — a masked secret must never emit multiline, +# tabbed, or invisible bytes into config/status/dump display output. _DISPLAY_CONTROL_RE = re.compile( r"[\x00-\x1f\x7f\x80-\x9f\u200b-\u200f\u202a-\u202e\u2060-\u2064]" ) @@ -616,40 +498,19 @@ def mask_secret( ) -> str: """Mask a secret for display, preserving ``head`` and ``tail`` characters. - Canonical helper for display-time redaction across Hermes — used by - ``hermes config``, ``hermes status``, ``hermes dump``, and anywhere - a secret needs to be shown truncated for debuggability while still - keeping the bulk hidden. + Canonical display-time helper (``hermes config`` / ``status`` / ``dump``). + Values shorter than ``floor`` return ``placeholder``; falsy input returns + ``empty`` (override for e.g. a dimmed "(not set)"). - Args: - value: The secret to mask. ``None``/empty returns ``empty``. - head: Leading characters to preserve. Default 4. - tail: Trailing characters to preserve. Default 4. - floor: Values shorter than ``head + tail + floor_margin`` are - fully masked (returns ``placeholder``). Default 12 — - matches the existing config/status/dump convention. - placeholder: Value returned for too-short inputs. Default ``"***"``. - empty: Value returned when ``value`` is falsy (None, ""). The - caller can override this to e.g. ``color("(not set)", - Colors.DIM)`` for user-facing display. - - Examples: - >>> mask_secret("sk-proj-abcdef1234567890") - 'sk-p...7890' - >>> mask_secret("short") # fully masked - '***' - >>> mask_secret("") # empty default - '' - >>> mask_secret("", empty="(not set)") # empty override - '(not set)' - >>> mask_secret("long-token", head=6, tail=4, floor=18) - '***' + >>> mask_secret("sk-proj-abcdef1234567890") + 'sk-p...7890' + >>> mask_secret("short") + '***' """ if not value: return empty - # Visible head/tail must not carry control bytes (newline, NUL, DEL, C1) - # into config/status/dump output (#55319, #55321). Strip them before - # slicing — the length check below then sees the displayable length. + # Strip control bytes before slicing so the visible head/tail can't carry + # them and the floor check sees the displayable length. value = _DISPLAY_CONTROL_RE.sub("", value) if not value: return empty @@ -667,12 +528,7 @@ def _mask_token(token: str) -> str: def _redact_query_string(query: str) -> str: - """Redact sensitive parameter values in a URL query string. - - Handles `k=v&k=v` format. Sensitive keys (case-insensitive) have values - replaced with `***`. Non-sensitive keys pass through unchanged. - Empty or malformed pairs are preserved as-is. - """ + """Replace values of sensitive ``k=v&k=v`` params with ``***``; others pass through.""" if not query: return query parts = [] @@ -689,11 +545,7 @@ def _redact_query_string(query: str) -> str: def _redact_url_query_params(text: str) -> str: - """Scan text for URLs with query strings and redact sensitive params. - - Catches opaque tokens that don't match vendor prefix regexes, e.g. - `https://example.com/cb?code=ABC123&state=xyz` → `...?code=***&state=xyz`. - """ + """Redact sensitive query params in every URL found in ``text``.""" def _sub(m: re.Match) -> str: scheme = m.group(1) authority = m.group(2) @@ -705,11 +557,7 @@ def _redact_url_query_params(text: str) -> str: def _redact_url_userinfo(text: str) -> str: - """Strip `user:password@` from HTTP/WS/FTP URLs. - - DB protocols (postgres, mysql, mongodb, redis, amqp) are handled - separately by `_DB_CONNSTR_RE`. - """ + """Mask the password in ``user:password@`` of HTTP/WS/FTP URLs.""" return _URL_USERINFO_RE.sub( lambda m: f"{m.group(1)}://{m.group(2)}:***@", text, @@ -730,10 +578,8 @@ def _canonical_url_param_name(name: str) -> str: def _redact_strict_url_credentials(text: str) -> str: """Redact credentials from absolute, relative, and network URL references. - This is intentionally stricter than display/log redaction and is used only - at explicit secret-egress boundaries. It preserves original keys, - separators, public parameters, hosts, and paths while masking sensitive - values and URL userinfo. + Stricter than display/log redaction; used only at explicit secret-egress + boundaries. Preserves keys, separators, public params, hosts, and paths. """ def _redact_param(match: re.Match) -> str: if _canonical_url_param_name(match.group(2)) not in _SENSITIVE_QUERY_PARAMS: @@ -754,19 +600,12 @@ def _redact_strict_url_credentials(text: str) -> str: def redact_cdp_url(value: object) -> str: """Mask secrets in a CDP/browser endpoint URL before it is logged. - The global ``redact_sensitive_text`` deliberately passes web-URL query - params and ``user:pass@`` userinfo through unmasked (OAuth callbacks, - magic-link / pre-signed URLs the agent is meant to follow -- see the - web-URL note above). CDP discovery endpoints are NOT such a workflow: - their query-string tokens and userinfo passwords are pure credentials - that must never reach the logs. So for CDP URLs we opt INTO the two URL - redactors that the global pass leaves off. - - This is the single source of truth for redacting a CDP URL that is passed - *directly* to a log or error message. Callers that instead need to redact an - exception whose text embeds the URL (e.g. a ``websockets`` connect error) - should route that through their own error-text helper, which delegates here - -- see ``tools.browser_supervisor._redact_cdp_error_text``. + ``redact_sensitive_text`` deliberately passes web-URL query params and + ``user:pass@`` through (OAuth callbacks, magic links the agent must follow). + CDP discovery endpoints are NOT such a workflow — their tokens are pure + credentials — so this opts INTO both URL redactors. Single source of truth + for CDP URLs passed directly to a log/error; error-text helpers that embed + the URL delegate here (``tools.browser_supervisor._redact_cdp_error_text``). """ text = redact_sensitive_text("" if value is None else str(value)) if not text: @@ -776,26 +615,13 @@ def redact_cdp_url(value: object) -> str: return text -def _redact_http_request_target_query_params(text: str) -> str: - """Redact sensitive query params in HTTP access-log request targets.""" - def _sub(m: re.Match) -> str: - prefix = m.group(1) - query = _redact_query_string(m.group(2)) - return f"{prefix}?{query}" - return _HTTP_REQUEST_TARGET_QUERY_RE.sub(_sub, text) - - def _redact_form_body(text: str) -> str: - """Redact sensitive values in a form-urlencoded body. + """Redact sensitive values when the ENTIRE text is a clean ``k=v&k=v`` body. - Only applies when the entire input looks like a pure form body - (k=v&k=v with no newlines, no other text). Single-line non-form - text passes through unchanged. This is a conservative pass — the - `_redact_url_query_params` function handles embedded query strings. + Conservative on purpose; embedded query strings are handled elsewhere. """ if not text or "\n" in text or "&" not in text: return text - # The body-body form check is strict: only trigger on clean k=v&k=v. if not _FORM_BODY_RE.match(text.strip()): return text return _redact_query_string(text.strip()) @@ -804,22 +630,13 @@ def _redact_form_body(text: str) -> str: def _mask_token_nonreusable(token: str) -> str: """Redact a prefix-matched credential to a NON-REUSABLE sentinel. - Unlike :func:`_mask_token` (which keeps head/tail chars — fine for logs - that are never fed back into a config), this emits a marker that: - - * cannot be mistaken for a usable-but-truncated key, so an agent that - reads it from a config file and writes it back does NOT corrupt the - stored credential into a dead 13-char string (issue #35519); and - * still does not leak the secret material (no head/tail chars). - - The vendor prefix label is preserved for debuggability so the agent can - still tell *which* credential is present (e.g. a GitHub PAT vs an OpenAI - key) without seeing any of its bytes. + Unlike :func:`_mask_token`, emits no head/tail chars: a truncated-looking + mask read from a config file and written back by an agent silently + corrupted the stored credential into a dead 13-char string. Only the vendor + prefix label (``ghp_``, ``sk-``) is kept so the credential KIND stays visible. """ if not token: return "«redacted-secret»" - # Preserve only the recognizable vendor prefix label (e.g. "ghp_", "sk-"), - # never any of the random secret body. label = "" for sub in _PREFIX_SUBSTRINGS: if token.startswith(sub): @@ -828,6 +645,87 @@ def _mask_token_nonreusable(token: str) -> str: return f"«redacted:{label}…»" if label else "«redacted-secret»" +def _redact_assignments(text: str) -> str: + """ENV / config / JSON / YAML assignment passes (skipped for code files). + + Every pass skips URL-bearing text where noted: web-URL query params are + intentionally passed through (see the note in redact_sensitive_text) and + the lowercase/config regexes would otherwise match ``token=``/``key=`` params. + """ + if "=" in text: + def _redact_env(m): + name, quote, value = m.group(1), m.group(2), m.group(3) + if not _should_redact_assignment(name, value, check_keyword=True): + return m.group(0) + return f"{name}={quote}{_mask_token(value)}{quote}" + text = _ENV_ASSIGN_RE.sub(_redact_env, text) + # Lowercase env names (``openai_key=…``); the uppercase regex is + # all-caps-only so it never matches URL params, this one would. + if "://" not in text: + text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text) + # Lowercase/dotted config keys. The keyword pre-gate is exact (every + # _CFG_*_RE match needs a secret keyword) and matters: _CFG_DOTTED_RE + # backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs + # (base64/hex blobs in compaction payloads). + if "://" not in text and _CFG_SECRET_WORD_RE.search(text): + text = _CFG_DOTTED_RE.sub(_redact_env, text) + text = _CFG_ANCHORED_RE.sub(_redact_env, text) + + # JSON fields: "apiKey": "***" + if ":" in text and '"' in text: + def _redact_json(m): + key, value = m.group(1), m.group(2) + if not _should_redact_assignment(key, value, check_keyword=False): + return m.group(0) + return f'{key}: "{_mask_token(value)}"' + text = _JSON_FIELD_RE.sub(_redact_json, text) + + # Unquoted YAML / colon config: password: *** (after JSON so quoted values + # are handled there; _YAML_ASSIGN_RE's lookahead skips quotes). + if ":" in text and "://" not in text: + def _redact_yaml(m): + key, sep, value = m.group(1), m.group(2), m.group(3) + if not _should_redact_assignment(key, value, check_keyword=True): + return m.group(0) + return f"{key}{sep}{_mask_token(value)}" + text = _YAML_ASSIGN_RE.sub(_redact_yaml, text) + return text + + +def _redact_url_credentials(text: str, code_file: bool) -> str: + """DB connection-string passwords and bare-token URL userinfo (``://`` text only).""" + def _redact_db(m): + # With code_file=True a pure ``{...}`` password group is an f-string + # template reference (f"postgresql://{user}:{pass}@{host}"), not a + # literal credential — preserve it. The regex forbids whitespace in the + # password group, so a single-line template's group(2) is exactly the + # brace expression. + pw = m.group(2) + if code_file and pw.startswith("{") and pw.endswith("}"): + return m.group(0) + return f"{m.group(1)}***{m.group(3)}" + text = _DB_CONNSTR_RE.sub(_redact_db, text) + # ``scheme://TOKEN@host`` — only the colon-less bare-token form; ``user:pass@`` + # and query-string tokens pass through (see the web-URL note below). + return _URL_BARE_TOKEN_RE.sub( + lambda m: f"{m.group(1)}{_mask_token(m.group(2))}{m.group(3)}", + text, + ) + + +def _redact_phone(m): + phone = m.group(1) + if len(phone) <= 8: + return phone[:2] + "****" + phone[-2:] + return phone[:4] + "****" + phone[-4:] + + +def _redact_telegram(m): + prefix = m.group(1) or "" + digits = m.group(2) + return f"{prefix}{digits}:***" + + def redact_sensitive_text( text: str, *, @@ -838,40 +736,29 @@ def redact_sensitive_text( ) -> str: """Apply all redaction patterns to a block of text. - Safe to call on any string -- non-matching text passes through unchanged. - Enabled by default. Disable via security.redact_secrets: false in config.yaml. - Set force=True for safety boundaries that must never return raw secrets - regardless of the user's global logging redaction preference. + Safe on any string; non-matching text passes through unchanged. Enabled by + default (``security.redact_secrets: false`` disables); ``force=True`` is for + safety boundaries that must never return raw secrets regardless. - Set redact_url_credentials=True at non-navigation egress boundaries to - additionally redact credential-named query parameters and ``user:pass@`` - URL userinfo. The default remains False because actionable OAuth callback, - magic-link, and pre-signed URLs must survive ordinary tool flows unchanged. + ``redact_url_credentials=True``: at non-navigation egress boundaries, also + redact credential-named query params and ``user:pass@`` userinfo. Default + False because actionable OAuth-callback / magic-link / pre-signed URLs must + survive ordinary tool flows unchanged. - Set code_file=True to skip the ENV-assignment and JSON-field regex - patterns when the text is known to be source code (e.g. MAX_TOKENS=*** - constants, "apiKey": "test" fixtures). Prefix patterns, auth headers, - private keys, DB connstrings, JWTs, and URL secrets are still redacted. + ``code_file=True``: skip the ENV-assignment and JSON-field passes for known + source code (``MAX_TOKENS=***`` constants, ``"apiKey": "test"`` fixtures). + Prefix patterns, auth headers, private keys, DSNs, JWTs are still redacted. - Set file_read=True for file *content* returned to the agent (read_file / - search_files / cat). Secrets are STILL redacted — they are never exposed — - but prefix-matched credentials are replaced with a non-reusable sentinel - (``«redacted:ghp_…»``) instead of a head/tail-preserving mask - (``ghp_S1...Pn2T``). The old mask looked like a real-but-truncated key, so - an agent reading it from config.yaml and writing it back silently corrupted - the stored credential into a dead 13-char value → 401 (issue #35519). The - sentinel is syntactically invalid as a token, so it can't be mistaken for a - usable key or written back as one. Implies code_file=True (config/data - files shouldn't trigger the source-code ENV/JSON false-positive paths). + ``file_read=True``: for file CONTENT returned to the agent. Secrets are still + redacted, but prefix-matched credentials become a non-reusable sentinel + (``«redacted:ghp_…»``) instead of a head/tail mask that looks like a real + truncated key (an agent wrote one back into config.yaml → dead credential → + 401). Implies ``code_file=True``. - Performance: each regex pattern is gated behind a cheap substring - pre-check (e.g. ``"=" in text`` for ENV assignments, ``"://" in text`` - for URLs, ``"eyJ" in text`` for JWTs). On a typical hermes log line - (no secrets) this drops the 13-pattern scan from ~5.6us to ~1.8us per - record (-68%). The pre-checks are conservative — false positives - still run the full regex, which then doesn't match. False negatives - are impossible because every regex requires the gated substring to - match. + Performance: every regex is gated behind a cheap substring pre-check + (``"=" in text``, ``"://" in text``, ``"eyJ" in text``, …) — conservative + (false positives just run the regex), never false-negative because every + regex requires the gated substring. """ if text is None: return None @@ -882,177 +769,53 @@ def redact_sensitive_text( if not (force or _REDACT_ENABLED): return text - # file_read content shouldn't hit the source-code ENV/JSON false-positive - # paths either (it's config/data, not log lines). if file_read: code_file = True - # Known prefixes (sk-, ghp_, etc.) — gate on substring presence + # Known prefixes (sk-, ghp_, etc.). Control/zero-width chars can split a + # token body so _PREFIX_RE alone misses it — mask those runs first. if _has_known_prefix_substring(text): _prefix_sub = _mask_token_nonreusable if file_read else _mask_token - # Control/zero-width chars (\\n, \\r, ESC, U+200B, …) split a token - # body so _PREFIX_RE cannot match across them — a secret smuggled as - # ``sk-abc\\x1bdef…`` leaks verbatim (issue #77484). Mask such runs by - # first matching on a control-stripped copy, then re-masking the - # corresponding span in the original (the stripped copy and the - # original are aligned 1:1 for non-control chars). text = _mask_control_split_tokens(text, _prefix_sub) text = _PREFIX_RE.sub(lambda m: _prefix_sub(m.group(1)), text) - # ENV assignments: OPENAI_API_KEY=*** (skip for code files — false positives) if not code_file: - if "=" in text: - def _redact_env(m): - name, quote, value = m.group(1), m.group(2), m.group(3) - # Programmatic env lookups reference variable *names*, not - # secret values — masking them corrupts code snippets in - # prose/log contexts (issue #2852): ``KEY=os.getenv('X')``. - if _ENV_LOOKUP_VALUE_RE.match(value): - return m.group(0) - # Keyword must sit at a word boundary within the key — - # ``author=Smith`` / ``press.secretary=…`` are prose, not - # credentials (ported from nearai/ironclaw#6129). All-caps - # keys (the _ENV_ASSIGN_RE shape) short-circuit to legacy - # embedded matching inside the helper. - if not _key_has_secret_keyword(name): - return m.group(0) - if not _assignment_value_requires_redaction(name, value): - return m.group(0) - return f"{name}={quote}{_mask_token(value)}{quote}" - text = _ENV_ASSIGN_RE.sub(_redact_env, text) - # Lowercase env names (``openai_key=…``). Skip URLs — the query - # string may contain ``token=``/``key=`` params that are - # intentionally passed through (see note near the bottom of this - # function; _redact_strict_url_credentials handles the opt-in - # case). The uppercase regex above is all-caps-only, so it never - # matches URL params; the lowercase one would (issue #77484). - if "://" not in text: - text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text) - # Lowercase/dotted config keys (issue #16413). Skip URLs entirely — - # web-URL query params are intentionally passed through (see note - # near the bottom of this function); _DB_CONNSTR_RE still guards - # connection-string passwords. - # - # Extra gate: every _CFG_*_RE match requires a secret keyword in - # the key, so a text without any secret keyword cannot match — - # skipping is exact. This matters because _CFG_DOTTED_RE - # backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs - # (e.g. base64/hex blobs in compaction payloads); the linear - # keyword scan prevents that pathological path on secret-free - # text. - if "://" not in text and _CFG_SECRET_WORD_RE.search(text): - text = _CFG_DOTTED_RE.sub(_redact_env, text) - text = _CFG_ANCHORED_RE.sub(_redact_env, text) + text = _redact_assignments(text) - # JSON fields: "apiKey": "***" (skip for code files — false positives) - if ":" in text and '"' in text: - def _redact_json(m): - key, value = m.group(1), m.group(2) - # Same programmatic-env-lookup exception as _redact_env above - # (issue #2852): "apiKey": "os.getenv('X')" is a code snippet, - # not a leaked secret value. - if _ENV_LOOKUP_VALUE_RE.match(value): - return m.group(0) - if not _assignment_value_requires_redaction(key, value): - return m.group(0) - return f'{key}: "{_mask_token(value)}"' - text = _JSON_FIELD_RE.sub(_redact_json, text) - - # Unquoted YAML / colon config: password: *** (after JSON so quoted - # values are handled there; the lookahead in _YAML_ASSIGN_RE skips - # quotes). Skip URLs — web-URL query params pass through by design. - if ":" in text and "://" not in text: - def _redact_yaml(m): - key, sep, value = m.group(1), m.group(2), m.group(3) - # Same programmatic-env-lookup exception as _redact_env above - # (issue #2852): api_key: os.getenv('X') is a code snippet, - # not a leaked secret value. - if _ENV_LOOKUP_VALUE_RE.match(value): - return m.group(0) - # Keyword must sit at a word boundary within the key — - # ``Secretary: J.Smith`` / ``tokenizer: cl100k_base`` are - # document text, not credentials (nearai/ironclaw#6129). - if not _key_has_secret_keyword(key): - return m.group(0) - if not _assignment_value_requires_redaction(key, value): - return m.group(0) - return f"{key}{sep}{_mask_token(value)}" - text = _YAML_ASSIGN_RE.sub(_redact_yaml, text) - - # Authorization headers — _AUTH_HEADER_RE matches any scheme after - # "[Proxy-]Authorization:" case-insensitively, so "uthorization" is the - # cheapest substring gate that covers every casing without a casefold(). + # Authorization headers — case-insensitive regex, so "uthorization" is the + # cheapest substring gate covering every casing without a casefold(). if "uthorization" in text or "UTHORIZATION" in text: text = _AUTH_HEADER_RE.sub( lambda m: m.group(1) + (m.group(2) or "") + _mask_token(m.group(3)), text, ) - # API-key style headers (x-api-key, api-key, …). Header values are - # colon-separated, so gate on ":" — the regex itself is the precise filter. + # API-key style headers (x-api-key, api-key, …) and Telegram bot tokens — + # both require ":"; the regexes are the precise filters. if ":" in text: text = _SECRET_HEADER_RE.sub( lambda m: m.group(1) + _mask_token(m.group(2)), text, ) - - # Telegram bot tokens — pattern requires ":" with digits prefix - if ":" in text: - def _redact_telegram(m): - prefix = m.group(1) or "" - digits = m.group(2) - return f"{prefix}{digits}:***" text = _TELEGRAM_RE.sub(_redact_telegram, text) - # Private key blocks if "BEGIN" in text and "-----" in text: text = _PRIVATE_KEY_RE.sub("[REDACTED PRIVATE KEY]", text) - # Database connection string passwords. With code_file=True, a password - # group that is a pure ``{...}`` brace expression is an f-string template - # reference (e.g. f"postgresql://{user}:{pass}@{host}"), not a literal - # credential — preserve it. Literal passwords are still redacted. The regex - # forbids whitespace in the password group, so a single-line template's - # group(2) is exactly the brace expression. See issue #33801. if "://" in text: - if code_file: - def _redact_db(m): - pw = m.group(2) - if pw.startswith("{") and pw.endswith("}"): - return m.group(0) - return f"{m.group(1)}***{m.group(3)}" - text = _DB_CONNSTR_RE.sub(_redact_db, text) - else: - text = _DB_CONNSTR_RE.sub(lambda m: f"{m.group(1)}***{m.group(3)}", text) - - # Bare-token userinfo in web/transport URLs: ``scheme://TOKEN@host``. - # The git-remote-with-embedded-password shape from #6396. Only the - # colon-less bare-token form is redacted — ``user:pass@`` and - # query-string tokens are left to pass through (see the web-URL note - # below). See _URL_BARE_TOKEN_RE for the false-positive guards. - text = _URL_BARE_TOKEN_RE.sub( - lambda m: f"{m.group(1)}{_mask_token(m.group(2))}{m.group(3)}", - text, - ) + text = _redact_url_credentials(text, code_file) # JWT tokens (eyJ... — base64-encoded JSON headers) if "eyJ" in text: text = _JWT_RE.sub(lambda m: _mask_token(m.group(0)), text) - # NOTE: Web-URL redaction (query params + userinfo + HTTP access-log - # request targets) is intentionally OFF. Many legitimate workflows pass - # opaque tokens through query strings — magic-link checkouts, OAuth - # callbacks the agent is meant to follow, pre-signed share URLs — and - # blanket-redacting param values by name breaks those skills mid-flow. - # Known credential shapes (sk-, ghp_, JWTs, etc.) inside URLs are still - # caught by _PREFIX_RE and _JWT_RE above. DB connection-string passwords - # are still caught by _DB_CONNSTR_RE. The ONE userinfo case still redacted - # is the colon-less bare-token form ``scheme://TOKEN@host`` (#6396, handled - # by _URL_BARE_TOKEN_RE in the ``://`` block above): a bare credential in - # userinfo is never a round-trip workflow token (those live in the query - # string), so masking it can't break a skill. The ``user:pass@`` form is - # left to pass through per #34029. - + # NOTE: Web-URL redaction (query params + ``user:pass@`` userinfo) is + # intentionally OFF by default: magic-link checkouts, OAuth callbacks, and + # pre-signed share URLs carry opaque tokens in query strings, and masking + # them by name breaks those skills mid-flow. Known credential shapes inside + # URLs are still caught by _PREFIX_RE / _JWT_RE, DSN passwords by + # _DB_CONNSTR_RE, and colon-less ``scheme://TOKEN@host`` by _URL_BARE_TOKEN_RE + # (a bare userinfo credential is never a round-trip workflow token). if redact_url_credentials: text = _redact_strict_url_credentials(text) @@ -1062,74 +825,47 @@ def redact_sensitive_text( # E.164 phone numbers (Signal, WhatsApp) if "+" in text: - def _redact_phone(m): - phone = m.group(1) - if len(phone) <= 8: - return phone[:2] + "****" + phone[-2:] - return phone[:4] + "****" + phone[-4:] text = _SIGNAL_PHONE_RE.sub(_redact_phone, text) return text -# Commands whose stdout is an environment-variable dump (KEY=value lines), -# NOT source code. For these, terminal-output redaction must run the -# ENV-assignment pass (code_file=False) so opaque tokens with no recognized -# vendor prefix (e.g. ``MY_SERVICE_TOKEN=abc123randomstring``) are still -# masked. For all other commands, code_file=True is used to avoid mangling -# legitimate source/config dumps (``MAX_TOKENS=100``, ``"apiKey": "x"`` -# fixtures, ``postgresql://{user}`` f-string templates). See issue #43025. +# Commands whose stdout is an env-var dump (KEY=value lines), NOT source code. +# Terminal redaction runs the ENV-assignment pass (code_file=False) for these so +# opaque tokens with no vendor prefix (``MY_SERVICE_TOKEN=abc123…``) are still +# masked; everything else uses code_file=True to avoid mangling source/config +# dumps (``MAX_TOKENS=100``, ``postgresql://{user}`` templates). _ENV_DUMP_COMMANDS = frozenset({"env", "printenv", "set", "export", "declare"}) -# Commands that read file contents to stdout. When the target is a ``.env`` -# file, the output is a credential dump — the same as ``printenv`` — so the -# ENV-assignment pass must run (code_file=False). Per AGENTS.md, ``.env`` is -# for secrets only; behavioral settings belong in config.yaml, so running -# the generic ENV redactor on ``.env`` content is the correct behavior. +# Commands that read file contents to stdout. A ``.env`` target is a credential +# dump (per AGENTS.md ``.env`` holds only secrets), so the ENV pass must run. _FILE_READ_COMMANDS = frozenset({ "cat", "head", "tail", "type", "bat", "less", "more", "nl", "zcat", "tac", "view", "batcat", }) -# Basenames that are treated as ``.env`` files for redaction purposes are -# imported at module top as ``_ENV_FILE_BASENAMES`` (see the -# ``agent.file_safety`` import). - def _command_reads_env_file(command: str | None) -> bool: - """Return True if ``command`` reads a ``.env`` file to stdout. + """True if ``command`` reads a ``.env``-style file (by basename) to stdout. - Detects file-read commands (``cat``, ``head``, ``tail``, etc.) where any - argument's basename is a ``.env``-style file. Template files - (``.env.example``, ``.env.sample``, ...) are not in the basename list and - therefore never match. Handles pipelines and command sequences. - - Conservative defense-in-depth, not a boundary — indirect reads - (``sudo cat .env``, ``/bin/cat .env``, ``$(cat .env)``, redirection, - ``sed``/``awk``/``xxd`` readers) are not detected, matching the - precedent of ``is_env_dump_command`` below. + Template files (``.env.example``) are not in the basename list. Handles + pipelines/sequences. Defense-in-depth, not a boundary: indirect reads + (``sudo cat .env``, ``$(cat .env)``, ``sed``/``awk`` readers) are not + detected, matching ``is_env_dump_command``. """ if not command: return False - segments = re.split(r"[|;&]+", command) - for seg in segments: - seg = seg.strip() - if not seg: - continue - # Use plain split() instead of shlex.split — shlex treats backslashes - # as escape chars, which mangles Windows paths (``C:\Users\...\.env``). - # We only need the command name and filename, so shell quoting is not - # a concern here. - tokens = seg.split() + for seg in re.split(r"[|;&]+", command): + # Plain split() rather than shlex: shlex treats backslashes as escapes + # and mangles Windows paths (``C:\Users\...\.env``); only the command + # name and filename matter here. + tokens = seg.strip().split() if not tokens or tokens[0] not in _FILE_READ_COMMANDS: continue - # Check all arguments (skip flags like -n, -A, etc.) for arg in tokens[1:]: if arg.startswith("-"): continue - # Strip shell quotes that plain split() leaves attached - # (``cat ".env"`` / ``cat '.env'``), then any leading path to - # get the basename. Handle both / and \. + # Strip quotes split() leaves attached, then any / or \ path prefix. arg = arg.strip("\"'") basename = arg.rsplit("/", 1)[-1].rsplit("\\", 1)[-1] if basename.lower() in _ENV_FILE_BASENAMES: @@ -1138,19 +874,15 @@ def _command_reads_env_file(command: str | None) -> bool: def is_env_dump_command(command: str | None) -> bool: - """Return True if ``command`` dumps environment variables to stdout. + """True if ``command`` dumps environment variables to stdout. - Detects ``env`` / ``printenv`` / ``set`` / ``export`` / ``declare`` as the - first token of any segment in a pipeline or sequence (``;`` / ``&&`` / - ``||`` / ``|``). Conservative: a parse failure or anything unrecognized - returns False (callers then fall back to the safer code_file=True path, - which still masks prefix-shaped keys). + Detects ``env``/``printenv``/``set``/``export``/``declare`` as the first + token of any pipeline/sequence segment. Conservative: anything unrecognized + returns False (callers fall back to the safer code_file=True path). """ if not command or not isinstance(command, str): return False - # Split on shell separators, then inspect the first token of each segment. - segments = re.split(r"[|;&]+", command) - for seg in segments: + for seg in re.split(r"[|;&]+", command): seg = seg.strip() if not seg: continue @@ -1166,25 +898,12 @@ def is_env_dump_command(command: str | None) -> bool: def redact_terminal_output( output: str, command: str | None = None, *, force: bool = False ) -> str: - """Redact secrets from terminal/process stdout. + """Redact secrets from terminal/process stdout — the single policy for ALL + terminal-output surfaces (foreground ``terminal`` and background ``process``). - Single redaction policy for ALL terminal-output surfaces — foreground - ``terminal`` results AND background ``process(action=poll/log/wait)`` - output — so they can't diverge. Picks ``code_file`` based on whether - ``command`` is an environment dump or reads a ``.env`` file: - - - env-dump command (``env``/``printenv``/``set``/``export``/``declare``) - → ``code_file=False`` so the ENV-assignment pass masks opaque tokens. - - file-read command targeting a ``.env`` file (``cat .env``, - ``head .env.local``, etc.) → ``code_file=False`` for the same reason. - Per AGENTS.md, ``.env`` files contain only secrets, so the generic - ENV pass is the right one (keys whose names carry no secret keyword - can still slip through it — same limit as the env-dump path). - - anything else (or unknown command) → ``code_file=True`` to avoid - false positives on source/config dumps. - - ``force=True`` bypasses the global ``security.redact_secrets`` preference - for safety boundaries that must never emit raw credentials. + ``code_file`` is False (ENV-assignment pass runs) only when ``command`` is an + env dump or reads a ``.env`` file; otherwise True to avoid false positives + on source/config dumps. ``force=True`` bypasses the global opt-out. """ if not output: return output @@ -1193,25 +912,14 @@ def redact_terminal_output( return redact_sensitive_text(output, force=force, code_file=code_file) -# Substrings used to gate ``_PREFIX_RE`` execution. If none of these appear in -# the input string, the prefix regex cannot match anything, so we skip it. -# False positives are fine (they just run the regex, which then matches -# nothing) — the bound is "no false negatives" and that holds because every -# pattern in ``_PREFIX_PATTERNS`` has at least one of these as a literal -# substring of its leading characters. -# -# Derived automatically from ``_PREFIX_PATTERNS`` at module load time so a -# future PR that adds a new prefix to the regex list can't silently break -# the screen. +# --------------------------------------------------------------------------- +# Prefix pre-screen — derived from _PREFIX_PATTERNS at load time so a new +# prefix can't silently break the gate. No false negatives: every pattern has +# its literal prefix as a substring of any match. +# --------------------------------------------------------------------------- def _extract_literal_prefix(pattern: str) -> str: - """Return the leading literal characters of a regex pattern. - - Stops at the first regex metacharacter (``[``, ``(``, ``\\``, ``.``, - ``?``, ``*``, ``+``, ``|``, ``{``, ``^``, ``$``). Returns the literal - that any match of the pattern MUST contain as a substring, so the - pre-screen never produces false negatives. - """ + """Leading literal chars of a regex (up to the first metacharacter).""" meta = "[(\\.?*+|{^$" for i, ch in enumerate(pattern): if ch in meta: @@ -1219,14 +927,23 @@ def _extract_literal_prefix(pattern: str) -> str: return pattern +def _skip_char_class(pattern: str, i: int) -> int: + """Given ``pattern[i] == "["``, return the index just past the closing ``]``.""" + i += 1 + if i < len(pattern) and pattern[i] == "]": + i += 1 + while i < len(pattern) and pattern[i] != "]": + if pattern[i] == "\\": + i += 1 + i += 1 + return i + + def _has_top_level_alternation(pattern: str) -> bool: """True if ``pattern`` contains a ``|`` outside any group or class. - A top-level alternation defeats the literal-prefix guarantee: - ``_extract_literal_prefix`` stops at ``|``, so for ``ab|.*`` it - returns ``ab`` even though the ``.*`` branch is not bound by that - prefix and matches anything. Grouped alternation after the prefix - (``ab(?:x|y)``) keeps the guarantee and stays allowed. + Defeats the literal-prefix guarantee: for ``ab|.*`` the prefix ``ab`` binds + only the first branch. Grouped alternation (``ab(?:x|y)``) stays allowed. """ depth = 0 i = 0 @@ -1236,13 +953,7 @@ def _has_top_level_alternation(pattern: str) -> bool: i += 2 continue if ch == "[": - i += 1 - if i < len(pattern) and pattern[i] == "]": - i += 1 - while i < len(pattern) and pattern[i] != "]": - if pattern[i] == "\\": - i += 1 - i += 1 + i = _skip_char_class(pattern, i) elif ch == "(": depth += 1 elif ch == ")": @@ -1256,19 +967,13 @@ def _has_top_level_alternation(pattern: str) -> bool: def _has_nested_unbounded_repeat(pattern: str) -> bool: """True if an unbounded quantifier applies to a group containing one. - ``(a+)+``, ``(?:x*)*``, ``(a{2,})+`` — the canonical catastrophic- - backtracking (ReDoS) shape. Registered patterns run against every log - line, tool output, and transcript chunk, so a pathological pattern from - a buggy plugin would stall the host process, not just the plugin. - - Detects structural nesting only; ambiguity between overlapping - alternation branches (``(a|aa)+``) is not statically detected and - remains the plugin author's responsibility. + ``(a+)+`` / ``(?:x*)*`` / ``(a{2,})+`` — the canonical ReDoS shape. Registered + patterns run on every log line and tool output, so a pathological plugin + pattern would stall the host. Structural nesting only; overlapping + alternation branches (``(a|aa)+``) are the plugin author's responsibility. """ def _unbounded_quantifier_follows(j: int) -> bool: - # Is pattern[j:] an unbounded quantifier (* + {m,}) for the atom - # that just ended at j? if j >= len(pattern): return False if pattern[j] in "*+": @@ -1280,8 +985,7 @@ def _has_nested_unbounded_repeat(pattern: str) -> bool: return body[:-1].isdigit() and body.endswith(",") return False - # Stack of flags: does the group at this depth contain an unbounded - # repeat? Index 0 is the top level. + # Per-depth flag: does the group at this depth contain an unbounded repeat? contains_unbounded = [False] i = 0 while i < len(pattern): @@ -1290,13 +994,7 @@ def _has_nested_unbounded_repeat(pattern: str) -> bool: i += 2 continue if ch == "[": - i += 1 - if i < len(pattern) and pattern[i] == "]": - i += 1 - while i < len(pattern) and pattern[i] != "]": - if pattern[i] == "\\": - i += 1 - i += 1 + i = _skip_char_class(pattern, i) elif ch == "(": contains_unbounded.append(False) elif ch == ")": @@ -1318,30 +1016,19 @@ _PREFIX_SUBSTRINGS = tuple( def _has_known_prefix_substring(text: str) -> bool: - """Return True if ``text`` contains any known credential prefix substring. - - Used as a cheap pre-check before invoking the expensive ``_PREFIX_RE``. - """ + """Cheap pre-check before the expensive ``_PREFIX_RE``.""" return any(p in text for p in _PREFIX_SUBSTRINGS) # --------------------------------------------------------------------------- # Plugin-registered redaction patterns # --------------------------------------------------------------------------- -# -# Every new vendor token format has historically required a core PR appending -# to ``_PREFIX_PATTERNS`` above (fw_, retaindb_, hsk-, mem0_, brv_, ...). -# This registry lets plugins add their provider's format instead. It is -# ADDITIVE-ONLY by design: a plugin can extend what gets masked but has no -# API to remove or weaken a built-in pattern, so a plugin can only ever -# over-redact, never expose. The operator's global opt-out -# (``security.redact_secrets: false`` / HERMES_REDACT_SECRETS) applies to -# plugin patterns exactly as it does to built-ins. - -# Keyed by registration source (e.g. "plugin:my-plugin") so the plugin -# lifecycle/ownership-ledger work (#64229) has a clean seam to drop ONE -# plugin's patterns on unload. There is deliberately no public removal -# API — additive-only stands; unload is a host-owned lifecycle concern. +# Lets plugins add their provider's token format instead of a core PR. ADDITIVE- +# ONLY by design: a plugin can extend what gets masked but has no API to remove +# or weaken a built-in, so it can only over-redact, never expose. The operator's +# global opt-out applies to plugin patterns exactly as to built-ins. +# Keyed by registration source ("plugin:my-plugin") so plugin unload has a clean +# seam to drop ONE plugin's patterns; unload is a host-owned lifecycle concern. _PLUGIN_PREFIX_PATTERNS: dict = {} _registry_lock = threading.Lock() @@ -1354,49 +1041,52 @@ def _plugin_patterns() -> list: def _rebuild_prefix_matcher() -> None: """Recompile the prefix alternation and pre-screen substrings. - ``redact_sensitive_text`` and ``_mask_token_nonreusable`` look these - globals up at call time, so swapping the module attributes (atomic - under the GIL) propagates immediately to every caller. + Callers look these globals up at call time, so swapping the module + attributes (atomic under the GIL) propagates immediately. """ global _PREFIX_RE, _PREFIX_SUBSTRINGS combined = _PREFIX_PATTERNS + _plugin_patterns() - _PREFIX_RE = re.compile( - r"(? reject when True, warning message with (source, pattern) args). +_PATTERN_REJECT_RULES = ( + ( + _has_top_level_alternation, + "%s: skipping redaction pattern %r — top-level alternation " + "escapes the literal-prefix guarantee (in 'ab|.*' the " + "prefix binds only the first branch); wrap alternation in " + "a group after the prefix, e.g. 'ab(?:x|y)'", + ), + ( + _has_nested_unbounded_repeat, + "%s: skipping redaction pattern %r — nested unbounded " + "quantifiers (e.g. '(a+)+') can backtrack catastrophically, " + "and registered patterns run on every log line and tool " + "output", + ), + ( + lambda pattern: len(_extract_literal_prefix(pattern)) < 2, + "%s: skipping redaction pattern %r — must start with at " + "least 2 literal characters (needed for the pre-screen " + "substring gate)", + ), +) + + def register_redaction_patterns(patterns, source: str = "plugin") -> int: """Additively register credential-token regexes with the redaction engine. - Each accepted pattern joins the vendor-prefix alternation used by - ``redact_sensitive_text`` (same masking, same head/tail rules, same - non-reusable sentinel on ``file_read``) — everywhere built-in patterns - apply: logs, terminal output, transport errors, transcripts. + Accepted patterns join the vendor-prefix alternation everywhere built-ins + apply (same masking, same ``file_read`` sentinel). Invalid entries are + warned and skipped, never raised — a broken plugin must not break startup. + Each pattern must: be a non-empty string that compiles; have no top-level + alternation; not nest unbounded quantifiers (ReDoS); start with >= 2 literal + chars (pre-screen anchor; also rules out ``.*``). Duplicates are skipped. - Per-pattern validation (invalid entries are warned and skipped, never - raised — a broken plugin must not break startup): - - * must be a non-empty string that compiles as a regex; - * must not contain a top-level alternation (``ab|.*`` would escape - the literal-prefix guarantee below through its unprefixed branch; - grouped alternation after the prefix, ``ab(?:x|y)``, is allowed); - * must not nest unbounded quantifiers (``(a+)+``-style patterns can - backtrack catastrophically, and registered patterns run against - every log line and tool output — see - ``_has_nested_unbounded_repeat``); - * must start with at least 2 literal characters (the pre-screen - substring gate in ``_has_known_prefix_substring`` needs a literal - anchor; it also structurally rules out redact-everything patterns - like ``.*``); - * duplicates of built-in or already-registered patterns are skipped. - - Args: - patterns: iterable of regex strings (e.g. ``[r"nvapi-[A-Za-z0-9_-]{20,}"]``). - source: attribution label for log lines (e.g. ``"plugin:my-plugin"``). - - Returns: - The number of patterns actually accepted. + Returns the number of patterns actually accepted. """ accepted = [] for pattern in patterns or []: @@ -1412,31 +1102,13 @@ def register_redaction_patterns(patterns, source: str = "plugin") -> int: source, pattern, exc, ) continue - if _has_top_level_alternation(pattern): - logger.warning( - "%s: skipping redaction pattern %r — top-level alternation " - "escapes the literal-prefix guarantee (in 'ab|.*' the " - "prefix binds only the first branch); wrap alternation in " - "a group after the prefix, e.g. 'ab(?:x|y)'", - source, pattern, - ) - continue - if _has_nested_unbounded_repeat(pattern): - logger.warning( - "%s: skipping redaction pattern %r — nested unbounded " - "quantifiers (e.g. '(a+)+') can backtrack catastrophically, " - "and registered patterns run on every log line and tool " - "output", - source, pattern, - ) - continue - if len(_extract_literal_prefix(pattern)) < 2: - logger.warning( - "%s: skipping redaction pattern %r — must start with at " - "least 2 literal characters (needed for the pre-screen " - "substring gate)", - source, pattern, - ) + rejected = False + for reject, message in _PATTERN_REJECT_RULES: + if reject(pattern): + logger.warning(message, source, pattern) + rejected = True + break + if rejected: continue if pattern in _PREFIX_PATTERNS or pattern in _plugin_patterns() or pattern in accepted: logger.debug("%s: redaction pattern %r already registered", source, pattern) @@ -1460,31 +1132,8 @@ def _reset_plugin_redaction_patterns() -> None: _rebuild_prefix_matcher() -_HTTP_METHOD_SUBSTRINGS = ( - "GET ", - "POST ", - "PUT ", - "PATCH ", - "DELETE ", - "HEAD ", - "OPTIONS ", - "TRACE ", - "CONNECT ", -) - - -def _has_http_method_substring(text: str) -> bool: - """Cheap pre-check before scanning for access-log request targets.""" - upper = text.upper() - return any(method in upper for method in _HTTP_METHOD_SUBSTRINGS) - - class RedactingFormatter(logging.Formatter): """Log formatter that redacts secrets from all log messages.""" - def __init__(self, fmt=None, datefmt=None, style='%', **kwargs): - super().__init__(fmt, datefmt, style, **kwargs) - def format(self, record: logging.LogRecord) -> str: - original = super().format(record) - return redact_sensitive_text(original) + return redact_sensitive_text(super().format(record)) diff --git a/agent/replay_cleanup.py b/agent/replay_cleanup.py index 780fe761bb..5423472eac 100644 --- a/agent/replay_cleanup.py +++ b/agent/replay_cleanup.py @@ -1,18 +1,11 @@ """Replay-history sanitization shared across resume code paths. -When a session's last turn dies mid-tool-loop — the process is killed by a -restart/shutdown command, a stale-timeout fires, or an interrupt lands before -the tool result is written — the persisted transcript can end with a dangling -``assistant(tool_calls)`` (no matching ``tool`` answer) or an interrupted -``assistant→tool`` block. On resume the model sees that broken tail and -re-issues the unanswered call, producing an endless "thinking"/reboot loop -(#49201, #29086). - -These pure helpers strip those tails before the history is replayed to the -model. They were originally local to ``gateway/run.py`` (which fixed the -messaging-gateway path) and are extracted here so every resume surface — the -messaging gateway AND the TUI/WebUI gateway — shares the same cleanup instead -of the WebUI path silently skipping it. +A session whose last turn died mid-tool-loop (process killed by a restart +command, stale timeout, interrupt before the tool result was written) persists +a dangling ``assistant(tool_calls)`` or interrupted ``assistant→tool`` tail. On +resume the model re-issues the unanswered call → endless "thinking"/reboot loop. +These pure helpers strip those tails before replay, for EVERY resume surface +(messaging gateway and TUI/WebUI gateway alike). """ from __future__ import annotations @@ -39,16 +32,36 @@ def is_interrupted_tool_result(content: Any) -> bool: return False +def _call_name(call: Dict[str, Any]) -> str: + return str((call.get("function") or {}).get("name") or "") + + +def _call_id(call: Dict[str, Any]) -> str: + return str(call.get("id") or call.get("call_id") or "") + + +def _any_side_effecting(calls: List[Dict[str, Any]]) -> bool: + return any(tool_may_have_side_effect(_call_name(call)) for call in calls) + + +def _orphan_recovery(name: str, unknown_text: str, none_text: str) -> tuple: + """(effect_disposition, content) for an interrupted/dangling call named ``name``.""" + if tool_may_have_side_effect(name): + return "unknown", unknown_text + return "none", none_text + + def strip_interrupted_tool_tails( agent_history: List[Dict[str, Any]], ) -> List[Dict[str, Any]]: """Strip interrupted assistant→tool sequences from replay history. - Older interrupted gateway turns can be followed by a queued real user - message, so the interrupted assistant/tool block is not necessarily the - final tail by the time we rebuild replay history. Remove any contiguous - assistant(tool_calls) + tool-result block that contains an interrupted tool - result, while preserving successful tool-call sequences intact. + The interrupted block is not necessarily the final tail (a queued real user + message may follow it), so every contiguous assistant(tool_calls)+tool-result + block containing an interrupted result is handled; successful sequences stay + intact. Read-only blocks are dropped; blocks with a side-effecting call are + KEPT with the interrupted results rewritten as orphan-recovery notices, since + the effect may already have happened and erasing it would hide that. """ if not agent_history: return agent_history @@ -69,18 +82,8 @@ def strip_interrupted_tool_tails( for m in tool_results ): calls = msg.get("tool_calls") or [] - if any( - tool_may_have_side_effect( - str((call.get("function") or {}).get("name") or "") - ) - for call in calls - ): - call_names = { - str(call.get("id") or call.get("call_id") or ""): str( - (call.get("function") or {}).get("name") or "" - ) - for call in calls - } + if _any_side_effecting(calls): + call_names = {_call_id(call): _call_name(call) for call in calls} cleaned.append(msg) for tool_result in tool_results: if not is_interrupted_tool_result(tool_result.get("content", "")): @@ -88,14 +91,11 @@ def strip_interrupted_tool_tails( continue recovered = dict(tool_result) name = call_names.get(str(tool_result.get("tool_call_id") or ""), "") - recovered["effect_disposition"] = ( - "unknown" if tool_may_have_side_effect(name) else "none" - ) - recovered["content"] = ( + recovered["effect_disposition"], recovered["content"] = _orphan_recovery( + name, "[Orphan recovery: interrupted side-effecting tool may have " - "executed; its effect is UNKNOWN. Inspect state before retrying.]" - if recovered["effect_disposition"] == "unknown" - else "[Orphan recovery: interrupted read-only tool did not complete.]" + "executed; its effect is UNKNOWN. Inspect state before retrying.]", + "[Orphan recovery: interrupted read-only tool did not complete.]", ) cleaned.append(recovered) i = j @@ -122,24 +122,13 @@ def strip_dangling_tool_call_tail( ) -> List[Dict[str, Any]]: """Strip a trailing ``assistant(tool_calls)`` block left with NO answers. - When a tool call itself kills the gateway process (``docker restart``, - ``systemctl restart``, ``kill``, ``hermes gateway restart``), the process - is terminated by SIGKILL *mid-call* — before the tool result is ever - written and before the orderly shutdown rewind - (``_drop_trailing_empty_response_scaffolding``) can run. The last thing - persisted is the ``assistant`` message that issued the ``tool_calls``, - with zero matching ``tool`` rows. - - On resume the model sees an unanswered tool call at the tail and naturally - re-issues it — which restarts the gateway again, producing the infinite - reboot loop in #49201. ``strip_interrupted_tool_tails`` does not catch - this because there is no tool result to inspect for an interrupt marker. - - This strips that dangling tail at the source so there is nothing for the - model to re-execute. It only acts when the tail is an - ``assistant(tool_calls)`` whose calls have NO corresponding ``tool`` - results — a completed assistant→tool pair (any tool answers present) is - left untouched so genuine mid-progress tool loops still resume. + A tool call that kills the gateway process itself (``docker restart``, + ``hermes gateway restart``) is SIGKILLed mid-call, before any tool result or + the orderly shutdown rewind; the persisted tail is the assistant message with + zero matching ``tool`` rows, which ``strip_interrupted_tool_tails`` cannot + detect (no result to inspect). Only acts when the tail has NO tool answers — + a partially answered block still resumes. Read-only tails are dropped; + side-effecting ones get synthetic UNKNOWN-effect results instead of erasure. """ if not agent_history: return agent_history @@ -153,26 +142,18 @@ def strip_dangling_tool_call_tail( return agent_history tool_calls = last.get("tool_calls") or [] - if any( - tool_may_have_side_effect( - str((call.get("function") or {}).get("name") or "") - ) - for call in tool_calls - ): + if _any_side_effecting(tool_calls): recovered = list(agent_history) for call in tool_calls: - function = call.get("function") or {} - name = str(function.get("name") or "unknown") - call_id = str(call.get("id") or call.get("call_id") or "") - disposition = "unknown" if tool_may_have_side_effect(name) else "none" - content = ( + name = str((call.get("function") or {}).get("name") or "unknown") + disposition, content = _orphan_recovery( + name, "[Orphan recovery: this tool may have executed before Hermes stopped; " - "its effect is UNKNOWN. Inspect current state before retrying.]" - if disposition == "unknown" - else "[Orphan recovery: this read-only tool did not complete and had no effect.]" + "its effect is UNKNOWN. Inspect current state before retrying.]", + "[Orphan recovery: this read-only tool did not complete and had no effect.]", ) recovered.append(make_tool_result_message( - name, content, call_id, effect_disposition=disposition, + name, content, _call_id(call), effect_disposition=disposition, )) logger.warning( "Recovered dangling side-effecting tool call(s) as UNKNOWN instead of erasing them" @@ -189,31 +170,23 @@ def strip_dangling_tool_call_tail( def sanitize_replay_history( agent_history: List[Dict[str, Any]], ) -> List[Dict[str, Any]]: - """Apply both replay-tail strippers in the canonical order. - - Convenience entry point for resume code paths: removes interrupted - assistant→tool blocks anywhere in the history, then removes a dangling - unanswered ``assistant(tool_calls)`` tail. Returns the same list object - when there is nothing to strip. - """ + """Both replay-tail strippers in canonical order (interrupted blocks, then + dangling tail). Returns the same list object when nothing is stripped.""" if not agent_history: return agent_history return strip_dangling_tool_call_tail(strip_interrupted_tool_tails(agent_history)) # ────────────────────────────────────────────────────────────────────── -# Stale dangerous-confirmation text expiry (#59607) +# Stale dangerous-confirmation text expiry # ────────────────────────────────────────────────────────────────────── -# How long a high-risk confirmation phrase remains valid. -# Short on purpose: dangerous side effects should not survive any restart -# or session resumption gap. The user can always re-confirm if needed. +# Short on purpose: a dangerous confirmation must not survive any restart or +# resume gap. The user can always re-confirm. _DANGEROUS_CONFIRMATION_EXPIRY_SECONDS = 60.0 -# Confirmation phrases that unlock destructive host actions. -# Substring match (case-insensitive) so that user variants (e.g. trailing -# punctuation, additional context) still match. Add new patterns here when -# new high-risk actions are introduced. +# Confirmation phrases that unlock destructive host actions; case-insensitive +# substring match so trailing punctuation / extra context still matches. _DANGEROUS_CONFIRMATION_PATTERNS: tuple = ( "confirm forced restart", "confirm forced reboot", @@ -229,9 +202,8 @@ _DANGEROUS_CONFIRMATION_PATTERNS: tuple = ( "確認重啟", ) -# Replacement text for an expired confirmation. Redacting in place (rather -# than deleting the message) preserves strict user/assistant role -# alternation in the replayed history. +# Redacting in place (rather than deleting the message) preserves strict +# user/assistant role alternation in the replayed history. _EXPIRED_CONFIRMATION_SENTINEL = ( "[A high-risk confirmation previously given here has EXPIRED and must " "not be acted on. Ask the user to re-confirm explicitly before " @@ -240,12 +212,7 @@ _EXPIRED_CONFIRMATION_SENTINEL = ( def is_dangerous_confirmation(content: Any) -> bool: - """Return True if a user-message text matches a known dangerous confirmation. - - Used by ``strip_stale_dangerous_confirmations`` to decide which - transcript rows to expire. Substring + case-insensitive so that - ``"Please confirm forced restart, the host is critical"`` still matches. - """ + """True if user-message text contains a known dangerous confirmation phrase.""" if not isinstance(content, str): return False text = content.strip().lower() @@ -258,38 +225,15 @@ def strip_stale_dangerous_confirmations( now: float, expiry_seconds: float = _DANGEROUS_CONFIRMATION_EXPIRY_SECONDS, ) -> List[Dict[str, Any]]: - """Expire stale dangerous-confirmation text in user messages (#59607). + """Expire stale dangerous-confirmation text in user messages. - When a high-risk side effect (e.g. host restart via ``shutdown.exe``) - runs, the user's plain-text confirmation phrase is persisted in the - conversation transcript. If the host restart killed the gateway - process before the assistant's tool result was written, the - transcript tail ends on the assistant's text response — and the - dangerous confirmation text remains in the user role. - - On the next inbound message — possibly a casual "are you there?" from - the user minutes later — the LLM sees the stale confirmation and may - interpret the new turn as a fresh re-confirmation, re-executing the - destructive action. This is the failure mode reported in #59607. - - Expired confirmations are REDACTED IN PLACE, not removed: deleting a - user message from the incident tail (``user(confirm) → - assistant("OK, restarting")``) would leave two consecutive assistant - messages, violating the strict role-alternation invariant providers - enforce. The message survives with its role intact; only the trigger - text is replaced by a sentinel that tells the model the confirmation - has expired. - - Messages without a timestamp are left untouched (backward - compatibility: legacy transcripts and in-memory test scaffolding have - no timestamps). User messages that contain dangerous confirmation - text but are within the expiry window are also left untouched — they - represent a fresh confirmation that has not yet been acted on. - - Complements 75ed07ace (which strips the *assistant* side of the - broken tail) by handling the *user* side: a stale plain-text - confirmation that the assistant has not yet responded to in a way - the resume logic recognises. + If a host restart killed the gateway before the tool result was written, the + user's confirmation phrase survives in the transcript; a casual "are you + there?" minutes later can read to the model as a fresh re-confirmation and + re-execute the destructive action. Expired confirmations are REDACTED IN + PLACE (deleting the message would leave two consecutive assistant turns). + Messages without a timestamp (legacy transcripts, test scaffolding) and + confirmations still inside the expiry window are left untouched. """ if not agent_history: return agent_history @@ -312,10 +256,8 @@ def strip_stale_dangerous_confirmations( ) redacted = dict(msg) redacted["content"] = _EXPIRED_CONFIRMATION_SENTINEL - # Drop the api_content sidecar: it carries the exact bytes - # previously sent — i.e. the dangerous confirmation this - # redaction exists to expire. Replaying it verbatim would - # undo the redaction on the wire. + # The api_content sidecar carries the exact bytes previously sent + # — the confirmation itself; replaying it would undo the redaction. drop_stale_api_content(redacted) cleaned.append(redacted) continue diff --git a/agent/runtime_cwd.py b/agent/runtime_cwd.py index bcd776e65b..905794bd2c 100644 --- a/agent/runtime_cwd.py +++ b/agent/runtime_cwd.py @@ -1,13 +1,10 @@ """Single source of truth for the agent working directory. `TERMINAL_CWD` is the runtime carrier for the configured working directory -(design #19214/#19242: `terminal.cwd` is bridged once to `TERMINAL_CWD` at -gateway/cron startup). The local-CLI backend deliberately leaves it unset and -relies on the launch dir. Reading it in one place keeps the system prompt, the -tool surfaces, and context-file discovery agreeing on where the agent lives. - -Multi-session gateways can pin a logical cwd via the `_SESSION_CWD` -contextvar; CLI/cron fall through to `TERMINAL_CWD`/launch cwd. +(`terminal.cwd` is bridged to it once at gateway/cron startup; the local CLI +leaves it unset and relies on the launch dir). Reading it in one place keeps the +system prompt, tool surfaces, and context-file discovery agreeing on where the +agent lives. Multi-session gateways can pin a logical cwd via `_SESSION_CWD`. """ import logging @@ -22,18 +19,15 @@ _UNSET: Any = object() _SESSION_CWD: ContextVar = ContextVar("HERMES_SESSION_CWD", default=_UNSET) -# The Python package/source root (this file lives at /agent/runtime_cwd.py). -# When a backend is launched from, or self-spawns into, this tree (the desktop -# app default), an os.getcwd() fallback would inject this repo's contributor -# AGENTS.md as authoritative project context. Context discovery must never -# resolve here. +# The package/source root (/agent/runtime_cwd.py). A backend launched from +# or self-spawned into this tree (desktop default) must never let an os.getcwd() +# fallback inject this repo's contributor AGENTS.md as project context. _PACKAGE_ROOT = Path(__file__).resolve().parent.parent def _is_install_tree(p: Path) -> bool: - # True only when p IS the package root or sits inside it. Ancestors of the - # package root (a user home that happens to contain the checkout, a --user - # site-packages parent) are legitimate workspaces and must not be blocked. + """True only when ``p`` IS the package root or sits inside it — ancestors + (a home dir containing the checkout) are legitimate workspaces.""" try: p = p.resolve() except Exception: @@ -61,9 +55,9 @@ def _terminal_cwd_env() -> str: """Scope-aware TERMINAL_CWD read (tools.terminal_scope.terminal_env). Under gateway multiplexing the per-turn terminal scope carries the active - profile's cwd; the process-global env var may hold another profile's - value. Only an import failure falls back: an active refusal scope must - raise, not silently resolve the launch profile's cwd. + profile's cwd; the process-global env var may hold another profile's. Only + an ImportError falls back: an active refusal scope must raise, not silently + resolve the launch profile's cwd. """ try: from tools.terminal_scope import terminal_env @@ -75,51 +69,49 @@ def _terminal_cwd_env() -> str: def scope_terminal_cwd() -> str: """Public wrapper — the scope-aware TERMINAL_CWD value (may be empty). - Shared by agent_init / skill_utils / code_execution_tool so every cwd - consumer reads through the per-turn terminal scope under gateway - multiplexing instead of the process-global env var. + Shared by agent_init / skill_utils / code_execution_tool so every cwd consumer + reads through the per-turn terminal scope under gateway multiplexing. """ return _terminal_cwd_env() -def resolve_agent_cwd() -> Path: +def _resolve_configured_cwd(*, override_is_final: bool) -> Path | None: + """Session override, then TERMINAL_CWD; each validated as a real directory. + + ``override_is_final``: a set-but-missing session override yields None + instead of falling through to TERMINAL_CWD. + """ override = _session_cwd_override() if override: p = Path(override).expanduser() if p.is_dir(): return p logger.warning("configured working directory does not exist: %s", override) + if override_is_final: + return None raw = _terminal_cwd_env().strip() if raw: p = Path(raw).expanduser() if p.is_dir(): return p logger.warning("TERMINAL_CWD does not exist: %s", raw) - return Path(os.getcwd()) + return None + + +def resolve_agent_cwd() -> Path: + """Configured cwd, else the launch dir (os.getcwd() — its OSError on a + deleted cwd deliberately propagates; the caller owns that guard).""" + p = _resolve_configured_cwd(override_is_final=False) + return p if p is not None else Path(os.getcwd()) def resolve_context_cwd() -> Path | None: - # None means "no configured cwd": build_context_files_prompt then falls back - # to the launch dir (os.getcwd()), correct for a local CLI launched inside a - # real project. A configured path is validated here (previously it was passed - # through unchecked, diverging from resolve_agent_cwd). An explicitly - # configured path is otherwise honored verbatim — including the Hermes - # source tree itself, which is a legitimate workspace when the user is - # developing Hermes (per-surface policy for fallback-picked directories - # lives in build_context_files_prompt; see #64590). - override = _session_cwd_override() - if override: - p = Path(override).expanduser() - if not p.is_dir(): - logger.warning("configured working directory does not exist: %s", override) - else: - return p - return None - raw = _terminal_cwd_env().strip() - if raw: - p = Path(raw).expanduser() - if not p.is_dir(): - logger.warning("TERMINAL_CWD does not exist: %s", raw) - else: - return p - return None + """Configured cwd for context-file discovery, or None for "no configured cwd". + + None makes build_context_files_prompt fall back to the launch dir (correct + for a local CLI launched inside a real project). A configured path is + validated here; an existing one is honored verbatim — including the Hermes + source tree itself, a legitimate workspace when developing Hermes + (fallback-directory policy lives in build_context_files_prompt). + """ + return _resolve_configured_cwd(override_is_final=True) diff --git a/agent/skill_bundles.py b/agent/skill_bundles.py index 2727b5837c..957bc23fe6 100644 --- a/agent/skill_bundles.py +++ b/agent/skill_bundles.py @@ -1,87 +1,38 @@ """Skill bundles — aliases that load multiple skills under one slash command. -A skill bundle is a small YAML file that names a set of skills to load -together. Invoking ``/`` from the CLI or gateway loads every -referenced skill's full content into a single user message, the same way -``/`` does — but for N skills at once. - -Storage -------- -Bundles live in ``~/.hermes/skill-bundles/*.yaml`` (and the equivalent -profile-aware directory under ``HERMES_HOME``). Each file looks like:: - - name: backend-dev - description: Backend feature work — code review, testing, PR workflow. - skills: - - github-code-review - - test-driven-development - - github-pr-workflow - instruction: | - Optional extra guidance to inject above the skill bodies. - -The file's stem is treated as a fallback name when ``name:`` is absent, so -dropping a YAML into the directory is enough to register a new bundle. - -Conflict resolution -------------------- -If a bundle and a skill share the same slash name, the bundle wins. The -slash command dispatch checks bundles first, then falls back to skills. -This is the intended behavior — a user who names a bundle ``research`` -explicitly wants ``/research`` to mean their bundle, not whatever skill -happens to share the slug. - -Public API ----------- -- :func:`get_skill_bundles` — return ``{"/slug": bundle_info}`` -- :func:`resolve_bundle_command_key` — map a user-typed command to its slug -- :func:`build_bundle_invocation_message` — produce the full user message -- :func:`reload_bundles` — re-scan disk and return a diff -- :func:`list_bundles` — return rich info for display (``hermes bundles``) -- :func:`save_bundle` / :func:`delete_bundle` — file-level operations +Bundles are YAML files in ``/skill-bundles/`` (``name``, +``description``, ``skills: [...]``, optional ``instruction``; the file stem is +the fallback name). ``/`` loads every member skill into one user +message. If a bundle and a skill share a slug, the bundle wins — slash dispatch +checks bundles first, on purpose. """ from __future__ import annotations import logging import os -import re from pathlib import Path from typing import Any, Dict, List, Optional, Tuple import yaml from hermes_constants import get_hermes_home +from agent.skill_commands import diff_command_snapshots, slugify_skill_name as _slugify logger = logging.getLogger(__name__) -# Slug normalization — matches agent/skill_commands.py so a bundle and a -# skill called "Foo Bar" both resolve to "/foo-bar". -_BUNDLE_INVALID_CHARS = re.compile(r"[^a-z0-9-]") -_BUNDLE_MULTI_HYPHEN = re.compile(r"-{2,}") - _bundles_cache: Dict[str, Dict[str, Any]] = {} _bundles_cache_mtime: Optional[float] = None def _bundles_dir() -> Path: - """Return the canonical bundles directory under HERMES_HOME. - - Honors ``HERMES_BUNDLES_DIR`` for tests; falls back to - ``/skill-bundles``. - """ + """Bundles directory: ``HERMES_BUNDLES_DIR`` override (tests) or ``/skill-bundles``.""" override = os.environ.get("HERMES_BUNDLES_DIR") if override: return Path(override).expanduser() return get_hermes_home() / "skill-bundles" -def _slugify(name: str) -> str: - cmd = name.lower().replace(" ", "-").replace("_", "-") - cmd = _BUNDLE_INVALID_CHARS.sub("", cmd) - cmd = _BUNDLE_MULTI_HYPHEN.sub("-", cmd).strip("-") - return cmd - - def _iter_bundle_files() -> List[Path]: base = _bundles_dir() if not base.exists(): @@ -93,19 +44,9 @@ def _iter_bundle_files() -> List[Path]: def _max_mtime(files: List[Path]) -> float: - """Highest mtime across the bundle files plus the dir itself. - - Watching the directory mtime catches deletions; watching individual - files catches edits. Together they're a cheap freshness check. - """ - base = _bundles_dir() + """Highest mtime across the bundle files plus the dir itself (dir mtime catches deletions).""" mtimes = [] - if base.exists(): - try: - mtimes.append(base.stat().st_mtime) - except OSError: - pass - for f in files: + for f in [_bundles_dir(), *files]: try: mtimes.append(f.stat().st_mtime) except OSError: @@ -114,11 +55,7 @@ def _max_mtime(files: List[Path]) -> float: def _load_bundle_file(path: Path) -> Optional[Dict[str, Any]]: - """Parse a single bundle YAML file. Returns ``None`` on any error. - - Errors are logged at WARNING level. We don't raise — a broken bundle - shouldn't take down slash command discovery. - """ + """Parse one bundle YAML; ``None`` (logged) on any error so a broken bundle can't break discovery.""" try: raw = path.read_text(encoding="utf-8") except OSError as exc: @@ -166,12 +103,7 @@ def _load_bundle_file(path: Path) -> Optional[Dict[str, Any]]: def scan_bundles() -> Dict[str, Dict[str, Any]]: - """Scan the bundles directory and rebuild the cache. - - Returns the same mapping as :func:`get_skill_bundles` — ``"/slug"`` → - bundle info dict. Later bundles with a duplicate slug are skipped with - a warning (first wins, alphabetical order). - """ + """Rebuild the ``"/slug"`` -> bundle info cache; duplicate slugs keep the first (alphabetical).""" global _bundles_cache, _bundles_cache_mtime files = _iter_bundle_files() out: Dict[str, Dict[str, Any]] = {} @@ -193,25 +125,15 @@ def scan_bundles() -> Dict[str, Dict[str, Any]]: def get_skill_bundles() -> Dict[str, Dict[str, Any]]: - """Return the current bundle mapping, rescanning when disk changed. - - Cheap to call repeatedly: only rescans when the bundles directory or - any bundle file's mtime is newer than the cached snapshot. - """ - files = _iter_bundle_files() - current_mtime = _max_mtime(files) + """Current bundle mapping; rescans only when a bundle file or the dir mtime changed.""" + current_mtime = _max_mtime(_iter_bundle_files()) if not _bundles_cache or _bundles_cache_mtime != current_mtime: scan_bundles() return _bundles_cache def resolve_bundle_command_key(command: str) -> Optional[str]: - """Resolve a user-typed command to its canonical bundle slash key. - - Hyphens and underscores are treated interchangeably to mirror the - skill-command behavior (Telegram converts hyphens to underscores in - bot command names). - """ + """Resolve a user-typed command to its ``/slug`` key (``_`` ≡ ``-``, as Telegram rewrites hyphens).""" if not command: return None cmd_key = f"/{command.replace('_', '-')}" @@ -219,35 +141,17 @@ def resolve_bundle_command_key(command: str) -> Optional[str]: def reload_bundles() -> Dict[str, Any]: - """Re-scan the bundles directory and return a diff. - - Mirrors :func:`agent.skill_commands.reload_skills` so callers can use - the same display logic. Returns a dict with ``added``, ``removed``, - ``unchanged``, and ``total`` keys. - """ + """Re-scan and return an ``added``/``removed``/``unchanged``/``total`` diff (same shape as reload_skills).""" def _snapshot(cmds: Dict[str, Dict[str, Any]]) -> Dict[str, str]: return {k.lstrip("/"): (v or {}).get("description", "") for k, v in cmds.items()} before = _snapshot(_bundles_cache) - new = scan_bundles() - after = _snapshot(new) - - added_names = sorted(set(after) - set(before)) - removed_names = sorted(set(before) - set(after)) - unchanged = sorted(set(after) & set(before)) - - return { - "added": [{"name": n, "description": after[n]} for n in added_names], - "removed": [{"name": n, "description": before[n]} for n in removed_names], - "unchanged": unchanged, - "total": len(after), - } + return diff_command_snapshots(before, _snapshot(scan_bundles())) def list_bundles() -> List[Dict[str, Any]]: """Return a sorted list of bundle info dicts for display.""" - bundles = get_skill_bundles() - return sorted(bundles.values(), key=lambda b: b["slug"]) + return sorted(get_skill_bundles().values(), key=lambda b: b["slug"]) def build_bundle_invocation_message( @@ -256,34 +160,20 @@ def build_bundle_invocation_message( task_id: str | None = None, platform: str | None = None, ) -> Optional[Tuple[str, List[str], List[str]]]: - """Build the user message content for a bundle slash command invocation. + """Build the user message for a bundle invocation. - Returns ``(message, loaded_skill_names, missing_skill_names)`` or - ``None`` if the bundle wasn't found. - - A bundle that references skills the user doesn't have installed still - loads — the agent gets a note about which ones were skipped. This is - the same forgiving stance ``build_preloaded_skills_prompt`` uses for - ``-s`` CLI preloading. - - Disabled skills are also skipped: bundles load members via - ``_load_skill_payload`` directly, bypassing the scan-time disabled - filter in ``get_skill_commands()``, so the disabled list must be - re-applied here. ``platform`` scopes the check to a specific - platform's ``skills.platform_disabled`` config (gateway dispatch - passes it explicitly because the gateway handles multiple platforms - in one process); when *None*, the platform resolves from session env - vars and the global disabled list still applies. Mirrors the - stacked-skill gate in gateway dispatch (#58888). + Returns ``(message, loaded_skill_names, missing_skill_names)`` or ``None`` + if the bundle wasn't found. Uninstalled members are skipped with a note. + Disabled members are skipped too: bundles load via ``_load_skill_payload``, + bypassing the scan-time disabled filter, so the list is re-applied here. + ``platform`` scopes that check (gateway passes it; None resolves from env). """ - bundles = get_skill_bundles() - info = bundles.get(cmd_key) + info = get_skill_bundles().get(cmd_key) if not info: return None - # Late import to avoid pulling tools/* at module import time and to - # keep skill_bundles cheap to import in test environments. - from agent.skill_commands import _load_skill_payload, _build_skill_message + # Late import keeps skill_bundles cheap to import (no tools/* at import time). + from agent.skill_commands import _load_skill_payload, _render_skill_block, _scaffold_header try: from agent.skill_utils import get_disabled_skill_names @@ -298,10 +188,8 @@ def build_bundle_invocation_message( seen: set[str] = set() bundle_name = info["name"] - skills = info["skills"] - extra_instruction = info.get("instruction") or "" - for skill_id in skills: + for skill_id in info["skills"]: identifier = (skill_id or "").strip() if not identifier or identifier in seen: continue @@ -311,66 +199,36 @@ def build_bundle_invocation_message( if not loaded: missing.append(identifier) continue - loaded_skill, skill_dir, skill_name = loaded + skill_name = loaded[2] - # Per-platform / global disabled gate. Checked against the loaded - # skill's canonical name (identifiers may be paths or aliases). + # Gate on the loaded skill's canonical name (identifiers may be paths or aliases). if skill_name in disabled_names or identifier in disabled_names: disabled.append(skill_name or identifier) continue - try: - from tools.skill_usage import bump_use - bump_use(skill_name, task_id=task_id) - except Exception: - pass - - activation_note = ( - f'[Loaded as part of the "{bundle_name}" skill bundle.]' - ) - skill_blocks.append( - _build_skill_message( - loaded_skill, - skill_dir, - activation_note, - session_id=task_id, - ) - ) + skill_blocks.append(_render_skill_block( + loaded, + f'[Loaded as part of the "{bundle_name}" skill bundle.]', + task_id, + )) loaded_names.append(skill_name) if not skill_blocks: return None - # Header — tells the agent this is a bundle, lists the skills, and - # provides any author-supplied instruction. - header_lines = [ - f'[IMPORTANT: The user has invoked the "{bundle_name}" skill bundle, ' - f"loading {len(loaded_names)} skills together. Treat every skill below " - "as active guidance for this turn.]", - "", - f"Bundle: {bundle_name}", - f"Skills loaded: {', '.join(loaded_names)}", - ] - if missing: - header_lines.append(f"Skills missing (skipped): {', '.join(missing)}") - if disabled: - header_lines.append( - f"Skills disabled for this platform (skipped): {', '.join(disabled)}" - ) - if extra_instruction: - header_lines.extend(["", f"Bundle instruction: {extra_instruction}"]) - if user_instruction: - header_lines.extend( - ["", f"User instruction: {user_instruction}"] - ) - - header = "\n".join(header_lines) + header = _scaffold_header( + f'"{bundle_name}" skill bundle', + loaded_names, + lead_lines=[f"Bundle: {bundle_name}"], + missing=missing, + disabled=disabled, + extra_instruction=info.get("instruction") or "", + user_instruction=user_instruction, + ) return ("\n\n".join([header, *skill_blocks]), loaded_names, missing) -# --------------------------------------------------------------------------- -# File-level CRUD helpers — used by `hermes bundles` CLI subcommand. -# --------------------------------------------------------------------------- +# ── File-level CRUD — used by `hermes bundles` ───────────────────────────── def bundle_path_for(name: str) -> Path: @@ -388,10 +246,10 @@ def save_bundle( instruction: str = "", overwrite: bool = False, ) -> Path: - """Write a bundle to disk and invalidate the cache. + """Write a bundle to disk and refresh the cache. - Raises ``FileExistsError`` if the target exists and ``overwrite`` is - False. Raises ``ValueError`` if the inputs are unusable. + Raises ``FileExistsError`` if the target exists and not ``overwrite``; + ``ValueError`` for unusable inputs. """ name = (name or "").strip() if not name: @@ -415,15 +273,12 @@ def save_bundle( yaml.safe_dump(payload, sort_keys=False, allow_unicode=True), encoding="utf-8", ) - scan_bundles() # refresh cache + scan_bundles() return path def delete_bundle(name: str) -> Path: - """Delete a bundle by name. Returns the deleted path. - - Raises ``FileNotFoundError`` if the bundle doesn't exist. - """ + """Delete a bundle by name and return its path; ``FileNotFoundError`` if absent.""" path = bundle_path_for(name) if not path.exists(): raise FileNotFoundError(f"No bundle at {path}") @@ -434,5 +289,4 @@ def delete_bundle(name: str) -> Path: def get_bundle(name: str) -> Optional[Dict[str, Any]]: """Look up a bundle by name (slug-normalized).""" - slug = _slugify(name) - return get_skill_bundles().get(f"/{slug}") + return get_skill_bundles().get(f"/{_slugify(name)}") diff --git a/agent/skill_commands.py b/agent/skill_commands.py index 9851fdb750..b623af6bdf 100644 --- a/agent/skill_commands.py +++ b/agent/skill_commands.py @@ -1,8 +1,4 @@ -"""Shared slash command helpers for skills. - -Shared between CLI (cli.py) and gateway (gateway/run.py) so both surfaces -can invoke skills via /skill-name commands. -""" +"""Shared slash command helpers for skills (CLI and gateway both invoke /skill-name).""" import json import logging @@ -15,9 +11,8 @@ from typing import Any, Dict, Optional from hermes_constants import display_hermes_home from agent.prompt_cache_boundary import register_stable_prefix from agent.skill_preprocessing import ( - expand_inline_shell as _expand_inline_shell, load_skills_config as _load_skills_config, - substitute_template_vars as _substitute_template_vars, + preprocess_skill_content, ) logger = logging.getLogger(__name__) @@ -26,8 +21,7 @@ _skill_commands: Dict[str, Dict[str, Any]] = {} _skill_commands_platform: Optional[str] = None _skill_commands_home: Optional[str] = None # Guards the (map, platform-tag, home-tag) triple so publication and the -# freshness lookup always see a consistent snapshot. Scanning itself stays -# outside this lock. +# freshness lookup always see a consistent snapshot. Scanning stays outside. _publish_lock = threading.Lock() # Patterns for sanitizing skill names into clean hyphen-separated slugs. _SKILL_INVALID_CHARS = re.compile(r"[^a-z0-9-]") @@ -36,20 +30,14 @@ _SKILL_MULTI_HYPHEN = re.compile(r"-{2,}") # --------------------------------------------------------------------------- # Skill-scaffolding markers and the canonical extractor. # -# When a user invokes a /skill (or /bundle), Hermes expands the turn into a -# model-facing message that embeds the full skill body plus scaffolding. That -# expanded text is what flows into the agent loop — and into memory providers -# via MemoryManager. Providers that store or embed the raw user turn (mem0, -# openviking, hindsight, retaindb, byterover, honcho, supermemory) would -# otherwise capture the entire skill body instead of what the user actually -# asked. ``extract_user_instruction_from_skill_message`` recovers just the -# user's instruction so memory stays clean. +# A /skill (or /bundle) turn is expanded into a model-facing message embedding +# the full skill body. Memory providers that store the raw user turn would +# capture the body instead of what the user asked, so +# ``extract_user_instruction_from_skill_message`` recovers just the instruction. # -# These markers MUST stay byte-identical to the builders below -# (``_build_skill_message`` here, ``build_bundle_invocation_message`` in -# agent/skill_bundles.py). They are co-located with the single-skill builder -# on purpose, and the bundle markers are asserted against the bundle builder in -# tests/openviking_plugin/test_openviking.py::test_skill_markers_match_hermes_scaffolding. +# These markers MUST stay byte-identical to the builders (``_build_skill_message`` +# here, ``build_bundle_invocation_message`` in agent/skill_bundles.py); the +# bundle markers are asserted in tests/openviking_plugin/test_openviking.py. # --------------------------------------------------------------------------- _SKILL_INVOCATION_PREFIX = "[IMPORTANT: The user has invoked the " _SINGLE_SKILL_MARKER = "The full skill content is loaded below.]" @@ -66,28 +54,34 @@ _BUNDLE_FIRST_SKILL_BLOCK = "\n\n[Loaded as part of the " _SKILL_NAME_RE = re.compile(re.escape(_SKILL_INVOCATION_PREFIX) + r'"([^"]*)"') # SQL LIKE pattern matching a skill-expanded turn, for listing queries that -# have to recognize scaffolding before the row reaches Python. The prefix -# contains no LIKE wildcards (`%`, `_`), so it needs no ESCAPE clause. +# recognize scaffolding before the row reaches Python. The prefix contains no +# LIKE wildcards, so it needs no ESCAPE clause. SKILL_SCAFFOLD_SQL_LIKE = _SKILL_INVOCATION_PREFIX + "%" # Marks where a preview query joined the head and tail of a long scaffolded -# message. ``describe_skill_invocation`` may hand back a span that runs across -# the joint (a bundle instruction cut off by the head window); callers cut the -# description there rather than show the skill body on the far side. +# message; ``describe_skill_invocation`` cuts a description there rather than +# show the skill body on the far side. SKILL_EXCERPT_JOINT = "\x1e" +def slugify_skill_name(name: str) -> str: + """Normalize a skill/bundle name to a ``/command`` slug (``Foo Bar`` -> ``foo-bar``). + + Strips non-alnum chars (``+``, ``/``) that would make invalid Telegram + command names downstream. + """ + cmd = name.lower().replace(" ", "-").replace("_", "-") + cmd = _SKILL_INVALID_CHARS.sub("", cmd) + return _SKILL_MULTI_HYPHEN.sub("-", cmd).strip("-") + + def append_user_instruction(parts: list, instruction: str) -> str: """Append the instruction line to ``parts``; return the stable prefix. - Shared by every builder that ends a static skill scaffold with the - caller-supplied volatile instruction (single-skill invocations, cron job - prompts). The returned prefix ends exactly at the instruction marker, so - registering it with ``agent.prompt_cache_boundary`` lets the Anthropic - cache planner put a breakpoint on the scaffold instead of caching the - whole message as one atomic block (#81867). Keeping construction in one - place guarantees the registered prefix stays a byte-prefix of the built - message — the invariant the request-time split depends on. + The prefix ends exactly at the instruction marker so, registered with + ``agent.prompt_cache_boundary``, the Anthropic cache planner can break on + the scaffold instead of caching the whole message as one atomic block. + Single construction site guarantees the prefix is a byte-prefix of the message. """ stable_prefix = "\n".join(parts) + "\n" + _SINGLE_SKILL_INSTRUCTION parts.append(f"{_SINGLE_SKILL_INSTRUCTION}{instruction}") @@ -97,14 +91,9 @@ def append_user_instruction(parts: list, instruction: str) -> str: def extract_user_instruction_from_skill_message(content: Any) -> Optional[str]: """Recover the user's instruction from a slash-skill-expanded turn. - Returns: - - The original string unchanged when it is NOT skill scaffolding - (a normal user message passes straight through). - - The extracted user instruction when the scaffolding carried one. - - ``None`` when the content is skill scaffolding with no user - instruction (i.e. a bare ``/skill`` invocation). Callers that feed - memory providers should skip the turn in that case — there is no - user content worth storing. + Returns the string unchanged when it is NOT scaffolding, the extracted + instruction when the scaffolding carried one, or ``None`` for a bare + ``/skill`` invocation (nothing worth storing in memory). """ if not isinstance(content, str): return None @@ -124,36 +113,25 @@ def extract_user_instruction_from_skill_message(content: Any) -> Optional[str]: def describe_skill_invocation(content: Any, separator: str = " — ") -> Optional[str]: """Render a slash-skill-expanded turn the way the user typed it. - The expanded message embeds the whole skill body, so any surface that - summarizes a user turn from its raw content — session titles, sidebar - previews, the ``/rewind`` picker — otherwise shows the skill's own prose - as if the user had written it. That is how a skill's opening line ends up - as a session title. - - Returns ``"/work — fix the title leak"``, or ``"/work"`` for a bare - invocation, or ``None`` when *content* is not skill scaffolding (the - caller should then summarize it as an ordinary message). - - *separator* joins the command and the instruction. Previews use the - default em dash; pass ``" "`` for the literal invocation the user typed, - which is what chat transcripts render. + Returns ``"/work — fix the title leak"``, ``"/work"`` for a bare invocation, + or ``None`` when *content* is not skill scaffolding. Surfaces that summarize + a user turn (session titles, previews, ``/rewind``) use this so the skill's + own prose never masquerades as the user's. Pass ``separator=" "`` for the + literal invocation as typed (chat transcripts). """ if not isinstance(content, str) or not content.startswith(_SKILL_INVOCATION_PREFIX): return None match = _SKILL_NAME_RE.match(content) name = (match.group(1) if match else "").strip() - # Bundle headers already carry their typed "/a /b" keys; a single skill is - # a bare name. + # Bundle headers already carry their typed "/a /b" keys; a single skill is a bare name. label = name if name.startswith("/") else f"/{name}" instruction = extract_user_instruction_from_skill_message(content) if instruction and instruction is not content: - # An excerpted message (head + tail, joined by SKILL_EXCERPT_JOINT) can - # put the joint inside the matched span — keep only the side the - # instruction marker was found on. - instruction = instruction.split(SKILL_EXCERPT_JOINT)[0] - instruction = " ".join(instruction.split()) + # An excerpt (head + tail joined by SKILL_EXCERPT_JOINT) can put the + # joint inside the span — keep only the side the marker was found on. + instruction = " ".join(instruction.split(SKILL_EXCERPT_JOINT)[0].split()) if instruction: return f"{label}{separator}{instruction}" if name else instruction @@ -161,46 +139,36 @@ def describe_skill_invocation(content: Any, separator: str = " — ") -> Optiona def _extract_single_skill_user_instruction(message: str) -> Optional[str]: - # Single-skill format appends the user instruction after the skill body, so - # the last occurrence is the user-provided one; the body may quote this text. + # The instruction follows the skill body, so the LAST marker is the user's + # (the body may quote the marker text). marker_idx = message.rfind(_SINGLE_SKILL_INSTRUCTION) if marker_idx < 0: return None - instruction = message[marker_idx + len(_SINGLE_SKILL_INSTRUCTION):] - runtime_idx = instruction.find(_RUNTIME_NOTE) - if runtime_idx >= 0: - instruction = instruction[:runtime_idx] - instruction = instruction.strip() - return instruction or None + return _cut_at(instruction, _RUNTIME_NOTE) def _extract_bundle_user_instruction(message: str) -> Optional[str]: - # Bundle format puts the user instruction before the loaded skills, so the - # first occurrence is the user-provided one. + # Bundles put the instruction before the loaded skills, so the FIRST marker is the user's. marker_idx = message.find(_BUNDLE_USER_INSTRUCTION) if marker_idx < 0: return None - instruction = message[marker_idx + len(_BUNDLE_USER_INSTRUCTION):] - first_skill_idx = instruction.find(_BUNDLE_FIRST_SKILL_BLOCK) - if first_skill_idx >= 0: - instruction = instruction[:first_skill_idx] - instruction = instruction.strip() - return instruction or None + return _cut_at(instruction, _BUNDLE_FIRST_SKILL_BLOCK) + + +def _cut_at(text: str, stop_marker: str) -> Optional[str]: + idx = text.find(stop_marker) + if idx >= 0: + text = text[:idx] + return text.strip() or None def _resolve_skill_commands_platform() -> Optional[str]: - """Return the current platform scope used for disabled-skill filtering. + """Current platform scope for disabled-skill filtering, or None (CLI, RL, scripts). - Used to detect when the active platform has shifted so - :func:`get_skill_commands` can drop a stale cache that was populated - for a different platform's ``skills.platform_disabled`` view (#14536). - - Resolves from (in order) ``HERMES_PLATFORM`` env var and - ``HERMES_SESSION_PLATFORM`` from the gateway session context. Returns - ``None`` when no platform scope is active (e.g. classic CLI, RL - rollouts, standalone scripts). + A change here invalidates the scan cache so each platform sees its own + ``skills.platform_disabled`` view. """ try: from gateway.session_context import get_session_env @@ -215,14 +183,10 @@ def _resolve_skill_commands_platform() -> Optional[str]: def _resolve_skill_commands_home() -> str: - """Return the effective Hermes home the skill scan should be scoped to. + """Effective Hermes home the scan is scoped to. - A gateway session can switch between profiles that each carry their own - ``skills.external_dirs`` (via ``set_hermes_home_override``), but the - module-level scan only tracked ``_resolve_skill_commands_platform()``. - Switching profiles without a platform change left the previous profile's - skill list cached, so ``get_skill_commands()`` reported a cache miss for - skills that only exist under the new profile (#88023). + Profiles carry their own ``skills.external_dirs``; a profile switch without + a platform change must still invalidate the scan cache. """ from hermes_constants import get_hermes_home @@ -253,10 +217,8 @@ def _load_skill_payload(skill_identifier: str, task_id: str | None = None) -> tu skill_name = str(loaded_skill.get("name") or normalized) skill_path = str(loaded_skill.get("path") or "") skill_dir = None - # Prefer the absolute skill_dir returned by skill_view() — this is - # correct for both local and external skills. Fall back to the old - # SKILLS_DIR-relative reconstruction only when skill_dir is absent - # (e.g. legacy skill_view responses). + # Prefer the absolute skill_dir from skill_view() (correct for external + # skills too); fall back to SKILLS_DIR-relative reconstruction for legacy responses. abs_skill_dir = loaded_skill.get("skill_dir") if abs_skill_dir: skill_dir = Path(abs_skill_dir) @@ -269,13 +231,20 @@ def _load_skill_payload(skill_identifier: str, task_id: str | None = None) -> tu return loaded_skill, skill_dir, skill_name -def _inject_skill_config(loaded_skill: dict[str, Any], parts: list[str]) -> None: - """Resolve and inject skill-declared config values into the message parts. +def _bump_use(skill_name: str, task_id: str | None) -> None: + """Track active usage for Curator lifecycle management; never fatal.""" + try: + from tools.skill_usage import bump_use + bump_use(skill_name, task_id=task_id) + except Exception: + pass - If the loaded skill's frontmatter declares ``metadata.hermes.config`` - entries, their current values (from config.yaml or defaults) are appended - as a ``[Skill config: ...]`` block so the agent knows the configured values - without needing to read config.yaml itself. + +def _inject_skill_config(loaded_skill: dict[str, Any], parts: list[str]) -> None: + """Append a ``[Skill config: ...]`` block with resolved ``metadata.hermes.config`` values. + + Lets the agent see configured values without reading config.yaml itself. + Non-critical: any failure leaves the message without the block. """ try: from agent.skill_utils import ( @@ -284,7 +253,6 @@ def _inject_skill_config(loaded_skill: dict[str, Any], parts: list[str]) -> None resolve_skill_config_values, ) - # The loaded_skill dict contains the raw content which includes frontmatter raw_content = str(loaded_skill.get("raw_content") or loaded_skill.get("content") or "") if not raw_content: return @@ -305,7 +273,37 @@ def _inject_skill_config(loaded_skill: dict[str, Any], parts: list[str]) -> None lines.append("]") parts.extend(lines) except Exception: - pass # Non-critical — skill still loads without config injection + pass + + +def _setup_note(loaded_skill: dict[str, Any]) -> Optional[str]: + if loaded_skill.get("setup_skipped"): + return ( + "Required environment setup was skipped. Continue loading the skill " + "and explain any reduced functionality if it matters." + ) + if loaded_skill.get("gateway_setup_hint"): + return loaded_skill["gateway_setup_hint"] + if loaded_skill.get("setup_needed") and loaded_skill.get("setup_note"): + return loaded_skill["setup_note"] + return None + + +def _supporting_files(loaded_skill: dict[str, Any], skill_dir: Path | None) -> list[str]: + """Skill-relative support file paths: from ``linked_files`` or a disk walk.""" + supporting = [] + for entries in (loaded_skill.get("linked_files") or {}).values(): + if isinstance(entries, list): + supporting.extend(entries) + + if not supporting and skill_dir: + for subdir in ("references", "templates", "scripts", "assets"): + subdir_path = skill_dir / subdir + if subdir_path.exists(): + for f in sorted(subdir_path.rglob("*")): + if f.is_file() and not f.is_symlink(): + supporting.append(str(f.relative_to(skill_dir))) + return supporting def _build_skill_message( @@ -319,22 +317,17 @@ def _build_skill_message( """Format a loaded skill into a user/system message payload.""" from tools.skills_tool import _skills_dir - content = str(loaded_skill.get("content") or "") - - # ── Template substitution and inline-shell expansion ── - # Done before anything else so downstream blocks (setup notes, - # supporting-file hints) see the expanded content. - skills_cfg = _load_skills_config() - if skills_cfg.get("template_vars", True): - content = _substitute_template_vars(content, skill_dir, session_id) - if skills_cfg.get("inline_shell", False): - timeout = int(skills_cfg.get("inline_shell_timeout", 10) or 10) - content = _expand_inline_shell(content, skill_dir, timeout) + # Preprocess first so downstream blocks see the expanded content. + content = preprocess_skill_content( + str(loaded_skill.get("content") or ""), + skill_dir, + session_id, + skills_cfg=_load_skills_config(), + ) parts = [activation_note, "", content.strip()] - # ── Inject the absolute skill directory so the agent can reference - # bundled scripts without an extra skill_view() round-trip. ── + # Absolute skill dir lets the agent run bundled scripts without a skill_view() round-trip. if skill_dir: parts.append("") parts.append(f"[Skill directory: {skill_dir}]") @@ -344,52 +337,18 @@ def _build_skill_message( "with the terminal tool using the absolute path." ) - # ── Inject resolved skill config values ── _inject_skill_config(loaded_skill, parts) - if loaded_skill.get("setup_skipped"): - parts.extend( - [ - "", - "[Skill setup note: Required environment setup was skipped. Continue loading the skill and explain any reduced functionality if it matters.]", - ] - ) - elif loaded_skill.get("gateway_setup_hint"): - parts.extend( - [ - "", - f"[Skill setup note: {loaded_skill['gateway_setup_hint']}]", - ] - ) - elif loaded_skill.get("setup_needed") and loaded_skill.get("setup_note"): - parts.extend( - [ - "", - f"[Skill setup note: {loaded_skill['setup_note']}]", - ] - ) - - supporting = [] - linked_files = loaded_skill.get("linked_files") or {} - for entries in linked_files.values(): - if isinstance(entries, list): - supporting.extend(entries) - - if not supporting and skill_dir: - for subdir in ("references", "templates", "scripts", "assets"): - subdir_path = skill_dir / subdir - if subdir_path.exists(): - for f in sorted(subdir_path.rglob("*")): - if f.is_file() and not f.is_symlink(): - rel = str(f.relative_to(skill_dir)) - supporting.append(rel) + setup_note = _setup_note(loaded_skill) + if setup_note: + parts.extend(["", f"[Skill setup note: {setup_note}]"]) + supporting = _supporting_files(loaded_skill, skill_dir) if supporting and skill_dir: try: skill_view_target = str(skill_dir.relative_to(_skills_dir())) except ValueError: - # Skill is from an external dir — use the skill name instead - skill_view_target = skill_dir.name + skill_view_target = skill_dir.name # external dir — use the skill name parts.append("") parts.append( "[This skill has supporting files (paths relative to the skill " @@ -406,12 +365,8 @@ def _build_skill_message( stable_prefix = None if user_instruction: parts.append("") - # Everything before the caller-supplied instruction is a stable - # scaffold; declare the exact boundary so the Anthropic cache planner - # can put a breakpoint on it instead of caching the whole message as - # one atomic block (#81867). The static instruction prose stays on - # the stable side; the volatile instruction (webhook payload, ticket - # IDs, timestamps) and any runtime note ride in the tail. + # Everything before the volatile instruction is a stable scaffold; the + # registered boundary lets the cache planner break there (see append_user_instruction). stable_prefix = append_user_instruction(parts, user_instruction) if runtime_note: @@ -424,24 +379,122 @@ def _build_skill_message( return message -def scan_skill_commands() -> Dict[str, Dict[str, Any]]: - """Scan ~/.hermes/skills/ and return a mapping of /command -> skill info. +def _render_skill_block( + loaded: tuple[dict[str, Any], Path | None, str], + activation_note: str, + task_id: str | None, +) -> str: + """Bump usage and build the message block for one loaded skill.""" + loaded_skill, skill_dir, skill_name = loaded + _bump_use(skill_name, task_id) + return _build_skill_message(loaded_skill, skill_dir, activation_note, session_id=task_id) - Returns: - Dict mapping "/skill-name" to {name, description, skill_md_path, skill_dir}. + +def _scaffold_header( + subject: str, + loaded_names: list[str], + *, + lead_lines: list[str] | None = None, + missing: list[str] | None = None, + disabled: list[str] | None = None, + extra_instruction: str = "", + user_instruction: str = "", +) -> str: + """Header for multi-skill messages (bundles and stacked invocations). + + ``subject`` (e.g. ``'"name" skill bundle'``) must end in " skill bundle" so + the bundle-format extractor in extract_user_instruction_from_skill_message() + applies unchanged. + """ + lines = [ + f"[IMPORTANT: The user has invoked the {subject}, " + f"loading {len(loaded_names)} skills together. Treat every skill below " + "as active guidance for this turn.]", + "", + *(lead_lines or []), + f"Skills loaded: {', '.join(loaded_names)}", + ] + if missing: + lines.append(f"Skills missing (skipped): {', '.join(missing)}") + if disabled: + lines.append( + f"Skills disabled for this platform (skipped): {', '.join(disabled)}" + ) + if extra_instruction: + lines.extend(["", f"Bundle instruction: {extra_instruction}"]) + if user_instruction: + lines.extend(["", f"User instruction: {user_instruction}"]) + return "\n".join(lines) + + +_SCAN_SKIP_PARTS = {'.git', '.github', '.hub', '.archive'} + + +def _scan_skill_md(skill_md: Path, disabled: set, seen_names: set, commands: Dict[str, Dict[str, Any]], resolve_command) -> None: + """Register one SKILL.md in *commands* (no-op when filtered or colliding).""" + from tools.skills_tool import _parse_frontmatter, skill_matches_platform, skill_matches_environment + + if any(part in _SCAN_SKIP_PARTS for part in skill_md.parts): + return + frontmatter, body = _parse_frontmatter(skill_md.read_text(encoding='utf-8')) + # OS gate is hard; environment gate (kanban/docker/s6) is offer-time only. + if not skill_matches_platform(frontmatter) or not skill_matches_environment(frontmatter): + return + name = frontmatter.get('name', skill_md.parent.name) + if name in seen_names or name in disabled: + return + description = frontmatter.get('description', '') + if not description: + for line in body.strip().split('\n'): + line = line.strip() + if line and not line.startswith('#'): + description = line[:80] + break + seen_names.add(name) + cmd_name = slugify_skill_name(name) + if not cmd_name: + return + # A collision with a core command (name or alias, via resolve_command) skips + # auto-registration; the skill stays loadable via /skill . + if resolve_command(cmd_name) is not None: + logger.warning( + "Skill %r generates slash command '/%s' which " + "collides with a core Hermes command; skipping " + "auto-registration. Use '/skill %s' instead.", + name, cmd_name, name, + ) + return + # Dedup on the slug too: "git_helper" and "git-helper" normalize the same. + # First-wins preserves project > local > external precedence. + cmd_key = f"/{cmd_name}" + if cmd_key in commands: + logger.warning( + "Skill %r maps to slash command %s already claimed " + "by %r; keeping the first and skipping this one.", + name, cmd_key, commands[cmd_key]["name"], + ) + return + commands[cmd_key] = { + "name": name, + "description": description or f"Invoke the {name} skill", + "skill_md_path": str(skill_md), + "skill_dir": str(skill_md.parent), + } + + +def scan_skill_commands() -> Dict[str, Dict[str, Any]]: + """Scan skill dirs and return {"/skill-name": {name, description, skill_md_path, skill_dir}}. + + Builds into a local map and publishes once at the end: writing straight into + the global exposed partial results to overlapping scans, which then logged + bogus "already claimed" collisions against their own incumbents. """ global _skill_commands, _skill_commands_platform, _skill_commands_home platform = _resolve_skill_commands_platform() home = _resolve_skill_commands_home() - # Build into a local map and publish once, at the end. Writing straight - # into the global made a scan's partial results visible to everything - # else in the process: a second, overlapping scan deduped against its own - # (empty) ``seen_names`` but collided against the first scan's already- - # published slugs, logging one bogus "already claimed" warning per skill — - # each naming the same skill as its own incumbent (#74574). commands: Dict[str, Dict[str, Any]] = {} try: - from tools.skills_tool import _skills_dir, _parse_frontmatter, skill_matches_platform, skill_matches_environment, _get_disabled_skill_names + from tools.skills_tool import _skills_dir, _get_disabled_skill_names from agent.skill_utils import ( get_external_skills_dirs, get_project_skills_dirs, @@ -452,13 +505,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: disabled = _get_disabled_skill_names() seen_names: set = set() - # Scan project dirs first (highest precedence), then local, then external. - # Project dirs iterate through the quarantine chokepoint. + # Precedence: project (through the quarantine chokepoint) > local > external. + # Resolve the local dir at call time: import-time SKILLS_DIR is frozen to + # the launch home, but a multiplexed profile scope may have changed it. project_dirs = list(get_project_skills_dirs()) dirs_to_scan = list(project_dirs) - # Resolve at call time: the import-time SKILLS_DIR is frozen to the - # launch home, so a multiplexed profile scope (set_hermes_home_override) - # would still scan the default profile's skills (#67277). skills_dir = _skills_dir() if skills_dir.exists(): dirs_to_scan.append(skills_dir) @@ -471,82 +522,15 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: else iter_skill_index_files(scan_dir, "SKILL.md") ) for skill_md in _iter: - if any(part in {'.git', '.github', '.hub', '.archive'} for part in skill_md.parts): - continue try: - content = skill_md.read_text(encoding='utf-8') - frontmatter, body = _parse_frontmatter(content) - # Skip skills incompatible with the current OS platform - if not skill_matches_platform(frontmatter): - continue - # Skip skills not relevant to the current runtime env - # (kanban/docker/s6). Offer-time only; explicit load bypasses. - if not skill_matches_environment(frontmatter): - continue - name = frontmatter.get('name', skill_md.parent.name) - if name in seen_names: - continue - # Respect user's disabled skills config - if name in disabled: - continue - description = frontmatter.get('description', '') - if not description: - for line in body.strip().split('\n'): - line = line.strip() - if line and not line.startswith('#'): - description = line[:80] - break - seen_names.add(name) - # Normalize to hyphen-separated slug, stripping - # non-alnum chars (e.g. +, /) to avoid invalid - # Telegram command names downstream. - cmd_name = name.lower().replace(' ', '-').replace('_', '-') - cmd_name = _SKILL_INVALID_CHARS.sub('', cmd_name) - cmd_name = _SKILL_MULTI_HYPHEN.sub('-', cmd_name).strip('-') - if not cmd_name: - continue - # Skip if this skill's auto-generated /command collides - # with a core Hermes slash command (name or alias). The - # skill remains fully loadable via /skill . - # Uses resolve_command() so aliases and case variants are - # covered without maintaining a separate cache. - if resolve_command(cmd_name) is not None: - logger.warning( - "Skill %r generates slash command '/%s' which " - "collides with a core Hermes command; skipping " - "auto-registration. Use '/skill %s' instead.", - name, cmd_name, name, - ) - continue - # Dedup on the resolved slug, not just the raw name: two - # distinct frontmatter names can normalize to the same - # slug (e.g. "git_helper" vs "git-helper"). First-wins - # preserves local-before-external precedence. - cmd_key = f"/{cmd_name}" - if cmd_key in commands: - logger.warning( - "Skill %r maps to slash command %s already claimed " - "by %r; keeping the first and skipping this one.", - name, cmd_key, commands[cmd_key]["name"], - ) - continue - commands[cmd_key] = { - "name": name, - "description": description or f"Invoke the {name} skill", - "skill_md_path": str(skill_md), - "skill_dir": str(skill_md.parent), - } + _scan_skill_md(skill_md, disabled, seen_names, commands, resolve_command) except Exception: continue except Exception: pass - # Publish the finished map and the platform/home it was scanned for as - # ONE step. Bare assignments are not atomic together: a reader landing - # between them sees the NEW map still carrying the OLD platform tag, and - # if that stale tag happens to match its own platform it accepts the map - # without rescanning — serving another platform's disabled-skill view, - # exactly the leak #14536 closed. Only the publish/lookup pair is locked; - # the scan above (file I/O, deferred imports) stays outside it. + # Publish map + tags as ONE step: a reader landing between bare assignments + # could accept the new map under a stale platform tag and serve another + # platform's disabled-skill view. with _publish_lock: _skill_commands = commands _skill_commands_platform = platform @@ -557,16 +541,12 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: def get_skill_commands() -> Dict[str, Dict[str, Any]]: """Return the current skill commands mapping (scan first if empty). - Rescans when the active platform scope changes (e.g. a gateway - process serving Telegram and Discord concurrently) so each platform - sees its own ``skills.platform_disabled`` view (#14536), and when the - active profile's Hermes home changes (e.g. Desktop switching profiles - mid-session) so each profile sees its own ``skills.external_dirs`` (#88023). + Rescans when the platform scope changes (one gateway serving Telegram and + Discord) or the active profile's home changes (Desktop profile switch), so + each sees its own ``platform_disabled`` / ``external_dirs`` view. """ current_platform = _resolve_skill_commands_platform() current_home = _resolve_skill_commands_home() - # Read the map and its tags under the same lock that publishes them, so - # the freshness decision is made against a consistent snapshot. with _publish_lock: commands = _skill_commands is_fresh = ( @@ -576,88 +556,55 @@ def get_skill_commands() -> Dict[str, Dict[str, Any]]: ) if is_fresh: return commands - # Scan outside the lock — it does file I/O and deferred imports, and - # concurrent scans are already safe (each builds its own map). + # Scan outside the lock — file I/O and deferred imports; concurrent scans + # are safe since each builds its own map. return scan_skill_commands() -def reload_skills() -> Dict[str, Any]: - """Re-scan the skills directory and return a diff of what changed. +def diff_command_snapshots(before: Dict[str, str], after: Dict[str, str]) -> Dict[str, Any]: + """Diff two {name: description} snapshots into added/removed/unchanged/total. - Rescans ``~/.hermes/skills/`` and any ``skills.external_dirs`` so the - slash-command map (``agent.skill_commands._skill_commands``) reflects - skills added or removed on disk. - - This does NOT invalidate the skills system-prompt cache. Skills are - called by name via ``/skill-name``, ``skills_list``, or ``skill_view`` - — they don't need to be in the system prompt for the model to use them. - Keeping the prompt cache intact preserves prefix caching across the - reload, so a user invoking ``/reload-skills`` pays no cache-reset cost. - - Returns: - Dict with keys:: - - { - "added": [{"name": str, "description": str}, ...], - "removed": [{"name": str, "description": str}, ...], - "unchanged": [skill names present before and after], - "total": total skill count after rescan, - "commands": total /slash-skill count after rescan, - } - - ``description`` is the skill's full SKILL.md frontmatter - ``description:`` field. Note: the system prompt skill index - truncates this to the first 57 chars; see ``extract_skill_description``. + Removed entries carry the pre-rescan description (the file may be gone). """ - # Snapshot pre-reload state (name -> description) from the current - # slash-command cache. Using dicts lets the post-rescan diff carry - # descriptions for newly-visible or just-removed skills without a - # second disk walk. - def _snapshot(cmds: Dict[str, Dict[str, Any]]) -> Dict[str, str]: - out: Dict[str, str] = {} - for slash_key, info in cmds.items(): - bare = slash_key.lstrip("/") - out[bare] = (info or {}).get("description") or "" - return out - - before = _snapshot(_skill_commands) - - # Rescan the skills dir. ``scan_skill_commands`` resets - # ``_skill_commands = {}`` internally and repopulates it. - new_commands = scan_skill_commands() - - after = _snapshot(new_commands) - added_names = sorted(set(after) - set(before)) removed_names = sorted(set(before) - set(after)) - unchanged = sorted(set(after) & set(before)) - - added = [{"name": n, "description": after[n]} for n in added_names] - # For removed skills, use the description we had cached pre-rescan - # (the skill file is gone so we can't re-read it). - removed = [{"name": n, "description": before[n]} for n in removed_names] - return { - "added": added, - "removed": removed, - "unchanged": unchanged, + "added": [{"name": n, "description": after[n]} for n in added_names], + "removed": [{"name": n, "description": before[n]} for n in removed_names], + "unchanged": sorted(set(after) & set(before)), "total": len(after), - "commands": len(new_commands), } +def reload_skills() -> Dict[str, Any]: + """Re-scan skill dirs and return a diff of the slash-command map. + + Does NOT invalidate the skills system-prompt cache: skills are called by + name, so keeping the prompt cache intact means ``/reload-skills`` costs no + cache reset. + + Returns ``{"added": [{name, description}], "removed": [...], "unchanged": + [names], "total": int, "commands": int}``; ``description`` is the full + frontmatter field (the system prompt index truncates it). + """ + def _snapshot(cmds: Dict[str, Dict[str, Any]]) -> Dict[str, str]: + return { + slash_key.lstrip("/"): (info or {}).get("description") or "" + for slash_key, info in cmds.items() + } + + before = _snapshot(_skill_commands) + new_commands = scan_skill_commands() + result = diff_command_snapshots(before, _snapshot(new_commands)) + result["commands"] = len(new_commands) + return result + + def resolve_skill_command_key(command: str) -> Optional[str]: - """Resolve a user-typed /command to its canonical skill_cmds key. + """Resolve a user-typed /command to its canonical ``/slug`` key, or None. - Skills are always stored with hyphens — ``scan_skill_commands`` normalizes - spaces and underscores to hyphens when building the key. Hyphens and - underscores are treated interchangeably in user input: this matches - ``_check_unavailable_skill`` and accommodates Telegram bot-command names - (which disallow hyphens, so ``/claude-code`` is registered as - ``/claude_code`` and comes back in the underscored form). - - Returns the matching ``/slug`` key from ``get_skill_commands()`` or - ``None`` if no match. + Underscores map to hyphens: Telegram disallows hyphens in bot commands, so + ``/claude-code`` comes back as ``/claude_code``. """ if not command: return None @@ -671,17 +618,8 @@ def build_skill_invocation_message( task_id: str | None = None, runtime_note: str = "", ) -> Optional[str]: - """Build the user message content for a skill slash command invocation. - - Args: - cmd_key: The command key including leading slash (e.g., "/gif-search"). - user_instruction: Optional text the user typed after the command. - - Returns: - The formatted message string, or None if the skill wasn't found. - """ - commands = get_skill_commands() - skill_info = commands.get(cmd_key) + """Build the user message for a skill slash command, or None if not found.""" + skill_info = get_skill_commands().get(cmd_key) if not skill_info: return None @@ -690,13 +628,7 @@ def build_skill_invocation_message( return None loaded_skill, skill_dir, skill_name = loaded - - # Track active usage for Curator lifecycle management (#17782) - try: - from tools.skill_usage import bump_use - bump_use(skill_name, task_id=task_id) - except Exception: - pass # Non-critical — skill invocation proceeds regardless + _bump_use(skill_name, task_id) activation_note = ( f'[IMPORTANT: The user has invoked the "{skill_name}" skill, indicating they want ' @@ -714,35 +646,18 @@ def build_skill_invocation_message( # --------------------------------------------------------------------------- # Stacked slash-skill invocations — `/skill-a /skill-b do XYZ` loads every -# leading skill (up to _MAX_STACKED_SKILLS), not just the first. -# -# Inspired by Claude Code v2.1.199 (July 2, 2026): "Stacked slash-skill -# invocations like /skill-a /skill-b do XYZ now load all leading skills -# (up to 5), not just the first." -# -# The generated message deliberately reuses the BUNDLE scaffolding markers -# ("skill bundle," header + "[Loaded as part of the " block prefix) so -# extract_user_instruction_from_skill_message() recovers the user's -# instruction without any new marker plumbing — memory providers keep -# storing what the user actually asked, not N skill bodies. +# leading skill (up to _MAX_STACKED_SKILLS). The message reuses the BUNDLE +# scaffolding markers so the memory extractor needs no new plumbing. # --------------------------------------------------------------------------- _MAX_STACKED_SKILLS = 5 def split_stacked_skill_commands(rest: str) -> tuple[list[str], str]: - """Consume additional leading ``/skill`` tokens from *rest*. + """Consume further leading ``/skill`` tokens from *rest* (text after the first matched command). - *rest* is the text that follows the FIRST matched skill command (the - caller has already resolved that one). Leading whitespace-delimited - tokens that start with ``/`` and resolve to installed skill commands are - consumed, up to ``_MAX_STACKED_SKILLS`` total leading skills (i.e. at - most ``_MAX_STACKED_SKILLS - 1`` extra keys here). Parsing stops at the - first token that is not a resolvable skill command — that token and - everything after it become the user instruction. - - Returns: - ``(extra_cmd_keys, remaining_instruction)`` where ``extra_cmd_keys`` - are canonical ``/slug`` keys from :func:`get_skill_commands`. + Stops at the first token that is not a resolvable skill command (or a + repeat); that token onward is the user instruction. Returns + ``(extra_cmd_keys, remaining_instruction)``. """ keys: list[str] = [] remaining = rest or "" @@ -768,13 +683,8 @@ def build_stacked_skill_invocation_message( ) -> Optional[tuple[str, list[str], list[str]]]: """Build the user message for a stacked multi-skill slash invocation. - Args: - cmd_keys: Canonical ``/slug`` keys, in the order the user typed them. - user_instruction: Text remaining after the leading skill commands. - - Returns: - ``(message, loaded_skill_names, missing_skill_names)`` or ``None`` - when no skill could be loaded at all. + Returns ``(message, loaded_skill_names, missing_skill_names)`` or ``None`` + when no skill could be loaded at all. """ commands = get_skill_commands() @@ -789,57 +699,29 @@ def build_stacked_skill_invocation_message( seen.add(cmd_key) skill_info = commands.get(cmd_key) - if not skill_info: - missing.append(cmd_key.lstrip("/")) - continue - - loaded = _load_skill_payload(skill_info["skill_dir"], task_id=task_id) + loaded = _load_skill_payload(skill_info["skill_dir"], task_id=task_id) if skill_info else None if not loaded: missing.append(cmd_key.lstrip("/")) continue - loaded_skill, skill_dir, skill_name = loaded - - # Track active usage for Curator lifecycle management (#17782) - try: - from tools.skill_usage import bump_use - bump_use(skill_name, task_id=task_id) - except Exception: - pass # Non-critical - - # NOTE: must start with "[Loaded as part of the " — that prefix is - # the bundle block marker the memory-scaffolding extractor cuts on. - activation_note = ( - f'[Loaded as part of the stacked skill invocation "{skill_name}".]' - ) - skill_blocks.append( - _build_skill_message( - loaded_skill, - skill_dir, - activation_note, - session_id=task_id, - ) - ) + skill_name = loaded[2] + # Must start with "[Loaded as part of the " — the bundle block marker. + skill_blocks.append(_render_skill_block( + loaded, + f'[Loaded as part of the stacked skill invocation "{skill_name}".]', + task_id, + )) loaded_names.append(skill_name) if not skill_blocks: return None - # Header — must contain " skill bundle," so the bundle-format extractor - # in extract_user_instruction_from_skill_message() applies unchanged. typed = " ".join(k for k in cmd_keys if k) - header_lines = [ - f'[IMPORTANT: The user has invoked the "{typed}" stacked skill bundle, ' - f"loading {len(loaded_names)} skills together. Treat every skill below " - "as active guidance for this turn.]", - "", - f"Skills loaded: {', '.join(loaded_names)}", - ] - if missing: - header_lines.append(f"Skills missing (skipped): {', '.join(missing)}") - if user_instruction: - header_lines.extend(["", f"User instruction: {user_instruction}"]) - - header = "\n".join(header_lines) + header = _scaffold_header( + f'"{typed}" stacked skill bundle', + loaded_names, + missing=missing, + user_instruction=user_instruction, + ) return ("\n\n".join([header, *skill_blocks]), loaded_names, missing) @@ -847,16 +729,11 @@ def build_preloaded_skills_prompt( skill_identifiers: list[str], task_id: str | None = None, ) -> tuple[str, list[str], list[str]]: - """Load one or more skills for session-wide CLI/TUI preloading. + """Load skills for session-wide CLI/TUI preloading. - Returns (prompt_text, loaded_skill_names, missing_identifiers). - - Disabled skills are treated the same as missing ones: this loads via a - raw identifier straight into ``_load_skill_payload``, bypassing - ``get_skill_commands()``'s scan-time disabled filter — mirrors the - bundle-invocation gate (#59156). Without this, ``hermes -s `` or - a deployment's ``HERMES_TUI_SKILLS`` env var could force-load a skill an - operator disabled via ``skills.disabled``/``skills.platform_disabled``. + Returns (prompt_text, loaded_skill_names, missing_identifiers). Disabled + skills count as missing: this path bypasses the scan-time disabled filter, + so ``hermes -s `` must not force-load an operator-disabled skill. """ prompt_parts: list[str] = [] loaded_names: list[str] = [] @@ -876,36 +753,17 @@ def build_preloaded_skills_prompt( seen.add(identifier) loaded = _load_skill_payload(identifier, task_id=task_id) - if not loaded: + if not loaded or loaded[2] in disabled_names or identifier in disabled_names: missing.append(identifier) continue - - loaded_skill, skill_dir, skill_name = loaded - - if skill_name in disabled_names or identifier in disabled_names: - missing.append(identifier) - continue - - # Track active usage for Curator lifecycle management (#17782) - try: - from tools.skill_usage import bump_use - bump_use(skill_name, task_id=task_id) - except Exception: - pass # Non-critical - - activation_note = ( + skill_name = loaded[2] + prompt_parts.append(_render_skill_block( + loaded, f'[IMPORTANT: The user launched this CLI session with the "{skill_name}" skill ' "preloaded. Treat its instructions as active guidance for the duration of this " - "session unless the user overrides them.]" - ) - prompt_parts.append( - _build_skill_message( - loaded_skill, - skill_dir, - activation_note, - session_id=task_id, - ) - ) + "session unless the user overrides them.]", + task_id, + )) loaded_names.append(skill_name) return "\n\n".join(prompt_parts), loaded_names, missing diff --git a/agent/skill_utils.py b/agent/skill_utils.py index e23647cce8..3e6706e510 100644 --- a/agent/skill_utils.py +++ b/agent/skill_utils.py @@ -1,8 +1,6 @@ """Lightweight skill metadata utilities shared by prompt_builder and skills_tool. -This module intentionally avoids importing the tool registry, CLI config, or any -heavy dependency chain. It is safe to import at module level without triggering -tool registration or provider resolution. +Import-light by design: no tool registry, CLI config, or provider resolution. """ import ast @@ -10,15 +8,13 @@ import logging import os import re import sys -from pathlib import Path -from typing import Any, Dict, List, Optional, Set, Tuple +from pathlib import Path, PurePath +from typing import Any, Callable, Dict, List, Optional, Set, Tuple from hermes_constants import get_config_path, get_skills_dir, is_termux logger = logging.getLogger(__name__) -# ── Platform mapping ────────────────────────────────────────────────────── - PLATFORM_MAP = { "macos": "darwin", "linux": "linux", @@ -45,27 +41,19 @@ EXCLUDED_SKILL_DIRS = frozenset( ) ) -# Supporting files live inside a skill package and are loaded explicitly via -# skill_view(skill, file_path=...). They are not standalone skills and must not -# be scanned for active SKILL.md/DESCRIPTION.md entries, even if a Curator or -# archive workflow preserves a complete old skill package under references/. +# Progressive-disclosure support dirs inside a skill package: loaded explicitly +# via skill_view(skill, file_path=...), never scanned as standalone skills. SKILL_SUPPORT_DIRS = frozenset(("references", "templates", "assets", "scripts")) # ── Org-shared skills (sync contract) ─────────────────────────── # Org mirrors live under ~/.hermes/skills/_org//. Resolution is -# TOKEN-GATED via a marker file the sync client writes after verifying the -# token (skills_sync_client.pull_org_skills): only the marked org's mirror is -# scanned. No marker ⇒ no org skills load. The marker is plain data (org_id -# string) so this module stays import-light; the VERIFICATION lives in the -# sync client, which is the only writer. Offline grace: the marker persists, -# so already-pulled org skills keep working without connectivity; a VERIFIED -# org change (or personal-org token) rewrites/removes it. - +# TOKEN-GATED via a marker the sync client writes after verifying the token: +# only the marked org's mirror is scanned; no marker ⇒ no org skills load. +# The marker persists offline so already-pulled org skills keep working. ORG_MIRROR_DIR_NAME = "_org" ORG_ACTIVE_MARKER = ".active_org" ORG_PROVENANCE_FILE = ".org-provenance.json" -# Records the fingerprint of each skill exactly as upstream sent it, so a -# later local edit is detectable and an org pull can refuse to clobber it. +# Fingerprint of each skill as upstream sent it, so a local edit is detectable. ORG_BASELINE_FILE = ".org-baseline.json" @@ -75,70 +63,53 @@ def read_active_org_id(skills_dir: Path) -> Optional[str]: marker = skills_dir / ORG_MIRROR_DIR_NAME / ORG_ACTIVE_MARKER if not marker.exists(): return None - val = marker.read_text(encoding="utf-8").strip() - return val or None + return marker.read_text(encoding="utf-8").strip() or None except OSError: return None +def _org_rel_parts(path, skills_dir: Path) -> Optional[Tuple[str, ...]]: + try: + return Path(path).resolve().relative_to(Path(skills_dir).resolve()).parts + except (OSError, ValueError): + return None + + def is_org_mirror_path(path, skills_dir: Path) -> bool: """True when *path* is inside the org mirror (``_org/``).""" - try: - rel = Path(path).resolve().relative_to(Path(skills_dir).resolve()) - except (OSError, ValueError): - return False - return bool(rel.parts) and rel.parts[0] == ORG_MIRROR_DIR_NAME + parts = _org_rel_parts(path, skills_dir) + return bool(parts) and parts[0] == ORG_MIRROR_DIR_NAME def org_id_of_path(path, skills_dir: Path) -> Optional[str]: """The ```` segment for a path under ``_org//...``.""" - try: - rel = Path(path).resolve().relative_to(Path(skills_dir).resolve()) - except (OSError, ValueError): - return None - if len(rel.parts) >= 2 and rel.parts[0] == ORG_MIRROR_DIR_NAME: - return rel.parts[1] + parts = _org_rel_parts(path, skills_dir) + if parts and len(parts) >= 2 and parts[0] == ORG_MIRROR_DIR_NAME: + return parts[1] return None def is_excluded_skill_path(path, *, root: Optional[Path] = None) -> bool: - """True if *path* should be skipped by active skill scanners. + """True if *path* (Path or str) should be skipped by active skill scanners. - Use this on every ``SKILL.md`` path produced by direct ``rglob`` scans to - prune dependency, virtualenv, VCS, cache, and progressive-disclosure - support-package paths. Centralising the check here keeps every - skill-scanning site in sync with the shared exclusion set. - - Accepts a Path or string. + Apply to every SKILL.md from a direct ``rglob`` scan so all scanning sites + share one exclusion set (dependency/VCS/cache dirs + support packages). """ - try: - parts = path.parts # Path - except AttributeError: - from pathlib import PurePath - parts = PurePath(str(path)).parts + parts = PurePath(str(path)).parts return any(part in EXCLUDED_SKILL_DIRS for part in parts) or is_skill_support_path( path, root=root ) def is_skill_support_path(path, *, root: Optional[Path] = None) -> bool: - """True if *path* is under a support dir of an actual skill root. + """True if *path* is under a support dir sitting directly inside a skill root. - ``references/``, ``templates/``, ``assets/``, and ``scripts/`` are - progressive-disclosure support areas when they sit directly inside a skill - directory containing ``SKILL.md``. They are not active discovery roots for - standalone skills. A preserved package such as - ``some-skill/references/old-skill-package/SKILL.md`` is documentation data - unless the caller explicitly loads it via ``file_path``. - - Legitimate categories or skill names such as ``skills/scripts/foo`` remain - discoverable because their ``scripts`` component is not directly under a - directory that contains ``SKILL.md``. + ``skills/scripts/foo`` stays discoverable: its ``scripts`` component is not + directly under a directory containing ``SKILL.md``. """ path_obj = path if isinstance(path, Path) else Path(str(path)) parts = path_obj.parts - # Last component may be a file or candidate skill directory name. Only - # components before the leaf can be containing support directories. + # Only components before the leaf can be containing support directories. for idx, part in enumerate(parts[:-1]): if part not in SKILL_SUPPORT_DIRS or idx == 0: continue @@ -174,25 +145,14 @@ def yaml_load(content: str): def parse_frontmatter(content: str) -> Tuple[Dict[str, Any], str]: - """Parse YAML frontmatter from a markdown string. + """Parse YAML frontmatter from markdown; returns (frontmatter_dict, body). - Uses yaml with CSafeLoader for full YAML support (nested metadata, lists) - with a fallback to simple key:value splitting for robustness. - - A single leading UTF-8 BOM (U+FEFF) is stripped before parsing. Windows - GUI editors (Notepad, PowerShell ``>``) prepend one when saving a SKILL.md - as UTF-8, and ``read_text(encoding="utf-8")`` preserves it (only - ``utf-8-sig`` strips it). Left in place, the BOM defeats the ``---`` fence - check below and the whole frontmatter is silently discarded — name, - description, ``platforms`` gating, env-var setup, and conditional - activation all vanish. See CONTRIBUTING.md "File encoding". - - Returns: - (frontmatter_dict, remaining_body) + Falls back to simple key:value splitting for malformed YAML. A single + leading UTF-8 BOM is stripped first: Windows editors prepend one and it + would otherwise defeat the ``---`` fence check and silently drop the + frontmatter (see CONTRIBUTING.md "File encoding"). """ frontmatter: Dict[str, Any] = {} - - # Strip only a leading BOM; a BOM mid-content is data, not a marker. if content.startswith("\ufeff"): content = content[1:] body = content @@ -212,7 +172,6 @@ def parse_frontmatter(content: str) -> Tuple[Dict[str, Any], str]: if isinstance(parsed, dict): frontmatter = parsed except Exception: - # Fallback: simple key:value parsing for malformed YAML for line in yaml_content.strip().split("\n"): if ":" not in line: continue @@ -238,139 +197,92 @@ def skill_matches_platform_list(platforms: Any) -> bool: mapped = PLATFORM_MAP.get(normalized, normalized) if current.startswith(mapped): return True - # Termux runs a Linux userland on Android. Accept linux-tagged - # skills regardless of whether sys.platform is "linux" (pre-3.13 - # Termux) or "android" (Python 3.13+ Termux, and any other - # Android runtime). - if running_in_termux and mapped == "linux": - return True - # Explicit termux/android tags match a Termux session too. - if running_in_termux and mapped in ("termux", "android"): + # Termux is a Linux userland on Android: accept linux-tagged skills + # whether sys.platform is "linux" (pre-3.13) or "android" (3.13+), + # plus explicit termux/android tags. + if running_in_termux and mapped in ("linux", "termux", "android"): return True return False def skill_matches_platform(frontmatter: Dict[str, Any]) -> bool: - """Return True when the skill is compatible with the current OS. - - Skills declare platform requirements via a top-level ``platforms`` list - in their YAML frontmatter:: - - platforms: [macos] # macOS only - platforms: [macos, linux] # macOS and Linux - - If the field is absent or empty the skill is compatible with **all** - platforms (backward-compatible default). - - Termux note: on Termux/Android, ``sys.platform`` is ``"linux"`` on - older Pythons but became ``"android"`` on Python 3.13+. Termux is a - Linux userland riding on the Android kernel, so skills tagged - ``linux`` are treated as compatible in Termux regardless of which - ``sys.platform`` value Python reports. Individual Linux commands - inside a skill may still misbehave (no systemd, BusyBox utils, no - apt/dnf, etc.) but that is on the skill, not on platform gating. - """ + """True when the skill's ``platforms:`` list (absent = all) matches this OS.""" return skill_matches_platform_list(frontmatter.get("platforms")) # ── Environment matching ────────────────────────────────────────────────── +# An ``environments:`` tag is a *relevance* gate for offer surfaces (index, +# autocomplete, slash commands), not a compatibility gate: an explicit load +# (skill_view, --skills) always succeeds. Detection is cached per process. -# Recognized environment tags and how each is detected. An environment tag is -# a *relevance* gate, not a hard-compatibility gate (that is what ``platforms:`` -# is for). A skill tagged for an environment it isn't relevant to is hidden from -# the skills index / offer surfaces so it does not add noise for users who will -# never need it — but it can ALWAYS still be loaded explicitly (``skill_view``, -# ``--skills``), because an explicit request is explicit consent. -# -# Detection is cached for the process lifetime via ``_ENV_DETECT_CACHE``. _KNOWN_ENVIRONMENTS = frozenset({"kanban", "docker", "s6"}) _ENV_DETECT_CACHE: Dict[str, bool] = {} -def _detect_environment(env: str) -> bool: - """Return True when the named runtime environment is currently active. +def _detect_kanban() -> bool: + # Mirror the signals tools/kanban_tools.py gates on: a dispatcher-spawned + # worker (HERMES_KANBAN_TASK/BOARD in env — but only when this execution + # owns the task; a delegate_task child or in-process cron job sees the + # worker's vars without being that worker) or a profile opted into the + # kanban toolset. + if os.getenv("HERMES_KANBAN_TASK") or os.getenv("HERMES_KANBAN_BOARD"): + try: + from agent.delegation_context import is_dispatcher_owned_worker_context - Cached per process, EXCEPT ``kanban``: that verdict is context-dependent - (a delegate_task child or an in-process cron job sees the worker's - HERMES_KANBAN_* vars without owning them), so caching it process-wide would - freeze whichever context asked first and leak it to the others. + if is_dispatcher_owned_worker_context(): + return True + except Exception: + return True + try: + from tools.kanban_tools import _profile_has_kanban_toolset + + return bool(_profile_has_kanban_toolset()) + except Exception: + return False + + +def _detect_docker() -> bool: + try: + from hermes_constants import is_container + + return is_container() + except Exception: + return False + + +def _detect_s6() -> bool: + # The Hermes Docker image runs s6-overlay as PID 1; either marker means + # we're inside an s6-supervised container. + return os.path.isdir("/run/s6") or os.path.isdir("/package/admin/s6-overlay") + + +_ENV_DETECTORS: Dict[str, Callable[[], bool]] = { + "kanban": _detect_kanban, + "docker": _detect_docker, + "s6": _detect_s6, +} + + +def _detect_environment(env: str) -> bool: + """True when the named runtime environment is active. + + Cached per process EXCEPT ``kanban``: that verdict is context-dependent + (delegate children / in-process cron see the worker's vars), so a + process-wide cache would leak the first asker's answer to the others. """ if env != "kanban" and env in _ENV_DETECT_CACHE: return _ENV_DETECT_CACHE[env] - - result = True - if env == "kanban": - # Kanban is "active" either as a dispatcher-spawned worker (the - # dispatcher sets ``HERMES_KANBAN_TASK`` / ``HERMES_KANBAN_BOARD`` in the - # worker env) or as an orchestrator profile that has opted into the - # kanban toolset. Mirror the same signals the kanban tools themselves - # gate on (``tools/kanban_tools.py``) so the offer filter agrees with - # tool availability. - if os.getenv("HERMES_KANBAN_TASK") or os.getenv("HERMES_KANBAN_BOARD"): - # ...but only when this execution actually owns the dispatcher's - # task. A delegate_task child or a cron job fired in-process from a - # worker sees the worker's vars without being that worker. - try: - from agent.delegation_context import ( - is_dispatcher_owned_worker_context, - ) - - _owns_dispatcher_task = is_dispatcher_owned_worker_context() - except Exception: - _owns_dispatcher_task = True - else: - _owns_dispatcher_task = False - if _owns_dispatcher_task: - result = True - else: - try: - from tools.kanban_tools import _profile_has_kanban_toolset - - result = bool(_profile_has_kanban_toolset()) - except Exception: - result = False - elif env == "docker": - try: - from hermes_constants import is_container - - result = is_container() - except Exception: - result = False - elif env == "s6": - # The Hermes Docker image runs s6-overlay as PID 1 (/init). s6 plants - # its runtime scaffolding under /run/s6 and ships its admin tree under - # /package/admin/s6-overlay. Either marker means we're inside an - # s6-supervised container. - result = os.path.isdir("/run/s6") or os.path.isdir( - "/package/admin/s6-overlay" - ) - + detector = _ENV_DETECTORS.get(env) + result = detector() if detector else True _ENV_DETECT_CACHE[env] = result return result def skill_matches_environment(frontmatter: Dict[str, Any]) -> bool: - """Return True when the skill is relevant to the current runtime environment. + """True when ANY declared ``environments:`` tag is active (absent = all). - Skills may declare an ``environments`` list in their YAML frontmatter:: - - environments: [kanban] # only relevant when kanban is active - environments: [s6] # only relevant inside the s6 Docker image - environments: [docker] # only relevant inside any container - - If the field is absent or empty the skill is relevant in **all** - environments (backward-compatible default). - - This is an OFFER-time filter: it controls whether a skill shows up in the - skills index / autocomplete / slash-command list. It is intentionally NOT - enforced by ``skill_view`` or ``--skills`` preloading — an explicit load is - explicit consent, and load-bearing force-loads (e.g. a dispatcher pinning - a task to a specialist skill via ``--skills``) must always succeed - regardless of how the offer surfaces filter the skill. - - A skill matches when ANY of its declared environments is currently active - (OR semantics, mirroring ``platforms``). Unknown env tags fail open. + Offer-time filter only; unknown tags fail open. """ environments = frontmatter.get("environments") if not environments: @@ -381,10 +293,7 @@ def skill_matches_environment(frontmatter: Dict[str, Any]) -> bool: normalized = str(env).lower().strip() if not normalized: continue - if normalized not in _KNOWN_ENVIRONMENTS: - # Tag we don't understand — don't hide the skill over it. - return True - if _detect_environment(normalized): + if normalized not in _KNOWN_ENVIRONMENTS or _detect_environment(normalized): return True return False @@ -401,12 +310,7 @@ def _raw_config_cache_clear() -> None: def _load_raw_config() -> Dict[str, Any]: - """Read config.yaml with a shared mtime+size keyed cache. - - This module intentionally avoids importing ``hermes_cli.config`` on the - skill prompt/build path. A tiny local cache gives the same repeated-read - win without pulling the heavier CLI config stack into startup. - """ + """Read config.yaml with an mtime+size keyed cache (no hermes_cli.config import).""" config_path = get_config_path() if not config_path.exists(): return {} @@ -435,35 +339,30 @@ def _load_raw_config() -> Dict[str, Any]: return parsed -# Skills that must stay available regardless of configuration. The -# `hermes-agent` skill is the agent's own operating manual — it drives -# configuring, extending, and troubleshooting Hermes itself, and the system -# prompt unconditionally points at it. Disabling it leaves the agent unable -# to help with Hermes, so disable requests for these names are ignored -# everywhere the disabled list is consulted. +def _skills_cfg() -> Optional[Dict[str, Any]]: + """The ``skills:`` mapping from config.yaml, or None when absent/malformed.""" + parsed = _load_raw_config() + skills_cfg = parsed.get("skills") if parsed else None + return skills_cfg if isinstance(skills_cfg, dict) else None + + +def _expand_path(entry: str) -> Path: + """Expand ``~`` and ``${VAR}`` in a config path entry.""" + return Path(os.path.expanduser(os.path.expandvars(entry))) + + +# Always available regardless of config: `hermes-agent` is the agent's own +# operating manual and the system prompt points at it unconditionally. ESSENTIAL_SKILLS: frozenset = frozenset({"hermes-agent"}) def get_disabled_skill_names(platform: str | None = None) -> Set[str]: - """Read disabled skill names from config.yaml. + """Disabled skill names from config.yaml: global list ∪ platform list. - Args: - platform: Explicit platform name (e.g. ``"telegram"``). When - *None*, resolves from ``HERMES_PLATFORM`` or - ``HERMES_SESSION_PLATFORM`` env vars. Returns the global - disabled list, unioned with the platform-specific list when a - platform is resolved (a globally-disabled skill stays disabled - on every platform). - - Reads the config file directly (no CLI config imports) to stay - lightweight. + *platform* defaults to ``HERMES_PLATFORM`` / ``HERMES_SESSION_PLATFORM``. """ - parsed = _load_raw_config() - if not parsed: - return set() - - skills_cfg = parsed.get("skills") - if not isinstance(skills_cfg, dict): + skills_cfg = _skills_cfg() + if skills_cfg is None: return set() from gateway.session_context import get_session_env @@ -487,11 +386,9 @@ def get_disabled_skill_names(platform: str | None = None) -> Set[str]: def parse_config_string_list(value) -> List[str]: """Normalize a config value that may hold a JSON-array string into a list. - ``hermes config set`` and JSON-mode editor saves store lists as quoted - JSON strings (``'["a","b"]'`` or the Python-literal ``"['a']"``). Treating - such a string as a single name makes a curated disabled list silently - filter nothing (#86661); parsing it restores the intended list. A scalar - string still means one name (#13026). + ``hermes config set`` stores lists as quoted JSON/Python-literal strings; + treating one as a single name would silently filter nothing (#86661). A + scalar string still means one name (#13026). """ if value is None: return [] @@ -516,12 +413,9 @@ def _normalize_string_set(values) -> Set[str]: # ── External skills directories ────────────────────────────────────────── -# (config_path_str, mtime_ns) -> resolved external dirs list. Keyed by -# mtime_ns so a config.yaml edit mid-run is picked up automatically; -# otherwise every call would re-read + re-YAML-parse the 15KB config, -# which becomes the dominant cost of ``hermes`` startup when ~120 skills -# each trigger a category lookup during banner construction (10+ seconds -# of pure waste). +# (config_path_str, mtime_ns) -> resolved external dirs. Called once per skill +# during banner / tool-registry scans; re-parsing config each time dominated +# cold-start. _EXTERNAL_DIRS_CACHE: Dict[Tuple[str, int], List[Path]] = {} @@ -532,23 +426,15 @@ def _external_dirs_cache_clear() -> None: def get_external_skills_dirs() -> List[Path]: - """Read ``skills.external_dirs`` from config.yaml and return validated paths. + """Validated, deduplicated ``skills.external_dirs`` (existing dirs only). - Each entry is expanded (``~`` and ``${VAR}``) and resolved to an absolute - path. Only directories that actually exist are returned. Duplicates and - paths that resolve to the local ``~/.hermes/skills/`` are silently skipped. - - Cached in-process, keyed on ``config.yaml`` mtime — the function is - called once per skill during banner / tool-registry scans, and YAML - parsing a non-trivial config dominates ``hermes`` cold-start time - when the cache is absent. + Entries are expanded (``~``, ``${VAR}``); relative paths resolve against + HERMES_HOME; the local ``~/.hermes/skills/`` is skipped. """ config_path = get_config_path() if not config_path.exists(): return [] - # Cache key: (absolute path, mtime_ns). stat() is ~2us vs ~85ms for - # the full YAML parse, so the fast path is nearly free. try: stat = config_path.stat() cache_key: Tuple[str, int] = (str(config_path), stat.st_mtime_ns) @@ -558,15 +444,10 @@ def get_external_skills_dirs() -> List[Path]: if cache_key is not None: cached = _EXTERNAL_DIRS_CACHE.get(cache_key) if cached is not None: - # Return a copy so callers can't mutate the cached list. - return list(cached) + return list(cached) # copy so callers can't mutate the cache - parsed = _load_raw_config() - if not parsed: - return [] - - skills_cfg = parsed.get("skills") - if not isinstance(skills_cfg, dict): + skills_cfg = _skills_cfg() + if skills_cfg is None: return [] raw_dirs = skills_cfg.get("external_dirs") @@ -591,17 +472,9 @@ def get_external_skills_dirs() -> List[Path]: entry = str(entry).strip() if not entry: continue - # Expand ~ and environment variables - expanded = os.path.expanduser(os.path.expandvars(entry)) - p = Path(expanded) - # Resolve relative paths against HERMES_HOME, not cwd - if not p.is_absolute(): - p = (hermes_home / p).resolve() - else: - p = p.resolve() - if p == local_skills: - continue - if p in seen: + p = _expand_path(entry) + p = (hermes_home / p).resolve() if not p.is_absolute() else p.resolve() + if p == local_skills or p in seen: continue if p.is_dir(): seen.add(p) @@ -615,23 +488,13 @@ def get_external_skills_dirs() -> List[Path]: def get_skill_create_dir() -> Optional[Path]: - """Return the configured ``skills.create_dir``, or ``None`` when unset. + """Configured ``skills.create_dir`` (need not exist yet), or None when unset. - When set, agent-created skills (``skill_manage`` action=create) land in - this directory instead of the profile-local ``~/.hermes/skills/``, and - every user-facing instruction string that names the creation path renders - this directory instead of the default. - - The entry is expanded (``~`` and ``${VAR}``); relative paths resolve - against HERMES_HOME. A value that resolves to the local skills dir is - treated as unset (that is already the default behaviour). The directory - does NOT need to exist yet — skill creation mkdirs it on first write. + Relative paths resolve against HERMES_HOME; a value equal to the local + skills dir counts as unset. """ - parsed = _load_raw_config() - if not parsed: - return None - skills_cfg = parsed.get("skills") - if not isinstance(skills_cfg, dict): + skills_cfg = _skills_cfg() + if skills_cfg is None: return None raw = skills_cfg.get("create_dir") if not raw or not isinstance(raw, (str, os.PathLike)): @@ -642,8 +505,7 @@ def get_skill_create_dir() -> Optional[Path]: from hermes_constants import get_hermes_home - expanded = os.path.expanduser(os.path.expandvars(entry)) - p = Path(expanded) + p = _expand_path(entry) if not p.is_absolute(): p = get_hermes_home() / p try: @@ -659,12 +521,10 @@ def get_skill_create_dir() -> Optional[Path]: def display_skill_create_dir() -> str: - """User-facing display string for where new skills are created. + """User-facing path where new skills are created (``~/`` shorthand when possible). - Renders the configured ``skills.create_dir`` (with ``~/`` shorthand when - under the user's home) or the default ``/skills/`` path. Used by - instruction text (tool schema descriptions, prompts, docs strings) so a - configured creation dir changes every instruction that names the path. + Used by tool schema descriptions and prompts so a configured + ``skills.create_dir`` changes every instruction that names the path. """ from hermes_constants import display_hermes_home @@ -678,18 +538,10 @@ def display_skill_create_dir() -> str: def get_all_skills_dirs() -> List[Path]: - """Return all skill directories: local ``~/.hermes/skills/`` first, then external. + """Skill dirs: local ``~/.hermes/skills/`` first, then create_dir, then external. - The local dir is always first (and always included even if it doesn't exist - yet — callers handle that). When ``skills.create_dir`` is configured, it - follows immediately after the local dir (so agent-created skills are - discovered, trusted, and modifiable). External dirs follow in config order. - - NOTE: trusted project-local dirs (``./.hermes/skills`` at the git root) are - NOT part of this list — they have *higher* precedence than the local dir, - so callers that need them use :func:`get_project_skills_dirs` and scan - those roots first. See ``get_scan_ordered_skills_dirs`` for the full - precedence-ordered list. + Trusted project-local dirs are NOT included — they have *higher* precedence + than the local dir; see :func:`get_project_skills_dirs`. """ dirs = [get_skills_dir()] create_dir = get_skill_create_dir() @@ -702,34 +554,13 @@ def get_all_skills_dirs() -> List[Path]: # ── Project-local skills directories ────────────────────────────────────── -# -# Repo-local skills, mirroring what OpenCode (.opencode/skill/, .agents/skills/) -# and Codex (.codex/skills/, .agents/skills/) do: a project checkout can carry -# its own skills, active only for sessions started inside that project. -# -# Two candidate roots at the project root (found by walking up from cwd to the -# first directory containing ``.git``): -# /.hermes/skills/ — Hermes-native location -# /.agents/skills/ — cross-tool convention shared with other harnesses -# -# TRUST GATE: unlike AGENTS.md (plain instruction text), skills are load-on- -# demand procedure documents an agent will follow — auto-sourcing them from any -# cloned repo is a prompt-injection vector. Project skills therefore only load -# when the project root is listed in ``skills.trusted_project_dirs`` in -# config.yaml (Codex-style per-path trust). Untrusted dirs are still -# *discoverable* via get_untrusted_project_skills_root() so the CLI can print -# a one-line "run `hermes skills trust`" notice. -# -# PRECEDENCE: trusted project skills override same-named profile/bundled -# skills (index scans project dirs first; skill_view resolves cross-tier -# collisions in favor of the project tier). This matches both competitor -# harnesses and is the point of the feature: vendored repo skills win inside -# their repo. -# -# CACHE SAFETY: cwd is fixed for the life of a session, and the trust list is -# read from config at agent build time — the resolved dirs are stable for the -# conversation, so the skills index (and with it the system prompt) stays -# byte-stable. Same contract as AGENTS.md injection and project plugins. +# A checkout can carry skills at /.hermes/skills/ or /.agents/skills/ +# (root = nearest ancestor with .git). TRUST GATE: skills are procedure docs an +# agent will follow, so auto-sourcing them from any cloned repo is a prompt- +# injection vector — they load only when the root is in +# ``skills.trusted_project_dirs``. Trusted project skills override same-named +# profile/bundled skills. cwd and the trust list are fixed for a session, so +# the skills index (and system prompt) stays byte-stable. PROJECT_SKILLS_SUBDIRS = ( os.path.join(".hermes", "skills"), @@ -741,18 +572,11 @@ _PROJECT_ROOT_MAX_DEPTH = 64 def find_project_root(start: Optional[Path] = None) -> Optional[Path]: - """Locate the enclosing project root: nearest ancestor containing ``.git``. + """Nearest ancestor containing ``.git`` (dir or worktree file), or None. - Returns None when cwd is not inside a git checkout. ``.git`` may be a dir - (normal clone) or a file (worktree/submodule) — both count. - - When *start* is not given, the surface's working directory wins over the - process cwd: ``TERMINAL_CWD`` is the same per-surface workdir the terminal - tool and cron jobs use (a cron job sets it from its per-job ``workdir`` - without chdir'ing the scheduler process). This is what lets - non-interactive surfaces inherit a prior interactive trust decision by - project identity — and a surface with no workdir in a trusted repo simply - resolves no project and loads nothing (#48975). + Without *start*, the surface's ``TERMINAL_CWD`` wins over process cwd so + cron/API surfaces inherit an interactive trust decision by project identity; + a surface with no workdir resolves no project (#48975). """ try: if start is None: @@ -767,11 +591,9 @@ def find_project_root(start: Optional[Path] = None) -> Optional[Path]: for _ in range(_PROJECT_ROOT_MAX_DEPTH): try: if (cur / ".git").exists(): - # A git checkout AT the home dir (dotfiles-style) would make - # every session project-scoped; treat home itself as non-project. - if cur == home: - return None - return cur + # A dotfiles checkout AT home would make every session + # project-scoped; treat home itself as non-project. + return None if cur == home else cur except OSError: return None if cur.parent == cur: @@ -782,11 +604,8 @@ def find_project_root(start: Optional[Path] = None) -> Optional[Path]: def _project_trusted_dirs_from_config() -> Set[Path]: """Resolved set of trusted project roots from ``skills.trusted_project_dirs``.""" - parsed = _load_raw_config() - if not parsed: - return set() - skills_cfg = parsed.get("skills") - if not isinstance(skills_cfg, dict): + skills_cfg = _skills_cfg() + if skills_cfg is None: return set() raw = skills_cfg.get("trusted_project_dirs") if isinstance(raw, str): @@ -799,7 +618,7 @@ def _project_trusted_dirs_from_config() -> Set[Path]: if not entry: continue try: - result.add(Path(os.path.expanduser(os.path.expandvars(entry))).resolve()) + result.add(_expand_path(entry).resolve()) except OSError: continue return result @@ -816,9 +635,7 @@ def is_project_root_trusted(root: Path) -> bool: def _candidate_project_skills_dirs(root: Path) -> List[Path]: """Existing skill dirs under *root*, excluding the profile's own skills dir. - The exclusion matters when HERMES_HOME itself lives inside a git checkout: - ``/.hermes/skills`` would otherwise double as both the profile-local - and the project tier. + Matters when HERMES_HOME itself lives inside a git checkout. """ local_skills = get_skills_dir().resolve() dirs: List[Path] = [] @@ -832,35 +649,24 @@ def _candidate_project_skills_dirs(root: Path) -> List[Path]: return dirs -def get_project_skills_dirs() -> List[Path]: - """Trusted project-local skill dirs for the current cwd (may be empty). +def _project_discovery_disabled() -> bool: + skills_cfg = _skills_cfg() + return skills_cfg is not None and skills_cfg.get("project_discovery") is False - Empty when: not in a git checkout, no project skills dirs exist, project - discovery is disabled (``skills.project_discovery: false``), or the - project root is not trusted. - """ - parsed = _load_raw_config() - skills_cfg = parsed.get("skills") if isinstance(parsed, dict) else None - if isinstance(skills_cfg, dict) and skills_cfg.get("project_discovery") is False: + +def get_project_skills_dirs() -> List[Path]: + """Trusted project-local skill dirs for the current cwd (may be empty).""" + if _project_discovery_disabled(): return [] root = find_project_root() - if root is None: - return [] - if not is_project_root_trusted(root): + if root is None or not is_project_root_trusted(root): return [] return _candidate_project_skills_dirs(root) def get_untrusted_project_skills_root() -> Optional[Tuple[Path, int]]: - """When cwd's project has skills but is NOT trusted: (root, skill_count). - - Used by the CLI to print a one-line notice pointing at - ``hermes skills trust``. Returns None when there is nothing to notify - about (no project, no skills, already trusted, or discovery disabled). - """ - parsed = _load_raw_config() - skills_cfg = parsed.get("skills") if isinstance(parsed, dict) else None - if isinstance(skills_cfg, dict) and skills_cfg.get("project_discovery") is False: + """(root, skill_count) when cwd's project has skills but is NOT trusted, else None.""" + if _project_discovery_disabled(): return None root = find_project_root() if root is None or is_project_root_trusted(root): @@ -876,39 +682,17 @@ def get_untrusted_project_skills_root() -> Optional[Tuple[Path, int]]: return root, count -def get_scan_ordered_skills_dirs() -> List[Path]: - """All skill dirs in precedence order: project → local → external. - - First-wins name deduplication over this order gives project skills - priority over profile-local and external ones. - """ - dirs = list(get_project_skills_dirs()) - dirs.extend(get_all_skills_dirs()) - return dirs - - # ── Project skill quarantine (scan-time injection defense) ──────────────── -# -# Trust (`hermes skills trust`) is a REPO-level decision made once; the repo's -# skill content keeps changing underneath it with every pull. The hub install -# path runs skills_guard on install, but project skills are read straight from -# a checkout — without this gate a `git pull` could inject a malicious skill -# into an already-trusted repo with no scan anywhere (#48974). -# -# Every project SKILL.md's parent dir is scanned with the same skills_guard -# scanner the hub uses (content-hash cached, so the cost is one scan per -# skill per content change). A "dangerous" verdict quarantines the skill: it -# is excluded from the index, skills_list, skill_view, and slash commands. -# "caution" loads (matches hub behavior for prose-level keyword hits) — the -# quarantine is for high-confidence findings only. -# -# The scan cache lives under HERMES_HOME, never inside the repo (we don't -# write artifacts into the user's checkout). +# Trust is a repo-level decision made once, but repo content changes with every +# pull — without this gate a `git pull` could inject a malicious skill into an +# already-trusted repo with no scan anywhere (#48974). Every project SKILL.md +# is scanned with the hub's skills_guard scanner (content-hash cached); a +# "dangerous" verdict excludes the skill from index, list, view, and slash +# commands ("caution" loads, as on the hub). Scan cache lives under HERMES_HOME, +# never inside the repo. _PROJECT_SCAN_SOURCE = "project-local" -# (skill_dir_resolved) -> quarantined bool, keyed per-process; scan_skill_cached -# already re-scans on content change via the bundle hash, this only avoids -# re-reading the attestation JSON on every index/list/view call in one run. +# skill_dir -> quarantined; avoids re-reading the attestation JSON per call. _PROJECT_QUARANTINE_CACHE: Dict[str, bool] = {} @@ -921,9 +705,8 @@ def _project_scan_cache_dir() -> Path: def is_quarantined_project_skill(skill_md) -> bool: """True when a project skill's scan verdict is ``dangerous``. - Fail-closed: a scanner crash or missing scanner quarantines the skill - (repo-sourced content with no completed scan must not load). Non-project - callers should not call this — it scans unconditionally. + Fail-closed: a scanner crash or missing scanner quarantines the skill. + Scans unconditionally — non-project callers should not call this. """ skill_dir = Path(skill_md).parent try: @@ -959,32 +742,22 @@ def is_quarantined_project_skill(skill_md) -> bool: return quarantined -def _project_quarantine_cache_clear() -> None: - """Test hook.""" - _PROJECT_QUARANTINE_CACHE.clear() - - def iter_project_skill_files(project_dir: Path): """Yield non-quarantined SKILL.md files under a trusted project dir. - The single iteration chokepoint for the project tier: every consumer - (index, skills_list, slash commands) iterates through here so the - quarantine cannot be bypassed by a new call site forgetting the check. + The single iteration chokepoint for the project tier, so the quarantine + cannot be bypassed by a new call site forgetting the check. """ for skill_md in iter_skill_index_files(project_dir, "SKILL.md"): - if is_quarantined_project_skill(skill_md): - continue - yield skill_md + if not is_quarantined_project_skill(skill_md): + yield skill_md def normalize_skill_lookup_name(identifier: str) -> str: - """Normalize a skill identifier to a ``skill_view()``-safe relative path. + """Translate a trusted absolute skill path to the relative form ``skill_view()`` accepts. - Slash commands and cron jobs may store absolute paths to skills that live - under ``~/.hermes/skills/`` (including via symlinks) or configured - ``skills.external_dirs``. ``skill_view()`` rejects absolute names for - security, so callers must translate trusted absolute paths to their - relative form first. + Slash commands and cron jobs may store absolute paths under the skills + root or ``skills.external_dirs``; skill_view() rejects absolute names. """ raw_identifier = (identifier or "").strip() if not raw_identifier: @@ -994,15 +767,10 @@ def normalize_skill_lookup_name(identifier: str) -> str: if not identifier_path.is_absolute(): return raw_identifier.lstrip("/") - # Look the primary skills root up on tools.skills_tool at CALL time - # (not via get_skills_dir()): callers and tests patch - # ``tools.skills_tool.SKILLS_DIR`` and skill_view() itself resolves - # against ``_skills_dir()`` — which honors that patch and otherwise - # follows the live profile-scoped HERMES_HOME (the import-time - # SKILLS_DIR is frozen to the launch home, #67277) — so normalization - # must agree with the exact root skill_view() will enforce. Import - # deferred to avoid a module cycle (tools.skills_tool imports - # agent.skill_utils). + # Resolve the primary root via tools.skills_tool at CALL time: tests patch + # ``tools.skills_tool.SKILLS_DIR`` and skill_view() enforces ``_skills_dir()`` + # (which also follows the live profile-scoped HERMES_HOME, #67277), so + # normalization must agree with that exact root. Import deferred (cycle). try: from tools import skills_tool as _skills_tool primary_root = _skills_tool._skills_dir() @@ -1010,20 +778,15 @@ def normalize_skill_lookup_name(identifier: str) -> str: primary_root = get_skills_dir() trusted_roots = [primary_root] - try: - trusted_roots.extend(get_project_skills_dirs()) - except Exception: - pass - try: - trusted_roots.extend(get_external_skills_dirs()) - except Exception: - pass + for getter in (get_project_skills_dirs, get_external_skills_dirs): + try: + trusted_roots.extend(getter()) + except Exception: + pass - # Prefer the lexical path under a trusted skill root before resolving - # symlinks. Slash-command discovery can legitimately find a skill via - # ~/.hermes/skills/ where is a symlink to a checked-out - # skill elsewhere. Resolving first turns that trusted visible path into - # an arbitrary absolute path that skill_view() refuses to load. + # Prefer the lexical path under a trusted root before resolving symlinks: + # ~/.hermes/skills/ may be a symlink to a checkout elsewhere, and + # resolving first would turn that trusted path into one skill_view rejects. for root in trusted_roots: try: return str(identifier_path.relative_to(root)) @@ -1050,85 +813,59 @@ def _resolve_for_skill_ownership(path) -> Path: def is_external_skill_path(path) -> bool: - """Return True when ``path`` lives under a configured external skills dir. + """True when ``path`` lives under an external or trusted project skills dir. - ``skills.external_dirs`` are externally owned: Hermes can discover and view - their skills, and foreground user-directed tool calls may still edit them, - but autonomous lifecycle maintenance must treat them as read-only. This - helper centralizes the ownership boundary so curator/reporting/tool paths do - not each need to re-interpret the config. + Those dirs are externally owned: autonomous lifecycle maintenance must + treat them as read-only (user-directed tool calls may still edit them). """ candidate = _resolve_for_skill_ownership(path) roots: List[Path] = list(get_external_skills_dirs()) - # Trusted project-local dirs are repo-owned — same read-only boundary - # for autonomous lifecycle maintenance as configured external dirs. try: roots.extend(get_project_skills_dirs()) except Exception: pass for root in roots: - resolved_root = _resolve_for_skill_ownership(root) try: - candidate.relative_to(resolved_root) + candidate.relative_to(_resolve_for_skill_ownership(root)) return True except ValueError: continue return False -# ── Condition extraction ────────────────────────────────────────────────── +# ── Frontmatter metadata extraction ─────────────────────────────────────── + + +def _hermes_metadata(frontmatter: Dict[str, Any]) -> Dict[str, Any]: + """``metadata.hermes`` mapping from frontmatter, or ``{}`` when malformed.""" + metadata = frontmatter.get("metadata") + if not isinstance(metadata, dict): + return {} + hermes = metadata.get("hermes") or {} + return hermes if isinstance(hermes, dict) else {} def extract_skill_conditions(frontmatter: Dict[str, Any]) -> Dict[str, List]: """Extract conditional activation fields from parsed frontmatter.""" - metadata = frontmatter.get("metadata") - # Handle cases where metadata is not a dict (e.g., a string from malformed YAML) - if not isinstance(metadata, dict): - metadata = {} - hermes = metadata.get("hermes") or {} - if not isinstance(hermes, dict): - hermes = {} + hermes = _hermes_metadata(frontmatter) return { "fallback_for_toolsets": hermes.get("fallback_for_toolsets", []), "requires_toolsets": hermes.get("requires_toolsets", []), "fallback_for_tools": hermes.get("fallback_for_tools", []), "requires_tools": hermes.get("requires_tools", []), - # Gateway-channel gate (maintainer-directed, skills-index slim): - # list of session platforms (e.g. ["msteams"]) the skill is FOR. - # Unlike top-level ``platforms:`` (host OS), this hides the skill - # from the index on every other channel — the teams-meeting - # pipeline has no business in a desktop or telegram session's - # index. Empty/absent = visible everywhere (backward compat). + # Gateway-channel gate: session platforms the skill is FOR (hidden from + # the index elsewhere). Unlike ``platforms:`` (host OS). Empty = everywhere. "session_platforms": hermes.get("session_platforms", []), } -# ── Skill config extraction ─────────────────────────────────────────────── - - def extract_skill_config_vars(frontmatter: Dict[str, Any]) -> List[Dict[str, Any]]: - """Extract config variable declarations from parsed frontmatter. + """Extract ``metadata.hermes.config`` declarations (key/description/default/prompt). - Skills declare config.yaml settings they need via:: - - metadata: - hermes: - config: - - key: wiki.path - description: Path to the LLM Wiki knowledge base directory - default: "~/wiki" - prompt: Wiki directory path - - Returns a list of dicts with keys: ``key``, ``description``, ``default``, - ``prompt``. Invalid or incomplete entries are silently skipped. + Entries missing ``key`` or ``description`` are skipped; ``prompt`` defaults + to the description. """ - metadata = frontmatter.get("metadata") - if not isinstance(metadata, dict): - return [] - hermes = metadata.get("hermes") - if not isinstance(hermes, dict): - return [] - raw = hermes.get("config") + raw = _hermes_metadata(frontmatter).get("config") if not raw: return [] if isinstance(raw, dict): @@ -1144,35 +881,28 @@ def extract_skill_config_vars(frontmatter: Dict[str, Any]) -> List[Dict[str, Any key = str(item.get("key", "")).strip() if not key or key in seen: continue - # Must have at least key and description desc = str(item.get("description", "")).strip() if not desc: continue - entry: Dict[str, Any] = { - "key": key, - "description": desc, - } + entry: Dict[str, Any] = {"key": key, "description": desc} default = item.get("default") if default is not None: entry["default"] = default prompt_text = item.get("prompt") - if isinstance(prompt_text, str) and prompt_text.strip(): - entry["prompt"] = prompt_text.strip() - else: - entry["prompt"] = desc + entry["prompt"] = ( + prompt_text.strip() + if isinstance(prompt_text, str) and prompt_text.strip() + else desc + ) seen.add(key) result.append(entry) return result def discover_all_skill_config_vars() -> List[Dict[str, Any]]: - """Scan all enabled skills and collect their config variable declarations. + """Config var declarations across all enabled, platform-compatible skills. - Walks every skills directory, parses each SKILL.md frontmatter, and returns - a deduplicated list of config var dicts. Each dict also includes a - ``skill`` key with the skill name for attribution. - - Disabled and platform-incompatible skills are excluded. + Deduplicated by key; each dict carries a ``skill`` attribution key. """ all_vars: List[Dict[str, Any]] = [] seen_keys: set = set() @@ -1183,19 +913,15 @@ def discover_all_skill_config_vars() -> List[Dict[str, Any]]: continue for skill_file in iter_skill_index_files(skills_dir, "SKILL.md"): try: - raw = skill_file.read_text(encoding="utf-8") - frontmatter, _ = parse_frontmatter(raw) + frontmatter, _ = parse_frontmatter(skill_file.read_text(encoding="utf-8")) except Exception: continue skill_name = frontmatter.get("name") or skill_file.parent.name - if str(skill_name) in disabled: - continue - if not skill_matches_platform(frontmatter): + if str(skill_name) in disabled or not skill_matches_platform(frontmatter): continue - config_vars = extract_skill_config_vars(frontmatter) - for var in config_vars: + for var in extract_skill_config_vars(frontmatter): if var["key"] not in seen_keys: var["skill"] = str(skill_name) all_vars.append(var) @@ -1204,17 +930,14 @@ def discover_all_skill_config_vars() -> List[Dict[str, Any]]: return all_vars -# Storage prefix: all skill config vars are stored under skills.config.* -# in config.yaml. Skill authors declare logical keys (e.g. "wiki.path"); -# the system adds this prefix for storage and strips it for display. +# Skill config vars are stored under skills.config. in config.yaml. SKILL_CONFIG_PREFIX = "skills.config" def _resolve_dotpath(config: Dict[str, Any], dotted_key: str): - """Walk a nested dict following a dotted key. Returns None if any part is missing.""" - parts = dotted_key.split(".") + """Walk a nested dict following a dotted key; None if any part is missing.""" current = config - for part in parts: + for part in dotted_key.split("."): if isinstance(current, dict) and part in current: current = current[part] else: @@ -1225,25 +948,20 @@ def _resolve_dotpath(config: Dict[str, Any], dotted_key: str): def resolve_skill_config_values( config_vars: List[Dict[str, Any]], ) -> Dict[str, Any]: - """Resolve current values for skill config vars from config.yaml. + """Map logical skill config keys to current values (or declared defaults). - Skill config is stored under ``skills.config.`` in config.yaml. - Returns a dict mapping **logical** keys (as declared by skills) to their - current values (or the declared default if the key isn't set). - Path values are expanded via ``os.path.expanduser``. + Path-like string values are ``~``/``${VAR}`` expanded. """ config = _load_raw_config() resolved: Dict[str, Any] = {} for var in config_vars: logical_key = var["key"] - storage_key = f"{SKILL_CONFIG_PREFIX}.{logical_key}" - value = _resolve_dotpath(config, storage_key) + value = _resolve_dotpath(config, f"{SKILL_CONFIG_PREFIX}.{logical_key}") if value is None or (isinstance(value, str) and not value.strip()): value = var.get("default", "") - # Expand ~ in path-like values if isinstance(value, str) and ("~" in value or "${" in value): value = os.path.expanduser(os.path.expandvars(value)) @@ -1266,8 +984,6 @@ def _normalize_skill_description(frontmatter: Dict[str, Any]) -> str: def extract_skill_description(frontmatter: Dict[str, Any]) -> str: """Extract a system-prompt-length description from parsed frontmatter.""" desc = _normalize_skill_description(frontmatter) - if not desc: - return "" if len(desc) > SKILL_PROMPT_DESC_LIMIT: return desc[:SKILL_PROMPT_DESC_LIMIT - 3] + "..." return desc @@ -1275,8 +991,7 @@ def extract_skill_description(frontmatter: Dict[str, Any]) -> str: def is_skill_description_truncated_for_prompt(frontmatter: Dict[str, Any]) -> bool: """True when the description will be truncated in the system prompt skill index.""" - desc = _normalize_skill_description(frontmatter) - return len(desc) > SKILL_PROMPT_DESC_LIMIT + return len(_normalize_skill_description(frontmatter)) > SKILL_PROMPT_DESC_LIMIT # ── File iteration ──────────────────────────────────────────────────────── @@ -1285,17 +1000,9 @@ def is_skill_description_truncated_for_prompt(frontmatter: Dict[str, Any]) -> bo def iter_skill_index_files(skills_dir: Path, filename: str): """Walk skills_dir yielding sorted paths matching *filename*. - Excludes Hermes metadata, VCS, virtualenv/dependency, cache, and skill - support directories. Support directories (references/templates/assets/ - scripts) can contain arbitrary markdown and even archived package - ``SKILL.md`` files, but they are progressive-disclosure data loaded through - ``skill_view(..., file_path=...)`` rather than active skill roots. - - M2 org mirrors (``_org/``): TOKEN-GATED resolution. Only the active org's - subdir (per the sync-client-written ``.active_org`` marker) is walked; - every other ``_org//`` (stale mirror from a previous org, or no - marker at all) is pruned — leave an org and its skills stop resolving, - without any manual cleanup. + Prunes EXCLUDED_SKILL_DIRS and support dirs of skill roots. Org mirrors + (``_org/``) are TOKEN-GATED: only the active org's subdir is walked, so + leaving an org stops its skills resolving without manual cleanup. """ skills_dir_str = str(skills_dir) active_org = read_active_org_id(skills_dir) @@ -1306,7 +1013,6 @@ def iter_skill_index_files(skills_dir: Path, filename: str): if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None: dirs.remove(ORG_MIRROR_DIR_NAME) elif root == org_root: - # Inside _org/: descend ONLY into the active org's mirror. dirs[:] = [d for d in dirs if d == active_org] dirs[:] = [ d @@ -1326,10 +1032,7 @@ _NAMESPACE_RE = re.compile(r"^[a-zA-Z0-9_-]+$") def parse_qualified_name(name: str) -> Tuple[Optional[str], str]: - """Split ``'namespace:skill-name'`` into ``(namespace, bare_name)``. - - Returns ``(None, name)`` when there is no ``':'``. - """ + """Split ``'namespace:skill-name'`` into ``(namespace, bare_name)``; ``(None, name)`` without ``':'``.""" if ":" not in name: return None, name return tuple(name.split(":", 1)) # type: ignore[return-value] @@ -1337,6 +1040,4 @@ def parse_qualified_name(name: str) -> Tuple[Optional[str], str]: def is_valid_namespace(candidate: Optional[str]) -> bool: """Check whether *candidate* is a valid namespace (``[a-zA-Z0-9_-]+``).""" - if not candidate: - return False - return bool(_NAMESPACE_RE.match(candidate)) + return bool(candidate) and bool(_NAMESPACE_RE.match(candidate)) diff --git a/agent/subdirectory_hints.py b/agent/subdirectory_hints.py index 11e4a7d4f9..6ad23a2372 100644 --- a/agent/subdirectory_hints.py +++ b/agent/subdirectory_hints.py @@ -1,16 +1,11 @@ """Progressive subdirectory hint discovery. -As the agent navigates into subdirectories via tool calls (read_file, terminal, -search_files, etc.), this module discovers and loads project context files -(AGENTS.md, CLAUDE.md, .cursorrules) from those directories. Discovered hints -are appended to the tool result so the model gets relevant context at the moment -it starts working in a new area of the codebase. - -This complements the startup context loading in ``prompt_builder.py`` which only -loads from the CWD. Subdirectory hints are discovered lazily and injected into -the conversation without modifying the system prompt (preserving prompt caching). - -Inspired by Block/goose's SubdirectoryHintTracker. +As the agent navigates into subdirectories via tool calls, this module loads +project context files (AGENTS.md, CLAUDE.md, .cursorrules) from those +directories and appends them to the tool result — context arrives without +touching the system prompt (preserving prompt caching). Complements the +startup CWD-only loading in ``prompt_builder.py``. Inspired by goose's +SubdirectoryHintTracker. """ import hashlib @@ -24,32 +19,21 @@ from agent.prompt_builder import _scan_context_content logger = logging.getLogger(__name__) -# Context files to look for in subdirectories, in priority order. -# Same filenames as prompt_builder.py but we load ALL found (not first-wins) -# since different subdirectories may use different conventions. +# Same filenames as prompt_builder.py, in priority order (first match wins per dir). _HINT_FILENAMES = [ "AGENTS.override.md", "AGENTS.md", "agents.md", "CLAUDE.md", "claude.md", ".cursorrules", ] - -# Maximum chars per hint file to prevent context bloat _MAX_HINT_CHARS = 8_000 - -# Tool argument keys that typically contain file paths _PATH_ARG_KEYS = {"path", "file_path", "workdir"} - -# Tools that take shell commands where we should extract paths _COMMAND_TOOLS = {"terminal"} - -# How many parent directories to walk up when looking for hints. -# Prevents scanning all the way to / for deeply nested paths. +# Ancestor levels walked per path — bounds the scan for deeply nested paths. _MAX_ANCESTOR_WALK = 5 -# Directory names that never contain authoritative project context. -# Backups, vendored deps, VCS internals, and caches routinely hold *copies* of -# AGENTS.md; loading those duplicates real context and inflates the prompt. +# Directories that hold *copies* of context files (backups, vendored deps, +# VCS internals, caches), never authoritative project context. _EXCLUDED_DIR_NAMES = frozenset({ "node_modules", "venv", ".venv", "__pycache__", ".git", ".hg", ".svn", @@ -61,7 +45,7 @@ _EXCLUDED_DIR_NAMES = frozenset({ def _is_ancestor_or_same(a: Path, b: Path) -> bool: - """Check if *a* is the same as or an ancestor of *b* (parent directory check).""" + """True if *a* is *b* or one of its ancestors.""" try: b.relative_to(a) return True @@ -72,34 +56,21 @@ def _is_ancestor_or_same(a: Path, b: Path) -> bool: class SubdirectoryHintTracker: """Track which directories the agent visits and load hints on first access. - Usage:: - - tracker = SubdirectoryHintTracker(working_dir="/path/to/project") - - # After each tool call: - hints = tracker.check_tool_call("read_file", {"path": "backend/src/main.py"}) - if hints: - tool_result += hints # append to the tool result string + Usage: after each tool call, ``hints = tracker.check_tool_call(name, args)`` + and append the returned text to the tool result. """ def __init__(self, working_dir: Optional[str] = None): self.working_dir = Path(working_dir or os.getcwd()).resolve() - self._loaded_dirs: Set[Path] = set() - # Content digests already injected — prevents re-sending the same file - # reachable through symlinks, hardlinks, or duplicated copies. + # The working dir is pre-marked loaded (startup context handles it). + self._loaded_dirs: Set[Path] = {self.working_dir} + # Content digests already injected: the same file reached through + # symlinks/hardlinks/copies is never re-sent. self._loaded_digests: Set[str] = set() - # Pre-mark the working dir as loaded (startup context handles it) - self._loaded_dirs.add(self.working_dir) self._seed_working_dir_digest() def _seed_working_dir_digest(self) -> None: - """Record the CWD context file's digest so it is never re-injected. - - ``prompt_builder`` already loads the working directory's context file at - startup. Seeding its digest here means the same content reached through - a different path (a symlink farm, a shared workspace) is recognised as a - duplicate instead of being sent a second time. - """ + """Record the CWD context file's digest (prompt_builder already loaded it).""" for filename in _HINT_FILENAMES: candidate = self.working_dir / filename try: @@ -119,23 +90,14 @@ class SubdirectoryHintTracker: tool_name: str, tool_args: Dict[str, Any], ) -> Optional[str]: - """Check tool call arguments for new directories and load any hint files. - - Returns formatted hint text to append to the tool result, or None. - """ - dirs = self._extract_directories(tool_name, tool_args) - if not dirs: - return None - + """Return formatted hint text for newly visited directories, or None.""" all_hints = [] - for d in dirs: + for d in self._extract_directories(tool_name, tool_args): hints = self._load_hints_for_directory(d) if hints: all_hints.append(hints) - if not all_hints: return None - return "\n\n" + "\n\n".join(all_hints) def _extract_directories( @@ -143,39 +105,30 @@ class SubdirectoryHintTracker: ) -> list: """Extract directory paths from tool call arguments.""" candidates: Set[Path] = set() - - # Direct path arguments for key in _PATH_ARG_KEYS: val = args.get(key) if isinstance(val, str) and val.strip(): self._add_path_candidate(val, candidates) - - # Shell commands — extract path-like tokens if tool_name in _COMMAND_TOOLS: cmd = args.get("command", "") if isinstance(cmd, str): self._extract_paths_from_command(cmd, candidates) - return list(candidates) def _add_path_candidate(self, raw_path: str, candidates: Set[Path]): - """Resolve a raw path and add its directory + ancestors to candidates. + """Add a raw path's directory and its ancestors to candidates. - Walks up from the resolved directory toward the filesystem root, - stopping at the first directory already in ``_loaded_dirs`` (or after - ``_MAX_ANCESTOR_WALK`` levels). This ensures that reading - ``project/src/main.py`` discovers ``project/AGENTS.md`` even when - ``project/src/`` has no hint files of its own. + Walks up toward the root, stopping at the first already-loaded + directory or after ``_MAX_ANCESTOR_WALK`` levels, so reading + ``project/src/main.py`` still discovers ``project/AGENTS.md``. """ try: p = Path(raw_path).expanduser() if not p.is_absolute(): p = self.working_dir / p p = p.resolve() - # Use parent if it's a file path (has extension or doesn't exist as dir) if p.suffix or (p.exists() and p.is_file()): p = p.parent - # Walk up ancestors — stop at already-loaded or root for _ in range(_MAX_ANCESTOR_WALK): if p in self._loaded_dirs: break @@ -189,32 +142,34 @@ class SubdirectoryHintTracker: pass def _extract_paths_from_command(self, cmd: str, candidates: Set[Path]): - """Extract path-like tokens from a shell command string.""" + """Extract path-like tokens (contain / or .; not flags or URLs) from a shell command.""" try: tokens = shlex.split(cmd) except ValueError: tokens = cmd.split() - for token in tokens: - # Skip flags if token.startswith("-"): continue - # Must look like a path (contains / or .) if "/" not in token and "." not in token: continue - # Skip URLs if token.startswith(("http://", "https://", "git@")): continue self._add_path_candidate(token, candidates) - def _is_valid_subdir(self, path: Path) -> bool: - """Check if path is a valid directory to scan for hints. + def _within_working_dir(self, path: Path) -> bool: + """Reject paths outside the working-dir tree. - Only allow subdirectories within the working directory tree. - This prevents loading AGENTS.md from outside the active workspace - (e.g. ~/.codex/AGENTS.md, ~/.claude/CLAUDE.md), which causes - cross-agent context contamination and instruction mixup. + Loading ~/.codex/AGENTS.md or ~/.claude/CLAUDE.md would mix another + agent's instructions into this session. ``is_relative_to`` handles + symlinked paths; the ancestor check is a best-effort fallback. """ + try: + return path.is_relative_to(self.working_dir) + except (OSError, ValueError): + return _is_ancestor_or_same(self.working_dir, path) + + def _is_valid_subdir(self, path: Path) -> bool: + """Directory inside the working-dir tree, not yet loaded, not an excluded copy dir.""" try: if not path.is_dir(): return False @@ -222,59 +177,31 @@ class SubdirectoryHintTracker: return False if path in self._loaded_dirs: return False - # Reject paths outside the working directory tree. - # path.resolve() may differ from working_dir.resolve() due to symlinks, - # but path.is_relative_to(working_dir) handles both absolute and - # symlinked paths correctly on Python 3.9+. - try: - if not path.is_relative_to(self.working_dir): - return False - except (OSError, ValueError): - # Older Python or path resolution error — fall back to parent - # check as a best-effort safeguard. - if not _is_ancestor_or_same(self.working_dir, path): - return False - if self._is_excluded(path): + if not self._within_working_dir(path): return False - return True + return not self._is_excluded(path) def _is_excluded(self, path: Path) -> bool: - """True when the path sits inside a directory that holds copies, not context. + """True when a segment *below* the working dir is an excluded copy dir. - Directories the user is deliberately working inside are never excluded — - if ``working_dir`` is itself under ``vendor/``, that segment is legitimate - and only segments *below* the working dir are screened. + Only segments under ``working_dir`` are screened: a user deliberately + working inside ``vendor/`` keeps that segment legitimate. """ try: rel_parts = path.relative_to(self.working_dir).parts except ValueError: - # Paths outside the working dir are already rejected by - # _is_valid_subdir before this runs; treat as excluded defensively. - return True + return True # outside the tree — already rejected upstream return any(part in _EXCLUDED_DIR_NAMES for part in rel_parts) def _load_hints_for_directory(self, directory: Path) -> Optional[str]: - """Load hint files from a directory. Returns formatted text or None. - - Only loads hints from directories within the working directory tree. - """ + """Load the first hint file in *directory*; formatted text or None.""" self._loaded_dirs.add(directory) - - # Reject paths outside the working directory tree. - try: - if not directory.is_relative_to(self.working_dir): - logger.debug( - "Skipping hint files in %s — outside working_dir %s", - directory, self.working_dir, - ) - return None - except (OSError, ValueError): - if not _is_ancestor_or_same(self.working_dir, directory): - logger.debug( - "Skipping hint files in %s — outside working_dir %s", - directory, self.working_dir, - ) - return None + if not self._within_working_dir(directory): + logger.debug( + "Skipping hint files in %s — outside working_dir %s", + directory, self.working_dir, + ) + return None found_hints = [] for filename in _HINT_FILENAMES: @@ -288,10 +215,6 @@ class SubdirectoryHintTracker: content = hint_path.read_text(encoding="utf-8").strip() if not content: continue - # Skip content we've already injected. The same AGENTS.md is - # routinely reachable through several paths (symlinked shared - # workspaces, hardlinks, copied backups); re-sending it burns - # context for zero new information. digest = hashlib.sha256(content.encode("utf-8")).hexdigest() if digest in self._loaded_digests: logger.debug( @@ -301,14 +224,13 @@ class SubdirectoryHintTracker: ) break self._loaded_digests.add(digest) - # Same security scan as startup context loading + # Same security scan as startup context loading. content = _scan_context_content(content, filename) if len(content) > _MAX_HINT_CHARS: content = ( content[:_MAX_HINT_CHARS] + f"\n\n[...truncated {filename}: {len(content):,} chars total]" ) - # Best-effort relative path for display rel_path = str(hint_path) try: rel_path = str(hint_path.relative_to(self.working_dir)) @@ -320,20 +242,17 @@ class SubdirectoryHintTracker: except (ValueError, RuntimeError): pass # keep absolute found_hints.append((rel_path, content)) - # First match wins per directory (like startup loading) - break + break # first match wins per directory (like startup loading) except Exception as exc: logger.debug("Could not read %s: %s", hint_path, exc) if not found_hints: return None - sections = [] - for rel_path, content in found_hints: - sections.append( - f"[Subdirectory context discovered: {rel_path}]\n{content}" - ) - + sections = [ + f"[Subdirectory context discovered: {rel_path}]\n{content}" + for rel_path, content in found_hints + ] logger.debug( "Loaded subdirectory hints from %s: %s", directory, diff --git a/agent/think_scrubber.py b/agent/think_scrubber.py index 374035e997..82a27e3b2a 100644 --- a/agent/think_scrubber.py +++ b/agent/think_scrubber.py @@ -1,29 +1,10 @@ """Stateful scrubber for reasoning/thinking blocks in streamed assistant text. -``run_agent._strip_think_blocks`` is regex-based and correct for a complete -string, but when it runs *per-delta* in ``_fire_stream_delta`` it destroys -the state that downstream consumers (CLI ``_stream_delta``, gateway -``GatewayStreamConsumer._filter_and_accumulate``) rely on. - -Concretely, when MiniMax-M2.7 streams - - delta1 = "" - delta2 = "Let me check their config" - delta3 = "" - -the per-delta regex erases delta1 entirely (case 2: unterminated-open at -boundary matches ``^...``), so the downstream state machine never -sees the open tag, treats delta2 as regular content, and leaks reasoning -to the user. Consumers that don't run their own state machine (ACP, -api_server, TTS) never had any defence at all — they just emitted -whatever survived the upstream regex. - -This module centralises the tag-suppression state machine at the -upstream layer so every stream_delta_callback sees text that has -already had reasoning blocks removed. Partial tags at delta -boundaries are held back until the next delta resolves them, and -end-of-stream flushing surfaces any held-back prose that turned out -not to be a real tag. +The regex ``run_agent._strip_think_blocks`` is correct for a complete string but, +run per-delta, erases an opening ```` that arrives alone in one delta, so +downstream state machines never see the open tag and leak reasoning. This class +centralises tag suppression upstream: partial tags at delta boundaries are held +back until resolved, and ``flush()`` releases held-back prose that was not a tag. Usage:: @@ -33,25 +14,15 @@ Usage:: if visible: emit(visible) tail = scrubber.flush() # at end of stream - if tail: - emit(tail) -The scrubber is re-entrant per agent instance. Call ``reset()`` at -the top of each new turn so a hung block from an interrupted prior -stream cannot taint the next turn's output. +Call ``reset()`` at the top of each turn so an interrupted block cannot taint +the next turn. Tags handled (case-insensitive): ````, ````, +````, ````, ````. -Tag variants handled (case-insensitive): - ````, ````, ````, ````, - ````. - -Block-boundary rule for opens: an opening tag is only treated as a -reasoning-block opener when it appears at the start of the stream, -after a newline (optionally followed by whitespace), or when only -whitespace has been emitted on the current line. This prevents prose -that *mentions* the tag name (e.g. ``"use tags here"``) from -being incorrectly suppressed. Closed pairs (``X``) are -always suppressed regardless of boundary; a closed pair is an -intentional, bounded construct. +Boundary rule: an opening tag only starts a block at a block boundary (stream +start, after a newline, or with only whitespace emitted on the current line), so +prose that *mentions* ```` is not suppressed. Closed pairs +(``X``) are always suppressed — a closed pair is intentional. """ from __future__ import annotations @@ -64,16 +35,10 @@ __all__ = ["StreamingThinkScrubber"] class StreamingThinkScrubber: """Stateful scrubber for streaming reasoning/thinking blocks. - State machine: - - ``_in_block``: True while inside an opened block, waiting for - a close tag. All text inside is discarded. - - ``_buf``: held-back partial-tag tail. Emitted / discarded on - the next ``feed()`` call or by ``flush()``. - - ``_last_emitted_ended_newline``: True iff the most recent - emission to the consumer ended with ``\\n``, or nothing has - been emitted yet (start-of-stream counts as a boundary). Used - to decide whether an open tag at buffer position 0 is at a - block boundary. + State: ``_in_block`` (inside an open block; text discarded), ``_buf`` + (held-back partial-tag tail), ``_last_emitted_ended_newline`` (True iff the + last emission ended with ``\\n`` or nothing has been emitted yet — decides + whether an open tag at buffer position 0 sits at a block boundary). """ _OPEN_TAG_NAMES: Tuple[str, ...] = ( @@ -84,31 +49,33 @@ class StreamingThinkScrubber: "REASONING_SCRATCHPAD", ) - # Materialise literal tag strings so the hot path does string - # operations, not regex compilation per feed(). + # Literal tag strings so the hot path does string ops, not regex per feed(). _OPEN_TAGS: Tuple[str, ...] = tuple(f"<{name}>" for name in _OPEN_TAG_NAMES) _CLOSE_TAGS: Tuple[str, ...] = tuple(f"" for name in _OPEN_TAG_NAMES) - - # Pre-compute the longest tag (for partial-tag hold-back bound). _MAX_TAG_LEN: int = max(len(tag) for tag in _OPEN_TAGS + _CLOSE_TAGS) def __init__(self) -> None: + self.reset() + + def reset(self) -> None: + """Reset all state. Call at the top of every new turn.""" self._in_block: bool = False self._buf: str = "" self._last_emitted_ended_newline: bool = True - def reset(self) -> None: - """Reset all state. Call at the top of every new turn.""" - self._in_block = False - self._buf = "" - self._last_emitted_ended_newline = True + def _emit(self, out: list[str], text: str) -> None: + """Append visible prose to *out* (orphan close tags stripped) and track the newline flag.""" + if text: + text = self._strip_orphan_close_tags(text) + if text: + out.append(text) + self._last_emitted_ended_newline = text.endswith("\n") def feed(self, text: str) -> str: """Feed one delta; return the scrubbed visible portion. - May return an empty string when the entire delta is reasoning - content or is being held back pending resolution of a partial - tag at the boundary. + Returns "" when the whole delta is reasoning content or is held back + pending resolution of a partial tag at the boundary. """ if not text: return "" @@ -118,130 +85,67 @@ class StreamingThinkScrubber: while buf: if self._in_block: - # Hunt for the earliest close tag. - close_idx, close_len = self._find_first_tag( - buf, self._CLOSE_TAGS, - ) + close_idx, close_len = self._find_first_tag(buf, self._CLOSE_TAGS) if close_idx == -1: - # No close yet — hold back a potential partial - # close-tag prefix; discard everything else. + # No close yet: hold back a possible partial close-tag prefix, drop the rest. held = self._max_partial_suffix(buf, self._CLOSE_TAGS) self._buf = buf[-held:] if held else "" return "".join(out) - # Found close: discard block content + tag, continue. buf = buf[close_idx + close_len:] self._in_block = False + continue + + # Priority 1: closed X pair anywhere (no boundary gating — + # even inline pairs are almost certainly leaked reasoning). + # Priority 2: unterminated open tag at a block boundary (gated so + # prose that mentions '' isn't over-stripped). Earliest wins. + pair = self._find_earliest_closed_pair(buf) + open_idx, open_len = self._find_open_at_boundary(buf, out) + if pair is not None and (open_idx == -1 or pair[0] <= open_idx): + self._emit(out, buf[:pair[0]]) + buf = buf[pair[1]:] + continue + if open_idx != -1: + self._emit(out, buf[:open_idx]) + self._in_block = True + buf = buf[open_idx + open_len:] + continue + + # No resolvable tag: hold back any partial-tag prefix at the tail + # so a tag split across deltas isn't missed, then emit the rest. + held = max( + self._max_partial_suffix(buf, self._OPEN_TAGS), + self._max_partial_suffix(buf, self._CLOSE_TAGS), + ) + if held: + self._emit(out, buf[:-held]) + self._buf = buf[-held:] else: - # Priority 1 — closed X pair anywhere in - # buf. Closed pairs are always an intentional, - # bounded construct (even mid-line prose containing - # an open/close pair is almost certainly a model - # leaking reasoning inline), so no boundary gating. - pair = self._find_earliest_closed_pair(buf) - # Priority 2 — unterminated open tag at a block - # boundary. Boundary-gated so prose that mentions - # '' isn't over-stripped. - open_idx, open_len = self._find_open_at_boundary( - buf, out, - ) - - # Pick whichever match comes earliest in the buffer. - if pair is not None and ( - open_idx == -1 or pair[0] <= open_idx - ): - start_idx, end_idx = pair - preceding = buf[:start_idx] - if preceding: - preceding = self._strip_orphan_close_tags(preceding) - if preceding: - out.append(preceding) - self._last_emitted_ended_newline = ( - preceding.endswith("\n") - ) - buf = buf[end_idx:] - continue - - if open_idx != -1: - # Unterminated open at boundary — emit preceding, - # enter block, continue loop with remainder. - preceding = buf[:open_idx] - if preceding: - preceding = self._strip_orphan_close_tags(preceding) - if preceding: - out.append(preceding) - self._last_emitted_ended_newline = ( - preceding.endswith("\n") - ) - self._in_block = True - buf = buf[open_idx + open_len:] - continue - - # No resolvable tag structure in buf. Hold back any - # partial-tag prefix at the tail so a split tag - # across deltas isn't missed, then emit the rest. - held = self._max_partial_suffix(buf, self._OPEN_TAGS) - held_close = self._max_partial_suffix( - buf, self._CLOSE_TAGS, - ) - held = max(held, held_close) - if held: - emit_text = buf[:-held] - self._buf = buf[-held:] - else: - emit_text = buf - self._buf = "" - if emit_text: - emit_text = self._strip_orphan_close_tags(emit_text) - if emit_text: - out.append(emit_text) - self._last_emitted_ended_newline = ( - emit_text.endswith("\n") - ) - return "".join(out) + self._emit(out, buf) + return "".join(out) return "".join(out) def flush(self) -> str: """End-of-stream flush. - If still inside an unterminated block, held-back content is - discarded — leaking partial reasoning is worse than a - truncated answer. Otherwise the held-back partial-tag tail is - emitted verbatim (it turned out not to be a real tag prefix). - - Always treats the next ``feed()`` as a fresh stream boundary. - Intra-turn retries (thinking-only prefill, empty-response - retry) flush then stream again without calling ``reset()``; - leaving ``_last_emitted_ended_newline`` False made a new - stream's opening ```` look mid-line and leak into the - visible reply. + Inside an unterminated block the held-back content is discarded (leaking + partial reasoning is worse than a truncated answer); otherwise the + held-back tail is emitted verbatim. Always resets the boundary flag: + intra-turn retries flush then stream again without ``reset()``, and a + stale False flag made the new stream's opening ```` look mid-line. """ - if self._in_block: - self._buf = "" - self._in_block = False - # Next feed() is a new stream — start-of-stream is a boundary. - self._last_emitted_ended_newline = True - return "" - tail = self._buf + tail = "" if self._in_block else self._buf self._buf = "" - # Same for the non-block path: do NOT derive the boundary flag - # from the flushed tail (e.g. a held-back '<'). End-of-stream - # means the next feed() starts a new model response. + self._in_block = False self._last_emitted_ended_newline = True - if not tail: - return "" - return self._strip_orphan_close_tags(tail) + return self._strip_orphan_close_tags(tail) if tail else "" # ── internal helpers ─────────────────────────────────────────────── @staticmethod - def _find_first_tag( - buf: str, tags: Tuple[str, ...], - ) -> Tuple[int, int]: - """Return (earliest_index, tag_length) over *tags*, or (-1, 0). - - Case-insensitive match. - """ + def _find_first_tag(buf: str, tags: Tuple[str, ...]) -> Tuple[int, int]: + """Return (earliest_index, tag_length) over *tags* (case-insensitive), or (-1, 0).""" buf_lower = buf.lower() best_idx = -1 best_len = 0 @@ -253,14 +157,10 @@ class StreamingThinkScrubber: return best_idx, best_len def _find_earliest_closed_pair(self, buf: str): - """Return (start_idx, end_idx) of the earliest closed pair, else None. + """Return (start_idx, end_idx) of the earliest ``...`` pair, else None. - A closed pair is ``...`` of any variant. Matches are - case-insensitive and non-greedy (the closest close tag after - an open tag wins), matching the regex ``.*?`` - semantics of ``_strip_think_blocks`` case 1. When two tag - variants could both match, the one whose open tag appears - earlier wins. + Case-insensitive and non-greedy (closest close after the open wins), + matching ``_strip_think_blocks`` case 1; the earliest open tag wins. """ buf_lower = buf.lower() best: "tuple[int, int] | None" = None @@ -270,23 +170,15 @@ class StreamingThinkScrubber: open_idx = buf_lower.find(open_lower) if open_idx == -1: continue - close_idx = buf_lower.find( - close_lower, open_idx + len(open_lower), - ) + close_idx = buf_lower.find(close_lower, open_idx + len(open_lower)) if close_idx == -1: continue - end_idx = close_idx + len(close_lower) if best is None or open_idx < best[0]: - best = (open_idx, end_idx) + best = (open_idx, close_idx + len(close_lower)) return best - def _find_open_at_boundary( - self, buf: str, already_emitted: list[str], - ) -> Tuple[int, int]: - """Return the earliest block-boundary open-tag (idx, len). - - Returns (-1, 0) if no boundary-legal opener is present. - """ + def _find_open_at_boundary(self, buf: str, already_emitted: list[str]) -> Tuple[int, int]: + """Return the earliest block-boundary open-tag (idx, len), or (-1, 0).""" buf_lower = buf.lower() best_idx = -1 best_len = 0 @@ -305,50 +197,30 @@ class StreamingThinkScrubber: search_start = idx + 1 return best_idx, best_len - def _is_block_boundary( - self, buf: str, idx: int, already_emitted: list[str], - ) -> bool: + def _is_block_boundary(self, buf: str, idx: int, already_emitted: list[str]) -> bool: """True iff position *idx* in *buf* is a block boundary. - A block boundary is: - - buf position 0 AND the most recent emission ended with - a newline (or nothing has been emitted yet) - - any position whose preceding text on the current line - (since the last newline in buf) is whitespace-only, AND - if there is no newline in the preceding buf portion, the - most recent prior emission ended with a newline + Boundary = position 0 with the prior emission ending in a newline (or + nothing emitted yet), or any position whose preceding text on the current + line is whitespace-only (when no newline precedes it in *buf*, the prior + emission must also have ended with a newline). """ + prior_newline = ( + already_emitted[-1].endswith("\n") if already_emitted else self._last_emitted_ended_newline + ) if idx == 0: - # Check whether the last already-emitted chunk in THIS - # feed() call ended with a newline, otherwise fall back - # to the cross-feed flag. - if already_emitted: - return already_emitted[-1].endswith("\n") - return self._last_emitted_ended_newline + return prior_newline preceding = buf[:idx] last_nl = preceding.rfind("\n") if last_nl == -1: - # No newline in buf before the tag — boundary only if the - # prior emission ended with a newline AND everything since - # is whitespace. - if already_emitted: - prior_newline = already_emitted[-1].endswith("\n") - else: - prior_newline = self._last_emitted_ended_newline return prior_newline and preceding.strip() == "" - # Newline present — text between it and the tag must be - # whitespace-only. return preceding[last_nl + 1:].strip() == "" @classmethod - def _max_partial_suffix( - cls, buf: str, tags: Tuple[str, ...], - ) -> int: - """Return the longest buf-suffix that is a prefix of any tag. + def _max_partial_suffix(cls, buf: str, tags: Tuple[str, ...]) -> int: + """Longest buf-suffix that is a strict prefix of any tag (case-insensitive). - Only prefixes strictly shorter than the tag itself count - (full-length suffixes are the tag and are handled as matches, - not held-back partials). Case-insensitive. + Full-length matches are real tags handled elsewhere, not held-back partials. """ if not buf: return 0 @@ -364,12 +236,7 @@ class StreamingThinkScrubber: @classmethod def _strip_orphan_close_tags(cls, text: str) -> str: - """Remove any close tags from *text* (orphan-close handling). - - An orphan close tag has no matching open in the current - scrubber state; it's always noise, stripped with any trailing - whitespace so the surrounding prose flows naturally. - """ + """Remove close tags with no matching open (always noise) plus trailing whitespace.""" if " Path: d = repo / ".hermes" / "skills" / "evil-skill" @@ -231,7 +221,7 @@ class TestQuarantine: (evil_dir / "SKILL.md").write_text( "---\nname: evil-skill\ndescription: now actually benign\n---\nbody\n" ) - su._project_quarantine_cache_clear() + su._PROJECT_QUARANTINE_CACHE.clear() assert su.is_quarantined_project_skill(evil_dir / "SKILL.md") is False def test_scan_cache_outside_repo(self, project_env): diff --git a/tests/agent/test_prompt_cache_boundary.py b/tests/agent/test_prompt_cache_boundary.py index d155473f91..fc222bed78 100644 --- a/tests/agent/test_prompt_cache_boundary.py +++ b/tests/agent/test_prompt_cache_boundary.py @@ -17,8 +17,8 @@ from unittest.mock import patch import agent.skill_bundles as skill_bundles import agent.skill_commands as skill_commands import tools.skills_tool as skills_tool +import agent.prompt_cache_boundary as prompt_cache_boundary from agent.prompt_cache_boundary import ( - clear_stable_prefixes, find_stable_prefix, register_stable_prefix, ) @@ -36,9 +36,11 @@ SKILL_BODY = "Inspect the report carefully and preserve the stable instructions. @pytest.fixture(autouse=True) def _isolated_registry(): - clear_stable_prefixes() + with prompt_cache_boundary._lock: + prompt_cache_boundary._prefixes.clear() yield - clear_stable_prefixes() + with prompt_cache_boundary._lock: + prompt_cache_boundary._prefixes.clear() def _write_skill(skills_dir, name, body=SKILL_BODY): diff --git a/tests/agent/test_prompt_caching.py b/tests/agent/test_prompt_caching.py index 38c50a0c7b..e3ce192802 100644 --- a/tests/agent/test_prompt_caching.py +++ b/tests/agent/test_prompt_caching.py @@ -88,7 +88,7 @@ def test_t20880_tool_heavy_native_loop_reproduction(): assert final_tool_marked assert shared_transaction_endpoint - assert after_exchange.marker_count <= 4 + assert _count_cache_markers(after_exchange.messages, after_exchange.tools) <= 4 class TestPromptCachePlan: @@ -114,7 +114,7 @@ class TestPromptCachePlan: assert plan.tools is not tools assert "cache_control" not in tools[-1] assert plan.tools[-1]["cache_control"] == MARKER - assert plan.marker_count == 4 + assert _count_cache_markers(plan.messages, plan.tools) == 4 def test_unmarkable_endpoint_does_not_consume_a_slot(self): messages = [ @@ -129,7 +129,7 @@ class TestPromptCachePlan: direct_native_tool_cache=True, ) - assert plan.marker_count == 2 + assert _count_cache_markers(plan.messages, plan.tools) == 2 assert "cache_control" not in plan.messages[-1] def test_static_prefix_equal_to_whole_prompt_emits_no_empty_block(self): @@ -184,7 +184,7 @@ class TestPromptCachePlan: native_anthropic=True, direct_native_tool_cache=True, ) - assert plan.marker_count == 3 + assert _count_cache_markers(plan.messages, plan.tools) == 3 assert len(plan.tools) == 0