diff --git a/.github/actions/plugin-validate/action.yml b/.github/actions/plugin-validate/action.yml new file mode 100644 index 0000000000..5484aaf56c --- /dev/null +++ b/.github/actions/plugin-validate/action.yml @@ -0,0 +1,56 @@ +name: Hermes Plugin Validate +description: >- + Validate a Hermes Agent plugin (plugin.yaml manifest schema AND + declared-vs-actually-registered capabilities) using + `hermes plugins validate`. Drop this into your plugin repo's CI: + + - uses: actions/checkout@ + - uses: NousResearch/hermes-agent/.github/actions/plugin-validate@main + with: + path: . + + The caller's job owns checkout; this action installs Python + hermes-agent + (git install — a supported CI-context install route) and runs the + validator against your plugin directory. + +inputs: + path: + description: Path to the plugin directory (containing plugin.yaml). + default: "." + hermes-ref: + description: hermes-agent git ref (branch/tag/sha) to install and validate with. + default: "main" + +runs: + using: composite + steps: + - name: Set up Python + uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 + with: + python-version: "3.11" + + - name: Install hermes-agent + shell: bash + env: + _HERMES_REF: ${{ inputs.hermes-ref }} + run: | + set -euo pipefail + # CI-context install from git; the ref lets plugin authors validate + # against a pinned hermes release instead of main. + pip install "git+https://github.com/NousResearch/hermes-agent@${_HERMES_REF}" + + - name: Validate plugin + shell: bash + env: + _PLUGIN_PATH: ${{ inputs.path }} + run: | + set -uo pipefail + # `hermes plugins validate` checks the plugin.yaml manifest schema + # and loads the plugin in a scratch subprocess to verify that the + # capabilities it DECLARES match what it actually registers. + if hermes plugins validate "$_PLUGIN_PATH"; then + echo "✅ PASS: plugin at '$_PLUGIN_PATH' validated cleanly" + else + echo "❌ FAIL: plugin at '$_PLUGIN_PATH' failed validation (see output above)" + exit 1 + fi diff --git a/.github/workflows/deploy-site.yml b/.github/workflows/deploy-site.yml index ccb0c84b02..d024b7772d 100644 --- a/.github/workflows/deploy-site.yml +++ b/.github/workflows/deploy-site.yml @@ -9,6 +9,9 @@ on: - 'website/**' - 'skills/**' - 'optional-skills/**' + # Catalog entry/removal merges must republish /docs/api/plugin-catalog.json — + # installed clients fetch it for live catalog refresh. + - 'plugin-catalog/**' - '.github/workflows/deploy-site.yml' workflow_dispatch: inputs: @@ -153,6 +156,9 @@ jobs: - name: Extract skill metadata for dashboard run: python3 website/scripts/extract-skills.py + - name: Extract plugin catalog for the Plugins page + run: python3 website/scripts/extract-plugins.py + - name: Regenerate per-skill docs pages + catalogs run: python3 website/scripts/generate-skill-docs.py diff --git a/.github/workflows/plugin-catalog-ci.yml b/.github/workflows/plugin-catalog-ci.yml new file mode 100644 index 0000000000..84efbc0761 --- /dev/null +++ b/.github/workflows/plugin-catalog-ci.yml @@ -0,0 +1,142 @@ +name: Plugin Catalog CI + +# Admission gate for plugin-catalog entries. Fires ONLY on PRs touching +# plugin-catalog/** so it can never go red on unrelated PRs. +# +# Two gates: +# structural — cheap schema check, no hermes install needed +# pinned-source-validate — supply-chain gate: the pinned sha MUST be +# reachable in the entry's repo, and the plugin +# at that exact commit must pass +# `hermes plugins validate`. + +on: + pull_request: + paths: + - "plugin-catalog/**" + +permissions: + contents: read + +jobs: + structural: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 + with: + python-version: "3.11" + + - name: Install PyYAML + uses: ./.github/actions/retry + with: + command: pip install pyyaml==6.0.2 + + - name: Validate catalog files (structural) + run: | + set -euo pipefail + # Validating the whole directory is simpler than diffing and keeps + # the invariant that EVERYTHING in plugin-catalog/ stays valid. + python3 scripts/validate_plugin_catalog.py plugin-catalog/ + + pinned-source-validate: + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + fetch-depth: 0 # need the merge-base to diff changed catalog files + + - uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0 + with: + python-version: "3.11" + + - name: Find changed catalog entries + id: changed + run: | + set -euo pipefail + MERGE_BASE=$(git merge-base "origin/${{ github.base_ref }}" HEAD) + # Added + modified entry files only; deletions and removed.yaml + # have nothing to clone. + CHANGED=$(git diff --name-only --diff-filter=AM "$MERGE_BASE"...HEAD \ + -- 'plugin-catalog/*.yaml' 'plugin-catalog/*.yml' \ + | grep -v '/removed\.yaml$' || true) + echo "Changed catalog entries:" + echo "${CHANGED:-}" + { + echo 'files<<__EOF__' + echo "$CHANGED" + echo '__EOF__' + } >> "$GITHUB_OUTPUT" + + - name: Install hermes-agent from the PR's own checkout + if: steps.changed.outputs.files != '' + uses: ./.github/actions/retry + with: + command: pip install -e . + + - name: Clone each entry at its pinned sha and validate + if: steps.changed.outputs.files != '' + env: + CHANGED_FILES: ${{ steps.changed.outputs.files }} + run: | + set -euo pipefail + FAILED=0 + while IFS= read -r entry; do + [ -z "$entry" ] && continue + echo "::group::validate $entry" + + # Parse repo / sha / subdir from the entry yaml. + eval "$(python3 - "$entry" <<'PYEOF' + import shlex + import sys + + import yaml + + with open(sys.argv[1], encoding="utf-8") as fh: + data = yaml.safe_load(fh) or {} + print(f"REPO={shlex.quote(str(data.get('repo', '')))}") + print(f"SHA={shlex.quote(str(data.get('sha', '')))}") + print(f"SUBDIR={shlex.quote(str(data.get('subdir', '') or ''))}") + PYEOF + )" + echo "repo=$REPO sha=$SHA subdir=$SUBDIR" + + CLONE_DIR=$(mktemp -d) + # Full clone (no --depth 1): the pinned sha may not be the branch tip. + if ! git clone "$REPO" "$CLONE_DIR"; then + echo "::error file=$entry::clone failed for $REPO" + FAILED=1; echo "::endgroup::"; continue + fi + + # SUPPLY-CHAIN GATE: the pinned sha must be reachable in the repo. + if ! git -C "$CLONE_DIR" checkout --detach "$SHA"; then + echo "::error file=$entry::pinned sha $SHA is not reachable in $REPO" + FAILED=1; echo "::endgroup::"; continue + fi + + PLUGIN_DIR="$CLONE_DIR${SUBDIR:+/$SUBDIR}" + # Native plugin.yaml OR portable Agent Plugins v1 plugin.json + # (#81196; native manifest wins when both exist). + if [ ! -f "$PLUGIN_DIR/plugin.yaml" ] && [ ! -f "$PLUGIN_DIR/plugin.yml" ] && [ ! -f "$PLUGIN_DIR/plugin.json" ]; then + echo "::error file=$entry::no plugin.yaml or plugin.json at subdir '$SUBDIR' of $REPO@$SHA" + FAILED=1; echo "::endgroup::"; continue + fi + + # Manifest schema + declared-vs-registered capability check. + if hermes plugins validate "$PLUGIN_DIR"; then + echo "✅ PASS: $entry" + else + echo "::error file=$entry::hermes plugins validate failed" + FAILED=1 + fi + echo "::endgroup::" + done <<< "$CHANGED_FILES" + + if [ "$FAILED" -ne 0 ]; then + echo "❌ FAIL: one or more catalog entries failed pinned-source validation" + exit 1 + fi + echo "✅ PASS: all changed catalog entries validated at their pinned shas" diff --git a/.gitignore b/.gitignore index f545754285..ea92770db4 100644 --- a/.gitignore +++ b/.gitignore @@ -1,361 +1,364 @@ - -!apps/desktop/src/global.d.ts -!apps/desktop/src/plugins/*/plugin.js -!apps/desktop/src/vite-env.d.ts -!hermes_cli/data/ -!hermes_cli/data/plugin_index.json -# — ignore so `git status` stays clean and update's autostash skips them. -# (launch-time stale-bytecode sweep). Runtime state, never a code change. -# `data/` pattern above would otherwise swallow it. -# `npm run sync-assets` (see web/package.json). -# also created in-repo when an agent operates in this checkout). Plans, audit -# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). -# automation-blueprints-index.json is a build artifact emitted by -# bootstrap installer. It is Hermes-managed runtime state, never a code change — -# Bundled community plugin index seed (shipped as package data) — the bare -# by `hermes update` / launch-time self-heal. Runtime state, never a code change -# by accident via 3a69e34702, removed in the #72002 salvage). -# Checkout fingerprint the __pycache__ tree was last validated against -# CLI config (may contain sensitive SSH paths) -# committed to the repo root. See the hermes-release skill. -# Cross-process web UI build lock (flock target, always empty) -# cut (passed to `gh release create --notes-file`); the GitHub Release itself -# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded -# Desktop/bootstrap install marker written into the managed checkout root by the -# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx -# every build). -# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, -# git for the same reason as skills-index.json (large, generated, change -# ignore it so `hermes update`'s `git stash push --include-untracked` does not -# image-provider (fal.media) URL — they are NEVER committed to the repo. The -# infographic-check CI job is what actually enforces this. -# Installer-written method stamp in the managed checkout root (scripts/install.sh). -# interrupted; consumed by launch-time recovery. Never commit it (was tracked -# Interrupted-update breadcrumb + recovery lock written next to the shared venv -# Local editor / agent tooling (machine-specific; keep in global config, not the repo) -# logs, and per-session caches are never artifacts of the codebase. -# Nix -# No trailing slash: also matches node_modules SYMLINKS (worktrees often -# Per-release changelog drafts. These exist only transiently during a release -# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) -# Playwright visual regression baselines — cached from main in CI, not committed -# PR body is the archive. See the hermes-agent-dev skill's -# PR infographics are rendered locally and embedded in PR descriptions via the -# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1). -# Private keys -# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. -# Release script temp files -# Repo-root build/debug artifacts that must never be committed -# resolves the stale .js OVER the .tsx — never track these) -# Runtime marker written by hermes update when a lazy dependency refresh is -# Runtime metadata only — never a code change. Ignore so `git status` stays clean -# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is -# sibling exists, so the stale-shadow hazard above cannot apply. -# sidestepped by an `infograficos/` directory (#70552). .gitignore is only -# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) -# skills.json + skills-meta.json are build artifacts emitted by -# slip into a commit and break `npm ci` on CI with ENOTDIR). -# Spelling variants are listed because a single `infographic/` pattern was -# stores the published notes. They are not a build artifact and must never be -# symlink node_modules to the main checkout; the dir-only pattern let one -# the first line of defence and cannot stop `git add -f` at all — the -# the route name, so each route gets its own tree and two can run at once. -# Tool Search live-test harness output — non-deterministic model transcripts, -# treat it as a local edit and autostash it on every run (#38529). -# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then -# walkthroughs). Throwaway artifacts, never part of the app. -# Web UI assets — synced from @nous-research/ui at build time via -# Web UI build output -# website/scripts/extract-automation-blueprints.py during prebuild. -# website/scripts/extract-skills.py during prebuild — keep them out of -# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; -# -%SystemDrive%/ -*.pem -*.ppk -*.pyc* -*.tsbuildinfo -*-snapshots/ -.act-sandbox-agent.* -.bytecode-fingerprint -.bytecode-fingerprint.tmp -.codex/ -.cursor/ -.direnv/ -.DS_Store -.env -.env.development -.env.development.local -.env.local -.env.production.local -.env.test -.env.test.local -.gemini/ -.hermes/ -.hermes-bootstrap-complete -.hermes-docker/ -.hermes-sandbox/ -.hermes-sandbox-e2e*/ -.lazy-refresh-incomplete -.mcp.json -.nix-stamps/ -.notebooklm-cli-venv/ -.notebooklm-home/ -.notebooklm-playwright/ -.op.env -.pip-cache/ -.pytest_cache/ -.pytest-cache/ -.release_notes.md -.skills_prompt_snapshot.json -.update-incomplete -.update-incomplete.lock -.uv-cache/ -.venv -.venv/ -.vscode/ -.web_ui_build.lock -.worktrees/ -.zed/ -/*.png.bak -/.hermes-runtime/ -/.install_method -/_pycache/ -/bin/ -/default.tar.gz -/log.txt -/sqlite_leak_fix.png -/venv.old/ -/venv.stale.runtime-*/ -/venv/ -__pycache__/ -__pycache__/model_tools.cpython-310.pyc -__pycache__/web_tools.cpython-310.pyc -act/ -agent-browser/ -apps/desktop/build/ -apps/desktop/demo/ -apps/desktop/dist/ -apps/desktop/release/ -apps/desktop/src/**/*.d.ts -apps/desktop/src/**/*.js -apps/desktop/src/**/*.js.map -apps/shared/src/**/*.d.ts -apps/shared/src/**/*.js -apps/shared/src/**/*.js.map -browser-use/ -cli-config.yaml -compose.hermes.local.yml -config/mcporter.json -data/ -data/* -docs/superpowers/* -environments/benchmarks/evals/ -examples/ -export* -hermes-*/* -hermes_agent.egg-info/ -hermes_cli/scripts/ -hermes_cli/tui_dist/* -hermes_cli/web_dist/ -ignored/ -images/ -infografico/ -infograficos/ -infographic/ -infographics/ -logs/ -mini-swe-agent/ -models-dev-upstream/ -native/fts5_cjk/*.so -node_modules -opencode.json -playwright-report/ -privvy* -RELEASE_v*.md -result -run_datagen_kimik2-thinking.sh -run_datagen_megascience_glm4-6.sh -run_datagen_sonnet.sh -source-data/* -run_datagen_megascience_glm4-6.sh -data/* -# No trailing slash: also matches node_modules SYMLINKS (worktrees often -# symlink node_modules to the main checkout; the dir-only pattern let one -# slip into a commit and break `npm ci` on CI with ENOTDIR). -node_modules -browser-use/ -agent-browser/ -# Private keys -*.ppk -*.pem -privvy* -images/ -__pycache__/ -hermes_agent.egg-info/ -wandb/ -testlogs -playwright-report/ -test-results/ -# Playwright visual regression baselines — cached from main in CI, not committed -*-snapshots/ - -# CLI config (may contain sensitive SSH paths) -cli-config.yaml - -# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) -skills/.hub/ -ignored/ -.worktrees/ -environments/benchmarks/evals/ - -# Web UI build output -hermes_cli/web_dist/ -# Cross-process web UI build lock (flock target, always empty) -.web_ui_build.lock -apps/desktop/build/ -apps/desktop/dist/ - -# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then -# resolves the stale .js OVER the .tsx — never track these) -apps/desktop/src/**/*.js -apps/desktop/src/**/*.js.map -apps/desktop/src/**/*.d.ts -# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, -# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx -# sibling exists, so the stale-shadow hazard above cannot apply. -!apps/desktop/src/plugins/*/plugin.js -!apps/desktop/src/global.d.ts -!apps/desktop/src/vite-env.d.ts - -# Build/debug artifacts that must never be committed -/log.txt -/sqlite_leak_fix.png -/*.png.bak -*.tar.gz -*.tgz -apps/shared/src/**/*.js -apps/shared/src/**/*.js.map -apps/shared/src/**/*.d.ts -apps/desktop/release/ -# stage-and-swap Desktop rebuild output (#86443); removed after the swap, but -# a killed build must not leave the checkout dirty -apps/desktop/.staging-*/ -*.tsbuildinfo - -# Web UI assets — synced from @nous-research/ui at build time via -# `npm run sync-assets` (see web/package.json). -web/public/fonts/ -web/public/ds-assets/ - -# Release script temp files -.release_notes.md -mini-swe-agent/ - -# Nix -.direnv/ -.nix-stamps/ -result -website/static/api/skills-index.json -# skills.json + skills-meta.json are build artifacts emitted by -# website/scripts/extract-skills.py during prebuild — keep them out of -# git for the same reason as skills-index.json (large, generated, change -# every build). -website/static/api/skills.json -website/static/api/skills-meta.json -# automation-blueprints-index.json is a build artifact emitted by -# website/scripts/extract-automation-blueprints.py during prebuild. -website/static/api/automation-blueprints-index.json -models-dev-upstream/ - -# Local editor / agent tooling (machine-specific; keep in global config, not the repo) -.codex/ -.cursor/ -.gemini/ -.zed/ -.mcp.json -opencode.json -config/mcporter.json - -hermes_cli/tui_dist/* -hermes_cli/scripts/ -docs/superpowers/* -# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; -# also created in-repo when an agent operates in this checkout). Plans, audit -# logs, and per-session caches are never artifacts of the codebase. -.hermes/ - -# Desktop/bootstrap install marker written into the managed checkout root by the -# bootstrap installer. It is Hermes-managed runtime state, never a code change — -# ignore it so `hermes update`'s `git stash push --include-untracked` does not -# treat it as a local edit and autostash it on every run (#38529). -.hermes-bootstrap-complete - -# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) -.hermes-sandbox/ -# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is -# the route name, so each route gets its own tree and two can run at once. -.hermes-sandbox-e2e*/ - -# Interrupted-update breadcrumb + recovery lock written next to the shared venv -# by `hermes update` / launch-time self-heal. Runtime state, never a code change -# — ignore so `git status` stays clean and update's autostash skips them. -.update-incomplete -.update-incomplete.lock - -# Checkout fingerprint the __pycache__ tree was last validated against -# (launch-time stale-bytecode sweep). Runtime state, never a code change. -.bytecode-fingerprint -.bytecode-fingerprint.tmp - -# Installer-written method stamp in the managed checkout root (scripts/install.sh). -# Runtime metadata only — never a code change. Ignore so `git status` stays clean -# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). -/.install_method - -# Tool Search live-test harness output — non-deterministic model transcripts, -# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. - -scripts/out/ - -# Per-release changelog drafts. These exist only transiently during a release -# cut (passed to `gh release create --notes-file`); the GitHub Release itself -# stores the published notes. They are not a build artifact and must never be -# committed to the repo root. See the hermes-release skill. - -# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded -# walkthroughs). Throwaway artifacts, never part of the app. - -# PR infographics are rendered locally and embedded in PR descriptions via the -# image-provider (fal.media) URL — they are NEVER committed to the repo. The -# PR body is the archive. See the hermes-agent-dev skill's -# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1). -# -# Spelling variants are listed because a single `infographic/` pattern was -# sidestepped by an `infograficos/` directory (#70552). .gitignore is only -# the first line of defence and cannot stop `git add -f` at all — the -# infographic-check CI job is what actually enforces this. -# Runtime marker written by hermes update when a lazy dependency refresh is -# interrupted; consumed by launch-time recovery. Never commit it (was tracked -# by accident via 3a69e34702, removed in the #72002 salvage). - -# Disposable profile created by scripts/probe_active_session_exclusivity.py -.probe-home/ -skills/.hub/ -source-data/* -temp_vision_images/ -test_durations.json -testlogs -test-results/ -tests/quick_test_dataset.jsonl -tests/sample_dataset.jsonl -tmp/ -wandb/ -web/public/ds-assets/ -web/public/fonts/ -website/static/api/automation-blueprints-index.json -website/static/api/skills.json -website/static/api/skills-index.json + +!apps/desktop/src/global.d.ts +!apps/desktop/src/plugins/*/plugin.js +!apps/desktop/src/vite-env.d.ts +!hermes_cli/data/ +# — ignore so `git status` stays clean and update's autostash skips them. +# (launch-time stale-bytecode sweep). Runtime state, never a code change. +# `data/` pattern above would otherwise swallow it. +# `npm run sync-assets` (see web/package.json). +# also created in-repo when an agent operates in this checkout). Plans, audit +# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). +# automation-blueprints-index.json is a build artifact emitted by +# bootstrap installer. It is Hermes-managed runtime state, never a code change — +# Bundled community plugin index seed (shipped as package data) — the bare +# by `hermes update` / launch-time self-heal. Runtime state, never a code change +# by accident via 3a69e34702, removed in the #72002 salvage). +# Checkout fingerprint the __pycache__ tree was last validated against +# CLI config (may contain sensitive SSH paths) +# committed to the repo root. See the hermes-release skill. +# Cross-process web UI build lock (flock target, always empty) +# cut (passed to `gh release create --notes-file`); the GitHub Release itself +# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded +# Desktop/bootstrap install marker written into the managed checkout root by the +# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx +# every build). +# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, +# git for the same reason as skills-index.json (large, generated, change +# ignore it so `hermes update`'s `git stash push --include-untracked` does not +# image-provider (fal.media) URL — they are NEVER committed to the repo. The +# infographic-check CI job is what actually enforces this. +# Installer-written method stamp in the managed checkout root (scripts/install.sh). +# interrupted; consumed by launch-time recovery. Never commit it (was tracked +# Interrupted-update breadcrumb + recovery lock written next to the shared venv +# Local editor / agent tooling (machine-specific; keep in global config, not the repo) +# logs, and per-session caches are never artifacts of the codebase. +# Nix +# No trailing slash: also matches node_modules SYMLINKS (worktrees often +# Per-release changelog drafts. These exist only transiently during a release +# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) +# Playwright visual regression baselines — cached from main in CI, not committed +# PR body is the archive. See the hermes-agent-dev skill's +# PR infographics are rendered locally and embedded in PR descriptions via the +# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1). +# Private keys +# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. +# Release script temp files +# Repo-root build/debug artifacts that must never be committed +# resolves the stale .js OVER the .tsx — never track these) +# Runtime marker written by hermes update when a lazy dependency refresh is +# Runtime metadata only — never a code change. Ignore so `git status` stays clean +# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is +# sibling exists, so the stale-shadow hazard above cannot apply. +# sidestepped by an `infograficos/` directory (#70552). .gitignore is only +# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) +# skills.json + skills-meta.json are build artifacts emitted by +# slip into a commit and break `npm ci` on CI with ENOTDIR). +# Spelling variants are listed because a single `infographic/` pattern was +# stores the published notes. They are not a build artifact and must never be +# symlink node_modules to the main checkout; the dir-only pattern let one +# the first line of defence and cannot stop `git add -f` at all — the +# the route name, so each route gets its own tree and two can run at once. +# Tool Search live-test harness output — non-deterministic model transcripts, +# treat it as a local edit and autostash it on every run (#38529). +# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then +# walkthroughs). Throwaway artifacts, never part of the app. +# Web UI assets — synced from @nous-research/ui at build time via +# Web UI build output +# website/scripts/extract-automation-blueprints.py during prebuild. +# website/scripts/extract-skills.py during prebuild — keep them out of +# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; +# +%SystemDrive%/ +*.pem +*.ppk +*.pyc* +*.tsbuildinfo +*-snapshots/ +.act-sandbox-agent.* +.bytecode-fingerprint +.bytecode-fingerprint.tmp +.codex/ +.cursor/ +.direnv/ +.DS_Store +.env +.env.development +.env.development.local +.env.local +.env.production.local +.env.test +.env.test.local +.gemini/ +.hermes/ +.hermes-bootstrap-complete +.hermes-docker/ +.hermes-sandbox/ +.hermes-sandbox-e2e*/ +.lazy-refresh-incomplete +.mcp.json +.nix-stamps/ +.notebooklm-cli-venv/ +.notebooklm-home/ +.notebooklm-playwright/ +.op.env +.pip-cache/ +.pytest_cache/ +.pytest-cache/ +.release_notes.md +.skills_prompt_snapshot.json +.update-incomplete +.update-incomplete.lock +.uv-cache/ +.venv +.venv/ +.vscode/ +.web_ui_build.lock +.worktrees/ +.zed/ +/*.png.bak +/.hermes-runtime/ +/.install_method +/_pycache/ +/bin/ +/default.tar.gz +/log.txt +/sqlite_leak_fix.png +/venv.old/ +/venv.stale.runtime-*/ +/venv/ +__pycache__/ +__pycache__/model_tools.cpython-310.pyc +__pycache__/web_tools.cpython-310.pyc +act/ +agent-browser/ +apps/desktop/build/ +apps/desktop/demo/ +apps/desktop/dist/ +apps/desktop/release/ +apps/desktop/src/**/*.d.ts +apps/desktop/src/**/*.js +apps/desktop/src/**/*.js.map +apps/shared/src/**/*.d.ts +apps/shared/src/**/*.js +apps/shared/src/**/*.js.map +browser-use/ +cli-config.yaml +compose.hermes.local.yml +config/mcporter.json +data/ +data/* +docs/superpowers/* +environments/benchmarks/evals/ +examples/ +export* +hermes-*/* +hermes_agent.egg-info/ +hermes_cli/scripts/ +hermes_cli/tui_dist/* +hermes_cli/web_dist/ +ignored/ +images/ +infografico/ +infograficos/ +infographic/ +infographics/ +logs/ +mini-swe-agent/ +models-dev-upstream/ +native/fts5_cjk/*.so +node_modules +opencode.json +playwright-report/ +privvy* +RELEASE_v*.md +result +run_datagen_kimik2-thinking.sh +run_datagen_megascience_glm4-6.sh +run_datagen_sonnet.sh +source-data/* +run_datagen_megascience_glm4-6.sh +data/* +# No trailing slash: also matches node_modules SYMLINKS (worktrees often +# symlink node_modules to the main checkout; the dir-only pattern let one +# slip into a commit and break `npm ci` on CI with ENOTDIR). +node_modules +browser-use/ +agent-browser/ +# Private keys +*.ppk +*.pem +privvy* +images/ +__pycache__/ +hermes_agent.egg-info/ +wandb/ +testlogs +playwright-report/ +test-results/ +# Playwright visual regression baselines — cached from main in CI, not committed +*-snapshots/ + +# CLI config (may contain sensitive SSH paths) +cli-config.yaml + +# Skills Hub state (lives in ~/.hermes/skills/.hub/ at runtime, but just in case) +skills/.hub/ +ignored/ +.worktrees/ +environments/benchmarks/evals/ + +# Web UI build output +hermes_cli/web_dist/ +# Cross-process web UI build lock (flock target, always empty) +.web_ui_build.lock +apps/desktop/build/ +apps/desktop/dist/ + +# tsc-emitted artifacts (a stray `tsc -b` compiles into src/, and vite then +# resolves the stale .js OVER the .tsx — never track these) +apps/desktop/src/**/*.js +apps/desktop/src/**/*.js.map +apps/desktop/src/**/*.d.ts +# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins, +# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx +# sibling exists, so the stale-shadow hazard above cannot apply. +!apps/desktop/src/plugins/*/plugin.js +!apps/desktop/src/global.d.ts +!apps/desktop/src/vite-env.d.ts + +# Build/debug artifacts that must never be committed +/log.txt +/sqlite_leak_fix.png +/*.png.bak +*.tar.gz +*.tgz +apps/shared/src/**/*.js +apps/shared/src/**/*.js.map +apps/shared/src/**/*.d.ts +apps/desktop/release/ +# stage-and-swap Desktop rebuild output (#86443); removed after the swap, but +# a killed build must not leave the checkout dirty +apps/desktop/.staging-*/ +*.tsbuildinfo + +# Web UI assets — synced from @nous-research/ui at build time via +# `npm run sync-assets` (see web/package.json). +web/public/fonts/ +web/public/ds-assets/ + +# Release script temp files +.release_notes.md +mini-swe-agent/ + +# Nix +.direnv/ +.nix-stamps/ +result +website/static/api/skills-index.json +# skills.json + skills-meta.json are build artifacts emitted by +# website/scripts/extract-skills.py during prebuild — keep them out of +# git for the same reason as skills-index.json (large, generated, change +# every build). +website/static/api/skills.json +website/static/api/skills-meta.json +# Plugin catalog JSON is generated from plugin-catalog/ during prebuild. +website/static/api/plugins.json +website/static/api/plugin-catalog.json +website/static/api/plugins-meta.json +# automation-blueprints-index.json is a build artifact emitted by +# website/scripts/extract-automation-blueprints.py during prebuild. +website/static/api/automation-blueprints-index.json +models-dev-upstream/ + +# Local editor / agent tooling (machine-specific; keep in global config, not the repo) +.codex/ +.cursor/ +.gemini/ +.zed/ +.mcp.json +opencode.json +config/mcporter.json + +hermes_cli/tui_dist/* +hermes_cli/scripts/ +docs/superpowers/* +# Working directory for the Hermes Agent's session state (~/.hermes/ at runtime; +# also created in-repo when an agent operates in this checkout). Plans, audit +# logs, and per-session caches are never artifacts of the codebase. +.hermes/ + +# Desktop/bootstrap install marker written into the managed checkout root by the +# bootstrap installer. It is Hermes-managed runtime state, never a code change — +# ignore it so `hermes update`'s `git stash push --include-untracked` does not +# treat it as a local edit and autostash it on every run (#38529). +.hermes-bootstrap-complete + +# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent) +.hermes-sandbox/ +# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is +# the route name, so each route gets its own tree and two can run at once. +.hermes-sandbox-e2e*/ + +# Interrupted-update breadcrumb + recovery lock written next to the shared venv +# by `hermes update` / launch-time self-heal. Runtime state, never a code change +# — ignore so `git status` stays clean and update's autostash skips them. +.update-incomplete +.update-incomplete.lock + +# Checkout fingerprint the __pycache__ tree was last validated against +# (launch-time stale-bytecode sweep). Runtime state, never a code change. +.bytecode-fingerprint +.bytecode-fingerprint.tmp + +# Installer-written method stamp in the managed checkout root (scripts/install.sh). +# Runtime metadata only — never a code change. Ignore so `git status` stays clean +# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855). +/.install_method + +# Tool Search live-test harness output — non-deterministic model transcripts, +# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo. + +scripts/out/ + +# Per-release changelog drafts. These exist only transiently during a release +# cut (passed to `gh release create --notes-file`); the GitHub Release itself +# stores the published notes. They are not a build artifact and must never be +# committed to the repo root. See the hermes-release skill. + +# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded +# walkthroughs). Throwaway artifacts, never part of the app. + +# PR infographics are rendered locally and embedded in PR descriptions via the +# image-provider (fal.media) URL — they are NEVER committed to the repo. The +# PR body is the archive. See the hermes-agent-dev skill's +# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1). +# +# Spelling variants are listed because a single `infographic/` pattern was +# sidestepped by an `infograficos/` directory (#70552). .gitignore is only +# the first line of defence and cannot stop `git add -f` at all — the +# infographic-check CI job is what actually enforces this. +# Runtime marker written by hermes update when a lazy dependency refresh is +# interrupted; consumed by launch-time recovery. Never commit it (was tracked +# by accident via 3a69e34702, removed in the #72002 salvage). + +# Disposable profile created by scripts/probe_active_session_exclusivity.py +.probe-home/ +skills/.hub/ +source-data/* +temp_vision_images/ +test_durations.json +testlogs +test-results/ +tests/quick_test_dataset.jsonl +tests/sample_dataset.jsonl +tmp/ +wandb/ +web/public/ds-assets/ +web/public/fonts/ +website/static/api/automation-blueprints-index.json +website/static/api/skills.json +website/static/api/skills-index.json website/static/api/skills-meta.json # ── generated icon assets (see scripts/generate_icons.py) ────────────────── # Regenerated on demand by every consuming pipeline (website/desktop/ diff --git a/agent/AGENTS.md b/agent/AGENTS.md index e1f162fa77..edd8072bcc 100644 --- a/agent/AGENTS.md +++ b/agent/AGENTS.md @@ -59,7 +59,10 @@ Adding one: register in that table (no `if name == ...` chain); `tools/todo_tool commands (`agent/skill_commands.py`) inject as a user message; subdirectory `AGENTS.md` hints (`agent/subdirectory_hints.py`) append to the tool result (head+tail truncated past `_MAX_HINT_CHARS = 32_000`, with a warning). - **Strict role alternation.** Never two same-role messages in a row; never a synthetic user - message injected mid-loop. Cron deliveries live in their own session for this reason. + message injected mid-loop. The one exception is `/steer`, delivered as a standalone user row + after a tool result (`assistant(tool_calls) → tool → user` is legal on every provider path) — + never smeared onto the already-persisted tool row, which append-only persistence would leave + divergent from the live request. Cron deliveries live in their own session for this reason. - **Context files** (`agent/prompt_builder.py`) load from the CWD only at startup and are capped (`CONTEXT_FILE_MAX_CHARS` / dynamic cap from the context window / `context_file_max_chars`). Never load an install-tree `AGENTS.md` as project context (PR #64611); subdirectory hints reject diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index f097dee742..43e862e52e 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -19,7 +19,7 @@ from hermes_cli.timeouts import get_provider_request_timeout from agent.message_sanitization import ( _FULL_ARGS_LOG_BOUND, coalesce_tool_call_id, tool_call_id_variants, tool_result_id_variants ) -from agent.prompt_builder import format_steer_marker +from agent.prompt_builder import STEER_DISPLAY_KIND, steer_user_row from agent.tool_dispatch_helpers import _trajectory_normalize_msg, make_tool_result_message from agent.trajectory import convert_scratchpad_to_think from agent.credential_pool import ( @@ -533,6 +533,9 @@ def _merge_consecutive_users(messages: List[Dict]) -> Tuple[List[Dict], int]: # A summary carrier followed by a new user row is a deliberate durable shape after # retry/rewind; never mutate the persisted carrier (sanitizers merge copies later). and split_user_originated_turn(prev)[0] is None + # A /steer row that ended the previous run is already persisted; merging the next + # prompt into it would rewrite it in place and re-break replay parity. + and prev.get("display_kind") != STEER_DISPLAY_KIND # Only merge plain-text content; leave multimodal (list) content alone. and isinstance(prev.get("content", ""), str) and isinstance(msg.get("content", ""), str) ): @@ -1121,6 +1124,13 @@ def restore_primary_runtime(agent) -> bool: return False # primary still in rate-limit cooldown, stay on fallback rt = agent._primary_runtime primary_provider = str((rt or {}).get("provider") or "").strip().lower() + primary_model = str((rt or {}).get("model") or "").strip() + from agent.fallback_cooldown import _is_entitlement_rejected + if primary_model and _is_entitlement_rejected(agent, primary_provider, primary_model): + # The primary slug was rejected as unentitled for this account (#106475): restoring + # here would announce a recovery that was never verified and re-fail every turn. + # Stay on the fallback; the user sees the terminal entitlement error instead. + return False primary_runtime_base_url = str((rt or {}).get("base_url") or "") def _matches_primary(candidate) -> bool: @@ -3147,9 +3157,25 @@ def _requeue_pending_steer(agent, steer_text: str) -> None: def apply_pending_steer_to_tool_results(agent, messages: list, num_tool_msgs: int) -> None: - """Append pending /steer text to the last ``role:"tool"`` message of this batch (bounded by - ``num_tool_msgs``), marked as user-origin. Modifies existing content only, so role - alternation is preserved.""" + """Persist any pending /steer text as a standalone user message. + + Called at the end of a tool-call batch, before the next API call. + + The steer is emitted as a NEW ``role:"user"`` message appended after the + last tool result (marker text included), so: + + - the model still sees the self-describing out-of-band marker (same text, + same provenance semantics); + - message-role alternation stays legal — ``assistant(tool_calls) → tool → + user`` is the documented "user jumped in mid-run" pattern that + ``repair_message_sequence`` deliberately keeps; + - the appended dict carries no ``_DB_PERSISTED_MARKER`` yet, so the next + ``_flush_messages_to_session_db`` writes it to the session store — the + steer text finally becomes part of the durable transcript instead of + being smeared onto an already-persisted tool row that append-only + persistence never rewrites (replayed histories then diverge from the + live request bytes and break the provider prompt cache). + """ if num_tool_msgs <= 0 or not messages: return steer_text = agent._drain_pending_steer() @@ -3159,22 +3185,14 @@ def apply_pending_steer_to_tool_results(agent, messages: list, num_tool_msgs: in tail = range(len(messages) - 1, max(len(messages) - num_tool_msgs - 1, -1), -1) target = next((messages[j] for j in tail if isinstance(messages[j], dict) and messages[j].get("role") == "tool"), None) if target is None: - # No tool result in this batch (e.g. all skipped by interrupt). + # No tool result in this batch (e.g. all skipped by interrupt); + # requeue so the fallback path delivers it as a normal next-turn + # user message (which persists like any other user turn). _requeue_pending_steer(agent, steer_text) return - marker = format_steer_marker(steer_text) - existing_content = target.get("content", "") - if isinstance(existing_content, str): - target["content"] = existing_content + marker - else: - # Anthropic multimodal content blocks: preserve them and append a text block. - try: - target["content"] = [*(existing_content or []), {"type": "text", "text": marker.lstrip()}] - except Exception: - # Fall back to string replacement if content shape is unexpected. - target["content"] = f"{existing_content}{marker}" + messages.append(steer_user_row(steer_text)) _ra().logger.info( - "Delivered /steer to agent after tool batch (%d chars): %s", len(steer_text), + "Delivered /steer to agent after tool batch (%d chars) as new user message: %s", len(steer_text), steer_text[:120] + ("..." if len(steer_text) > 120 else ""), ) diff --git a/agent/background_review.py b/agent/background_review.py index 9ed654f098..46b2394694 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -613,7 +613,9 @@ def _prior_tool_keys(prior_snapshot: List[Dict]) -> Tuple[set, set]: def _action_lines(data: Dict, detail: Dict, verbose: bool) -> List[str]: """Summary line(s) for one successful notify-tool result (``[]`` when nothing to report).""" if data.get("staged"): - return [] + # The fork's own review summary is never published back, so an unattended-review + # consolidation proposal must surface here or it is silently lost (#105921). + return [data["message"]] if data.get("proposal_staged") and data.get("message") else [] message = data.get("message", "") target = data.get("target", "") or detail.get("target", "") is_skill = detail.get("tool") == "skill_manage" @@ -842,6 +844,23 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio return kwargs +# Above any live registry generation: _publish_tool_snapshot refuses an older-generation rebuild, +# so the compaction-boundary refresh_agent_mcp_tools(content_aware=True) cannot rebuild the fork's +# tools[] from the live registry and drop the inherited provider/plugin tools (#103579). +_FROZEN_TOOL_SNAPSHOT_GENERATION = 2_147_483_647 + + +def _inherit_parent_tool_surface(review_agent: Any, agent: Any) -> None: + """Same-model fork: advertise the parent's exact tools[] (its last outbound payload — an + empty list included) so the request prefix matches byte-for-byte, then freeze the snapshot + generation. Dispatch stays behind the review whitelist; advertising is not permission.""" + # getattr: /btw and review callers build bare object.__new__ agents in tests without ``tools``. + review_agent.tools = copy.deepcopy(getattr(agent, "tools", None) or []) + review_agent.valid_tool_names = {tool["function"]["name"] for tool in review_agent.tools} + review_agent._tool_snapshot_generation = _FROZEN_TOOL_SNAPSHOT_GENERATION + + + def build_cache_parity_fork( agent: Any, task_cfg: Optional[Dict[str, Any]] = None, *, max_iterations: int, write_origin: str = "background_review", @@ -887,6 +906,7 @@ def build_cache_parity_fork( if not _routed: review_agent._cached_system_prompt = agent._cached_system_prompt review_agent.session_start = agent.session_start + _inherit_parent_tool_surface(review_agent, agent) _detach_fork_compression(review_agent) # Compaction bounds a single request; this bounds the WHOLE review (checked in # conversation_loop via _review_input_budget_exhausted). @@ -934,14 +954,17 @@ def _track_review_fork(agent: Any, review_agent: Any, *, register: bool) -> None agent._active_children.remove(review_agent) -def _review_tool_whitelist(review_agent: Any, task_cfg: Optional[Dict[str, Any]]) -> Tuple[set, set]: +def _review_tool_whitelist( + review_agent: Any, task_cfg: Optional[Dict[str, Any]], review_memory: bool = False, +) -> Tuple[set, set]: """``(whitelist, configured_extra_tools)`` for the review fork — DISPATCH-side only, so the advertised ``tools[]`` stays byte-identical to the parent's (prompt-cache parity).""" from model_tools import get_tool_definitions - # Gate the built-in memory tool on the profile's memory flags so a memory-disabled profile - # is never contaminated by the review LLM. + # Gate the built-in memory tool on BOTH the profile's memory flags and the trigger that fired + # (#105921): a skill-nudge review never gets the memory tool, so an unattended fork cannot + # act on the memory tool's "consolidate now" hint and delete entries no one reviewed. memory_on = review_agent._memory_enabled or review_agent._user_profile_enabled - review_toolsets = ["memory", "skills"] if memory_on else ["skills"] + review_toolsets = ["memory", "skills"] if memory_on and review_memory else ["skills"] whitelist = {t["function"]["name"] for t in get_tool_definitions(enabled_toolsets=review_toolsets, quiet_mode=True)} # Read-only file tools: denying read_file/search_files caused a per-review denial storm that # starved the loop (read_file also registers the read with the read-before-write guard). @@ -992,25 +1015,34 @@ def _release_fork_clients(review_agent: Any) -> None: def _run_review_fork( agent: Any, messages_snapshot: List[Dict], prompt: str, task_cfg: Optional[Dict[str, Any]], - review_run: Optional[_BackgroundReviewRun], st: _ReviewForkState, + review_run: Optional[_BackgroundReviewRun], st: _ReviewForkState, review_memory: bool = False, + explicit: bool = False, ) -> None: """Fork phase (inside thread-scoped silence): build the fork, run the prompt under the tool whitelist, snapshot its messages/usage, release its clients. Partial progress lands on ``st`` - so the caller's error path still sees usage and the fork to clean up.""" - st.review_agent, _rt, _routed = build_cache_parity_fork(agent, task_cfg, max_iterations=_REVIEW_MAX_ITERATIONS) + so the caller's error path still sees usage and the fork to clean up. ``explicit`` (/refine) + keeps the ``background_review`` origin (curator/skill guards still apply) but marks the fork + attended, so the unattended-only memory delete gate leaves the full operation set available.""" + st.review_agent, _rt, _routed = build_cache_parity_fork( + agent, task_cfg, max_iterations=_REVIEW_MAX_ITERATIONS) + st.review_agent._review_attended = explicit _track_review_fork(agent, st.review_agent, register=True) from hermes_cli.plugins import set_thread_tool_whitelist, clear_thread_tool_whitelist - review_whitelist, configured_extra_tools = _review_tool_whitelist(st.review_agent, task_cfg) + review_whitelist, configured_extra_tools = _review_tool_whitelist(st.review_agent, task_cfg, review_memory) extra_list = ", ".join(sorted(configured_extra_tools)) deny_extra = f" Configured extra tools also allowed: {extra_list}." if configured_extra_tools else "" prompt_extra = f" Exception — these configured tools are also allowed: {extra_list}." if configured_extra_tools else "" + # Keep the deny/prompt wording in sync with the whitelist: a memory-less review must not + # tell the model that memory is available, or it will burn iterations on denied calls. + memory_phrase_deny = " and memory for notes (add only)" if "memory" in review_whitelist else "" + memory_phrase_prompt = "memory and skill " if "memory" in review_whitelist else "skill " set_thread_tool_whitelist( review_whitelist, deny_msg_fmt=( "Background review denied non-whitelisted tool: " "{tool_name}. Allowed here: skill_view/skills_list/read_file/search_files to read, " - "skill_manage(action='patch'|...) to change skills, and " - "memory for notes." + deny_extra + " Do not retry {tool_name}." + "skill_manage(action='patch'|...) to change skills" + + memory_phrase_deny + "." + deny_extra + " Do not retry {tool_name}." ), ) with suppress(Exception): @@ -1022,7 +1054,7 @@ def _run_review_fork( # Routed -> digest (cache cold anyway); same model -> full snapshot (warm cache reads). st.review_agent.run_conversation( user_message=( - prompt + "\n\nYou can only call memory and skill " + prompt + "\n\nYou can only call " + memory_phrase_prompt + "management tools. Other tools will be denied " "at runtime — do not attempt them." + prompt_extra ), @@ -1056,6 +1088,7 @@ def _publish_review_summary(agent: Any, actions: List[str]) -> None: def _run_review_in_thread( agent: Any, messages_snapshot: List[Dict], prompt: str, task_cfg: Optional[Dict[str, Any]] = None, review_run: Optional[_BackgroundReviewRun] = None, + review_memory: bool = False, explicit: bool = False, ) -> None: """Daemon-thread worker: build the fork, run the prompt, surface the action summary via ``agent._safe_print`` / ``background_review_callback``. ``review_run`` (from @@ -1090,7 +1123,7 @@ def _run_review_in_thread( # their console output (#55769 / #55925). ``thread_scoped_silence`` routes only this thread's writes # to devnull and leaves all other threads on the real streams. with thread_scoped_silence(): - _run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st) + _run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st, review_memory, explicit) # A buggy/legacy tool response shape must NOT take down the whole review (the outer # except would discard every action the fork DID complete), so coerce to an empty list. try: @@ -1147,11 +1180,14 @@ def spawn_background_review_thread( agent: Any, messages_snapshot: List[Dict], review_memory: bool = False, review_skills: bool = False, focus: Optional[str] = None, task_cfg: Optional[Dict[str, Any]] = None, review_run: Optional[_BackgroundReviewRun] = None, + explicit: bool = False, ): """Return ``(target, prompt)``; the caller builds the ``threading.Thread`` so test patches of ``run_agent.threading.Thread`` keep working. ``focus`` (``/refine [instructions]``) is appended to the chosen prompt; automatic reviews pass ``None``. ``task_cfg`` is the pre-loaded - ``auxiliary.background_review`` block; when omitted it is read once here.""" + ``auxiliary.background_review`` block; when omitted it is read once here. ``explicit`` + (/refine) propagates to the fork's write origin so user-requested reviews keep the full + memory operation set.""" if task_cfg is None: task_cfg = _background_review_task_config() # Per-agent overrides (agent._MEMORY_REVIEW_PROMPT etc.) keep working. @@ -1164,7 +1200,9 @@ def spawn_background_review_thread( ) def _target() -> None: # resolves _run_review_in_thread at call time (tests patch it) - _run_review_in_thread(agent, messages_snapshot, prompt, task_cfg=task_cfg, review_run=review_run) + _run_review_in_thread( + agent, messages_snapshot, prompt, task_cfg=task_cfg, review_run=review_run, + review_memory=review_memory, explicit=explicit) return _target, prompt diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 1f72060750..3f3e83004a 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1731,6 +1731,10 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider: return True if not fb_provider or not fb_model: return True + from agent.fallback_cooldown import _is_entitlement_rejected + if _is_entitlement_rejected(agent, fb_provider, fb_model): + logger.info("Fallback skip: %s/%s was rejected as unentitled for this account", fb_provider, fb_model) + return True local_skip_reason = _fallback_entry_unavailable_without_network(agent, fb) if local_skip_reason: unavailable.add(fb_key) @@ -2174,11 +2178,17 @@ def cleanup_task_resources(agent, task_id: str) -> None: def _build_partial_stream_stub(role, full_content, full_reasoning, model_name, usage_obj, *, - dropped_tool_names=None): + dropped_tool_names=None, overflow_terminal=False): """Stub for an SSE stream that ended without ``finish_reason`` after delivering content. Tagged ``PARTIAL_STREAM_STUB_ID`` + ``FINISH_REASON_LENGTH`` so the loop enters its continuation/retry path instead of accepting - truncated output as a complete turn (#32086).""" + truncated output as a complete turn (#32086). + + ``overflow_terminal`` (``full_content=None``): the stream died on a + context-overflow error. Seeding the recovered text as a continuation stub + would grow every later request into the same overflow (#106260); the loop + treats the marker as terminal and ends the turn via the recovery contract. + """ return SimpleNamespace( id=PARTIAL_STREAM_STUB_ID, model=model_name, @@ -2190,6 +2200,7 @@ def _build_partial_stream_stub(role, full_content, full_reasoning, model_name, u )], usage=usage_obj, _dropped_tool_names=dropped_tool_names or None, + _overflow_terminal=overflow_terminal, ) @@ -3260,22 +3271,37 @@ class _StreamingCall(StreamingWaitMonitor): logger.warning( "Partial stream dropped tool call(s) %s after %s chars of text; surfaced warning to user: %s", _partial_names, len(_partial_text or ""), error) - else: - logger.warning( - "Partial stream delivered before error; returning length-truncated stub with %s chars of " - "recovered content so the loop can continue from where the stream died: %s", - len(_partial_text or ""), error) - # Classify content filtering (MiniMax 1027, Azure content_filter, Anthropic refusal) - # before the error is swallowed into the stub: the loop reads the tag and falls back. - _stub = _build_partial_stream_stub("assistant", _partial_text, None, - getattr(self.agent, "model", "unknown"), None, dropped_tool_names=_partial_names) + # Classify the error before it is swallowed into the stub: the loop reads the + # content-filter tag and falls back; a context overflow must not be continued at all. + _cls = None with contextlib.suppress(Exception): from agent.error_classifier import classify_api_error _cls = classify_api_error( error, provider=str(getattr(self.agent, "provider", "") or ""), model=str(getattr(self.agent, "model", "") or "")) - if _cls.reason == FailoverReason.content_policy_blocked: - _stub._content_filter_terminated = True _reset_stale_streak(self.agent) # deltas fired => provider responsive: clear the breaker + # #106260: continuing after a context-overflow error re-sends a larger request into the + # same overflow. Return an EMPTY stub marked terminal so the loop ends the turn instead. + # Scope is context_overflow ONLY: payload_too_large (413) has its own byte-scored recovery + # owner (turn_overflow._recover_payload_too_large, #88960/#47339) that must not be bypassed. + if _cls is not None and _cls.reason == FailoverReason.context_overflow: + logger.warning( + "Partial stream ended on a context-overflow error after %s chars; " + "NOT seeding a continuation stub (transcript is already over budget): %s", + len(_partial_text or ""), error, + ) + return _build_partial_stream_stub( + "assistant", None, None, getattr(self.agent, "model", "unknown"), None, + dropped_tool_names=_partial_names, overflow_terminal=True, + ) + if not _partial_names: + logger.warning( + "Partial stream delivered before error; returning length-truncated stub with %s chars of " + "recovered content so the loop can continue from where the stream died: %s", + len(_partial_text or ""), error) + _stub = _build_partial_stream_stub("assistant", _partial_text, None, + getattr(self.agent, "model", "unknown"), None, dropped_tool_names=_partial_names) + if _cls is not None and _cls.reason == FailoverReason.content_policy_blocked: + _stub._content_filter_terminated = True return _stub def run(self): diff --git a/agent/context_compressor.py b/agent/context_compressor.py index e54c69eb59..15d3543e07 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -25,6 +25,7 @@ from agent.context_engine import ContextEngine, sanitize_memory_context from agent.context_compressor_summary import SummaryDispatchMixin from agent.error_classifier import FailoverReason, classify_api_error from agent.micro_compaction import MicroCompactionMixin +from agent.prompt_builder import STEER_DISPLAY_KIND from agent.model_metadata import ( MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough, strip_opaque_replay_items, @@ -1456,7 +1457,25 @@ def _sum_clarify(name, args, content, content_len, line_count): # min_prune_chars guard and skips the >=200-char dedup. max_summary_chars = _PRUNE_MIN_CHARS - 1 truncation_marker = "...[truncated]" - response = _json_dict(content).get("user_response") + parsed = _json_dict(content) + response = parsed.get("user_response") + # Batch clarify (``questions=[...]``) nests each answer inside ``responses[].user_response`` + # rather than the top level; without this every batch answer was lost and the summarizer only + # saw "asked user a question" (#106077). + if response is None: + batch_responses = parsed.get("responses") + if isinstance(batch_responses, list) and batch_responses: + collected = [] + for entry in batch_responses: + if not isinstance(entry, dict): + continue + single = entry.get("user_response") + # multi_select emits a list of strings; flatten it so the summary keeps every choice. + if isinstance(single, str) and single: + collected.append(single) + elif isinstance(single, list) and all(isinstance(s, str) and s for s in single): + collected.extend(single) + response = collected if collected else None is_answer_shaped = (isinstance(response, str) and bool(response)) or ( isinstance(response, list) and bool(response) and all(isinstance(s, str) and s for s in response) ) @@ -3644,7 +3663,9 @@ Write only the summary body. Do not include any preamble or prefix.""" return False # display_kind rows (internal notifications, hidden scaffolding) are not human input # and must not anchor the tail or seed auto-focus. Mirrors is_user_originated_turn. - if message.get("display_kind") or cls._is_context_summary_message(message): + # A /steer row is typed for the renderer and the alternation repair, but it IS human input. + display_kind = message.get("display_kind") + if (display_kind and display_kind != STEER_DISPLAY_KIND) or cls._is_context_summary_message(message): return False return not cls._is_blank_user_turn(message) @@ -4777,10 +4798,10 @@ def split_user_originated_turn(message: Any) -> tuple[Optional[Dict[str, Any]], candidate = None if display_kind and display_kind != "hidden" else ContextCompressor._strip_context_summary_handoff_message(message) if candidate is None: return handoff, None - elif message.get("display_kind"): + elif message.get("display_kind") and message.get("display_kind") != STEER_DISPLAY_KIND: return None, None else: - candidate = message.copy() + candidate = message.copy() # includes a typed /steer row: full user authority for key in ( COMPRESSED_SUMMARY_METADATA_KEY, COMPRESSED_SUMMARY_HAS_USER_TURN_KEY, MICRO_COMPACT_MARKER_KEY, diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 12359913ed..87d36ba406 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -1934,8 +1934,8 @@ def _steer_markers() -> Tuple[str, str]: def _message_contains_busy_steer(message: Any) -> bool: """Return whether *message* carries a busy-steer marker. - Steer follow-ups live as markers inside ``role=tool`` results, so they carry user intent that - ``_is_real_user_message`` alone would miss.""" + Steer follow-ups are now their own ``role=user`` rows (caught by ``_is_real_user_message``); in + transcripts persisted before that they ride inside ``role=tool`` results, so those still count.""" text = _message_text(message) if not text: return False diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index ef1e1399bf..70e9a12762 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -29,6 +29,9 @@ from agent.prompt_caching import ( strip_anthropic_tool_cache_control, ) from agent.runtime_cwd import resolve_agent_cwd +from agent.surface_switch import ( + identity_line_value, note_inert_pinned_tools, split_runtime_boundary, stage_surface_switch_note, +) from agent.turn_context import PreflightCompressionTimedOut, build_turn_context from agent.turn_retry_state import TurnRetryState # Phase helpers of the turn loop, bound at import so a source-tree swap cannot load a @@ -124,8 +127,13 @@ def _maybe_inject_run_budget_wrapup(agent: Any, messages: List[Dict[str, Any]]) (time.time() - started) < 0.8 * float(budget) ): return False + from agent.context_compressor import _DB_PERSISTED_MARKER for msg in reversed(messages): if isinstance(msg, dict) and msg.get("role") == "tool": + # Only the current tool-result tail is mutable; an older turn may already be + # cached (same contract as _maybe_inject_iteration_budget_warning). + if msg.get(_DB_PERSISTED_MARKER): + return False existing = msg.get("content", "") if isinstance(existing, str): msg["content"] = existing + f"\n\n{RUN_BUDGET_WRAPUP_NOTICE}" @@ -677,6 +685,7 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) except Exception: pass agent._cached_system_prompt = agent._build_system_prompt(system_message) + stage_surface_switch_note(agent, agent._cached_system_prompt, conversation_history) # Persist so the NEXT turn restores the new bytes verbatim (cache break is # once per capability change). on_session_start not re-fired: continuation. _persist_system_prompt( @@ -688,13 +697,27 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # Continuing session — reuse the exact system prompt from the # previous turn so the Anthropic cache prefix matches. agent._cached_system_prompt = stored_prompt + # The reused bytes may describe the surface this conversation STARTED on; correct that + # at the tail of the request instead of rebuilding the prompt in front of it (#104414). + announced_switch = stage_surface_switch_note(agent, stored_prompt, conversation_history) # Same contract for tools[]: pin the array to the order this session already # sent (tools freeze) instead of re-probing every check_fn on a fresh AIAgent. + # The pin holds ON the announcing turn too. tools[] is serialized AHEAD of the system + # prompt this branch just preserved, so dropping the previous surface's toolset would + # change the request at token 0 and re-prefill everything behind it — the exact cost + # #104414 is about, paid on the exact turn we are here to make cheap. The merge still + # ADDS what the new surface brought (a tui -> desktop switch pays a break no freeze can + # avoid), and what it carries FORWARD is named in the note instead, so a tool that can + # only answer ``tool_error("desktop only")`` here does not read as a live capability. try: saved_tools = session_row.get("tool_names") if session_row else None if saved_tools: - from tools.mcp_tool_agent import restore_agent_tool_prefix + from tools.mcp_tool_agent import agent_tool_names, restore_agent_tool_prefix + # Captured BEFORE the pin merges the previous surface's tools back in. + built_for_this_surface = agent_tool_names(agent) if announced_switch else [] restore_agent_tool_prefix(agent, json.loads(saved_tools)) + if announced_switch: + note_inert_pinned_tools(agent, built_for_this_surface) except Exception: logger.debug("tool prefix restore skipped", exc_info=True) # Prompt-section callbacks are new-session-only; recover their frozen bytes @@ -727,6 +750,12 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) # First turn of a new session (or recovering from a broken stored prompt). agent._cached_system_prompt = agent._build_system_prompt(system_message) + # The rebuilt prompt describes the CURRENT surface, but a surface note left in the + # transcript by an earlier switch does not — retire it here too, or a rebuild for an + # unrelated reason (a model switch) would leave the newest interface statement in the + # request naming a surface the conversation has left (#104414). + stage_surface_switch_note(agent, agent._cached_system_prompt, conversation_history) + # Plugin hook: on_session_start — fired once for a brand-new session, not on continuation. try: from hermes_cli.lifecycle import invoke_hook as _invoke_hook @@ -756,23 +785,12 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: """Return False when the persisted runtime-identity lines are stale.""" - identity, runtime_marker, runtime = prompt.rpartition(f"\n\n{RUNTIME_ENVIRONMENT_HEADING}\n\n") - # Legacy prose may quote the heading, but only the new renderer ends in this boundary. - runtime_marker = runtime_marker if prompt.endswith(RUNTIME_ENVIRONMENT_END) else "" - # The final runtime block can contain embedder prose, not authoritative identity. - lines = (identity if runtime_marker else prompt).splitlines() - - def line_value(label: str) -> str: - """Last matching line wins — safe ONLY for volatile-tier fields at the END of the - prompt (embedded project context could shadow earlier fields; see ``host_info_value``).""" - prefix = f"{label}:" - matches = [line[len(prefix):].strip() for line in lines if line.startswith(prefix)] - return matches[-1] if matches else "" + _identity, runtime_marker, runtime = split_runtime_boundary(prompt) def host_info_value(label: str) -> str: """New prompts delimit runtime hints; legacy prompts put them before context.""" prefix = f"{label}:" - host_lines = runtime.split("\n\n", 1)[0].splitlines() if runtime_marker else lines + host_lines = (runtime.split("\n\n", 1)[0] if runtime_marker else prompt).splitlines() for idx, line in enumerate(host_lines): if line.startswith("User home directory:"): for candidate in host_lines[idx + 1: idx + 4]: @@ -780,10 +798,11 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: return candidate[len(prefix):].strip() return "" - # Model/provider identity, then cwd drift, then runtime-surface drift (reusing a - # desktop-built prompt on a terminal session would inject the wrong runtime hints). + # Model/provider identity, then cwd drift. A cwd change is a real content change (context + # files, the workspace snapshot and the coding posture are all resolved from it), so it + # still rebuilds; the runtime surface does not (agent/surface_switch.py). for label, attr in (("Model", "model"), ("Provider", "provider")): - stored = line_value(label) + stored = identity_line_value(prompt, label) current = str(getattr(agent, attr, "") or "").strip() if stored and current and stored != current: return False @@ -792,9 +811,10 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: stored_cwd = host_info_value("Current working directory") if stored_cwd and stored_cwd != str(resolve_agent_cwd()): return False - stored_platform = line_value("Platform") - current_platform = str(getattr(agent, "platform", "") or "").strip() - return not (stored_platform and current_platform and stored_platform != current_platform) + # Platform is deliberately NOT an identity field: a surface switch does not invalidate the + # stored bytes, it only makes their interface section out of date, and that is corrected by + # agent.surface_switch.stage_surface_switch_note without touching the cached prefix (#104414). + return True # Named so _is_synthetic_compression_user_turn can recognize a crash-persisted nudge by @@ -1284,6 +1304,11 @@ class _LoopState: failed: bool = False codex_ack_continuations: int = 0 length_continue_retries: int = 0 + # Per-turn backstop for the refunding restarts (redirect / rebuilt-for-fallback). + # Unlike ``retry_count`` (rebound to 0 each iteration) this accumulates for the whole + # turn so a runaway interrupt/redirect that keeps re-arming a restart flag cannot + # refund the iteration budget forever and hold the turn lease indefinitely. + restart_count: int = 0 _outer_error_count: int = 0 # outer-loop exceptions this turn (#92450), see _MAX_OUTER_LOOP_ERRORS truncated_tool_call_retries: int = 0 truncated_response_parts: List[str] = field(default_factory=list) @@ -1392,7 +1417,7 @@ def _run_api_retry_loop(agent, s: _LoopState) -> Optional[Dict[str, Any]]: return None -def run_conversation( +def _run_conversation_turn( agent, user_message: Any, system_message: str = None, @@ -1540,6 +1565,46 @@ def run_conversation( return result +def run_conversation( + agent, + user_message: Any, + system_message: str = None, + conversation_history: List[Dict[str, Any]] = None, + task_id: str = None, + stream_callback: Optional[callable] = None, + persist_user_message: Optional[Any] = None, + persist_user_timestamp: Optional[float] = None, + persist_user_display_kind: Optional[str] = None, + persist_user_display_metadata: Optional[Dict[str, Any]] = None, + persist_user_platform_id: Optional[str] = None, + moa_config: Optional[dict[str, Any]] = None, +) -> Dict[str, Any]: + """Run one turn (see ``_run_conversation_turn``) and export the current-turn boundary. + + Every envelope that leaves the loop — success, partial/error, interrupt, retry-exhausted, + tool-limit, preflight timeout, codex runtime — passes through here, so the + ``{turn_id, current_turn_user_idx}`` pair is stamped beside the exact ``messages`` it + addresses, after every history rewrite including post-turn micro-compaction. + """ + from agent.turn_context import export_current_turn_boundary + + result = _run_conversation_turn( + agent, + user_message, + system_message=system_message, + conversation_history=conversation_history, + task_id=task_id, + stream_callback=stream_callback, + persist_user_message=persist_user_message, + persist_user_timestamp=persist_user_timestamp, + persist_user_display_kind=persist_user_display_kind, + persist_user_display_metadata=persist_user_display_metadata, + persist_user_platform_id=persist_user_platform_id, + moa_config=moa_config, + ) + return export_current_turn_boundary(agent, result, user_message) + + __all__ = ["run_conversation"] diff --git a/agent/error_classifier.py b/agent/error_classifier.py index ad4e9ae546..e4d1709498 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -125,6 +125,7 @@ _RATE_LIMIT_PATTERNS = ( _OVERLOADED_PATTERNS = ( "overloaded", "temporarily overloaded", "service is temporarily overloaded", "service may be temporarily overloaded", "server is overloaded", "server overloaded", + "server overload", "server_overload", "service overloaded", "service is overloaded", "upstream overloaded", "currently overloaded", "at capacity", "over capacity", ) @@ -150,9 +151,14 @@ _PAYLOAD_TOO_LARGE_PATTERNS = ( # Per-image size/dimension 400s (Anthropic 5 MB / 8000 px; MiniMax "media # exceeds size limit" #76039) — a specific 400 before the request hits 413. A # non-image media hit is harmless: the shrink pass finds no image parts. +# "patches after processing": OpenAI Codex Responses rejects an image whose +# tile-patch budget (ceil(w/32)×ceil(h/32)) exceeds its 30000-patch ceiling +# with wording that names no image-size vocabulary — without this pattern it +# fell to format_error (non-retryable), bypassing the shrink recovery (#106337). _IMAGE_TOO_LARGE_PATTERNS = ( "image exceeds", "image too large", "image_too_large", "image size exceeds", "image dimensions exceed", "dimensions exceed max allowed size", "max allowed size: 8000", "media exceeds", "media too large", + "patches after processing", ) # Undecodable image bytes → strip-and-retry, never shrink. xAI wordings @@ -169,9 +175,34 @@ _IMAGE_CORRUPT_PATTERNS = ( _MULTIMODAL_TOOL_CONTENT_PATTERNS = ( "text is not set", "tool message content must be a string", "tool content must be a string", "tool message must be a string", "expected string, got list", "expected string, got array", - "tool_call.content must be string", + # Console Go / pydantic-v2 relays behind opencode-go (422, param ``messages.N.tool.content.str``, #104731). + "tool_call.content must be string", "tool.content.str", "input should be a valid string", ) +# Local-inference memory/resource-ceiling rejections (oMLX/MLX memory guard, +# llama.cpp/vLLM OOM, Metal/CUDA allocation ceilings). The server aborts on a +# prefill memory PEAK, not a window limit, yet its remediation tail says +# "reduce context length" — so without this list the request routes into +# compression, which cannot lower a prefill peak: it burns the compression +# budget, re-hits the wedged server each attempt and ends in a session reset. +# Every token names memory/allocation in BYTES, never a token count, so the +# list is disjoint from _CONTEXT_OVERFLOW_PATTERNS. Must be checked BEFORE +# both overflow AND the usage-limit disambiguation ("memory limit exceeded" +# contains "limit exceeded", which would otherwise read as billing). oMLX +# reworded the accounting sentence in 0.5.7 ("predicted peak would require / +# exceed"); the 0.5.6 wording is still in the field, so both stay. (#52261) +_MEMORY_CEILING_PATTERNS = ( + "memory guard", "memory limit exceeded", "memory_guard_tier", "dynamic ceiling", + "memory ceiling", "available memory", "out of memory", "insufficient memory", + "prefill would require", "predicted peak would", "prefill safety cap", "metal_cap", +) + +# Structured codes identifying the same rejection at the source, before an +# OpenAI-compatible proxy flattens the body and drops the wording. +_MEMORY_CEILING_ERROR_CODES = frozenset({ + "prefill_memory_exceeded", "prefill_memory_aborted", "omlx_prefill_memory_exceeded", +}) + # Bare "max_tokens" is load-bearing: the output-cap-retry path keys off it; # empty-response advisories mentioning it are intercepted earlier. Groups: # generic; vLLM; Ollama; llama.cpp; Chinese; Z.AI (1210); Bedrock; Together. @@ -380,7 +411,8 @@ _IMAGE_TOOL_RULES = ( # Overflow signals arriving as 5xx (llama.cpp reports overflow as 500; busy / # model-load OOM as 503). Empty-response advisories must not enter compression. _OVERFLOW_AS_5XX_RULES = ( - (_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW), + (_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_MEMORY_CEILING_PATTERNS, _V_OVERLOADED), + (_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW), ) # 404: Nous API surfaces credit depletion as a paid model vanishing from the @@ -398,7 +430,8 @@ _400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + ( ) # Status-less message path, head (before usage-limit disambiguation). -_MESSAGE_HEAD_RULES = ((_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE),) + _IMAGE_TOOL_RULES +_MESSAGE_HEAD_RULES = ((_MEMORY_CEILING_PATTERNS, _V_OVERLOADED), + (_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE)) + _IMAGE_TOOL_RULES # Status-less tail. Overload before rate_limit/billing so "overloaded" backs off # instead of rotating; policy block before model_not_found; timeout/connection @@ -419,6 +452,7 @@ _ERROR_CODE_VERDICTS: Dict[str, Verdict] = { **dict.fromkeys(_BILLING_ERROR_CODES, _V_BILLING), **dict.fromkeys(("model_not_found", "model_not_available", "invalid_model"), _V_MODEL_NOT_FOUND), **dict.fromkeys(("context_length_exceeded", "max_tokens_exceeded"), _V_CONTEXT_OVERFLOW), + **dict.fromkeys(_MEMORY_CEILING_ERROR_CODES, _V_OVERLOADED), "invalid_encrypted_content": _V_INVALID_ENCRYPTED, } @@ -713,6 +747,10 @@ def _classify_400(c: _Ctx) -> Verdict: "error=%.200s", c.num_messages, c.approx_tokens, msg, ) return _V_FORMAT_ERROR + # Memory ceiling by code: _by_status runs before _by_error_code, so a + # 400 whose wording a proxy stripped would fall through to format_error. + if code in _MEMORY_CEILING_ERROR_CODES: + return _V_OVERLOADED verdict = _first_match(msg, _400_TAIL_RULES) if verdict is not None: return verdict @@ -732,6 +770,7 @@ def _classify_400(c: _Ctx) -> Verdict: _STATUS_HANDLERS: Dict[int, Callable[[_Ctx], Verdict]] = { 400: _classify_400, 401: lambda c: _V_AUTH_ROTATE, 402: lambda c: _classify_402(c.msg, dict), 403: _status_403, 404: _status_404, 408: lambda c: _V_TIMEOUT, 413: lambda c: _V_PAYLOAD_TOO_LARGE, + 422: lambda c: _first_match(c.msg, _IMAGE_TOOL_RULES) or _V_FORMAT_ERROR, 429: _status_429, 500: _status_5xx, 502: _status_5xx, 503: lambda c: _first_match(c.msg, _OVERFLOW_AS_5XX_RULES) or _V_OVERLOADED, 529: lambda c: _first_match(c.msg, _OVERFLOW_AS_5XX_RULES) or _V_OVERLOADED, diff --git a/agent/fallback_cooldown.py b/agent/fallback_cooldown.py index 5fbe74eaae..5ed9261f7b 100644 --- a/agent/fallback_cooldown.py +++ b/agent/fallback_cooldown.py @@ -1,9 +1,12 @@ -"""Primary rate-limit cooldown arming, shared by fallback switches.""" +"""Primary rate-limit cooldown arming and per-session model rejection markers, shared by the +fallback walk (chat_completion_helpers) and restore_primary_runtime (agent_runtime_helpers).""" import logging import time from agent.error_classifier import FailoverReason +logger = logging.getLogger(__name__) + _RATE_LIMIT_FAILOVER_REASONS = frozenset({FailoverReason.rate_limit, FailoverReason.billing, FailoverReason.upstream_rate_limit}) @@ -26,3 +29,56 @@ def _arm_rate_limit_cooldown(agent, reason: "FailoverReason | None") -> int | No return backoff_seconds +# Codex ChatGPT-account entitlement 400 — the account can never use the named slug, so with +# nothing to rotate it is a config error, not a transient failure (#106475). +_CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER = "model is not supported when using codex with a chatgpt account" + + +def _mark_entitlement_rejected_model(agent, api_error) -> bool: + """Record a Codex ChatGPT-account 400 that rejects the current model for this account. + + With a single credential there is no pool to rotate (#71970 covers that case), so the + (provider, model) pair is treated as dead for the session: the fallback walk skips it and + restore_primary_runtime stops switching back — otherwise every turn re-fails on the primary, + announces an unverified "Primary model restored", and oscillates forever (#106475). + """ + if getattr(api_error, "status_code", None) != 400: + return False + pool = getattr(agent, "_credential_pool", None) + if pool is not None and len(pool.entries()) > 1: + return False # another account in the pool may be entitled; leave rotation to it + haystack = str(getattr(api_error, "message", "") or api_error).lower() + if _CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER not in haystack: + return False + provider = str(getattr(agent, "provider", "") or "").strip().lower() + model = str(getattr(agent, "model", "") or "").strip() + if not provider or not model: + return False + rejected = getattr(agent, "_entitlement_rejected_models", None) + if rejected is None: + rejected = agent._entitlement_rejected_models = set() + if (provider, model) in rejected: + return True + rejected.add((provider, model)) + logger.warning( + "Model entitlement rejection: this account is not entitled to %s via %s; " + "treating it as unavailable for this session", + model, provider, + ) + agent._buffer_status( + f"🚫 This account is not entitled to {model} via {provider}; it will be skipped " + "until restart. Switch to an entitled model via /model or `hermes model`." + ) + return True + + +def _is_entitlement_rejected(agent, provider: str, model: str) -> bool: + """True when (provider, model) — as configured or normalized — was rejected as unentitled + for this account (see _mark_entitlement_rejected_model).""" + rejected = getattr(agent, "_entitlement_rejected_models", None) or () + if not rejected: + return False + if (provider, model) in rejected: + return True + from hermes_cli.model_normalize import normalize_model_for_provider + return (provider, normalize_model_for_provider(model, provider)) in rejected diff --git a/agent/interrupt_control.py b/agent/interrupt_control.py index f99aca9dc1..ced3a70175 100644 --- a/agent/interrupt_control.py +++ b/agent/interrupt_control.py @@ -215,8 +215,8 @@ class InterruptControlMixin: return True def steer(self, text: str) -> bool: - """Append user text to the LAST tool result once the batch finishes (no interrupt); multiple calls - concatenate with newlines. Returns False for empty text.""" + """Queue user text for delivery as its own user row after the current tool batch finishes (no + interrupt); multiple calls concatenate with newlines. Returns False for empty text.""" if not text or not text.strip(): return False cleaned = text.strip() diff --git a/agent/interrupt_scope.py b/agent/interrupt_scope.py new file mode 100644 index 0000000000..e08a66a823 --- /dev/null +++ b/agent/interrupt_scope.py @@ -0,0 +1,65 @@ +"""Host-owned cancellation for agents created deep inside synchronous work. + +A host that runs a blocking command on a worker thread (Hermes Console) never sees +the ``AIAgent`` a CLI subcommand forks inside it, so it cannot call ``interrupt()`` +when the user cancels. The host binds an :class:`InterruptScope` around the work; +every ``run_conversation()`` under that scope registers its agent, and +``scope.cancel()`` hard-interrupts them from any thread. Agents registering after +the cancel are interrupted immediately, so a cancel never loses the race with a +turn that has not started yet (#106179). +""" + +from __future__ import annotations + +import threading +from contextlib import contextmanager, nullcontext +from contextvars import ContextVar +from typing import Any, Iterator, Optional + +from agent.interrupt_compat import request_hard_interrupt + +_ACTIVE_SCOPE: ContextVar[Optional["InterruptScope"]] = ContextVar("hermes_interrupt_scope", default=None) + + +class InterruptScope: + def __init__(self) -> None: + self._lock = threading.Lock() + self._agents: list[Any] = [] + self.reason: Optional[str] = None + + def cancel(self, reason: str) -> None: + """Latch ``reason`` and hard-interrupt every agent running under this scope.""" + with self._lock: + self.reason = reason + agents = list(self._agents) + for agent in agents: + request_hard_interrupt(agent, reason, tool_reason="host cancelled the command") + + @contextmanager + def track(self, agent: Any) -> Iterator[None]: + with self._lock: + self._agents.append(agent) + reason = self.reason + if reason is not None: + request_hard_interrupt(agent, reason, tool_reason="host cancelled the command") + try: + yield + finally: + with self._lock: + self._agents.remove(agent) + + +@contextmanager +def bind_interrupt_scope(scope: Optional[InterruptScope]) -> Iterator[None]: + """Make ``scope`` the owner of every agent turn started in this context.""" + token = _ACTIVE_SCOPE.set(scope) + try: + yield + finally: + _ACTIVE_SCOPE.reset(token) + + +def track_in_interrupt_scope(agent: Any): + """Register ``agent`` with the bound scope for the duration of its turn (no-op without one).""" + scope = _ACTIVE_SCOPE.get() + return nullcontext() if scope is None else scope.track(agent) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index a27ea3c137..aa6b60eb3c 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -556,8 +556,10 @@ def is_local_endpoint(base_url: str) -> bool: return False if host is None: return False - # Unqualified hostnames (no dots) are local by definition — Docker Compose service names, /etc/hosts entries, mDNS. - if host in _LOCAL_HOSTS or host.endswith(_CONTAINER_LOCAL_SUFFIXES) or (host and "." not in host): + # Unqualified hostnames (no dots) are local by definition — Docker Compose service names, /etc/hosts + # entries, mDNS — as is `*.local` (RFC 6762 mDNS, LAN-only). IPv6 literals have no dots either, so + # they are excluded here and classified by scope below (a global address is not local). + if host in _LOCAL_HOSTS or host.endswith(_CONTAINER_LOCAL_SUFFIXES) or host.endswith(".local") or (host and "." not in host and ":" not in host): return True try: addr = ipaddress.ip_address(host) diff --git a/agent/periodic_scheduler.py b/agent/periodic_scheduler.py index a1ba07d666..6495172b4a 100644 --- a/agent/periodic_scheduler.py +++ b/agent/periodic_scheduler.py @@ -3,15 +3,20 @@ Replaces the per-child ``while not stop.wait(interval): body()`` daemon threads (delegate heartbeat, durable turn-lease refresher, turn-liveness watchdog). With ~130 in-process subagents those added 2-3 sleeping OS -threads per child; this module runs every periodic body on ONE daemon -thread ordered by a heap of due times. +threads per child. This module keeps ONE daemon thread that only orders due +times; every due body runs on its own short-lived daemon worker. A blocked +callback therefore cannot delay unrelated lease/liveness timers, while +steady-state thread use stays near zero (workers exist only while a body is +actually running, never one per scheduled handle). Semantics match the loop they replace: the first call happens ``interval`` seconds after :func:`schedule`, and each following call ``interval`` seconds after the previous body *returned* (drift-free wrt. body duration was never a property of the old loops either). A body that returns ``False`` stops itself; a body that raises is logged at debug and rescheduled — one bad -callback must never kill the shared thread. +callback must never kill the shared thread. A handle never overlaps itself: +it is re-queued only once its in-flight run has returned. A worker-start +failure never retires the handle: it is re-queued and logged at warning. """ from __future__ import annotations @@ -26,18 +31,20 @@ from typing import Callable, Optional logger = logging.getLogger(__name__) _THREAD_NAME = "hermes-periodic-scheduler" +_CALLBACK_THREAD_PREFIX = "hermes-periodic-callback" class ScheduledHandle: """Cancel token for one scheduled periodic callback.""" - __slots__ = ("_fn", "_interval", "_cancelled", "_scheduler") + __slots__ = ("_fn", "_interval", "_cancelled", "_scheduler", "_runner") def __init__(self, scheduler: "PeriodicScheduler", fn: Callable[[], object], interval: float): self._scheduler = scheduler self._fn = fn self._interval = interval self._cancelled = False + self._runner: Optional[threading.Thread] = None @property def cancelled(self) -> bool: @@ -56,12 +63,11 @@ class PeriodicScheduler: self._heap: list = [] # (due, seq, handle) self._seq = itertools.count() self._thread: Optional[threading.Thread] = None - self._running: Optional[ScheduledHandle] = None def schedule(self, fn: Callable[[], object], interval: float) -> ScheduledHandle: handle = ScheduledHandle(self, fn, float(interval)) with self._cond: - heapq.heappush(self._heap, (time.monotonic() + handle._interval, next(self._seq), handle)) + self._requeue(handle) if self._thread is None or not self._thread.is_alive(): self._thread = threading.Thread(target=self._run, name=_THREAD_NAME, daemon=True) self._thread.start() @@ -72,8 +78,52 @@ class PeriodicScheduler: with self._cond: handle._cancelled = True self._cond.notify() - if wait and self._running is handle and threading.current_thread() is not self._thread: - self._cond.wait_for(lambda: self._running is not handle, timeout=wait) + runner = handle._runner + if wait and runner is not None and threading.current_thread() is not runner: + runner.join(wait) + + def _dispatch(self, handle: ScheduledHandle) -> None: + """Start ``handle``'s body on its own worker. Called with ``_cond`` held + so ``cancel`` can never observe a half-set runner.""" + runner = threading.Thread( + target=self._run_callback, + args=(handle,), + name=f"{_CALLBACK_THREAD_PREFIX}-{id(handle):x}", + daemon=True, + ) + handle._runner = runner + try: + runner.start() + except Exception: + handle._runner = None + logger.warning( + "failed to start periodic callback worker %r; retrying in %s s", + handle._fn, + handle._interval, + exc_info=True, + ) + if not handle._cancelled: + self._requeue(handle) + self._cond.notify() + + def _requeue(self, handle: ScheduledHandle) -> None: + """Push ``handle``'s next due time (``_cond`` held).""" + heapq.heappush(self._heap, (time.monotonic() + handle._interval, next(self._seq), handle)) + + def _run_callback(self, handle: ScheduledHandle) -> None: + stop = False + try: + stop = handle._fn() is False + except Exception: + logger.debug("periodic callback %r raised", handle._fn, exc_info=True) + finally: + with self._cond: + handle._runner = None + if stop: + handle._cancelled = True + elif not handle._cancelled: + self._requeue(handle) + self._cond.notify() def _run(self) -> None: while True: @@ -91,28 +141,13 @@ class PeriodicScheduler: self._cond.wait(delay) continue heapq.heappop(self._heap) - self._running = handle + self._dispatch(handle) break - stop = False - try: - stop = handle._fn() is False - except Exception: - logger.debug("periodic callback %r raised", handle._fn, exc_info=True) - with self._cond: - self._running = None - if stop: - handle._cancelled = True - elif not handle._cancelled: - heapq.heappush( - self._heap, - (time.monotonic() + handle._interval, next(self._seq), handle), - ) - self._cond.notify_all() _DEFAULT = PeriodicScheduler() def schedule(fn: Callable[[], object], interval: float) -> ScheduledHandle: - """Run ``fn()`` every ``interval`` seconds on the shared scheduler thread.""" + """Run ``fn()`` every ``interval`` seconds via the shared scheduler.""" return _DEFAULT.schedule(fn, interval) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 43f3a125e0..1b09d5f154 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -13,7 +13,7 @@ import sys import threading from collections import OrderedDict from pathlib import Path -from typing import Optional +from typing import Any, Dict, Optional from hermes_constants import ( get_hermes_home, get_skills_dir, is_wsl, reset_hermes_home_override, set_hermes_home_override, @@ -514,10 +514,22 @@ STEER_MARKER_CLOSE = "[/OUT-OF-BAND USER MESSAGE]" def format_steer_marker(steer_text: str) -> str: - """Wrap a mid-turn steer for appending to a tool result (see note above).""" + """Wrap a mid-turn steer in the self-describing marker (see note above).""" return f"\n\n{STEER_MARKER_OPEN}\n{steer_text}\n{STEER_MARKER_CLOSE}" +STEER_DISPLAY_KIND = "steer" + + +def steer_user_row(steer_text: str) -> Dict[str, Any]: + """The standalone ``role:user`` row a mid-turn /steer is delivered as (after the newest tool + result). Its own row — never smeared onto the already-persisted tool row, which append-only + persistence would leave divergent from the live request — and typed so the alternation repair + never merges the next real prompt into it and history renderers can label it.""" + return {"role": "user", "content": format_steer_marker(steer_text).lstrip(), + "display_kind": STEER_DISPLAY_KIND} + + STEER_CHANNEL_NOTE = ( # Only what the marker cannot say about itself: it is the ONLY trusted shape and carries full user authority. # Dieted (#95681, maintainer-directed). History: #40240 added this note when the marker was bare and diff --git a/agent/session_persistence.py b/agent/session_persistence.py index d8fe1ece2c..2d39226f55 100644 --- a/agent/session_persistence.py +++ b/agent/session_persistence.py @@ -7,7 +7,7 @@ import logging import re from contextlib import nullcontext -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Tuple from agent.context_compressor import ( COMPRESSED_SUMMARY_METADATA_KEY, @@ -74,6 +74,19 @@ def _override_replaces_content(msg: Dict, content: Any, override: Any) -> bool: ) +def durable_user_row_content(agent, msg: Dict, content: Any, api_content: Any) -> Tuple[Any, Any]: + """``(content, api_content)`` as the current turn's user row is written: the persist override is the + clean transcript, the live content is what the wire sent — so when they differ and nothing else was + injected, the live bytes ARE the sidecar. Shared by the flush and the turn-start stamp so the stamp + matches the row the flush wrote.""" + override = getattr(agent, "_persist_user_message_override", None) + if _override_replaces_content(msg, content, override): + if api_content is None and isinstance(content, str) and content != override: + api_content = content + content = override + return content, api_content + + def _summary_display_kind(msg: Dict) -> Any: """Standalone handoffs are hidden so they never occupy the active user slot in retry/undo dispatch; merge-into-tail carriers keep their prior visibility.""" @@ -143,12 +156,7 @@ def _db_flush_row(agent, msg: Dict, is_current_turn_user: bool) -> Dict[str, Any api_content = msg.get("api_content") if isinstance(msg.get("api_content"), str) else None timestamp = msg.get("timestamp") if is_current_turn_user and role == "user": - override = getattr(agent, "_persist_user_message_override", None) - if _override_replaces_content(msg, content, override): - # Live content is what the wire sent, the override is the clean transcript; keep the sent bytes. - if api_content is None and isinstance(content, str) and content != override: - api_content = content - content = override + content, api_content = durable_user_row_content(agent, msg, content, api_content) ov_timestamp = getattr(agent, "_persist_user_message_timestamp", None) timestamp = timestamp if ov_timestamp is None else ov_timestamp if api_content == content: diff --git a/agent/surface_switch.py b/agent/surface_switch.py new file mode 100644 index 0000000000..5b642ecd0b --- /dev/null +++ b/agent/surface_switch.py @@ -0,0 +1,124 @@ +"""Surface switch without a prompt rebuild (#104414). + +A surface switch (desktop <-> TUI, a session resumed under a different host) changes only which +interface renders the reply, but the surface guidance and the ``Platform:`` trailer are embedded +in the persisted system prompt. Rebuilding for it diverged the prompt within its first blocks, +so the ENTIRE request behind it re-prefilled — a 220K-token session came back at a 1% cache hit. +The stored bytes are therefore kept and the CURRENT surface's guidance is delivered on the +per-turn user-message channel instead: that lands after the cached prefix and is stamped into +the byte-stable ``api_content`` sidecar so later turns replay it unchanged. The prompt itself +converges at the next rebuild boundary (compaction). +""" +from __future__ import annotations + +import logging +from typing import Any, List + +from agent.message_content import flatten_message_text +from agent.prompt_builder import RUNTIME_ENVIRONMENT_END, RUNTIME_ENVIRONMENT_HEADING + +logger = logging.getLogger("run_agent") + +_SURFACE_SWITCH_NOTE_PREFIX = "[System: This conversation is now being answered on a different interface: " +# Closes the surface name in the note; platform names are free-form for plugin platforms, so the +# terminator (not ".") delimits the parse. +_SURFACE_NAME_END = " — any earlier interface guidance" +# Only the newest note matters, and the note is re-stamped on the switch turn, so a bounded tail +# scan is enough; without a bound every turn of a never-switched session walks the whole transcript. +_NOTE_SCAN_TAIL = 200 + + +def split_runtime_boundary(prompt: str) -> tuple: + """``(identity, runtime_marker, runtime)`` of a persisted prompt. Legacy prose may quote + the runtime heading, but only the new renderer ENDS in the boundary; when the marker is + empty the whole prompt is identity.""" + identity, runtime_marker, runtime = prompt.rpartition(f"\n\n{RUNTIME_ENVIRONMENT_HEADING}\n\n") + return (identity, runtime_marker, runtime) if prompt.endswith(RUNTIME_ENVIRONMENT_END) else (prompt, "", "") + + +def identity_line_value(prompt: str, label: str) -> str: + """Last ``Label: value`` line in the identity portion (the final runtime block is embedder + prose, never identity). Last match wins — safe only for the volatile-tier trailer fields.""" + prefix = f"{label}:" + matches = [line[len(prefix):].strip() for line in split_runtime_boundary(prompt)[0].splitlines() + if line.startswith(prefix)] + return matches[-1] if matches else "" + + +def _last_announced_surface(conversation_history: Any) -> str: + """The surface named by the NEWEST switch note in the transcript ("" when none). + + Once a switch has been announced, that note — not the stored prompt's ``Platform:`` trailer — + is the last thing the model was told it runs on. Reading it back is also what keeps a fresh + AIAgent per turn (the gateway shape) from stacking one copy of the note per turn.""" + for msg in reversed((conversation_history or [])[-_NOTE_SCAN_TAIL:]): + # The note only ever lands on a user row: in its api_content sidecar, or as a text part + # when the content is a multimodal list (which cannot take the string sidecar). + if not isinstance(msg, dict) or msg.get("role") != "user": + continue + sidecar = msg.get("api_content") + text = (sidecar if isinstance(sidecar, str) else "") + "\n" + flatten_message_text(msg.get("content")) + if _SURFACE_SWITCH_NOTE_PREFIX in text: + tail = text.rsplit(_SURFACE_SWITCH_NOTE_PREFIX, 1)[1] + return tail.split(_SURFACE_NAME_END, 1)[0].strip() + return "" + + +def note_inert_pinned_tools(agent: Any, built_for_this_surface: List[str]) -> None: + """Name, at the end of the staged note, the pinned tools THIS surface did not build. + + The freeze keeps a previous surface's tools on the wire deliberately — removing them is the one + thing that would still re-prefill the request behind the preserved prompt — so the model has + to be TOLD they are inert here, or it plans around a ``focus_pane`` a terminal turn can only + answer with ``tool_error("desktop only")``.""" + from tools.mcp_tool_agent import agent_tool_names + surface_names = set(built_for_this_surface) + inert = [name for name in agent_tool_names(agent) if name not in surface_names] + note = getattr(agent, "_surface_switch_note", "") or "" + if not inert or not note: + return + agent._surface_switch_note = note + ( + "\n[System: These tools stay listed for this conversation (dropping them would discard " + "the cached request prefix) but were not loaded for this interface — expect a call to " + f"one of them to fail: {', '.join(inert)}.]" + ) + + +def stage_surface_switch_note(agent: Any, prompt: str, conversation_history: Any) -> bool: + """Stage a one-shot correction when the request would otherwise misdescribe the surface. + + Compares the runtime surface against what the model was last told — the newest switch note + if one exists, else ``prompt``'s own ``Platform:`` trailer. Consulting the note matters in + both directions: switching BACK to the prompt's surface (desktop -> tui -> desktop) leaves the + trailer agreeing with the runtime while a stale note still says otherwise, and a rebuild for + an unrelated reason (a model switch) refreshes the prompt but not that note. The full surface + guidance is attached only when the prompt itself is out of date; otherwise the note just + retires the stale one. Returns whether it staged. + + MoA and codex_app_server turns never stamp the ``api_content`` sidecar, so the note could not + be read back and would be re-sent every turn; those modes keep the stored prompt and skip the + note entirely.""" + if getattr(agent, "provider", None) == "moa" or getattr(agent, "api_mode", None) == "codex_app_server": + return False + current = str(getattr(agent, "platform", "") or "").strip() + if not current: + return False + described = identity_line_value(prompt, "Platform") + told = _last_announced_surface(conversation_history) or described + if not told or told == current: + return False + from agent.system_prompt import platform_hint + hint = platform_hint(agent) if described != current else "" + where = "the guidance below" if hint else "the interface section in the system prompt above" + note = ( + f"{_SURFACE_SWITCH_NOTE_PREFIX}{current}{_SURFACE_NAME_END} in this conversation is " + f"superseded — follow {where} for formatting, file delivery and any interface-specific " + "capability.]" + ) + agent._surface_switch_note = f"{note}\n{hint}" if hint else note + logger.info( + "Session %s switched surface %s -> %s; keeping the stored system prompt and delivering " + "the new surface guidance as a turn note (prefix cache preserved).", + agent.session_id, told, current, + ) + return True diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 34338b16d5..3778515924 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -381,7 +381,7 @@ def _active_profile_line(agent: Any) -> str: ) -def _platform_hint(agent: Any) -> str: +def platform_hint(agent: Any) -> str: """Built-in/plugin platform hint + Telegram rich-messages opt-in + config override + desktop TUI clarifier.""" platform_key = (agent.platform or "").lower().strip() @@ -579,7 +579,7 @@ def _post_workspace_parts(agent: Any) -> List[str]: pass # Probe failure must never block prompt build. if getattr(agent, "_bot_mode_protocol", True): parts.extend(_bot_mode_parts(agent)) - parts += [_active_profile_line(agent), _platform_hint(agent)] + parts += [_active_profile_line(agent), platform_hint(agent)] return parts @@ -736,7 +736,7 @@ def format_tools_for_system_message(agent: Any) -> str: __all__ = ["build_system_prompt_parts", "build_system_prompt", "invalidate_system_prompt", - "restore_plugin_prompt_sections", "format_tools_for_system_message"] + "platform_hint", "restore_plugin_prompt_sections", "format_tools_for_system_message"] # ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ---- diff --git a/agent/tool_executor.py b/agent/tool_executor.py index a34d284312..6b15d2f637 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -173,10 +173,12 @@ def _resolve_concurrent_tool_timeout() -> float | None: def _flush_session_db_after_tool_progress(agent, messages: list, *, stage: str) -> bool: """Flush tool-call progress to the session DB before projecting it to any UI: tool side effects can kill/restart the process before turn-end persistence runs.""" + from agent.conversation_loop import _maybe_inject_run_budget_wrapup from agent.turn_iteration_prep import _maybe_inject_iteration_budget_warning # Persist exactly the checkpoint text the next model call will see, before stamping # this tool result as durable. Already-written rows must never be rewritten later. + _maybe_inject_run_budget_wrapup(agent, messages) _maybe_inject_iteration_budget_warning(agent, messages) try: persisted = agent._flush_messages_to_session_db(messages) is not False diff --git a/agent/turn_api_error.py b/agent/turn_api_error.py index 13cd554a50..305570c8d4 100644 --- a/agent/turn_api_error.py +++ b/agent/turn_api_error.py @@ -282,6 +282,10 @@ def settle_unrecovered_error( ) and not is_context_length_error if is_client_error: + # A Codex ChatGPT-account entitlement 400 names the model: with nothing to rotate the + # slug is dead for this account, so record it before the fallback walk runs (#106475). + from agent.fallback_cooldown import _mark_entitlement_rejected_model + _mark_entitlement_rejected_model(agent, api_error) # Copilot self-heal BEFORE fallback: a stale credential yields a 400 # ``model_not_available_for_integrator`` / ``model_not_supported``, not a 401. # Fresh token + client rebuild, one retry, SAME provider. diff --git a/agent/turn_context.py b/agent/turn_context.py index 6ecb93a313..584d75d211 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -114,14 +114,25 @@ def extract_api_content_sidecar(msg: Mapping[str, Any]) -> Optional[str]: return v if isinstance(v, str) else None -def consume_gateway_turn_context_notes(agent: Any) -> str: - """Pop the gateway's per-turn must-deliver notes off the agent (one-shot, so the - system prompt stays byte-stable and a cached agent never replays a stale note).""" - notes = getattr(agent, "_gateway_turn_context_notes", "") or "" - if hasattr(agent, "_gateway_turn_context_notes"): +def _pop_turn_note(agent: Any, attr: str) -> str: + """One-shot per-turn note: read and clear, so the system prompt stays byte-stable and a + cached agent never replays a stale note.""" + note = getattr(agent, attr, "") or "" + if hasattr(agent, attr): with suppress(Exception): - agent._gateway_turn_context_notes = "" - return notes if isinstance(notes, str) else "" + setattr(agent, attr, "") + return note if isinstance(note, str) else "" + + +def consume_gateway_turn_context_notes(agent: Any) -> str: + """Pop the gateway's per-turn must-deliver notes.""" + return _pop_turn_note(agent, "_gateway_turn_context_notes") + + +def consume_surface_switch_note(agent: Any) -> str: + """Pop the surface-switch note staged by the system-prompt restore (#104414); rides the same + user-message channel as the gateway notes, behind the cached prefix.""" + return _pop_turn_note(agent, "_surface_switch_note") def append_notes_to_multimodal_content(content: Any, notes: str) -> bool: @@ -233,6 +244,44 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in return fallback +def export_current_turn_boundary(agent: Any, result: Any, user_message: Any) -> Any: + """Stamp ``{turn_id, current_turn_user_idx}`` on a result envelope, proven against the + exact ``result["messages"]`` projection it travels with. + + Hosts that settle their own transcript by index (hermes-webui) must never guess which + row is the current user turn after this loop rewrote history (alternation repair, + compaction, post-turn micro-compaction): a guessed index or a text match can relabel an + identical historical prompt and claim its old answer as this turn's. So the producer + exports the coordinate, computed on the final list, only when the addressed row is this + turn's user message verbatim. Otherwise the keys are omitted and hosts fail closed. + + A preflight-timeout envelope carries the prior history without this turn's row (#7100), so a + repeated prompt would resolve to its historical copy: nothing is exported there. + """ + if not isinstance(result, dict) or result.get("turn_exit_reason") == "context_compression_timeout": + return result + messages = result.get("messages") + turn_id = str(getattr(agent, "_current_turn_id", "") or "") + if not isinstance(messages, list) or not turn_id or user_message is None: + return result + idx = reanchor_current_turn_user_idx(messages, user_message) + if idx < 0 or idx >= len(messages): + return result + row = messages[idx] + if not (isinstance(row, dict) and row.get("role") == "user"): + return result + from agent.context_compressor import user_originated_turn_view + + live_view = user_originated_turn_view(row) + if row.get("content") != user_message and not ( + isinstance(live_view, dict) and live_view.get("content") == user_message + ): + return result # rewritten (merge-into-tail) row: not a proven boundary + result["turn_id"] = turn_id + result["current_turn_user_idx"] = idx + return result + + def compression_made_progress( orig_len: int, new_len: int, orig_tokens: int, new_tokens: int ) -> bool: @@ -663,11 +712,15 @@ def _collect_pre_llm_call_context( def _merge_gateway_notes( agent: Any, messages: List[Any], current_turn_user_idx: int, plugin_user_context: str ) -> str: - """Gateway must-deliver notes ride the user-message injection channel (one-shot, - gateway-staged) so the ephemeral system prompt stays byte-stable. Multimodal (list) - content can't take the string sidecar — append a durable text part instead.""" - _gateway_notes = consume_gateway_turn_context_notes(agent) - if not _gateway_notes: + """Must-deliver per-turn notes ride the user-message injection channel (one-shot) so the + ephemeral system prompt stays byte-stable: the gateway's staged notes, then the + surface-switch correction. Multimodal (list) content can't take the string sidecar — + append a durable text part instead.""" + _turn_notes = "\n\n".join( + part for part in (consume_gateway_turn_context_notes(agent), + consume_surface_switch_note(agent)) if part + ) + if not _turn_notes: return plugin_user_context _gw_turn_content = ( messages[current_turn_user_idx].get("content") @@ -676,10 +729,10 @@ def _merge_gateway_notes( else None ) if isinstance(_gw_turn_content, list): - append_notes_to_multimodal_content(_gw_turn_content, _gateway_notes) + append_notes_to_multimodal_content(_gw_turn_content, _turn_notes) return plugin_user_context return ( - plugin_user_context + "\n\n" + _gateway_notes if plugin_user_context else _gateway_notes + plugin_user_context + "\n\n" + _turn_notes if plugin_user_context else _turn_notes ) @@ -728,30 +781,44 @@ def _stamp_api_content_sidecar( """api_content sidecar — persist what you send: injected context lives only in the API copy, so stamp the exact sent bytes on the live dict for replay.""" _turn_user_msg = messages[current_turn_user_idx] - _api_content = compose_user_api_content( - _turn_user_msg.get("content", ""), ext_prefetch_cache, plugin_user_context + live_content = _turn_user_msg.get("content") + from agent.session_persistence import _persist_lock, durable_user_row_content + # Match the row the flush wrote (persist override = clean transcript), not the live bytes. + durable_content, _api_content = durable_user_row_content( + agent, _turn_user_msg, live_content, + compose_user_api_content(live_content or "", ext_prefetch_cache, plugin_user_context), ) - if _api_content is None or _api_content == _turn_user_msg.get("content"): + if _api_content is None or _api_content == durable_content: return _turn_user_msg["api_content"] = _api_content - # In-place preflight compaction already inserted this turn's user row and the - # crash persist identity-skips compacted dicts, so backfill the stamp onto the row - # directly. Rotation mode flushes to the child session later. - if not (preflight_compressed and getattr(agent, "_last_compaction_in_place", False)): - return - _db = getattr(agent, "_session_db", None) - if _db is not None: + + # When another writer materialized this turn's user row BEFORE the sidecar existed — in-place + # preflight compaction, or a close/early flush that raced the prologue (#102194) — the crash + # persist marker-skips the message and the stamp never reaches the DB, so the next turn replays + # clean content and the request prefix diverges here. Both writers stamp ``_row_id`` on the live + # dict, which is at once the proof a row exists and the address to update. + # + # Never widen this to an unconditional positional backfill — see set_latest_user_api_content. + # + # ``_row_id`` is read under ``_session_persist_lock``: a close flush holds it while it commits + # the row and only then writes ``_row_id`` back (``sync_flushed_message_markers``). Read outside + # it, the stamp can land in between, see no id, return — and the flush then marks the message + # persisted with ``api_content = NULL``, leaving no writer to correct the row. + with _persist_lock(agent): + _row_id = _turn_user_msg.get("_row_id") + _in_place_compacted = preflight_compressed and bool(getattr(agent, "_last_compaction_in_place", False)) + _db = getattr(agent, "_session_db", None) + if _db is None or not (isinstance(_row_id, int) or _in_place_compacted): + return try: - _db.set_latest_user_api_content( - agent.session_id, _turn_user_msg.get("content"), _api_content - ) + if isinstance(_row_id, int): + _db.set_message_api_content(agent.session_id, _row_id, durable_content, _api_content) + else: + # Compacted copies carry no row id; positional is safe only because + # archive_and_compact just made this message the newest active user row. + _db.set_latest_user_api_content(agent.session_id, durable_content, _api_content) except Exception: - logger.warning( - "in-place compaction api_content backfill failed " - "for session=%s", - agent.session_id or "none", - exc_info=True, - ) + logger.warning("api_content backfill failed for session=%s", agent.session_id or "none", exc_info=True) def _persist_turn_start( @@ -806,6 +873,8 @@ def build_turn_context( # warning and a needless first-turn prefix cache miss. (Issue #45499.) set_session_context(agent.session_id) set_current_write_origin(getattr(agent, "_memory_write_origin", "assistant_tool")) + from tools.skill_provenance import set_review_attended + set_review_attended(getattr(agent, "_review_attended", False)) agent._restore_primary_runtime() _publish_runtime_main(agent) _refresh_mcp_tools_between_turns(agent) diff --git a/agent/turn_explainers.py b/agent/turn_explainers.py index 4e66025ffa..f564572443 100644 --- a/agent/turn_explainers.py +++ b/agent/turn_explainers.py @@ -41,6 +41,16 @@ _EXIT_REASON_EXPLANATIONS: Dict[str, str] = { "the request was interrupted mid-call before a reply was " "received. Send `continue` to retry." ), + "redirect_restart_limit_exceeded": ( + "the request was cancelled by a new correction on every attempt, " + "so the turn stopped instead of retrying forever. Your last " + "correction is queued as the next message." + ), + "rebuilt_restart_limit_exceeded": ( + "every provider in the fallback chain kept failing over, so the " + "turn stopped instead of retrying forever. Send `continue` or " + "switch provider." + ), "budget_exhausted": ( "the per-turn iteration/cost budget was exhausted before a " "final answer. Send `continue` to keep going." diff --git a/agent/turn_facade.py b/agent/turn_facade.py index 5f147c62cf..128ecd1eca 100644 --- a/agent/turn_facade.py +++ b/agent/turn_facade.py @@ -47,6 +47,7 @@ class TurnFacadeMixin: from agent.prompt_cache_scope import declared_conversation_scope_safe from agent.review_idle_queue import QUEUE as _review_queue from agent.subagent_lifecycle import bind_subagent_parent + from agent.interrupt_scope import track_in_interrupt_scope from agent.turn_facade_lease import admit_durable_turn_lease from hermes_cli.observability.relay_shared_metrics import finish_task_run, start_task_run @@ -115,7 +116,8 @@ class TurnFacadeMixin: ) # Keep the ContextVar scope local (agent tokens may be observed from another thread). - with bind_subagent_parent(self), scoped_runtime_main({}): + # A host that owns this thread (Hermes Console) may cancel the turn cross-thread. + with bind_subagent_parent(self), scoped_runtime_main({}), track_in_interrupt_scope(self): try: if lease is not None: lease.start() diff --git a/agent/turn_facade_lease.py b/agent/turn_facade_lease.py index da33122f6f..9ae9c93a02 100644 --- a/agent/turn_facade_lease.py +++ b/agent/turn_facade_lease.py @@ -4,7 +4,8 @@ One process at a time may load -> run -> flush a session shared through state.db resume, gateway, background delivery). ``admit_durable_turn_lease`` acquires the row lease (or returns the early result the façade must hand back); ``DurableTurnLease`` owns the periodic refresher, the turn-liveness watchdog wiring, and the lease-loss / stall interrupt plumbing. Both -timers run on the shared scheduler thread (``agent/periodic_scheduler.py``), not per-turn threads. +timers run via the shared scheduler (``agent/periodic_scheduler.py``; timer thread orders, +bodies run on per-handle workers), not per-turn threads. """ import logging import os @@ -172,7 +173,7 @@ class DurableTurnLease: _set_interrupt(False, agent._execution_thread_id) def refresh_tick(self): - """One periodic renewal (every ``refresh_interval`` on the shared scheduler); a miss or + """One periodic renewal (every ``refresh_interval`` via the shared scheduler); a miss or error interrupts the turn. Returning False stops the timer. The holder-qualified UPDATE fences a late refresher from a successor lease. The façade's diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index a6541073d1..9d7daee131 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -1,7 +1,7 @@ """Outer-iteration bookkeeping for the conversation turn loop, in call order: ``begin_iteration`` (pending redirect, interrupt / review-budget / iteration-budget exits), -``prepare_iteration`` (``agent:step`` callback, skill-nudge counter, pre-API ``/steer`` drain -into the newest tool result — never a user message —, run-budget wrap-up notice, tool_call +``prepare_iteration`` (``agent:step`` callback, skill-nudge counter, pre-API ``/steer`` drain as a +standalone user row after the newest tool result, run-budget wrap-up notice, tool_call argument sanitization, interrupt-scaffold ghost-row drop, role-alternation repair), ``announce_api_call`` (verbose summary / quiet spinner) and, after the retry loop, ``apply_retry_restarts`` (consumes the ``TurnRetryState`` restart flags). Nothing here @@ -17,7 +17,7 @@ from dataclasses import dataclass from typing import Any, Dict from agent.display import KawaiiSpinner -from agent.turn_context import reanchor_current_turn_user_idx +from agent.turn_context_compaction import _reanchor logger = logging.getLogger("agent.conversation_loop") @@ -89,11 +89,14 @@ class IterationPrep: action: str messages: Any request_logger: Any + current_turn_user_idx: Any -def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> IterationPrep: +def prepare_iteration( + agent: Any, *, messages: Any, api_call_count: Any, user_message: Any = None, current_turn_user_idx: Any = None, +) -> IterationPrep: """Prepare ``messages`` for this iteration in the original order. Every mutation here is - cache-safe by construction: steer text lands in the newest tool result, the ghost-row + cache-safe by construction: steer text is appended as a new (not yet persisted) user row, the ghost-row filter only drops hidden scaffold placeholders, and repair runs BEFORE the request build.""" from agent.conversation_loop import ( _INTERRUPT_SCAFFOLD_MARKER, _maybe_inject_run_budget_wrapup @@ -125,18 +128,20 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera except Exception: logger.debug("Nous key pre-expiry adoption failed", exc_info=True) - # Drain a /steer sent during the last API call into the newest tool message so - # it lands THIS iteration. Never put in a user message (breaks alternation). + # Drain a /steer sent during the last API call so it lands THIS iteration. Delivered as a + # standalone user row after the newest tool result (never smeared onto the tool row: that + # row is already persisted append-only, so replay would diverge from the live request and + # break the prompt cache — same contract as apply_pending_steer_to_tool_results). _pre_api_steer = agent._drain_pending_steer() if _pre_api_steer: - _inject_steer_into_newest_tool_result(agent, messages, _pre_api_steer) + _inject_steer_after_newest_tool_result(agent, messages, _pre_api_steer) - # One-shot run-budget wrap-up notice at 80% of agent.run_budget_seconds, via the - # same cache-safe channel as /steer (newest tool result); off with no budget. + # One-shot run-budget wrap-up notice at 80% of agent.run_budget_seconds, appended to the + # newest tool result; off with no budget. if getattr(agent, "run_budget_seconds", None): _maybe_inject_run_budget_wrapup(agent, messages) - # Use the same cache-safe channel as /steer; never add a synthetic user/system row. + # Appended to the newest tool result; never a synthetic user/system row. _maybe_inject_iteration_budget_warning(agent, messages) request_logger = getattr(agent, "logger", None) or logger # same name as the origin module @@ -182,7 +187,23 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera repaired_seq, agent.session_id or "-", ) - return IterationPrep(action="fallthrough", messages=messages, request_logger=request_logger) + # The merge shrank the list, so the index recorded at turn start can point past this + # turn's user row: prefetch would inject into a historical row and index-settling hosts + # (hermes-webui) would write the current turn to the FRONT of the context. Re-anchor as + # the compression-restart path does (last verbatim row wins, never a historical copy); + # without the text the index cannot be re-derived and is left detectably stale. + if user_message is not None: + _reanchored_idx = _reanchor(agent, messages, user_message) + if _reanchored_idx != current_turn_user_idx: + request_logger.info( + "Re-anchored current_turn_user_idx %s -> %s after alternation repair (session=%s)", + current_turn_user_idx, _reanchored_idx, agent.session_id or "-", + ) + current_turn_user_idx = _reanchored_idx + return IterationPrep( + action="fallthrough", messages=messages, request_logger=request_logger, + current_turn_user_idx=current_turn_user_idx, + ) def _previous_tool_round(messages: Any) -> list: @@ -208,37 +229,18 @@ def _previous_tool_round(messages: Any) -> list: return [] -def _inject_steer_into_newest_tool_result(agent: Any, messages: Any, steer_text: str) -> None: - """Append the steer marker to the newest tool message; with no tool message, put the - text back so the post-tool-execution drain delivers it later.""" +def _inject_steer_after_newest_tool_result(agent: Any, messages: Any, steer_text: str) -> None: + """Append the steer marker as a standalone user row after the newest tool message; with no + tool message, put the text back so the post-tool-execution drain delivers it later.""" for _si in range(len(messages) - 1, -1, -1): _sm = messages[_si] if isinstance(_sm, dict) and _sm.get("role") == "tool": - from agent.prompt_builder import format_steer_marker - marker = format_steer_marker(steer_text) - existing = _sm.get("content", "") - if isinstance(existing, str): - _sm["content"] = existing + marker - else: - # Multimodal content blocks — append a text block. - with suppress(Exception): - blocks = list(existing) if existing else [] - blocks.append({"type": "text", "text": marker}) - _sm["content"] = blocks - logger.debug( - "Pre-API-call steer drain: injected into tool msg at index %d", _si - ) + from agent.prompt_builder import steer_user_row + messages.insert(_si + 1, steer_user_row(steer_text)) + logger.debug("Pre-API-call steer drain: appended user row after tool msg at index %d", _si) return - _lock = getattr(agent, "_pending_steer_lock", None) - if _lock is not None: - with _lock: - if agent._pending_steer: - agent._pending_steer = agent._pending_steer + "\n" + steer_text - else: - agent._pending_steer = steer_text - else: - existing = getattr(agent, "_pending_steer", None) - agent._pending_steer = (existing + "\n" + steer_text) if existing else steer_text + from agent.agent_runtime_helpers import _requeue_pending_steer + _requeue_pending_steer(agent, steer_text) @dataclass @@ -373,6 +375,7 @@ class RetryRestartVerdict: current_turn_user_idx: Any final_response: Any retry_count: Any + restart_count: Any api_call_count: Any _preflight_compression_blocked: Any _turn_exit_reason: Any @@ -381,13 +384,20 @@ class RetryRestartVerdict: def apply_retry_restarts( agent: Any, *, _retry: Any, response: Any, interrupted: Any, messages: Any, conversation_history: Any, user_message: Any, api_kwargs: Any, current_turn_user_idx: Any, - final_response: Any, retry_count: Any, api_call_count: Any, length_continue_retries: Any, + final_response: Any, retry_count: Any, max_retries: Any, api_call_count: Any, + restart_count: Any, length_continue_retries: Any, _preflight_compression_blocked: Any, _turn_exit_reason: Any, ) -> RetryRestartVerdict: """Consume the ``TurnRetryState`` restart flags after the retry loop, in the original priority order. Refunds the iteration budget/count for restarts that produced no valid assistant item; ``restart_with_rebuilt_messages`` is the single consumer that clears - ``_preflight_compression_blocked`` so the fallback gets a fresh preflight (#84733).""" + ``_preflight_compression_blocked`` so the fallback gets a fresh preflight (#84733). + + The two refunding restart paths (redirect and rebuilt-for-fallback) are bounded by + ``max_retries`` via ``restart_count`` (a per-turn accumulator) so a runaway + interrupt/redirect that keeps re-arming a restart flag cannot refund the budget + forever and hold the turn lease indefinitely.""" + from agent.conversation_loop import ( _HANDOFF_SKIP_FINAL_RESPONSE, _should_skip_model_call_for_reference_handoff ) @@ -395,12 +405,30 @@ def apply_retry_restarts( def _verdict(action: str) -> RetryRestartVerdict: return RetryRestartVerdict( action=action, current_turn_user_idx=current_turn_user_idx, - final_response=final_response, retry_count=retry_count, api_call_count=api_call_count, + final_response=final_response, retry_count=retry_count, restart_count=restart_count, + api_call_count=api_call_count, _preflight_compression_blocked=_preflight_compression_blocked, _turn_exit_reason=_turn_exit_reason, ) if _retry.restart_with_redirected_messages: + restart_count += 1 + if restart_count > max_retries: + # A redirect/interrupt keeps re-arming this flag: stop refunding the iteration + # budget and re-issuing the same logical iteration, or a runaway turn holds the + # turn lease indefinitely (redirect restarts previously had no bound). + _turn_exit_reason = "redirect_restart_limit_exceeded" + logger.warning( + "Redirected-message restart limit (%s) exceeded; ending turn instead of " + "refunding the iteration budget indefinitely.", + max_retries, + ) + # The correction that tripped the cap was never applied; hand it back as the + # next user turn (result["pending_steer"]) instead of losing it to clear_interrupt(). + _unapplied = agent._drain_pending_redirect() + if _unapplied: + agent.steer(_unapplied) + return _verdict("break") # Cancelled request produced no valid assistant item: reuse the same logical # iteration after the outer loop appends partial context + correction. api_call_count -= 1 @@ -438,11 +466,22 @@ def apply_retry_restarts( # In-loop compression rebuilt `messages`; re-anchor the current-turn index # like the prologue, AFTER the handoff guard (it may re-append this turn's # ask). A stale anchor injects prefetch into a historical row. - current_turn_user_idx = reanchor_current_turn_user_idx(messages, user_message) - agent._persist_user_message_idx = current_turn_user_idx + current_turn_user_idx = _reanchor(agent, messages, user_message) return _verdict("continue") if _retry.restart_with_rebuilt_messages: + restart_count += 1 + if restart_count > max_retries: + # A stall/failure keeps re-escalating to the fallback chain: stop refunding the + # iteration budget and re-issuing, or a runaway turn holds the turn lease + # indefinitely (rebuilt restarts previously had no bound). + _turn_exit_reason = "rebuilt_restart_limit_exceeded" + logger.warning( + "Rebuilt-message restart limit (%s) exceeded; ending turn instead of " + "refunding the iteration budget indefinitely.", + max_retries, + ) + return _verdict("break") # A stall/failure escalated to the fallback chain: re-issue against the # active fallback provider, refunding budget/count for the stalled attempt. api_call_count -= 1 diff --git a/agent/turn_liveness.py b/agent/turn_liveness.py index 538ca2a583..7b0562815d 100644 --- a/agent/turn_liveness.py +++ b/agent/turn_liveness.py @@ -86,8 +86,8 @@ def resolve_turn_liveness_settings( class TurnLivenessWatchdog: - """Sampled-idle watchdog bound to one conversation turn (polls on the - shared periodic scheduler thread). + """Sampled-idle watchdog bound to one conversation turn (via the shared + periodic scheduler; timer thread orders, body runs on its own worker). ``activity_lock`` must be the SAME lock ``AIAgent._touch_activity`` stamps the activity clock with; run_agent owns the lease state and callbacks. @@ -110,7 +110,7 @@ class TurnLivenessWatchdog: self._deactivate_turn = deactivate_turn def schedule(self): - """Start polling on the shared periodic scheduler thread; returns the cancel handle. + """Start polling via the shared periodic scheduler; returns the cancel handle. Scheduled at turn entry, after the turn-active flag and activity clock are stamped.""" from agent.periodic_scheduler import schedule diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index fbc9b0731e..df56ca9a04 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -10,6 +10,7 @@ mutate ``agent`` / ``messages`` / ``api_messages`` in place. Logger name stays from __future__ import annotations import logging +import math import re import time from dataclasses import dataclass @@ -60,6 +61,17 @@ def _image_error_max_dimension(error: Exception) -> Optional[int]: except Exception: pass text = " ".join(parts).lower() + # OpenAI Codex Responses reports a tile-patch budget (ceil(w/32)×ceil(h/32)) + # instead of a pixel ceiling. A square image is the worst case for the budget, + # so a per-side cap of isqrt(limit)*32 px keeps isqrt(limit)² ≤ limit — for the + # 30000-patch ceiling that is 5536 px. Without this the caller falls back to + # 8000 px and a 6000 px image that already exceeds the budget is skipped (#106337). + if "patches after processing" in text: + match = re.search(r"exceeding the limit of\s*(\d{2,7})", text) + if not match: + return None + max_dimension = math.isqrt(int(match.group(1))) * 32 + return max_dimension if 512 <= max_dimension <= 8000 else None if "image" not in text or "dimension" not in text or "max allowed size" not in text: return None match = re.search(r"max allowed size(?:\s+for [^:]+)?:\s*(\d{3,5})\s*pixels?", text) diff --git a/agent/turn_truncation.py b/agent/turn_truncation.py index 6c81d581d3..335cb1c3c7 100644 --- a/agent/turn_truncation.py +++ b/agent/turn_truncation.py @@ -20,6 +20,7 @@ from agent.message_sanitization import close_interrupted_tool_sequence from agent.repetition_guard import is_repetition_dominated from agent.turn_api_call import stop_thinking_spinner from agent.turn_retry_state import TurnRetryState +from agent.usage_pricing import normalize_usage from hermes_constants import PARTIAL_STREAM_STUB_ID logger = logging.getLogger("agent.conversation_loop") @@ -28,6 +29,14 @@ _CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_message _THINK_TAG_RE = re.compile(r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', re.IGNORECASE) _TRUNCATED_FINAL = "Response truncated due to output length limit" _FIRST_TRUNCATED_FINAL = "First response truncated due to output length limit" +# #106260: a stream that died on a context-overflow error after partial delivery must not seed a +# continuation — the transcript already cannot fit, and appending the partial stub grows every +# later request into the same overflow. End the turn via the recovery contract instead. +_CONTEXT_OVERFLOW_PARTIAL_FINAL = ( + "The request no longer fits the model's context window, so the partial " + "response was not continued. Continue in a fresh session (/new; gateway " + "chats are reset automatically)." +) _THINKING_EXHAUSTED = ( "💭 Reasoning exhausted the output token budget — no visible response was produced.", @@ -52,6 +61,27 @@ _CEILING_NO_TEXT = ( "continuation attempt — its reasoning consumed the entire budget each time.\n\nTo fix this:\n" "→ Lower reasoning effort: `/reasoning low` or `/reasoning none`\n→ Or raise max_tokens for this model" ) +# Below this many free tokens the prompt itself filled the window: a continuation nudge + +# fragment costs ~100 tokens per attempt, so retrying only shrinks the room (#106120). +_MIN_CONTINUATION_HEADROOM = 512 +_WINDOW_FILLED = ( + "⚠️ **Context window full.** The prompt used {prompt:,} of this model's {ctx:,}-token " + "context window, leaving no room to answer in. This is a context-window limit, not an " + "output-length limit.\n\nTo fix this:\n→ Compress the conversation with `/compress` or start " + "a new session\n→ Or raise the model's context window (e.g. Ollama `num_ctx`)" +) + + +def _prompt_filled_window(agent: Any, response: Any) -> Optional[tuple[int, int]]: + """``(prompt_tokens, context_length)`` when this response's usage shows the prompt left + less than ``_MIN_CONTINUATION_HEADROOM`` in the window compression resolves for the + model; ``None`` (keep continuing) when either number is unknown.""" + ctx = int(getattr(getattr(agent, "context_compressor", None), "context_length", 0) or 0) + usage = getattr(response, "usage", None) + if not (ctx and usage): + return None + prompt = normalize_usage(usage, provider=agent.provider, api_mode=agent.api_mode).prompt_tokens + return (prompt, ctx) if prompt and ctx - prompt < _MIN_CONTINUATION_HEADROOM else None def normalize_response_for_agent(agent: Any, response: Any) -> Any: @@ -65,11 +95,12 @@ def normalize_response_for_agent(agent: Any, response: Any) -> Any: def partial_result( messages: List[Dict[str, Any]], api_call_count: int, final_response: str, - error: Optional[str] = None, *, failed: bool = False, + error: Optional[str] = None, *, failed: bool = False, compression_exhausted: bool = False, ) -> Dict[str, Any]: """Typed incomplete-turn result (``partial`` unless ``failed``); ``error`` defaults to - ``final_response``.""" - return { + ``final_response``. ``compression_exhausted`` carries the #98722 typed bit the gateway + consumes to reset/move future input to a clean session (see run_turn.py).""" + result = { "final_response": final_response, "messages": messages, "api_calls": api_call_count, @@ -77,6 +108,9 @@ def partial_result( ("failed" if failed else "partial"): True, "error": final_response if error is None else error, } + if compression_exhausted: + result["compression_exhausted"] = True + return result @dataclass @@ -113,6 +147,7 @@ class _Trunc(TruncationVerdict): current_turn_user_idx: Any action: str = "fallthrough" result: Optional[Dict[str, Any]] = None + window_filled: Optional[tuple[int, int]] = None # (prompt_tokens, context_length) def done(self, action: str, result: Optional[Dict[str, Any]] = None) -> TruncationVerdict: self.action, self.result = action, result @@ -121,16 +156,20 @@ class _Trunc(TruncationVerdict): def end_turn( self, final_response: str, error: Optional[str] = None, *, result_messages: Optional[List[Dict[str, Any]]] = None, cleanup: bool = True, - failed: bool = False, + failed: bool = False, compression_exhausted: bool = False, ) -> TruncationVerdict: - """Persist and end the turn as partial (or ``failed``).""" + """Persist and end the turn as partial (or ``failed``). + + ``compression_exhausted`` forwards the #98722 typed bit so the gateway can + move future input off a bloated session (run_turn.py consumes it). + """ agent = self.agent if cleanup: agent._cleanup_task_resources(self.effective_task_id) agent._persist_session(self.messages, self.conversation_history) return self.done("return", partial_result( self.messages if result_messages is None else result_messages, self.api_call_count, - final_response, error, failed=failed, + final_response, error, failed=failed, compression_exhausted=compression_exhausted, )) @property @@ -214,7 +253,8 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) - append_message(messages, interim_msg) st.truncated_response_parts.append(_interim_content) - if n < 4: + filled = st.window_filled + if n < 4 and filled is None: _dropped_tools = getattr(st.response, "_dropped_tool_names", None) if st.is_stub and _dropped_tools: agent._vprint( @@ -237,6 +277,8 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) - # The one-shot reasoning-off override must not leak into the next turn. agent._ephemeral_reasoning_off = False agent._vprint( + f"{agent.log_prefix}⚠️ Not continuing — each attempt would only grow the prompt." + if filled is not None else f"{agent.log_prefix}⚠️ Response still truncated after {n} continuation attempts — " + ("keeping the partial response received so far." if partial_response else "no visible text was produced."), @@ -256,6 +298,12 @@ def _continue_text(st: _Trunc, _retry: TurnRetryState, assistant_message: Any) - "role": "assistant", "content": partial_response, "finish_reason": "length" }) agent._session_messages = messages + if filled is not None: + notice = _WINDOW_FILLED.format(prompt=filled[0], ctx=filled[1]) + return st.end_turn( + f"{partial_response}\n\n{notice}" if partial_response else notice, + f"Prompt used {filled[0]} of {filled[1]} context tokens; no room to answer", + ) return st.end_turn( partial_response or _CEILING_NO_TEXT, "Response remained truncated after 4 continuation attempts", @@ -319,13 +367,44 @@ def recover_from_truncation( truncated_tool_call_retries=truncated_tool_call_retries, retry_count=retry_count, compression_attempts=compression_attempts, ) + st.window_filled = _prompt_filled_window(agent, response) agent._vprint( f"{agent.log_prefix}⚠️ Response truncated — stream ended before completion" if st.is_stub else + f"{agent.log_prefix}⚠️ Response truncated (finish_reason='length') - the prompt filled the " + f"context window ({st.window_filled[0]:,}/{st.window_filled[1]:,} tokens)" + if st.window_filled else f"{agent.log_prefix}⚠️ Response truncated (finish_reason='length') - model hit max output tokens", force=True, ) + # #106260: a context-overflow error after partial delivery must not seed a + # continuation. _partial_stream_stub marks such stubs _overflow_terminal and + # leaves content empty; continuing would only re-send a larger request into + # the same overflow. The stub path never raises, so this class never reached + # recover_from_overflow's compress-and-retry on main either — ending the turn + # replaces a growth loop, not a compression attempt. + if getattr(st.response, "_overflow_terminal", False): + agent._flush_status_buffer() + agent._vprint( + f"{agent.log_prefix}⚠️ Stream ended on a context-overflow error after " + "partial delivery — not continuing (the request no longer fits the model's " + "context window).", + force=True, + ) + # Prior tool batches can leave a tool-result tail; this path never reaches + # finalize_turn (same as the truncated-tool-call terminal above). + close_interrupted_tool_sequence(st.messages, _CONTEXT_OVERFLOW_PARTIAL_FINAL) + # Carry the #98722 typed exhaustion bit so the gateway resets/moves future + # input to a clean session instead of leaving this bloated one authoritative + # for the next turn. + return st.end_turn( + _CONTEXT_OVERFLOW_PARTIAL_FINAL, + error=_CONTEXT_OVERFLOW_PARTIAL_FINAL, + failed=True, + compression_exhausted=True, + ) + _trunc_msg = normalize_response_for_agent(agent, response) _trunc_content = getattr(_trunc_msg, "content", None) if _trunc_msg else None _trunc_has_tool_calls = bool(getattr(_trunc_msg, "tool_calls", None)) if _trunc_msg else False diff --git a/agent/verification_evidence.py b/agent/verification_evidence.py index 468deb2a9e..69e7f2c1aa 100644 --- a/agent/verification_evidence.py +++ b/agent/verification_evidence.py @@ -111,6 +111,14 @@ def _db_path() -> Path: return get_hermes_home() / "verification_evidence.db" +def _ledger_enabled() -> bool: + """The ledger exists only to feed verify-on-stop; when that guard is off nothing may + record, read, or even create the database (an unconsumed ledger is pure disk churn).""" + from agent.verification_stop import verify_on_stop_enabled + + return verify_on_stop_enabled() + + def _connect() -> sqlite3.Connection: from hermes_state_wal import apply_wal_with_fallback @@ -451,6 +459,8 @@ def record_terminal_result( *, command: str, cwd: str | Path | None, session_id: str | None, exit_code: int, output: str = "" ) -> Optional[dict[str, Any]]: """Record a foreground terminal result when it is verification evidence.""" + if not _ledger_enabled(): + return None evidence = classify_verification_command(command, cwd=cwd, session_id=session_id, exit_code=exit_code, output=output) return None if evidence is None else _insert_evidence(evidence) @@ -465,6 +475,8 @@ def record_verify_run( canonical test command would. ``root`` is re-resolved through project facts so it matches what :func:`verification_status` derives later. """ + if not _ledger_enabled(): + return None resolved = str(Path(root).resolve()) return _insert_evidence(VerificationEvidence( command=command, canonical_command="hermes verify", kind="verify", @@ -511,6 +523,8 @@ def mark_workspace_edited( *, session_id: str | None, cwd: str | Path | None, paths: list[str] | tuple[str, ...] | None = None ) -> Optional[dict[str, Any]]: """Mark verification evidence stale after a successful file edit.""" + if not _ledger_enabled(): + return None facts = _project_facts(cwd) if not facts: return None @@ -547,6 +561,8 @@ def verification_status(*, session_id: str | None, cwd: str | Path | None) -> di Evidence recorded before the latest edit is reported as ``stale``. """ + if not _ledger_enabled(): + return {"status": "disabled", "evidence": None} facts = _project_facts(cwd) if not facts: return {"status": "not_applicable", "evidence": None} diff --git a/apps/desktop/DESIGN.md b/apps/desktop/DESIGN.md index df16b73f0f..8267f59e04 100644 --- a/apps/desktop/DESIGN.md +++ b/apps/desktop/DESIGN.md @@ -86,6 +86,15 @@ Menus and popovers use their own shared `shadow-md` + dashed targets and local blur. These are semantic surface classes, not licenses for call-site shadow or border inventions. +## Window glass + +Glass defaults to **29% Tint, Sidebar only** in both light and dark appearances. +Fade defaults to zero so the content column and text stay opaque. Native frost +keeps its platform/appearance defaults. Explicitly saved settings take precedence; +changing defaults must not overwrite a user's existing choices. The shared +`apps/shared/src/translucency.ts` resolver owns these defaults for both the +renderer and Electron's first window paint. + ## Stroke & color tokens | Token | Use | @@ -271,6 +280,10 @@ Sizes: `default`, `xs`, `overlay` (titlebar glyph counts). ## Motion +- Visible windows keep animating when another app takes focus. Hidden/minimized + windows and inactive panes may pause; background polling stays focus-gated. +- Animated integer counts reuse `AnimatedInt` in `src/components/ui/diff-count.tsx`. + Its spring updates the DOM directly without per-frame React renders. - Quick, functional transitions (~100ms on controls). Respect `prefers-reduced-motion` for anything beyond a fade. - Choreographed exits (e.g. onboarding's "matrix" fade-down) stagger per-element diff --git a/apps/desktop/e2e/glyph-spinner.spec.ts b/apps/desktop/e2e/glyph-spinner.spec.ts index 8d2e97f552..c7228eafd6 100644 --- a/apps/desktop/e2e/glyph-spinner.spec.ts +++ b/apps/desktop/e2e/glyph-spinner.spec.ts @@ -173,11 +173,9 @@ test.describe('GlyphSpinner (compositor animation)', () => { expect(parked.playState).toBe('paused') expect(parked.willChange).toBe('auto') - // 2. The global gate: window blur / minimize / document-hidden, which - // main.tsx drives by arming this attribute on the root. The strip must - // be named in that rule, or every spinner keeps animating behind an - // inactive window — the CPU burn the original ticker's pause - // controller existed to avoid. + // 2. The global gate: window minimize / document-hidden, which + // main.tsx drives by arming this attribute on the root. Visible windows + // keep animating even when another app has focus. const globallyPaused = await page.evaluate(strip => { const root = document.documentElement const had = root.hasAttribute('data-renderer-animations-paused') diff --git a/apps/desktop/electron/hud-overlay.test.ts b/apps/desktop/electron/hud-overlay.test.ts index 351a2a741a..b08f04412d 100644 --- a/apps/desktop/electron/hud-overlay.test.ts +++ b/apps/desktop/electron/hud-overlay.test.ts @@ -4,7 +4,7 @@ import { test } from 'vitest' import { applyHudElectronOverlay } from './hud-overlay' -test('macOS uses the floating panel level and all-spaces visibility', () => { +test('macOS uses the floating window level and all-spaces visibility', () => { const calls: string[] = [] const win = { diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index e909613c05..3ce8728b56 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -353,6 +353,7 @@ import { import { missingRendererAssets } from './renderer-bundle' import { loadRendererLoadErrorPage } from './renderer-load-error-page' import { attachRendererConsoleCapture, formatRendererBoundaryReport } from './renderer-log' +import { fetchRosterSourceData } from './roster-source-fetch' import { classifyStoredSecret, readSecretStoragePolicy, @@ -10286,6 +10287,7 @@ async function bootstrapSshConnectionInner(profile, sshConfig, reuseToken, sourc const lifecycle = platform.os === 'Windows' ? connectWindowsRemote : remoteLifecycle.connect result = await lifecycle({ ssh, + platform, profile: resolveRemoteSshDashboardProfile(sshConfig.remoteProfile, profile), remoteHermesPath: sshConfig.remoteHermesPath || '', ownershipId: sshOwnershipKey(profile), @@ -13821,13 +13823,12 @@ function spawnHudWindow(sessionId, profile) { minimizable: false, maximizable: false, fullscreenable: false, - // Same rationale as the pet overlay: on Windows/Linux keep the helper out - // of the taskbar/alt-tab list; on macOS use an NSPanel so the frameless - // window never becomes the app's cmd-tab anchor. + // Keep the interactive macOS HUD as an ordinary NSWindow. NSPanel defaults + // hidesOnDeactivate to true, which removes the HUD while the user works in + // another app; the floating/all-spaces setup below supplies overlay behavior. skipTaskbar: !IS_MAC, hasShadow: false, alwaysOnTop: true, - type: IS_MAC ? 'panel' : undefined, // Clips the vibrancy layer to the HUD's silhouette rather than a hard // rectangle — the frost stops where the window's corners do. roundedCorners: true, @@ -15348,11 +15349,13 @@ async function enumerateRegistryAgentSources(registry = readDesktopConnectionsRe ) ) - const body: any = await getJsonForBackend(descriptor, '/api/profiles', { timeoutMs: 8_000 }) + const { body, installId } = await fetchRosterSourceData( + () => getJsonForBackend(descriptor, '/api/profiles', { timeoutMs: 8_000 }), + () => probeConnectionInstallId(connection.id, descriptor) + ) - // Cached with a TTL, so the 5s roster poll usually pays zero extra - // requests for the backend-identity probe. - const installId = await probeConnectionInstallId(connection.id, descriptor) + // The install-id probe is TTL-cached, so the 5s roster poll usually + // pays zero extra requests; on a miss it runs beside /api/profiles. const profiles = Array.isArray(body?.profiles) ? body.profiles.map(p => String(p?.name || '').trim()).filter(Boolean) diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index b75487eb39..583f4edcab 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -999,7 +999,14 @@ test('connect() spawns fresh when there is no lockfile, adopts the served token' [/cat .*\.log/, 'HERMES_DASHBOARD_READY port=51999\n'] ]) - const result = await connect(connectDeps(ssh, { adoptServedToken: async () => 'the-served-token' })) + const result = await connect( + connectDeps(ssh, { + adoptServedToken: async () => 'the-served-token', + platform: { os: 'Linux', arch: 'x86_64' } + }) + ) + + assert.equal(ssh.calls.filter(command => command === 'uname -s; uname -m').length, 0) assert.equal(result.reused, false) assert.equal(result.remotePort, 51999) assert.equal(result.localPort, 50001) diff --git a/apps/desktop/electron/remote-lifecycle.ts b/apps/desktop/electron/remote-lifecycle.ts index 112fdddda3..a9b2d9b4fd 100644 --- a/apps/desktop/electron/remote-lifecycle.ts +++ b/apps/desktop/electron/remote-lifecycle.ts @@ -1396,7 +1396,7 @@ async function connect(deps) { const log = msg => rememberLog(`[ssh-lifecycle] ${msg}`) assertBootstrapNotSuperseded(signal) - const platform = await probeRemotePlatform(ssh) + const platform = deps.platform ?? (await probeRemotePlatform(ssh)) log(`remote platform ${platform.os}/${platform.arch}`) const hermesHome = await probeRemoteHermesHome(ssh) await assertRemoteInstallUpdateClear(ssh, hermesHome) diff --git a/apps/desktop/electron/roster-source-fetch.test.ts b/apps/desktop/electron/roster-source-fetch.test.ts new file mode 100644 index 0000000000..a071fc6188 --- /dev/null +++ b/apps/desktop/electron/roster-source-fetch.test.ts @@ -0,0 +1,61 @@ +import assert from 'node:assert/strict' + +import { test } from 'vitest' + +import { fetchRosterSourceData } from './roster-source-fetch' + +test('roster source starts profile and install-id reads together and preserves both results', async () => { + let releaseProfiles!: (value: { profiles: Array<{ name: string }> }) => void + let releaseInstallId!: (value: string | undefined) => void + + const profiles = new Promise<{ profiles: Array<{ name: string }> }>(resolve => { + releaseProfiles = resolve + }) + + const installId = new Promise(resolve => { + releaseInstallId = resolve + }) + + const started: string[] = [] + + const pending = fetchRosterSourceData( + () => { + started.push('profiles') + + return profiles + }, + () => { + started.push('install-id') + + return installId + } + ) + + assert.deepEqual(started, ['profiles', 'install-id']) + releaseInstallId('install-1') + releaseProfiles({ profiles: [{ name: 'default' }] }) + assert.deepEqual(await pending, { + body: { profiles: [{ name: 'default' }] }, + installId: 'install-1' + }) +}) + +test('roster source preserves a profile-read failure after starting the install-id read', async () => { + const profilesError = new Error('profiles unavailable') + let installIdStarted = false + + await assert.rejects( + fetchRosterSourceData( + async () => { + throw profilesError + }, + async () => { + installIdStarted = true + + return undefined + } + ), + profilesError + ) + assert.equal(installIdStarted, true) +}) diff --git a/apps/desktop/electron/roster-source-fetch.ts b/apps/desktop/electron/roster-source-fetch.ts new file mode 100644 index 0000000000..bbe77e535d --- /dev/null +++ b/apps/desktop/electron/roster-source-fetch.ts @@ -0,0 +1,8 @@ +export async function fetchRosterSourceData( + fetchProfiles: () => Promise, + fetchInstallId: () => Promise +): Promise<{ body: T; installId: string | undefined }> { + const [body, installId] = await Promise.all([fetchProfiles(), fetchInstallId()]) + + return { body, installId } +} diff --git a/apps/desktop/electron/translucency.test.ts b/apps/desktop/electron/translucency.test.ts index e9c9ed0c61..d525c7c864 100644 --- a/apps/desktop/electron/translucency.test.ts +++ b/apps/desktop/electron/translucency.test.ts @@ -603,9 +603,7 @@ describe('what an update actually changes natively', () => { }) it('leaves a window alone when glass is selected but off', () => { - // The light default carries one point of fade. Someone who dragged the - // tint to zero asked for an opaque window, and that point must not follow - // them there — off has to mean exactly 1, not 0.9999. + // A saved fade must not follow the tint to zero: off means opaque. expect(windowOpacityFor({ ...glass(0), fade: 1 })).toBe(1) expect(windowOpacityFor({ ...glass(0), fade: 40 })).toBe(1) }) @@ -615,12 +613,7 @@ describe('what an update actually changes natively', () => { }) }) -/** - * The shipped defaults, per platform. These are the numbers a fresh profile - * gets before anyone opens Settings, so they are the ones most people will - * ever see — and they differ by platform because the lever means different - * things behind macOS vibrancy and Windows acrylic. - */ +/** Fresh profiles share the sidebar treatment, with native frost per platform. */ describe('the defaults a fresh profile lands on', () => { const mac = (appearance: 'dark' | 'light') => defaultTranslucencyValues(appearance, false) const win = (appearance: 'dark' | 'light') => defaultTranslucencyValues(appearance, true) @@ -641,21 +634,16 @@ describe('the defaults a fresh profile lands on', () => { expect(defaultTranslucencyState('dark', false, false).mode).toBe('clear') }) - it('tints light more heavily than dark, on both platforms', () => { - // A dark field already separates from what is behind it; a bright one - // needs real thinning before the desktop reads as a layer underneath. - expect(mac('light').intensity).toBeGreaterThan(mac('dark').intensity) - expect(win('light').intensity).toBeGreaterThan(win('dark').intensity) + it('keeps tint consistent across appearances and platforms', () => { + for (const values of [mac('light'), mac('dark'), win('light'), win('dark')]) { + expect(values.intensity).toBe(mac('light').intensity) + } }) - it('asks far less of Windows, which composites its own tint in DWM', () => { - expect(win('light').intensity).toBeLessThan(mac('light').intensity) - expect(win('dark').intensity).toBeLessThan(mac('dark').intensity) - }) - - it('never fades a Windows window — setOpacity dims the composited backdrop', () => { - expect(win('light').fade).toBe(0) - expect(win('dark').fade).toBe(0) + it('keeps the content column opaque at the native level', () => { + for (const values of [mac('light'), mac('dark'), win('light'), win('dark')]) { + expect(windowOpacityFor({ ...values, mode: 'glass' })).toBe(1) + } }) it('defaults each platform onto a frost that platform can actually render', () => { @@ -665,9 +653,9 @@ describe('the defaults a fresh profile lands on', () => { } }) - it('opens the whole window, not just the sidebar rail', () => { + it('uses the normalized scope default for every appearance and platform', () => { for (const values of [mac('light'), mac('dark'), win('light'), win('dark')]) { - expect(values.scope).toBe('window') + expect(values.scope).toBe(normalizeScope(undefined)) } }) }) @@ -680,9 +668,14 @@ describe('the defaults a fresh profile lands on', () => { describe('resolving the book for the painted appearance', () => { const empty = normalizeBook(null, true) - it('falls all the way through to the platform default', () => { - expect(resolveTranslucency(empty, 'dark', false).intensity).toBe(defaultTranslucencyValues('dark', false).intensity) - expect(resolveTranslucency(empty, 'dark', true).intensity).toBe(defaultTranslucencyValues('dark', true).intensity) + it('agrees with the native first-window defaults in either appearance', () => { + for (const appearance of ['light', 'dark'] as const) { + for (const isWindows of [false, true]) { + expect(resolveTranslucency(empty, appearance, isWindows)).toEqual( + defaultTranslucencyState(appearance, true, isWindows) + ) + } + } }) it('scopes an edit to the appearance it was made in', () => { @@ -692,14 +685,17 @@ describe('resolving the book for the painted appearance', () => { expect(resolveTranslucency(book, 'dark', false).intensity).toBe(defaultTranslucencyValues('dark', false).intensity) }) - it('carries a v1 state into BOTH appearances via base', () => { - // Someone who tuned a window before appearances were split keeps exactly - // what was on screen, in either appearance, until they edit one of them. - const migrated = normalizeBook({ intensity: 40, mode: 'glass' }, true) + it('preserves a saved whole-window treatment in both appearances', () => { + const saved = { intensity: 40, scope: 'window', mode: 'glass' } as const + const migrated = normalizeBook(saved, true) - expect(migrated.base.intensity).toBe(40) - expect(resolveTranslucency(migrated, 'light', false).intensity).toBe(40) - expect(resolveTranslucency(migrated, 'dark', false).intensity).toBe(40) + expect(migrated.base).toEqual({ intensity: saved.intensity, scope: saved.scope }) + + for (const appearance of ['light', 'dark'] as const) { + for (const isWindows of [false, true]) { + expect(resolveTranslucency(migrated, appearance, isWindows)).toMatchObject(saved) + } + } }) it('lets an appearance override base without disturbing the other', () => { diff --git a/apps/desktop/electron/wsl-path-bridge.test.ts b/apps/desktop/electron/wsl-path-bridge.test.ts index 52073af3c3..04ebf57267 100644 --- a/apps/desktop/electron/wsl-path-bridge.test.ts +++ b/apps/desktop/electron/wsl-path-bridge.test.ts @@ -39,13 +39,22 @@ test('parseDefaultDistro strips the default-marker and blank lines', () => { // ── wslPosixToWindowsAccessible ────────────────────────────────────── -test('wslPosixToWindowsAccessible maps a drvfs mount to its Windows drive', () => { - assert.equal(wslPosixToWindowsAccessible('/mnt/c/Users/alex', 'Ubuntu'), 'C:\\Users\\alex') - assert.equal(wslPosixToWindowsAccessible('/mnt/d', 'Ubuntu'), 'D:\\') -}) +test('wslPosixToWindowsAccessible resolves a distro only for paths that need a UNC share', () => { + let distroProbes = 0 -test('wslPosixToWindowsAccessible maps an in-distro POSIX path to a UNC share', () => { - assert.equal(wslPosixToWindowsAccessible('/home/alex/proj', 'Ubuntu'), '\\\\wsl.localhost\\Ubuntu\\home\\alex\\proj') + const resolveDistro = () => { + distroProbes += 1 + + return 'Ubuntu' + } + + assert.equal(wslPosixToWindowsAccessible('/mnt/c/Users/alex', undefined, resolveDistro), 'C:\\Users\\alex') + assert.equal(distroProbes, 0) + assert.equal( + wslPosixToWindowsAccessible('/home/alex/proj', undefined, resolveDistro), + '\\\\wsl.localhost\\Ubuntu\\home\\alex\\proj' + ) + assert.equal(distroProbes, 1) }) test('wslPosixToWindowsAccessible leaves non-absolute / already-Windows paths alone', () => { diff --git a/apps/desktop/electron/wsl-path-bridge.ts b/apps/desktop/electron/wsl-path-bridge.ts index feff32a2f7..cab863f742 100644 --- a/apps/desktop/electron/wsl-path-bridge.ts +++ b/apps/desktop/electron/wsl-path-bridge.ts @@ -133,7 +133,11 @@ function wslUncBase(distro: string): string { * (drvfs mount), any other absolute POSIX path → `\\wsl.localhost\\...`. * Non-absolute or already-Windows paths pass through. */ -export function wslPosixToWindowsAccessible(posixPath: string, distro: string = resolveDefaultWslDistro()): string { +export function wslPosixToWindowsAccessible( + posixPath: string, + distro?: string, + resolveDistro: () => string = resolveDefaultWslDistro +): string { const value = String(posixPath || '').trim() const normalized = value.replace(/\\/g, '/') @@ -151,7 +155,7 @@ export function wslPosixToWindowsAccessible(posixPath: string, distro: string = const relative = normalized.replace(/^\/+/, '').replace(/\//g, '\\') - return `${wslUncBase(distro)}\\${relative}` + return `${wslUncBase(distro ?? resolveDistro())}\\${relative}` } /** Native folder dialog `defaultPath`: open a WSL cwd in the Windows picker. */ @@ -174,7 +178,7 @@ export function resolvePickerDefaultPath( const value = String(defaultPath).trim() return value.startsWith('/') && !WIN_DRIVE_RE.test(value) - ? wslPosixToWindowsAccessible(value, distro ?? resolveDefaultWslDistro()) + ? wslPosixToWindowsAccessible(value, distro) : defaultPath } @@ -191,6 +195,6 @@ export function resolveLocalReadPath(dirPath: string, distro?: string, profile?: } return IS_WINDOWS && value.startsWith('/') && !WIN_DRIVE_RE.test(value) - ? wslPosixToWindowsAccessible(value, distro ?? resolveDefaultWslDistro()) + ? wslPosixToWindowsAccessible(value, distro) : value } diff --git a/apps/desktop/src/app/chat/composer/status-stack/index.tsx b/apps/desktop/src/app/chat/composer/status-stack/index.tsx index abf9727877..b1e74267dc 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/index.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/index.tsx @@ -8,6 +8,7 @@ import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/u import { BillingBanner } from '@/components/billing-banner' import { composerDockCard } from '@/components/chat/composer-dock' import { StatusSection } from '@/components/chat/status-section' +import { usePaneVisible } from '@/components/pane-shell/pane-visibility' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { GlyphSpinner } from '@/components/ui/glyph-spinner' @@ -141,15 +142,26 @@ export function ComposerStatusStack({ onSubmit, queue, sessionId }: ComposerStat // dead `localhost:5174` chips stick around. On-disk file previews are kept. const visiblePreviews = previews.filter(item => hasRunningBackground || !isLocalhostPreview(item.target)) + // Keep-alive keeps every ever-active tab mounted, so without this gate each + // background tile's safety-net poll fires every 5s — N sessions means N + // gateway round-trips plus shared-map churn forever. Hidden tabs skip the + // poll (event-driven refreshes in use-message-stream still land through the + // store) and resume it on reveal via `paneVisible` in the dep array. + const paneVisible = usePaneVisible() + useEffect(() => { - if (!sessionId || !hasRunningBackground) { + if (!sessionId || !hasRunningBackground || !paneVisible) { return } - const timer = setInterval(() => void refreshBackgroundProcesses(sessionId), BACKGROUND_POLL_MS) + const timer = setInterval(() => { + if (document.visibilityState === 'visible') { + void refreshBackgroundProcesses(sessionId) + } + }, BACKGROUND_POLL_MS) return () => clearInterval(timer) - }, [hasRunningBackground, sessionId]) + }, [hasRunningBackground, sessionId, paneVisible]) const openAgents = () => navigate(AGENTS_ROUTE) diff --git a/apps/desktop/src/app/chat/composer/status-stack/polling-guard.test.tsx b/apps/desktop/src/app/chat/composer/status-stack/polling-guard.test.tsx index 4df2aea889..771e3849b3 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/polling-guard.test.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/polling-guard.test.tsx @@ -2,8 +2,9 @@ import { cleanup, render } from '@testing-library/react' import { MemoryRouter } from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility' import { I18nProvider } from '@/i18n' -import { resetBackgroundPollingGuard } from '@/store/composer-status' +import { $backgroundStatusBySession, resetBackgroundPollingGuard } from '@/store/composer-status' import { $gateway } from '@/store/gateway' import { ComposerStatusStack } from './index' @@ -76,3 +77,82 @@ describe('ComposerStatusStack dead-runtime remount', () => { second.unmount() }) }) + +// #73287: keep-alive keeps every ever-active tab mounted, so each background +// tile's 5s safety-net poll used to fire even while its tab was hidden — N +// sessions meant N gateway round-trips plus shared-map churn forever. +describe('ComposerStatusStack hidden-pane poll', () => { + const SID_BG = 'sess-bg-poll' + + const runningList = vi.fn(async (method: string) => + method === 'process.list' + ? { processes: [{ command: 'dev server', session_id: 'bg1', status: 'running' }] } + : {} + ) + const processListCalls = () => runningList.mock.calls.filter(([method]) => method === 'process.list').length + + function renderStackBg(visible: boolean) { + return render( + + + + + + + + ) + } + + beforeEach(() => { + vi.useFakeTimers() + // jsdom defaults to 'prerender', which the in-tick document check skips. + vi.spyOn(document, 'visibilityState', 'get').mockReturnValue('visible') + runningList.mockClear() + $gateway.set({ request: runningList } as never) + }) + + afterEach(() => { + cleanup() + vi.useRealTimers() + vi.restoreAllMocks() + $gateway.set(null as never) + $backgroundStatusBySession.set({}) + resetBackgroundPollingGuard() + }) + + it('hidden tile skips the 5s poll; visible tile polls; reveal resumes', async () => { + const hidden = renderStackBg(false) + await vi.advanceTimersByTimeAsync(0) + await vi.advanceTimersByTimeAsync(15_000) + // Mount seed only — the interval never armed while hidden. + expect(processListCalls()).toBe(1) + hidden.unmount() + + const shown = renderStackBg(true) + await vi.advanceTimersByTimeAsync(0) + await vi.advanceTimersByTimeAsync(15_000) + // Mount seed + three 5s ticks. + expect(processListCalls()).toBe(1 + 1 + 3) + shown.unmount() + }) + + it('revealing a hidden tile arms the poll', async () => { + const view = renderStackBg(false) + await vi.advanceTimersByTimeAsync(0) + await vi.advanceTimersByTimeAsync(15_000) + expect(processListCalls()).toBe(1) + + view.rerender( + + + + + + + + ) + await vi.advanceTimersByTimeAsync(10_000) + expect(processListCalls()).toBeGreaterThan(1) + view.unmount() + }) +}) diff --git a/apps/desktop/src/app/chat/scroll-to-bottom-button.test.tsx b/apps/desktop/src/app/chat/scroll-to-bottom-button.test.tsx index 3d3d618c6d..00bc80ff95 100644 --- a/apps/desktop/src/app/chat/scroll-to-bottom-button.test.tsx +++ b/apps/desktop/src/app/chat/scroll-to-bottom-button.test.tsx @@ -3,7 +3,12 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { clearAllPrompts, setApprovalRequest } from '@/store/prompts' import { $activeSessionId } from '@/store/session' -import { onScrollToBottomRequest, resetThreadScroll, setThreadAtBottom } from '@/store/thread-scroll' +import { + onScrollToBottomRequest, + publishThreadMessagesBelow, + resetThreadScroll, + setThreadAtBottom +} from '@/store/thread-scroll' import { ScrollToBottomButton } from './scroll-to-bottom-button' @@ -28,11 +33,12 @@ describe('ScrollToBottomButton', () => { expect(screen.queryByRole('button')).toBeNull() }) - it('is a plain jump-to-bottom control when scrolled up with no approval', () => { + it('shows the messages below the viewport when scrolled up with no approval', () => { setThreadAtBottom(false) + publishThreadMessagesBelow(12, { paneVisible: true }) render() - expect(screen.getByRole('button', { name: 'Scroll to bottom' })).toBeTruthy() + expect(screen.getByRole('button', { name: 'Scroll to bottom · 12 messages' }).textContent).toBe('12 messages') expect(screen.queryByText('Approval needed')).toBeNull() }) diff --git a/apps/desktop/src/app/chat/scroll-to-bottom-button.tsx b/apps/desktop/src/app/chat/scroll-to-bottom-button.tsx index 85d01cb700..6141e35dc3 100644 --- a/apps/desktop/src/app/chat/scroll-to-bottom-button.tsx +++ b/apps/desktop/src/app/chat/scroll-to-bottom-button.tsx @@ -1,12 +1,14 @@ import { useStore } from '@nanostores/react' +import { useReducedMotion } from 'motion/react' import { useRef } from 'react' import { Codicon } from '@/components/ui/codicon' +import { AnimatedInt } from '@/components/ui/diff-count' import { useI18n } from '@/i18n' import { triggerHaptic } from '@/lib/haptics' import { cn } from '@/lib/utils' import { $approvalRequest } from '@/store/prompts' -import { $threadJumpButtonVisible, requestScrollToBottom } from '@/store/thread-scroll' +import { $threadJumpButtonVisible, $threadMessagesBelow, requestScrollToBottom } from '@/store/thread-scroll' /** * Floating "jump to bottom" control. Sits centered just above the composer, @@ -14,7 +16,8 @@ import { $threadJumpButtonVisible, requestScrollToBottom } from '@/store/thread- * the thread's bottom clearance uses (`--composer-measured-height`, which * covers the whole dock), so it never overlaps the queue / subagent * / background cards. Visible only while the user has scrolled meaningfully - * away from the bottom; clicking re-arms sticky-bottom and pins the viewport. + * away from the bottom, with an animated count of messages below the viewport. + * Clicking re-arms sticky-bottom and pins the viewport. * * When the turn is BLOCKED on an approval, this same control morphs into an * "Approval needed" pill — the only response surface is the inline Run/Reject @@ -31,6 +34,8 @@ import { $threadJumpButtonVisible, requestScrollToBottom } from '@/store/thread- export function ScrollToBottomButton({ sessionId }: { sessionId: string | null }) { const { t } = useI18n() const visible = useStore($threadJumpButtonVisible) + const count = useStore($threadMessagesBelow) + const reducedMotion = useReducedMotion() const request = useStore($approvalRequest) // Scrolled away while an approval is pending → the inline Run/Reject bar is // below the fold. Relabel so the user knows the session needs them, not just @@ -43,17 +48,20 @@ export function ScrollToBottomButton({ sessionId }: { sessionId: string | null } } const state = visible ? 'in' : hasShownRef.current ? 'out' : 'idle' - const label = approval ? t.assistant.approval.jumpToApproval : t.assistant.thread.scrollToBottom + const countLabel = t.sidebar.messageCount(count) + const [beforeCount, afterCount] = countLabel.split(String(count)) + + const label = approval ? t.assistant.approval.jumpToApproval : `${t.assistant.thread.scrollToBottom} · ${countLabel}` return ( ) } diff --git a/apps/desktop/src/app/chat/sidebar/projects/overview-row.test.tsx b/apps/desktop/src/app/chat/sidebar/projects/overview-row.test.tsx index 5467b7be86..baf7ef1fab 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/overview-row.test.tsx +++ b/apps/desktop/src/app/chat/sidebar/projects/overview-row.test.tsx @@ -17,7 +17,8 @@ vi.mock('@/i18n', () => ({ projects: { enter: (label: string) => `Enter ${label}`, reorder: (label: string) => `Reorder ${label}`, - toggle: (label: string, open: boolean) => `${open ? 'Show' : 'Hide'} ${label} sessions` + toggle: (label: string, open: boolean) => `${open ? 'Show' : 'Hide'} ${label} sessions`, + autoDiscovered: 'Auto-discovered' } } } @@ -92,4 +93,26 @@ describe('ProjectOverviewRow', () => { expect(container.querySelector('[data-sessions-project="p1"]')).toBeTruthy() }) + + it('explicit projects keep the folder-library glyph and a plain accessible name', () => { + const explicit = { id: 'p1', label: 'Explicit' } as unknown as SidebarProjectTree + + const { container } = render() + + expect(container.querySelector('.codicon-folder-library')).toBeTruthy() + expect(container.querySelector('.codicon-repo')).toBeNull() + expect(screen.getByRole('button', { name: 'Enter Explicit' })).toBeTruthy() + }) + + it('auto-discovered repos get the repo glyph, an "Auto-discovered" tooltip, and an accessible name that says so', () => { + const auto = { id: '/Users/dev/my-repo', label: 'my-repo', isAuto: true } as unknown as SidebarProjectTree + + const { container } = render() + + expect(container.querySelector('.codicon-repo')).toBeTruthy() + expect(container.querySelector('.codicon-folder-library')).toBeNull() + + const link = screen.getByRole('button', { name: 'Enter my-repo (Auto-discovered)' }) + expect(tipTrigger(link)).toBeTruthy() + }) }) diff --git a/apps/desktop/src/app/chat/sidebar/projects/overview-row.tsx b/apps/desktop/src/app/chat/sidebar/projects/overview-row.tsx index 8f92767b13..88577de1e5 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/overview-row.tsx +++ b/apps/desktop/src/app/chat/sidebar/projects/overview-row.tsx @@ -3,6 +3,7 @@ import { useRef } from 'react' import { type NewSessionSplitHandler, startNewSessionDrag } from '@/app/chat/new-session-drag' import { Codicon } from '@/components/ui/codicon' +import { Tip } from '@/components/ui/tooltip' import type { SessionInfo } from '@/hermes' import { useI18n } from '@/i18n' import { cn } from '@/lib/utils' @@ -27,7 +28,10 @@ import { WorkspaceAddButton } from './workspace-header' // A bare color dot (no icon) or an icon glyph — tinted by `color` when set, else // the lead's default tertiary. The glyph wrapper centers + caps size either way. -export function projectIcon({ color, icon, isNoProject }: SidebarProjectTree) { +// Auto-discovered repos (git lanes Desktop found by scanning disk, not rows in +// projects.db) get the `repo` glyph so a glance tells explicit projects +// (`folder-library`) apart from incidental disk/session findings. +export function projectIcon({ color, icon, isAuto, isNoProject }: SidebarProjectTree) { if (color && !icon) { return ( @@ -38,7 +42,10 @@ export function projectIcon({ color, icon, isNoProject }: SidebarProjectTree) { return ( - + ) } @@ -115,6 +122,22 @@ export function ProjectOverviewRow({ {projectIcon(project)} ) + const labelLink = ( + onEnter?.(project.id)} + > + {project.label} + + ) + const shell = ( onEnter?.(project.id)} - > - {project.label} - - } + label={project.isAuto ? {labelLink} : labelLink} lead={lead} // The label is grab surface too, not just the lead's grabber — same // listeners, minus the controls that keep their own gestures. A project diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx index cfea238630..694a352fad 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx @@ -9,7 +9,12 @@ import { useBackgroundSync } from './use-background-sync' const noop = () => undefined const requestGateway = async () => ({ sessions: [] }) -function render(activeGatewayProfile: string, activeConnectionId: string, refreshSessions: () => Promise) { +function render( + activeGatewayProfile: string, + activeConnectionId: string, + refreshSessions: () => Promise, + gatewayRequest = requestGateway +) { return renderHook( ({ connectionId, profile }: { connectionId: string; profile: string }) => { useBackgroundSync({ @@ -26,7 +31,7 @@ function render(activeGatewayProfile: string, activeConnectionId: string, refres refreshHermesConfig: noop, refreshMessagingSessions: noop, refreshSessions, - requestGateway + requestGateway: gatewayRequest }) }, { initialProps: { connectionId: activeConnectionId, profile: activeGatewayProfile } } @@ -47,6 +52,23 @@ describe('useBackgroundSync profile-scoped session refresh', () => { vi.useRealTimers() }) + it('coalesces change ticks while the live status request is pending', async () => { + $changeEventsAvailable.set(true) + let release!: (value: { sessions: [] }) => void + const pending = new Promise<{ sessions: [] }>(resolve => { release = resolve }) + const request = vi.fn(() => pending) + render('default', 'local', async () => undefined, request) + await act(async () => undefined) + + for (let tick = 1; tick <= 8; tick += 1) { + await act(async () => { $sessionsChangeTick.set(tick) }) + } + + expect(request).toHaveBeenCalledTimes(1) + await act(async () => { release({ sessions: [] }) }) + expect(request).toHaveBeenCalledTimes(2) + }) + it('refreshes the session list after the active gateway profile changes', async () => { const refreshSessions = vi.fn(async () => undefined) const hook = render('default', 'local', refreshSessions) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index 9bd5e47403..5363eeb149 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -575,7 +575,6 @@ export function useBackgroundSync({ }: BackgroundSyncParams): void { const changeEventsAvailable = useStore($changeEventsAvailable) const cronChangeTick = useStore($cronChangeTick) - const sessionsChangeTick = useStore($sessionsChangeTick) const activeTranscriptBusy = useStore($busy) const activeTranscriptRefreshPendingRef = useRef(null) // Tile reconcile state (#93942 slice 1): shared sequence guard + per-tile @@ -682,9 +681,16 @@ export function useBackgroundSync({ let cancelled = false let inFlight = false + let refreshPending = false const refreshLiveStatuses = async () => { + if (cancelled) { + return + } + if (inFlight) { + refreshPending = true + return } @@ -701,9 +707,16 @@ export function useBackgroundSync({ // still work as before; leave the current sidebar state untouched. } finally { inFlight = false + + if (refreshPending && !cancelled) { + refreshPending = false + void refreshLiveStatuses() + } } } + const unsubscribe = $sessionsChangeTick.listen(() => void refreshLiveStatuses()) + const dispose = visiblePoll( changeEventsAvailable ? LIVE_SESSION_STATUS_BACKSTOP_INTERVAL_MS : LIVE_SESSION_STATUS_POLL_INTERVAL_MS, () => void refreshLiveStatuses() @@ -713,11 +726,12 @@ export function useBackgroundSync({ return () => { cancelled = true + unsubscribe() dispose() } - // sessionsChangeTick: each sessions.changed broadcast re-seeds immediately - // via the effect re-run (already coalesced to 2s server-side). - }, [activeGatewayProfile, changeEventsAvailable, gatewayState, requestGateway, sessionsChangeTick]) + // Keep the in-flight guard alive across change ticks; a slow response must + // not create a new request (and invalidate the old result) on every tick. + }, [activeConnectionId, activeGatewayProfile, changeEventsAvailable, gatewayState, requestGateway]) // sessions.changed also means the *stored* list may have new rows (a cron // run's session, an inbound messaging turn creating a thread). The full list diff --git a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.test.tsx b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.test.tsx index 4f6d214ca1..d20e0cb89c 100644 --- a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.test.tsx +++ b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.test.tsx @@ -174,7 +174,7 @@ class FakeWebSocket { } const primaryConn = { - authMode: 'token' as const, + authMode: 'token' as 'oauth' | 'token', baseUrl: 'https://vps.example.com', connectionId: 'primary-vps', profile: 'default', @@ -182,6 +182,8 @@ const primaryConn = { wsUrl: 'wss://vps.example.com/api/ws?token=t' } +const remotePrimaryConn = { ...primaryConn, mode: 'remote' as const } + const coderConn = { authMode: 'token' as const, baseUrl: 'https://coder.example.com', @@ -1626,6 +1628,99 @@ describe('useGatewayBoot remote reconnect loop (real hook, fake socket)', () => expect($desktopBoot.get().error).toBeNull() }) + it('RETRY CONTRACT: a resolved remote whose first renderer gateway dial fails retries even when main still reports stale non-retryable ready progress', async () => { + // Renderer reload against a saved direct remote: Electron has already + // reported a successful backend.ready snapshot, so that stale progress + // cannot classify the renderer-owned WebSocket dial which follows it. + const desktop = fakeDesktop() + desktop.getConnection = vi.fn(async () => remotePrimaryConn) + desktop.getBootProgress = vi.fn(async () => ({ + error: null, + fakeMode: false, + message: 'Hermes is ready', + phase: 'backend.ready', + progress: 100, + retryable: false, + running: true, + timestamp: 1 + })) + ;(window as { hermesDesktop?: unknown }).hermesDesktop = desktop + FakeWebSocket.mode = 'fail' + + render() + await flushAsync() + + // getConnection resolved the saved REMOTE descriptor; only its first + // renderer-owned gateway.connect() failed. + expect(desktop.getConnection).toHaveBeenCalledTimes(1) + expect(FakeWebSocket.instances).toHaveLength(1) + expect($desktopBoot.get().error).toBeNull() + + // The same endpoint becomes reachable before the bounded retry fires. + FakeWebSocket.mode = 'open' + await advanceBackoff() + + expect(desktop.getConnection).toHaveBeenCalledTimes(2) + expect($gatewayState.get()).toBe('open') + expect($desktopBoot.get().error).toBeNull() + }) + + it('RETRY CONTRACT: an invalid remote WebSocket URL is not a dial failure — it stays terminal under the stale ready snapshot', async () => { + const desktop = fakeDesktop() + desktop.getConnection = vi.fn(async () => ({ ...remotePrimaryConn, wsUrl: 'not a WebSocket URL' })) + desktop.getGatewayWsUrl = vi.fn(async () => 'not a WebSocket URL') + desktop.getBootProgress = vi.fn(async () => ({ + error: null, + fakeMode: false, + message: 'Hermes is ready', + phase: 'backend.ready', + progress: 100, + retryable: false, + running: true, + timestamp: 1 + })) + ;(window as { hermesDesktop?: unknown }).hermesDesktop = desktop + FakeWebSocket.mode = 'fail' + + render() + await flushAsync() + + expect($desktopBoot.get().error).toBeTruthy() + expect(desktop.getConnection).toHaveBeenCalledTimes(1) + await advanceBackoff() + expect(desktop.getConnection).toHaveBeenCalledTimes(1) + }) + + it('RETRY CONTRACT: a post-connect failure stays terminal even when its socket closes before boot catches it — a closed socket after a good dial is not a dial failure', async () => { + const desktop = fakeDesktop() + desktop.getConnection = vi.fn(async () => remotePrimaryConn) + desktop.getBootProgress = vi.fn(async () => ({ + error: null, + fakeMode: false, + message: 'Hermes is ready', + phase: 'backend.ready', + progress: 100, + retryable: false, + running: true, + timestamp: 1 + })) + ;(window as { hermesDesktop?: unknown }).hermesDesktop = desktop + + const refreshHermesConfig = vi.fn(async () => { + FakeWebSocket.instances[0]?.drop() + throw new Error('post-connect initialization failed') + }) + + render() + await flushAsync() + + expect(refreshHermesConfig).toHaveBeenCalledTimes(1) + expect($desktopBoot.get().error).toBeTruthy() + expect(desktop.getConnection).toHaveBeenCalledTimes(1) + await advanceBackoff() + expect(desktop.getConnection).toHaveBeenCalledTimes(1) + }) + it('FIX #82679: boot retries are BOUNDED — a persistently dead remote ends in the recovery overlay, not a spinner', async () => { const desktop = fakeDesktop() desktop.getConnection = vi.fn(async () => { diff --git a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts index 5f3096316d..c0932b00e6 100644 --- a/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts +++ b/apps/desktop/src/app/gateway/hooks/use-gateway-boot.ts @@ -1,4 +1,9 @@ -import { isGatewayReauthRequired, JsonRpcGatewayError, resolveGatewayWsUrl } from '@hermes/shared' +import { + isGatewayReauthRequired, + isGatewayWebSocketUrl, + JsonRpcGatewayError, + resolveGatewayWsUrl +} from '@hermes/shared' import { useEffect, useRef } from 'react' import { shouldApplyPostBootProgressError } from '@/components/boot-failure-reauth' @@ -108,15 +113,14 @@ const RECONNECT_ESCALATE_AFTER_MS = 300_000 // only a STREAK of unanswered pings rebuilds the transport. const GATEWAY_LIVENESS_PROBE_TIMEOUT_MS = 5_000 -// Bounded self-heal for a failed REMOTE boot (#82679): when the primary boot -// fails on a transient remote fault (dropped SSH/HTTP registered connection, -// mint timeout — main tags those `retryable` on the boot progress), the -// renderer re-attempts the whole boot with the same full-jitter backoff the -// post-boot reconnect loop uses, up to this many attempts. Retries are -// bounded and end in the real recovery affordance (the boot-failure overlay -// with Retry / Settings), never an infinite spinner. Local failures and -// confirmed reauth rejections never enter this loop — a missing capability -// differs from a transient failure. +// Bounded self-heal for a failed REMOTE boot (#82679): main classifies every +// fault it can see (via getBootProgress().retryable); the renderer adds the one +// it cannot — a valid remote WebSocket dial that fails before becoming usable. +// The renderer re-attempts the whole boot with the same full-jitter backoff the +// post-boot reconnect loop uses, up to this many attempts. Retries are bounded +// and end in the real recovery affordance (the boot-failure overlay with +// Retry / Settings), never an infinite spinner. Local failures and confirmed +// reauth rejections never enter this loop. const BOOT_RETRY_MAX_ATTEMPTS = 5 // Base delay for boot retries. Deliberately slower than the socket reconnect // loop's 300ms: each attempt may rebuild an SSH master + remote dashboard. @@ -1017,6 +1021,11 @@ export function useGatewayBoot({ }) async function boot() { + // Where this boot attempt got to — a historical fact, not a late read of + // gateway.connectionState. A socket can close after a successful dial; + // later initialization errors must not be reclassified as boot dials. + let stage: 'resolving' | 'minting' | 'dialing' | 'connected' = 'resolving' + try { // A profile-pinned helper window (the HUD) dials its target profile's // backend directly — ensureBackend spawns/reuses it from the pool. @@ -1035,6 +1044,8 @@ export function useGatewayBoot({ return } + stage = 'minting' + setDesktopBootStep({ phase: 'renderer.gateway.connect', message: translateNow('boot.steps.connectingGateway'), @@ -1059,17 +1070,24 @@ export function useGatewayBoot({ // Mint a fresh WS URL right before connecting. For OAuth gateways the // ticket is single-use with a short TTL, so the ticket baked into // conn.wsUrl is stale; resolveGatewayWsUrl() re-mints it rather than - // connecting with a dead ticket. Auth rejection asks for sign-in; - // connectivity failures remain retryable. Bounded like the reconnect - // path (#93454) so a wedged mint fails into boot retry instead of - // hanging "Starting Hermes…" forever. + // connecting with a dead ticket. Auth rejection asks for sign-in. This + // await is bounded like the reconnect path (#93454) so a wedged mint + // reaches the recovery affordance instead of hanging "Starting Hermes…". const wsUrl = await withTimeout( resolveGatewayWsUrl(desktop, conn), RECONNECT_ATTEMPT_TIMEOUT_MS, 'Timed out minting the gateway WebSocket URL' ) + // Only a valid WebSocket dial against a remote descriptor counts as a + // transient renderer-side failure; URL and capability failures stay + // terminal at their own boundaries. + if (conn.mode === 'remote' && isGatewayWebSocketUrl(wsUrl)) { + stage = 'dialing' + } + await gateway.connect(wsUrl) + stage = 'connected' if (cancelled) { return @@ -1116,15 +1134,16 @@ export function useGatewayBoot({ if (!cancelled) { const message = err instanceof Error ? err.message : String(err) - // Transient remote failure (dropped SSH/HTTP registered connection, - // mint timeout): self-heal with bounded, jittered retries instead of - // parking on "Desktop boot failed" until the user re-enters the same - // connection details (#82679). Main already cleared the failed cached - // descriptor, so the next getConnection() rebuilds the connection — - // exactly what manual re-entry forced. Exhausted retries, local - // failures, and confirmed reauth rejections end in the real recovery - // affordance (the boot-failure overlay), never an infinite spinner. - if (bootRetryAttempt < BOOT_RETRY_MAX_ATTEMPTS && (await bootFailureIsRetryable()) && !cancelled) { + // Main's classification (#82679) still decides every failure it can + // see. The one it cannot see is the renderer-owned WebSocket dial: + // after a renderer reload main serves its cached descriptor with a + // stale `backend.ready / retryable:false` snapshot, so a remote dial + // that never became usable is retryable on its own. Anything after a + // successful dial keeps the terminal recovery surface. + const canRetry = bootRetryAttempt < BOOT_RETRY_MAX_ATTEMPTS + const retryable = canRetry && (stage === 'dialing' || (await bootFailureIsRetryable())) + + if (retryable && !cancelled) { const delay = reconnectBackoffDelayMs(bootRetryAttempt, { baseDelayMs: BOOT_RETRY_BASE_DELAY_MS }) bootRetryAttempt += 1 resumeDesktopBootForRetry(translateNow('boot.steps.retryingRemoteBackend')) diff --git a/apps/desktop/src/app/hud/hud-shell.tsx b/apps/desktop/src/app/hud/hud-shell.tsx index c4d280dbc0..1cdbc89ae8 100644 --- a/apps/desktop/src/app/hud/hud-shell.tsx +++ b/apps/desktop/src/app/hud/hud-shell.tsx @@ -2,6 +2,7 @@ import { useStore } from '@nanostores/react' import { type CSSProperties, useCallback, useEffect, useRef, useState } from 'react' import { useNavigate } from 'react-router' +import { useViewedInterval } from '@/hooks/use-viewed-interval' import { chatMessageText } from '@/lib/chat-messages' import { $activeSessionAwaitingInput } from '@/store/prompts' import { $busy, $messages } from '@/store/session' @@ -231,7 +232,11 @@ export function HudShell() { // hysteresis so the layout can't flutter while it's dragged along the line. const [edge, setEdge] = useState<'bottom' | 'top'>('top') - useEffect(() => { + // Viewed-gated (#88275): window position only changes while the user can + // see the window, so an ungated 300ms poll keeps the idle renderer hot + // forever. useViewedInterval parks the timer when hidden/unfocused and + // leading-ticks on return, so a drag is still picked up within one tick. + const measureEdge = useCallback(() => { // Measured on the WINDOW, and flush-only. Flipping is what lets the bar // reach the top of the screen at all: the window's top edge can sit against // the menu bar, and the flip moves the composer to that edge. Keying it off @@ -242,33 +247,33 @@ export function HudShell() { const FLIP_ON = 0 const FLIP_OFF = 4 - const measure = () => { - // TRYING IT: the bar stays on top and the transcript always hangs below, - // wherever the HUD is parked. Flip the constant to re-enable the - // edge-aware layout (the CSS for both orientations is still here). - if (HUD_THREAD_ALWAYS_BELOW) { - setEdge('top') + // TRYING IT: the bar stays on top and the transcript always hangs below, + // wherever the HUD is parked. Flip the constant to re-enable the + // edge-aware layout (the CSS for both orientations is still here). + if (HUD_THREAD_ALWAYS_BELOW) { + setEdge('top') - return - } - - // availTop ≈ menu bar / notch inset on macOS; screenY is in full-screen - // coordinates, so "parked at the top" means screenY ≈ availTop, not 0. - const availTop = (window.screen as { availTop?: number }).availTop ?? 0 - const topGap = window.screenY - availTop - - setEdge(prev => (topGap <= FLIP_ON ? 'top' : topGap >= FLIP_OFF ? 'bottom' : prev)) + return } - measure() - const timer = setInterval(measure, 300) - window.addEventListener('resize', measure) + // availTop ≈ menu bar / notch inset on macOS; screenY is in full-screen + // coordinates, so "parked at the top" means screenY ≈ availTop, not 0. + const availTop = (window.screen as { availTop?: number }).availTop ?? 0 + const topGap = window.screenY - availTop + + setEdge(prev => (topGap <= FLIP_ON ? 'top' : topGap >= FLIP_OFF ? 'bottom' : prev)) + }, []) + + useViewedInterval(measureEdge, 300) + + useEffect(() => { + measureEdge() + window.addEventListener('resize', measureEdge) return () => { - clearInterval(timer) - window.removeEventListener('resize', measure) + window.removeEventListener('resize', measureEdge) } - }, []) + }, [measureEdge]) const rootRef = useRef(null) diff --git a/apps/desktop/src/app/hud/hud-shell.viewed-interval.test.tsx b/apps/desktop/src/app/hud/hud-shell.viewed-interval.test.tsx new file mode 100644 index 0000000000..ffbe045acc --- /dev/null +++ b/apps/desktop/src/app/hud/hud-shell.viewed-interval.test.tsx @@ -0,0 +1,96 @@ +// @vitest-environment jsdom +import { act, cleanup, render } from '@testing-library/react' +import { MemoryRouter } from 'react-router' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('../contrib/wiring', () => ({ WiredPane: () => null })) + +class ResizeObserverStub { + observe() {} + unobserve() {} + disconnect() {} +} +Object.assign(globalThis, { ResizeObserver: ResizeObserverStub }) + +import { HudShell } from './hud-shell' + +const EDGE_POLL_MS = 300 + +/** Count callbacks of the edge-measure poll only (the transcript band arms an + * unrelated 500ms mount probe), so the assertion is about this timer. */ +function countEdgePollFires(): { fires: number } { + const counter = { fires: 0 } + const real = window.setInterval.bind(window) + + vi.spyOn(window, 'setInterval').mockImplementation(((cb: TimerHandler, ms?: number, ...rest: unknown[]) => { + if (typeof cb !== 'function' || ms !== EDGE_POLL_MS) { + return real(cb, ms, ...rest) + } + + return real( + () => { + counter.fires += 1 + cb() + }, + ms, + ...rest + ) + }) as typeof window.setInterval) + + return counter +} + +let visibility: DocumentVisibilityState = 'visible' +let focused = true + +describe('HudShell edge poll', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.spyOn(document, 'visibilityState', 'get').mockImplementation(() => visibility) + vi.spyOn(document, 'hasFocus').mockImplementation(() => focused) + }) + + afterEach(() => { + cleanup() + vi.useRealTimers() + vi.restoreAllMocks() + }) + + // #88275: the window's screen position only moves while someone can see the + // window, so the 300ms edge poll must park while the document is hidden and + // run while it is viewed — an ungated interval keeps the idle renderer hot. + it('parks while the window is hidden and runs while it is viewed', async () => { + visibility = 'hidden' + focused = false + const hidden = countEdgePollFires() + + render( + + + + ) + await act(async () => { + await vi.advanceTimersByTimeAsync(3_000) + }) + + expect(hidden.fires).toBe(0) + + cleanup() + vi.mocked(window.setInterval).mockRestore() + + visibility = 'visible' + focused = true + const viewed = countEdgePollFires() + + render( + + + + ) + await act(async () => { + await vi.advanceTimersByTimeAsync(3_000) + }) + + expect(viewed.fires).toBeGreaterThan(0) + }) +}) diff --git a/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx b/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx index f40fcedf6c..c0c5936c78 100644 --- a/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx +++ b/apps/desktop/src/app/right-sidebar/terminal/persistent.tsx @@ -210,7 +210,7 @@ export function PersistentTerminal({ onAddSelectionToChat }: PersistentTerminalP scheduleMeasure('ancestor-mutation') }) - pauseController = createRendererLoopPauseController(handleVisibilityChange) + pauseController = createRendererLoopPauseController(handleVisibilityChange, { pauseWhenUnfocused: true }) if (measure('initial')) { scheduleMeasure('settle') diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index 0673798c18..edbf615359 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -5837,3 +5837,50 @@ describe('usePromptActions reloadFromMessage failed-submit rollback (#95745)', ( expect(latest?.awaitingResponse).toBe(false) }) }) + +describe('usePromptActions live-owner refusal (#106217)', () => { + afterEach(() => { + cleanup() + clearNotifications() + }) + + it('stamps the 4090 SESSION_NOT_OWNED refusal as a non-retryable gateway error surface', async () => { + // Another surface (TUI) holds the lease: the gateway refuses prompt.submit + // with the machine reason in error.data. The inline error bubble must + // carry that as a structured descriptor so the card can drop Retry and + // offer "Start new session" without sniffing the English prose. + let latest: Record | undefined + + const requestGateway = vi.fn(async (method: string) => { + if (method === 'prompt.submit') { + throw new JsonRpcGatewayError( + 'Session 20260909_095312_6b93f5 already has a live owner (tui, pid 32977, lease age 22m).', + { code: 4090, data: { reason: 'SESSION_NOT_OWNED' } } + ) + } + + return {} as never + }) + + let handle: HarnessHandle | null = null + await actRender( + (handle = h)} + onSeedState={next => { + latest = next + }} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + /> + ) + + expect(await handle!.submitText('continue here')).toBe(false) + + const bubble = (latest?.messages as { error?: string; errorSurface?: Record }[]).at(-1) + + expect(bubble?.error).toMatch(/already has a live owner/) + expect(bubble?.errorSurface).toEqual({ layer: 'gateway', code: 'SESSION_NOT_OWNED', retryable: false }) + // Not a stale-runtime symptom: no resume/re-mint attempt hides the refusal. + expect(requestGateway.mock.calls.map(c => c[0])).toEqual(['prompt.submit']) + }) +}) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts index fc6115650b..bea099da65 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts @@ -45,6 +45,7 @@ import { inlineErrorMessage, isProviderSetupError, isSessionBusyError, + isSessionNotOwnedError, isTargetSessionBusy, releaseSubmitInFlight, SessionRecoveryAborted, @@ -855,6 +856,9 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { const message = inlineErrorMessage(err, copy.promptFailed) const occurredAt = Date.now() / 1000 + // Another surface owns the session (#106217): a deterministic gateway + // refusal, so the error card drops Retry and offers a new session. + const notOwned = isSessionNotOwnedError(err) updateSessionState( sessionId, @@ -867,6 +871,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { role: 'assistant', parts: [], error: message || copy.promptFailed, + ...(notOwned && { errorSurface: { layer: 'gateway', code: 'SESSION_NOT_OWNED', retryable: false } }), branchGroupId: state.pendingBranchGroup ?? undefined, completedAt: occurredAt, timestamp: occurredAt diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts index c5bfb83da7..ef315bbcc7 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts @@ -1,4 +1,5 @@ import type { AppendMessage } from '@assistant-ui/react' +import { JsonRpcGatewayError } from '@hermes/shared' import { translateNow, type Translations } from '@/i18n' import type { ChatMessage } from '@/lib/chat-messages' @@ -281,6 +282,17 @@ export function isSessionBusyError(error: unknown): boolean { return /session busy/i.test(error instanceof Error ? error.message : String(error)) } +// prompt.submit refused because another surface (TUI, messaging gateway) +// holds this session's lease (4090 / SESSION_NOT_OWNED, #106217). The gateway +// stamps the machine reason in `error.data.reason`; the prose fallback covers +// backends older than that contract. Deterministic until the owner lets go — +// Retry reproduces it, so the card offers "Start new session" instead. +export function isSessionNotOwnedError(error: unknown): boolean { + const reason = error instanceof JsonRpcGatewayError ? (error.data as { reason?: unknown } | undefined)?.reason : null + + return reason === 'SESSION_NOT_OWNED' || /already has a live owner/i.test(error instanceof Error ? error.message : '') +} + const sleep = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)) // Retry a gateway call across transient "session busy" so it never reaches the diff --git a/apps/desktop/src/app/session/session-state-cache.test.ts b/apps/desktop/src/app/session/session-state-cache.test.ts index ecdeb7f22e..c00ac5a964 100644 --- a/apps/desktop/src/app/session/session-state-cache.test.ts +++ b/apps/desktop/src/app/session/session-state-cache.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' import type { ClientSessionState } from '@/app/types' import type { ChatMessage } from '@/lib/chat-messages' @@ -64,6 +64,26 @@ describe('SessionStateCache', () => { expect(owners.get('stored-a')).toBe('runtime-new-owner') }) + it('does not remeasure unchanged warm transcripts on repeated stream flushes', () => { + const stringify = vi.spyOn(JSON, 'stringify') + + const cache = new SessionStateCache( + { isReferenced: () => false, onEvict: () => undefined }, + { maxBytes: Number.POSITIVE_INFINITY, maxCount: 24 } + ) + + for (let index = 0; index < 24; index += 1) { + cache.set(`runtime-${index}`, settled(`stored-${index}`, 'x'.repeat(4096))) + } + + for (let flush = 0; flush < 10; flush += 1) { + cache.prune() + } + + expect(stringify).toHaveBeenCalledTimes(24) + stringify.mockRestore() + }) + it('uses transcript bytes as well as count', () => { const evicted: string[] = [] diff --git a/apps/desktop/src/app/session/session-state-cache.ts b/apps/desktop/src/app/session/session-state-cache.ts index 04208fed72..6047342f44 100644 --- a/apps/desktop/src/app/session/session-state-cache.ts +++ b/apps/desktop/src/app/session/session-state-cache.ts @@ -45,6 +45,7 @@ export class SessionStateCache extends Map { readonly #maxBytes: number readonly #maxCount: number readonly #recency = new Map() + readonly #transcriptWeights = new WeakMap() #clock = 0 constructor(callbacks: SessionStateCacheCallbacks, limits: SessionStateCacheLimits = {}) { @@ -91,7 +92,7 @@ export class SessionStateCache extends Map { continue } - const weight = transcriptBytes(state) + const weight = this.#weight(state) candidates.push({ bytes: weight, runtimeId, state, touched: this.#recency.get(runtimeId) ?? 0 }) bytes += weight } @@ -145,6 +146,19 @@ export class SessionStateCache extends Map { ) } + #weight(state: ClientSessionState): number { + const cached = this.#transcriptWeights.get(state.messages) + + if (cached !== undefined) { + return cached + } + + const weight = transcriptBytes(state) + this.#transcriptWeights.set(state.messages, weight) + + return weight + } + #touch(runtimeId: string): void { this.#clock += 1 this.#recency.set(runtimeId, this.#clock) diff --git a/apps/desktop/src/app/settings/model-settings.test.tsx b/apps/desktop/src/app/settings/model-settings.test.tsx index 4fe3b0619f..c965910168 100644 --- a/apps/desktop/src/app/settings/model-settings.test.tsx +++ b/apps/desktop/src/app/settings/model-settings.test.tsx @@ -409,6 +409,30 @@ describe('ModelSettings', () => { // Banner present on load, no switch required. expect(await screen.findByText(/still run on/)).toBeTruthy() }) + + it('does not flag an aux slot pinned to a local/LAN endpoint and shows its base_url', async () => { + getAuxiliaryModels.mockResolvedValueOnce({ + main: { provider: 'ollama-cloud', model: 'glm-5.3-flash' }, + tasks: [ + { + task: 'title_generation', + provider: 'openai', + model: 'llama3.2:3b', + base_url: 'http://byron.local:11434/v1', + local_endpoint: true + }, + { task: 'vision', provider: 'openai', model: 'gpt-4o-mini', base_url: 'https://api.example.com/v1', local_endpoint: false } + ] + }) + + await renderModelSettings() + + // The public custom endpoint still bills a provider, so the banner stays — + // but it names only that one task, not the free LAN pin. + expect(await screen.findByText(/1 auxiliary task \(/)).toBeTruthy() + // The row shows where the pinned task actually points. + expect(screen.getByText(/http:\/\/byron\.local:11434\/v1/)).toBeTruthy() + }) }) describe('ModelSettings MoA preset editor', () => { diff --git a/apps/desktop/src/app/settings/model-settings.tsx b/apps/desktop/src/app/settings/model-settings.tsx index baed8f7d81..580b881ec9 100644 --- a/apps/desktop/src/app/settings/model-settings.tsx +++ b/apps/desktop/src/app/settings/model-settings.tsx @@ -18,6 +18,7 @@ import { } from '@/hermes' import type { AuxiliaryModelsResponse, + AuxiliaryTaskAssignment, MoaConfigResponse, MoaModelSlot, ModelOptionProvider, @@ -145,6 +146,30 @@ export const moaConfigComplete = (config: MoaConfigResponse): boolean => moaSlotComplete(preset.aggregator) ) +// Persistent mismatch: any aux slot pinned to a provider different from the +// current main, regardless of whether the user just switched. Catches the +// "I pinned aux months ago and forgot, now it bills a dead provider" case. +// A pin on a private/LAN endpoint (per-task base_url, e.g. a home Ollama box) +// never bills a provider, so the backend's `local_endpoint` verdict exempts it. +export function staleAuxAssignments( + tasks: readonly AuxiliaryTaskAssignment[], + mainProvider: string +): StaleAuxAssignment[] { + const main = mainProvider.toLowerCase() + + if (!main) { + return [] + } + + return tasks + .filter(entry => { + const p = (entry.provider ?? '').toLowerCase() + + return p && p !== 'auto' && p !== main && !entry.local_endpoint + }) + .map(entry => ({ task: entry.task, provider: entry.provider, model: entry.model })) +} + interface StaleAuxWarningProps { applying: boolean onReset: () => void @@ -500,24 +525,10 @@ export function ModelSettings({ onMainModelChanged, scopeProfile }: ModelSetting const auxiliaryTaskLabel = useCallback((key: string) => m.tasks[key]?.label ?? key, [m.tasks]) - // Persistent mismatch: any aux slot pinned to a provider different from the - // current main, regardless of whether the user just switched. Catches the - // "I pinned aux months ago and forgot, now it bills a dead provider" case. - const persistentStaleAux = useMemo(() => { - const mainProvider = (mainModel?.provider ?? '').toLowerCase() - - if (!mainProvider || !auxiliary) { - return [] - } - - return auxiliary.tasks - .filter(entry => { - const p = (entry.provider ?? '').toLowerCase() - - return p && p !== 'auto' && p !== mainProvider - }) - .map(entry => ({ task: entry.task, provider: entry.provider, model: entry.model })) - }, [auxiliary, mainModel]) + const persistentStaleAux = useMemo( + () => staleAuxAssignments(auxiliary?.tasks ?? [], mainModel?.provider ?? ''), + [auxiliary, mainModel] + ) // Capabilities of the APPLIED main model — gates the profile-default // reasoning/speed controls the same way the composer picker gates per-model @@ -1069,6 +1080,9 @@ export function ModelSettings({ onMainModelChanged, scopeProfile }: ModelSetting description={ {isAuto ? m.autoUseMain : `${current.provider} · ${current.model || m.providerDefault}`} + {!isAuto && current.base_url && ( + · {current.base_url} + )} } title={ diff --git a/apps/desktop/src/app/settings/plugin-install-modal.tsx b/apps/desktop/src/app/settings/plugin-install-modal.tsx index 1034e20367..7c1f3cf8b8 100644 --- a/apps/desktop/src/app/settings/plugin-install-modal.tsx +++ b/apps/desktop/src/app/settings/plugin-install-modal.tsx @@ -32,6 +32,7 @@ import { } from '@/store/plugin-install-request' import { $activeGatewayProfile, $profileScope } from '@/store/profile' import { $connection } from '@/store/session' +import { runGatewayRestart } from '@/store/system-actions' type ProbeResult = Awaited['probePluginRepo']>>> @@ -91,6 +92,8 @@ export function PluginInstallModal() { setPhase('probing') setProbe(null) setInstallError(null) + // Reviewed catalog picks streamline the ceremony: enable defaults ON + // (installing a reviewed entry to not use it is the rare case). setEnableAgent(payload.enable ?? true) setForceReinstall(payload.force ?? false) @@ -151,7 +154,7 @@ export function PluginInstallModal() { } }, [request, resetState, runProbe]) - const profileLabel = activeProfile || profileScope || 'default' + const profileLabel = request?.profile || activeProfile || profileScope || 'default' const agentTargetHint = connection?.mode === 'remote' ? m.agentTargetRemote(profileLabel) : m.agentTargetLocal(profileLabel) @@ -183,22 +186,34 @@ export function PluginInstallModal() { const errors: string[] = [] const successes: string[] = [] + let agentInstalled = false try { if (installAgent && probe.agent) { const result = await installAgentPlugin(requestGateway, { identifier: request.repo, force: forceReinstall, - enable: enableAgent + enable: enableAgent, + catalogName: request.catalogName, + profile: request.profile }) if (result.ok) { successes.push(m.agentSuccess(result.pluginName ?? request.repo)) + agentInstalled = true if (result.missingEnv?.length) { + const firstVar = result.missingEnv[0] + notify({ kind: 'warning', - message: m.missingEnv(result.missingEnv.join(', ')) + message: m.missingEnv(result.missingEnv.join(', ')), + // Deep-link straight to the credential card instead of leaving + // the user to hunt through Settings → Tools & Keys by hand. + action: { + label: m.missingEnvAction, + onClick: () => navigate(`/settings?tab=keys&key=${encodeURIComponent(firstVar)}`) + } }) } @@ -234,8 +249,19 @@ export function PluginInstallModal() { notify({ kind: 'success', message }) } + // An enabled agent plugin only takes effect after a gateway restart — + // offer the restart right here instead of a dim hint to run later. + if (agentInstalled && enableAgent) { + notify({ + kind: 'success', + message: m.restartToApply, + action: { label: m.restartNow, onClick: () => void runGatewayRestart() } + }) + } + closePluginInstallRequest() - navigate('/settings?tab=plugins') + // Catalog picks come from Capabilities → Plugins; land back there. + navigate(request.catalogName ? '/skills?tab=plugins' : '/settings?tab=plugins') return } @@ -305,12 +331,21 @@ export function PluginInstallModal() {
{request.repo}
+ {request.catalogName && ( +

+ {m.catalogPinned(request.catalogName, request.sha?.slice(0, 8) ?? '')} +

+ )}
-
{m.securityHeading}
-

{m.securityIntro}

+
+ {request.catalogName ? m.reviewedHeading : m.securityHeading} +
+

+ {request.catalogName ? m.reviewedIntro : m.securityIntro} +

{sourceLinks && ( @@ -414,12 +449,14 @@ export function PluginInstallModal() { )} - + {!request.catalogName && ( + + )}
)} diff --git a/apps/desktop/src/app/settings/plugins-settings.test.tsx b/apps/desktop/src/app/settings/plugins-settings.test.tsx index 917e7e093a..355770de44 100644 --- a/apps/desktop/src/app/settings/plugins-settings.test.tsx +++ b/apps/desktop/src/app/settings/plugins-settings.test.tsx @@ -1,65 +1,34 @@ -import { QueryClientProvider } from '@tanstack/react-query' -import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' +import { cleanup, render, screen } from '@testing-library/react' +import { MemoryRouter } from 'react-router' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -const { requestGateway, getProfiles } = vi.hoisted(() => ({ - requestGateway: vi.fn(), - getProfiles: vi.fn<() => Promise<{ profiles: { name: string; is_default: boolean }[] }>>(async () => ({ - profiles: [] - })) +const { requestGateway } = vi.hoisted(() => ({ + requestGateway: vi.fn(async () => ({ plugins: [] })) })) vi.mock('@/app/gateway/hooks/use-gateway-request', () => ({ useGatewayRequest: () => ({ requestGateway }) })) -vi.mock('@/hermes', async importOriginal => ({ - ...(await importOriginal>()), - getProfiles -})) - import { $pluginRecords } from '@/contrib/plugins-store' -import { queryClient } from '@/lib/query-client' -import { - $agentPluginBusy, - $agentPlugins, - $agentPluginsError, - $agentPluginsStatus, - type AgentPluginRow -} from '@/store/agent-plugins' -import { $activeGatewayProfile } from '@/store/profile' -import { $connection, $gatewayState } from '@/store/session' +import { $agentPlugins, $agentPluginsStatus } from '@/store/agent-plugins' +import { $gatewayState } from '@/store/session' import { PluginsSettings } from './plugins-settings' -const legacyRow = { - name: 'Legacy plugin', - version: '0.20.0', - description: 'Returned by a pre-key backend', - source: 'user', - status: 'disabled' -} satisfies AgentPluginRow - const renderSettings = () => render( - + - + ) beforeEach(() => { - requestGateway.mockReset() - getProfiles.mockReset() - getProfiles.mockResolvedValue({ profiles: [] }) - queryClient.clear() + requestGateway.mockClear() $pluginRecords.set({}) - $agentPlugins.set([legacyRow]) + $agentPlugins.set([]) $agentPluginsStatus.set('ready') - $agentPluginsError.set(null) - $agentPluginBusy.set(null) $gatewayState.set('idle') - $connection.set(null) - $activeGatewayProfile.set('default') }) afterEach(() => { @@ -68,166 +37,93 @@ afterEach(() => { }) describe('PluginsSettings', () => { - it('renders and searches plugin rows returned without a canonical key', () => { - renderSettings() - - expect(screen.getByText('Legacy plugin')).toBeTruthy() - - fireEvent.change(screen.getByRole('textbox'), { target: { value: 'pre-key' } }) - - expect(screen.getByText('Legacy plugin')).toBeTruthy() - }) - - it('renders keyless rows read-only instead of falling back to name-addressed toggles', () => { - // Name-addressed toggles flip every same-named plugin across category - // dirs (image_gen/fal vs video_gen/fal) — the reason toggles moved to - // canonical keys. A pre-contract-v6 row must never reach the RPC. - renderSettings() - - const toggle = screen.getByRole('switch', { name: 'Enable Legacy plugin' }) - - expect(toggle.hasAttribute('disabled') || toggle.getAttribute('aria-disabled') === 'true').toBe(true) - - fireEvent.click(toggle) - - expect(requestGateway).not.toHaveBeenCalledWith('plugins.manage', expect.objectContaining({ action: 'toggle' })) - }) - - it('keeps duplicate-named keyless rows distinct (no React key collision)', () => { - const sibling = { - ...legacyRow, - description: 'A second plugin category with the same legacy name' - } - - const consoleError = vi.spyOn(console, 'error').mockImplementation(() => undefined) - - $agentPlugins.set([legacyRow, sibling]) - - renderSettings() - - expect(screen.getAllByRole('switch', { name: 'Enable Legacy plugin' })).toHaveLength(2) - expect(screen.getByText(sibling.description)).toBeTruthy() - expect(consoleError.mock.calls.flat().join(' ')).not.toContain('same key') - }) - - it('keeps using the canonical key when the backend provides one', async () => { - const keyedRow = { ...legacyRow, key: 'image_gen/legacy' } - - $agentPlugins.set([keyedRow]) - requestGateway.mockResolvedValue({ ok: true, plugin: { ...keyedRow, status: 'enabled' } }) - - renderSettings() - fireEvent.click(screen.getByRole('switch', { name: 'Enable Legacy plugin' })) - - await waitFor(() => - expect(requestGateway).toHaveBeenCalledWith('plugins.manage', { - action: 'toggle', - key: 'image_gen/legacy', - enable: true - }) - ) - }) - - it('hides repo-bundled built-ins and keeps the count pill in sync', () => { - // The Agent plugins section is the control panel for plugins the USER - // installed — built-ins (browser backends, cron providers, model - // providers…) ship enabled-by-default and are configured elsewhere. + it('points agent-plugin management at Capabilities instead of duplicating the list', () => { + // Agent plugins are profile-scoped and managed in Capabilities → Plugins; + // Settings keeps desktop plugins only, plus a pointer. $agentPlugins.set([ - legacyRow, - { ...legacyRow, name: 'browserbase', key: 'browser/browserbase', source: 'bundled' }, - { ...legacyRow, name: 'chronos', key: 'cron_providers/chronos', source: 'bundled' }, - { ...legacyRow, name: 'deepinfra', key: 'model-providers/deepinfra', source: 'bundled' } + { + description: 'Should NOT be listed here anymore', + key: 'demo-plugin', + name: 'demo-plugin', + source: 'git', + status: 'enabled', + version: '1.0.0' + } ]) renderSettings() - expect(screen.getByText('Legacy plugin')).toBeTruthy() - expect(screen.queryByText('browserbase')).toBeNull() - expect(screen.queryByText('chronos')).toBeNull() - expect(screen.queryByText('deepinfra')).toBeNull() - // Count pill reflects the filtered list, not the raw RPC row count. - expect(screen.getByText('1 installed', { exact: false })).toBeTruthy() + expect(screen.queryByText('demo-plugin')).toBeNull() + expect(screen.getByText(/managed per profile in Capabilities/)).toBeTruthy() + expect(screen.getByRole('link', { name: /Capabilities/ }).getAttribute('href')).toContain('/skills?tab=plugins') }) - it('hides legacy other-surface categories even when the backend omits source', () => { - // Older backends may not report source reliably — the key-prefix - // fallback still hides categories other surfaces own. - $agentPlugins.set([{ ...legacyRow, name: 'deepinfra', key: 'model-providers/deepinfra', source: 'user' }]) - - renderSettings() - - expect(screen.queryByText('deepinfra')).toBeNull() - }) - - it('shows no profile selector with a single profile', async () => { - getProfiles.mockResolvedValue({ profiles: [{ name: 'default', is_default: true }] }) - - renderSettings() - - await waitFor(() => expect(getProfiles).toHaveBeenCalled()) - expect(screen.queryByText('Applies to:')).toBeNull() - }) - - it('lists the active profile scope without a profile param and reloads scoped on change', async () => { - getProfiles.mockResolvedValue({ - profiles: [ - { name: 'default', is_default: true }, - { name: 'work', is_default: false } - ] - }) - requestGateway.mockResolvedValue({ plugins: [legacyRow] }) - $gatewayState.set('open') - - renderSettings() - - // Active profile scope: no profile param — older backends unchanged. - await waitFor(() => expect(requestGateway).toHaveBeenCalledWith('plugins.manage', { action: 'list' })) - await waitFor(() => expect(screen.getByText('Applies to:')).toBeTruthy()) - }) - - it('sends toggles through the selected profile scope', async () => { - // jsdom's scrollIntoView is missing/non-functional; Radix Select calls it - // when the dropdown opens. - Element.prototype.scrollIntoView = vi.fn() - - const keyedRow = { ...legacyRow, key: 'image_gen/legacy' } - - getProfiles.mockResolvedValue({ - profiles: [ - { name: 'default', is_default: true }, - { name: 'work', is_default: false } - ] - }) - requestGateway.mockImplementation(async (method: string, params?: Record) => { - if (params?.action === 'list') { - return { plugins: [keyedRow] } + it('flags a unified-root desktop half whose agent half is missing on this backend', () => { + $pluginRecords.set({ + 'pixel-overlay': { + id: 'pixel-overlay', + name: 'Pixel Overlay', + kind: 'disk', + status: 'loaded', + file: '/home/user/.hermes/plugins/pixel-overlay/desktop/plugin.js' } - - return { ok: true, plugin: { ...keyedRow, status: 'enabled' } } }) + $agentPlugins.set([]) // connected backend has no agent half + $agentPluginsStatus.set('ready') + + renderSettings() + + expect(screen.getByText('agent half missing here')).toBeTruthy() + }) + + it('does not flag when the agent half exists on the connected backend', () => { + $pluginRecords.set({ + 'pixel-overlay': { + id: 'pixel-overlay', + name: 'Pixel Overlay', + kind: 'disk', + status: 'loaded', + file: '/home/user/.hermes/plugins/pixel-overlay/desktop/plugin.js' + } + }) + $agentPlugins.set([ + { + description: '', + key: 'pixel-overlay', + name: 'pixel-overlay', + source: 'user', + status: 'enabled', + version: '1.0.0' + } + ]) + + renderSettings() + + expect(screen.queryByText('agent half missing here')).toBeNull() + }) + + it('does not flag standalone desktop plugins (not from the unified root)', () => { + $pluginRecords.set({ + standalone: { + id: 'standalone', + name: 'Standalone Theme', + kind: 'disk', + status: 'loaded', + file: '/home/user/.config/hermes-desktop/desktop-plugins/standalone/plugin.js' + } + }) + $agentPlugins.set([]) + + renderSettings() + + expect(screen.queryByText('agent half missing here')).toBeNull() + }) + + it('loads the connected backend plugin list once the gateway opens (badge data)', () => { $gatewayState.set('open') renderSettings() - await waitFor(() => expect(screen.getByText('Applies to:')).toBeTruthy()) - - // Select the non-active profile scope. - fireEvent.click(screen.getByRole('combobox')) - fireEvent.click(await screen.findByText('work')) - - await waitFor(() => - expect(requestGateway).toHaveBeenCalledWith('plugins.manage', { action: 'list', profile: 'work' }) - ) - - fireEvent.click(screen.getByRole('switch', { name: 'Enable Legacy plugin' })) - - await waitFor(() => - expect(requestGateway).toHaveBeenCalledWith('plugins.manage', { - action: 'toggle', - key: 'image_gen/legacy', - enable: true, - profile: 'work' - }) - ) + expect(requestGateway).toHaveBeenCalledWith('plugins.manage', expect.objectContaining({ action: 'list' })) }) }) diff --git a/apps/desktop/src/app/settings/plugins-settings.tsx b/apps/desktop/src/app/settings/plugins-settings.tsx index b85c7f917d..af3c882b45 100644 --- a/apps/desktop/src/app/settings/plugins-settings.tsx +++ b/apps/desktop/src/app/settings/plugins-settings.tsx @@ -1,47 +1,27 @@ import { useStore } from '@nanostores/react' -import { useQuery } from '@tanstack/react-query' -import { type ReactNode, useEffect, useState } from 'react' +import { type ReactNode, useEffect } from 'react' +import { Link } from 'react-router' import { useGatewayRequest } from '@/app/gateway/hooks/use-gateway-request' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' -import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from '@/components/ui/select' import { Switch } from '@/components/ui/switch' import { Tip } from '@/components/ui/tooltip' import { $pluginRecords, type PluginRecord, setPluginEnabled } from '@/contrib/plugins-store' import { discoverRuntimePlugins } from '@/contrib/runtime-loader' -import { getProfiles } from '@/hermes' import { useI18n } from '@/i18n' import { triggerHaptic } from '@/lib/haptics' import { FolderOpen, Monitor, Package, RefreshCw } from '@/lib/icons' -import { normalize } from '@/lib/text' -import { - $agentPluginBusy, - $agentPlugins, - $agentPluginsError, - $agentPluginsStatus, - type AgentPluginRow, - type GatewayRequest, - isDesktopRelevantPlugin, - loadAgentPlugins, - toggleAgentPlugin -} from '@/store/agent-plugins' +import { $agentPlugins, $agentPluginsStatus, loadAgentPlugins } from '@/store/agent-plugins' import { notifyError } from '@/store/notifications' import { openPluginInstallRequest } from '@/store/plugin-install-request' -import { $activeGatewayProfile } from '@/store/profile' -import { $connection, $gatewayState } from '@/store/session' +import { $gatewayState } from '@/store/session' -import { EmptyState, ListRowSkeleton, Pill, SettingsContent, SettingsSection } from './primitives' +import { EmptyState, Pill, SettingsContent, SettingsSection } from './primitives' import { useDeepLinkHighlight } from './use-deep-link-highlight' const KIND_ORDER: Record = { disk: 0, runtime: 1, bundled: 2 } -// User-installed plugins first — mirrors `hermes plugins list --user`. -const SOURCE_ORDER: Record = { user: 0, git: 0, project: 1, entrypoint: 2 } - -const agentPluginRowKey = (row: AgentPluginRow) => - row.key ?? [row.name, row.source, row.version, row.description].join('\0') - /** Deep-link anchor for a plugin row (`?tab=plugins&plugin=`). */ export const pluginElementId = (target: string) => `plugin-${target}` @@ -73,31 +53,6 @@ async function revealPluginsDir() { } } -// Agent plugins live under the BACKEND's hermes home (profile-aware), so the -// path comes from the gateway — not from the renderer's local HERMES_HOME. -// Callers gate on a local connection: openDir mkdir-creates the path, which -// must never happen for a directory that belongs to a remote box. -async function revealAgentPluginsDir(request: GatewayRequest) { - try { - const result = await request<{ home?: string }>('config.get', { key: 'profile' }) - const home = (result?.home ?? '').trim() - - if (!home) { - notifyError('The backend did not report its home directory', 'Could not open the plugins folder') - - return - } - - const opened = await window.hermesDesktop?.openDir?.(`${home}/plugins`) - - if (opened && !opened.ok) { - notifyError(opened.error ?? 'unknown error', 'Could not open the plugins folder') - } - } catch (err) { - notifyError(err, 'Could not open the plugins folder') - } -} - // Compact row: name + pills and a wrapping description on the left, controls // pinned top-right. Same type scale as ListRow, without its wide control grid. function PluginLine({ @@ -128,190 +83,61 @@ function PluginLine({ ) } -function AgentPluginRowView({ row, profile }: { row: AgentPluginRow; profile: string | null }) { - const { t } = useI18n() - const p = t.settings.plugins - const { requestGateway } = useGatewayRequest() - const busy = useStore($agentPluginBusy) - const key = row.key +/** Folder name when a desktop plugin entry lives in the UNIFIED agent-plugins + * root (`~/.hermes/plugins//desktop/plugin.js`) — i.e. it is the + * desktop half of a bundled agent+desktop package. Null for standalone + * desktop plugins. */ +function unifiedPackageName(file?: string): null | string { + if (!file) { + return null + } - // Pre-contract-v6 backends return rows without a canonical key. Name-addressed - // toggles silently flip every same-named plugin across category dirs - // (image_gen/fal vs video_gen/fal), so keyless rows are read-only — the - // backend-contract skew toast tells the user to update. - const toggle = ( - { - if (!key) { - return - } + const match = /[\\/]plugins[\\/]([^\\/]+)[\\/]desktop[\\/]plugin\.js$/.exec(file) - triggerHaptic('selection') - void toggleAgentPlugin(requestGateway, key, on, p.agent.toggleFailed(row.name), profile) - }} - /> - ) + return match ? match[1] : null +} - return ( - {toggle}} - description={row.description || (row.version ? `v${row.version}` : undefined)} - id={pluginElementId(key ?? row.name)} - title={ - <> - {row.name} - {p.agent.sources[row.source] ?? row.source} - {row.portable && {p.agent.portable}} - +/** Open the dual-target install modal pre-filled to install ONLY the agent + * half of a bundled package (drift repair). Provenance comes from the + * package's catalog sidecar when present; otherwise the git remote of the + * plugin folder is unknown and we fall back to asking the user via the + * standard flow with the folder name as identifier hint. */ +async function repairAgentHalf(record: PluginRecord, packageName: string) { + let repo = '' + let catalogName: string | undefined + let sha: string | undefined + + try { + const pluginDir = record.file?.replace(/[\\/]desktop[\\/]plugin\.js$/, '') + + const raw = pluginDir + ? await window.hermesDesktop?.readFileText?.(`${pluginDir}/.hermes-catalog.json`) + : null + + if (raw) { + const sidecar = JSON.parse(typeof raw === 'string' ? raw : (raw as { content?: string }).content ?? '') as { + catalog_name?: string + repo?: string + sha?: string } - /> - ) -} -function AgentPluginsSection() { - const { t } = useI18n() - const p = t.settings.plugins - const { requestGateway } = useGatewayRequest() - const gatewayState = useStore($gatewayState) - const connection = useStore($connection) - const rows = useStore($agentPlugins) - const status = useStore($agentPluginsStatus) - const error = useStore($agentPluginsError) - const [query, setQuery] = useState('') - - // 'Applies to' profile scope: which profile's plugins we list/toggle. - // Defaults to the app-wide active profile; overriding it here lets the user - // manage ANY profile's plugins without switching the whole app (same - // pattern as the Capabilities scope selector in app/skills). null = the - // active profile — the RPC is sent without a profile param so older - // backends keep working unchanged. - const activeProfile = useStore($activeGatewayProfile) - const [scopeOverride, setScopeOverride] = useState(null) - const scopeProfile = scopeOverride ?? activeProfile ?? null - // The param we actually send: omit it for the active profile. - const requestProfile = scopeOverride && scopeOverride !== activeProfile ? scopeOverride : null - - const { data: profilesData } = useQuery({ - queryKey: ['agent-plugins-profiles'], - queryFn: getProfiles, - staleTime: 60_000 - }) - - const profiles = profilesData?.profiles ?? [] - - // An app-wide profile switch retargets the default scope — drop the - // override so the list reloads for the profile the user just switched to. - useEffect(() => { - setScopeOverride(null) - }, [activeProfile]) - - useEffect(() => { - if (gatewayState !== 'open') { - return + repo = sidecar.repo ?? '' + catalogName = sidecar.catalog_name + sha = sidecar.sha } + } catch { + // No sidecar (raw-git bundled install) — fall through to the name hint. + } - void loadAgentPlugins(requestGateway, requestProfile) - }, [gatewayState, requestGateway, requestProfile]) - - const needle = normalize(query) - - const sorted = rows - .filter(isDesktopRelevantPlugin) - .filter( - row => - !needle || - row.name.toLowerCase().includes(needle) || - (row.key ?? '').toLowerCase().includes(needle) || - row.description.toLowerCase().includes(needle) - ) - .sort((a, b) => (SOURCE_ORDER[a.source] ?? 9) - (SOURCE_ORDER[b.source] ?? 9) || a.name.localeCompare(b.name)) - - return ( - -

- {p.agent.blurb} -

- - {profiles.length > 1 && ( -
- - {p.agent.appliesTo} - - -
- )} - - {connection?.mode !== 'remote' && !requestProfile && ( -
- -
- )} - - setQuery(event.target.value)} - placeholder={p.agent.search} - spellCheck={false} - value={query} - /> - - {status === 'loading' || status === 'idle' ? ( -
- - - -
- ) : status === 'error' ? ( - - ) : sorted.length === 0 ? ( - needle ? ( -

- {p.agent.noMatches} -

- ) : ( - - ) - ) : ( -
- {sorted.map(row => ( - - ))} -
- )} -
- ) + openPluginInstallRequest({ + catalogName, + legacyHint: 'agent', + repo: repo || packageName, + sha + }) } -function PluginRow({ record }: { record: PluginRecord }) { +function PluginRow({ record, agentHalfMissing }: { record: PluginRecord; agentHalfMissing?: boolean }) { const { t } = useI18n() const p = t.settings.plugins @@ -349,6 +175,18 @@ function PluginRow({ record }: { record: PluginRecord }) { {record.name} {p.kinds[record.kind]} {record.status === 'error' && {p.failed}} + {agentHalfMissing && ( + + + + )} } /> @@ -359,6 +197,25 @@ export function PluginsSettings() { const { t } = useI18n() const p = t.settings.plugins const records = useStore($pluginRecords) + const { requestGateway } = useGatewayRequest() + const gatewayState = useStore($gatewayState) + // The agent-plugin list for the CURRENTLY connected backend's active + // profile — used only to flag bundled packages whose desktop half is local + // but whose agent half is not installed where the app is now pointing (one + // desktop app, N agents: switching gateway/profile makes this drift visible + // instead of silent). Management of agent plugins lives in Capabilities → + // Plugins; this page keeps just the badge. + const agentRows = useStore($agentPlugins) + const agentStatus = useStore($agentPluginsStatus) + const agentNames = new Set(agentRows.flatMap(row => [row.name, row.key ?? row.name])) + + useEffect(() => { + if (gatewayState !== 'open') { + return + } + + void loadAgentPlugins(requestGateway) + }, [gatewayState, requestGateway]) // Deep-link from settings search (?plugin=): rows render as soon // as their store hydrates, so "ready" is simply target-present; the polling @@ -406,14 +263,31 @@ export function PluginsSettings() { ) : (
- {rows.map(record => ( - - ))} + {rows.map(record => { + const packageName = unifiedPackageName(record.file) + + return ( + + ) + })}
)} - + +

+ {p.agent.movedToCapabilities}{' '} + + {p.agent.openCapabilities} + +

+
) } diff --git a/apps/desktop/src/app/skills/index.tsx b/apps/desktop/src/app/skills/index.tsx index 81e1dfb1e9..c7e38e6ccc 100644 --- a/apps/desktop/src/app/skills/index.tsx +++ b/apps/desktop/src/app/skills/index.tsx @@ -68,12 +68,13 @@ import type { SetStatusbarItemGroup } from '../shell/statusbar-controls' import { EmbeddedHubPicker } from './embedded-hub-picker' import { McpTab } from './mcp-tab' +import { PluginsTab } from './plugins-tab' import { $skillsSortDesc, $toolsetsSortDesc } from './store' // 'hub' is gone as a top-level tab — the Skills Hub browser lives inside the // Skills tab now (EmbeddedHubPicker below the installed list). Legacy // `?tab=hub` links fall back to 'skills' via useRouteEnumParam. -const SKILLS_MODES = ['skills', 'toolsets', 'mcp'] as const +const SKILLS_MODES = ['skills', 'toolsets', 'mcp', 'plugins'] as const // Skills + toolsets live in the RQ cache so switching tabs/pages paints the // cached lists instantly (no reload flash) and mount only fires a deduped @@ -847,14 +848,15 @@ export function SkillsView({ onTabChange={id => setMode(id as (typeof SKILLS_MODES)[number])} // MCP manages a handful of entries with the editor right there — // searching it is noise. - searchHidden={mode === 'mcp'} + searchHidden={mode === 'mcp' || mode === 'plugins'} searchHints={searchHints} searchPlaceholder={mode === 'skills' ? t.skills.searchSkills : t.skills.searchToolsets} searchValue={query} tabs={[ { id: 'skills', label: t.skills.tabSkills, meta: skills?.length ?? null }, { id: 'toolsets', label: t.skills.tabToolsets, meta: toolsets ? visibleToolsetCount(toolsets) : null }, - { id: 'mcp', label: t.skills.tabMcp } + { id: 'mcp', label: t.skills.tabMcp }, + { id: 'plugins', label: t.skills.tabPlugins } ]} > {/* One shared column: the scope selector sits above whichever tab is @@ -864,7 +866,12 @@ export function SkillsView({ {profileScopeSelector}
- {mode === 'mcp' ? ( + {mode === 'plugins' ? ( + // Agent plugins for the scoped profile: installed list on top, + // the live catalog picker underneath (same shape as Skills). + // Keyed on scope so a profile/connection switch reloads the list. + + ) : mode === 'mcp' ? ( // The gateway instance backs ONLY the live `reload.mcp` RPC, and // it is the ACTIVE gateway's socket — for a scope pinned to a // different backend that RPC would hot-reload the wrong diff --git a/apps/desktop/src/app/skills/mcp-tab.tsx b/apps/desktop/src/app/skills/mcp-tab.tsx index 4f984a4d4f..020d3e8e2a 100644 --- a/apps/desktop/src/app/skills/mcp-tab.tsx +++ b/apps/desktop/src/app/skills/mcp-tab.tsx @@ -31,6 +31,7 @@ import { testMcpServer } from '@/hermes' import { type Translations, useI18n } from '@/i18n' +import { startCompletionPoll } from '@/lib/completion-poll' import { compactNumber } from '@/lib/format' import { brandFor } from '@/lib/mcp-brands' import { estimateServerTokens, serverUsageCount } from '@/lib/mcp-cost' @@ -1725,36 +1726,25 @@ function McpLogs({ }) { const [lines, setLines] = useState(null) // A profile switch reroutes getLogs to the new backend; keying the effect on - // the active profile tears down the old poll (its `cancelled` flag blocks a - // late setLines) so profile A's logs never flash in B. + // the active profile tears down the old poll (stop suppresses a late + // publish) so profile A's logs never flash in B. const activeProfile = useStore($activeGatewayProfile) useEffect(() => { - let cancelled = false + setLines(null) - const poll = async () => { - try { + return startCompletionPoll({ + delayMs: LOG_POLL_MS, + poll: async () => { const response = source === 'stdio' ? await getLogs({ file: 'mcp', lines: 500 }) : await getLogs({ file: 'agent', lines: 300, search: server ?? 'mcp' }) - if (!cancelled) { - setLines(source === 'stdio' && server ? filterStdioSections(response.lines, server) : response.lines) - } - } catch { - // Backend momentarily unavailable — keep the last tail. - } - } - - setLines(null) - void poll() - const timer = window.setInterval(() => void poll(), LOG_POLL_MS) - - return () => { - cancelled = true - window.clearInterval(timer) - } + return source === 'stdio' && server ? filterStdioSections(response.lines, server) : response.lines + }, + publish: setLines + }) }, [server, source, activeProfile]) return diff --git a/apps/desktop/src/app/skills/plugins-tab.test.tsx b/apps/desktop/src/app/skills/plugins-tab.test.tsx new file mode 100644 index 0000000000..183987c7d8 --- /dev/null +++ b/apps/desktop/src/app/skills/plugins-tab.test.tsx @@ -0,0 +1,318 @@ +import { cleanup, render, screen, waitFor } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { $agentPlugins, $agentPluginsStatus } from '@/store/agent-plugins' +import { $pluginInstallRequest, closePluginInstallRequest } from '@/store/plugin-install-request' + +import { PluginsTab } from './plugins-tab' + +const requestGateway = vi.fn(async () => ({ plugins: [] })) + +vi.mock('@/app/gateway/hooks/use-gateway-request', () => ({ + useGatewayRequest: () => ({ requestGateway }) +})) + +describe('PluginsTab', () => { + beforeEach(() => { + $agentPlugins.set([]) + $agentPluginsStatus.set('ready') + closePluginInstallRequest() + requestGateway.mockClear() + }) + + afterEach(cleanup) + + it('lists the scoped profile agent plugins with toggles', () => { + $agentPlugins.set([ + { + description: 'A test plugin', + key: 'demo-plugin', + name: 'demo-plugin', + source: 'git', + status: 'enabled', + version: '1.0.0' + } + ]) + + render() + + expect(screen.getByText('demo-plugin')).toBeTruthy() + expect(screen.getByRole('switch', { name: 'demo-plugin' }).getAttribute('aria-checked')).toBe('true') + }) + + it('hides bundled plugins (managed from their own surfaces)', () => { + $agentPlugins.set([ + { + description: '', + key: 'image_gen/fal', + name: 'fal', + source: 'bundled', + status: 'enabled', + version: '' + } + ]) + + render() + + expect(screen.queryByText('fal')).toBeNull() + expect(screen.getByText(/No agent plugins installed/)).toBeTruthy() + }) + + it('loads the plugin list scoped to the selected profile', () => { + render() + + expect(requestGateway).toHaveBeenCalledWith( + 'plugins.manage', + expect.objectContaining({ action: 'list', profile: 'workbot' }) + ) + }) + + it('opens the dual-target install modal from a catalog pick message', async () => { + render() + + window.dispatchEvent( + new MessageEvent('message', { + data: { + name: 'weather-plugin', + repo: 'https://github.com/example/weather-plugin', + sha: 'a'.repeat(40), + subdir: '', + tier: 'community', + type: 'hermes-plugin-pick' + }, + origin: 'https://hermes-agent.nousresearch.com' + }) + ) + + await waitFor(() => { + const request = $pluginInstallRequest.get() + + expect(request).not.toBeNull() + expect(request?.catalogName).toBe('weather-plugin') + expect(request?.repo).toBe('https://github.com/example/weather-plugin') + expect(request?.profile).toBe('workbot') + expect(request?.sha).toBe('a'.repeat(40)) + }) + }) + + it('ignores pick messages from foreign origins', () => { + render() + + window.dispatchEvent( + new MessageEvent('message', { + data: { + name: 'evil-plugin', + repo: 'https://github.com/evil/evil-plugin', + type: 'hermes-plugin-pick' + }, + origin: 'https://evil.example.com' + }) + ) + + expect($pluginInstallRequest.get()).toBeNull() + }) + + it('toggles by canonical key through plugins.manage', async () => { + $agentPlugins.set([ + { + description: '', + key: 'image_gen/legacy', + name: 'Legacy plugin', + source: 'user', + status: 'disabled', + version: '0.20.0' + } + ]) + requestGateway.mockResolvedValueOnce({ + ok: true, + plugin: { key: 'image_gen/legacy', name: 'Legacy plugin', status: 'enabled' } + } as never) + + render() + + screen.getByRole('switch', { name: 'Legacy plugin' }).click() + + await waitFor(() => + expect(requestGateway).toHaveBeenCalledWith( + 'plugins.manage', + expect.objectContaining({ action: 'toggle', key: 'image_gen/legacy', enable: true }) + ) + ) + }) + + it('renders keyless rows read-only (no name-addressed toggle RPC)', () => { + // Name-addressed toggles flip every same-named plugin across category + // dirs — pre-contract-v6 rows must never reach the RPC. + $agentPlugins.set([ + { + description: 'Returned by a pre-key backend', + name: 'Legacy plugin', + source: 'user', + status: 'disabled', + version: '0.20.0' + } + ]) + + render() + + const toggle = screen.getByRole('switch', { name: 'Legacy plugin' }) + + expect(toggle.hasAttribute('disabled') || toggle.getAttribute('aria-disabled') === 'true').toBe(true) + + toggle.click() + + expect(requestGateway).not.toHaveBeenCalledWith( + 'plugins.manage', + expect.objectContaining({ action: 'toggle' }) + ) + }) + + it('appends the subdir fragment for multi-plugin repos', async () => { + render() + + window.dispatchEvent( + new MessageEvent('message', { + data: { + name: 'nested-plugin', + repo: 'https://github.com/example/plugins-monorepo', + subdir: 'nested-plugin', + type: 'hermes-plugin-pick' + }, + origin: 'https://hermes-agent.nousresearch.com' + }) + ) + + await waitFor(() => { + expect($pluginInstallRequest.get()?.repo).toBe( + 'https://github.com/example/plugins-monorepo#nested-plugin' + ) + }) + }) +}) + +describe('PluginsTab catalog UX', () => { + beforeEach(() => { + $agentPlugins.set([]) + $agentPluginsStatus.set('ready') + closePluginInstallRequest() + requestGateway.mockClear() + }) + + afterEach(cleanup) + + it('shows an Update chip when the catalog pin moved past the installed SHA', () => { + $agentPlugins.set([ + { + catalog_name: 'demo-weather', + catalog_sha: 'b'.repeat(40), + catalog_tier: 'community', + description: '', + installed_sha: 'a'.repeat(40), + key: 'demo-weather', + name: 'demo-weather', + source: 'git', + status: 'enabled', + update_available: true, + version: '1.0.0' + } + ]) + + render() + + expect(screen.getByRole('button', { name: `Update to ${'b'.repeat(8)}` })).toBeTruthy() + }) + + it('re-pins through plugins.manage update when the chip is clicked', async () => { + $agentPlugins.set([ + { + catalog_name: 'demo-weather', + catalog_sha: 'b'.repeat(40), + catalog_tier: 'community', + description: '', + installed_sha: 'a'.repeat(40), + key: 'demo-weather', + name: 'demo-weather', + source: 'git', + status: 'enabled', + update_available: true, + version: '1.0.0' + } + ]) + requestGateway.mockResolvedValue({ ok: true, unchanged: false, plugins: [] } as never) + + render() + + screen.getByRole('button', { name: `Update to ${'b'.repeat(8)}` }).click() + + await waitFor(() => + expect(requestGateway).toHaveBeenCalledWith( + 'plugins.manage', + expect.objectContaining({ action: 'update', name: 'demo-weather', profile: 'workbot' }) + ) + ) + }) + + it('refuses a catalog pick that is already installed and current', async () => { + $agentPlugins.set([ + { + catalog_name: 'demo-weather', + description: '', + installed_sha: 'a'.repeat(40), + key: 'demo-weather', + name: 'demo-weather', + source: 'git', + status: 'enabled', + update_available: false, + version: '1.0.0' + } + ]) + + render() + + window.dispatchEvent( + new MessageEvent('message', { + data: { + name: 'demo-weather', + repo: 'https://github.com/example/demo-weather', + type: 'hermes-plugin-pick' + }, + origin: 'https://hermes-agent.nousresearch.com' + }) + ) + + // The modal must NOT open — the pick is refused with a toast. + await new Promise(resolve => setTimeout(resolve, 20)) + expect($pluginInstallRequest.get()).toBeNull() + }) + + it('still opens the modal for an installed pick when an update is available', async () => { + $agentPlugins.set([ + { + catalog_name: 'demo-weather', + description: '', + installed_sha: 'a'.repeat(40), + key: 'demo-weather', + name: 'demo-weather', + source: 'git', + status: 'enabled', + update_available: true, + version: '1.0.0' + } + ]) + + render() + + window.dispatchEvent( + new MessageEvent('message', { + data: { + name: 'demo-weather', + repo: 'https://github.com/example/demo-weather', + type: 'hermes-plugin-pick' + }, + origin: 'https://hermes-agent.nousresearch.com' + }) + ) + + await waitFor(() => expect($pluginInstallRequest.get()).not.toBeNull()) + }) +}) diff --git a/apps/desktop/src/app/skills/plugins-tab.tsx b/apps/desktop/src/app/skills/plugins-tab.tsx new file mode 100644 index 0000000000..600b73aeac --- /dev/null +++ b/apps/desktop/src/app/skills/plugins-tab.tsx @@ -0,0 +1,298 @@ +import { useStore } from '@nanostores/react' +import { memo, useEffect, useMemo, useState } from 'react' + +import { useGatewayRequest } from '@/app/gateway/hooks/use-gateway-request' +import { Button } from '@/components/ui/button' +import { Switch } from '@/components/ui/switch' +import { Tip } from '@/components/ui/tooltip' +import type { ProfileScope } from '@/hermes' +import { useI18n } from '@/i18n' +import { Loader2, Package } from '@/lib/icons' +import { cn } from '@/lib/utils' +import { + $agentPluginBusy, + $agentPlugins, + $agentPluginsError, + $agentPluginsStatus, + type AgentPluginRow, + isDesktopRelevantPlugin, + loadAgentPlugins, + toggleAgentPlugin, + updateAgentPlugin +} from '@/store/agent-plugins' +import { notify } from '@/store/notifications' +import { $paneHeightOverride, setPaneHeightOverride } from '@/store/panes' +import { openPluginInstallRequest } from '@/store/plugin-install-request' + +import { PanelEmpty } from '../overlays/panel' + +// The REAL Plugin Catalog page (docs site) embedded as a one-click picker — +// the same pattern as the Skills tab's EmbeddedHubPicker. `?embed=picker` +// hides the docs chrome and adds "+ Add to this Agent" per card, which posts +// { type: 'hermes-plugin-pick', name, repo, sha, subdir, tier, installCmd } +// to the parent window. We validate the origin and open the shared +// dual-target install modal (agent half → catalog-pinned install into the +// scoped profile; desktop half → local app), so bundled agent+desktop +// packages install both halves in one flow. +const CATALOG_ORIGIN = 'https://hermes-agent.nousresearch.com' +const CATALOG_PICKER_URL = `${CATALOG_ORIGIN}/docs/plugins?embed=picker` + +const CATALOG_PANE_ID = 'capabilities-plugin-catalog' +const CATALOG_DEFAULT_PX = 380 +const CATALOG_COLLAPSED_PX = 4 + +interface PluginPickMessage { + installCmd?: string + name?: string + repo?: string + sha?: string + subdir?: string + tier?: string + type?: string +} + +/** Derive the bare profile name a `plugins.manage` call should target. */ +function profileParam(scope: ProfileScope): null | string { + if (!scope) { + return null + } + + return typeof scope === 'string' ? scope : (scope.profile ?? null) +} + +function PluginRow({ + row, + busy, + onToggle, + onUpdate +}: { + row: AgentPluginRow + busy: boolean + onToggle: (enable: boolean) => void + onUpdate?: () => void +}) { + const { t } = useI18n() + const address = row.key ?? '' + const canToggle = Boolean(address) + const enabled = row.status === 'enabled' + + return ( +
+ +
+
+ {row.name} + {row.version && v{row.version}} + {row.portable && ( + + {t.skills.plugins.portableBadge} + + )} + {row.catalog_name && ( + + + {row.catalog_tier === 'official' + ? t.skills.plugins.tierOfficial + : t.skills.plugins.tierCommunity} + + + )} + {row.update_available && onUpdate && ( + + )} +
+ {row.description && ( +
+ {row.description} +
+ )} +
+
+ {busy && } + {canToggle ? ( + + ) : ( + + + + + + )} +
+
+ ) +} + +/** Agent plugins for the Capabilities page: the scoped profile's installed + * plugins on top (toggleable), the live catalog picker underneath — same + * management-plus-discovery shape as the Skills tab. */ +export const PluginsTab = memo(function PluginsTab({ profile }: { profile: ProfileScope }) { + const { t } = useI18n() + const p = t.skills.plugins + const { requestGateway } = useGatewayRequest() + + const rows = useStore($agentPlugins) + const status = useStore($agentPluginsStatus) + const error = useStore($agentPluginsError) + const busyKey = useStore($agentPluginBusy) + + const scope = profileParam(profile) + + useEffect(() => { + void loadAgentPlugins(requestGateway, scope) + }, [requestGateway, scope]) + + const visible = useMemo(() => rows.filter(isDesktopRelevantPlugin), [rows]) + + // Catalog picker viewport (persisted height, collapse toggle) — same pane + // store contract as EmbeddedHubPicker. + const heightOverride = useStore($paneHeightOverride(CATALOG_PANE_ID)) + const height = heightOverride ?? CATALOG_DEFAULT_PX + const open = height > CATALOG_COLLAPSED_PX + const [pickerMounted, setPickerMounted] = useState(open) + + if (open && !pickerMounted) { + setPickerMounted(true) + } + + useEffect(() => { + if (!open) { + return undefined + } + + const onMessage = (event: MessageEvent) => { + if (event.origin !== CATALOG_ORIGIN) { + return + } + + const data = event.data as null | PluginPickMessage + + if (!data || data.type !== 'hermes-plugin-pick' || !data.name || !data.repo) { + return + } + + // Already installed at (or past) this pin in the scoped profile → + // tell the user instead of re-running the install ceremony. Rows with + // update_available keep their explicit Update chip in the list above. + const existing = $agentPlugins + .get() + .find(row => row.catalog_name === data.name || row.name === data.name) + + if (existing && !existing.update_available) { + notify({ kind: 'success', message: t.skills.plugins.alreadyInstalled(String(data.name)) }) + + return + } + + // Open the shared dual-target install modal: it probes the repo for + // agent/desktop halves, installs the agent half at the catalog pin + // into the scoped profile, and offers the desktop half locally. + openPluginInstallRequest({ + catalogName: String(data.name), + profile: scope, + repo: data.subdir ? `${String(data.repo)}#${String(data.subdir)}` : String(data.repo), + sha: data.sha ? String(data.sha) : undefined + }) + } + + window.addEventListener('message', onMessage) + + return () => window.removeEventListener('message', onMessage) + }, [open, scope, t]) + + return ( +
+
+ {status === 'error' ? ( + void loadAgentPlugins(requestGateway, scope)} size="sm"> + {t.skills.refresh} + + } + description={error ?? undefined} + icon="error" + title={p.loadFailed} + /> + ) : visible.length === 0 && status === 'ready' ? ( + + ) : ( +
+ {visible.map(row => ( + { + if (!row.key) { + return + } + + void toggleAgentPlugin(requestGateway, row.key, enable, p.toggleFailed(row.name), scope) + }} + onUpdate={ + row.update_available + ? () => { + void updateAgentPlugin(requestGateway, row.name, p.updateFailed(row.name), scope).then( + applied => { + if (applied) { + notify({ kind: 'success', message: p.updated(row.name) }) + } + } + ) + } + : undefined + } + row={row} + /> + ))} +
+ )} +
+ +
+
+ {p.catalogTitle} + +
+ {pickerMounted && ( +
+
+