From d8d9639d8e2a4a5438efca702db7f02bc2daee34 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:28:54 -0700 Subject: [PATCH] fix(telemetry): execution_backend skips background review / curator fork calls hermes.execution_backend.count still counted terminal/browser/code calls made under the background_review write origin (the review and curator forks), while hermes.terminal.outcome and file_edit already skip them. Owner ruling: match terminal.outcome. record_execution_backend now returns early when is_background_review() is set (one contextvar read, before any work). Doc: the execution_backend row lists the forks as not counted. Probe probe_internal.py (terminal_tool inside a background-review origin): before: [execution_backend {backend: local, kind: terminal, outcome: success} x1] after: [] test_background_review_fork_terminal_calls_are_not_backend_usage is RED on base (2 rows != 1), green here. --- hermes_cli/observability/shared_metrics_loop.py | 7 +++++-- tests/hermes_cli/test_shared_metrics_loop.py | 16 ++++++++++++++++ .../docs/developer-guide/relay-shared-metrics.md | 2 +- 3 files changed, 22 insertions(+), 3 deletions(-) diff --git a/hermes_cli/observability/shared_metrics_loop.py b/hermes_cli/observability/shared_metrics_loop.py index 4804273423..19892e8e2a 100644 --- a/hermes_cli/observability/shared_metrics_loop.py +++ b/hermes_cli/observability/shared_metrics_loop.py @@ -333,8 +333,11 @@ def unmetered_backend_calls() -> Iterator[None]: def record_execution_backend(kind: str, backend: Any, result: Any = None, *, error_class: str | None = None) -> Any: """Count one tool call that reached its backend; returns ``result`` unchanged. ``backend`` may be a - callable so resolving it costs nothing while collection is off.""" - if _UNMETERED.get(): + callable so resolving it costs nothing while collection is off. Calls the background review / + curator forks make are Hermes' own work, not the user's (terminal.outcome skips them too).""" + from tools.skill_provenance import is_background_review + + if _UNMETERED.get() or is_background_review(): return result _emit( "EXECUTION_BACKEND_MARK", execution_backend_fields, kind=kind, backend=backend, result=result, diff --git a/tests/hermes_cli/test_shared_metrics_loop.py b/tests/hermes_cli/test_shared_metrics_loop.py index f095276046..3d29db97d8 100644 --- a/tests/hermes_cli/test_shared_metrics_loop.py +++ b/tests/hermes_cli/test_shared_metrics_loop.py @@ -208,6 +208,22 @@ def test_foreground_terminal_timeout_is_a_failed_timeout(home): ] +def test_background_review_fork_terminal_calls_are_not_backend_usage(home): + import tools.skill_provenance as provenance + from tools.terminal_tool import terminal_tool + + token = provenance.set_current_write_origin(provenance.BACKGROUND_REVIEW) + try: + assert json.loads(terminal_tool("echo review"))["exit_code"] == 0 + finally: + provenance.reset_current_write_origin(token) + terminal_tool("echo user-work") + + assert _rows(home, "hermes.execution_backend.count") == [ + ({"backend": "local", "error_class": "none", "kind": "terminal", "outcome": "success"}, 1), + ] + + def test_path_completion_listing_is_not_user_backend_work(home): import tui_gateway.methods_complete as mc from tools.terminal_tool import terminal_tool diff --git a/website/docs/developer-guide/relay-shared-metrics.md b/website/docs/developer-guide/relay-shared-metrics.md index a4d0d92a64..202d82f369 100644 --- a/website/docs/developer-guide/relay-shared-metrics.md +++ b/website/docs/developer-guide/relay-shared-metrics.md @@ -307,7 +307,7 @@ itself ships. | `hermes.memory.op.count` | op (`add`/`replace`/`remove`/`read`/`search`/`other`), provider (`builtin`, a bundled memory plugin, else `plugin`), origin (`foreground`/`background_review`), outcome (`success`/`failed`/`rejected`) | Is the learning loop writing memory, who asks for it (the user's turn or the background review), and how often writes are refused or fail. Never the memory text. | | `hermes.curator.run.count` | trigger (`scheduled`/`manual`), outcome (`success`/`failed`/`skipped`), archived/merged/patched/created buckets | Does the skill curator run, and does it actually consolidate anything. Dry runs report `skipped`; a scheduled check that finds another process already running the pass records nothing. Never skill names. | | `hermes.delegation.run.count` | subagent-count bucket, depth (`1`–`3`, `gte_4`), mode (`foreground`/`background`), outcome (`success`/`partial`/`failed`/`cancelled`) | How wide and deep delegate_task fan-outs go and how often every child finishes. One row per call, however many completion units it splits into. | -| `hermes.execution_backend.count` | kind (`terminal`/`browser`/`code`), backend, outcome, error class | Which sandboxes carry real work and how reliable each is. Terminal backends are the `terminal.backend` values (else `other`); browser backends are `local`, `lightpanda`, `cdp`, `camofox`, `extension` or a bundled cloud provider (else `other`); execute_code is `local` or `remote`. A command's own nonzero exit is still a backend success, but a foreground command that hits its timeout is `failed`/`timeout`; terminal and execute_code calls a guard refuses before they reach the backend, and Hermes' own listings for TUI/Desktop path completion, are not counted. | +| `hermes.execution_backend.count` | kind (`terminal`/`browser`/`code`), backend, outcome, error class | Which sandboxes carry real work and how reliable each is. Terminal backends are the `terminal.backend` values (else `other`); browser backends are `local`, `lightpanda`, `cdp`, `camofox`, `extension` or a bundled cloud provider (else `other`); execute_code is `local` or `remote`. A command's own nonzero exit is still a backend success, but a foreground command that hits its timeout is `failed`/`timeout`; terminal and execute_code calls a guard refuses before they reach the backend, Hermes' own listings for TUI/Desktop path completion, and calls made by the background review and curator forks are not counted. | | `hermes.platform.health` | platform, event (`connect_ok`/`connect_failed`/`reconnect`/`disconnect`), error class (`auth`/`network`/`rate_limited`/`config`/`other`) | Which messaging platforms fail to connect or drop, and why. Classified from exception types, HTTP statuses and Hermes's own fatal codes, never error text. | | `hermes.platform.delivery` | platform, outcome (`sent`/`failed`), failure class (`rate_limited`/`too_long`/`auth`/`network`/`forbidden`/`other`) | How often replies fail to reach the user per platform (one count per logical reply, retries included). | | `hermes.gateway.reply_latency` | platform, first-response bucket (`lt_2s` … `gte_60s`) | Time from an accepted inbound message to the first visible reply text (stream first chunk or final message). |