diff --git a/hermes_cli/observability/shared_metrics_loop.py b/hermes_cli/observability/shared_metrics_loop.py index 4804273423..19892e8e2a 100644 --- a/hermes_cli/observability/shared_metrics_loop.py +++ b/hermes_cli/observability/shared_metrics_loop.py @@ -333,8 +333,11 @@ def unmetered_backend_calls() -> Iterator[None]: def record_execution_backend(kind: str, backend: Any, result: Any = None, *, error_class: str | None = None) -> Any: """Count one tool call that reached its backend; returns ``result`` unchanged. ``backend`` may be a - callable so resolving it costs nothing while collection is off.""" - if _UNMETERED.get(): + callable so resolving it costs nothing while collection is off. Calls the background review / + curator forks make are Hermes' own work, not the user's (terminal.outcome skips them too).""" + from tools.skill_provenance import is_background_review + + if _UNMETERED.get() or is_background_review(): return result _emit( "EXECUTION_BACKEND_MARK", execution_backend_fields, kind=kind, backend=backend, result=result, diff --git a/tests/hermes_cli/test_shared_metrics_loop.py b/tests/hermes_cli/test_shared_metrics_loop.py index f095276046..3d29db97d8 100644 --- a/tests/hermes_cli/test_shared_metrics_loop.py +++ b/tests/hermes_cli/test_shared_metrics_loop.py @@ -208,6 +208,22 @@ def test_foreground_terminal_timeout_is_a_failed_timeout(home): ] +def test_background_review_fork_terminal_calls_are_not_backend_usage(home): + import tools.skill_provenance as provenance + from tools.terminal_tool import terminal_tool + + token = provenance.set_current_write_origin(provenance.BACKGROUND_REVIEW) + try: + assert json.loads(terminal_tool("echo review"))["exit_code"] == 0 + finally: + provenance.reset_current_write_origin(token) + terminal_tool("echo user-work") + + assert _rows(home, "hermes.execution_backend.count") == [ + ({"backend": "local", "error_class": "none", "kind": "terminal", "outcome": "success"}, 1), + ] + + def test_path_completion_listing_is_not_user_backend_work(home): import tui_gateway.methods_complete as mc from tools.terminal_tool import terminal_tool diff --git a/website/docs/developer-guide/relay-shared-metrics.md b/website/docs/developer-guide/relay-shared-metrics.md index a4d0d92a64..202d82f369 100644 --- a/website/docs/developer-guide/relay-shared-metrics.md +++ b/website/docs/developer-guide/relay-shared-metrics.md @@ -307,7 +307,7 @@ itself ships. | `hermes.memory.op.count` | op (`add`/`replace`/`remove`/`read`/`search`/`other`), provider (`builtin`, a bundled memory plugin, else `plugin`), origin (`foreground`/`background_review`), outcome (`success`/`failed`/`rejected`) | Is the learning loop writing memory, who asks for it (the user's turn or the background review), and how often writes are refused or fail. Never the memory text. | | `hermes.curator.run.count` | trigger (`scheduled`/`manual`), outcome (`success`/`failed`/`skipped`), archived/merged/patched/created buckets | Does the skill curator run, and does it actually consolidate anything. Dry runs report `skipped`; a scheduled check that finds another process already running the pass records nothing. Never skill names. | | `hermes.delegation.run.count` | subagent-count bucket, depth (`1`–`3`, `gte_4`), mode (`foreground`/`background`), outcome (`success`/`partial`/`failed`/`cancelled`) | How wide and deep delegate_task fan-outs go and how often every child finishes. One row per call, however many completion units it splits into. | -| `hermes.execution_backend.count` | kind (`terminal`/`browser`/`code`), backend, outcome, error class | Which sandboxes carry real work and how reliable each is. Terminal backends are the `terminal.backend` values (else `other`); browser backends are `local`, `lightpanda`, `cdp`, `camofox`, `extension` or a bundled cloud provider (else `other`); execute_code is `local` or `remote`. A command's own nonzero exit is still a backend success, but a foreground command that hits its timeout is `failed`/`timeout`; terminal and execute_code calls a guard refuses before they reach the backend, and Hermes' own listings for TUI/Desktop path completion, are not counted. | +| `hermes.execution_backend.count` | kind (`terminal`/`browser`/`code`), backend, outcome, error class | Which sandboxes carry real work and how reliable each is. Terminal backends are the `terminal.backend` values (else `other`); browser backends are `local`, `lightpanda`, `cdp`, `camofox`, `extension` or a bundled cloud provider (else `other`); execute_code is `local` or `remote`. A command's own nonzero exit is still a backend success, but a foreground command that hits its timeout is `failed`/`timeout`; terminal and execute_code calls a guard refuses before they reach the backend, Hermes' own listings for TUI/Desktop path completion, and calls made by the background review and curator forks are not counted. | | `hermes.platform.health` | platform, event (`connect_ok`/`connect_failed`/`reconnect`/`disconnect`), error class (`auth`/`network`/`rate_limited`/`config`/`other`) | Which messaging platforms fail to connect or drop, and why. Classified from exception types, HTTP statuses and Hermes's own fatal codes, never error text. | | `hermes.platform.delivery` | platform, outcome (`sent`/`failed`), failure class (`rate_limited`/`too_long`/`auth`/`network`/`forbidden`/`other`) | How often replies fail to reach the user per platform (one count per logical reply, retries included). | | `hermes.gateway.reply_latency` | platform, first-response bucket (`lt_2s` … `gte_60s`) | Time from an accepted inbound message to the first visible reply text (stream first chunk or final message). |