diff --git a/acp_adapter/server.py b/acp_adapter/server.py index d9a5f0beaa..719766c1da 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -638,6 +638,10 @@ class HermesACPAgent(SlashCommandsMixin, acp.Agent): try: if state.agent: request_hard_interrupt(state.agent) + # Background delegations are detached from the turn's fan-out; they end with the cancel. + from tools.async_delegation import interrupt_for_session + interrupt_for_session(parent_session_id=str(getattr(state.agent, "session_id", "") or ""), + reason="acp_cancel") except Exception: logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True) logger.info("Cancelled session %s", session_id) diff --git a/agent/anthropic_credentials.py b/agent/anthropic_credentials.py index a514fa1aa9..27de5c798e 100644 --- a/agent/anthropic_credentials.py +++ b/agent/anthropic_credentials.py @@ -231,8 +231,8 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]: def claude_code_credentials_path() -> Path: """Claude Code's shared OAuth file; every profile reads/writes this same path. Honours ``CLAUDE_CONFIG_DIR`` - like the Claude CLI itself (blank = unset, as in ``hermes_cli.foreign_sessions``), so pointing it at an - empty directory opts a Hermes process out of borrowing the login.""" + like the Claude CLI itself (blank = unset, as in ``hermes_cli.foreign_sessions``). The supported opt-out of + borrowing the login is ``auth.adopt_external_logins: false`` in config.yaml.""" override = os.environ.get("CLAUDE_CONFIG_DIR", "").strip() root = Path(override).expanduser() if override else Path.home() / ".claude" return root / ".credentials.json" @@ -246,7 +246,13 @@ def _read_claude_code_credentials_from_file() -> Optional[Dict[str, Any]]: def read_claude_code_credentials() -> Optional[Dict[str, Any]]: """Read refreshable Claude Code OAuth credentials (Keychain and/or file). When both exist: prefer the only non-expired one (Claude Code 2.1.x refreshes one source but not the other), else the later ``expiresAt`` so a - refresh uses the freshest refreshToken. ~/.claude.json primaryApiKey is deliberately excluded.""" + refresh uses the freshest refreshToken. ~/.claude.json primaryApiKey is deliberately excluded. + + This is the only reader of the borrowed login, so ``auth.adopt_external_logins: false`` is enforced here: + every resolver, pool seed/sync and 401 refresher then sees "no Claude Code login" and never touches the file.""" + from agent.credential_sources import adopt_external_logins_enabled + if not adopt_external_logins_enabled(): + return None kc_creds = _read_claude_code_credentials_from_keychain() file_creds = _read_claude_code_credentials_from_file() if not (kc_creds and file_creds): diff --git a/agent/background_review.py b/agent/background_review.py index 5979b1134b..51bf3c70f2 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -151,9 +151,12 @@ _REVIEW_MAX_ITERATIONS = 16 # Aggregate INPUT-token budget for one review fork (checked in conversation_loop's # ``_review_input_budget_exhausted``). Request #1 replays the full snapshot as a warm cache read # (both compression gates deferred until the first response); compaction then bounds each -# request, but nothing else caps the SUM across the tool loop. 2x the historical 300k foreground -# trigger. Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. -_REVIEW_MAX_INPUT_TOKENS_DEFAULT = 600_000 +# request, but nothing else caps the SUM across the tool loop. The default leaves 25% of the +# review model's context window available and never exceeds the historical cloud-scale ceiling. +# Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. +_REVIEW_MAX_INPUT_TOKENS_CAP = 600_000 +_REVIEW_INPUT_CONTEXT_FRACTION = 0.75 +_REVIEW_MAX_INPUT_TOKENS_FALLBACK = 120_000 def _task_block(cfg: Any) -> Dict[str, Any]: @@ -175,13 +178,27 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) -> return {} -def _review_input_token_budget(task_cfg: Optional[Dict[str, Any]] = None) -> Optional[int]: - """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables).""" - raw = _background_review_task_config(task_cfg).get("max_input_tokens", _REVIEW_MAX_INPUT_TOKENS_DEFAULT) +def _context_derived_review_input_budget(review_agent: Any = None) -> int: + """Default budget: 75% of the review fork's resolved context window, capped at the historical + 600k ceiling. The fork's ``context_compressor.context_length`` is already resolved by + ``AIAgent.__init__`` (config overrides, catalog, endpoint probe) — no second lookup here. + Unknown window → a conservative fixed fallback so unattended review work stays bounded.""" + context_window = getattr(getattr(review_agent, "context_compressor", None), "context_length", None) + if not isinstance(context_window, int) or isinstance(context_window, bool) or context_window <= 0: + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + return min(_REVIEW_MAX_INPUT_TOKENS_CAP, max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION))) + + +def _review_input_token_budget( + task_cfg: Optional[Dict[str, Any]] = None, review_agent: Any = None, +) -> Optional[int]: + """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables). Unset + or malformed ``max_input_tokens`` → derived from ``review_agent``'s context window.""" + task = _background_review_task_config(task_cfg) try: - budget = int(raw) - except (TypeError, ValueError): - budget = _REVIEW_MAX_INPUT_TOKENS_DEFAULT + budget = int(task["max_input_tokens"]) + except (KeyError, TypeError, ValueError): + return _context_derived_review_input_budget(review_agent) return budget if budget > 0 else None @@ -978,7 +995,7 @@ def build_cache_parity_fork( _detach_fork_compression(review_agent) # Compaction bounds a single request; this bounds the WHOLE review (checked in # conversation_loop via _review_input_budget_exhausted). - review_agent._review_input_token_budget = _review_input_token_budget(task_cfg) + review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, review_agent) return review_agent, _rt, _routed diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 17cc31e747..e8c4d9ebba 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1701,6 +1701,7 @@ _FALLBACK_REASON_LABELS = { FailoverReason.provider_policy_blocked: "provider policy blocked the request", FailoverReason.content_policy_blocked: "content policy blocked the request", FailoverReason.format_error: "request format rejected", + FailoverReason.role_alternation: "adjacent same-role messages rejected", FailoverReason.invalid_encrypted_content: "encrypted reasoning state rejected", FailoverReason.multimodal_tool_content_unsupported: "multimodal tool content unsupported", FailoverReason.thinking_signature: "thinking signature rejected", diff --git a/agent/context_compressor.py b/agent/context_compressor.py index d2cf2a9d80..e2750fc09d 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1016,7 +1016,12 @@ _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") # MEDIA delivery directives must not reach the summarizer — if one leaks into the summary, the downstream # model may re-emit it as an active directive on the next turn, triggering bogus attachment sends (#14665). _MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") -_HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)") +# Pre-#44454 alias. A summarizer that still emits it must be replaced, not prepended. +_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" +_TASK_SNAPSHOT_HEADINGS = (HISTORICAL_TASK_HEADING, _LEGACY_ACTIVE_TASK_HEADING) +_HISTORICAL_TASK_SECTION_RE = re.compile( + rf"(?ms)^(?:{'|'.join(re.escape(heading) for heading in _TASK_SNAPSHOT_HEADINGS)})\s*\n.*?(?=^## |\Z)" +) def _redact_compaction_text(text: Any) -> str: @@ -3531,7 +3536,7 @@ PREVIOUS SUMMARY: NEW TURNS TO INCORPORATE: {content_to_summarize}{_memory_section} -Update the summary using this exact structure. PRESERVE all existing information that is still relevant. ADD new completed actions to the numbered list (continue numbering). Move items from "In Progress" to "Completed Actions" when done. Move answered questions to "Resolved Questions". Update "Active State" to reflect current state. Remove information only if it is clearly obsolete. CRITICAL: Update "## Active Task" to reflect the user's most recent unfulfilled input — this includes any question, decision request, or discussion turn that the assistant has not yet answered. Only write "None" if the last exchange was fully resolved. +Update the summary using this exact structure. PRESERVE all existing information that is still relevant. ADD new completed actions to the numbered list (continue numbering). Move items from "In Progress" to "Completed Actions" when done. Move answered questions to "Resolved Questions". Update "Active State" to reflect current state. Remove information only if it is clearly obsolete. CRITICAL: Update "{HISTORICAL_TASK_HEADING}" to reflect the user's most recent unfulfilled input — this includes any question, decision request, or discussion turn that the assistant has not yet answered. Only write "None" if the last exchange was fully resolved. {_template_sections}""" else: @@ -3784,8 +3789,8 @@ Write only the summary body. Do not include any preamble or prefix.""" """Reject user attribution when the source transcript has no user.""" if has_user_turn: return - match = re.search(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n(.*?)(?=\n##\s|\Z)", summary) - task_snapshot = match.group(1).strip() if match else "" + match = _HISTORICAL_TASK_SECTION_RE.search(summary) + task_snapshot = match.group(0).split("\n", 1)[-1].strip() if match else "" # The "User asked:" scan can false-positive on quoted tool output; acceptable, since # the RuntimeError only costs one retry on the existing fallback path. if task_snapshot != _NO_USER_TASK_SENTINEL or re.search(r"\bUser\s+asked\s*:", summary, re.IGNORECASE): @@ -3906,7 +3911,18 @@ Write only the summary body. Do not include any preamble or prefix.""" # this regex on the next compaction (deleting every following section). replacement = f"{HISTORICAL_TASK_HEADING}\n{snapshot}\n\n" if _HISTORICAL_TASK_SECTION_RE.search(body): - return _HISTORICAL_TASK_SECTION_RE.sub(lambda _m: replacement, body, count=1).strip() + # Replace the first task section and drop every later one: a summarizer that emits both the + # canonical heading and the legacy alias would otherwise leave a second, undisclaimed task section. + seen = False + + def _collapse(_m: re.Match) -> str: + nonlocal seen + if seen: + return "" + seen = True + return replacement + + return _HISTORICAL_TASK_SECTION_RE.sub(_collapse, body).strip() return f"{replacement}{body}".strip() @classmethod diff --git a/agent/credential_pool.py b/agent/credential_pool.py index 458c8c62aa..ede5f7d0e1 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -306,6 +306,40 @@ def label_from_token(token: str, fallback: str) -> str: return fallback +def _codex_principal_identity(access_token: Any) -> Optional[Tuple[str, str]]: + """``(chatgpt_account_id, sub)`` of a Codex access token, or None when either claim is missing. + + Decoded without signature verification: this only decides whether two credentials Hermes + already holds belong to the same principal, never whether a token is valid. Both claims are + required because members of one ChatGPT workspace share ``chatgpt_account_id`` yet have their + own subjects and quotas. + """ + claims = _decode_jwt_claims(access_token) + auth_claims = claims.get("https://api.openai.com/auth") if isinstance(claims, dict) else None + account_id = auth_claims.get("chatgpt_account_id") if isinstance(auth_claims, dict) else None + subject = claims.get("sub") if isinstance(claims, dict) else None + if not (isinstance(account_id, str) and account_id.strip() and isinstance(subject, str) and subject.strip()): + return None + return account_id.strip(), subject.strip() + + +def _codex_entry_tracks_singleton(entry: PooledCredential, singleton_tokens: Dict[str, Any]) -> bool: + """Whether a Codex pool entry may adopt the auth.json singleton's token pair. + + ``device_code`` IS the singleton. ``manual:device_code`` is ambiguous: a legacy alias of the + singleton (same account, must follow its rotations) or an independent account added with + ``hermes auth add openai-codex`` (must never be overwritten — adopting turned two logins into + one account, both hitting the same usage limit). Same principal proves the alias; unknown + identity fails closed. + """ + if entry.source == "device_code": + return True + if entry.source != SOURCE_MANUAL_DEVICE_CODE: + return False + entry_identity = _codex_principal_identity(entry.access_token) + return entry_identity is not None and entry_identity == _codex_principal_identity(singleton_tokens.get("access_token")) + + def _next_priority(entries: List[PooledCredential]) -> int: return max((entry.priority for entry in entries), default=-1) + 1 @@ -370,6 +404,23 @@ def _parse_absolute_timestamp(value: Any) -> Optional[float]: return None +def _singleton_predates_entry(state: Any, entry: "PooledCredential") -> bool: + """True only when the auth.json singleton is PROVABLY older than *entry*. + + Both sides stamp ``last_refresh`` on every successful rotation. When + either side lacks a parseable stamp this returns False (cannot prove), + which keeps the historical adopt-on-difference behavior (#70111) intact + for legacy writers. + """ + entry_ts = _parse_absolute_timestamp(entry.last_refresh) + if entry_ts is None: + return False + state_ts = _parse_absolute_timestamp(state.get("last_refresh") if isinstance(state, dict) else None) + if state_ts is None: + return False + return state_ts < entry_ts + + def _normalize_error_context(error_context: Optional[Dict[str, Any]]) -> Dict[str, Any]: if not isinstance(error_context, dict): return {} @@ -1043,6 +1094,8 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin) tokens = state.get("tokens") if isinstance(state, dict) else None if not isinstance(tokens, dict): return entry + if is_codex and not _codex_entry_tracks_singleton(entry, tokens): + return entry store_access = tokens.get("access_token", "") store_refresh = tokens.get("refresh_token", "") entry_refresh = entry.refresh_token or "" @@ -1063,6 +1116,23 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin) entry.id, ) should_adopt = True + if should_adopt and _singleton_predates_entry(state, entry): + # #106705: manual:* entries never write back to the singleton + # (#39236), so after a pool-side rotation the singleton sits + # one chain behind. Adopting it would replay the consumed + # refresh token. ``last_refresh`` is stamped on every + # successful rotation on both sides; when either side lacks a + # parseable stamp this falls through to the historical + # adopt-on-difference above (#70111). + logger.info( + "Pool entry %s: auth.json singleton predates this entry's " + "rotation (last_refresh %s < %s); keeping pool chain to " + "avoid replaying the consumed refresh token", + entry.id, + state.get("last_refresh") if isinstance(state, dict) else None, + entry.last_refresh, + ) + should_adopt = False if should_adopt: logger.debug( "Pool entry %s: syncing %s tokens from auth.json (refreshed by another process)", @@ -2182,11 +2252,16 @@ def _seed_anthropic_singletons(seed: _Seeder) -> None: read_claude_code_credentials, read_hermes_oauth_credentials, ) + from agent.credential_sources import adopt_external_logins_enabled - for source_name, creds in ( - ("hermes_pkce", read_hermes_oauth_credentials()), - ("claude_code", read_claude_code_credentials()), - ): + sources = [("hermes_pkce", read_hermes_oauth_credentials())] + if adopt_external_logins_enabled(): + sources.append(("claude_code", read_claude_code_credentials())) + else: + # Singleton-seeded rows are otherwise never pruned; the opt-out must also drop the row an + # earlier (adopting) process persisted, or it keeps rotating a login Hermes no longer reads. + seed.changed |= _retain_sources_not_in(seed.entries, {"claude_code"}) + for source_name, creds in sources: if creds and creds.get("accessToken"): seed.upsert(source_name, { "auth_type": AUTH_TYPE_OAUTH, diff --git a/agent/credential_sources.py b/agent/credential_sources.py index ba90e312f1..7a8ab1644c 100644 --- a/agent/credential_sources.py +++ b/agent/credential_sources.py @@ -7,14 +7,47 @@ Readers live in ``agent.credential_pool``; what is unified here is **removal**: dispatcher suppresses ``(provider, source_id)`` in auth.json so the seeding branch skips the upsert. Adding a source: wire a reader branch in ``_seed_from_*``, gate it behind ``is_source_suppressed``, register a step here. + +Also home to the one policy switch over *borrowed* CLI logins (Codex CLI's +``~/.codex/auth.json``, Claude Code's ``~/.claude/.credentials.json``): +``auth.adopt_external_logins`` in config.yaml. """ from __future__ import annotations +import logging import os from dataclasses import dataclass, field from typing import Callable, List, Optional +logger = logging.getLogger(__name__) + +EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE = ( + "External CLI logins (Codex CLI, Claude Code) are not adopted: auth.adopt_external_logins is false. " + "Hermes uses only its own logins; run `hermes auth add ` to add one." +) +_notice_logged = False + + +def adopt_external_logins_enabled() -> bool: + """``auth.adopt_external_logins`` (default True). + + Codex and Claude OAuth refresh tokens are single-use and rotate, so once Hermes borrows a CLI's + token pair the two programs hold one token family and whichever refreshes first logs the other + out. When the user opts out, Hermes never reads or refreshes those files and says so once per + process (INFO) the first time it would have.""" + global _notice_logged + try: + from hermes_cli.config import load_config_readonly + auth_cfg = (load_config_readonly() or {}).get("auth") + except Exception: + return True + enabled = not isinstance(auth_cfg, dict) or bool(auth_cfg.get("adopt_external_logins", True)) + if not enabled and not _notice_logged: + _notice_logged = True + logger.info(EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE) + return enabled + @dataclass class RemovalResult: diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 29bb23bee0..d6b0088d32 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -50,6 +50,7 @@ class FailoverReason(enum.Enum): provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged format_error = "format_error" # 400 bad request — abort or strip + retry + role_alternation = "role_alternation" # Strict chat template rejected adjacent same-role messages — merge them for this destination and retry invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry multimodal_tool_content_unsupported = "multimodal_tool_content_unsupported" # Provider rejected list-type content in tool messages (e.g. Xiaomi MiMo) — downgrade to text and retry reasoning_mandatory = "reasoning_mandatory" # Route rejects reasoning: {enabled: false} — send the disable no more this session and retry @@ -277,6 +278,19 @@ _INVALID_MESSAGE_BODY_PATTERNS = ( "messages: at least one message is required", _NO_USER_QUERY_SIGNAL, ) +# Strict-alternation chat templates (llama.cpp / vLLM Jinja templates, Mistral, some +# OpenRouter routes) 400 when two adjacent messages share a role. Deterministic for the +# request shape, and the only bad thing is the adjacency, so the caller that produced it +# (the MoA aggregator appends ``user(guidance)`` after ``user(task)`` on iteration 1 — +# #112358) merges the pair for THAT destination and retries once. Checked before the +# request-validation table: the body usually also carries ``invalid_request_error``. +_ROLE_ALTERNATION_PATTERNS = ( + "roles must alternate", "role must alternate", "must alternate between", + "consecutive user messages", "consecutive messages with the same role", + "consecutive messages of the same role", "same role in a row", "multiple user messages in a row", + "adjacent messages with the same role", +) + # Proxy-side rejection of the model's own tool-call JSON (Ollama "invalid tool call arguments", # OpenRouter-wrapped "function_call arguments"). Checked before the generic 400 validation and # overflow heuristics: on a large session the bare message would otherwise read as overflow. @@ -428,6 +442,9 @@ _V_OVERLOADED, _V_SERVER_ERROR, _V_TIMEOUT, _V_UNKNOWN = map(_v, (_R.overloaded, _V_IMAGE_TOO_LARGE, _V_IMAGE_CORRUPT = _v(_R.image_too_large), _v(_R.image_corrupt) _V_MULTIMODAL, _V_INVALID_ENCRYPTED = _v(_R.multimodal_tool_content_unsupported), _v(_R.invalid_encrypted_content) _V_REASONING_MANDATORY = _v(_R.reasoning_mandatory, should_compress=False, should_fallback=False) +# Same recovery hints as format_error: consumers without a merge-and-retry step (the main loop +# already merges adjacent users before the call) keep aborting to the fallback chain. +_V_ROLE_ALTERNATION = _v(_R.role_alternation, **_ABORT_FALLBACK) # The MODEL emitted unparseable tool-call JSON and the proxy (Ollama, OpenRouter) rejected it: no # other provider can fix that output, so falling back only replays the same broken turn 4-5 times # (20-60s per occurrence, #12770). Abort this call; the loop's argument repair handles the retry. @@ -524,7 +541,8 @@ _400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + ( # Status-less message path, head (before usage-limit disambiguation). _MESSAGE_HEAD_RULES = ((_MEMORY_CEILING_PATTERNS, _V_OVERLOADED), - (_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE)) + _IMAGE_TOOL_RULES + (_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE), + (_ROLE_ALTERNATION_PATTERNS, _V_ROLE_ALTERNATION)) + _IMAGE_TOOL_RULES # Status-less tail. Overload before rate_limit/billing so "overloaded" backs off # instead of rotating; policy block before model_not_found; timeout/connection @@ -966,6 +984,8 @@ def _classify_400(c: _Ctx) -> Verdict: return _V_SERVER_ERROR if any(p in msg for p in _MALFORMED_TOOL_ARGS_PATTERNS): return _V_MALFORMED_TOOL_ARGS + if any(p in msg for p in _ROLE_ALTERNATION_PATTERNS): + return _V_ROLE_ALTERNATION # Before overflow: GPT-5's "Unsupported parameter: 'max_tokens'" contains it. if any(p in msg for p in _400_VALIDATION_PATTERNS) or code in _400_VALIDATION_CODES: return _V_FORMAT_ERROR diff --git a/agent/moa_alternation.py b/agent/moa_alternation.py new file mode 100644 index 0000000000..4e94608b8c --- /dev/null +++ b/agent/moa_alternation.py @@ -0,0 +1,64 @@ +"""Reactive same-role merge for the MoA aggregator request (#112358, last atom). + +The aggregator request deliberately ends ``user(task), user(guidance)`` on iteration 1 of a +turn: the guidance is its own trailing message so every earlier message stays byte-stable and +the provider prefix cache keeps growing (``moa_loop._attach_reference_guidance``). Strict- +alternation chat templates (llama.cpp / vLLM Jinja templates, Mistral, some OpenRouter routes) +400 on that adjacency. Merging proactively for everyone would re-introduce the divergence the +split shape removed and only move the 400, so the merge is reactive and destination-scoped: + +* a 400 classified ``FailoverReason.role_alternation`` → retry ONCE with adjacent same-role + messages merged; +* the destination (``base_url``/provider + model) is remembered on the facade for the rest of + the session, so later iterations pre-merge for it and never pay the 400 again; +* destinations that accepted the split shape are never touched — their prefix stays byte-stable. +""" + +from __future__ import annotations + +import logging +from typing import Any + +logger = logging.getLogger(__name__) + + +def destination_key(runtime: dict[str, Any]) -> tuple[str, str]: + """``(route, model)`` identity of an aggregator destination: the base_url when the slot + resolved one (two providers can share a model id), else the provider slug.""" + route = str(runtime.get("base_url") or runtime.get("provider") or "").strip().rstrip("/") + return route, str(runtime.get("model") or "").strip() + + +def merge_same_role_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Return ``messages`` with adjacent user turns folded into one (the only same-role adjacency + the aggregator request produces); other rows are shared, never mutated. Returns the input + object itself when nothing merged so callers can detect a no-op.""" + from agent.agent_runtime_helpers import _UNMERGEABLE, _merge_user_content + + merged: list[dict[str, Any]] = [] + changed = False + for message in messages: + prev = merged[-1] if merged else None + content: Any = _UNMERGEABLE + if prev is not None and prev.get("role") == "user" and message.get("role") == "user": + content = _merge_user_content(prev.get("content", ""), message.get("content", "")) + if content is _UNMERGEABLE: + merged.append(message) + continue + merged[-1] = {**prev, "content": content} + changed = True + return merged if changed else messages + + +def is_role_alternation_rejection(exc: Exception, runtime: dict[str, Any]) -> bool: + """True when the aggregator destination rejected the request for adjacent same-role messages.""" + from agent.error_classifier import FailoverReason, classify_api_error + + try: + classified = classify_api_error( + exc, provider=str(runtime.get("provider") or ""), model=str(runtime.get("model") or ""), + base_url=str(runtime.get("base_url") or ""), + ) + except Exception: # pragma: no cover - classification must never mask the original error + return False + return classified.reason is FailoverReason.role_alternation diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 21b747fac8..5ce755d512 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -22,6 +22,7 @@ from typing import Any from agent.auxiliary_client import call_llm from agent.message_content import flatten_message_text +from agent.moa_alternation import destination_key, is_role_alternation_rejection, merge_same_role_messages from agent.transports import get_transport from agent.usage_pricing import CanonicalUsage @@ -957,6 +958,10 @@ class MoAChatCompletions: self._fanout_turn_sig: str | None = None self._fanout_last_state_sig: str | None = None self._privacy_mode: str = "" # normalized moa.privacy_filter, refreshed per create() + # Destinations (route, model) that 400'd on adjacent same-role messages this session: + # their aggregator requests are pre-merged; every other destination keeps the split, + # cache-stable shape (agent/moa_alternation.py). + self._merge_same_role_destinations: set[tuple[str, str]] = set() def consume_reference_usage(self) -> tuple[Any, Any]: """Pop pending fan-out ``(CanonicalUsage, cost_usd_or_None)`` and reset both @@ -1081,10 +1086,6 @@ class MoAChatCompletions: ) trace = self._pending_trace if trace is not None: - # Trace the exact aggregator INPUT (persisted copy redacted; live input raw). - trace["aggregator_input_messages"] = ( - _redact_trace_messages([dict(m) for m in agg_messages]) if getattr(self, "_privacy_mode", "") else agg_messages - ) trace["aggregator_label"] = _slot_label(aggregator) # stream=True returns the RAW token stream (consumer reassembles + retries); # the non-streaming path forwards no stream/stream_options/timeout. The @@ -1097,13 +1098,40 @@ class MoAChatCompletions: stream_kwargs["timeout"] = api_kwargs["timeout"] # Pop the runtime's extra_body so the explicit kwarg never collides with **agg_runtime. agg_extra_body = _merge_slot_extra_body(agg_runtime.pop("extra_body", None), api_kwargs.get("extra_body")) - agg_response = call_llm( - task="moa_aggregator", messages=agg_messages, temperature=prepared["aggregator_temperature"], + destination = destination_key(agg_runtime) + # Facades built via __new__ (tests, swaps) have no __init__ state. + remembered = getattr(self, "_merge_same_role_destinations", None) + if remembered is None: + remembered = self._merge_same_role_destinations = set() + merged = destination in remembered + if merged: + agg_messages = merge_same_role_messages(agg_messages) + send = functools.partial( + call_llm, task="moa_aggregator", temperature=prepared["aggregator_temperature"], max_tokens=api_kwargs.get("max_tokens"), tools=tools, extra_body=agg_extra_body, reasoning_config=_aggregator_reasoning_config(aggregator), # same policy as direct create() **stream_kwargs, **agg_runtime, ) + try: + agg_response = send(messages=agg_messages) + except Exception as exc: + # Strict-alternation template rejected ``user(task), user(guidance)``: merge the pair for + # THIS destination only and retry once; remember it so later iterations pre-merge. + retry_messages = None if merged else merge_same_role_messages(agg_messages) + if retry_messages is None or retry_messages is agg_messages or not is_role_alternation_rejection(exc, agg_runtime): + raise + remembered.add(destination) + logger.warning( + "MoA aggregator %s rejected adjacent same-role messages — merging them for this " + "destination for the rest of the session and retrying once: %.200s", _slot_label(aggregator), exc, + ) + agg_messages = retry_messages + agg_response = send(messages=agg_messages) if trace is not None: + # Trace the exact aggregator INPUT as sent (persisted copy redacted; live input raw). + trace["aggregator_input_messages"] = ( + _redact_trace_messages([dict(m) for m in agg_messages]) if getattr(self, "_privacy_mode", "") else agg_messages + ) # Streaming output lands as the turn's assistant message; the trace marks it. trace["aggregator_streamed"] = stream output = None diff --git a/agent/turn_failure_copy.py b/agent/turn_failure_copy.py index ffd7f06bea..fec973dd59 100644 --- a/agent/turn_failure_copy.py +++ b/agent/turn_failure_copy.py @@ -154,6 +154,10 @@ _NONRETRYABLE_COPY: Dict[str, str] = { "{label} rejected this request as malformed, so the model didn't answer. Start a clean " "session with /new or switch models with /model; if it keeps happening, run `hermes doctor`." ), + FailoverReason.role_alternation.value: ( + "{label} requires user and assistant turns to strictly alternate and rejected this " + "conversation's shape. Start a clean session with /new or switch models with /model." + ), FailoverReason.ssl_cert_verification.value: ( "Hermes couldn't verify {label}'s security certificate, so the connection was refused. " "This is usually a corporate proxy or an outdated certificate store on this computer — " @@ -253,6 +257,17 @@ _ONE_OFF_COPY: Dict[str, str] = { "your settings (compression.enabled). Run /compress to shrink it now, /new to start " "fresh, or pick a model with a bigger context window." ), + # Wording deliberately avoids the overflow phrases gateway/run_turn.py matches on + # (``_CONTEXT_OVERFLOW_ERROR_PHRASES``): this failure is transient, so the user's + # message must stay in the transcript and the session must not be auto-reset. + "server_context_rejection": ( + "The model server rejected this request as too large, but this conversation is only " + "about {tokens:,} tokens — well under the {window:,}-token window Hermes knows for " + "{model} — so shrinking it would not help. Another request on the same server (for " + "example a background memory review from an earlier session) was probably holding its " + "capacity, or the server runs {model} with a smaller window than Hermes assumes. Wait a " + "moment and send /retry; if it keeps happening, check the server's context setting." + ), "stream_dropped_tool_call": ( "The connection to {label} kept dropping while the model was writing a large action, " "so nothing was run. Check your network and send /retry; asking for the file in smaller " diff --git a/agent/turn_overflow.py b/agent/turn_overflow.py index 925a328ccb..e4b0066843 100644 --- a/agent/turn_overflow.py +++ b/agent/turn_overflow.py @@ -10,6 +10,7 @@ estimators that tests patch on the loop module are imported lazily inside the ha from __future__ import annotations import logging +import re import time from dataclasses import dataclass from typing import Any, Dict, List, Optional, Tuple @@ -33,6 +34,11 @@ logger = logging.getLogger("agent.conversation_loop") _RETRY_HINT = " 💡 Try /new to start a fresh conversation, or /compress to retry compression." +# A provider "context exceeded" whose request (incl. the output reservation) is under this share +# of the known window is not explained by the transcript; the rough estimator is ±30%, so a real +# overflow (request ≈ window) never lands below half. +_UNEXPLAINED_REJECTION_FRACTION = 0.5 + _GITHUB_MODELS_HINT = ( " 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every", " request at ~8K tokens. Hermes' system prompt + tool schemas baseline", @@ -88,7 +94,8 @@ class _Recovery(OverflowVerdict): def fail_turn( self, final_response: str, *, notices: tuple = (), log: Optional[tuple] = None, - compression_exhausted: bool = True, **extra: Any, + compression_exhausted: bool = True, reason: str = "context_overflow", + retryable: bool = False, **extra: Any, ) -> OverflowVerdict: """End the turn as failed/partial. ``notices`` flush the buffered retry trace first so the user sees what compression attempts were made.""" @@ -108,7 +115,7 @@ class _Recovery(OverflowVerdict): "error": final_response, "partial": True, "failed": True, - }, "context_overflow", False) + }, reason, retryable) if compression_exhausted: # Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The # bloated transcript remains intact while future input can move to a clean session instead of @@ -346,12 +353,12 @@ def _adopt_provider_context_limit(st: _Recovery, error_msg: str, old_ctx: int) - if is_minimax_provider and "context window exceeds limit (" in error_msg: agent._buffer_vprint( f"Provider reported overflow amount only; " - f"keeping context_length at {old_ctx:,} tokens and compressing." + f"keeping context_length at {old_ctx:,} tokens." ) else: agent._buffer_vprint( f"⚠️ Context length exceeded, but provider did not report a max context length; " - f"keeping context_length at {old_ctx:,} tokens and compressing." + f"keeping context_length at {old_ctx:,} tokens." ) return None @@ -387,6 +394,35 @@ def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: st new_ctx = _adopt_provider_context_limit(st, error_msg, old_ctx) + # A rejection the transcript cannot explain (#114644): the request sits far below the window + # Hermes knows for this model (after adopting any limit the server reported), so compressing + # would destroy history for nothing. Single-slot local servers reject like this while ANOTHER + # request — a background review from an earlier session — holds their context. Name that, + # keep the turn retryable and transient: no "conversation too long", no gateway auto-reset. + # Only when the server quoted NO measurement of its own: "prompt is too long: 233153 tokens + # > 200000" is the server's count and beats the local estimate. + window = agent.context_compressor.context_length + request_tokens = st.request_tokens() + max(0, int(getattr(agent, "max_tokens", 0) or 0)) + if ( + not re.search(r"\d{4,}", error_msg) + and isinstance(window, int) and window > 0 + and request_tokens < window * _UNEXPLAINED_REJECTION_FRACTION + ): + return st.fail_turn( + site_copy("server_context_rejection", model=agent.model, tokens=request_tokens, window=window), + notices=( + f"❌ The server rejected the request as too large, but it is only ~{request_tokens:,} " + f"tokens against a {window:,}-token window — not compressing.", + " 💡 Wait a moment and /retry; another request on the same server (e.g. a background " + "review) may have been holding its context.", + ), + log=( + "%sProvider context rejection not explained by request size (~%s of %s tokens); " + "skipping compression: %s", agent.log_prefix, f"{request_tokens:,}", f"{window:,}", error_msg[:200], + ), + compression_exhausted=False, reason=FailoverReason.server_error.value, retryable=True, + ) + exhausted = st.count_attempt() if exhausted is not None: return exhausted diff --git a/apps/desktop/AGENTS.md b/apps/desktop/AGENTS.md index ae540cca9d..34f6bab173 100644 --- a/apps/desktop/AGENTS.md +++ b/apps/desktop/AGENTS.md @@ -69,6 +69,15 @@ rename sibling of `dropTilesForProfile`. Add any new profile-keyed localStorage family to BOTH, or a rename leaves it pointing at a backend that no longer exists ("Couldn't open this session" on every restore). +When an id is verifiably gone anyway (`goneSessionVerdict` → `'draft'`), the +window drops to a fresh draft without toasting or looping — and the unsent +text stashed under the dead key follows it: the verdict calls +`announceGoneSessionDraft(id)` and the composer's swap onto the fresh scope +consumes it once (`adoptGoneSessionDraft`, `store/composer.ts`), seeding the +composer and publishing the inline, undoable `$restoredDraftNotice`. Offer, +don't hijack: no navigation beyond the drop itself, no focus steal, no toast, +and an already non-empty fresh draft is never clobbered. + ## Server truth is cached, not owned The renderer paints from a cache of backend truth, so it must reconcile, not diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx index c413d81be7..50c61d19da 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx @@ -4,9 +4,12 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility' import { + $restoredDraftNotice, + announceGoneSessionDraft, announceNewSessionDraftKey, clearSessionDraft, type ComposerAttachment, + dismissRestoredDraftNotice, mainComposerScope, stashSessionDraft, takeSessionDraft @@ -122,6 +125,42 @@ describe('useComposerDraft — attachment scope stays coherent with the committe clearSessionDraft('session-created') }) + it('carries the unsent draft of a GONE session into the fresh chat once, with an undoable notice (#111868)', () => { + stashSessionDraft('session-gone', 'typed into a session that no longer exists', []) + + const { rerender } = render( + undefined} sessionId="session-gone" /> + ) + + // The resume's gone verdict announces the dead key, then drops the window + // to a fresh draft (route → /new, scope → the pre-session bucket). + announceGoneSessionDraft('session-gone') + act(() => { + rerender( undefined} sessionId="" />) + }) + + expect(takeSessionDraft(null).text).toBe('typed into a session that no longer exists') + expect(takeSessionDraft('session-gone').text).toBe('') + expect($restoredDraftNotice.get()).toEqual({ + fromKey: 'session-gone', + text: 'typed into a session that no longer exists' + }) + + // Fires once: a later trip through the fresh draft finds nothing to move + // and does not re-publish the notice the user already dismissed. + dismissRestoredDraftNotice() + act(() => { + rerender( undefined} sessionId="session-A" />) + }) + act(() => { + rerender( undefined} sessionId="" />) + }) + + expect($restoredDraftNotice.get()).toBeNull() + expect(takeSessionDraft(null).text).toBe('typed into a session that no longer exists') + clearSessionDraft(null) + }) + it('leaves the pre-session draft in its bucket when the user opens another session from a fresh chat', () => { stashSessionDraft(null, 'still composing a new chat', []) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts index 423572b1b6..3d6516c473 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts @@ -14,6 +14,7 @@ import { isElementInHiddenPane } from '@/components/pane-shell/pane-visibility' import { sanitizeComposerInput } from '@/lib/composer-input-sanitize' import { useStoreSelector } from '@/lib/use-session-slice' import { + adoptGoneSessionDraft, adoptNewSessionDraft, type ComposerAttachment, type ComposerDraftSyncMode, @@ -464,6 +465,14 @@ export function useComposerDraft({ // the route flips the scope, so it is not a usable signal here. if (!draftScopeRef.current && activeQueueSessionKey) { adoptNewSessionDraft(activeQueueSessionKey) + } else if (!activeQueueSessionKey) { + // The reverse handoff: a session the user was typing into turned out + // to be gone and the window dropped to a fresh draft (#111868). The + // outgoing composer's cleanup has already stashed the live text under + // the dead key — whether that was this instance's previous scope or an + // unmounted one's — so move it into the fresh draft when the gone + // verdict announced it. No announcement, no-op. + adoptGoneSessionDraft() } draftScopeRef.current = activeQueueSessionKey diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index 957aba3f69..dfb440c6f4 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -71,6 +71,7 @@ import { shouldConvertPasteToAttachment } from './large-paste' import { ActionBadges } from './micro-actions' import { chipTypedPathOnSpace, pathifyRefs } from './path-refs' import { QueuePanel } from './queue-panel' +import { RestoredDraftNotice } from './restored-draft-notice' import { beginComposerComposition, composerPlainText, @@ -1439,6 +1440,11 @@ export function ChatBar({ additions beside the "+" menu and before the controls. All four render nothing until something contributes. */} + {queueEdit && editingQueuedPrompt && ( diff --git a/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx b/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx new file mode 100644 index 0000000000..609a4d3cab --- /dev/null +++ b/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx @@ -0,0 +1,63 @@ +import { useStore } from '@nanostores/react' + +import { Button } from '@/components/ui/button' +import { useI18n } from '@/i18n' +import { $restoredDraftNotice, dismissRestoredDraftNotice, undoRestoredDraft } from '@/store/composer' + +interface RestoredDraftNoticeProps { + /** The composer is showing the fresh draft (no session scope). */ + freshDraft: boolean + /** Clear the editor after Undo emptied the fresh draft. */ + onUndone: () => void + /** Latest live editor text — Undo only applies while it is still what was restored. */ + readLiveText: () => string +} + +/** + * "Restored your unsent message" strip above the fresh draft's input + * (#111868). Offers, never hijacks: no focus move, no navigation, no toast — + * the text is simply in the composer with a way to put it back. Renders + * nothing outside the fresh draft, so opening another session hides it + * without consuming the Undo. + */ +export function RestoredDraftNotice({ freshDraft, onUndone, readLiveText }: RestoredDraftNoticeProps) { + const notice = useStore($restoredDraftNotice) + const { t } = useI18n() + + if (!notice || !freshDraft) { + return null + } + + return ( +
+
{t.composer.restoredDraftNotice}
+
+ + +
+
+ ) +} diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts index db87364455..876ea7d9f3 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts @@ -17,6 +17,7 @@ import { resolveDesktopCommand } from '@/lib/desktop-slash-commands' import { isMissingRpcMethod } from '@/lib/gateway-rpc' +import { applyReasoningSlashResult, reasoningSlashParams } from '@/lib/reasoning-slash' import { setSessionYolo } from '@/lib/yolo-session' import { openCommandPalettePage } from '@/store/command-palette' import { setComposerDraft } from '@/store/composer' @@ -771,6 +772,46 @@ export function useSlashCommand(deps: SlashCommandDeps) { compressInFlightRef.current.delete(sessionId) } }, + // /reasoning runs the gateway's `config.set key=reasoning` — the Ink + // TUI's path. Through slash.exec the display words only reached + // config.yaml and the Thinking gate ($showReasoning) waited for the + // next config refresh; the effort level was set on a throwaway CLI. + reasoning: async ctx => { + const resolved = await withSlashOutput(ctx) + + if (!resolved) { + return + } + + const { render: renderSlashOutput, sessionId } = resolved + const params = reasoningSlashParams(ctx.arg, sessionId) + + try { + if (!params) { + const current = await requestGateway<{ display?: string; value?: string }>('config.get', { + key: 'reasoning', + session_id: sessionId + }) + + renderSlashOutput(`reasoning: ${current.value || 'medium'} · display ${current.display || 'hide'}`) + + return + } + + const result = await requestGateway<{ value?: string }>('config.set', params) + + applyReasoningSlashResult(result.value) + renderSlashOutput(`reasoning: ${result.value || params.value}`) + } catch (err) { + if (isMissingRpcMethod(err)) { + await runExec(ctx) + + return + } + + renderSlashOutput(`error: ${err instanceof Error ? err.message : String(err)}`) + } + }, // /yolo maps to the status-bar YOLO control — a per-session approval // bypass, same scope as the TUI's Shift+Tab. With no session yet we arm // it locally; the session-create path applies it on the first message. diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index d211801b9b..a59532638a 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -27,7 +27,7 @@ import { isMissingRpcMethod } from '@/lib/gateway-rpc' import { recoverInFlightTurnJournal } from '@/lib/inflight-turn-journal' import { setSessionYolo } from '@/lib/yolo-session' import { $clarifyRequests } from '@/store/clarify' -import { announceNewSessionDraftKey, migrateSessionDraft } from '@/store/composer' +import { announceGoneSessionDraft, announceNewSessionDraftKey, migrateSessionDraft } from '@/store/composer' import { clearQueuedPrompts, migrateQueuedPrompts } from '@/store/composer-queue' import { $connectionRequests } from '@/store/connection-request' import { @@ -2171,6 +2171,11 @@ export function useSessionActions({ return } + // The id is verifiably dead, but the text the user typed into it is + // still stashed under that key (#111868). Announce it so the + // composer's swap onto the fresh draft carries it over with an + // inline, undoable notice instead of leaving it stranded. + announceGoneSessionDraft(storedSessionId) startFreshSessionDraft(true) return diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 0f956724ee..ea6ff57a25 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2133,6 +2133,8 @@ export const ar = defineLocale({ attachments: count => `${count} مرفق`, editingInComposer: 'جار التحرير في صندوق الكتابة', editingQueuedInComposer: 'جار تحرير رسالة في الطابور', + restoredDraftNotice: 'تمت استعادة رسالتك غير المُرسلة', + restoredDraftUndo: 'تراجع', queueEdit: 'تحرير الرسالة المجدولة', queueSendNext: 'إرسالها تاليا', queueSteer: 'توجيه — تصحيح الدور الجاري فورا', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 0ac7d3bf0a..e00258d837 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -3021,6 +3021,8 @@ export const en: Translations = { attachments: count => `${count} attachment${count === 1 ? '' : 's'}`, editingInComposer: 'Editing in composer', editingQueuedInComposer: 'Editing queued turn in composer', + restoredDraftNotice: 'Restored your unsent message', + restoredDraftUndo: 'Undo', queueEdit: 'Edit', queueSendNext: 'Next', queueSteer: 'Steer — redirect the live turn now', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 532acf74ee..d79db66fa0 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2499,6 +2499,8 @@ export const ja = defineLocale({ attachments: count => `${count} 件の添付`, editingInComposer: 'コンポーザーで編集中', editingQueuedInComposer: 'コンポーザーでキュー済みターンを編集中', + restoredDraftNotice: '未送信のメッセージを復元しました', + restoredDraftUndo: '元に戻す', queueEdit: '編集', queueSendNext: '次に送信', queueSteer: 'ステア — 現在のターンを今すぐ修正', diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index 028202725e..e7263f59a5 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -2764,6 +2764,8 @@ export const ru = defineLocale({ attachments: count => `${count} ${RU_NOUN(count, 'вложение', 'вложения', 'вложений')}`, editingInComposer: 'Редактирование в композере', editingQueuedInComposer: 'Редактирование хода в очереди в композере', + restoredDraftNotice: 'Восстановлено ваше неотправленное сообщение', + restoredDraftUndo: 'Отменить', queueEdit: 'Изменить', queueSendNext: 'Дальше', queueSteer: 'Направить — изменить текущий ход сейчас', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index eedc45dfad..fabcf2a5e7 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2602,6 +2602,8 @@ export interface Translations { attachments: (count: number) => string editingInComposer: string editingQueuedInComposer: string + restoredDraftNotice: string + restoredDraftUndo: string queueEdit: string queueSendNext: string queueSend: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 4518b02ae2..2ef86a0837 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2482,6 +2482,8 @@ export const zhHant = defineLocale({ attachments: count => `${count} 個附件`, editingInComposer: '在輸入框中編輯', editingQueuedInComposer: '在輸入框中編輯排隊回合', + restoredDraftNotice: '已還原你未送出的訊息', + restoredDraftUndo: '復原', queueEdit: '編輯', queueSendNext: '下一個', queueSteer: '引導 — 立即修正目前回合', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index cd24e038de..31adf4ce57 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3156,6 +3156,8 @@ export const zh = defineLocale({ attachments: count => `${count} 个附件`, editingInComposer: '正在输入框中编辑', editingQueuedInComposer: '正在输入框中编辑排队回合', + restoredDraftNotice: '已恢复你未发送的消息', + restoredDraftUndo: '撤销', queueEdit: '编辑', queueSendNext: '下一个', queueSteer: '引导 — 立即修正当前回合', diff --git a/apps/desktop/src/lib/desktop-slash-commands.ts b/apps/desktop/src/lib/desktop-slash-commands.ts index f1decc521d..b3e3668c43 100644 --- a/apps/desktop/src/lib/desktop-slash-commands.ts +++ b/apps/desktop/src/lib/desktop-slash-commands.ts @@ -66,6 +66,7 @@ export type DesktopActionId = | 'new' | 'pet' | 'profile' + | 'reasoning' | 'skin' | 'stop' | 'title' @@ -188,6 +189,12 @@ const DESKTOP_COMMAND_SPECS: readonly DesktopCommandSpec[] = [ surface: action('branch') }, { name: '/yolo', description: 'Toggle YOLO — auto-approve dangerous commands', surface: action('yolo') }, + { + name: '/reasoning', + description: 'Reasoning effort or display [ [--global]|show|hide|full|clamp]', + surface: action('reasoning'), + argumentMode: 'options' + }, { name: '/wake', description: 'Control the desktop wake-word listener [on|off|status]', diff --git a/apps/desktop/src/lib/desktop-slash-registry.json b/apps/desktop/src/lib/desktop-slash-registry.json index d8a05d279d..91917627df 100644 --- a/apps/desktop/src/lib/desktop-slash-registry.json +++ b/apps/desktop/src/lib/desktop-slash-registry.json @@ -22,7 +22,6 @@ "/platforms": "terminal", "/plugins": "terminal", "/quit": "terminal", - "/reasoning": "advanced", "/redraw": "terminal", "/reload": "terminal", "/reload-mcp": "advanced", diff --git a/apps/desktop/src/lib/reasoning-slash.test.ts b/apps/desktop/src/lib/reasoning-slash.test.ts new file mode 100644 index 0000000000..755e6944ea --- /dev/null +++ b/apps/desktop/src/lib/reasoning-slash.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from 'vitest' + +import { resolveDesktopCommand } from '@/lib/desktop-slash-commands' +import { applyReasoningSlashResult, reasoningSlashParams } from '@/lib/reasoning-slash' +import { $showReasoning, setShowReasoningFromConfig } from '@/store/reasoning-disclosure' + +// `/reasoning hide` typed into the Desktop composer must gate Thinking blocks +// in the transcript immediately (#111761, #49664): the renderer mirrors the +// gateway's `config.set key=reasoning` answer instead of waiting for the next +// config refresh. +describe('/reasoning slash command', () => { + it('runs as a desktop action instead of the slash worker', () => { + expect(resolveDesktopCommand('/reasoning')?.surface.kind).toBe('action') + }) + + it('builds the gateway config.set payload, honoring the scope flags', () => { + expect(reasoningSlashParams('hide', 's1')).toEqual({ key: 'reasoning', session_id: 's1', value: 'hide' }) + expect(reasoningSlashParams('high --global', 's1')).toEqual({ + key: 'reasoning', + scope: 'global', + session_id: 's1', + value: 'high' + }) + expect(reasoningSlashParams(' ', 's1')).toBeNull() + }) + + it('flips the Thinking gate on hide/show and leaves it alone for effort levels', () => { + setShowReasoningFromConfig(true) + applyReasoningSlashResult('hide') + expect($showReasoning.get()).toBe(false) + applyReasoningSlashResult('high') + expect($showReasoning.get()).toBe(false) + applyReasoningSlashResult('show') + expect($showReasoning.get()).toBe(true) + }) +}) diff --git a/apps/desktop/src/lib/reasoning-slash.ts b/apps/desktop/src/lib/reasoning-slash.ts new file mode 100644 index 0000000000..71f8c5ac49 --- /dev/null +++ b/apps/desktop/src/lib/reasoning-slash.ts @@ -0,0 +1,53 @@ +import { setShowReasoningFromConfig } from '@/store/reasoning-disclosure' + +// `/reasoning` typed into the composer — the Ink TUI's path +// (ui-tui/src/app/slash/commands/session.ts). The gateway's `config.set +// key=reasoning` handles both the display words (show/hide/full/clamp) and the +// effort levels; through `slash.exec` the words only reached config.yaml and +// the transcript kept rendering Thinking until the next config refresh. + +const GLOBAL_FLAGS = new Set(['--global', '-g', 'global']) +const SESSION_FLAGS = new Set(['--session', '-s', 'session']) + +export type ReasoningSlashParams = { + key: 'reasoning' + scope?: 'global' | 'session' + session_id: string + value: string +} + +/** Build the `config.set` params for `/reasoning `; `null` when there is nothing to set. */ +export function reasoningSlashParams(arg: string, sessionId: string): null | ReasoningSlashParams { + let scope: ReasoningSlashParams['scope'] + const values: string[] = [] + + for (const part of arg.trim().split(/\s+/).filter(Boolean)) { + const flag = part.toLowerCase() + + if (GLOBAL_FLAGS.has(flag)) { + scope = 'global' + } else if (SESSION_FLAGS.has(flag)) { + scope = 'session' + } else { + values.push(part) + } + } + + if (!values.length) { + return null + } + + return { key: 'reasoning', session_id: sessionId, value: values.join(' '), ...(scope ? { scope } : {}) } +} + +/** + * Mirror the gateway's answer into the renderer: `hide`/`show` are the + * `display.show_reasoning` words; effort levels leave the display gate alone. + */ +export function applyReasoningSlashResult(value: unknown): void { + if (value === 'hide') { + setShowReasoningFromConfig(false) + } else if (value === 'show') { + setShowReasoningFromConfig(true) + } +} diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts index 71feaa6a58..8d5ae2269c 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts @@ -502,6 +502,22 @@ describe('gateway mirror', () => { expect(log.length).toBeLessThanOrEqual(16) expect(log.at(-1)?.text).toMatch(/^99:/) expect(log.at(-1)?.text.length).toBeLessThanOrEqual(1200) + expect(log.at(-1)?.truncated).toBe(true) + expect(log.at(-1)?.text).toContain('[truncated]') + expect(log.at(-1)?.text).not.toContain(long) + }) + + it('marks truncated sync text instead of silently slicing it', async () => { + const { chat } = await loadRoom() + const long = `plan:${'x'.repeat(2000)}` + const compacted = chat.compactGroupChatSyncText(long) + + expect(compacted.truncated).toBe(true) + expect(compacted.text).toContain('[truncated]') + expect(compacted.text.length).toBeLessThanOrEqual(1200) + expect(compacted.text.startsWith('plan:')).toBe(true) + expect(compacted.text).not.toBe(long) + expect(chat.compactGroupChatSyncText('short').truncated).toBeUndefined() }) it('preserves threads and budgets escaped Unicode', async () => { @@ -522,6 +538,30 @@ describe('gateway mirror', () => { expect(snapshot.rooms['name:Unicode'].log.at(-1)?.thread).toBe('thread-15') }) + // #114341: the mirror is the only on-disk copy of a room. A head trim — + // by message count or by the byte budget — must say how many earlier + // entries it dropped, or a reader concludes the user never said it. + it('counts the head entries the mirror does not carry', async () => { + const { chat } = await loadRoom() + const entry = (index: number, text: string) => ({ at: index, from: { kind: 'user', name: 'You' }, text }) + + const snapshot = chat.groupChatSyncSnapshot({ + Fits: { log: Array.from({ length: 3 }, (_, index) => entry(index, `short ${index}`)) }, + ByCount: { log: Array.from({ length: 40 }, (_, index) => entry(index, `m${index}`)) }, + ByBytes: { log: Array.from({ length: 16 }, (_, index) => entry(index, `${index} ${'🧠'.repeat(1200)}`)) } + } as unknown as Record) + + const byCount = snapshot.rooms['name:ByCount'] + const byBytes = snapshot.rooms['name:ByBytes'] + + expect(snapshot.rooms['name:Fits'].omitted).toBeUndefined() + expect(byCount.log).toHaveLength(16) + expect(byCount.omitted).toBe(24) + expect(byBytes.log.length).toBeLessThan(16) + expect(byBytes.omitted).toBe(16 - byBytes.log.length) + expect(chat.groupChatGatewayJsonSize(snapshot)).toBeLessThanOrEqual(48000) + }) + it('omits empty runtime rooms', async () => { const { chat } = await loadRoom() diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.ts index 8acb8508b8..4458e667f6 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.ts @@ -52,6 +52,7 @@ const GROUP_CHAT_SYNC_META_KEY = 'hermes-bots-groups' const GROUP_CHAT_SYNC_MAX_BYTES = 48000 const GROUP_CHAT_SYNC_MESSAGES = 16 const GROUP_CHAT_SYNC_TEXT_CHARS = 1200 +const GROUP_CHAT_SYNC_TRUNCATION_MARK = '… [truncated]' const GROUP_CHAT_SYNC_IMAGE_CHARS = 24000 let groupChatSyncTimer: ReturnType | null = null @@ -62,6 +63,9 @@ interface GroupChatSyncRoom { log: GroupMessage[] members?: GroupMember[] name?: string + /** At least this many earlier room entries exist that the projection does + * not carry (head-trimmed to the message/byte budget). */ + omitted?: number revision?: number roomId?: string } @@ -91,6 +95,36 @@ const groupChatSyncRetryTimers = new Map>( const groupChatSyncRetryCounts = new Map() export let groupChatSyncDisposed = false +/** Cut one sync-projection line to the per-message budget and mark the cut. + * Receivers used to see a silent mid-sentence slice with no signal that the + * body continued. Keep the mark inside the same char budget so CJK/envelope + * accounting does not grow. */ +export function compactGroupChatSyncText(text: string, limit = GROUP_CHAT_SYNC_TEXT_CHARS) { + const raw = String(text || '') + + if (raw.length <= limit) { + return { text: raw } + } + + const budget = Math.max(0, limit - GROUP_CHAT_SYNC_TRUNCATION_MARK.length) + + return { + text: `${raw.slice(0, budget)}${GROUP_CHAT_SYNC_TRUNCATION_MARK}`, + truncated: true as const + } +} + +/** #114341: the ui_meta mirror is the only on-disk copy of a room, so a + * head-trimmed log must say how many earlier entries it does not carry — + * a bare slice reads as "the user never said it". */ +function noteGroupChatSyncOmitted(room: GroupChatSyncRoom, total: number) { + const omitted = total - room.log.length + + if (omitted > 0) { + room.omitted = omitted + } +} + /** Conservative byte count for the gateway's ensure_ascii JSON encoding. * Python also inserts separator spaces, so reserve one extra byte per JS * structural separator on top of escaped Unicode code-point widths. */ @@ -216,29 +250,38 @@ export function groupChatSyncSnapshot( } for (const [name, room] of ranked) { - const log: GroupMessage[] = room.log.slice(-GROUP_CHAT_SYNC_MESSAGES).map(entry => ({ - ...(entry?.id - ? { - id: String(entry.id).slice(0, 160) - } - : {}), - from: { - kind: entry?.from?.kind === 'member' ? 'member' : 'user', - name: String(entry?.from?.name || (entry?.from?.kind === 'member' ? 'Bot' : 'You')).slice(0, 128), - ...(entry?.from?.source + const log: GroupMessage[] = room.log.slice(-GROUP_CHAT_SYNC_MESSAGES).map(entry => { + const compacted = compactGroupChatSyncText(String(entry?.text || '')) + + return { + ...(entry?.id ? { - source: String(entry.from.source).slice(0, 128) + id: String(entry.id).slice(0, 160) + } + : {}), + from: { + kind: entry?.from?.kind === 'member' ? 'member' : 'user', + name: String(entry?.from?.name || (entry?.from?.kind === 'member' ? 'Bot' : 'You')).slice(0, 128), + ...(entry?.from?.source + ? { + source: String(entry.from.source).slice(0, 128) + } + : {}) + }, + text: compacted.text, + at: Number(entry?.at || 0), + ...(entry?.thread + ? { + thread: String(entry.thread).slice(0, 128) + } + : {}), + ...(compacted.truncated + ? { + truncated: true } : {}) - }, - text: String(entry?.text || '').slice(0, GROUP_CHAT_SYNC_TEXT_CHARS), - at: Number(entry?.at || 0), - ...(entry?.thread - ? { - thread: String(entry.thread).slice(0, 128) - } - : {}) - })) + } + }) const compact: GroupChatSyncRoom = { name: String(name).slice(0, 64), @@ -286,9 +329,11 @@ export function groupChatSyncSnapshot( const key = groupChatRoomKey(name, room) rooms[key] = compact + noteGroupChatSyncOmitted(compact, room.log.length) while (compact.log.length > 1 && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { compact.log.shift() + noteGroupChatSyncOmitted(compact, room.log.length) } if (compact.image && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { @@ -418,6 +463,8 @@ export function mergeGroupChatSyncSnapshots( } const remoteRevision = Math.max(0, Number(remoteRoom?.revision || 0)) + // Either writer's head trim is a lower bound on what the union still lacks. + const omitted = Math.max(Number(remoteRoom?.omitted || 0), Number(localRoom?.omitted || 0)) const localRevision = changed.has(key) ? Math.max(0, Number(writeRevision || 0)) @@ -473,6 +520,11 @@ export function mergeGroupChatSyncSnapshots( }), members, revision: Math.max(remoteRevision, localRevision), + ...(omitted > 0 + ? { + omitted + } + : {}), ...(typeof image === 'string' && image ? { image @@ -531,6 +583,7 @@ function groupChatSyncEnvelope( for (const [key, room] of ranked) { while ((room.log?.length || 0) > 1 && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { room.log.shift() + room.omitted = (room.omitted || 0) + 1 } if (room.image && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts index 9b9d348690..f8b2664dfe 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts @@ -3,7 +3,6 @@ import { recordGroupActivity } from './group-activity' import { $groupChats, appendGroupChatEntry, - GROUP_CHAT_HISTORY_LIMIT, GROUP_CHAT_MAX_CONTINUATIONS, GROUP_CHAT_MAX_MESSAGES, groupThreadOf, @@ -12,7 +11,7 @@ import { } from './group-chat' import type { GroupChatRoom } from './group-chat' import { groupMemberKey } from './group-membership' -import { buildGroupChatTurnPrompt, formatGroupChatLine } from './group-round-prompt' +import { buildGroupChatTurnPrompt, formatGroupDeltaLines } from './group-round-prompt' import { isGroupPassText, runGroupChatMemberTurn } from './group-turns' import type { Attachment, GroupMember, GroupMessage } from './types' @@ -94,7 +93,7 @@ function prepareGroupRoundMember(context: GroupRoundMemberContext, member: Group groupName: context.group, members, viewer: member, - deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member, context.group)) + deltaLines: formatGroupDeltaLines(delta, member, context.group) }) // Images riding this delta (user attachments — member entries don't diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts index 7694facea8..e4c68a6a31 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts @@ -1,5 +1,5 @@ import { botMentionTag } from './data' -import { groupSpeakerLabel } from './group-chat' +import { GROUP_CHAT_HISTORY_LIMIT, groupSpeakerLabel } from './group-chat' import { groupMemberKey } from './group-membership' import type { GroupMember, GroupMessage, GroupMessageAuthor } from './types' @@ -50,6 +50,23 @@ export function formatGroupChatLine(entry: GroupMessage, viewer: GroupChatLineVi return `${groupSpeakerLabel(entry.from.name, group)}${suffix}${source}: ${relabelMemberControlFrames(entry.text)}${attached}` } +/** #114341: a member's turn renders only the last GROUP_CHAT_HISTORY_LIMIT + * delta lines while the watermark commit advances past the whole tail, so + * the head of an over-long delta is never delivered on any later turn + * either. Mark the cut — without it a member has no way to know its view + * of the room is partial (typically missing the very user instruction + * that started the exchange). */ +export function formatGroupDeltaLines(delta: GroupMessage[], viewer: GroupChatLineViewer, group?: null | string) { + const omitted = delta.length - GROUP_CHAT_HISTORY_LIMIT + const lines = delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map(entry => formatGroupChatLine(entry, viewer, group)) + + if (omitted > 0) { + lines.unshift(`… ${omitted} earlier room message${omitted === 1 ? '' : 's'} omitted since your last turn`) + } + + return lines +} + function viewerNameOf(viewer: GroupChatLineViewer): string { return typeof viewer === 'string' ? viewer : viewer?.name || '' } diff --git a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts index 36d273a62a..a12a1fbd9d 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts @@ -393,6 +393,51 @@ describe('per-member delta', () => { expect(room.gateway.calls.at(-1)?.prompt).toContain('unseen-99') }) + // #114341: the turn renders only the last GROUP_CHAT_HISTORY_LIMIT entries + // of the delta while the watermark advances past the whole tail, so the + // head is never delivered later either. The cut must be visible to the + // member (naming how many entries it did not see); a delta that fits + // carries no marker. + it('names the omitted head of an over-long delta in the turn prompt', async () => { + const room = await loadRoom({ turn: () => '(pass)' }) + const members = [MEMBERS[0]] + const thread = room.rounds.sendToGroupChat('Head', members, 'seen-0')! + await settle(room, 'Head') + const limit = room.chat.GROUP_CHAT_HISTORY_LIMIT + + for (let i = 1; i <= limit + 5; i++) { + room.chat.appendGroupChatEntry('Head', { kind: 'user', name: 'You' }, `unseen-${i}`, thread) + } + + const seen = room.chat.$groupChats.get().Head.watermarks[`${thread}::research`] || 0 + const omitted = log(room, 'Head').slice(seen).length - limit + expect(omitted).toBeGreaterThan(0) + + await room.rounds.runGroupChatRounds('Head', members, thread) + const prompt = room.gateway.calls.at(-1)?.prompt || '' + + expect(prompt).toMatch(new RegExp(`${omitted} earlier room messages omitted`)) + expect(prompt).toContain(`unseen-${limit + 5}`) + expect(prompt).not.toContain(`unseen-${omitted}\n`) + }) + + it('adds no omission marker when the delta fits the window', async () => { + const room = await loadRoom({ turn: () => '(pass)' }) + const members = [MEMBERS[0]] + const thread = room.rounds.sendToGroupChat('Fits', members, 'seen-0')! + await settle(room, 'Fits') + + for (let i = 1; i < room.chat.GROUP_CHAT_HISTORY_LIMIT; i++) { + room.chat.appendGroupChatEntry('Fits', { kind: 'user', name: 'You' }, `unseen-${i}`, thread) + } + + await room.rounds.runGroupChatRounds('Fits', members, thread) + const prompt = room.gateway.calls.at(-1)?.prompt || '' + + expect(prompt).toContain('unseen-1') + expect(prompt).not.toMatch(/omitted/) + }) + it('feeds a second send only the NEW messages', async () => { const room = await loadRoom() const member: GroupMember[] = [{ name: 'research', title: '' }] diff --git a/apps/desktop/src/plugins/hermes-bots/types.ts b/apps/desktop/src/plugins/hermes-bots/types.ts index df397e1cb6..c5eff2606d 100644 --- a/apps/desktop/src/plugins/hermes-bots/types.ts +++ b/apps/desktop/src/plugins/hermes-bots/types.ts @@ -159,6 +159,8 @@ export interface GroupMessage { text: string /** Messages predating threading carry the sentinel thread `'legacy'`. */ thread?: string + /** Set on the ui_meta projection when `text` was cut to the sync budget. */ + truncated?: boolean } export interface GroupHold { diff --git a/apps/desktop/src/store/composer.test.ts b/apps/desktop/src/store/composer.test.ts index 0055e5ae33..a0f13584d6 100644 --- a/apps/desktop/src/store/composer.test.ts +++ b/apps/desktop/src/store/composer.test.ts @@ -2,8 +2,11 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { $composerAttachments, + $restoredDraftNotice, $voiceConversationStartRequest, addComposerAttachment, + adoptGoneSessionDraft, + announceGoneSessionDraft, clearSessionDraft, type ComposerAttachment, createComposerAttachmentOccurrenceId, @@ -15,6 +18,7 @@ import { stashSessionDraft, takeSessionDraft, takeVoiceConversationStart, + undoRestoredDraft, updateComposerAttachment } from './composer' @@ -267,6 +271,38 @@ describe('session drafts', () => { expect(takeSessionDraft('session-a').attachments[0]?.label).toBe('doc.pdf') }) + it('restores a gone session draft only into an EMPTY fresh chat, and Undo puts it back where it was (#111868)', () => { + // Never clobber what the user is already typing in the new chat. + stashSessionDraft('session-a', 'from the dead session', []) + stashSessionDraft(null, 'already composing here', []) + announceGoneSessionDraft('session-a') + + expect(adoptGoneSessionDraft()).toBe(false) + expect($restoredDraftNotice.get()).toBeNull() + expect(takeSessionDraft(null).text).toBe('already composing here') + expect(takeSessionDraft('session-a').text).toBe('from the dead session') + + // Empty fresh chat → restored; Undo (text untouched) returns it to the + // dead key, so the same recovery path can find it again later. + clearSessionDraft(null) + announceGoneSessionDraft('session-a') + + expect(adoptGoneSessionDraft()).toBe(true) + expect(takeSessionDraft(null).text).toBe('from the dead session') + expect(undoRestoredDraft('from the dead session')).toBe(true) + expect(takeSessionDraft(null).text).toBe('') + expect(takeSessionDraft('session-a').text).toBe('from the dead session') + expect($restoredDraftNotice.get()).toBeNull() + + // Once the user has edited the restored text, Undo would destroy their + // work: it only dismisses. + announceGoneSessionDraft('session-a') + adoptGoneSessionDraft() + + expect(undoRestoredDraft('from the dead session, edited')).toBe(false) + expect(takeSessionDraft(null).text).toBe('from the dead session') + }) + it('migrates a tip-keyed draft onto the post-compression tip', () => { const tipBefore = '20260720_062637_ad96b3' const tipAfter = '20260720_071049_a28905' diff --git a/apps/desktop/src/store/composer.ts b/apps/desktop/src/store/composer.ts index bdaa5d5083..7a581b2272 100644 --- a/apps/desktop/src/store/composer.ts +++ b/apps/desktop/src/store/composer.ts @@ -187,6 +187,17 @@ export interface SessionDraft { const draftKey = (scope: string | null | undefined) => scope?.trim() || NEW_SESSION_DRAFT_KEY +/** Inline "Restored your unsent message" notice for the fresh draft (see + * `adoptGoneSessionDraft`). `null` = nothing to show. */ +export interface RestoredDraftNotice { + /** The dead stored-session key the text came from. */ + fromKey: string + /** The text as restored — Undo only applies while the draft still equals it. */ + text: string +} + +export const $restoredDraftNotice = atom(null) + const cloneDraft = (draft: SessionDraft): SessionDraft => ({ attachments: draft.attachments.map(attachment => ({ ...attachment })), text: draft.text @@ -389,6 +400,10 @@ export function stashSessionDraft(scope: string | null | undefined, text: string if (text.trim() || attachments.length > 0) { draftsBySession.set(key, cloneDraft({ attachments, text })) + } else if (key === NEW_SESSION_DRAFT_KEY) { + // The fresh draft was sent or emptied — a restore notice has nothing left + // to undo. + $restoredDraftNotice.set(null) } persistDraftTexts() @@ -463,6 +478,94 @@ export function adoptNewSessionDraft(toKey: string | null | undefined): boolean return !!announced && announced === toKey?.trim() && migrateSessionDraft(null, toKey) } +/** + * Recovery for the unsent text of a session that turned out to be GONE + * (#111868): deleted, or a stale id from a wiped / renamed backend. + * + * The resume path already drops such a window to a fresh draft without + * toasting or looping (62af32efe7c, bounded by `goneSessionVerdict`). The + * composer's draft stash is keyed per stored session, so the text the user + * typed into the dead id is not lost — but nothing will ever open that key + * again, so it is invisible. The gone verdict announces the dead key here; + * the composer's scope swap (concrete id → the `__new__` bucket) consumes + * it AFTER the outgoing cleanup stashed the live editor text, so even + * keystrokes still inside the persist debounce ride along. + * + * Offer, don't hijack: the fresh draft is seeded and an inline notice with + * Undo is published — no navigation, no focus steal, no toast. Fires once: + * the source key is cleared by the move, so re-opening the dead id later + * finds nothing to restore. + */ +let announcedGoneSessionDraftKey: string | null = null + +export function announceGoneSessionDraft(fromKey: string | null | undefined): void { + announcedGoneSessionDraftKey = fromKey?.trim() || null +} + +/** + * Consume the announcement when a composer enters the fresh-draft scope. + * Moves the dead key's draft into the `__new__` bucket and publishes the + * notice. Declines (no notice) when nothing was announced, the key holds no + * text, or the user is already composing a new chat — never clobber what + * they are typing. Keyed on the announcement, not on the composer observing + * an id → fresh transition: the composer can remount across the drop (a + * loading route mounts no composer), so the dead scope may never have been + * this instance's previous scope. + */ +export function adoptGoneSessionDraft(): boolean { + const announced = announcedGoneSessionDraftKey + announcedGoneSessionDraftKey = null + + if (!announced) { + return false + } + + const source = draftsBySession.get(draftKey(announced)) + + if (!source?.text.trim()) { + return false + } + + const dest = draftsBySession.get(NEW_SESSION_DRAFT_KEY) + + if (dest && (dest.text.trim() || dest.attachments.length > 0)) { + return false + } + + const { attachments, text } = source + stashSessionDraft(null, text, attachments) + clearSessionDraft(announced) + $restoredDraftNotice.set({ fromKey: announced, text }) + + return true +} + +export function dismissRestoredDraftNotice(): void { + $restoredDraftNotice.set(null) +} + +/** + * Undo the restore: put the text back under the dead key (where it was, + * still recoverable by the same path) and empty the fresh draft. Only while + * the live text is still exactly what was restored — once the user has + * edited it, Undo would destroy their work, so it only dismisses the notice. + * Returns whether the fresh draft was emptied (the caller repaints). + */ +export function undoRestoredDraft(liveText: string): boolean { + const notice = $restoredDraftNotice.get() + $restoredDraftNotice.set(null) + + if (!notice || liveText !== notice.text) { + return false + } + + const current = draftsBySession.get(NEW_SESSION_DRAFT_KEY) + stashSessionDraft(notice.fromKey, notice.text, current?.attachments ?? []) + clearSessionDraft(null) + + return true +} + export function setComposerDraft(value: string) { $composerDraft.set(value) } diff --git a/cli.py b/cli.py index fee1f3af8a..254f14dbc9 100644 --- a/cli.py +++ b/cli.py @@ -4119,6 +4119,15 @@ _TRANSIENT_PROVIDER_REASONS = frozenset({ "rate_limit", "upstream_rate_limit", "billing", "overloaded", "server_error", "timeout", }) +# ``failure_reason`` values a retry can never heal: the credential was rejected, the model does +# not exist for this account, or the TLS chain is broken. A Kanban worker exits +# ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` so the dispatcher parks the card after ONE spawn with +# the provider's words as the reason, instead of re-spawning into the same wall until +# ``kanban.failure_limit`` is spent. ``billing`` stays transient: credit comes back. +_TERMINAL_PROVIDER_REASONS = frozenset({ + "auth", "auth_permanent", "model_not_found", "ssl_cert_verification", +}) + def _single_query_exit_code(result) -> int: """Map a one-shot turn result onto a process exit code, for both `-q` and `-Q`. @@ -4129,6 +4138,8 @@ def _single_query_exit_code(result) -> int: failed purely on a provider rate-limit / billing wall exits ``KANBAN_RATE_LIMIT_EXIT_CODE`` (EX_TEMPFAIL): the dispatcher books that run ``rate_limited`` and requeues the task WITHOUT counting a failure, so a quota window or a provider outage cannot trip the breaker. + One that failed on a terminal provider error (credential revoked, model gone) exits + ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` (EX_CONFIG): the dispatcher blocks the card at once. """ if not isinstance(result, dict): return 1 @@ -4136,9 +4147,14 @@ def _single_query_exit_code(result) -> int: return 130 if not (result.get("failed") or result.get("partial") or result.get("completed") is False): return 0 - if os.environ.get("HERMES_KANBAN_TASK") and result.get("failure_reason") in _TRANSIENT_PROVIDER_REASONS: - from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE - return KANBAN_RATE_LIMIT_EXIT_CODE + if os.environ.get("HERMES_KANBAN_TASK"): + reason = result.get("failure_reason") + if reason in _TRANSIENT_PROVIDER_REASONS: + from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE + return KANBAN_RATE_LIMIT_EXIT_CODE + if reason in _TERMINAL_PROVIDER_REASONS: + from hermes_cli.kanban_db import KANBAN_TERMINAL_PROVIDER_EXIT_CODE + return KANBAN_TERMINAL_PROVIDER_EXIT_CODE return 1 diff --git a/contributors/emails/Zoeille@users.noreply.github.com b/contributors/emails/Zoeille@users.noreply.github.com new file mode 100644 index 0000000000..ec7ac6dcaa --- /dev/null +++ b/contributors/emails/Zoeille@users.noreply.github.com @@ -0,0 +1 @@ +Zoeille diff --git a/contributors/emails/artiefisher123@gmail.com b/contributors/emails/artiefisher123@gmail.com new file mode 100644 index 0000000000..f7afbbe764 --- /dev/null +++ b/contributors/emails/artiefisher123@gmail.com @@ -0,0 +1 @@ +Shotflame diff --git a/contributors/emails/dev@mikesoft.it b/contributors/emails/dev@mikesoft.it new file mode 100644 index 0000000000..dfcbd18b41 --- /dev/null +++ b/contributors/emails/dev@mikesoft.it @@ -0,0 +1 @@ +TheStreamCode diff --git a/contributors/emails/drkpxl@users.noreply.github.com b/contributors/emails/drkpxl@users.noreply.github.com new file mode 100644 index 0000000000..59387bb2af --- /dev/null +++ b/contributors/emails/drkpxl@users.noreply.github.com @@ -0,0 +1 @@ +drkpxl diff --git a/contributors/emails/eloktev@users.noreply.github.com b/contributors/emails/eloktev@users.noreply.github.com new file mode 100644 index 0000000000..8885873168 --- /dev/null +++ b/contributors/emails/eloktev@users.noreply.github.com @@ -0,0 +1 @@ +eloktev diff --git a/contributors/emails/openclaw@Jonas-Mac-Studio.local b/contributors/emails/openclaw@Jonas-Mac-Studio.local new file mode 100644 index 0000000000..1966aa8adf --- /dev/null +++ b/contributors/emails/openclaw@Jonas-Mac-Studio.local @@ -0,0 +1 @@ +jonameijers diff --git a/contributors/emails/yapache@gmail.com b/contributors/emails/yapache@gmail.com new file mode 100644 index 0000000000..7eba2f413e --- /dev/null +++ b/contributors/emails/yapache@gmail.com @@ -0,0 +1 @@ +Caelier diff --git a/cron/AGENTS.md b/cron/AGENTS.md index ec38f3dcdb..cddd8e5fc6 100644 --- a/cron/AGENTS.md +++ b/cron/AGENTS.md @@ -75,7 +75,10 @@ zero outside a kanban task (footprint ladder rung 3). Isolation: **board** is the hard boundary — workers get `HERMES_KANBAN_BOARD` pinned in their env and cannot see other boards; **tenant** is a soft namespace within a board (workspace-path + memory-key isolation, one fleet serving several businesses). After `kanban.failure_limit` consecutive -non-success attempts on a task (default 2) the dispatcher auto-blocks it to stop spin loops. +non-success attempts on a task (default 2) the dispatcher auto-blocks it to stop spin loops; a +worker exit of `KANBAN_TERMINAL_PROVIDER_EXIT_CODE` (78 — credential revoked, model gone; the +worker's own `failure_reason` classification via `cli._TERMINAL_PROVIDER_REASONS`) trips it on +the first attempt, sticky, because no retry can heal it (#114587). Process-identity note: `kanban --preserve-cache` contains "serve" — never classify processes by argv substring (root). Worker liveness is `(worker_pid, worker_started_at)` — the start-time fingerprint (`gateway.status.get_process_start_time`) recorded at claim time — never bare PID existence, or a diff --git a/cron/scheduler_thread.py b/cron/scheduler_thread.py index 2552842909..85f54e5bb7 100644 --- a/cron/scheduler_thread.py +++ b/cron/scheduler_thread.py @@ -23,11 +23,22 @@ class SupervisedTickerThread: stop_event: threading.Event, name: str = "cron-scheduler") -> None: self._target, self._args, self._kwargs = target, args, dict(kwargs or {}) self._stop_event, self._name = stop_event, name + # An external provider's start() (Chronos) arms remote one-shots and RETURNS by design; + # only a target that escaped with an exception is a dead ticker worth respawning. + self._crashed = False self._thread = self._spawn() self.restarts = 0 + def _run(self) -> None: + try: + self._target(*self._args, **self._kwargs) + except BaseException: + self._crashed = True + raise + def _spawn(self) -> threading.Thread: - return threading.Thread(target=self._target, args=self._args, kwargs=self._kwargs, daemon=True, name=self._name) + self._crashed = False + return threading.Thread(target=self._run, daemon=True, name=self._name) def start(self) -> None: self._thread.start() @@ -39,8 +50,8 @@ class SupervisedTickerThread: self._thread.join(timeout) def restart_if_dead(self) -> bool: - """Respawn the ticker when it ended without ``stop_event``; True when a restart happened.""" - if self._stop_event.is_set() or self._thread.is_alive(): + """Respawn the ticker when it crashed without ``stop_event``; True when a restart happened.""" + if self._stop_event.is_set() or self._thread.is_alive() or not self._crashed: return False self.restarts += 1 logger.error( diff --git a/gateway/AGENTS.md b/gateway/AGENTS.md index 3f7fc691c9..d41911e2d2 100644 --- a/gateway/AGENTS.md +++ b/gateway/AGENTS.md @@ -133,6 +133,15 @@ gateway under the backend, and do NOT "fix" update locks by widening the tree-ki ## Profile scope (adapters, turns, and everything between turns) +- **One identity per inbound event.** `gateway/session_identity.py::resolve_identity` answers + "which bot received it / who may admit it / where does it run" ONCE per event and pins a frozen + `RoutingIdentity` on the source (wire-invisible, like `_transport_adapter_ref`); + `_transport_owner`, `_authorization_home_for_source`, `_resolve_profile_home_for_source`, + `_session_key_profile` and `_resolve_profile_for_key` read it when present and fall back to + their old chain only for sources nothing resolved (restored rows, hand-built sources). Never + derive a second answer next to the identity; extend the object. `transport_profile` ≠ + `runtime_profile` is normal (shared bot → routed satellite). A source copy goes through + `session_identity.replace_source` so the identity travels with it. - **Token locks.** An adapter that connects with a unique credential (bot token, API key) calls `acquire_scoped_lock()` from `gateway.status` in `connect()`/`start()` and `release_scoped_lock()` in `disconnect()`/`stop()`, so two profiles cannot share one credential. Canonical: diff --git a/gateway/authz_mixin.py b/gateway/authz_mixin.py index 6ea612b7bd..dc9a98af87 100644 --- a/gateway/authz_mixin.py +++ b/gateway/authz_mixin.py @@ -293,13 +293,18 @@ class GatewayAuthorizationMixin: return (adapter, profile) if registered else None def _authorization_home_for_source(self, source: SessionSource): - """HERMES_HOME whose allowlist admits *source*: the ingress-stamped transport home, else the home of - the profile owning the adapter that delivers it. ``None`` = authorize in the ambient scope - (multiplex off, or no live adapter — the check then fails closed on its own). + """HERMES_HOME whose allowlist admits *source*: the identity's transport home (or the + ingress-stamped one), else the home of the profile owning the adapter that delivers it. + ``None`` = authorize in the ambient scope (multiplex off, or no live adapter — the check then + fails closed on its own). Inside a routed satellite's turn the ambient scope is the satellite's, whose ``.env`` has no token/allowlist; every authorization decision made mid-turn (``/topic``, sibling ``/stop``, plugin injection, voice, auto-resume) must read the admitting bot's allowlist instead.""" + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.authorization_home if identity.multiplexed else None stamped = getattr(source, "_authorization_profile_home", None) if stamped is not None: return Path(stamped) diff --git a/gateway/platforms/ADDING_A_PLATFORM.md b/gateway/platforms/ADDING_A_PLATFORM.md index 9887ad5dd9..3fb182a3c7 100644 --- a/gateway/platforms/ADDING_A_PLATFORM.md +++ b/gateway/platforms/ADDING_A_PLATFORM.md @@ -141,7 +141,12 @@ def check__requirements() -> bool: ### Key patterns to follow -- Use `self.build_source(...)` to construct `SessionSource` objects +- Use `self.build_source(...)` to construct `SessionSource` objects (never `SessionSource(...)` + directly — the transport provenance and profile route are stamped there) +- Derive every adapter-side session key (batching, per-chat queues, busy detection) through + `self._event_session_key(event)` / `self._source_session_key(source)`, never the free + `build_session_key()` — the seam keys in the owning profile's namespace under a multiplexed + gateway; the advisory lint (`scripts/check_profile_scope_patterns.py`, pattern P32) flags both - Call `self.handle_message(event)` to dispatch inbound messages to the gateway - Use `MessageEvent`, `MessageType` from `gateway.platforms.event` and `SendResult` from base - Use `cache_image_from_bytes`, `cache_audio_from_bytes`, `cache_document_from_bytes` for attachments diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index 45a709f46d..5995878b6c 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -2307,9 +2307,13 @@ class BasePlatformAdapter(ABC): def _session_key_profile(self, source: Optional[Any] = None) -> Optional[str]: """Profile namespace for an adapter-derived session key. Ingress runs BEFORE the runner stamps ``source.profile``, so without this every bot in a multiplexed gateway shares one - ``agent:main:`` lane. Order: ``source.profile`` → ``_owner_profile`` → session-store - resolver; getattr-guarded (object.__new__ in tests), type-checked (no MagicMock in the - key).""" + ``agent:main:`` lane. Order: pinned ``RoutingIdentity`` → ``source.profile`` → + ``_owner_profile`` → session-store resolver; getattr-guarded (object.__new__ in tests), + type-checked (no MagicMock in the key).""" + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.session_key_profile for candidate in ( getattr(source, "profile", None) if source is not None else None, getattr(self, "_owner_profile", None)): diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index 6c9ebcf8bd..a32778c10c 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -914,12 +914,6 @@ class WeixinAdapter(OwnAccessPolicyMixin, BasePlatformAdapter): else: await self.handle_message(event) - def _text_batch_key(self, event: MessageEvent) -> str: - from gateway.session import build_session_key - return build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), profile=event.source.profile) - async def _collect_media(self, item: Dict[str, Any], media_paths: List[str], media_types: List[str]) -> None: spec = _INBOUND_MEDIA.get(item.get("type")) path, mime = await self._download_media(item, spec) if spec else (None, "") diff --git a/gateway/platforms/yuanbao.py b/gateway/platforms/yuanbao.py index a426c3ad02..3f61c4fda1 100644 --- a/gateway/platforms/yuanbao.py +++ b/gateway/platforms/yuanbao.py @@ -64,7 +64,6 @@ from gateway.platforms.yuanbao_proto import ( encode_send_private_heartbeat, encode_send_group_heartbeat, encode_query_group_info, encode_get_group_member_list, next_seq_no, ) -from gateway.session import build_session_key from gateway.session_transcript import TranscriptReadError logger = logging.getLogger(__name__) @@ -1666,11 +1665,8 @@ class DispatchMiddleware(InboundMiddleware): async def handle(self, ctx: InboundContext, next_fn) -> None: adapter = ctx.adapter - _sk = build_session_key( - ctx.source, - group_sessions_per_user=adapter.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=adapter.config.extra.get("thread_sessions_per_user", False), - ) + # The adapter seam: keyed in the owner profile's namespace, same as ``handle_message``. + _sk = adapter._source_session_key(ctx.source) async def _dispatch_inbound_event() -> None: if any(mt.startswith(("application/", "text/")) for mt in ctx.media_types): diff --git a/gateway/run.py b/gateway/run.py index a3050c1b0b..a4836197c2 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -4336,11 +4336,16 @@ class GatewayRunner( return None def _resolve_profile_home_for_source(self, source: SessionSource) -> "Path": - """Resolve which profile's HERMES_HOME serves this source: ``source.profile``, then - ``_profile_name_for_source`` (sources bypassing ``build_source``), then the active profile.""" + """Resolve which profile's HERMES_HOME serves this source: the pinned identity's runtime + home, else ``source.profile``, then ``_profile_name_for_source`` (sources bypassing + ``build_source``), then the active profile.""" from gateway.profile_routing import ProfileRouteRejected + from gateway.session_identity import identity_of from hermes_cli.profiles import get_active_profile_name, get_profile_dir, profile_exists from hermes_constants import get_hermes_home + identity = identity_of(source) + if identity is not None: + return identity.runtime_home explicit_profile = None # explicitly requested (source or routing) vs. default fallback try: name = (source.profile or "").strip() or self._profile_name_for_source(source) diff --git a/gateway/run_adapters.py b/gateway/run_adapters.py index bb9aa3b9e9..bedf66a852 100644 --- a/gateway/run_adapters.py +++ b/gateway/run_adapters.py @@ -1348,12 +1348,18 @@ class GatewayAdapterLifecycleMixin: """``scope_factory(profile_home)`` or a nullcontext when the profile home is unknown.""" return scope_factory(profile_home) if profile_home is not None else contextlib.nullcontext() - @staticmethod - def _stamp_event_profile(event, profile_name: str) -> None: - """Best-effort: stamp ``source.profile`` on an inbound event that has none yet.""" + def _stamp_event_profile(self, event, profile_name: str) -> None: + """Best-effort: pin the secondary's identity on an inbound event (stamps ``source.profile`` + when none yet). A source that cannot resolve keeps today's fallback readers.""" + source = getattr(event, "source", None) + if source is None: + return with suppress(Exception): - if getattr(event, "source", None) is not None and not event.source.profile: - event.source.profile = profile_name + from gateway.session_identity import resolve_identity + resolve_identity(source, runner=self, transport_profile=profile_name) + with suppress(Exception): + if not source.profile: + source.profile = profile_name def _make_profile_message_handler(self, profile_name: str): """Message handler that stamps source.profile, then delegates under the profile scope @@ -1415,23 +1421,14 @@ class GatewayAdapterLifecycleMixin: return _handler def _admit_primary_source(self, source, default_home: Path) -> Optional[Path]: - """Stamp the transport home (authorization) and routed profile on a primary-adapter source and - return the runtime home to scope the turn under; ``None`` when the route targets an unserved - profile. ``_authorization_profile_home`` is in-process only (serialization ignores dynamic attrs); - route ≠ admitting bot.""" - source._authorization_profile_home = default_home - if ( - not getattr(source, "profile", None) - and getattr(source, "profile_route_rejected", False) is not True - and not self._stamp_routed_profile(source) - ): - source.profile_route_rejected = True - if getattr(source, "profile_route_rejected", False) is True: + """Resolve the primary-adapter source's identity (transport home for authorization, routed + profile for the runtime) and return the runtime home to scope the turn under; ``None`` when + the route targets an unserved profile. Route ≠ admitting bot.""" + from gateway.session_identity import IdentityUnresolved, resolve_identity + try: + return resolve_identity(source, runner=self, primary_home=default_home).runtime_home + except IdentityUnresolved: return None - return ( - self._resolve_profile_home_for_source(source) - if getattr(source, "profile", None) else default_home - ) def _stamp_routed_profile(self, source) -> bool: """Stamp ``source.profile`` from ``profile_routes``; False when the route is rejected.""" diff --git a/gateway/run_agent_cache.py b/gateway/run_agent_cache.py index 462a50ee12..19a7e0b086 100644 --- a/gateway/run_agent_cache.py +++ b/gateway/run_agent_cache.py @@ -481,6 +481,13 @@ class GatewayAgentCacheMixin: session_key, interrupt_reason=interrupt_reason, invalidation_reason=invalidation_reason, ) from gateway.run import _AGENT_PENDING_SENTINEL + # The turn's hard interrupt reaches only its in-turn children; background delegations were + # detached at dispatch and would otherwise run to completion and wake the session later. + # Each interrupted unit still returns as a completion (status=interrupted, partial output). + from tools.async_delegation import interrupt_for_session + interrupt_for_session( + session_key=session_key, reason=invalidation_reason, + parent_session_id=str(getattr(running_agent, "session_id", "") or "")) if running_agent and running_agent is not _AGENT_PENDING_SENTINEL: # Plugins holding a per-turn external resource (an outbound RPC blocked on a tool result # the loop will never consume) learn the turn is gone. Fires for /stop and the /new diff --git a/gateway/run_topics.py b/gateway/run_topics.py index 05b6abf70a..cc467c1b43 100644 --- a/gateway/run_topics.py +++ b/gateway/run_topics.py @@ -5,7 +5,6 @@ from __future__ import annotations import asyncio -import dataclasses import logging import re import time @@ -434,12 +433,10 @@ class GatewayTopicThreadsMixin: return copied_source = source with suppress(Exception): - copied_source = dataclasses.replace(source) - # Keep the live transport owner; multiplex routes may run under a + # Keep the live transport owner and identity; multiplex routes may run under a # profile that does not own the Discord adapter/token. - transport_ref = getattr(source, "_transport_adapter_ref", None) - if transport_ref is not None: - setattr(copied_source, "_transport_adapter_ref", transport_ref) + from gateway.session_identity import replace_source + copied_source = replace_source(source) future = safe_schedule_threadsafe( make_coro(copied_source), loop, logger=logger, log_message=f"{label} failed to schedule", ) diff --git a/gateway/session_identity.py b/gateway/session_identity.py new file mode 100644 index 0000000000..d746cb6e4f --- /dev/null +++ b/gateway/session_identity.py @@ -0,0 +1,183 @@ +"""One frozen routing identity per inbound gateway event. + +A multiplexed gateway answers three questions about every event, and until now answered them in +three places that only agreed because they read the same fallback chain: WHICH bot received it +(``_transport_owner``), WHO may admit it (``_authorization_home_for_source``) and WHERE the turn +runs (``_resolve_profile_home_for_source`` / ``_session_key_profile``). :func:`resolve_identity` +answers all three once and pins the result on the source as a wire-invisible dynamic attribute +(like ``_transport_adapter_ref``); the existing helpers read it when present and keep their +fallback chain when absent, so a source built outside the runner still resolves as before. + +``SessionSource.profile`` stays the serialized runtime profile — ``None`` on the wire means the +receiving bot's own profile — so nothing here changes the wire format or any historical +``agent:main`` key. +""" +from __future__ import annotations + +import dataclasses +import weakref +from dataclasses import dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Any, Optional + +if TYPE_CHECKING: + from gateway.session import SessionSource + +_IDENTITY_ATTR = "_identity" +# Wire-invisible provenance copied alongside the identity when a source is duplicated. +_PROVENANCE_ATTRS = ("_transport_adapter_ref", "_authorization_profile_home", _IDENTITY_ATTR) + + +class IdentityUnresolved(RuntimeError): + """Under multiplexing the event's runtime profile could not be established (an explicit + ``profile_routes`` entry targets a profile this gateway does not serve). Callers drop the + event; it must never fall through to the default profile.""" + + +@dataclass(frozen=True) +class RoutingIdentity: + """Everything a turn needs to know about who it is, resolved once at ingress. + + ``transport_profile`` owns the receiving adapter (its credential and allowlist); + ``runtime_profile`` is the profile that executes the turn — the same name unless a + ``profile_routes`` entry re-homed the event. Both are explicit (``"default"`` is spelled out); + ``None`` never means default here. ``multiplexed`` is False for a standalone gateway, whose + keys stay in the legacy ``agent:main`` namespace whatever profile it was launched with. + """ + + transport_profile: str + runtime_profile: str + authorization_home: Path + runtime_home: Path + multiplexed: bool = True + # Receiving adapter; None for restored/synthetic sources (no live provenance → fail closed). + # Provenance, not identity: two events from the same bot share one identity. + transport: Optional[weakref.ref] = field(default=None, compare=False, hash=False) + + @property + def namespace(self) -> str: + """``agent:`` prefix for this identity's session keys — byte-identical to + :func:`gateway.session._session_key_namespace` for every historical key.""" + from gateway.session import _session_key_namespace + return _session_key_namespace(self.session_key_profile) + + @property + def store_path(self) -> Path: + return self.runtime_home / "state.db" + + @property + def session_key_profile(self) -> Optional[str]: + """The ``profile=`` argument :func:`gateway.session.build_session_key` expects for this + identity: the runtime profile under multiplexing, else ``None`` (legacy namespace).""" + return self.runtime_profile if self.multiplexed else None + + def adapter(self) -> Any: + """The live receiving adapter, or None when it is gone or was never known.""" + return self.transport() if self.transport is not None else None + + +def identity_of(source: Any) -> Optional[RoutingIdentity]: + """The identity pinned on *source* by :func:`resolve_identity`, if any.""" + identity = getattr(source, _IDENTITY_ATTR, None) + return identity if isinstance(identity, RoutingIdentity) else None + + +def replace_source(source: "SessionSource", **changes: Any) -> "SessionSource": + """:func:`dataclasses.replace` that keeps the wire-invisible provenance (transport ref, + authorization home, identity). A plain ``replace`` silently produces a source the runner + can only route through heuristics.""" + copied = dataclasses.replace(source, **changes) + for name in _PROVENANCE_ATTRS: + value = getattr(source, name, None) + if value is not None: + setattr(copied, name, value) + return copied + + +def _name(value: Any) -> Optional[str]: + text = value.strip() if isinstance(value, str) else "" + return text or None + + +def resolve_identity( + source: "SessionSource", *, runner: Any, adapter: Any = None, + transport_profile: Optional[str] = None, primary_home: Optional[Path] = None, +) -> RoutingIdentity: + """Resolve and pin the :class:`RoutingIdentity` of an inbound *source*. + + *adapter* is the receiving adapter when the caller holds it; otherwise the source's own + transport provenance is consulted. *transport_profile* names the receiving bot's owning + profile when the caller knows it by construction (the runner's per-profile handlers); + ``None`` = derive it from the adapter registry, primary when unknown. *primary_home* is the + primary bot's home for authorization (default: the process home, never a per-turn override). + + Stamps ``source.profile`` the way the ingress handlers always did (routed name, else a + secondary's own name; ``None`` stays ``None`` for the primary so the wire is unchanged) and + ``_authorization_profile_home`` for the existing authorization readers. + + Raises :class:`IdentityUnresolved` under multiplexing when the route is rejected. + """ + from gateway.profile_routing import ProfileRouteRejected + from hermes_constants import get_hermes_home, get_process_hermes_home + + multiplexed = bool(getattr(getattr(runner, "config", None), "multiplex_profiles", False)) + primary_profile = _name(getattr(runner, "_primary_profile_name", None)) + if primary_profile is None: + active = getattr(runner, "_active_profile_name", None) + primary_profile = (_name(active()) if callable(active) else None) or "default" + platform = getattr(source, "platform", None) + + owner_profile: Optional[str] = None + if adapter is None: + owner = runner._transport_owner(source) + if owner is not None: + adapter, owner_profile = owner + else: + if getattr(source, "_transport_adapter_ref", None) is None: + source._transport_adapter_ref = weakref.ref(adapter) + _registered, owner_profile = runner._owning_profile(adapter, platform) + transport_name = _name(transport_profile) or _name(owner_profile) or primary_profile + transport_ref = weakref.ref(adapter) if adapter is not None else None + + if not multiplexed: + home = Path(get_hermes_home()) + identity = RoutingIdentity( + transport_profile=primary_profile, runtime_profile=primary_profile, + authorization_home=home, runtime_home=home, multiplexed=False, transport=transport_ref) + setattr(source, _IDENTITY_ATTR, identity) + return identity + + if transport_name == primary_profile: + authorization_home = Path(primary_home) if primary_home is not None else Path(get_process_hermes_home()) + else: + from hermes_cli.profiles import get_profile_dir + authorization_home = get_profile_dir(transport_name) + source._authorization_profile_home = authorization_home + + where = f"{getattr(platform, 'value', platform)}/{getattr(source, 'chat_id', '')}" + if getattr(source, "profile_route_rejected", False) is True: + raise IdentityUnresolved(f"{where}: profile route rejected") + if _name(getattr(source, "profile", None)) is None: + adapter_profile = None if transport_name == primary_profile else transport_name + try: + # The primary keeps the historical one-argument call (its adapter profile is None). + routed = ( + runner._profile_name_for_source(source) if adapter_profile is None + else runner._profile_name_for_source(source, adapter_profile=adapter_profile)) + except ProfileRouteRejected as exc: + source.profile_route_rejected = True + raise IdentityUnresolved(f"{where}: {exc}") from exc + source.profile = routed or adapter_profile + + runtime_name = _name(source.profile) or primary_profile + # A routed runtime goes through the runner's resolver (missing-profile fallback + warning); + # a bot serving its own profile runs where it authorizes. + runtime_home = ( + authorization_home if runtime_name == transport_name + else Path(runner._resolve_profile_home_for_source(source))) + identity = RoutingIdentity( + transport_profile=transport_name, runtime_profile=runtime_name, + authorization_home=authorization_home, runtime_home=runtime_home, + multiplexed=True, transport=transport_ref) + setattr(source, _IDENTITY_ATTR, identity) + return identity diff --git a/gateway/session_recovery.py b/gateway/session_recovery.py index cdc13acd43..770f69295a 100644 --- a/gateway/session_recovery.py +++ b/gateway/session_recovery.py @@ -35,9 +35,14 @@ class SessionRecoveryMixin: def _resolve_profile_for_key(self, source: Optional[SessionSource] = None) -> Optional[str]: """Profile namespace for session keys: None when multiplexing is off (legacy - ``agent:main``), else ``source.profile`` or the active profile.""" + ``agent:main``), else the pinned identity's runtime profile, ``source.profile`` or the + active profile.""" if not getattr(self.config, "multiplex_profiles", False): return None + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.session_key_profile if source is not None and source.profile: return source.profile try: diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 93d23ead83..25e3ab1cfd 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -458,7 +458,13 @@ class GatewaySlashCommandsMixin( reason, session_key, len(fallback_keys), ", ".join(fallback_keys)) return EphemeralReply(t("gateway.stop.stopped")) - # No running agent anywhere for this scope. A platform status indicator can still be stuck — + # No running agent anywhere for this scope. Background delegations the session dispatched in an + # earlier turn still count as "active": stop them; each returns as an interrupted completion. + from tools.async_delegation import interrupt_for_session + if interrupt_for_session(session_key=session_key, reason="stop_command", + parent_session_id=str(getattr(session_entry, "session_id", "") or "")): + return EphemeralReply(t("gateway.stop.stopped")) + # A platform status indicator can still be stuck — # e.g. Slack's persistent assistant.threads.setStatus survives a gateway restart or a turn # that died without a final send. # Best-effort clear so /stop always dismisses a phantom "is thinking...". See #32295. diff --git a/hermes_cli/auth_codex.py b/hermes_cli/auth_codex.py index f4c029462b..db55aefe80 100644 --- a/hermes_cli/auth_codex.py +++ b/hermes_cli/auth_codex.py @@ -176,8 +176,14 @@ def _save_codex_tokens(tokens: Dict[str, str], last_refresh: str = None, label: def _recover_codex_tokens_from_cli(reason: str) -> Optional[Dict[str, str]]: - """Adopt a valid Codex CLI token pair into Hermes auth, if available.""" + """Adopt a valid Codex CLI token pair into Hermes auth, if available. + + Automatic adoption only; the interactive import offer in ``_login_openai_codex`` asks first and is + not subject to ``auth.adopt_external_logins``.""" + from agent.credential_sources import adopt_external_logins_enabled from hermes_cli.auth import _import_codex_cli_tokens, _save_codex_tokens + if not adopt_external_logins_enabled(): + return None imported = _import_codex_cli_tokens() # Require BOTH tokens before adopting: persisting a payload without a usable refresh_token # would only break the next refresh cycle. diff --git a/hermes_cli/auth_commands.py b/hermes_cli/auth_commands.py index 1b5396159c..c4ad8f5e16 100644 --- a/hermes_cli/auth_commands.py +++ b/hermes_cli/auth_commands.py @@ -28,6 +28,8 @@ _OAUTH_CAPABLE_PROVIDERS = {"anthropic", "nous", "openai-codex", "xai-oauth", "q # ...and default to it when ``--type`` is omitted. OpenRouter stays API-key-first: the documented # ``hermes auth add openrouter --api-key sk-or-...`` must keep working with no ``--type``. _OAUTH_DEFAULT_PROVIDERS = _OAUTH_CAPABLE_PROVIDERS - {"openrouter"} +# Providers whose sibling CLI login Hermes may borrow (``auth.adopt_external_logins``). +EXTERNAL_LOGIN_PROVIDERS = {"anthropic", "openai-codex"} def _get_custom_provider_entries() -> list[dict]: @@ -491,6 +493,15 @@ def auth_list_command(args) -> None: ) print(row.rstrip()) print() + if not provider_filter or provider_filter in EXTERNAL_LOGIN_PROVIDERS: + _print_external_login_notice() + + +def _print_external_login_notice() -> None: + """One line telling the user why no Codex CLI / Claude Code login shows up when adoption is off.""" + from agent.credential_sources import EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE, adopt_external_logins_enabled + if not adopt_external_logins_enabled(): + print(EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE) def auth_remove_command(args) -> None: @@ -607,6 +618,8 @@ def auth_status_command(args) -> None: if not status.get("logged_in"): reason = status.get("error") print(f"{provider}: logged out" + (f" ({reason})" if reason else "")) + if provider in EXTERNAL_LOGIN_PROVIDERS: + _print_external_login_notice() return print(f"{provider}: logged in") for key in ("auth_type", "client_id", "redirect_uri", "scope", "expires_at", "api_base_url"): diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index e87d9b4699..47fd759bb8 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -187,8 +187,7 @@ COMMAND_REGISTRY: list[CommandDef] = [ subcommands=("manual", "smart", "off")), CommandDef("reasoning", "Manage reasoning effort and display", "Configuration", args_hint="[level|show|hide|full|clamp] [--global]", - subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global"), - desktop="advanced"), + subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global")), CommandDef("fast", "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)", "Configuration", args_hint="[normal|fast|auto|cold|status] [--global]", subcommands=("normal", "fast", "auto", "cold", "status", "on", "off", "--global"), diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index e2a0dc1d61..d383fd5d23 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -747,14 +747,15 @@ DEFAULT_CONFIG = { "monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine # Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying # the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper). - # enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of - # replayed input tokens over the review loop (iterations capped at 16); the loop stops - # before crossing it. <= 0 = unlimited. + # enabled=false skips auto spawns (/refine still works). An explicit max_input_tokens caps + # the SUM of replayed input tokens over the review loop (iterations capped at 16); the loop + # stops before crossing it. When unset, the runtime derives a budget from the active model + # context window. <= 0 = unlimited. # reasoning_effort is IGNORED while the review stays on the main model: the fork inherits the # conversation's reasoning config verbatim so its request bytes keep the parent's warm # prompt-cache prefix (#30532). Set provider/model below to route the review to another model # if you want a different effort level; a one-time warning says so when the key is set. - "background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000}, + "background_review": {"enabled": True, **_aux(120)}, # No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset # (moa.presets..reference_models[].reasoning_effort / aggregator.reasoning_effort). "moa_reference": _aux(900, reasoning_effort=False), @@ -1647,6 +1648,14 @@ DEFAULT_CONFIG = { # Custom personalities: {"name": "system prompt"} or {"name": {"description", "system_prompt", # "tone", "style"}}. "personalities": {}, + "auth": { # Login policy (credentials themselves live in auth.json / .env). + # Borrow and refresh the Codex CLI (~/.codex/auth.json) and Claude Code (~/.claude/.credentials.json) + # logins automatically when Hermes has no usable login of its own. Their refresh tokens are single-use + # and rotate, so two programs on one login can log each other out; set false to make Hermes use only + # its own logins (`hermes auth add `). `hermes auth add openai-codex` still offers the import + # interactively. + "adopt_external_logins": True, + }, "security": { # Security: pre-exec scanning via tirith plus related guards. "allow_private_urls": False, # allow requests to private/internal IPs (OpenWrt, VPNs) # CIDR blocks a local TUN proxy answers DNS with (Mihomo/Clash fake-ip, Surge enhanced). diff --git a/hermes_cli/gateway_windows.py b/hermes_cli/gateway_windows.py index 56eae9a9fd..7a34788147 100644 --- a/hermes_cli/gateway_windows.py +++ b/hermes_cli/gateway_windows.py @@ -683,6 +683,22 @@ def _spawn_detached(script_path: Path | None = None, home: Path | None = None) - return proc.pid +def _stdin_is_interactive(*, isatty: bool, console_mode_ok: bool | None) -> bool: + """A human can answer a prompt only on a real console. The Windows CRT reports isatty()==True for + every character device — the NUL device included (`hermes gateway start < NUL`, stdin=DEVNULL) — so + isatty must be confirmed by GetConsoleMode accepting the handle (#113977). ``console_mode_ok`` is + None where that fact does not exist (not Windows) and isatty alone decides.""" + return isatty and console_mode_ok is not False + + +def _stdin_console_mode_ok() -> bool | None: + if sys.platform != "win32": + return None + kernel32 = ctypes.windll.kernel32 + handle = kernel32.GetStdHandle(-10) # STD_INPUT_HANDLE + return bool(kernel32.GetConsoleMode(handle, ctypes.byref(ctypes.c_ulong()))) + + def _install_choice_from_env(name: str) -> bool | None: raw = os.environ.get(name) if raw is None: @@ -785,7 +801,7 @@ def install( _start_or_report_running() else: print("ℹ Gateway not started and no auto-start service installed.") - print(" Run later with: hermes gateway start") + print(" Run in the foreground later with: hermes gateway run") return task_name = get_task_name() @@ -1493,17 +1509,26 @@ def start() -> None: return if not is_task_registered() and not is_startup_entry_installed(): - from hermes_cli.setup import prompt_yes_no + # Login persistence is a lasting system change: a bare ``start`` installs it only on an explicit + # answer — the HERMES_GATEWAY_INSTALL_START_ON_LOGIN override or a real TTY prompt — never on a + # non-TTY default (#113977). Declining still starts the gateway; the command is ``start``. + start_on_login = _install_choice_from_env("HERMES_GATEWAY_INSTALL_START_ON_LOGIN") + if start_on_login is None: + from hermes_cli.setup import is_interactive_stdin, is_noninteractive, prompt_yes_no - print("✗ Gateway service is not installed") - if not prompt_yes_no(" Install it now so the gateway starts on login?", True): - print(" Run: hermes gateway install") - return - install(force=False) - if not is_task_registered() and not is_startup_entry_installed(): - print("⚠ Gateway install did not complete in this process.") - print(" If a UAC prompt opened, approve it, then run: hermes gateway start") + print("✗ Gateway service is not installed") + if is_noninteractive() or not _stdin_is_interactive( + isatty=is_interactive_stdin(), console_mode_ok=_stdin_console_mode_ok() + ): + start_on_login = False + else: + start_on_login = prompt_yes_no(" Install it now so the gateway starts on login?", True) + if start_on_login: + # install() starts the gateway itself (start_now) and reports the outcome — including a UAC + # hand-off to an elevated child — so there is nothing left to spawn or to warn about here. + install(force=False, start_now=True, start_on_login=True) return + print("ℹ Login auto-start not installed; add it later with: hermes gateway install") elif is_task_registered(): reconcile_scheduled_task(get_task_name()) # like systemd's regenerate-on-stale before a start diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py index be81bf6709..fd3069bafb 100644 --- a/hermes_cli/kanban_db.py +++ b/hermes_cli/kanban_db.py @@ -305,6 +305,11 @@ DEFAULT_CRASH_GRACE_SECONDS = 30 # breaker must never trip on a throttle). 75 == BSD EX_TEMPFAIL. KANBAN_RATE_LIMIT_EXIT_CODE = 75 +# Worker exit "provider rejected the configuration": credential revoked (401/403), model gone +# (404), TLS chain broken — a retry cannot fix it, so the dispatcher parks the card blocked on +# the FIRST occurrence instead of spending ``failure_limit`` identical spawns. 78 == BSD EX_CONFIG. +KANBAN_TERMINAL_PROVIDER_EXIT_CODE = 78 + def _resolve_crash_grace_seconds() -> int: """``HERMES_KANBAN_CRASH_GRACE_SECONDS`` (0 = immediate, for tests) else default.""" diff --git a/hermes_cli/kanban_db_dispatch.py b/hermes_cli/kanban_db_dispatch.py index c74ed448fd..18dacbbcd9 100644 --- a/hermes_cli/kanban_db_dispatch.py +++ b/hermes_cli/kanban_db_dispatch.py @@ -246,6 +246,8 @@ def _exit_code_kind(code: int) -> "tuple[str, int]": return ("clean_exit", 0) if code == _kb.KANBAN_RATE_LIMIT_EXIT_CODE: return ("rate_limited", code) + if code == _kb.KANBAN_TERMINAL_PROVIDER_EXIT_CODE: + return ("terminal_provider", code) return ("nonzero_exit", code) @@ -1016,6 +1018,9 @@ class _DeadWorker: event_payload: dict protocol_violation: bool = False rate_limited: bool = False + terminal_provider: bool = False + """``KANBAN_TERMINAL_PROVIDER_EXIT_CODE``: the provider rejected the worker's + credential/model — trips the breaker on this first occurrence.""" @property def run_outcome(self) -> str: @@ -1084,6 +1089,18 @@ def _classify_dead_worker_exit( {"pid": pid, "claimer": claimer, "exit_code": code}, rate_limited=True, ) + if kind == "terminal_provider": + # The worker classified its own provider failure as unhealable (credential + # revoked, model gone): every further spawn would hit the same wall, so + # ``_account_crashes`` trips the breaker now instead of after ``failure_limit``. + return _DeadWorker( + kind, code, + f"pid {pid} exited on a terminal provider error (exit {code}): the provider rejected " + "this profile's credential or model — fix the configuration, then unblock.", + "crashed", + {"pid": pid, "claimer": claimer, "exit_kind": kind, "exit_code": code, "terminal_provider": True}, + terminal_provider=True, + ) if kind == "nonzero_exit": error_text = f"pid {pid} exited with code {code}" elif kind == "signaled": @@ -1103,9 +1120,9 @@ class _CrashSweep: crashed: list[str] = field(default_factory=list) rate_limited: list[str] = field(default_factory=list) - # ``(task_id, pid, claimer, protocol_violation, error_text)``: accounted - # after the txn via ``_record_task_failure`` (needs its own write_txn). - crash_details: list[tuple[str, int, str, bool, str]] = field(default_factory=list) + # ``(task_id, pid, claimer, dead_worker)``: accounted after the txn via + # ``_record_task_failure`` (needs its own write_txn). + crash_details: list[tuple[str, int, str, _DeadWorker]] = field(default_factory=list) # Worker-exit observer payloads, fired only after every reclaim/accounting # txn has committed. exited_hook_payloads: list[dict] = field(default_factory=list) @@ -1177,9 +1194,7 @@ def _reclaim_dead_workers(conn: sqlite3.Connection, board: Optional[str] = None) sweep.rate_limited.append(row["id"]) else: sweep.crashed.append(row["id"]) - sweep.crash_details.append( - (row["id"], pid, row["claim_lock"], dead.protocol_violation, dead.error_text) - ) + sweep.crash_details.append((row["id"], pid, row["claim_lock"], dead)) return sweep @@ -1188,16 +1203,18 @@ def _account_crashes(conn: sqlite3.Connection, crash_details: list) -> list[str] Protocol violations get a BOUNDED violation-only budget independent of ``consecutive_failures`` (per-task ``max_retries`` takes precedence); - systemic same-error crashes (>= 3 identical fingerprints this tick) trip - immediately. + systemic same-error crashes (>= 3 identical fingerprints this tick) and + terminal provider errors (credential revoked, model gone — a retry cannot + heal them) trip immediately. """ auto_blocked: list[str] = [] fp_counts: dict[str, int] = {} - for _, _, _, _, err_text in crash_details: - fp = _error_fingerprint(err_text) + for _, _, _, dead in crash_details: + fp = _error_fingerprint(dead.error_text) fp_counts[fp] = fp_counts.get(fp, 0) + 1 - for tid, pid, claimer, protocol_violation, error_text in crash_details: - if protocol_violation: + for tid, pid, claimer, dead in crash_details: + error_text = dead.error_text + if dead.protocol_violation: streak = _protocol_violation_streak(conn, tid) trow = conn.execute("SELECT max_retries FROM tasks WHERE id = ?", (tid,)).fetchone() if trow is None: @@ -1227,6 +1244,20 @@ def _account_crashes(conn: sqlite3.Connection, crash_details: list) -> list[str] "protocol_violation_limit": violation_limit, }, ) + elif dead.terminal_provider: + # A retry cannot heal a revoked credential or a missing model, so + # the whole ``failure_limit`` budget would be spent on identical + # failures. ``force_trip`` blocks now, sticky: ``recompute_ready`` + # must not auto-resume it before the operator fixes the provider. + tripped = _record_task_failure( + conn, tid, + error=error_text, + outcome="crashed", + force_trip=True, + release_claim=False, + end_run=False, + event_payload_extra={"pid": pid, "claimer": claimer, "terminal_provider": True}, + ) else: is_systemic = fp_counts.get(_error_fingerprint(error_text), 0) >= 3 extra = {"pid": pid, "claimer": claimer} diff --git a/hermes_cli/kanban_diagnostics.py b/hermes_cli/kanban_diagnostics.py index 9e255ab9f1..de123d9c7a 100644 --- a/hermes_cli/kanban_diagnostics.py +++ b/hermes_cli/kanban_diagnostics.py @@ -112,6 +112,18 @@ def _latest_event_ts(events: Iterable[Any], kinds: set[str]) -> int: return max([0, *(_event_ts(ev) for ev in events if _event_kind(ev) in kinds)]) +def _latest_gave_up_is_terminal_provider(events: Iterable[Any]) -> bool: + """True when the most recent breaker trip was a terminal provider error (credential + revoked, model gone) and nothing has resumed the task since.""" + for ev in reversed(list(events)): + kind = _event_kind(ev) + if kind == "gave_up": + return bool(_parse_payload(ev).get("terminal_provider")) + if kind in {"unblocked", "promoted", "completed", "claimed"}: + return False + return False + + def _cli_hint(label: str, command: str, *, suggested: bool = False) -> DiagnosticAction: return DiagnosticAction(kind="cli_hint", label=label, payload={"command": command}, suggested=suggested) @@ -369,8 +381,12 @@ def _rule_repeated_failures(task, events, runs, now, cfg) -> list[Diagnostic]: threshold = _positive_int(_failure_threshold(cfg), 3) failure_limit = _positive_int(cfg.get("failure_limit"), threshold) failures = _first_field(task, "consecutive_failures", "spawn_failures", 0) - if failures is None or failures < threshold: + # A terminal provider error (credential revoked, model gone) blocks the card after ONE + # attempt, below any threshold; it still needs an operator, so diagnose it now. + terminal_trip = _latest_gave_up_is_terminal_provider(events) + if not terminal_trip and (failures is None or failures < threshold): return [] + failures = failures or 0 last_err = _first_field(task, "last_failure_error", "last_spawn_error") assignee = _task_field(task, "assignee") @@ -397,7 +413,15 @@ def _rule_repeated_failures(task, events, runs, now, cfg) -> list[Diagnostic]: severity = "critical" if failures >= threshold * 2 else "error" err_snippet = _error_snippet(last_err) outcome_label = _OUTCOME_LABELS.get(most_recent_outcome or "", "failure") - if err_snippet: + if terminal_trip: + title = "Provider rejected this profile's credential or model — blocked after one attempt" + detail = ( + f"The worker's provider call failed with an error a retry cannot fix (revoked or invalid " + f"API key, model not found), so the dispatcher blocked the task instead of spending the " + f"{failure_limit}-attempt retry budget on it. Full last error:\n\n{err_snippet}\n\n" + f"Fix the assignee profile's provider credentials/model, then unblock the task." + ) + elif err_snippet: title = f"Agent {outcome_label} x{failures}: {err_snippet.splitlines()[0][:160]}" detail = ( f"This task has failed {failures} times in a row (most recent: {outcome_label}). Full " diff --git a/hermes_cli/web_server_config.py b/hermes_cli/web_server_config.py index b5fd93f441..b502ab914f 100644 --- a/hermes_cli/web_server_config.py +++ b/hermes_cli/web_server_config.py @@ -101,6 +101,14 @@ _SCHEMA_OVERRIDES: Dict[str, Dict[str, Any]] = { "description": "Refuse Docker sandboxes when egress is enabled but not configured/running", "category": "security", }, + "auth.adopt_external_logins": { + "type": "boolean", + "description": ( + "Borrow and refresh the Codex CLI / Claude Code logins when Hermes has no usable login of its own. " + "Off: Hermes uses only its own logins (`hermes auth add `)." + ), + "category": "security", + }, "tts.provider": _select( "Text-to-speech provider", "edge", "elevenlabs", "openai", "xai", "minimax", "mistral", "gemini", "neutts", "kittentts", "piper", @@ -197,6 +205,7 @@ _CATEGORY_MERGE: Dict[str, str] = { "session": "general", "nous": "agent", "connections": "agent", + "auth": "security", } diff --git a/plugin-catalog/aihubmix.yaml b/plugin-catalog/aihubmix.yaml new file mode 100644 index 0000000000..9d6aaff0d0 --- /dev/null +++ b/plugin-catalog/aihubmix.yaml @@ -0,0 +1,15 @@ +name: aihubmix +repo: https://github.com/AIhubmix/hermes-provider-aihubmix +sha: f0b6b37858970ced32b73447faaaaad6db899779 +subdir: aihubmix +description: AIHubMix — unified OpenAI-compatible access to 400+ models; /model picker filtered to tool-capable routes. +maintainer: AIHubMix +tier: community +category: models +docs_url: https://github.com/AIhubmix/hermes-provider-aihubmix#readme +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: + - AIHUBMIX_API_KEY diff --git a/plugin-catalog/artifact-relay.yaml b/plugin-catalog/artifact-relay.yaml new file mode 100644 index 0000000000..1063490bf1 --- /dev/null +++ b/plugin-catalog/artifact-relay.yaml @@ -0,0 +1,14 @@ +name: artifact-relay +repo: https://github.com/eloktev/hermes-artifact-relay +sha: c91fa9229121dd616a275a45447301eacca45e6f +description: Publish and read private Markdown or HTML reports through your Artifact Relay server. +maintainer: eloktev +tier: community +category: tools +docs_url: https://github.com/eloktev/hermes-artifact-relay#readme +platforms: [linux, macos, windows] +capabilities: + provides_tools: [artifact_read, artifact_publish] + provides_hooks: [] + provides_middleware: [] + requires_env: [ARTIFACT_RELAY_API_TOKEN] diff --git a/plugin-catalog/hermes-muse-code.yaml b/plugin-catalog/hermes-muse-code.yaml new file mode 100644 index 0000000000..ee8b9ee42c --- /dev/null +++ b/plugin-catalog/hermes-muse-code.yaml @@ -0,0 +1,15 @@ +name: hermes-muse-code +repo: https://github.com/TheStreamCode/hermes-muse-code +sha: e33501185e328b5597fd98c6de916495922ec727 +description: Muse Spark in Hermes billed to the Muse Code monthly login (device-code login, no API key). +maintainer: TheStreamCode +tier: community +category: models +requires_hermes: ">=0.21.3" +docs_url: https://github.com/TheStreamCode/hermes-muse-code +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] diff --git a/plugin-catalog/hermes-security-audit.yaml b/plugin-catalog/hermes-security-audit.yaml new file mode 100644 index 0000000000..561f6c033f --- /dev/null +++ b/plugin-catalog/hermes-security-audit.yaml @@ -0,0 +1,16 @@ +name: hermes-security-audit +repo: https://github.com/dafka007/hermes-security-audit +sha: 60a7661bbe2f255c507d620275c59782cabe0335 +description: Approval-gated local security audit for Hermes Agent using Gitleaks, OSV-Scanner, and Semgrep CE. +maintainer: dafka007 +tier: community +category: tools +docs_url: https://github.com/dafka007/hermes-security-audit#readme +platforms: [] +capabilities: + provides_tools: + - security_audit + provides_hooks: + - pre_tool_call + provides_middleware: [] + requires_env: [] diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml new file mode 100644 index 0000000000..c6a5bc8039 --- /dev/null +++ b/plugin-catalog/hermes-talk.yaml @@ -0,0 +1,24 @@ +name: hermes-talk +repo: https://github.com/TheSmokeDev/hermes-talk +sha: 7219f2fdff353df72f5323b219bfe759991b0365 +description: >- + Realtime voice for Hermes through terminal, Discord, dashboard and Desktop. + Live voice mode (TALK_VOICE_MODE=live, TALK_LIVE_AUTH=subscription) connects with your + own Codex OAuth session to the undocumented chatgpt.com/backend-api/codex/realtime/calls + endpoint, which is outside the vendor's published API terms and may change or stop + working without notice. +maintainer: TheSmokeDev +tier: community +category: voice +version: "0.21.0" +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/7219f2fdff353df72f5323b219bfe759991b0365/README.md +capabilities: + provides_tools: [] + provides_hooks: + - on_session_end + - subagent_start + - subagent_stop + - post_tool_call + - pre_approval_request + provides_middleware: [] + requires_env: [] diff --git a/plugin-catalog/morning-briefing.yaml b/plugin-catalog/morning-briefing.yaml new file mode 100644 index 0000000000..a1998abb90 --- /dev/null +++ b/plugin-catalog/morning-briefing.yaml @@ -0,0 +1,17 @@ +name: morning-briefing +repo: https://github.com/drkpxl/morning-briefing +sha: 347ce44a1e71173285b2943a7a97bbdece8327b2 +description: 'Personal daily newspaper for Hermes: gathers weather, calendar, AQI, curated + 36-hour news (timestamped Reddit RSS + X + web search, per-story QR codes), and + newsletter digests, renders one 8.5x11 newsprint page, and prints it every morning + via a cron job.' +maintainer: drkpxl +tier: community +category: automation +docs_url: https://github.com/drkpxl/morning-briefing#readme +platforms: [macos, linux] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] \ No newline at end of file diff --git a/plugin-catalog/openalex.yaml b/plugin-catalog/openalex.yaml index abdcb45958..90df1f0802 100644 --- a/plugin-catalog/openalex.yaml +++ b/plugin-catalog/openalex.yaml @@ -1,28 +1,26 @@ name: openalex repo: https://github.com/Adolanium/hermes-plugin-openalex -sha: 04e673e34bfd09ea9cf20626373ce48df096f813 -description: OpenAlex for Hermes. Search 250M scholarly works, authors, journals and institutions, with - a USD budget guard, per-call cost prediction, and response shaping that keeps a 2.8 MB paper inside - the context window. +sha: 47aafae87256ef634f32c18a69bf35d39c2c5682 +description: Scholarly search, citation traversal, grouped counts, and fulltext access through OpenAlex with session spending limits. maintainer: Adolanium tier: community category: tools docs_url: https://github.com/Adolanium/hermes-plugin-openalex#readme capabilities: provides_tools: - - openalex_account - - openalex_classify - - openalex_count - - openalex_fields - - openalex_fulltext - - openalex_get - - openalex_harvest - - openalex_related - - openalex_resolve - - openalex_search + - openalex_resolve + - openalex_get + - openalex_count + - openalex_search + - openalex_related + - openalex_account + - openalex_fields + - openalex_fulltext + - openalex_classify + - openalex_harvest provides_hooks: - - on_session_reset - - on_session_start + - on_session_start + - on_session_reset provides_middleware: [] requires_env: - - OPENALEX_API_KEY + - OPENALEX_API_KEY diff --git a/plugin-catalog/prism.yaml b/plugin-catalog/prism.yaml new file mode 100644 index 0000000000..fa8303810b --- /dev/null +++ b/plugin-catalog/prism.yaml @@ -0,0 +1,15 @@ +name: prism +repo: https://github.com/prismhq/hermes-prism-provider +sha: f68a9bff76a1dba30389f56a671292e48b22887a +description: Prism OpenAI-compatible inference with long-context DeepSeek models. +maintainer: Prism +tier: community +category: models +requires_hermes: ">=0.21.3" +docs_url: https://github.com/prismhq/hermes-prism-provider#readme +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: + - PRISM_API_KEY diff --git a/plugin-catalog/pstack.yaml b/plugin-catalog/pstack.yaml new file mode 100644 index 0000000000..b8c5480f55 --- /dev/null +++ b/plugin-catalog/pstack.yaml @@ -0,0 +1,15 @@ +name: pstack +repo: https://github.com/Zoeille/pstack +sha: 932d8174eba6581d3a117c2d2b7796e7eb27953e +description: "Engineering rigor stack: poteto-mode orchestrator with 7 playbooks (new-feature, bugfix, refactor, testing, one-shot, migrate, explore), plus how, architect, interrogate, swarm, and unslop skills." +maintainer: Zoeille +tier: community +category: tools +version: "0.1.0" +docs_url: https://github.com/Zoeille/pstack +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] diff --git a/plugin-catalog/search1api.yaml b/plugin-catalog/search1api.yaml new file mode 100644 index 0000000000..db01202012 --- /dev/null +++ b/plugin-catalog/search1api.yaml @@ -0,0 +1,19 @@ +name: search1api +repo: https://github.com/superagents-lab/hermes-search1api +sha: 14c861746c50ef5ab32edec8b5c9ad0815f40cec +description: Live web search, news, page crawl, sitemap, and trending via Search1API +maintainer: superagents-lab +tier: community +category: web +docs_url: https://github.com/superagents-lab/hermes-search1api#readme +version: "0.1.0" +capabilities: + provides_tools: + - search1api_search + - search1api_news + - search1api_crawl + - search1api_sitemap + - search1api_trending + provides_hooks: [] + provides_middleware: [] + requires_env: [] diff --git a/plugin-catalog/you.yaml b/plugin-catalog/you.yaml new file mode 100644 index 0000000000..e009e2968c --- /dev/null +++ b/plugin-catalog/you.yaml @@ -0,0 +1,20 @@ +name: you +repo: https://github.com/youdotcom-oss/agent-skills +sha: 08f3d80e7ca9518ccc64b38b365a78cc4a762691 +description: 'You.com skills for live web search, URL content extraction, cited research, finance + research, and integration discovery via the You.com MCP servers (api.you.com). Portable Agent + Plugins v1 package (mcp.json + skills/, five skills bundled); auth via YDC_API_KEY, OAuth, or + MPP/x402 — the free profile works keyless.' +maintainer: youdotcom-oss +tier: community +category: web +requires_hermes: ">=0.20" +docs_url: https://github.com/youdotcom-oss/agent-skills#readme +version: "0.6.0" +image: https://raw.githubusercontent.com/youdotcom-oss/agent-skills/08f3d80e7ca9518ccc64b38b365a78cc4a762691/assets/logo.png +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] diff --git a/plugins/cron_providers/chronos/__init__.py b/plugins/cron_providers/chronos/__init__.py index 0af25f24bc..9a846aa73d 100644 --- a/plugins/cron_providers/chronos/__init__.py +++ b/plugins/cron_providers/chronos/__init__.py @@ -15,6 +15,8 @@ from typing import Any, Dict from cron.scheduler_provider import CronScheduler +from ._nas_client import NasCronClientError + logger = logging.getLogger("cron.chronos") @@ -35,6 +37,12 @@ class ChronosCronScheduler(CronScheduler): self._armed: Dict[str, str] = {} self._lock = threading.Lock() self._client = None # lazily constructed (no network in is_available) + # Set when NAS answered 403 invalid_client: the Nous token in auth.json is not this + # instance's provisioned identity, so every arm would fail the same way for the life of + # the process. Once set, NAS is left alone and the built-in ticker fires jobs (#97494). + self._identity_rejected = False + self._stop_event = None + self._ticker_kwargs: Dict[str, Any] = {} @property def name(self) -> str: @@ -66,6 +74,10 @@ class ChronosCronScheduler(CronScheduler): def start(self, stop_event, *, adapters=None, loop=None, interval=60): """Arm all enabled jobs via NAS, then RETURN — no loop, no periodic wake (scale-to-zero).""" + # Kept so a later identity rejection (boot or mid-life re-arm) can hand this process's + # fires to the built-in ticker with the gateway's own adapters/loop. + self._ticker_kwargs = {"adapters": adapters, "loop": loop, "interval": interval} + self._stop_event = stop_event # A new lifecycle can't prove what an interrupted process did: classify unknown, never requeue. self.recover_interrupted() self._reconcile_logged(logger.warning, "start()") @@ -74,15 +86,23 @@ class ChronosCronScheduler(CronScheduler): pass def on_jobs_changed(self) -> None: - self._reconcile_logged(logger.debug, "on_jobs_changed") + if not self._identity_rejected: + self._reconcile_logged(logger.debug, "on_jobs_changed") def register_job(self, job: Dict[str, Any]) -> None: """Arm the first one-shot for a new job; may raise so creation can report it.""" - self._arm_one_shot(job) + try: + self._arm_one_shot(job) + except NasCronClientError as e: + if not e.identity_rejected: + raise + self._note_identity_rejected() # the job is stored; the ticker fires it def _arm_one_shot(self, job: Dict[str, Any]) -> None: """Arm one one-shot at next_run_at (agent computes the time; NAS executes). dedup_key=(job_id, fire_at) makes re-arming the same fire a no-op.""" + if self._identity_rejected: + return # the built-in ticker owns this process's fires; NAS would 403 again job_id = job["id"] fire_at = job.get("next_run_at") if not fire_at: @@ -93,11 +113,38 @@ class ChronosCronScheduler(CronScheduler): with self._lock: self._armed[job_id] = fire_at + def _note_identity_rejected(self) -> None: + """403 invalid_client is deterministic: NAS maps the bearer to a provisioned instance via an + ``agent:*`` client or the hosted bootstrap session, and a plain ``hermes auth`` login is + neither — re-logging in cannot fix it, which is what users try first (#97494). Without + NAS the jobs have no trigger at all (the misfire sweep runs them ``misfire_grace_minutes`` + late), so the built-in ticker takes over this process's fires.""" + with self._lock: + if self._identity_rejected: + return + self._identity_rejected = True + logger.warning( + "Chronos: NAS rejected this agent's Nous credential for agent-cron (403 invalid_client). " + "The Nous token in auth.json is not this instance's provisioned identity (an agent:* client " + "or the hosted bootstrap session), so no job can be armed. A normal `hermes auth` re-login " + "cannot fix this; the hosted credential has to be restored from the Nous Portal. Falling back " + "to the built-in cron ticker for this process so scheduled jobs keep firing on time.") + if self._stop_event is None: + return # start() never ran (e.g. a CLI `hermes cron add`); nothing to tick here + from agent.memory_provider import spawn_context_thread + from cron.scheduler_provider import InProcessCronScheduler + spawn_context_thread( + InProcessCronScheduler().start, name="cron-scheduler-chronos-fallback", + args=(self._stop_event,), kwargs=self._ticker_kwargs).start() + def _arm_logged(self, job: Dict[str, Any], what: str) -> None: """Best-effort arm: log a warning instead of raising (reconcile/fire must not die).""" try: self._arm_one_shot(job) except Exception as e: + if isinstance(e, NasCronClientError) and e.identity_rejected: + self._note_identity_rejected() + return logger.warning("Chronos failed to %s: %s", what, e) def _cancel(self, job_id: str) -> None: diff --git a/plugins/cron_providers/chronos/_nas_client.py b/plugins/cron_providers/chronos/_nas_client.py index 6777a271d5..2ef7843b5f 100644 --- a/plugins/cron_providers/chronos/_nas_client.py +++ b/plugins/cron_providers/chronos/_nas_client.py @@ -16,7 +16,22 @@ _LIST_PATH = "/api/agent-cron/list" class NasCronClientError(RuntimeError): - """Raised when a NAS agent-cron call fails (non-2xx or transport error).""" + """Raised when a NAS agent-cron call fails (non-2xx or transport error). + + ``status`` is the HTTP status (None on transport error) and ``error_code`` the OAuth-style + ``error`` field of a JSON error body (``invalid_client`` marks a deterministic identity + rejection the provider can act on, unlike a transient 5xx). + """ + + def __init__(self, message: str, *, status: int | None = None, error_code: str = "") -> None: + super().__init__(message) + self.status = status + self.error_code = error_code + + @property + def identity_rejected(self) -> bool: + """NAS refused the bearer as not belonging to a provisioned agent (never transient).""" + return self.status == 403 and self.error_code == "invalid_client" class NasCronClient: @@ -41,7 +56,12 @@ class NasCronClient: except Exception as e: raise NasCronClientError(f"{method} {path} failed: {e}") from e if resp.status_code // 100 != 2: - raise NasCronClientError(f"{method} {path} returned {resp.status_code}: {resp.text[:200]}") + error_code = "" + with contextlib.suppress(Exception): + error_code = str((resp.json() or {}).get("error") or "") + raise NasCronClientError( + f"{method} {path} returned {resp.status_code}: {resp.text[:200]}", + status=resp.status_code, error_code=error_code) with contextlib.suppress(Exception): return resp.json() if resp.content else {} return {} diff --git a/plugins/image_gen/_common.py b/plugins/image_gen/_common.py index efc8b8f4ad..aea9d21cda 100644 --- a/plugins/image_gen/_common.py +++ b/plugins/image_gen/_common.py @@ -242,6 +242,25 @@ class HttpFailure: response: Any = None +def record_token_usage(usage: Any, *, model: str, provider: str, base_url: Optional[str] = None) -> None: + """Record a token-billed image call against the ambient session as task ``image_generation``. + + Token-metered image models (OpenRouter chat-image and Image API models, OpenAI ``gpt-image``) + bill exactly like a chat completion, so they get a ``session_model_usage`` row through the + same chokepoint as auxiliary calls. ``usage`` is the response's usage block (dict or SDK + object). Per-image backends (FAL, xAI, Krea, ...) return no token usage and never call this; + a body without tokens is a no-op inside ``record_aux_usage``, as is running outside a turn. + """ + if not usage: + return + from types import SimpleNamespace + + from agent.aux_accounting import record_aux_usage + + record_aux_usage( + SimpleNamespace(model=model, usage=usage), "image_generation", provider=provider, base_url=base_url) + + def post_json( url: str, *, headers: Dict[str, str], payload: Dict[str, Any], timeout: Any, label: str, error_message: Callable[[Any, Exception], str] = requests_error_message, diff --git a/plugins/image_gen/openai/__init__.py b/plugins/image_gen/openai/__init__.py index 2c0742ee8a..be013eebbf 100644 --- a/plugins/image_gen/openai/__init__.py +++ b/plugins/image_gen/openai/__init__.py @@ -14,7 +14,7 @@ from agent.image_gen_provider import DEFAULT_ASPECT_RATIO, resolve_aspect_ratio, from plugins.image_gen._common import ( GPT_IMAGE_2_API_MODEL as API_MODEL, GPT_IMAGE_2_DEFAULT as DEFAULT_MODEL, GPT_IMAGE_2_TIERS, StaticImageGenProvider, collect_source_images, error_factory, import_openai, materialize_image, - openai_importable, prompt_required_error, resolve_static_model, size_for) + openai_importable, prompt_required_error, record_token_usage, resolve_static_model, size_for) logger = logging.getLogger(__name__) @@ -143,6 +143,9 @@ class OpenAIImageGenProvider(StaticImageGenProvider): logger.debug("OpenAI image %s failed", verb, exc_info=True) return fail(f"OpenAI image {'editing' if is_edit else 'generation'} failed: {exc}", "api_error") + # gpt-image bills per text/image token; the tier id is a Hermes label, the API model prices. + # Recorded before extraction/save: the tokens are billed whether or not an image came back. + record_token_usage(getattr(response, "usage", None), model=meta["api_model"], provider="openai") data = getattr(response, "data", None) or [] if not data: return fail("OpenAI returned no image data", "empty_response") diff --git a/plugins/image_gen/openrouter/__init__.py b/plugins/image_gen/openrouter/__init__.py index 0d1e6a237c..7b26a7c999 100644 --- a/plugins/image_gen/openrouter/__init__.py +++ b/plugins/image_gen/openrouter/__init__.py @@ -22,7 +22,7 @@ from typing import Any, Dict, List, Optional, Tuple from agent.image_gen_provider import ( DEFAULT_ASPECT_RATIO, ImageGenProvider, error_response, resolve_aspect_ratio, save_b64_image, save_url_image, success_response) -from plugins.image_gen._common import error_factory, load_image_gen_config, post_json +from plugins.image_gen._common import error_factory, load_image_gen_config, post_json, record_token_usage logger = logging.getLogger(__name__) @@ -680,6 +680,8 @@ class OpenRouterCompatImageProvider(ImageGenProvider): "model_access", retryable=True) return _fail(failure.error, "api_error", retryable=status in _IMAGE_API_FALLBACK_STATUSES) + # The provider billed these tokens on HTTP 200 whether or not an image came back / saved. + record_token_usage(_dict_at(body, "usage"), model=model_id, provider=self._name, base_url=base_url) entries = [e for e in _list_at(body, "data") if isinstance(e, dict)] if not entries: return _fail( @@ -724,6 +726,8 @@ class OpenRouterCompatImageProvider(ImageGenProvider): return fail(hint, "model_access"), "unavailable" return fail(failure.error, "api_error"), None + # The provider billed these tokens on HTTP 200 whether or not an image came back / saved. + record_token_usage(_dict_at(result, "usage"), model=model_id, provider=self._name, base_url=base_url) images = _extract_images(result) if not images: # Text but no image usually means the model didn't honor image output. diff --git a/plugins/platforms/raft/adapter.py b/plugins/platforms/raft/adapter.py index d3d45a5c18..2f4d26ee83 100644 --- a/plugins/platforms/raft/adapter.py +++ b/plugins/platforms/raft/adapter.py @@ -39,7 +39,6 @@ sys.path.insert(0, str(_Path(__file__).resolve().parents[3])) from gateway.config import Platform, PlatformConfig from gateway.platforms.base import BasePlatformAdapter, SendResult, merge_pending_message_event from gateway.platforms.event import MessageEvent, MessageType -from gateway.session import build_session_key from gateway.platforms._shared import coerce_port, profile_scoped as _profile_scoped logger = logging.getLogger(__name__) @@ -491,10 +490,7 @@ class RaftAdapter(BasePlatformAdapter): return if not self._message_handler: return - session_key = build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), - profile=self._session_key_profile(event.source)) + session_key = self._event_session_key(event) if session_key in self._active_sessions: logger.debug("[raft] Wake queued for busy session %s", session_key) merge_pending_message_event(self._pending_messages, session_key, event) diff --git a/plugins/platforms/slack/adapter.py b/plugins/platforms/slack/adapter.py index fc9d3d7d52..2532c02f44 100644 --- a/plugins/platforms/slack/adapter.py +++ b/plugins/platforms/slack/adapter.py @@ -5966,20 +5966,14 @@ class SlackAdapter(BasePlatformAdapter): def _build_thread_session_key( self, channel_id: str, thread_ts: str, user_id: str, team_id: str = "", *, chat_type: str = "group") -> Optional[str]: - """Thread session key via ``build_session_key()`` (honours per-user isolation). - ``chat_type`` must come from the event's ``channel_type``, not the ID prefix (MPIM ids - start with ``G``).""" - session_store = getattr(self, "_session_store", None) - if not session_store: + """Thread session key through the adapter seam (``_source_session_key``: per-user isolation + from the adapter config the runner seeded, owner-profile namespace). ``chat_type`` must come + from the event's ``channel_type``, not the ID prefix (MPIM ids start with ``G``).""" + if not getattr(self, "_session_store", None): return None try: - from gateway.session import build_session_key source = self._thread_session_source(channel_id, thread_ts, user_id, team_id, chat_type) - store_cfg = getattr(session_store, "config", None) - return build_session_key( - source, group_sessions_per_user=getattr(store_cfg, "group_sessions_per_user", True), - thread_sessions_per_user=getattr(store_cfg, "thread_sessions_per_user", False), - profile=self._session_key_profile(source)) + return self._source_session_key(source) except Exception: return None diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 30203b407d..3d64430f7e 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -6277,11 +6277,7 @@ class TelegramAdapter(BasePlatformAdapter): def _photo_batch_key(self, event: MessageEvent, msg: Message) -> str: """Return a batching key for Telegram photos/albums.""" - from gateway.session import build_session_key - session_key = build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), - profile=self._session_key_profile(event.source)) + session_key = self._event_session_key(event) media_group_id = getattr(msg, "media_group_id", None) return f"{session_key}:album:{media_group_id}" if media_group_id else f"{session_key}:photo-burst" diff --git a/scripts/check_profile_scope_patterns.py b/scripts/check_profile_scope_patterns.py index dc7e7f3821..79dbccfcad 100644 --- a/scripts/check_profile_scope_patterns.py +++ b/scripts/check_profile_scope_patterns.py @@ -49,7 +49,9 @@ def load_patterns(path: Path = PATTERNS) -> list[dict]: data = json.loads(path.read_text(encoding="utf-8")) out = [] for p in data["patterns"]: - out.append({**p, "_rx": re.compile(p["pattern_regex"], re.M)}) + # ``path_regex`` (optional) restricts a pattern to files whose repo-relative path matches. + path_rx = re.compile(p["path_regex"]) if p.get("path_regex") else None + out.append({**p, "_rx": re.compile(p["pattern_regex"], re.M), "_path_rx": path_rx}) return out @@ -63,6 +65,9 @@ def scan_text(rel: str, text: str, patterns: list[dict], lines: set[int] | None findings: list[Finding] = [] src_lines = text.split("\n") for p in patterns: + path_rx = p.get("_path_rx") + if path_rx is not None and not path_rx.search(rel): + continue for m in p["_rx"].finditer(text): line_no = text.count("\n", 0, m.start()) + 1 if lines is not None and line_no not in lines: diff --git a/scripts/ci/profile_scope_patterns.json b/scripts/ci/profile_scope_patterns.json index 50baf25932..94e786cc51 100644 --- a/scripts/ci/profile_scope_patterns.json +++ b/scripts/ci/profile_scope_patterns.json @@ -32,7 +32,7 @@ "P24: 62 hits on main", "P26: 252 hits on main" ], - "usage": "scripts/check_profile_scope_patterns.py --base origin/main [--head HEAD] | --files " + "usage": "scripts/check_profile_scope_patterns.py --base origin/main [--head HEAD] | --files ; optional path_regex restricts a pattern to matching repo-relative paths" }, "patterns": [ { @@ -158,8 +158,16 @@ "id": "P31", "class": "C6", "pattern_regex": "f\"agent:\\{|[\"']agent:[\"']\\s*\\+|session_key\\s*=\\s*f\"[a-z]+:", - "scope_hint": "Every adapter-built session key carries the agent:: namespace (profile 'main' is 'agent:main~'); yuanbao still builds keys with no profile component per the MindDragon probe.", + "scope_hint": "Every adapter-built session key carries the agent:: namespace (profile 'main' is 'agent:main~'); adapters derive keys through _source_session_key / _event_session_key (P32), never a hand-built prefix.", "why": "Rows for a served profile land in the root store; browser/computer_use caches never saw the namespace because turns pass the bare session id." + }, + { + "id": "P32", + "class": "C4", + "path_regex": "^(gateway|plugins)/platforms/(?!base\\.py$).+\\.py$", + "pattern_regex": "\\bbuild_session_key\\(|\\bSessionSource\\(", + "scope_hint": "Inside an adapter derive every key through self._source_session_key(source) / self._event_session_key(event) (owner-profile namespace, runner-seeded isolation flags) and build sources with self.build_source(...) so the transport provenance is kept; only platforms/base.py owns the free calls.", + "why": "Yuanbao keyed its per-group queue and RecallGuard with the free build_session_key() (no profile) while handle_message keyed under agent:: - two derivations of one identity, one lane shared across bots (#88715)." } ] } diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py index b78e7466f9..6a3e3f351f 100644 --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -525,6 +525,12 @@ def _run_one_file_once( # (venv without pytest, -k that matches nothing) can't report green. rc = 0 summary = _parse_pytest_summary(output) + crash = _describe_interpreter_crash(rc, output) if rc != 0 else None + if crash: + # Same convention as the timeout path: the diagnosis leads the + # captured output, so the failure dump reads correctly on its own. + summary["crashed"] = 1 + output = f"(interpreter crashed: {crash})\n{output}" subproc_wall = time.monotonic() - subproc_start return file, rc, output, summary, subproc_wall @@ -562,6 +568,35 @@ def _parse_pytest_summary(output: str) -> dict[str, int]: return result +def _describe_interpreter_crash(rc: int, output: str) -> Optional[str]: + """Return a one-line description when the pytest subprocess died instead of exiting. + + A native fault (sqlite stepping a connection another thread closed, + #113186) kills the interpreter mid-file: faulthandler prints ``Fatal + Python error: Segmentation fault`` and the process dies by signal, so + there is no summary line and every count parses to 0. Without this the + file is reported as "no tests ran (collection/import error)" under a + summary that says ``0 failed`` — the wrong diagnosis in both places. + """ + fatal = next( + (line.strip() for line in output.splitlines() if "Fatal Python error:" in line), + None, + ) + if fatal: + # The faulthandler banner is appended to the progress dots of the + # last test; keep only the banner. + fatal = fatal[fatal.index("Fatal Python error:"):] + if rc < 0: + import signal as _signal + + try: + name = _signal.Signals(-rc).name + except ValueError: + name = f"signal {-rc}" + return f"{fatal} ({name})" if fatal else f"killed by {name}" + return fatal + + def _format_file(file: Path, repo_root: Path) -> str: """Render a test-file path for display: strip the repo-root prefix when possible so output reads ``tests/acp_adapter/test_auth.py`` instead of @@ -624,6 +659,8 @@ def _print_progress( parts.append(f"{xf}xf") if xp: parts.append(f"{xp}xp") + if file_summary.get("crashed"): + parts.append("CRASHED") test_str = " ".join(parts) + ", " if parts else "" else: n_tests = test_counts.get(file, 0) @@ -1168,11 +1205,12 @@ def main() -> int: # nothing-ran guard, whereas a file that died before collection reports # nothing at all and must. tests_collected = 0 + files_crashed = 0 lock = threading.Lock() def _on_done(file: Path, started_at: float, fut: "Future[Tuple[Path, int, str, Dict[str, int], float]]") -> None: nonlocal files_done, tests_done, pass_count, fail_count, tests_passed, tests_failed, tests_skipped - nonlocal tests_collected + nonlocal tests_collected, files_crashed n_tests = test_counts.get(file, 0) try: fpath, rc, output, summary, subproc_wall = fut.result() @@ -1197,6 +1235,7 @@ def main() -> int: tests_passed += summary.get("passed", 0) tests_failed += summary.get("failed", 0) tests_skipped += summary.get("skipped", 0) + files_crashed += summary.get("crashed", 0) tests_collected += sum( summary.get(k, 0) for k in ("passed", "failed", "skipped", "errors", "xfailed", "xpassed") @@ -1245,7 +1284,13 @@ def main() -> int: print() pct = min(100, (tests_done / approx_total_tests * 100)) if approx_total_tests else 0 skipped_note = f", {tests_skipped} skipped" if tests_skipped else "" - print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") + # A crashed interpreter has no failed-test count; say so on the one line + # everyone reads, or "0 failed" + exit 1 looks like a runner bug. + crashed_note = ( + f", {files_crashed} file{'s' if files_crashed != 1 else ''} CRASHED" + if files_crashed else "" + ) + print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{crashed_note}{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") # Host-OS gating note: tests marked for another OS were skipped by the # conftest hook, not run. Say so explicitly — a green local run on Linux @@ -1268,7 +1313,7 @@ def main() -> int: # The summary line above reads green at a glance ("0 failed ... 100% # complete"), which has been misread as a successful verification, so say # it plainly AND fail the exit code. - no_tests_ran_at_all = bool(files) and tests_collected == 0 + no_tests_ran_at_all = bool(files) and tests_collected == 0 and not files_crashed if no_tests_ran_at_all: print() print( @@ -1336,11 +1381,17 @@ def main() -> int: print(output.rstrip()) print() # Split: files with actual test failures vs non-zero exit for other reasons - test_fail_files = [(f, s) for f, _o, s in failures if s.get("failed", 0) > 0] - all_passed_but_nonzero = [(f, s) for f, _o, s in failures + crashed_files = [(f, o, s) for f, o, s in failures if s.get("crashed")] + rest = [(f, s) for f, _o, s in failures if not s.get("crashed")] + test_fail_files = [(f, s) for f, s in rest if s.get("failed", 0) > 0] + all_passed_but_nonzero = [(f, s) for f, s in rest if s.get("failed", 0) == 0 and s.get("passed", 0) > 0] - no_tests_ran = [(f, s) for f, _o, s in failures + no_tests_ran = [(f, s) for f, s in rest if s.get("failed", 0) == 0 and s.get("passed", 0) == 0] + if crashed_files: + print(f"=== {len(crashed_files)} file{'s' if len(crashed_files) != 1 else ''} where the interpreter CRASHED mid-run (native fault — a real bug, not a collection error; the tests that did run are not counted) ===") + for file, output, _s in crashed_files: + print(f" {_format_file(file, repo_root)} {output.splitlines()[0]}") if test_fail_files: total_tf = sum(s.get("failed", 0) for _, s in test_fail_files) print(f"=== {len(test_fail_files)} file{'s' if len(test_fail_files) != 1 else ''} with test failures ({total_tf} test{'s' if total_tf != 1 else ''} failed) ===") diff --git a/tests/agent/test_anthropic_external_login_optout.py b/tests/agent/test_anthropic_external_login_optout.py new file mode 100644 index 0000000000..72a3fafa1b --- /dev/null +++ b/tests/agent/test_anthropic_external_login_optout.py @@ -0,0 +1,82 @@ +"""``auth.adopt_external_logins: false`` keeps Hermes off the Claude Code login (#113023). + +Claude Code's OAuth refresh token is single-use: once Hermes borrows and refreshes it, Hermes and +Claude Code hold one token family and whichever refreshes first logs the other out. With the opt-out +set, Hermes must neither read nor refresh ``~/.claude/.credentials.json``, must drop the pool row an +earlier adopting process persisted, and must say so in ``hermes auth list``. +""" +from __future__ import annotations + +import io +import json +import time +from contextlib import redirect_stdout +from types import SimpleNamespace +from unittest import mock + +import urllib.request + +from agent import anthropic_credentials as ac +from agent import credential_sources +from agent.credential_pool import load_pool +from hermes_cli.auth_commands import auth_list_command + + +def _write_config(hermes_home, adopt): + body = "model:\n provider: anthropic\n model: claude-sonnet-4-5\n" + if adopt is not None: + body += f"auth:\n adopt_external_logins: {'true' if adopt else 'false'}\n# arm {adopt}\n" + (hermes_home / "config.yaml").write_text(body) + + +def test_opt_out_never_reads_or_refreshes_claude_code_login(tmp_path, monkeypatch): + hermes_home, claude_dir = tmp_path / "hermes", tmp_path / "claude" + hermes_home.mkdir() + claude_dir.mkdir() + (hermes_home / ".env").write_text("") + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(claude_dir)) + monkeypatch.setattr(credential_sources, "_notice_logged", False, raising=False) + cred_file = claude_dir / ".credentials.json" + cred_file.write_text(json.dumps({"claudeAiOauth": { + "accessToken": "sk-ant-oat01-expired", "refreshToken": "sk-ant-ort01-cc", + "expiresAt": int(time.time() * 1000) - 3_600_000, "scopes": ["user:inference"]}})) + original_bytes = cred_file.read_bytes() + + posts: list = [] + + def fake_urlopen(req, timeout=None): + if "oauth/token" in req.full_url: # the Anthropic token endpoint; other seeders probe other hosts + posts.append(req.full_url) + body = json.dumps({"access_token": "sk-ant-oat01-fresh", "refresh_token": "sk-ant-ort01-fresh", + "expires_in": 3600}).encode() + resp = mock.MagicMock() + resp.read.return_value = body + resp.__enter__.return_value = resp + return resp + + monkeypatch.setattr(urllib.request, "urlopen", fake_urlopen) + + # Arm A (default): today's behaviour — the borrowed login is seeded into the pool. + _write_config(hermes_home, adopt=None) + assert [e.source for e in load_pool("anthropic").entries()] == ["claude_code"] + + # Arm B (opt-out): no read, no refresh POST, the persisted row is dropped, the status line explains why. + _write_config(hermes_home, adopt=False) + assert ac.resolve_anthropic_token() is None + assert posts == [] + assert cred_file.read_bytes() == original_bytes + assert [e.source for e in load_pool("anthropic").entries()] == [] + out = io.StringIO() + with redirect_stdout(out): + auth_list_command(SimpleNamespace(provider=None)) + assert credential_sources.EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE in out.getvalue() + + # Arm A again: flipping back adopts (and refreshes) exactly as before. + _write_config(hermes_home, adopt=True) + assert ac.resolve_anthropic_token() == "sk-ant-oat01-fresh" + assert len(posts) == 1 + out = io.StringIO() + with redirect_stdout(out): + auth_list_command(SimpleNamespace(provider=None)) + assert credential_sources.EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE not in out.getvalue() diff --git a/tests/agent/test_background_review_input_budget.py b/tests/agent/test_background_review_input_budget.py index b0d483f93d..846ac93111 100644 --- a/tests/agent/test_background_review_input_budget.py +++ b/tests/agent/test_background_review_input_budget.py @@ -224,16 +224,36 @@ def test_review_input_budget_exhausted_predicate_edge_cases(): @pytest.mark.parametrize( ("config_value", "expected"), [ - ({}, 600_000), ({"max_input_tokens": 1_000_000}, 1_000_000), ({"max_input_tokens": 0}, None), ({"max_input_tokens": -5}, None), - ({"max_input_tokens": "not-a-number"}, 600_000), ({"max_input_tokens": "300000"}, 300_000), ], ) def test_review_input_token_budget_resolution(config_value, expected): - """Config parsing: default, override, explicit disable, garbage fallback.""" + """Explicit settings retain their established override and unlimited semantics.""" from agent.background_review import _review_input_token_budget assert _review_input_token_budget(config_value) == expected + + +def test_review_input_token_budget_default_tracks_forks_context_window(): + """Unset or malformed ``max_input_tokens`` → 75% of the fork's RESOLVED window (a 65k local + model gets ~49k, not the cloud-scale 600k), capped at 600k; unknown window → 120k fallback.""" + from agent.background_review import _review_input_token_budget + + def fork(window): + return SimpleNamespace(context_compressor=SimpleNamespace(context_length=window)) + + assert _review_input_token_budget({}, fork(65_536)) == 49_152 + assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, fork(65_536)) == 49_152 + assert _review_input_token_budget({}, fork(2_000_000)) == 600_000 + assert _review_input_token_budget({}, fork(None)) == 120_000 + assert _review_input_token_budget({}, None) == 120_000 + + +def test_background_review_config_does_not_freeze_a_fixed_input_budget(): + """The config default must leave the budget resolver access to the active runtime.""" + from hermes_cli.config_defaults import DEFAULT_CONFIG + + assert "max_input_tokens" not in DEFAULT_CONFIG["auxiliary"]["background_review"] diff --git a/tests/agent/test_compression_small_ctx_threshold_floor.py b/tests/agent/test_compression_small_ctx_threshold_floor.py index 11797c5545..abaf240707 100644 --- a/tests/agent/test_compression_small_ctx_threshold_floor.py +++ b/tests/agent/test_compression_small_ctx_threshold_floor.py @@ -83,7 +83,10 @@ class TestReasoningExcludedFromSummarizer: out = comp._generate_summary([{"role": "user", "content": "hi"}]) assert out is not None assert "OUTPUT_TRACE" not in out - assert "## Active Task" in out + # Snapshot grounding rewrites the legacy "## Active Task" alias into the canonical, grounded + # section instead of prepending a second task section next to it. + assert cc.HISTORICAL_TASK_HEADING in out + assert "## Active Task" not in out # The iterative-update seed must be clean too, or the trace compounds # across every subsequent compaction. assert "OUTPUT_TRACE" not in (comp._previous_summary or "") diff --git a/tests/agent/test_context_compressor_task_heading.py b/tests/agent/test_context_compressor_task_heading.py new file mode 100644 index 0000000000..eacc4c7c82 --- /dev/null +++ b/tests/agent/test_context_compressor_task_heading.py @@ -0,0 +1,51 @@ +"""Task-snapshot heading identity across the summarizer prompt, the template, and grounding. + +The iterative-update instruction, the emitted template, and ``_ground_historical_task_snapshot`` must agree on +one heading. A leftover ``## Active Task`` section is not disclaimed by SUMMARY_PREFIX and reads as live work, +so grounding must replace it (and any duplicate task section) rather than prepend a second one. +""" + +from types import SimpleNamespace + +from agent.context_compressor import ContextCompressor, HISTORICAL_TASK_HEADING + +_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" + + +def _headings(text: str) -> list[str]: + return [line for line in text.splitlines() if line.startswith("## ")] + + +def test_update_instruction_names_the_emitted_heading(): + """The prompt built when a previous summary exists (the update path) names HISTORICAL_TASK_HEADING.""" + stub = SimpleNamespace( + tail_mode="lean", + _previous_summary="PREVIOUS SUMMARY BODY", + _bound_summary_input=lambda text: text, + ) + stub._summary_template_sections = ContextCompressor._summary_template_sections + stub._build_summary_prompt = ContextCompressor._build_summary_prompt.__get__(stub) + prompt = stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) + + assert f'Update "{HISTORICAL_TASK_HEADING}"' in prompt + assert f'"{_LEGACY_ACTIVE_TASK_HEADING}"' not in prompt + + +def test_grounding_collapses_alias_and_duplicate_task_sections(): + """A summarizer that emits the legacy alias, or both headings, ends up with exactly one grounded section.""" + body = ( + f"{HISTORICAL_TASK_HEADING}\nUser asked: 'stale canonical'\n\n" + "## Goal\nthing\n\n" + f"{_LEGACY_ACTIVE_TASK_HEADING}\nUser asked: 'stale alias'\n\n" + "## Constraints & Preferences\n- none\n" + ) + grounded = ContextCompressor._ground_historical_task_snapshot.__func__( + ContextCompressor, body, [{"role": "user", "content": "fresh ask"}] + ) + + headings = _headings(grounded) + assert headings.count(HISTORICAL_TASK_HEADING) == 1 + assert _LEGACY_ACTIVE_TASK_HEADING not in headings + assert headings[1:] == ["## Goal", "## Constraints & Preferences"] + assert "fresh ask" in grounded + assert "stale" not in grounded diff --git a/tests/agent/test_credential_pool_codex_singleton_isolation.py b/tests/agent/test_credential_pool_codex_singleton_isolation.py new file mode 100644 index 0000000000..f57ab0392c --- /dev/null +++ b/tests/agent/test_credential_pool_codex_singleton_isolation.py @@ -0,0 +1,106 @@ +"""Codex pool entries and the auth.json singleton: who may adopt whose tokens. + +``manual:device_code`` is ambiguous — a legacy alias of the singleton or an independent account +added with ``hermes auth add openai-codex``. Adopting the singleton into an independent account +silently turned two logins into one (both then hit the same usage limit; salvaged from #100423 and +#106788, cluster #92198 / #95297, issue #106705). +""" +from __future__ import annotations + +import base64 +import json +import time + +import pytest + +import hermes_cli.auth as auth_mod +from agent.credential_pool import load_pool + + +def _jwt(account: str, sub: str, exp: float) -> str: + def seg(obj: dict) -> str: + return base64.urlsafe_b64encode(json.dumps(obj).encode()).rstrip(b"=").decode() + + payload = {"exp": int(exp), "sub": sub, "https://api.openai.com/auth": {"chatgpt_account_id": account}} + return f"{seg({'alg': 'none'})}.{seg(payload)}.sig" + + +def _iso(ts: float) -> str: + return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(ts)) + + +def _write_store(home, singleton_tokens: dict, singleton_last_refresh: str, manual: dict) -> None: + home.mkdir(parents=True, exist_ok=True) + (home / "auth.json").write_text(json.dumps({ + "version": 1, + "active_provider": "openai-codex", + "providers": {"openai-codex": { + "tokens": singleton_tokens, "last_refresh": singleton_last_refresh, "auth_mode": "chatgpt"}}, + "credential_pool": {"openai-codex": [ + {"id": "seeded", "label": "device_code", "auth_type": "oauth", "priority": 0, + "source": "device_code", **singleton_tokens, "last_refresh": singleton_last_refresh}, + {"id": "manual", "label": "second", "auth_type": "oauth", "priority": 1, + "source": "manual:device_code", **manual}, + ]}, + }), encoding="utf-8") + + +@pytest.fixture +def home(tmp_path, monkeypatch): + home = tmp_path / "hermes" + monkeypatch.setenv("HERMES_HOME", str(home)) + return home + + +def _stub_refresh(monkeypatch, minted: str, minted_rt: str, posted: list) -> None: + def fake(access_token, refresh_token): + posted.append(refresh_token) + return {"access_token": minted, "refresh_token": minted_rt, "last_refresh": _iso(time.time())} + + monkeypatch.setattr(auth_mod, "refresh_codex_oauth_pure", fake) + + +def test_independent_manual_account_refreshes_with_its_own_pair(home, monkeypatch): + """An independent second account never adopts the singleton: its refresh POSTs its OWN refresh + token, and the persisted row still identifies the second principal afterwards.""" + now = time.time() + a_at = _jwt("acct-A", "user-A", now + 8 * 3600) + b_at = _jwt("acct-B", "user-B", now + 60) # expiring → the pool defers it to _refresh_entry() + _write_store(home, {"access_token": a_at, "refresh_token": "rt-A"}, _iso(now - 3600), + {"access_token": b_at, "refresh_token": "rt-B", "last_refresh": _iso(now - 7200)}) + posted: list = [] + b_new = _jwt("acct-B", "user-B", now + 8 * 3600) + _stub_refresh(monkeypatch, b_new, "rt-B2", posted) + + pool = load_pool("openai-codex") + refreshed = pool._refresh_entry(next(e for e in pool.entries() if e.id == "manual"), force=False) + + assert posted == ["rt-B"] + assert refreshed is not None and refreshed.refresh_token == "rt-B2" + on_disk = json.loads((home / "auth.json").read_text(encoding="utf-8")) + manual = next(e for e in on_disk["credential_pool"]["openai-codex"] if e["id"] == "manual") + assert (manual["access_token"], manual["refresh_token"]) == (b_new, "rt-B2") + # The singleton (account A) is untouched by account B's rotation. + assert on_disk["providers"]["openai-codex"]["tokens"] == {"access_token": a_at, "refresh_token": "rt-A"} + + +def test_same_account_alias_adopts_only_a_newer_singleton(home, monkeypatch): + """A legacy alias (same principal) must follow a singleton that was re-authed AFTER it, but must + not fall back onto a singleton older than its own rotation — that replays a consumed token.""" + now = time.time() + alias_at = _jwt("acct-A", "user-A", now + 8 * 3600) + stale_singleton_at = _jwt("acct-A", "user-A", now + 3600) + _write_store(home, {"access_token": stale_singleton_at, "refresh_token": "rt-consumed"}, _iso(now - 7200), + {"access_token": alias_at, "refresh_token": "rt-alias", "last_refresh": _iso(now - 60)}) + pool = load_pool("openai-codex") + alias = next(e for e in pool.entries() if e.id == "manual") + + synced = pool._sync_entry_from_auth_store(alias) + assert (synced.access_token, synced.refresh_token) == (alias_at, "rt-alias") + + # The user re-authenticates the singleton (newer stamp, same principal): the alias follows. + fresh_at = _jwt("acct-A", "user-A", now + 9 * 3600) + _write_store(home, {"access_token": fresh_at, "refresh_token": "rt-fresh"}, _iso(now + 5), + {"access_token": alias_at, "refresh_token": "rt-alias", "last_refresh": _iso(now - 60)}) + synced = load_pool("openai-codex")._sync_entry_from_auth_store(alias) + assert (synced.access_token, synced.refresh_token) == (fresh_at, "rt-fresh") diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index c71cc07f3b..8e8dab5aa1 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -64,7 +64,7 @@ class TestFailoverReason: "ssl_cert_verification", "context_overflow", "payload_too_large", "image_too_large", "image_corrupt", - "model_not_found", "format_error", + "model_not_found", "format_error", "role_alternation", "invalid_encrypted_content", "multimodal_tool_content_unsupported", "reasoning_mandatory", diff --git a/tests/agent/test_failed_turn_site_codes.py b/tests/agent/test_failed_turn_site_codes.py index afeed1cf39..6ce9ce279e 100644 --- a/tests/agent/test_failed_turn_site_codes.py +++ b/tests/agent/test_failed_turn_site_codes.py @@ -40,6 +40,55 @@ def test_overflow_exhaustion_is_non_retryable_context_overflow_with_slash_comman assert build_error_surface_from_result(result)["code"] == "context_overflow" +def _context_rejection(request_tokens: int, window: int = 65_536, error="HTTP 500: Context size has been exceeded."): + """Drive ``_recover_context_length`` with a provider "context exceeded" and a request the rough + estimator prices at ``request_tokens`` against a ``window``-token model (no output cap).""" + from unittest.mock import patch + + from agent.turn_overflow import _recover_context_length + from agent.turn_retry_state import TurnRetryState + + st = _recovery() + st.compression_attempts = 0 + st.agent.max_tokens = None + st.agent.context_compressor = SimpleNamespace(context_length=window) + st.agent.provider, st.agent.base_url, st.agent.tools = "lmstudio", "http://127.0.0.1:1234/v1", None + st.agent._buffer_vprint = st.agent._buffer_diagnostic_status = lambda *a, **k: None + compressed = [] + st.agent._compress_context = lambda msgs, *a, **k: (compressed.append(1) or [{"role": "user", "content": "x"}], None) + with patch("agent.model_metadata.estimate_request_tokens_rough", return_value=request_tokens), \ + patch("agent.model_metadata.estimate_messages_tokens_rough", side_effect=[request_tokens, 10]), \ + patch("agent.turn_overflow.time.sleep", lambda _: None): + verdict = _recover_context_length(st, TurnRetryState(), error) + return verdict, compressed + + +def test_context_rejection_far_below_the_window_is_not_blamed_on_the_conversation(): + """A single-slot local server rejecting a ~3k-token request (another thread held its + context) must not read as "this conversation has grown too long": no compression, transient + + retryable, and the gateway's overflow verdict (drop message / auto-reset) stays off.""" + from gateway.run_turn import is_context_overflow_failure_result + + verdict, compressed = _context_rejection(3_000) + result = verdict.result + assert verdict.action == "return" and not compressed + assert result["failed"] is True and not result.get("compression_exhausted") + assert result["failure_reason"] == "server_error" and result["failure_retryable"] is True + text = result["final_response"] + assert "grown too long" not in text and "/new" not in text + assert "3,000" in text and "65,536" in text and "background" in text and "/retry" in text + assert is_context_overflow_failure_result(result, history_len=2) is False + + +def test_context_rejection_near_the_window_still_compresses(): + """Control: a request that plausibly overflows keeps the compress-and-retry path, and so does + a small local estimate when the SERVER quoted its own count — its measurement wins.""" + verdict, compressed = _context_rejection(60_000) + assert compressed and verdict.action == "break" + verdict, compressed = _context_rejection(3_000, error="prompt is too long: 70000 tokens > 65536 maximum") + assert compressed and verdict.action == "break" + + def test_payload_and_context_overflow_share_one_next_step(): """413 and context-length exhaustion differ in cause text but never in what to do.""" a = _recovery().count_attempt(payload_too_large=True).result["final_response"] diff --git a/tests/agent/test_moa_alternation_recovery.py b/tests/agent/test_moa_alternation_recovery.py new file mode 100644 index 0000000000..e8d0c05a81 --- /dev/null +++ b/tests/agent/test_moa_alternation_recovery.py @@ -0,0 +1,90 @@ +"""MoA aggregator: strict-alternation destinations get their adjacent same-role messages merged +reactively and per destination (#112358, last atom). + +The guidance is deliberately a separate trailing ``user`` message so the prefix stays cache-stable; +a chat template that 400s on ``user(task), user(guidance)`` must get ONE merged retry, be remembered +for the session, and never change the bytes sent to destinations that accepted the split shape. +""" + +from types import SimpleNamespace + +import pytest + +from agent import moa_loop +from agent.error_classifier import FailoverReason, classify_api_error + +ALTERNATION_MSG = "Conversation roles must alternate user/assistant/user/assistant/..." + + +class _AlternationRejected(Exception): + status_code = 400 + + def __init__(self): + super().__init__(f"Error code: 400 - {{'error': {{'message': '{ALTERNATION_MSG}', 'type': 'invalid_request_error'}}}}") + self.response = SimpleNamespace(status_code=400, headers={}) + self.body = {"message": ALTERNATION_MSG, "type": "invalid_request_error"} + + +def _strict_destination(calls, strict_models): + """``call_llm`` double: 400s like a strict chat template when adjacent non-system messages share a role.""" + def call_llm(**kw): + calls.append(kw) + msgs = [m for m in kw["messages"] if m["role"] != "system"] + if kw["model"] in strict_models and any(a["role"] == b["role"] for a, b in zip(msgs, msgs[1:])): + raise _AlternationRejected() + return SimpleNamespace(choices=[]) + return call_llm + + +@pytest.fixture +def facade(monkeypatch): + monkeypatch.setattr( + moa_loop, "_slot_runtime", + lambda slot: {"provider": "custom", "model": slot["model"], "base_url": "http://strict.local/v1", + "api_mode": "chat_completions"}, + ) + f = moa_loop.MoAChatCompletions("default", agent=None) + f._pending_trace = None + return f + + +def _send(facade, model, messages, guidance="[Mixture of Agents reference context]\nadvice"): + prepared = facade.rebase_prepared_request( + {"guidance": guidance, "aggregator": {"provider": "custom", "model": model}, "aggregator_temperature": None}, + messages, + ) + return facade._call_prepared_aggregator(prepared, {"tools": None}) + + +def test_alternation_400_is_classified_and_retried_once_merged_then_remembered(monkeypatch, facade): + classified = classify_api_error(_AlternationRejected(), provider="custom", model="strict-model") + assert classified.reason is FailoverReason.role_alternation + + calls = [] + monkeypatch.setattr(moa_loop, "call_llm", _strict_destination(calls, {"strict-model"})) + task = [{"role": "system", "content": "sys"}, {"role": "user", "content": "task"}] + + _send(facade, "strict-model", task) # iteration 1 of turn 1: split shape rejected → one merged retry + assert [[m["role"] for m in c["messages"]] for c in calls] == [["system", "user", "user"], ["system", "user"]] + assert calls[-1]["messages"][-1]["content"] == "task\n\n[Mixture of Agents reference context]\nadvice" + + del calls[:] + _send(facade, "strict-model", [*task, {"role": "assistant", "content": "answer"}, {"role": "user", "content": "task2"}]) + # Remembered destination: iteration 1 of the next turn is pre-merged, no 400 paid again. + assert [[m["role"] for m in c["messages"]] for c in calls] == [["system", "user", "assistant", "user"]] + assert calls[0]["messages"][-1]["content"].startswith("task2\n\n[Mixture of Agents reference context]") + + +def test_destinations_that_accept_the_split_shape_keep_byte_identical_requests(monkeypatch, facade): + calls = [] + monkeypatch.setattr(moa_loop, "call_llm", _strict_destination(calls, {"strict-model"})) + task = [{"role": "system", "content": "sys"}, {"role": "user", "content": "task"}] + guidance = "[Mixture of Agents reference context]\nadvice" + + _send(facade, "strict-model", task, guidance) # teaches the facade about the strict destination + del calls[:] + _send(facade, "lenient-model", task, guidance) + + # The lenient destination on the SAME facade still gets the split, cache-stable shape in one request. + assert len(calls) == 1 + assert calls[0]["messages"] == [*task, {"role": "user", "content": guidance}] diff --git a/tests/cron/test_ticker_startup_survival.py b/tests/cron/test_ticker_startup_survival.py index 4884741228..ffb05da7d8 100644 --- a/tests/cron/test_ticker_startup_survival.py +++ b/tests/cron/test_ticker_startup_survival.py @@ -105,3 +105,17 @@ def test_housekeeping_restarts_a_dead_ticker(monkeypatch): stop.set() gateway_run._start_gateway_housekeeping(_OneTick(), interval=0, cron_thread=ticker) assert len(starts) == 2, "a stopped ticker must not be respawned" + + +def test_supervisor_leaves_a_returning_external_provider_alone(): + """An external provider's start() (Chronos) arms remote one-shots and RETURNS by design; the + supervisor must not read that as a dead ticker and re-run start() every housekeeping tick + (each rerun re-reconciles against NAS and logs a spurious "died without a stop request").""" + from cron.scheduler_thread import SupervisedTickerThread + + starts = [] + stop = threading.Event() + ticker = SupervisedTickerThread(lambda stop_event: starts.append(1), args=(stop,), stop_event=stop) + ticker.start() + _wait_until(lambda: not ticker.is_alive()) + assert ticker.restart_if_dead() is False and starts == [1] and ticker.restarts == 0 diff --git a/tests/gateway/test_adapter_session_key_seam.py b/tests/gateway/test_adapter_session_key_seam.py new file mode 100644 index 0000000000..1e807b6fa4 --- /dev/null +++ b/tests/gateway/test_adapter_session_key_seam.py @@ -0,0 +1,69 @@ +"""Every adapter-side session key goes through one seam (#88715, invariant 4). + +A key an adapter derives for batching / queueing / busy detection must be the key the runner +derives for the same event, for a primary bot and for a secondary-owned bot alike. Yuanbao's +``DispatchMiddleware`` used the free ``build_session_key()`` (no profile) while ``handle_message`` +keyed under ``agent::``, so a secondary bot's per-group queue and RecallGuard entries lived +in a lane the runner never popped. +""" + +import asyncio +from types import SimpleNamespace + +import pytest + +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.platforms.event import MessageEvent +from gateway.platforms.yuanbao import DispatchMiddleware, InboundContext, YuanbaoAdapter +from gateway.profile_routing import parse_profile_routes + + +def _yuanbao(owner): + adapter = YuanbaoAdapter(PlatformConfig(extra={ + "app_id": "k", "app_secret": "s", "ws_url": "wss://x", "api_domain": "https://x", + "group_sessions_per_user": True, "thread_sessions_per_user": False})) + adapter.set_owner_profile(owner) + return adapter + + +def _runner(adapter, owner): + from gateway.run import GatewayRunner + + runner = object.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=True) + runner.config.profile_routes = parse_profile_routes([]) + runner._primary_profile_name = "default" + runner.adapters = {} if owner else {Platform.YUANBAO: adapter} + runner._profile_adapters = {owner: {Platform.YUANBAO: adapter}} if owner else {} + adapter.gateway_runner = runner + return runner + + +@pytest.mark.parametrize("owner", [None, "acme"]) +def test_adapter_batch_key_equals_runner_session_key(owner): + adapter = _yuanbao(owner) + runner = _runner(adapter, owner) + source = adapter.build_source(chat_id="grp-1", chat_type="group", user_id="u1") + ctx = InboundContext(adapter=adapter, chat_type="group", chat_id="grp-1", raw_text="hi", msg_id="m1", source=source) + seen = [] + + async def go(): + async def _next(): + pass + + async def _capture(event): + seen.append(adapter._event_session_key(event)) + + adapter.handle_message = _capture + await DispatchMiddleware().handle(ctx, _next) + queued = list(adapter._group_queues) + await asyncio.sleep(0.05) # let the group consumer dispatch once + return queued + + queued = asyncio.run(go()) + runner_key = runner._session_key_for_source(source) + expected_ns = f"agent:{owner}" if owner else "agent:main" + assert runner_key.startswith(expected_ns + ":"), runner_key + assert queued == [runner_key], (queued, runner_key) + assert seen == [runner_key] + assert adapter._processing_msg_ids == {runner_key: "m1"} diff --git a/tests/gateway/test_session_identity.py b/tests/gateway/test_session_identity.py new file mode 100644 index 0000000000..201d551b76 --- /dev/null +++ b/tests/gateway/test_session_identity.py @@ -0,0 +1,147 @@ +"""Invariants for ``gateway/session_identity.py`` (#88715 phase 1). + +One frozen ``RoutingIdentity`` per inbound event, resolved by the real ``GatewayRunner`` resolvers +(no patched predicates) against a temp ``HERMES_HOME`` with two served profiles. +""" + +import dataclasses +import weakref +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.pairing import PairingStore +from gateway.platforms.base import BasePlatformAdapter +from gateway.profile_routing import parse_profile_routes +from gateway.session import SessionSource, build_session_key +from gateway.session_identity import ( + IdentityUnresolved, RoutingIdentity, identity_of, replace_source, resolve_identity, +) + + +class _Stub(BasePlatformAdapter): + pass + + +_Stub.__abstractmethods__ = frozenset() + + +def _stub(platform, runner, label): + adapter = _Stub.__new__(_Stub) + adapter.platform, adapter.gateway_runner, adapter.label = platform, runner, label + adapter.config = PlatformConfig(enabled=True, extra={}) + adapter._pending_messages, adapter._active_sessions = {}, {} + return adapter + + +def _runner(home, *, multiplex, routes=()): + from gateway.run import GatewayRunner + + runner = object.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=multiplex) + runner.config.platforms = {Platform.TELEGRAM: PlatformConfig(enabled=True, extra={})} + runner.config.profile_routes = parse_profile_routes(list(routes)) + runner.pairing_store = PairingStore(profile="default") + runner.pairing_stores = {} + runner._primary_profile_name = "default" + primary = _stub(Platform.TELEGRAM, runner, "PRIMARY") + team_b = _stub(Platform.TELEGRAM, runner, "TEAM_B") + team_b.set_owner_profile("team_b") + runner.adapters = {Platform.TELEGRAM: primary} + runner._profile_adapters = {"team_b": {Platform.TELEGRAM: team_b}, "ops": {}} + return SimpleNamespace(runner=runner, home=home, primary=primary, team_b=team_b) + + +@pytest.fixture +def mux(tmp_path, monkeypatch): + """Default home + satellite ``ops`` (routed through the shared bot for chat 72719239) + ``team_b`` + owning its own Telegram bot; a route to unserved ``ghost`` for chat 4040.""" + home = tmp_path / "hh" + for name in ("ops", "team_b"): + (home / "profiles" / name).mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(home)) + served = [("default", home), ("ops", home / "profiles" / "ops"), ("team_b", home / "profiles" / "team_b")] + rig = _runner(home, multiplex=True, routes=[ + {"name": "admin-dm", "platform": "telegram", "profile": "ops", "chat_id": "72719239"}, + {"name": "ghost", "platform": "telegram", "profile": "ghost", "chat_id": "4040"}, + ]) + with patch("hermes_cli.profiles.profiles_to_serve", return_value=served), \ + patch("hermes_cli.profiles.get_profile_dir", side_effect=lambda n: home if n == "default" else home / "profiles" / n), \ + patch("hermes_cli.profiles.profile_exists", return_value=True): + yield rig + + +def test_identity_is_one_frozen_value_that_every_reader_agrees_on(mux): + """Shared bot → routed satellite: transport stays default (authorization), runtime is ``ops``; + the pinned identity is frozen, equal by value, and the legacy readers all report the same + answer it does. A dedicated secondary resolves to itself on both axes.""" + routed = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + identity = resolve_identity(routed, runner=mux.runner) + + assert identity == RoutingIdentity( + transport_profile="default", runtime_profile="ops", authorization_home=mux.home, + runtime_home=mux.home / "profiles" / "ops") + assert identity.namespace == "agent:ops" and identity.store_path == mux.home / "profiles" / "ops" / "state.db" + assert identity.adapter() is mux.primary + with pytest.raises(dataclasses.FrozenInstanceError): + identity.runtime_profile = "default" + assert identity_of(routed) is identity + # Thin readers: same object, same answers — no second derivation. + assert mux.runner._authorization_home_for_source(routed) == mux.home + assert mux.runner._resolve_profile_home_for_source(routed) == mux.home / "profiles" / "ops" + assert mux.runner._transport_owner(routed) == (mux.primary, None) + assert mux.primary._source_session_key(routed) == "agent:ops:telegram:dm:72719239" + assert mux.runner._session_key_for_source(routed) == mux.primary._source_session_key(routed) + + own = mux.team_b.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + own_identity = resolve_identity(own, runner=mux.runner, transport_profile="team_b") + assert (own_identity.transport_profile, own_identity.runtime_profile) == ("team_b", "team_b") + assert own_identity.authorization_home == own_identity.runtime_home == mux.home / "profiles" / "team_b" + assert own_identity != identity + # Provenance is not identity: a second event from the same bot has the same identity. + again = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + assert resolve_identity(again, runner=mux.runner) == identity + assert hash(resolve_identity(again, runner=mux.runner)) == hash(identity) + + +def test_unresolved_under_multiplex_raises_and_never_means_default(mux, tmp_path, monkeypatch): + """A route to an unserved profile raises ``IdentityUnresolved`` (the runner drops the event); + outside multiplexing the same source resolves to an explicit default identity whose keys stay + byte-identical to the legacy ``agent:main`` namespace.""" + rejected = mux.primary.build_source(chat_id="4040", chat_type="dm", user_id="4040") + with pytest.raises(IdentityUnresolved): + resolve_identity(rejected, runner=mux.runner) + assert identity_of(rejected) is None + assert mux.runner._admit_primary_source(rejected, mux.home) is None + + solo_home = tmp_path / "solo" + solo_home.mkdir() + monkeypatch.setenv("HERMES_HOME", str(solo_home)) + solo = _runner(solo_home, multiplex=False) + source = solo.primary.build_source(chat_id="4040", chat_type="dm", user_id="4040") + identity = resolve_identity(source, runner=solo.runner) + assert (identity.transport_profile, identity.runtime_profile, identity.multiplexed) == ("default", "default", False) + assert identity.runtime_home == solo_home + assert identity.namespace == "agent:main" + assert solo.primary._source_session_key(source) == build_session_key(source) == "agent:main:telegram:dm:4040" + assert source.profile is None # wire format untouched + assert solo.runner._authorization_home_for_source(source) is None # ambient scope, as before + + +def test_replace_source_keeps_identity_and_transport_where_dataclasses_replace_drops_them(mux): + routed = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + identity = resolve_identity(routed, runner=mux.runner) + + bare = dataclasses.replace(routed) + assert identity_of(bare) is None and mux.runner._transport_owner(bare) is None + + copied = replace_source(routed, thread_id="7") + assert copied.thread_id == "7" and copied is not routed + assert identity_of(copied) is identity + assert mux.runner._transport_owner(copied) == (mux.primary, None) + assert isinstance(copied._transport_adapter_ref, weakref.ref) + assert Path(copied._authorization_profile_home) == mux.home + assert mux.primary._source_session_key(copied) == "agent:ops:telegram:dm:72719239:7" diff --git a/tests/gateway/test_slack.py b/tests/gateway/test_slack.py index 851acf5005..790014273f 100644 --- a/tests/gateway/test_slack.py +++ b/tests/gateway/test_slack.py @@ -3070,7 +3070,10 @@ class TestThreadReplyHandling: from gateway.session import SessionEntry # Deserialize a legacy routing entry so lifecycle flags have real defaults. + # The thread key with a per-user suffix comes from the adapter's isolation flags (the runner + # seeds them into PlatformConfig.extra); this store has no bearing on the key any more. session_key = "agent:main:slack:group:T_TEAM:C123:123.000:U_USER" + adapter_with_session_store.config.extra["thread_sessions_per_user"] = True mock_session_store._entries = {session_key: SessionEntry.from_dict({ "session_key": session_key, "session_id": "slack-thread-session", diff --git a/tests/gateway/test_stop_ends_background_delegations.py b/tests/gateway/test_stop_ends_background_delegations.py new file mode 100644 index 0000000000..17ba5010af --- /dev/null +++ b/tests/gateway/test_stop_ends_background_delegations.py @@ -0,0 +1,80 @@ +"""Invariant: ``/stop`` ends the session's BACKGROUND delegate_task children too — whether the session is +mid-turn (busy fast path) or idle after the dispatching turn already ended — instead of only the in-turn +children, leaving the detached unit to run to completion and wake the chat later. See #114456. + +Real ``GatewayRunner`` + real ``BasePlatformAdapter`` subclass; the background unit is a real +``tools.async_delegation`` registry record whose ``interrupt_fn`` is the observable. +""" +from unittest.mock import MagicMock + +import pytest + +import tools.async_delegation as ad +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.platforms.base import BasePlatformAdapter, SendResult +from gateway.platforms.event import MessageEvent, MessageType +from gateway.run import GatewayRunner +from gateway.session import SessionSource + + +class _Adapter(BasePlatformAdapter): + def __init__(self): + super().__init__(PlatformConfig(enabled=True, token="x"), Platform.TELEGRAM) + + @property + def name(self): + return "telegram" + + async def connect(self, *, is_reconnect=False): + return True + + async def disconnect(self): + pass + + async def send(self, chat_id, content, reply_to=None, metadata=None): + return SendResult(success=True) + + async def get_chat_info(self, chat_id): + return {"id": chat_id, "type": "private"} + + +@pytest.fixture(autouse=True) +def _reset_async_delegation(): + ad._reset_for_tests() + yield + ad._reset_for_tests() + + +def _seed_unit(session_key: str, parent_session_id: str = "") -> MagicMock: + """A live background unit as ``_dispatch_background`` registers it (running, routed by session_key).""" + fn = MagicMock() + with ad._records_lock: + ad._records["deleg_bg1"] = {"delegation_id": "deleg_bg1", "status": "running", "session_key": session_key, + "origin_ui_session_id": "", "parent_session_id": parent_session_id, "interrupt_fn": fn} + return fn + + +@pytest.mark.asyncio +@pytest.mark.parametrize("session_state", ["idle", "busy"]) +async def test_stop_ends_background_delegations_of_the_session(monkeypatch, session_state): + monkeypatch.setenv("TELEGRAM_ALLOWED_USERS", "u1") + adapter = _Adapter() + runner = GatewayRunner(config=GatewayConfig(platforms={Platform.TELEGRAM: PlatformConfig(enabled=True, token="x")})) + runner.adapters = {Platform.TELEGRAM: adapter} + source = SessionSource(platform=Platform.TELEGRAM, chat_id="c1", chat_type="dm", user_id="u1", user_name="tester") + key = adapter._event_session_key(MessageEvent(text="", message_type=MessageType.TEXT, source=source)) + stop_fn = _seed_unit(key) + other_fn = MagicMock() + with ad._records_lock: + ad._records["deleg_other"] = {"delegation_id": "deleg_other", "status": "running", "session_key": "agent:main:telegram:dm:c2", + "origin_ui_session_id": "", "parent_session_id": "", "interrupt_fn": other_fn} + + if session_state == "busy": + await runner._interrupt_and_clear_session(key, source, interrupt_reason="stop", invalidation_reason="stop_command") + else: + reply = await runner._handle_stop_command(MessageEvent(text="/stop", message_type=MessageType.TEXT, source=source)) + # The chat is told something WAS stopped, not "No active task to stop." + assert "Stopped" in str(getattr(reply, "text", reply)) + + stop_fn.assert_called_once() + other_fn.assert_not_called() # another chat's background work is untouched diff --git a/tests/hermes_cli/test_auth_codex_self_heal.py b/tests/hermes_cli/test_auth_codex_self_heal.py index 4c92cde2f5..fa6182d7cb 100644 --- a/tests/hermes_cli/test_auth_codex_self_heal.py +++ b/tests/hermes_cli/test_auth_codex_self_heal.py @@ -96,3 +96,34 @@ def test_self_heals_missing_singleton_access_token_from_codex_cli(tmp_path, monk assert tokens["refresh_token"] == "fresh-refresh" + + +def test_opt_out_never_adopts_codex_cli_login(tmp_path, monkeypatch): + """``auth.adopt_external_logins: false`` (#113023): the Codex CLI pair is a single-use refresh-token + family the user did not hand to Hermes. Both automatic recovery paths must leave it (and Hermes' own + auth.json) untouched and surface the real error instead.""" + hermes_home = tmp_path / "hermes" + codex_home = tmp_path / "codex" + hermes_home.mkdir() + codex_home.mkdir() + (hermes_home / "config.yaml").write_text("auth:\n adopt_external_logins: false\n") + hermes_auth = {"version": 1, "providers": {"openai-codex": { + "tokens": {"refresh_token": "stale-refresh"}, "auth_mode": "chatgpt"}}} + (hermes_home / "auth.json").write_text(json.dumps(hermes_auth)) + (codex_home / "auth.json").write_text(json.dumps({ + "tokens": {"access_token": "fresh-access", "refresh_token": "fresh-refresh"}})) + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + monkeypatch.setenv("CODEX_HOME", str(codex_home)) + + with pytest.raises(AuthError) as info: + resolve_codex_runtime_credentials() + assert info.value.code == "codex_auth_missing_access_token" + + def _rejected(*_a, **_k): + raise AuthError("bad", provider="openai-codex", code="invalid_grant", relogin_required=True) + + monkeypatch.setattr(auth_codex, "refresh_codex_oauth_pure", _rejected) + with pytest.raises(AuthError) as info: + _refresh_codex_auth_tokens(dict(STALE), 5.0) + assert info.value.relogin_required # surfaced, not papered over with the CLI pair + assert json.loads((hermes_home / "auth.json").read_text()) == hermes_auth diff --git a/tests/hermes_cli/test_gateway_windows.py b/tests/hermes_cli/test_gateway_windows.py index 584150cf1c..64ee166a8f 100644 --- a/tests/hermes_cli/test_gateway_windows.py +++ b/tests/hermes_cli/test_gateway_windows.py @@ -481,6 +481,76 @@ def test_reconcile_scheduled_task_reregisters_only_on_drift(monkeypatch, tmp_pat assert not any(c[0] in ("/Delete", "/Create") for c in calls) +def _arrange_uninstalled_start(monkeypatch): + """start() with no Scheduled Task / Startup entry; returns (install_calls, spawn_count).""" + installs, spawns = [], [] + monkeypatch.delenv("HERMES_GATEWAY_INSTALL_START_ON_LOGIN", raising=False) + monkeypatch.delenv("HERMES_NONINTERACTIVE", raising=False) + monkeypatch.setattr(gateway_windows, "_assert_windows", lambda: None) + monkeypatch.setattr(gateway_windows, "_print_start_attestation_warning", lambda: None) + monkeypatch.setattr(gateway_windows, "_gateway_pids", lambda: []) + monkeypatch.setattr(gateway_windows, "is_task_registered", lambda: False) + monkeypatch.setattr(gateway_windows, "is_startup_entry_installed", lambda: False) + monkeypatch.setattr(gateway_windows, "install", lambda **kwargs: installs.append(kwargs)) + monkeypatch.setattr(gateway_windows, "_spawn_detached", lambda: spawns.append(1) or 4242) + monkeypatch.setattr(gateway_windows, "_report_gateway_start", lambda via: None) + monkeypatch.setattr(gateway_windows, "_stdin_console_mode_ok", lambda: True) + return installs, spawns + + +def test_stdin_interactive_only_when_isatty_and_a_console_answers_get_console_mode(): + """Windows CRT isatty() is True for the NUL device (`hermes gateway start < NUL`, stdin=DEVNULL), so + isatty alone must not open the prompt; off Windows (no console-mode fact) isatty decides (#113977).""" + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=False) is False # NUL + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=True) is True # console + assert gateway_windows._stdin_is_interactive(isatty=False, console_mode_ok=True) is False # pipe + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=None) is True # POSIX tty + + +def test_start_with_nul_stdin_starts_the_gateway_but_never_installs_login_persistence(monkeypatch, capsys): + """isatty says TTY, GetConsoleMode says no console: `< NUL` gets the same treatment as a pipe.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: True) + monkeypatch.setattr(gateway_windows, "_stdin_console_mode_ok", lambda: False) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("no prompt on a NUL stdin")) + + gateway_windows.start() + + assert installs == [] and spawns == [1] + assert "hermes gateway install" in capsys.readouterr().out + + +def test_start_without_tty_starts_the_gateway_but_never_installs_login_persistence(monkeypatch, capsys): + """`hermes gateway start < /dev/null` must not answer the persistence question with a default Yes + (#113977); it starts the gateway once and points at the explicit install command.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: False) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("no prompt without a TTY")) + + gateway_windows.start() + + assert installs == [] and spawns == [1] + out = capsys.readouterr().out + assert "hermes gateway install" in out and "did not complete" not in out + + +def test_start_on_tty_hands_both_answers_to_install_and_honours_the_env_opt_out(monkeypatch): + """Yes → one install() carrying start_now+start_on_login (install spawns; start() must not spawn + again). HERMES_GATEWAY_INSTALL_START_ON_LOGIN=0 → no question, no install, a plain start.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: True) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: True) + + gateway_windows.start() + assert installs == [{"force": False, "start_now": True, "start_on_login": True}] and spawns == [] + + installs.clear() + monkeypatch.setenv("HERMES_GATEWAY_INSTALL_START_ON_LOGIN", "0") + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("env override must skip the prompt")) + gateway_windows.start() + assert installs == [] and spawns == [1] + + diff --git a/tests/hermes_cli/test_kanban_db.py b/tests/hermes_cli/test_kanban_db.py index f7e3304d65..9313dcfecf 100644 --- a/tests/hermes_cli/test_kanban_db.py +++ b/tests/hermes_cli/test_kanban_db.py @@ -388,6 +388,50 @@ def test_rate_limit_exit_requeues_without_counting_failure( assert "crashed" not in outcomes +@pytest.mark.parametrize("lane", ["ready", "review"]) +def test_terminal_provider_exit_blocks_after_one_attempt_in_either_lane(kanban_home, monkeypatch, lane): + """A worker that exits ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` (credential revoked, model + gone) parks the card ``blocked`` on the FIRST death — well below ``failure_limit`` and the + per-task ``max_retries`` — with the provider error as the reason, sticky against + ``recompute_ready``. Same booking for the implementation and the review lane (#114587).""" + import hermes_cli.kanban_db as _kb + from hermes_cli import kanban_db_dispatch as _kbd + + monkeypatch.setattr(_kb, "_pid_alive", lambda _pid: False) + monkeypatch.setenv("HERMES_KANBAN_CRASH_GRACE_SECONDS", "0") + + with kbc.connect() as conn: + host = _kb._claimer_id().split(":", 1)[0] + tid = kb.create_task(conn, title="terminal", assignee="a", max_retries=5) + claimed = kb.claim_task(conn, tid, claimer=f"{host}:w0") + if lane == "review": + assert kb.request_review(conn, tid, summary="done", reviewer="r", + expected_run_id=claimed.current_run_id) + assert kb.claim_review_task(conn, tid, claimer=f"{host}:r0") is not None + pid = 71000 + conn.execute("UPDATE tasks SET worker_pid=? WHERE id=?", (pid, tid)) + conn.commit() + _kbd._record_worker_exit(pid, _exited_status(_kb.KANBAN_TERMINAL_PROVIDER_EXIT_CODE)) + + crashed = kbd.detect_crashed_workers(conn) + assert tid in crashed + assert tid in getattr(_kbd.detect_crashed_workers, "_last_auto_blocked", []) + + task = kb.get_task(conn, tid) + assert task.status == "blocked" + assert task.consecutive_failures == 1 # one spawn, not failure_limit / max_retries of them + assert "terminal provider error" in (task.last_failure_error or "") + gave_up = conn.execute( + "SELECT payload FROM task_events WHERE task_id=? AND kind='gave_up'", (tid,), + ).fetchone() + assert json.loads(gave_up["payload"])["terminal_provider"] is True + + # Sticky: the breaker did not reach its counter limit, yet the card must stay parked + # until an operator fixes the provider and unblocks it. + kb.recompute_ready(conn) + assert kb.get_task(conn, tid).status == "blocked" + + def test_respawn_guard_defers_rate_limited_within_cooldown( diff --git a/tests/hermes_cli/test_single_query_exit_contract.py b/tests/hermes_cli/test_single_query_exit_contract.py index cef28eea59..79ac9748d5 100644 --- a/tests/hermes_cli/test_single_query_exit_contract.py +++ b/tests/hermes_cli/test_single_query_exit_contract.py @@ -13,7 +13,7 @@ from types import SimpleNamespace import pytest import cli -from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE +from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE, KANBAN_TERMINAL_PROVIDER_EXIT_CODE @pytest.fixture(autouse=True) @@ -53,6 +53,16 @@ def test_dispatcher_spawned_worker_signals_a_provider_outage_not_a_protocol_viol assert code == KANBAN_RATE_LIMIT_EXIT_CODE +@pytest.mark.parametrize("reason", ["auth", "auth_permanent", "model_not_found", "ssl_cert_verification"]) +def test_dispatcher_spawned_worker_signals_a_terminal_provider_error(monkeypatch, reason): + """A revoked credential / missing model cannot be retried into working: the worker says so + with EX_CONFIG so the dispatcher parks the card after one spawn. A person's run keeps 1.""" + monkeypatch.setenv("HERMES_KANBAN_TASK", "t_abc123") + assert _run_non_quiet(monkeypatch, {"failed": True, "failure_reason": reason}) == KANBAN_TERMINAL_PROVIDER_EXIT_CODE + monkeypatch.delenv("HERMES_KANBAN_TASK") + assert _run_non_quiet(monkeypatch, {"failed": True, "failure_reason": reason}) == 1 + + @pytest.mark.parametrize( ("turn_result", "expected"), [ diff --git a/tests/plugins/image_gen/test_openai_provider.py b/tests/plugins/image_gen/test_openai_provider.py index e2c8aa77c1..20f5152a5e 100644 --- a/tests/plugins/image_gen/test_openai_provider.py +++ b/tests/plugins/image_gen/test_openai_provider.py @@ -172,6 +172,40 @@ class TestGenerate: # gpt-image-2 rejects response_format — we must NOT send it. assert "response_format" not in call_kwargs + @pytest.mark.parametrize("has_image", [True, False]) + def test_token_usage_reaches_session_accounting(self, provider, has_image): + """gpt-image bills per token: the Images API ``usage`` block lands as one + ``image_generation`` row keyed on the API model, not the Hermes tier label — also + when the billed HTTP 200 carries no image data.""" + from agent import aux_accounting + + recorded = [] + + class _DB: + def record_auxiliary_usage(self, *args, **kwargs): + recorded.append((args, kwargs)) + + response = _fake_response(b64=_b64_png()) + if not has_image: + response.data = [] + response.usage = SimpleNamespace(input_tokens=23, output_tokens=1056, total_tokens=1079) + fake_client = MagicMock() + fake_client.images.generate.return_value = response + token = aux_accounting.set_accounting_context(_DB(), "sess-1") + try: + with _patched_openai(fake_client): + result = provider.generate("a cat", aspect_ratio="landscape") + finally: + aux_accounting.reset_accounting_context(token) + + assert result["success"] is has_image + if not has_image: + assert result["error_type"] == "empty_response" + ((session_id, task), kwargs), = recorded + assert (session_id, task) == ("sess-1", "image_generation") + assert (kwargs["model"], kwargs["billing_provider"]) == ("gpt-image-2", "openai") + assert (kwargs["input_tokens"], kwargs["output_tokens"]) == (23, 1056) + @pytest.mark.parametrize("api_model,quality", [ ("gpt-image-2", quality) for quality in ("low", "medium", "high") ] + [ diff --git a/tests/plugins/image_gen/test_openrouter_compat_provider.py b/tests/plugins/image_gen/test_openrouter_compat_provider.py index e5c51af09e..414980b1c8 100644 --- a/tests/plugins/image_gen/test_openrouter_compat_provider.py +++ b/tests/plugins/image_gen/test_openrouter_compat_provider.py @@ -703,6 +703,54 @@ class TestImageApiSurface: assert result["exact_aspect_ratio"] == "9:16" assert result["image"] == str(Path("/tmp/i.png")) + _USAGE = {"prompt_tokens": 1000, "completion_tokens": 128, "total_tokens": 1128} + + @pytest.mark.parametrize("surface, model, usage, images", [ + ("chat", "openai/gpt-5.4-image-2", _USAGE, True), # default chain: token-billed via /chat/completions + ("images", "krea/krea-2-medium", _USAGE, True), # curated Image API model + ("images", "krea/krea-2-medium", None, True), # flat-fee body without usage: no write + ("chat", "openai/gpt-5.4-image-2", _USAGE, False), # billed HTTP 200 with text but no image + ("images", "krea/krea-2-medium", _USAGE, False), # billed HTTP 200 with empty ``data`` + ]) + def test_token_usage_reaches_session_accounting(self, surface, model, usage, images): + """A response carrying token usage records one ``image_generation`` row on the ambient + session — also when it carries no image (the provider billed the tokens anyway); a body + without usage records nothing.""" + from agent import aux_accounting + + recorded = [] + + class _DB: + def record_auxiliary_usage(self, *args, **kwargs): + recorded.append((args, kwargs)) + + if surface == "chat": + response = _mock_chat_response([_PNG_DATA_URI] if images else []) + response.json.return_value["usage"] = dict(usage) + else: + response = _mock_image_api_response([] if not images else None, usage=usage) + token = aux_accounting.set_accounting_context(_DB(), "sess-1") + try: + with patch(_RUNTIME, return_value=_runtime_ok()), \ + patch("requests.post", return_value=response), \ + patch("plugins.image_gen.openrouter.save_b64_image", return_value=Path("/tmp/i.png")): + result = _openrouter_image_api().generate(prompt="p", aspect_ratio="portrait", model=model) + finally: + aux_accounting.reset_accounting_context(token) + + assert result["success"] is images + if not images: + assert result["error_type"] == "empty_response" + if usage is None: + assert recorded == [] + return + ((session_id, task), kwargs), = recorded + assert (session_id, task) == ("sess-1", "image_generation") + assert kwargs["model"] == model + assert kwargs["billing_provider"] == "openrouter" + assert (kwargs["input_tokens"], kwargs["output_tokens"]) == (1000, 128) + + def test_multiple_images_land_in_additional_images(self): entries = [ {"b64_json": "AA==", "media_type": "image/png"}, diff --git a/tests/plugins/test_chronos_cron.py b/tests/plugins/test_chronos_cron.py index 97486632bf..cb63781b97 100644 --- a/tests/plugins/test_chronos_cron.py +++ b/tests/plugins/test_chronos_cron.py @@ -100,6 +100,72 @@ def test_register_job_propagates_provision_failure(chronos): # -- reconcile ---------------------------------------------------------------- +def test_identity_rejection_hands_fires_to_the_builtin_ticker(temp_home, chronos, monkeypatch, caplog): + """Regression for #97494: NAS answering 403 invalid_client is a deterministic identity rejection + (the auth.json token is not this instance's provisioned agent), not a transient. The provider + must say so ONCE with the remedy, stop calling NAS, and start the built-in ticker so jobs still + fire on time instead of only via the late misfire sweep.""" + import threading + + from plugins.cron_providers.chronos._nas_client import NasCronClientError + + prov, fake = chronos + calls = [] + + def rejected(**kw): + calls.append(kw["job_id"]) + raise NasCronClientError( + "POST /api/agent-cron/provision returned 403: invalid_client", + status=403, error_code="invalid_client") + + fake.provision = rejected + jobs = [ + {"id": "a", "enabled": True, "next_run_at": "2026-06-18T12:00:00+00:00", "state": "scheduled"}, + {"id": "b", "enabled": True, "next_run_at": "2026-06-18T12:05:00+00:00", "state": "scheduled"}, + ] + monkeypatch.setattr("cron.jobs.load_jobs", lambda: jobs) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: next(j for j in jobs if j["id"] == jid)) + monkeypatch.setattr("cron.executions.recover_interrupted_executions", lambda: 0) + ticker_started = threading.Event() + monkeypatch.setattr( + "cron.scheduler_provider.InProcessCronScheduler.start", + lambda self, stop_event, **kw: ticker_started.set()) + + stop = threading.Event() + with caplog.at_level("WARNING", logger="cron.chronos"): + prov.start(stop, adapters={"x": 1}, loop=None, interval=7) + assert ticker_started.wait(2.0), "the built-in ticker must take over this process's fires" + assert calls == ["a"], "after the first rejection NAS is left alone" + identity_msgs = [r.message for r in caplog.records if "re-login" in r.message] + assert len(identity_msgs) == 1 and "built-in cron ticker" in identity_msgs[0] + + # Job creation and re-arms no longer reach NAS (and no longer fail the create). + prov.register_job({"id": "c", "next_run_at": "2026-06-18T12:10:00+00:00"}) + prov.on_jobs_changed() + assert calls == ["a"] + + +def test_transient_provision_failure_does_not_degrade(temp_home, chronos, monkeypatch): + """A 5xx / transport error is retried on the next reconcile; only 403 invalid_client degrades.""" + from plugins.cron_providers.chronos._nas_client import NasCronClientError + + prov, fake = chronos + attempts = [] + + def flaky(**kw): + attempts.append(kw["job_id"]) + raise NasCronClientError("POST /api/agent-cron/provision returned 502: upstream", status=502) + + fake.provision = flaky + jobs = [{"id": "a", "enabled": True, "next_run_at": "2026-06-18T12:00:00+00:00", "state": "scheduled"}] + monkeypatch.setattr("cron.jobs.load_jobs", lambda: jobs) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: jobs[0]) + + prov.reconcile() + prov.reconcile() + assert attempts == ["a", "a"] and prov._identity_rejected is False + + def test_reconcile_arms_all_enabled(temp_home, chronos, monkeypatch): prov, fake = chronos jobs = [ diff --git a/tests/scripts/test_check_profile_scope_patterns.py b/tests/scripts/test_check_profile_scope_patterns.py index 7e5bb66322..1e13b8bb03 100644 --- a/tests/scripts/test_check_profile_scope_patterns.py +++ b/tests/scripts/test_check_profile_scope_patterns.py @@ -59,6 +59,30 @@ def test_child_env_from_environ_is_flagged_and_the_scoped_builder_is_not(): assert [f.line for f in mod.scan_text("tools/x.py", hazard, patterns, lines={5})] == [5] +def test_adapter_key_outside_the_seam_is_flagged_only_under_platforms(): + """An adapter that derives a session key with the free ``build_session_key()`` bypasses the + owner-profile seam (``_source_session_key``); the same call in ``platforms/base.py`` (the seam + itself) or in the runner is legitimate and must stay silent.""" + mod = _load() + patterns = mod.load_patterns() + free_key = textwrap.dedent(''' + from gateway.session import build_session_key + + def _batch_key(self, event): + return build_session_key(event.source, profile=event.source.profile) + ''') + seam = textwrap.dedent(''' + def _batch_key(self, event): + return self._event_session_key(event) + ''') + flagged = mod.scan_text("plugins/platforms/acme/adapter.py", free_key, patterns) + assert [(f.line, f.pattern_id, f.pattern_class) for f in flagged] == [(5, "P32", "C4")] + assert [f.pattern_id for f in mod.scan_text("gateway/platforms/acme.py", free_key, patterns)] == ["P32"] + assert mod.scan_text("plugins/platforms/acme/adapter.py", seam, patterns) == [] + for owner in ("gateway/platforms/base.py", "gateway/run_startup.py", "gateway/session_recovery.py"): + assert not [f for f in mod.scan_text(owner, free_key, patterns) if f.pattern_id == "P32"], owner + + def test_lint_is_advisory_and_exits_zero_with_findings(tmp_path, capsys): mod = _load() bad = tmp_path / "bad.py" diff --git a/tests/scripts/test_run_tests_parallel.py b/tests/scripts/test_run_tests_parallel.py index edde659c5c..b97e2e7fd2 100644 --- a/tests/scripts/test_run_tests_parallel.py +++ b/tests/scripts/test_run_tests_parallel.py @@ -450,3 +450,38 @@ def test_drive_letter_colon_is_not_a_path_separator(tmp_path: Path) -> None: f"drive letter split off as a phantom root:\n{proc.stdout}" ) assert "Discovered 1 test files" in proc.stdout, proc.stdout + + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX signal death; Windows has no SIGSEGV exit") +def test_interpreter_crash_is_reported_as_a_crash_not_as_no_tests_ran(tmp_path: Path) -> None: + """A file whose interpreter dies by signal is classified as CRASHED (#113186). + + A native fault after some tests passed leaves no pytest summary line, so + every count parses to 0. The runner used to file that under "no tests ran + (collection/import error)" beneath a summary reading ``0 failed`` — two + wrong diagnoses for one real bug. The crash must be named on the summary + line and in the failure buckets, and the run must still exit non-zero. + """ + probe_dir = tmp_path / "probe" + probe_dir.mkdir() + (probe_dir / "test_probe_crash.py").write_text( + textwrap.dedent( + """ + import os, signal + + def test_before(): + assert True + + def test_crash(): + os.kill(os.getpid(), signal.SIGSEGV) + """ + ) + ) + + proc = _run_runner(probe_dir, "--file-retries", "0") + + assert proc.returncode != 0 + assert "1 file CRASHED" in proc.stdout + assert "SIGSEGV" in proc.stdout + assert "where no tests ran" not in proc.stdout + assert "NO TESTS RAN" not in proc.stdout diff --git a/tests/tools/test_delegate_interrupted_partial_output.py b/tests/tools/test_delegate_interrupted_partial_output.py new file mode 100644 index 0000000000..6eb573949e --- /dev/null +++ b/tests/tools/test_delegate_interrupted_partial_output.py @@ -0,0 +1,28 @@ +"""Invariant: a child stopped mid-work reports what it HAD so far. The loop's ``final_response`` for an +interrupted turn is the placeholder ``"Operation interrupted."`` (also appended as the closing assistant row); +the parent-visible entry must carry the child's last real assistant text as ``summary`` and keep the +placeholder as ``error`` — never a partial result that reads as an empty stop. See #114456. +""" +from types import SimpleNamespace + +from tools.delegate_tool_child_run import _SchemaOutcome, _build_result_entry + + +def test_interrupted_child_entry_carries_its_partial_output(): + messages = [ + {"role": "user", "content": "kickoff"}, + {"role": "assistant", "content": [{"type": "text", "text": "Audited 3 of 7 modules; two findings so far."}], + "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "terminal", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": "[Command interrupted]"}, + {"role": "assistant", "content": "Operation interrupted."}, + ] + result = {"final_response": "Operation interrupted.", "messages": messages, "api_calls": 2, + "completed": False, "interrupted": True} + child = SimpleNamespace(model="m", session_estimated_cost_usd=0.0, session_cost_status="unknown", + session_prompt_tokens=1, session_completion_tokens=1, _delegate_role="leaf") + + entry = _build_result_entry(child, result, 0, 12.5, _SchemaOutcome(None, None, [], 0)) + + assert entry["status"] == entry["exit_reason"] == "interrupted" + assert entry["summary"] == "Audited 3 of 7 modules; two findings so far." + assert entry["error"] == "Operation interrupted." diff --git a/tools/AGENTS.md b/tools/AGENTS.md index 75a973a354..588342f6fd 100644 --- a/tools/AGENTS.md +++ b/tools/AGENTS.md @@ -140,7 +140,12 @@ processes are killed at its teardown and their notices are suppressed in the par (`process_registry.transfer_ownership`) so the completion routes and reaps by the new owner; un-handed leftovers land on the result as `orphaned_processes`, exited-but-never-read notify processes as `unread_completions` (`_ChildRun.account_background_processes`, before `cleanup` kills them). **Child kernels:** a child's `execute_code` kernels are keyed `::child::`, pinned against the `max_session_kernels` LRU cap while the child runs and disposed by `cleanup` (`code_kernel.shutdown_kernels_for_delegated_child`) — never let a finished child's kernel squat the cap. **output_schema:** a miss after the one retry keeps `status: completed` with the raw text in `summary` plus `schema_valid: false` / `schema_errors` / `schema_note` — never discard a child's result. **Durability:** background delegation is process-local; work that must survive restart uses `cronjob` or -`terminal(background=True, notify_on_complete=True)`. API: `website/docs/developer-guide/subagent-lifecycle-api.md`. +`terminal(background=True, notify_on_complete=True)`. **Stop fan-out:** a turn's hard interrupt reaches only +`_active_children`; background units are detached at dispatch, so every stop surface (gateway `/stop` busy AND +idle paths, TUI `session.interrupt`, ACP `cancel`, CLI `/stop` via `interrupt_all`) calls +`async_delegation.interrupt_for_session` too. Depth>0 delegations are always synchronous +(`_model_background_value`), so the stop recurses through the child's own `_active_children` fan-out; an +interrupted entry's `summary` is the child's last real assistant text (`_build_result_entry`). API: `website/docs/developer-guide/subagent-lifecycle-api.md`. ## Tests diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 5634fc5116..0a04dc68b1 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -545,16 +545,14 @@ _DESCRIPTION_HEAD = ( "as a new message when subagents finish ({delivery}). Background results are delivered only " "BETWEEN your turns: finish whatever does not depend on them, then give a one-line status and END YOUR TURN. Never " "wait or poll on transcripts, artifact files, or CI for a child. " - "While children run, `action` (list/steer/stop) controls them live — steer when a transcript shows a " - "child drifting.\n\n" - "USE FOR: reasoning-heavy subtasks, work that would flood your context with intermediate data, or independent " - "parallel workstreams.\n" + "While children run, `action` (list/steer/stop) controls them live.\n\n" + "USE FOR: reasoning-heavy subtasks, work that would flood your context, or independent parallel workstreams.\n" "DO NOT USE FOR (use these instead):\n" "- Mechanical multi-step work with no reasoning needed -> execute_code\n" "- A single tool call -> call the tool directly\n" "- Tasks needing user interaction -> subagents cannot ask questions\n" "- Durable work that must survive this session -> cronjob or terminal(background=True, notify=True); /stop, /new, " - "or process exit discards running subagents.\n\n" + "or process exit halts running subagents (whole tree); each returns an 'interrupted' completion with partial output.\n\n" "RULES:\n" "- Children know nothing of this conversation: pass everything needed via 'context', including any required " "output language, tone, or style (e.g. \"respond in Chinese\").\n" diff --git a/tools/delegate_tool_child_run.py b/tools/delegate_tool_child_run.py index fa0f56cddc..a2b7f72a61 100644 --- a/tools/delegate_tool_child_run.py +++ b/tools/delegate_tool_child_run.py @@ -503,8 +503,18 @@ def _build_result_entry( # "(empty)" is run_agent's give-up sentinel after repeated empty LLM # responses (usually a transport bug) — a failure, not a success. usable_summary = bool(summary) and summary.strip() != "(empty)" + interrupt_note = "" if result.get("interrupted", False): status, exit_reason = "interrupted", "interrupted" + # The loop's final_response is a placeholder here ("Operation interrupted…", also appended as the closing + # assistant row); the completion must carry what the child actually had so far — its last real assistant + # text — and keep the placeholder as the error. + from agent.message_content import flatten_message_text + placeholders = {"", summary.strip(), "Operation interrupted."} + partial = next((t for m in reversed(result.get("messages") or []) if m.get("role") == "assistant" + and (t := flatten_message_text(m.get("content")).strip()) not in placeholders), "") + if partial: + interrupt_note, summary = summary.strip(), partial elif result.get("failed") or result.get("error"): # The loop returns the error text as final_response, which would otherwise read as "completed". Never report a # provider rejection as "max_iterations" — that is only truthful for real budget exhaustion. @@ -554,6 +564,8 @@ def _build_result_entry( _failure_reason = result.get("failure_reason") if isinstance(_failure_reason, str) and _failure_reason: entry["failure_reason"] = _failure_reason + elif interrupt_note: + entry["error"] = interrupt_note # Schema-validation outcome — emitted ONLY when a schema was requested, so # legacy (schema-less) payloads keep their exact shape. diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index fed9a956f3..a200af8991 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -621,6 +621,14 @@ def _interrupt_session_turn(sid: str, session: dict, *, request_id: str | None = if should_interrupt: from agent.interrupt_compat import request_hard_interrupt request_hard_interrupt(session.get("agent")) + # Background delegations are detached from the turn's interrupt fan-out; a stop ends them too + # (own UI sid + spawner id only — a viewer tab must not kill gateway work). Each returns as an + # interrupted completion with its partial output. + with contextlib.suppress(Exception): + from tools.async_delegation import interrupt_for_session + interrupt_for_session( + origin_ui_session_id=_lifecycle_own_sid(session, sid), reason="user_stop", + parent_session_id=str(getattr(session.get("agent"), "session_id", "") or "")) if not run_thread_alive: with session["history_lock"]: if session.get("running"): diff --git a/website/docs/developer-guide/chronos-managed-cron-contract.md b/website/docs/developer-guide/chronos-managed-cron-contract.md index fe331fea11..c7d4c34f40 100644 --- a/website/docs/developer-guide/chronos-managed-cron-contract.md +++ b/website/docs/developer-guide/chronos-managed-cron-contract.md @@ -219,6 +219,20 @@ If `callback_url` / `portal_url` is blank or the agent has no Nous login, `is_available()` returns False and the resolver falls back to the built-in in-process ticker — cron never loses its trigger. +**Identity rejection at runtime (`403 invalid_client`).** `is_available()` is +config-only, so it cannot tell whether the stored Nous token is the identity NAS +maps to a provisioned instance (hop 1 above). When `provision` answers 403 +`invalid_client` — the token in `auth.json` is a plain `hermes-cli` user login +rather than the `hermes-cli-vps` bootstrap session or an `agent:*` client — the +rejection is deterministic for the life of that credential: every arm, re-arm +and `list` would fail the same way, and a `hermes auth` re-login makes it +permanent (it *replaces* the bootstrap session; only NAS can re-mint one). The +provider therefore logs ONE warning naming that remedy, stops calling NAS, and +starts the built-in ticker for the rest of the process so jobs keep firing on +time instead of only through the late misfire sweep +(`cron.misfire_grace_minutes`). Transient failures (5xx, transport) do not +degrade; the next reconcile retries them. + ## Escape hatch (not default) The inbound `/api/cron/fire` verifier is pluggable (`get_fire_verifier()`). If diff --git a/website/docs/developer-guide/gateway-session-lifecycle.md b/website/docs/developer-guide/gateway-session-lifecycle.md index 21c78aa6a0..2df83f49cc 100644 --- a/website/docs/developer-guide/gateway-session-lifecycle.md +++ b/website/docs/developer-guide/gateway-session-lifecycle.md @@ -463,6 +463,17 @@ post-command drain starts it right away instead of the session idling until the message. Whether a wake pinned to a session that `/new` just closed may still run is decided at processing time (`_resolve_async_delegation_session`, fail-closed). +Both commands also end the session's **background delegations** (`tools.async_delegation. +interrupt_for_session`, selected by routing key and by the spawner's durable session id): +`_interrupt_and_clear_session` fans the stop out for the busy path, and `_handle_stop_command` +does the same for an idle session whose dispatching turn already ended (replying "Stopped" rather +than "No active task to stop"). The turn's own hard interrupt never reaches those units — they are +detached from `_active_children` at dispatch — so without the fan-out they run to completion and wake +the chat minutes later. Each stopped unit still finalizes normally and re-enters as its completion +notice with `status="interrupted"` and the child's partial output. `/new` and `/reset` already did this +in `_handle_reset_command`; the shared helper's earlier call is idempotent there (a hard interrupt +requested twice is one stop). + ### FIFO Invariant Each `/queue` invocation produces exactly one full agent turn, in FIFO order, with no diff --git a/website/docs/developer-guide/multiplexing-gateway.md b/website/docs/developer-guide/multiplexing-gateway.md index 613f4891f9..2d6c91e74e 100644 --- a/website/docs/developer-guide/multiplexing-gateway.md +++ b/website/docs/developer-guide/multiplexing-gateway.md @@ -168,13 +168,23 @@ Pairing stores are constructed per served profile. ## Per-bot session lanes Session keys are namespaced by profile (`agent:main` for default, -`agent:` for named profiles). Adapters carry `_owner_profile` -(installed at adapter configuration time, before any inbound event) because -adapter ingress runs before `SessionSource.profile` is stamped; -`_session_key_profile` resolves source stamp → owner profile → store -resolver. Text/media batching, active-session tracking, and the busy-session -guard are all keyed per lane, so two bots sharing a chat do not share a -session lane. +`agent:` for named profiles). Every inbound event carries ONE frozen +`RoutingIdentity` (`gateway/session_identity.py`), resolved by +`resolve_identity()` at the runner's ingress handlers and pinned on the source +as a wire-invisible attribute: `transport_profile` (the bot that received it — +credential, allowlist, `authorization_home`), `runtime_profile` (the routed +profile that executes — `runtime_home`, key `namespace`, `store_path`) and a +weak `transport` ref to the receiving adapter. `"default"` is spelled out; +`None` never means default. Under multiplexing a route to an unserved profile +raises `IdentityUnresolved` and the event is dropped. + +Adapters also carry `_owner_profile` (installed at adapter configuration time, +before any inbound event) because adapter ingress runs before the runner pins +the identity; `_session_key_profile` resolves identity → source stamp → owner +profile → store resolver. Text/media batching, active-session tracking, and the +busy-session guard are all keyed per lane, so two bots sharing a chat do not +share a session lane. Copy a source with `session_identity.replace_source`, not +`dataclasses.replace`, or the copy loses its transport and identity. ## Control plane diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index da7ef399bb..785dbb8256 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -91,7 +91,7 @@ Don't have a subscription yet? Get one at [portal.nousresearch.com/manage-subscr :::info Codex Note -The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. +The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. Automatic adoption of the Codex CLI login (when Hermes' own refresh fails) is controlled by `auth.adopt_external_logins` — see [Borrowed CLI logins](../user-guide/security.md#borrowed-cli-logins). If a token refresh fails with a terminal error (HTTP 4xx, `invalid_grant`, revoked grant, etc.), Hermes marks the refresh token as dead and stops replaying it so you don't see a flood of identical auth failures. The next request surfaces a typed re-auth message instead. Run `hermes auth add openai-codex` (or `hermes model` → **ChatGPT or Codex Subscription**) to start a fresh device-code login; the quarantine clears on the next successful exchange. @@ -163,7 +163,10 @@ Use Claude models directly through the Anthropic API — no OpenRouter proxy nee When no explicit environment credential is selected, Hermes-owned OAuth grants in the credential pool take precedence over a borrowed Claude Code login. The -borrowed login remains the fallback when no owned OAuth grant is available. +borrowed login remains the fallback when no owned OAuth grant is available — +unless `auth.adopt_external_logins: false` is set, in which case Hermes never +reads or refreshes Claude Code's credentials (see +[Borrowed CLI logins](../user-guide/security.md#borrowed-cli-logins)). Auxiliary authentication recovery refreshes the credential used by the failed request, not an unrelated ambient login; rotating a borrowed login can otherwise invalidate its owner's refresh token. diff --git a/website/docs/reference/faq.md b/website/docs/reference/faq.md index d635cefa8e..611ce9ffe4 100644 --- a/website/docs/reference/faq.md +++ b/website/docs/reference/faq.md @@ -323,6 +323,8 @@ Look at the CLI startup line — it shows the detected context length (e.g., ` **Local servers (llama.cpp, Ollama) that go silent instead of erroring:** when a provider rejects a request as too large, Hermes compacts the conversation and rebuilds the request. Hermes re-measures the *complete* rebuilt request (system prompt + tool schemas + messages) before retrying, and runs further bounded compaction passes if it is still over the threshold. If the request still cannot fit, the turn ends with `Context length exceeded: compression could not reduce the rebuilt request below the safe threshold` rather than sending an oversized request that llama.cpp would silently truncate (`stop processing: n_tokens = 65535, truncated = 1` in the server log). If you hit that message, the fix is almost always the configured `context_length` above: make it match the server's actual `-c` / `--ctx-size`. +**"The model server rejected this request as too large, but this conversation is only about N tokens…":** the server said "context exceeded" without quoting any measurement, while Hermes's own estimate of the request is far below the window it knows for the model — so it does **not** compress or blame the conversation, and the turn stays retryable. On single-slot local servers (LM Studio, Ollama) this is almost always another request holding the server's context at that moment — typically a background memory review from an earlier session (`thread=bg-review` in `logs/agent.log`). Wait a moment and `/retry`. If it recurs with no other Hermes process running, the server is loading the model with a smaller window than Hermes assumes: raise the server's context setting or lower `model.context_length` to match it. + To fix context detection, set it explicitly: ```yaml diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index 2fcc9068d9..3ff7329647 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -47,6 +47,7 @@ The center of the app. You get: - **The same conversation history** as every other Hermes surface — sessions started here resume in the CLI/TUI and vice versa. - **Drag-and-drop files** anywhere in the chat area to attach them to your next message. - **Independent background drafts** — hidden chat tabs can update their drafts without moving the caret or selection in the visible composer. +- **Unsent drafts survive a lost session** — a draft you typed into a chat whose session no longer exists (deleted elsewhere, or a stale tab after a profile rename or wiped backend) is carried into the fresh chat the app falls back to, with an inline **Restored your unsent message** strip above the input. **Undo** puts the text back where it was; nothing is sent, navigated, or focused on your behalf, and the strip appears once per lost draft. - **Directive chip actions** — hover an actionable reference (such as a URL) to reveal its action pill. A short grace period lets you move from the chip to the pill before it dismisses. The pill stays available while you move within it; after leaving, unrelated pointer movement does not delay dismissal. Clicking its action preserves the draft selection. - **A right-hand preview rail** — render web pages, files, and tool outputs side by side while you keep chatting. - **Hide vs. Close for stateful panes** — a zone holding a Browser (or HTML preview) or the Terminal pane offers **Hide** instead of Minimize. Hiding collapses the zone to its restore rail but keeps the pane's body mounted: a live page keeps its unsaved form input, timers, scroll position, and the agent's `drive_preview` automation target; a terminal keeps its running shell and scrollback; and **Restore** shows the same content without a reload. The hidden body is inert — it never takes keyboard shortcuts or steals composer focus. **Close** (the tab's ×) is what actually releases the page or shell. @@ -196,7 +197,7 @@ Manage providers, models, tools, and credentials from a real UI instead of editi - **xAI Grok OAuth** — Grok is a first-class OAuth provider in the launcher; sign in through the browser flow like the other OAuth providers. - **Tool-backend installs from the GUI** — run a tool backend's post-setup install steps directly from the app instead of dropping to a terminal. In the terminal backend picker, selecting a backend marked **Needs setup** asks for confirmation first; declining leaves the current backend selected. - **Terminal font picker** — choose an installed font in **Settings → Appearance**. Nerd Fonts such as `MesloLGS NF` render Powerlevel10k separators and icons in both interactive and agent terminals; the setting is saved per profile. -- **Reasoning Blocks** — **Settings → Chat → Reasoning Blocks** (`display.show_reasoning` in `config.yaml`) shows or hides the model's thinking in the transcript (off shows answers only). Open chats update as soon as the setting saves. +- **Reasoning Blocks** — **Settings → Chat → Reasoning Blocks** (`display.show_reasoning` in `config.yaml`) shows or hides the model's thinking in the transcript (off shows answers only). Open chats update as soon as the setting saves. Typing `/reasoning hide` or `/reasoning show` in the composer flips the same setting and the open transcript follows immediately. - **Reopen Last Chat on Launch** — by default the app picks up where you left off on cold start. Turn it off in **Settings → Appearance** (or set `display.resume_last_session: false` in `config.yaml`) to always begin with a fresh chat. Deep links and explicit destinations are never overridden either way. - **Auxiliary-model warning** — if you switch the main model to a new provider while auxiliary tasks (titling, summarization, and similar helpers) are still pinned to another provider, the app warns you so you don't unknowingly split work across two providers. - **Per-task reasoning effort** — each row under **Settings → Model → Auxiliary models** has a reasoning selector next to its provider/model pick: a level, **Off**, or **inherit · main model effort** (the default, which removes the task's override). It is saved as `auxiliary..reasoning_effort` in `config.yaml`, the same key `hermes model` writes, and shows in the row's summary when set. Use it to run frequent helpers such as compression or titling at low or no reasoning while the main agent stays at high. diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index 93a793abdc..fc7330a322 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -212,7 +212,7 @@ The dispatch handle lists each unit (`units[].delegation_id`, `group`, `task_ind - **Thread pool:** Uses `ThreadPoolExecutor` with the configured concurrency limit as max workers - **Progress display:** In CLI mode, a tree-view shows tool calls from each subagent in real-time with per-task completion lines. In gateway mode, progress is batched and relayed to the parent's progress callback. CLI and TUI completion notices use task-first titles such as `Subagent Task Completed: Review changes`; multi-task groups use the group name and task count. Unsuccessful or incomplete work gets a corresponding status label. These compact notices do not replace the full results delivered to the parent agent. - **Result ordering:** Within a unit, results are sorted by task index to match input order regardless of completion order; `TASK i/N` labels index the whole call -- **Cancellation:** Follow-up messages do not cancel a top-level background batch. `/stop` or closing/resetting the owning session cancels its active children. Synchronous orchestrator children still follow their parent's interrupt state +- **Cancellation:** Follow-up messages do not cancel a top-level background batch. `/stop` (gateway `/stop`, CLI `/stop`, the Desktop/TUI Stop button, an ACP cancel) or closing/resetting the owning session ends its background children and every synchronous descendant beneath them; each stopped child still returns as a completion with `status="interrupted"` and its partial output Synchronous single-task delegation from an orchestrator runs directly without thread pool overhead. @@ -548,11 +548,11 @@ delegate_task( :::warning Background completion durability is not durable execution Top-level model-facing `delegate_task` calls run in the background automatically where the session supports later delivery. Hermes returns a handle immediately, and the result re-enters the conversation after the child or batch finishes. Orchestrator subagents wait for their workers in the current turn because they must synthesize those results before returning. Stateless request/response endpoints fall back to synchronous execution when they cannot deliver a detached result later. -- Normal follow-up messages do not cancel background children. `/stop` cancels running background delegations, and closing or resetting the owning session discards its active children. +- Normal follow-up messages do not cancel background children. `/stop` on any surface (gateway `/stop` — also when the session is idle after the dispatching turn ended — CLI `/stop`, the Desktop/TUI Stop button, an ACP cancel) ends the session's running background delegations, and closing or resetting the owning session does the same. - Explicit session close/reset interrupts that session's background children. Closing a TUI viewer of a gateway-owned session does not kill the gateway's work. - A Hermes process restart does **not** resume a running child. Its attempt becomes `unknown` because Hermes cannot prove which side effects happened. - A child that completed before restart but whose result was not delivered is restored and routed back through the owning session's normal checks. -- Cancelled children return a structured result (`status="interrupted"`, `exit_reason="interrupted"`), but because the parent was interrupted too, that result often never makes it into a user-visible reply. +- Stopped children return a structured result (`status="interrupted"`, `exit_reason="interrupted"`) whose `summary` is the last text the child produced before the stop (the interrupt placeholder moves to `error`). The stop recurses down the spawn tree — an orchestrator child's synchronous workers are interrupted first and their partial results roll up into the child's own interrupted result — and each background unit's result re-enters the conversation right away as its normal completion notice (`Subagent Task Interrupted: …`), so nothing waits for the child to exhaust its budget. For **durable execution** that must survive session closure or process restart, use: @@ -566,7 +566,7 @@ For **durable execution** that must survive session closure or process restart, - Subagents inherit the parent's enabled toolsets; the model cannot select or widen them per call - **Nested delegation is opt-in** — only `role="orchestrator"` children can delegate further, and only when `max_spawn_depth` is raised from its default of 1 (flat). Disable globally with `orchestrator_enabled: false`. - Leaf subagents **cannot** call: `delegate_task`, `clarify`, `memory`, `send_message`, `cronjob`. Orchestrator subagents retain `delegate_task` but keep the other blocks. Both roles retain `execute_code` (programmatic tool calling) so children can batch mechanical work instead of burning reasoning iterations. -- **Cancellation follows ownership** — `/stop` or closing/resetting the owning session cancels its background children; synchronous descendants under orchestrators follow their parent's interrupt state +- **Cancellation follows ownership** — `/stop` or closing/resetting the owning session ends its background children and the synchronous descendants beneath them; each returns an interrupted completion carrying its partial output - Only the final summary enters the parent's context, keeping token usage efficient - Subagents inherit the parent's **API key, provider configuration, and credential pool** (enabling key rotation on rate limits) diff --git a/website/docs/user-guide/features/image-generation.md b/website/docs/user-guide/features/image-generation.md index f184308054..f3390d6bb7 100644 --- a/website/docs/user-guide/features/image-generation.md +++ b/website/docs/user-guide/features/image-generation.md @@ -319,6 +319,7 @@ If upscaling fails (network issue, rate limit), the original image is returned a 3. **Submission** — `_submit_fal_request()` routes via direct FAL credentials or the managed Nous gateway, according to the stored `image_gen.provider` selection. 4. **Upscaling** — runs only when the agent passed `upscale: true`; every model's catalog default is off. 5. **Delivery** — final image URL returned to the agent, which emits a `MEDIA:` tag that platform adapters convert to native media. +6. **Usage accounting** — token-billed image models (OpenRouter chat-image and Image API models such as `google/gemini-3.1-flash-lite-image`, OpenAI `gpt-image`) return real token counts, so each call is recorded in `session_model_usage` as task `image_generation` under the billing provider and model, and shows up in `hermes insights` and the dashboard's Usage analytics alongside other model calls. Per-image backends (FAL, xAI, Krea, ...) return no token usage and are not recorded there. ## Debugging diff --git a/website/docs/user-guide/features/kanban.md b/website/docs/user-guide/features/kanban.md index 99fd8c8c03..5fc2d4c3af 100644 --- a/website/docs/user-guide/features/kanban.md +++ b/website/docs/user-guide/features/kanban.md @@ -561,11 +561,24 @@ is part of the worker protocol. If the worker process exits with status 0 while the task is still `running`, the dispatcher treats that as a protocol violation and emits a `protocol_violation` event. A dispatcher-spawned worker whose turn failed -therefore exits non-zero: `1` for an ordinary failure, and `75` +therefore exits non-zero: `1` for an ordinary failure, `75` (`EX_TEMPFAIL`) when the provider was rate-limited, overloaded, returning 5xx or timing out, or the account hit a billing/quota wall — the dispatcher records that run as `rate_limited` and requeues the task without counting a failure, so a quota window is never -booked as a protocol violation. The worker also writes its exit code as the +booked as a protocol violation — and `78` (`EX_CONFIG`) when the provider +rejected something a retry cannot fix: the profile's credential (401/403, +revoked or invalid key), the model (404 / model not found) or the TLS chain. +That **terminal provider error** trips the circuit breaker on the first +occurrence: the dispatcher records the run as `crashed` with +`exit_kind: terminal_provider`, emits `gave_up` with `terminal_provider: true` +and parks the card `blocked` (sticky — `recompute_ready` will not auto-resume +it) with the provider's own words in `last_failure_error`, instead of +re-spawning into the same wall until `kanban.failure_limit` / `max_retries` +is spent. Both lanes get the same booking: a reviewer worker that dies on a +revoked key parks the card exactly like an implementer. `hermes kanban show` +surfaces it as *Provider rejected this profile's credential or model — blocked +after one attempt*; fix the assignee profile's provider (`hermes -p +auth` / `setup`), then `hermes kanban unblock `. The worker also writes its exit code as the last line of its own log (`[kanban-worker-exit] rc=`), so a per-tick `hermes kanban dispatch` process — which never reaped the worker and cannot read its exit status — books the same death the same way the gateway-embedded @@ -1407,7 +1420,7 @@ Every transition appends a row to `task_events`. Each row carries an optional `r | `respawn_guarded` | `{reason}` | Dispatcher refused to re-spawn this ready task this tick. Reasons: `infrastructure_cooldown` (the host refused the last spawn — no restart-safe systemd scope — and the cooldown has not elapsed; never counted against the card), `rate_limit_cooldown` (the last run hit a quota wall; same cooldown, never counted), `blocker_auth` (last failure was a quota/auth/429 error — wait for the rate window to reset), `recent_success` (a completed run happened in the last hour — wait for review before re-running), `active_pr` (a GitHub PR URL appears in a recent comment — a prior worker already opened a PR). The task stays in `ready`; the next tick gets another chance to spawn. If the underlying condition persists, the normal `consecutive_failures` circuit breaker will auto-block via `gave_up` after `failure_limit` failures. | | `spawn_failed` | `{error, failures}` | One spawn attempt failed (missing PATH, workspace unmountable, …). Counter increments; task returns to `ready` for retry. | | `protocol_violation` | `{pid, claimer, exit_code, protocol_violation, worker_output?}` | Worker exited successfully while the task was still `running`, usually because it answered without a terminal board call (`kanban_complete`, `kanban_request_review` or `kanban_block`). Emitted on every violation (the payload's `protocol_violation: true` marker is copied into the run metadata and feeds the violation-only retry budget). Below the budget — up to `_PROTOCOL_VIOLATION_FAILURE_LIMIT` (default 3) *consecutive* violations, per-task `max_retries` overriding — the task simply returns to `ready` for another attempt; when the streak reaches the bound the dispatcher also emits `gave_up` and auto-blocks. `worker_output` carries the worker's own last printed text (usually its explanation of why it stopped), also folded into `last_failure_error` and shown to the retry worker as the prior-attempt error. | -| `gave_up` | `{failures, effective_limit, limit_source, error}` | Circuit breaker fired after N consecutive non-successful attempts. Task auto-blocks with the last error. The effective limit resolves as task `max_retries`, then dispatcher `failure_limit` / `kanban.failure_limit`, then the built-in default. | +| `gave_up` | `{failures, effective_limit, limit_source, error, terminal_provider?}` | Circuit breaker fired after N consecutive non-successful attempts. Task auto-blocks with the last error. The effective limit resolves as task `max_retries`, then dispatcher `failure_limit` / `kanban.failure_limit`, then the built-in default. `terminal_provider: true` means the worker exited `78` on a provider error a retry cannot fix (credential revoked, model gone) and the breaker fired on that first attempt, sticky, regardless of the limit. | `hermes kanban tail ` shows these for a single task. `hermes kanban watch` streams them board-wide. diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index f15f752960..0e441c2a37 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -369,6 +369,26 @@ auxiliary: With `enabled: false`, automatic post-turn forks do not spawn; manual `/refine` still works. +### Capping review cost (`max_input_tokens`) + +The review loop replays the conversation on every provider request it makes, +so a single review can multiply input tokens across its tool iterations. +`max_input_tokens` caps the SUM of replayed input tokens for one review; the +loop stops before crossing it. `<= 0` means unlimited. + +```yaml +auxiliary: + background_review: + max_input_tokens: 48000 # <= 0 = unlimited +``` + +When the key is unset, the budget is derived from the review model's resolved +context window: 75% of the window, capped at 600,000 tokens — so it also binds +on small local models (a 65,536-token model gets 49,152), where a fixed +cloud-scale default would never bite. If the window cannot be resolved, a +conservative 120,000-token fallback applies. Note the key lives under +`auxiliary:`; a top-level `background_review:` block is not read. + Fork usage is persisted in `session_model_usage` with `task='background_review'` and a completion line is written to `agent.log` (`Background review complete: thread=bg-review calls=… in=… out=… result=…`). diff --git a/website/docs/user-guide/features/mixture-of-agents.md b/website/docs/user-guide/features/mixture-of-agents.md index 72ae9e465b..18abd47e9e 100644 --- a/website/docs/user-guide/features/mixture-of-agents.md +++ b/website/docs/user-guide/features/mixture-of-agents.md @@ -245,6 +245,8 @@ Both internal call types cache normally: - **Reference models** receive a trimmed, deterministic view of the conversation (system prompt and tool transcript stripped — see the loop above). Because that view is a stable function of the stable history, a reference model's prompt prefix repeats across iterations and caches normally. References are short advisory calls with no tools. - **The aggregator** is the acting model. The reference outputs are appended as their *own* trailing user message of private guidance — never merged into your message. Because that block sits at the tail — below the entire stable prefix (system prompt + your message + prior tool history) — it does not invalidate any cached prefix: every request in a tool loop is a byte-identical extension of the previous one minus its guidance block, so the aggregator gets a cache hit on everything above the injection and only the freshly appended tail is new. That is exactly how every normal turn behaves, where each new user message is also uncached tail tokens. Aggregators on the Anthropic Messages, Bedrock Converse or native Gemini wire merge the two adjacent user turns into one message, but as separate content blocks: your message's block is byte-identical to the one later iterations replay, and the guidance block follows it, so the cached prefix still runs through your message. + On the OpenAI-compatible wire the request ends `user(your message), user(guidance)` on the first iteration of a turn. A few chat templates that enforce strict user/assistant alternation (llama.cpp and vLLM Jinja templates, some OpenRouter routes) reject that with a 400 such as `Conversation roles must alternate`. Hermes recovers on its own: it retries that one request with the two adjacent user messages merged, remembers that aggregator destination (endpoint + model) for the rest of the session so later turns are merged up front, and leaves every other destination on the split, cache-stable shape. The merge is applied only where a destination demanded it, because merging everywhere would reintroduce the prefix divergence described above. + So MoA does not sacrifice prompt caching on either call type. Its only real cost is the extra reference calls (once per user turn with the default `fanout`) — you pay for multiple model perspectives, not for broken caches. The long-lived conversation prefix shared with the rest of Hermes is fully intact. ## Notes diff --git a/website/docs/user-guide/security.md b/website/docs/user-guide/security.md index 0832482cd0..1735befbb5 100644 --- a/website/docs/user-guide/security.md +++ b/website/docs/user-guide/security.md @@ -665,6 +665,17 @@ terminal: Paths are relative to `~/.hermes/`. Files are mounted to `/root/.hermes/` inside the container. This list is read by `tools/credential_files.py` (`terminal.credential_files`) — it lives under the `terminal:` block but is loaded by the credential-files module, not the core terminal backend, so it isn't part of the bundled `DEFAULT_CONFIG` snapshot. +### Borrowed CLI logins (Codex CLI, Claude Code) {#borrowed-cli-logins} + +When Hermes has no usable login of its own for `openai-codex` or `anthropic`, it can borrow the Codex CLI's `~/.codex/auth.json` and Claude Code's `~/.claude/.credentials.json` (or Keychain entry) and refresh them on your behalf. Both use single-use, rotating refresh tokens: once two programs hold one token family, whichever refreshes first invalidates the other's copy, which shows up as "I logged in once in the terminal and Hermes keeps failing" (or the reverse). If you run those CLIs alongside Hermes, give Hermes its own login and turn adoption off: + +```yaml +auth: + adopt_external_logins: false # default: true +``` + +With the switch off Hermes never reads or refreshes those files: the `claude_code` credential-pool row disappears, `hermes auth list` prints one line saying so, and the log carries one INFO line per process. Only automatic adoption is affected — `hermes auth add openai-codex` still asks before importing an existing Codex CLI login. Add your own logins with `hermes auth add anthropic` / `hermes auth add openai-codex`. + ### What Each Sandbox Filters | Sandbox | Default Filter | Passthrough Override | diff --git a/website/docs/user-guide/windows-native.md b/website/docs/user-guide/windows-native.md index 5422f76ec6..aec6abda7b 100644 --- a/website/docs/user-guide/windows-native.md +++ b/website/docs/user-guide/windows-native.md @@ -226,7 +226,7 @@ Flags used when spawning: `DETACHED_PROCESS | CREATE_NEW_PROCESS_GROUP | CREATE_ ```powershell hermes gateway status # Merged view: schtasks + Startup folder + running PID -hermes gateway start # Starts the scheduled task now +hermes gateway start # Starts the gateway in the background (asks about login auto-start only on a TTY when nothing is installed) hermes gateway stop # Writes the planned-stop marker, waits for the gateway to drain (≤ agent.restart_drain_timeout, capped at 30 s), then force-kills only if it is still alive hermes gateway restart # Same drain-first stop, then a fresh start hermes gateway uninstall # Removes schtasks entry, Startup shortcut, pid file @@ -234,6 +234,8 @@ hermes gateway uninstall # Removes schtasks entry, Startup shortcut, pid file `hermes gateway status` is idempotent — call it a thousand times in a row and it will never accidentally kill the gateway. (Pre-PR #21561 it silently did, via `os.kill(pid, 0)` colliding with `CTRL_C_EVENT` at the C level — see "process management internals" below if you care about the story.) +Login auto-start is only ever installed on an explicit answer: `hermes gateway install`, a `Y` on a real terminal, or `HERMES_GATEWAY_INSTALL_START_ON_LOGIN=1`. A scripted or piped `hermes gateway start` (no TTY, or `HERMES_NONINTERACTIVE=1`) starts the gateway without touching the Scheduled Task or the Startup folder; set `HERMES_GATEWAY_INSTALL_START_ON_LOGIN=0` to skip the question on a terminal too. + ### Why not a Windows Service? Services require admin rights to install and tie the gateway's lifecycle to machine boot, not user login. The typical Hermes user wants: log in → gateway available, log out → gateway gone. Scheduled Tasks do exactly that without elevation. If you genuinely want a service, use `nssm` or `sc create` manually — but you probably don't.