From 4ef56cef4c6eecc009e2284fe2f1df20664f357a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 03:08:01 -0700 Subject: [PATCH] fix(auth): bound the no-TTL key_cmd token cache; docs for key_cmd Follow-ups on the #85006 salvage: - A key_cmd token with no advertised expiry was cached for the life of the process. The "refresh on 401" contract it relied on has no implementation (SDK retries cover 429/5xx only), so an expired no-TTL token would 401 every request until restart. Cache on a bounded 15-minute window instead; helpers that want a longer cache can advertise their real expiry. - Test for the no-TTL path updated to pin the bounded-window contract; the remint test's $RANDOM (bash-only, empty under dash) replaced with date +%s%N so it exercises remint under any /bin/sh. - website/docs/integrations/providers.md: document key_cmd in the named custom providers section (contract, precedence, secrets.command contrast). --- agent/command_token_source.py | 20 +++++++++++-------- tests/agent/test_command_token_source.py | 25 ++++++++++++++++-------- website/docs/integrations/providers.md | 22 ++++++++++++++++++++- 3 files changed, 50 insertions(+), 17 deletions(-) diff --git a/agent/command_token_source.py b/agent/command_token_source.py index e12cb2e7e8..fac8aa64a4 100644 --- a/agent/command_token_source.py +++ b/agent/command_token_source.py @@ -50,6 +50,13 @@ _TOKEN_REFRESH_LEEWAY_SECONDS = 60.0 # A token helper reads a local credential cache and should answer in # milliseconds; anything approaching this budget is hung, not slow. _MINT_TIMEOUT_SECONDS = 15 +# When a helper advertises NO expiry, the token cannot be cached for the life +# of the process: nothing in the request path re-mints on 401 (the SDK retries +# 429/5xx only), so an expired no-TTL token would 401 every request until +# restart. Re-mint on a bounded window instead — the helper answers from a +# local credential cache in milliseconds, so a periodic re-run is cheap, and a +# helper that wants a longer cache can simply advertise its real expiry. +_NO_TTL_REFRESH_SECONDS = 900.0 class CommandTokenError(RuntimeError): @@ -150,23 +157,20 @@ class CommandTokenSource: self._label = label or "custom" self._lock = threading.Lock() self._token = "" - self._expires_at: Optional[float] = None + self._expires_at: float = 0.0 def __call__(self) -> str: with self._lock: - # ``expires_at is None`` means the command advertised no TTL: use - # the token and rely on the caller's 401 handling, rather than - # inventing an expiry. Same contract as buzz's is_expired(). - if self._token and ( - self._expires_at is None or time.monotonic() < self._expires_at - ): + if self._token and time.monotonic() < self._expires_at: return self._token token, ttl = _mint(self._command, self._label) self._token = token self._expires_at = ( time.monotonic() + max(ttl - _TOKEN_REFRESH_LEEWAY_SECONDS, 5.0) if ttl - else None + # No advertised TTL: bounded cache (see _NO_TTL_REFRESH_SECONDS) + # — there is no 401-driven re-mint hook to fall back on. + else time.monotonic() + _NO_TTL_REFRESH_SECONDS ) logger.debug( "Minted key_cmd token for provider %s (ttl=%s)", diff --git a/tests/agent/test_command_token_source.py b/tests/agent/test_command_token_source.py index b66278f222..e7210480ef 100644 --- a/tests/agent/test_command_token_source.py +++ b/tests/agent/test_command_token_source.py @@ -98,24 +98,33 @@ class TestCaching: assert source() == source() def test_expired_token_is_reminted(self): + # date +%s%N changes every run; $RANDOM would be bash-only (empty + # under dash, which is what /bin/sh is on Debian-family CI). source = CommandTokenSource( - """printf '{"access_token":"tok-%s","expires_in":3600}' $RANDOM""", "dbx" + """printf '{"access_token":"tok-%s","expires_in":3600}' "$(date +%s%N)" """, + "dbx", ) first = source() # Force the cache past its expiry. source._expires_at = 0.0 assert source() != first - def test_no_advertised_ttl_caches_indefinitely(self): - """No TTL means trust the token and refresh on 401. + def test_no_advertised_ttl_caches_on_a_bounded_window(self): + """No TTL means a bounded cache, not a process-lifetime one. - Inventing a synthetic expiry would re-run the command on a schedule - the issuer never asked for. + Nothing in the request path re-mints on 401 (SDK retries cover + 429/5xx only), so caching forever would wedge an expired token until + restart. The window keeps the helper from running per-request while + guaranteeing an eventual re-mint. """ + from agent.command_token_source import _NO_TTL_REFRESH_SECONDS + source = CommandTokenSource("date +%s%N", "dbx") - source() - assert source._expires_at is None - assert source() == source() + first = source() + assert 0 < source._expires_at - time.monotonic() <= _NO_TTL_REFRESH_SECONDS + assert source() == first # cached inside the window + source._expires_at = time.monotonic() - 1 # cross the window + assert source() != first # re-minted after it def test_advertised_ttl_sets_an_expiry(self): source = CommandTokenSource( diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 549c93049f..6137f8f559 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -1271,7 +1271,27 @@ providers: transport: anthropic_messages # for Anthropic-compatible proxies ``` -Each entry accepts: `api` (the endpoint base URL — `base_url`/`url` are accepted aliases), `name` (optional display name; defaults to the dict key), `key_env` or inline `api_key`, `transport` (`chat_completions` / `anthropic_messages` / `codex_responses`), `default_model`, `models`, `context_length`, `discover_models`, `extra_body`, `extra_headers`, `ssl_ca_cert` / `ssl_verify`, and `enabled: false` to hide an entry without deleting it. +Each entry accepts: `api` (the endpoint base URL — `base_url`/`url` are accepted aliases), `name` (optional display name; defaults to the dict key), `key_env` or inline `api_key` or `key_cmd` (see below), `transport` (`chat_completions` / `anthropic_messages` / `codex_responses`), `default_model`, `models`, `context_length`, `discover_models`, `extra_body`, `extra_headers`, `ssl_ca_cert` / `ssl_verify`, and `enabled: false` to hide an entry without deleting it. + +#### Command-minted credentials (`key_cmd`) + +Enterprise gateways often issue short-lived bearer tokens (SSO/OIDC brokers, cloud IAM, internal auth proxies) rather than static API keys, so a token copied into `.env` goes stale mid-session and requests start returning 401. `key_cmd` names a command that *prints* a token; Hermes runs it and caches the result until shortly before expiry, so long sessions keep working with no restart: + +```yaml +providers: + my-gateway: + base_url: "https://gateway.internal.example.com/v1" + api_mode: chat_completions + key_cmd: "my-auth-cli print-token --profile prod" +``` + +Works with any helper that prints a token — `databricks auth token`, `gcloud auth print-access-token`, `az account get-access-token`, `vault read`, or Claude Code-style `apiKeyHelper` scripts. + +The command must print **only** the token on stdout: either bare, or as JSON with an `access_token` field (`expires_in` is honored; absolute `expiry`/`expiresOn` ISO timestamps too). Multi-line output is rejected rather than guessed at. If no expiry is advertised, the token is re-minted on a bounded window. + +Precedence: an explicit `--api-key` flag still wins; otherwise `key_cmd` beats a static `api_key`/`key_env` on the same entry. The minted credential applies to the main agent turn and to auxiliary tasks (title generation, compression, vision, embedding) alike. + +Not to be confused with `secrets.command`, which runs a helper **once at startup** to populate env vars process-wide. Use that for a vault/keychain helper handing back many secrets; use `key_cmd` when one provider's credential must be re-minted *during* a session. :::note Legacy format Older configs used a top-level `custom_providers:` list instead. It still works — Hermes reads both — and `hermes update` auto-migrates it to the `providers:` dict (config v12). Field names differ slightly in the dict format: legacy `model` is `default_model`, and legacy `api_mode` is `transport`.