fix(models): peek past expired and superseded pricing entries

This commit is contained in:
Mariano Nicolini
2026-09-01 16:35:47 -03:00
parent d7520b2822
commit 3c4e84c166
2 changed files with 31 additions and 5 deletions

View File

@@ -2289,16 +2289,20 @@ def peek_cached_pricing(base_url: str) -> dict[str, dict[str, Any]]:
Accepts a ``/v1``-suffixed URL as well as the pre-``/v1`` root the fetchers
key on, and prefers an authenticated catalog. Scans rather than rebuilding a
key, because callers hold a base URL but no credential.
key, because callers hold a base URL but no credential — newest first, and
skipping expired entries, so a rotated credential does not keep answering
from the catalog its predecessor read.
"""
root = (base_url or "").rstrip("/")
if root.endswith("/v1"):
root = root[:-3].rstrip("/")
authed_prefix = root + _PRICING_AUTH_KEY_PREFIX
for key, cached in _pricing_cache.items():
if cached and key.startswith(authed_prefix):
return cached
return _pricing_cache.get(root) or {}
for key in reversed(list(_pricing_cache)):
if key.startswith(authed_prefix):
cached = _cached_catalog(key)
if cached:
return cached
return _cached_catalog(root) or {}
def _format_price_per_mtok(per_token_str: str) -> str:

View File

@@ -186,3 +186,25 @@ class TestNousCatalogExpiry:
monkeypatch.setattr(models_mod.time, "monotonic", lambda: now + 86_400)
fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
assert len(catalog) == 1
def test_peek_prefers_the_newest_credential(self, per_org_catalog):
"""After a rotation the older entry is still resident and, being
insertion-ordered, comes first."""
fetch_models_with_pricing(api_key="tok-a", base_url=BASE, cache_ttl_seconds=300)
fetch_models_with_pricing(api_key="tok-b", base_url=BASE, cache_ttl_seconds=300)
assert list(peek_cached_pricing(BASE)) == ["org-b/only"]
def test_peek_skips_an_expired_entry(self, catalog, monkeypatch):
"""Reading _pricing_cache directly walked straight past the TTL."""
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
fetch_models_with_pricing(
api_key="sk-test", base_url=BASE,
cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
)
now = models_mod.time.monotonic()
monkeypatch.setattr(
models_mod.time, "monotonic",
lambda: now + _NOUS_CATALOG_TTL_SECONDS + 1,
)
assert peek_cached_pricing(BASE) == {}