refactor(credits): throttle the re-warm, scope the seed thread, one subscription predicate

- rewarm_pricing_before_depleted_notice: a failed fetch caches {} for
  _FAILED_CATALOG_TTL_SECONDS and the peek reads that as cold, so every
  header in that window spawned a thread that read the auth store and hit
  the cached {}. Remember when the last warm started and decide inline
  until the window passes. Drop the dead try/except around the pure peek.
- _bg_seed now runs under spawn_context_thread: the warm it gained reads
  the profile's auth store, so the thread must carry the profile scope.
- _rerun_notice_policy replaces the idiom copied at three sites.
- _is_subscription_billed: the free-tier default filtered on any truthy
  billing_mode while _is_model_free keyed on == 'subscription'.
- The no-respawn guard test counts warm calls instead of enumerating
  finished threads (which always read 0).
This commit is contained in:
kshitijk4poor
2026-09-20 13:20:52 +05:30
committed by kshitij
parent 21cf53b888
commit 6cfbc5f891
3 changed files with 37 additions and 32 deletions

View File

@@ -389,6 +389,10 @@ def _hydrate_seed_state(agent, state) -> None:
latch = getattr(agent, "_credits_latch", None)
if isinstance(latch, dict) and state.used_fraction is not None:
latch["seen_below_90"] = True # ONLY this gate — priming seen_grant_unspent would revive the steady-state nag
_rerun_notice_policy(agent)
def _rerun_notice_policy(agent) -> None:
if callable(emit := getattr(agent, "_emit_credits_notices", None)):
emit()
@@ -407,39 +411,34 @@ def _warm_nous_pricing_cache() -> None:
def rewarm_pricing_before_depleted_notice(agent) -> bool:
"""Depleted account, model the peek cannot vouch for, Nous catalog COLD: start a background warm
whose completion re-runs the policy, and return True so the caller leaves the depleted decision
to that re-run instead of flashing a banner the warm catalog would suppress.
Needed because the session-start warm is one-shot while the Nous catalog expires after
``_NOUS_CATALOG_TTL_SECONDS`` — every inference header after that re-evaluated against a cold
peek and brought the banner back for a subscription-billed model. Returns False (decide now)
when the peek is warm, the agent is not on Nous, or a warm is already in flight — including the
warm's own re-run, which must decide against whatever the fetch produced (fail-open: a failed
fetch leaves the peek cold and the banner shows).
"""
"""Depleted account, model the peek cannot vouch for, Nous catalog cold: start a background warm
whose completion re-runs the policy and return True so the caller defers the depleted decision
to it — the session-start warm is one-shot but the catalog expires after
``_NOUS_CATALOG_TTL_SECONDS``, and a cold peek would bring the banner back for a
subscription-billed model. False = decide now: peek warm, not on Nous, a warm in flight (the
warm's own re-run included — fail-open, a failed fetch leaves the peek cold and the banner
shows), or a warm already failed within the failed-catalog window (no per-turn re-spawn)."""
base_url = getattr(agent, "base_url", "") or ""
if getattr(agent, "provider", "") != "nous" or not base_url:
return False
try:
from hermes_cli.models_pricing import peek_cached_pricing
from hermes_cli.models_pricing import _FAILED_CATALOG_TTL_SECONDS, peek_cached_pricing
if peek_cached_pricing(base_url):
return False
except Exception:
if peek_cached_pricing(base_url):
return False
inflight = getattr(agent, "_credits_pricing_warm", None)
if inflight is not None and inflight.is_alive():
previous = getattr(agent, "_credits_pricing_warm", None)
if previous is not None and (
previous.is_alive() or time.monotonic() - previous.started_at < _FAILED_CATALOG_TTL_SECONDS
):
return False
def _warm_then_rerun() -> None:
_warm_nous_pricing_cache()
if callable(emit := getattr(agent, "_emit_credits_notices", None)):
emit()
_rerun_notice_policy(agent)
from agent.memory_provider import spawn_context_thread
thread = spawn_context_thread(_warm_then_rerun, name="credits-pricing-warm")
thread.started_at = time.monotonic()
agent._credits_pricing_warm = thread
thread.start()
return True
@@ -472,14 +471,15 @@ def seed_credits_at_session_start(agent) -> bool:
# A live inference header beat us — don't clobber it, but DO re-run the policy:
# it evaluated against the cold cache and may be showing a banner the warm
# catalog now suppresses.
if callable(emit := getattr(agent, "_emit_credits_notices", None)):
emit()
_rerun_notice_policy(agent)
return
if (state := _credits_state_from_account(info)) is not None:
_hydrate_seed_state(agent, state)
except Exception:
logger.debug("credits ▸ session-start seed (background) failed", exc_info=True)
threading.Thread(target=_bg_seed, name="credits-seed", daemon=True).start()
from agent.memory_provider import spawn_context_thread
spawn_context_thread(_bg_seed, name="credits-seed").start() # the warm reads the profile's auth store
return True
except Exception:
logger.debug("credits ▸ session-start seed failed (fail-open)", exc_info=True) # innermost log: diagnosable dead seed

View File

@@ -214,11 +214,16 @@ def _zero_priced(pricing: Any, keys: tuple[str, str], default: str) -> bool:
return False
def _is_subscription_billed(entry: Any) -> bool:
"""The gateway bills this catalog row to a subscription the account holds, not to credits."""
return isinstance(entry, dict) and entry.get("billing_mode") == "subscription"
def _is_model_free(model_id: str, pricing: dict[str, dict[str, str]]) -> bool:
"""Return True if *model_id* costs no credits: zero-cost prompt AND completion pricing, or a row
the gateway bills to a subscription."""
entry = pricing.get(model_id)
return bool(entry) and (entry.get("billing_mode") == "subscription" or _zero_priced(entry, ("prompt", "completion"), "1"))
return bool(entry) and (_is_subscription_billed(entry) or _zero_priced(entry, ("prompt", "completion"), "1"))
def partition_nous_models_by_tier(
@@ -487,7 +492,7 @@ def recommended_nous_default_model() -> dict[str, Any]:
if free_tier:
model_ids, _unavailable = partition_nous_models_by_tier(model_ids, pricing, free_tier=True)
# Never default onto a subscription-billed row: spending that plan is the user's call.
model_ids = [mid for mid in model_ids if not pricing.get(mid, {}).get("billing_mode")] or model_ids
model_ids = [mid for mid in model_ids if not _is_subscription_billed(pricing.get(mid))] or model_ids
return {"provider": "nous", "model": pick_silent_default_model(model_ids, provider="nous"),
"free_tier": bool(free_tier)}

View File

@@ -312,18 +312,18 @@ def test_header_after_ttl_expiry_rewarms_instead_of_flashing_the_banner(monkeypa
def test_header_on_a_cold_catalog_still_warns_when_the_warm_fails(monkeypatch):
"""Fail-open guard rail: a warm that leaves the catalog cold decides against the cold peek — the
banner shows — and the warm's own re-run does not spawn another warm."""
import threading
banner shows — and neither the warm's own re-run nor the next header (still inside the
failed-catalog window) starts another warm."""
from agent import credits_tracker
_cold_pricing_cache(monkeypatch)
monkeypatch.setattr(credits_tracker, "_warm_nous_pricing_cache", lambda: None)
warms: list = []
monkeypatch.setattr(credits_tracker, "_warm_nous_pricing_cache", lambda: warms.append(1))
agent = _mixin_agent()
before = {t for t in threading.enumerate() if t.name == "credits-pricing-warm"}
agent._emit_credits_notices()
_join_pricing_warm(agent)
assert agent.shown == ["credits.depleted"]
assert len({t for t in threading.enumerate() if t.name == "credits-pricing-warm"} - before) == 0
assert agent._credits_pricing_warm is not None and not agent._credits_pricing_warm.is_alive()
agent._emit_credits_notices() # next header, catalog still cold
_join_pricing_warm(agent)
assert warms == [1]