refactor(agent/models): table-driven credits header parsing, unified account-usage fetchers
- credits_tracker: _MICROS_FIELDS/_USD_FIELDS/_BOOL_FIELDS drive parse_credits_headers; _sticky_notice() replaces repeated AgentNotice literals; collapsed defensive layers - account_usage: _USAGE_FETCHERS dispatch table replaces provider if/elif; shared fetch/parse helpers across the codex/anthropic/openrouter fetchers; drop dead _resolve_codex_usage_url - billing_links, fast_mode: compacted docstrings, collapsed if/return chains - model_metadata: fold effective_provider inference into one expression Notice keys, header names, URLs, log strings and dataclass defaults unchanged.
This commit is contained in:
File diff suppressed because it is too large
Load Diff
@@ -1,11 +1,10 @@
|
||||
"""Provider-agnostic billing/credit recovery links.
|
||||
|
||||
Maps a billing-classified failure onto a recovery link + label. *Detection*
|
||||
is not done here — that is :mod:`agent.error_classifier`
|
||||
(``FailoverReason.billing``), the single source of truth for "credit wall vs.
|
||||
rate limit / auth / transport". The resulting :class:`BillingBlock` rides the
|
||||
turn result and the gateway ``message.complete`` event so every surface (CLI,
|
||||
TUI, desktop) renders one structured signal instead of re-parsing error text.
|
||||
Maps a billing-classified failure onto a recovery link + label. *Detection* is
|
||||
:mod:`agent.error_classifier` (``FailoverReason.billing``), the single source of
|
||||
truth for "credit wall vs. rate limit / auth / transport". The :class:`BillingBlock`
|
||||
rides the turn result and the gateway ``message.complete`` event so every surface
|
||||
renders one structured signal instead of re-parsing error text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -20,10 +19,9 @@ from utils import base_url_host_matches
|
||||
class BillingBlock:
|
||||
"""Structured billing-wall descriptor shared across every surface.
|
||||
|
||||
``is_nous`` is the routing bit: Nous has a first-class in-app billing surface
|
||||
(desktop Settings → Billing, TUI/CLI ``/topup``), so surfaces prefer that over
|
||||
``billing_url``; third-party providers have no in-app flow, so ``billing_url``
|
||||
is the deep link the user actually needs.
|
||||
``is_nous`` is the routing bit: Nous has an in-app billing surface (desktop
|
||||
Settings → Billing, TUI/CLI ``/topup``) preferred over ``billing_url``;
|
||||
third-party providers have none, so ``billing_url`` is the deep link needed.
|
||||
"""
|
||||
|
||||
provider: str
|
||||
@@ -45,11 +43,10 @@ class _Provider:
|
||||
hosts: tuple[str, ...] = ()
|
||||
|
||||
|
||||
# Single source of truth: internal slug(s) + base_url host(s) → billing page.
|
||||
# Curated "add credits / manage billing" landing pages, not marketing homes.
|
||||
# Hosts back the OpenAI-compatible fallback where the slug is a generic bucket
|
||||
# (e.g. "openai_compatible") but base_url reveals the real upstream. An unknown
|
||||
# provider degrades to a readable label with no invented URL.
|
||||
# Single source of truth: slug(s) + base_url host(s) → curated "add credits"
|
||||
# page (not marketing homes). Hosts back the OpenAI-compatible fallback where the
|
||||
# slug is a generic bucket but base_url reveals the real upstream. Unknown
|
||||
# providers degrade to a readable label with no invented URL.
|
||||
_PROVIDERS: tuple[_Provider, ...] = (
|
||||
_Provider("OpenAI", "https://platform.openai.com/settings/organization/billing", ("openai",), ("api.openai.com",)),
|
||||
_Provider("Anthropic", "https://console.anthropic.com/settings/billing", ("anthropic",), ("api.anthropic.com",)),
|
||||
@@ -72,16 +69,15 @@ _BY_SLUG: dict[str, _Provider] = {slug: p for p in _PROVIDERS for slug in p.slug
|
||||
|
||||
def is_nous_inference_route(provider: str, base_url: str) -> bool:
|
||||
"""True when the failing route is the Nous-managed inference gateway."""
|
||||
if (provider or "").strip().lower() == "nous":
|
||||
return True
|
||||
return base_url_host_matches(str(base_url or ""), "inference-api.nousresearch.com")
|
||||
return (provider or "").strip().lower() == "nous" or base_url_host_matches(
|
||||
str(base_url or ""), "inference-api.nousresearch.com"
|
||||
)
|
||||
|
||||
|
||||
def _nous_billing_url() -> Optional[str]:
|
||||
"""Best-effort Nous portal billing URL (text-surface fallback; Nous prefers the in-app flow)."""
|
||||
try:
|
||||
from hermes_cli.nous_account import nous_portal_billing_url
|
||||
|
||||
return nous_portal_billing_url(None)
|
||||
except Exception:
|
||||
return "https://portal.nousresearch.com/billing"
|
||||
@@ -92,22 +88,14 @@ def _resolve_provider_link(slug: str, base_url: str) -> tuple[str, Optional[str]
|
||||
hit = _BY_SLUG.get(slug)
|
||||
if hit:
|
||||
return hit.label, hit.url
|
||||
|
||||
base = str(base_url or "")
|
||||
for p in _PROVIDERS:
|
||||
if any(base_url_host_matches(base, host) for host in p.hosts):
|
||||
return p.label, p.url
|
||||
|
||||
return slug.replace("_", " ").replace("-", " ").strip().title() or "your provider", None
|
||||
|
||||
|
||||
def build_billing_block(
|
||||
*,
|
||||
provider: str,
|
||||
base_url: str,
|
||||
model: str,
|
||||
message: str = "",
|
||||
) -> BillingBlock:
|
||||
def build_billing_block(*, provider: str, base_url: str, model: str, message: str = "") -> BillingBlock:
|
||||
"""Build the billing descriptor for a billing-classified failure.
|
||||
|
||||
``message`` is the guidance already assembled by the agent loop
|
||||
@@ -116,9 +104,7 @@ def build_billing_block(
|
||||
"""
|
||||
slug = (provider or "").strip().lower()
|
||||
model = (model or "").strip()
|
||||
|
||||
if is_nous_inference_route(slug, base_url):
|
||||
return BillingBlock(slug or "nous", "Nous Portal", model, _nous_billing_url(), True, message or "")
|
||||
|
||||
label, url = _resolve_provider_link(slug, base_url)
|
||||
return BillingBlock(slug, label, model, url, False, message or "")
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,17 +1,12 @@
|
||||
"""Bounded fast-mode windows (``/fast auto`` and ``/fast cold``).
|
||||
|
||||
``agent.service_tier`` is ``None`` (normal), ``"priority"`` (static fast),
|
||||
``"auto"`` or ``"cold"``. The static value is pinned into
|
||||
``agent.request_overrides`` at agent build time; the two bounded modes
|
||||
instead open a wall-clock window at each user-turn boundary and layer the
|
||||
provider's fast override onto the request kwargs only while it is open:
|
||||
|
||||
- ``auto`` — every user turn opens a window of ``agent.fast_auto_seconds``.
|
||||
- ``cold`` — only the first turn of a session (no prior history) opens it.
|
||||
|
||||
Only per-request params (``service_tier`` / ``speed``) vary between requests;
|
||||
the system prompt, tools, and messages are untouched, so the prompt cache is
|
||||
preserved across the window boundary.
|
||||
``agent.service_tier`` is ``None`` (normal), ``"priority"`` (static fast, pinned
|
||||
into ``agent.request_overrides`` at build time), ``"auto"`` (every user turn opens
|
||||
a window of ``agent.fast_auto_seconds``) or ``"cold"`` (only a session's first
|
||||
turn, no prior history, opens it). The provider's fast override is layered onto
|
||||
the request kwargs only while the window is open. Only per-request params
|
||||
(``service_tier`` / ``speed``) vary — system prompt, tools, and messages are
|
||||
untouched, so the prompt cache survives the boundary.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -44,19 +39,16 @@ def begin_turn(agent: Any, conversation_history: Any) -> None:
|
||||
def effective_request_overrides(agent: Any) -> dict[str, Any]:
|
||||
"""``agent.request_overrides`` plus the fast override while the window is open."""
|
||||
overrides = dict(getattr(agent, "request_overrides", None) or {})
|
||||
if getattr(agent, "service_tier", None) not in BOUNDED_MODES:
|
||||
return overrides
|
||||
if time.monotonic() >= getattr(agent, "_fast_until", 0.0):
|
||||
if getattr(agent, "service_tier", None) not in BOUNDED_MODES or time.monotonic() >= getattr(
|
||||
agent, "_fast_until", 0.0
|
||||
):
|
||||
return overrides
|
||||
from hermes_cli.models import resolve_fast_mode_overrides
|
||||
|
||||
base_url = getattr(agent, "base_url", None)
|
||||
if getattr(agent, "api_mode", None) == "anthropic_messages":
|
||||
base_url = getattr(agent, "_anthropic_base_url", None) or base_url
|
||||
fast = resolve_fast_mode_overrides(
|
||||
getattr(agent, "model", None),
|
||||
provider=getattr(agent, "provider", None),
|
||||
base_url=base_url,
|
||||
getattr(agent, "model", None), provider=getattr(agent, "provider", None), base_url=base_url
|
||||
)
|
||||
if fast:
|
||||
overrides.update(fast)
|
||||
|
||||
@@ -2714,11 +2714,8 @@ def get_model_context_length(
|
||||
# model has different limits per provider (claude-opus-4.6: 1M on
|
||||
# Anthropic, 128K on Copilot). Generic providers are inferred from the URL.
|
||||
effective_provider = provider
|
||||
if not effective_provider or effective_provider in {"openrouter", "custom"}:
|
||||
if base_url:
|
||||
inferred = _infer_provider_from_url(base_url)
|
||||
if inferred:
|
||||
effective_provider = inferred
|
||||
if base_url and (not effective_provider or effective_provider in {"openrouter", "custom"}):
|
||||
effective_provider = _infer_provider_from_url(base_url) or effective_provider
|
||||
|
||||
# 5a. Copilot live /models — account-specific models (claude-opus-4.6-1m)
|
||||
# absent from models.dev, and the provider-enforced limit for the rest.
|
||||
|
||||
Reference in New Issue
Block a user