refactor(agent/models): compact pricing snapshot, billing/subscription views, reasoning helpers

- usage_pricing: _snap() builder for official-docs pricing entries (table values identical, verified by dump), shared source/version dicts, drop dead DEFAULT_PRICING
- models_dev: _registry_models/_iter_model_entries/_extract_limit helpers replace repeated registry walking; drop dead ModelInfo.format_cost
- billing_view/subscription_view: OrgRoleCapability mixin replaces duplicated is_admin/can_change_plan; shared fetch_portal_state/parse_org_fields
- reasoning_effort/timeouts/summaries, thinking_timeout_guidance, portal_tags: dispatch tables and compacted comment essays; drop dead CODEX_RESPONSES_EFFORTS alias and _match_any
This commit is contained in:
Teknium
2026-09-02 09:20:52 -07:00
parent a3d33fe22f
commit 77743eac8a
13 changed files with 1233 additions and 3034 deletions

View File

@@ -1,32 +1,18 @@
"""Shared dollar-denominated usage model for the billing/subscription surfaces.
"""Shared dollar-denominated usage model for the ``/usage`` and ``/subscription`` bars.
The single source of truth behind the ``/usage`` and ``/subscription`` usage
bars (TUI + CLI). User feedback (Jun 2026): the terminal surfaces show
**dollars**, never "credits", and every usage bar must make the monthly
subscription allowance and separately-purchased top-up dollars distinctly
visible.
Terminal surfaces show **dollars**, never "credits"; the monthly subscription
allowance and separately-purchased top-up dollars must stay distinctly visible.
Data source: the NAS account-info fetch (``NousPortalAccountInfo``), whose
``paid_service_access_info`` carries the three dollar magnitudes we render
(despite the legacy ``*_credits`` field names, these are USD floats):
Data source: ``NousPortalAccountInfo.paid_service_access_info`` (USD floats despite
the legacy ``*_credits`` names): ``subscription_credits_remaining`` (plan $ left),
``purchased_credits_remaining`` (top-up $, rolls over), ``total_usable_credits``;
plus ``subscription.monthly_credits`` (plan bar denominator) and ``current_period_end``.
- ``subscription_credits_remaining`` -> plan dollars left this month
- ``purchased_credits_remaining`` -> top-up dollars left (rolls over)
- ``total_usable_credits`` -> total spendable
plus ``subscription.monthly_credits`` (the plan's monthly $ allowance, the
denominator for the "% used" plan bar) and ``current_period_end`` (renewal).
Design: two SEPARATE bars (decided with the user) rather than one crammed
three-segment bar — at terminal widths three same-glyph density segments are
unreadable. The plan bar is "spent vs allowance this month" (carries % used);
the top-up bar is "money you bought, doesn't expire". Each gets full
resolution and a single fill glyph, so the bar is never ambiguous and never
relies on color.
Fail-open everywhere: any missing/non-finite field degrades to fewer bars or a
magnitudes-only view; a logged-out / unreachable portal yields
``available=False`` and the surface shows nothing.
Design: two SEPARATE bars rather than one three-segment bar — at terminal widths
three same-glyph density segments are unreadable. Plan bar = spent vs allowance
(carries % used); top-up bar = purchased money that doesn't expire. Fail-open:
missing/non-finite fields degrade to fewer bars; logged-out / unreachable portal
yields ``available=False`` and the surface shows nothing.
"""
from __future__ import annotations
@@ -39,14 +25,17 @@ from typing import Any, Optional
logger = logging.getLogger(__name__)
# Below this TOTAL spendable ($), a paid account is flagged "low" — the alert
# state that nudges top-up/upgrade before a mid-run cutoff. Product threshold
# (user feedback): "any amount below $5 should be an alert status."
# Below this TOTAL spendable ($) a paid account is flagged "low" — the alert state
# that nudges top-up/upgrade before a mid-run cutoff (product: "below $5 is an alert").
LOW_BALANCE_THRESHOLD_USD = 5.0
def _finite(value: Any) -> Optional[float]:
"""Return value as a float iff it's a real finite number (not bool/NaN/Inf)."""
"""Return value as a float iff it's a real finite number (not bool/NaN/Inf).
NaN/Infinity slip past isinstance (json.loads parses bare NaN by default) and
would otherwise render as ``$nan`` with a falsely-confident gauge.
"""
if isinstance(value, bool) or not isinstance(value, (int, float)):
return None
f = float(value)
@@ -54,17 +43,33 @@ def _finite(value: Any) -> Optional[float]:
def _fmt_usd(value: Optional[float]) -> str:
"""``$X.YY`` for display. ``None`` -> ``$0.00`` (callers gate on presence)."""
"""``$X,XXX.YY`` for display. ``None`` -> ``$0.00`` (callers gate on presence)."""
return f"${(value or 0.0):,.2f}"
def format_renews(value: Optional[str]) -> Optional[str]:
"""Format an ISO date/timestamp as a human date, e.g. ``Jul 24, 2026``.
def nous_logged_in() -> bool:
"""Cheap local auth-state check: a Nous access token is present. Fail-closed."""
try:
from hermes_cli.auth import get_provider_auth_state
Accepts ``2026-07-24``, ``2026-07-24T11:05:01.000Z``, etc. Returns the raw
string unchanged if it can't be parsed (never raises), and ``None`` for
empty input.
"""
tok = (get_provider_auth_state("nous") or {}).get("access_token")
return isinstance(tok, str) and bool(tok.strip())
except Exception:
return False
def fetch_nous_account(timeout: float):
"""Wall-clock-bounded fresh portal account fetch. Raises on failure/timeout."""
import concurrent.futures
from hermes_cli.nous_account import get_nous_portal_account_info
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
return pool.submit(get_nous_portal_account_info, force_fresh=True).result(timeout=timeout)
def format_renews(value: Optional[str]) -> Optional[str]:
"""ISO date/timestamp -> ``Jul 24, 2026``; unparseable input is returned unchanged."""
if not value:
return None
from datetime import datetime
@@ -76,7 +81,6 @@ def format_renews(value: Optional[str]) -> Optional[str]:
try:
dt = datetime.fromisoformat(iso)
except ValueError:
# Fall back to a bare date prefix (YYYY-MM-DD) if present.
try:
dt = datetime.strptime(text[:10], "%Y-%m-%d")
except ValueError:
@@ -89,9 +93,8 @@ def format_renews(value: Optional[str]) -> Optional[str]:
class UsageBar:
"""One full-resolution bar: ``spent`` of ``total``, plus a remaining figure.
``kind`` is ``"plan"`` (monthly allowance, shows % used) or ``"topup"``
(purchased dollars, no denominator — ``spent`` is 0 and ``total`` ==
``remaining`` so it renders as a full bar of available balance).
``kind`` is ``"plan"`` (monthly allowance, shows % used) or ``"topup"`` (no
denominator — ``spent`` is 0 and ``total == remaining`` so it renders full).
"""
kind: str # "plan" | "topup"
@@ -117,11 +120,8 @@ class UsageBar:
class UsageModel:
"""Surface-agnostic dollar usage model shared by /usage and /subscription.
``status`` classifies the account for copy selection:
- ``"free"`` : no paid access / no subscription (free models only)
- ``"low"`` : paid, but total spendable < $5 (ALERT)
- ``"healthy"`` : paid, total spendable >= $5
- ``"depleted"`` : paid access lost (balance exhausted)
``status``: ``free`` (no paid access / no plan), ``low`` (paid, spendable < $5,
ALERT), ``healthy`` (paid, spendable >= $5), ``depleted`` (paid access lost).
"""
available: bool
@@ -141,11 +141,7 @@ class UsageModel:
def usage_model_from_account(account_info: Any) -> UsageModel:
"""Build a :class:`UsageModel` from a ``NousPortalAccountInfo``. Fail-open.
Returns ``UsageModel(available=False)`` when there's no usable account info
(logged out, no entitlement block). Never raises.
"""
"""Build a :class:`UsageModel` from a ``NousPortalAccountInfo``. Never raises."""
try:
if account_info is None or not getattr(account_info, "logged_in", False):
return UsageModel(available=False)
@@ -163,48 +159,39 @@ def usage_model_from_account(account_info: Any) -> UsageModel:
monthly = _finite(getattr(sub, "monthly_credits", None)) if sub is not None else None
has_subscription = bool(plan_name) or (monthly is not None and monthly > 0)
has_topup = bool(topup_remaining and topup_remaining > 0)
# Total spendable: prefer the server's total; else sum the parts we have.
# Prefer the server's total; else sum the parts we have.
if total_usable is not None:
total_spendable = total_usable
else:
parts = [v for v in (sub_remaining, topup_remaining) if v is not None]
total_spendable = sum(parts) if parts else None
# Status classification.
if paid is False:
status = "depleted"
elif not has_subscription and not (topup_remaining and topup_remaining > 0):
# No plan and no purchased balance -> free-models-only.
status = "free"
elif not has_subscription and not has_topup:
status = "free" # no plan and no purchased balance -> free-models-only
elif total_spendable is not None and total_spendable < LOW_BALANCE_THRESHOLD_USD:
status = "low"
else:
status = "healthy"
# Plan bar — only with a positive monthly allowance AND a remaining we
# can place on it. spent = cap - remaining, clamped (a debt/over-cap
# balance reads as fully spent rather than a nonsensical negative).
# Plan bar needs a positive allowance AND a remaining to place on it; spent is
# clamped so a debt/over-cap balance reads fully spent, not negative.
plan_bar: Optional[UsageBar] = None
if monthly is not None and monthly > 0 and sub_remaining is not None:
remaining = max(0.0, min(monthly, sub_remaining))
plan_bar = UsageBar(
kind="plan",
remaining_usd=remaining,
remaining_usd=max(0.0, min(monthly, sub_remaining)),
total_usd=monthly,
spent_usd=max(0.0, monthly - sub_remaining),
)
# Top-up bar — only when there are purchased dollars to show. No
# denominator (top-up has no monthly cap), so it renders full = balance.
# Top-up has no monthly cap, so the bar renders full = balance.
topup_bar: Optional[UsageBar] = None
if topup_remaining is not None and topup_remaining > 0:
topup_bar = UsageBar(
kind="topup",
remaining_usd=topup_remaining,
total_usd=topup_remaining,
spent_usd=0.0,
)
topup_bar = UsageBar(kind="topup", remaining_usd=topup_remaining, total_usd=topup_remaining, spent_usd=0.0)
return UsageModel(
available=True,
@@ -226,98 +213,51 @@ def usage_model_from_account(account_info: Any) -> UsageModel:
def build_usage_model(*, timeout: float = 10.0) -> UsageModel:
"""Fetch account-info and build the shared usage model. Fail-open.
Dev override: ``HERMES_DEV_CREDITS_FIXTURE`` short-circuits to a fixture so
every usage state is testable without a live account (mirrors the existing
``/usage`` credits-block fixture path).
``HERMES_DEV_CREDITS_FIXTURE`` short-circuits to a fixture so every usage state
is testable without a live account.
"""
fixture = _dev_fixture_usage_model()
if fixture is not None:
return fixture
try:
from hermes_cli.auth import get_provider_auth_state
tok = (get_provider_auth_state("nous") or {}).get("access_token")
if not (isinstance(tok, str) and tok.strip()):
return UsageModel(available=False)
except Exception:
if not nous_logged_in():
return UsageModel(available=False)
try:
import concurrent.futures
from hermes_cli.nous_account import get_nous_portal_account_info
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
account = pool.submit(get_nous_portal_account_info, force_fresh=True).result(timeout=timeout)
return usage_model_from_account(account)
return usage_model_from_account(fetch_nous_account(timeout))
except Exception:
logger.debug("usage ▸ portal fetch failed (fail-open)", exc_info=True)
return UsageModel(available=False)
# =============================================================================
# Dev fixtures (throwaway scaffolding — env-var driven, no live portal)
# =============================================================================
# ── Dev fixtures (throwaway scaffolding — env-var driven, no live portal) ────
def _plan_bar(remaining: float, spent: float) -> UsageBar:
return UsageBar(kind="plan", remaining_usd=remaining, total_usd=20.0, spent_usd=spent)
def _dev_fixture_usage_model() -> Optional[UsageModel]:
"""Map ``HERMES_DEV_CREDITS_FIXTURE`` to a usage model for offline UX work.
Recognized names: ``free | healthy | low | topup | depleted``. Returns
``None`` when the env var is unset (real portal path runs).
"""
"""``HERMES_DEV_CREDITS_FIXTURE`` -> fixture model (``free|healthy|low|topup|depleted``), else None."""
name = (os.getenv("HERMES_DEV_CREDITS_FIXTURE") or "").strip().lower()
if not name:
return None
if name == "free":
return UsageModel(available=True, status="free", plan_name=None)
if name in ("healthy", "mid"):
return UsageModel(
available=True,
status="healthy",
plan_name="Plus",
renews_at="2026-07-01",
subscription_remaining_usd=14.0,
total_spendable_usd=14.0,
plan_bar=UsageBar(kind="plan", remaining_usd=14.0, total_usd=20.0, spent_usd=6.0),
)
if name in ("topup", "top-up"):
return UsageModel(
available=True,
status="healthy",
plan_name="Plus",
renews_at="2026-07-01",
subscription_remaining_usd=14.0,
topup_remaining_usd=12.0,
total_spendable_usd=26.0,
plan_bar=UsageBar(kind="plan", remaining_usd=14.0, total_usd=20.0, spent_usd=6.0),
name = {"mid": "healthy", "top-up": "topup"}.get(name, name)
plus = dict(available=True, plan_name="Plus", renews_at="2026-07-01")
specs: dict[str, dict] = {
"free": dict(available=True, status="free", plan_name=None),
"healthy": dict(
**plus, status="healthy", subscription_remaining_usd=14.0, total_spendable_usd=14.0,
plan_bar=_plan_bar(14.0, 6.0),
),
"topup": dict(
**plus, status="healthy", subscription_remaining_usd=14.0, topup_remaining_usd=12.0,
total_spendable_usd=26.0, plan_bar=_plan_bar(14.0, 6.0),
topup_bar=UsageBar(kind="topup", remaining_usd=12.0, total_usd=12.0, spent_usd=0.0),
)
if name == "low":
return UsageModel(
available=True,
status="low",
plan_name="Plus",
renews_at="2026-07-01",
subscription_remaining_usd=3.4,
total_spendable_usd=3.4,
plan_bar=UsageBar(kind="plan", remaining_usd=3.4, total_usd=20.0, spent_usd=16.6),
)
if name == "depleted":
return UsageModel(
available=True,
status="depleted",
plan_name="Plus",
renews_at="2026-07-01",
subscription_remaining_usd=0.0,
total_spendable_usd=0.0,
plan_bar=UsageBar(kind="plan", remaining_usd=0.0, total_usd=20.0, spent_usd=20.0),
)
return None
),
"low": dict(
**plus, status="low", subscription_remaining_usd=3.4, total_spendable_usd=3.4,
plan_bar=_plan_bar(3.4, 16.6),
),
"depleted": dict(
**plus, status="depleted", subscription_remaining_usd=0.0, total_spendable_usd=0.0,
plan_bar=_plan_bar(0.0, 20.0),
),
}
spec = specs.get(name)
return UsageModel(**spec) if spec else None

View File

@@ -1,12 +1,10 @@
"""Surface-agnostic core for the Phase 2b Remote Spending screens.
"""Surface-agnostic core for the Remote Spending screens.
One fetch/parse per concern, consumed identically by the CLI handler
(``cli.py::_show_billing``), the TUI JSON-RPC methods
(``tui_gateway/server.py``), and any other surface. Mirrors the proven
``agent/account_usage.py::build_credits_view`` pattern: parse the server payload
into a frozen dataclass; **fail open** — when not logged in or the portal is
unreachable, return a struct with ``logged_in=False`` and let the surface degrade
gracefully (never crash).
(``cli.py::_show_billing``), the TUI JSON-RPC methods (``tui_gateway/server.py``),
and any other surface. Parse the server payload into a frozen dataclass and
**fail open**: when not logged in or the portal is unreachable, return a struct
with ``logged_in=False`` and let the surface degrade gracefully (never crash).
Money discipline: the server emits decimal STRINGS (``"142.5"``, not fixed 2dp).
We keep them as :class:`decimal.Decimal` end-to-end and only format for display.
@@ -19,22 +17,16 @@ import os
import uuid
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import Any, Optional
from typing import Any, Callable, Optional
logger = logging.getLogger(__name__)
# =============================================================================
# Decimal money helpers
# =============================================================================
# ── Decimal money helpers ────────────────────────────────────────────────────
def parse_money(value: Any) -> Optional[Decimal]:
"""Parse a server money value (decimal string) into :class:`Decimal`.
Returns None for missing/invalid input. Never raises. Accepts str/int (and,
defensively, float — though the server always sends strings).
"""
"""Server money value (decimal string; defensively int/float) -> Decimal, or None. Never raises."""
if value is None:
return None
try:
@@ -44,30 +36,32 @@ def parse_money(value: Any) -> Optional[Decimal]:
return None
def format_money(value: Optional[Decimal]) -> str:
"""Format a Decimal as ``$X`` / ``$X.YY`` for display.
def format_money(value: Optional[Decimal], *, grouped: bool = False) -> str:
"""``$X`` for whole dollars, ``$X.YY`` (exactly 2dp) otherwise; ``None`` -> ``—``.
Whole dollars show no decimals; any fractional amount shows exactly 2dp:
``Decimal("142.5")`` → ``"$142.50"``, ``Decimal("100")`` → ``"$100"``,
``Decimal("0.01")`` → ``"$0.01"``.
``grouped=True`` adds thousands separators (``$1,234.50``) to mirror the TUI's
``toLocaleString('en-US')`` on plan-catalog rows; the default is intentionally
ungrouped and asserted so across the other surfaces.
"""
if value is None:
return "—"
spec = ",f" if grouped else "f"
if value == value.to_integral_value():
# Whole dollars — no decimal point. format(..., "f") avoids 1E+3 for 1000.
return f"${format(value.to_integral_value(), 'f')}"
# Fractional — always show 2dp.
return f"${format(value.quantize(Decimal('0.01')), 'f')}"
# format(..., "f") avoids 1E+3 for 1000.
return f"${format(value.to_integral_value(), spec)}"
return f"${format(value.quantize(Decimal('0.01')), spec)}"
# =============================================================================
# Parsed sub-structures
# =============================================================================
def _optional_str(raw: dict, key: str) -> Optional[str]:
value = raw.get(key)
return value if isinstance(value, str) else None
# resolvedVia → the human answer to "why THIS card?". Keys are the server's card
# resolution rungs (NAS card-on-file ladder); absent/unknown rungs render no label
# so the display degrades cleanly on servers that don't send resolvedVia yet.
# ── Parsed sub-structures ────────────────────────────────────────────────────
# resolvedVia (server card-on-file ladder rung) → "why THIS card?". Unknown/absent
# rungs render no label so older servers degrade cleanly.
_CARD_PROVENANCE_LABELS = {
"subPin": "the card on your subscription",
"customerDefault": "your default card saved on the portal",
@@ -79,39 +73,34 @@ _CARD_PROVENANCE_LABELS = {
class CardInfo:
brand: str
last4: str
# NAS card-on-file field (post card-resolver): which ladder rung found the
# card. Defaults off so pre-resolver payloads parse unchanged.
resolved_via: Optional[str] = None
resolved_via: Optional[str] = None # ladder rung; None on pre-resolver payloads
@property
def masked(self) -> str:
# A Link payment method has no card number (last4 = "") — render the
# brand alone, not "Link ····".
# A Link payment method has no card number (last4 = "") — brand alone, not "Link ····".
if not self.last4:
return self.brand
return f"{self.brand} ····{self.last4}"
@property
def provenance(self) -> Optional[str]:
"""Human label for why this card was picked, or None (unknown rung /
server too old to say)."""
"""Human label for why this card was picked, or None (unknown rung / old server)."""
if self.resolved_via is None:
return None
return _CARD_PROVENANCE_LABELS.get(self.resolved_via)
@property
def display(self) -> str:
"""The one-line card display: ``Visa ····4242 — the card on your
subscription`` (or just the masked card when provenance is unknown)."""
"""``Visa ····4242 — the card on your subscription`` (masked only when provenance unknown)."""
label = self.provenance
return f"{self.masked} — {label}" if label else self.masked
@dataclass(frozen=True)
class PaymentMethodInfo:
"""The payment method on file. `kind` is "card", "link", or "unknown"
— anything else is normalised to "unknown" at parse time, so consumers
only ever see fields that belong to the kind they are looking at."""
"""Payment method on file. ``kind`` is "card", "link", or "unknown" — anything
else is normalised to "unknown" at parse time so consumers only see fields
that belong to the kind they are looking at."""
kind: str
brand: Optional[str] = None
@@ -119,8 +108,7 @@ class PaymentMethodInfo:
wallet: Optional[str] = None
email: Optional[str] = None
resolved_via: Optional[str] = None
#: What the server called it, when we did not recognise the kind.
raw_kind: Optional[str] = None
raw_kind: Optional[str] = None # what the server called an unrecognised kind
@dataclass(frozen=True)
@@ -146,13 +134,30 @@ class AutoReload:
card: Optional[AutoReloadCard] = None
@dataclass(frozen=True)
class BillingState:
"""Parsed ``GET /api/billing/state`` — the overview screen's data.
class OrgRoleCapability:
"""``is_admin`` / ``can_change_plan`` shared by the billing and subscription states."""
Fail-open: ``logged_in=False`` (and empty fields) when not logged in or the
portal is unreachable.
"""
role: Optional[str]
can_change_plan_raw: Optional[bool]
@property
def is_admin(self) -> bool:
"""Deprecated/display only — legacy OWNER/ADMIN check, NOT a capability check
(use :attr:`can_change_plan` to gate plan-change actions)."""
return (self.role or "").upper() in ("OWNER", "ADMIN")
@property
def can_change_plan(self) -> bool:
"""Server capability when supplied; otherwise the legacy role fallback."""
if self.can_change_plan_raw is not None:
return self.can_change_plan_raw
return self.is_admin
@dataclass(frozen=True)
class BillingState(OrgRoleCapability):
"""Parsed ``GET /api/billing/state``. Fail-open: ``logged_in=False`` (empty
fields) when not logged in or the portal is unreachable."""
logged_in: bool
org_id: Optional[str] = None
@@ -170,37 +175,13 @@ class BillingState:
monthly_cap: Optional[MonthlyCap] = None
auto_reload: Optional[AutoReload] = None
portal_url: Optional[str] = None
# When the fetch failed (vs cleanly not-logged-in), the message for the surface.
error: Optional[str] = None
@property
def is_admin(self) -> bool:
"""Deprecated/display only — a legacy OWNER/ADMIN check.
NOT a capability check; use :attr:`can_change_plan` for gating billing
plan-change actions.
"""
return (self.role or "").upper() in ("OWNER", "ADMIN")
@property
def can_change_plan(self) -> bool:
"""Server capability when supplied; otherwise the legacy role fallback."""
if self.can_change_plan_raw is not None:
return self.can_change_plan_raw
return self.is_admin
error: Optional[str] = None # set when the fetch failed (vs cleanly not-logged-in)
@property
def can_charge(self) -> bool:
"""True when the UI should offer charge/auto-reload actions.
Uses the server-granted plan-change capability (``can_change_plan``,
which itself falls back to the legacy OWNER/ADMIN role check when the
server omits ``canChangePlan``) AND the per-org kill-switch. This lets
the server grant charge capability to non-OWNER/ADMIN roles (e.g.
FINANCE_ADMIN) via ``canChangePlan``, instead of hard-coding the
deprecated 3-role admin check. (The server still enforces; this is
just for graying out actions the user can't take.)
"""
"""Offer charge/auto-reload actions: server-granted ``can_change_plan`` (so
e.g. FINANCE_ADMIN can be granted via ``canChangePlan``) AND the per-org
kill-switch. Display gating only — the server still enforces."""
return self.can_change_plan and self.cli_billing_enabled
@@ -211,11 +192,7 @@ def _parse_card(raw: Any) -> Optional[CardInfo]:
last4 = raw.get("last4")
if not (isinstance(brand, str) and isinstance(last4, str)):
return None
# Post-resolver fields — all optional so both payload generations parse.
resolved_via = raw.get("resolvedVia")
if not isinstance(resolved_via, str):
resolved_via = None
return CardInfo(brand=brand, last4=last4, resolved_via=resolved_via)
return CardInfo(brand=brand, last4=last4, resolved_via=_optional_str(raw, "resolvedVia"))
def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
@@ -224,33 +201,17 @@ def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
kind = raw.get("kind")
if not isinstance(kind, str):
return None
def _optional_string(key: str) -> Optional[str]:
value = raw.get(key)
return value if isinstance(value, str) else None
resolved_via = _optional_string("resolvedVia")
brand = _optional_string("brand")
last4 = _optional_string("last4")
# Settle the kind here, the way _parse_card settles a card, so nothing
# downstream has to re-check which fields this kind is allowed to have.
resolved_via = _optional_str(raw, "resolvedVia")
brand = _optional_str(raw, "brand")
last4 = _optional_str(raw, "last4")
# Settle the kind here (like _parse_card) so nothing downstream re-checks fields.
if kind == "card" and brand and last4:
return PaymentMethodInfo(
kind="card",
brand=brand,
last4=last4,
wallet=_optional_string("wallet"),
resolved_via=resolved_via,
kind="card", brand=brand, last4=last4, wallet=_optional_str(raw, "wallet"), resolved_via=resolved_via
)
if kind == "link":
return PaymentMethodInfo(
kind="link",
email=_optional_string("email"),
resolved_via=resolved_via,
)
return PaymentMethodInfo(
kind="unknown", raw_kind=kind, resolved_via=resolved_via
)
return PaymentMethodInfo(kind="link", email=_optional_str(raw, "email"), resolved_via=resolved_via)
return PaymentMethodInfo(kind="unknown", raw_kind=kind, resolved_via=resolved_via)
def _parse_monthly_cap(raw: Any) -> Optional[MonthlyCap]:
@@ -282,32 +243,29 @@ def _parse_auto_reload_card(raw: Any) -> Optional[AutoReloadCard]:
return None
if kind in ("canonical", "none"):
return AutoReloadCard(kind=kind)
payment_method_id = raw.get("paymentMethodId")
brand = raw.get("brand")
last4 = raw.get("last4")
return AutoReloadCard(
kind=kind,
payment_method_id=payment_method_id if isinstance(payment_method_id, str) else None,
brand=brand if isinstance(brand, str) else None,
last4=last4 if isinstance(last4, str) else None,
payment_method_id=_optional_str(raw, "paymentMethodId"),
brand=_optional_str(raw, "brand"),
last4=_optional_str(raw, "last4"),
)
def parse_org_fields(payload: dict[str, Any]) -> tuple[dict[str, Any], Optional[bool]]:
"""``(org dict or {}, canChangePlan if bool else None)`` — shared by both state parsers."""
raw_org = payload.get("org")
ccp = payload.get("canChangePlan")
return (raw_org if isinstance(raw_org, dict) else {}), (ccp if isinstance(ccp, bool) else None)
def billing_state_from_payload(
payload: dict[str, Any], *, portal_url: Optional[str] = None
) -> BillingState:
"""Map a raw ``/api/billing/state`` JSON dict into :class:`BillingState`."""
raw_org = payload.get("org")
org: dict[str, Any] = raw_org if isinstance(raw_org, dict) else {}
org, can_change_plan_raw = parse_org_fields(payload)
raw_bounds = payload.get("bounds")
bounds: dict[str, Any] = raw_bounds if isinstance(raw_bounds, dict) else {}
presets: list[Decimal] = []
for item in payload.get("chargePresets") or ():
parsed = parse_money(item)
if parsed is not None:
presets.append(parsed)
presets = [p for p in map(parse_money, payload.get("chargePresets") or ()) if p is not None]
return BillingState(
logged_in=True,
@@ -315,11 +273,7 @@ def billing_state_from_payload(
org_slug=org.get("slug"),
org_name=org.get("name"),
role=org.get("role"),
can_change_plan_raw=(
payload.get("canChangePlan")
if isinstance(payload.get("canChangePlan"), bool)
else None
),
can_change_plan_raw=can_change_plan_raw,
balance_usd=parse_money(payload.get("balanceUsd")),
cli_billing_enabled=bool(payload.get("cliBillingEnabled")),
charge_presets=tuple(presets),
@@ -333,95 +287,100 @@ def billing_state_from_payload(
)
# =============================================================================
# Fail-open builders (the surface front doors)
# =============================================================================
# ── Fail-open builders (the surface front doors) ─────────────────────────────
def fetch_portal_state(
endpoint: str,
label: str,
*,
failed: Callable[..., Any],
parse: Callable[[dict, Optional[str]], Any],
portal_fallback: Callable[[str], str],
timeout: float,
log: logging.Logger,
):
"""Shared fail-open fetch+parse for the billing/subscription overview builders.
``failed(**kw)`` builds the ``logged_in=False`` struct: bare on auth failure,
with ``error`` set on a portal/HTTP failure so the surface can show a clear
message. Prefers a server-supplied ``portalUrl`` (absolutized); else
``portal_fallback(portal_base_url)``.
"""
try:
import hermes_cli.nous_billing as nb
except Exception:
return failed(error="billing client unavailable")
try:
payload = getattr(nb, endpoint)(timeout=timeout)
except nb.BillingAuthError:
return failed()
except nb.BillingError as exc:
log.debug("%s ▸ /state fetch failed (fail-open)", label, exc_info=True)
return failed(error=str(exc))
except Exception:
log.debug("%s ▸ /state unexpected error (fail-open)", label, exc_info=True)
return failed(error=f"could not load {label} state")
raw_portal = payload.get("portalUrl") if isinstance(payload, dict) else None
portal_url = nb._absolutize_portal_url(raw_portal) if raw_portal else None
if not portal_url:
try:
portal_url = portal_fallback(nb.resolve_portal_base_url())
except Exception:
portal_url = None
return parse(payload, portal_url)
def build_billing_state(*, timeout: float = 15.0) -> BillingState:
"""Fetch + parse ``/api/billing/state``. Fail-open.
"""Fetch + parse ``/api/billing/state``. Fail-open (see :func:`fetch_portal_state`).
Returns ``BillingState(logged_in=False)`` when not logged in. On a portal/HTTP
failure, returns ``logged_in=False`` with ``error`` set so the surface can show
a clear message rather than crashing.
Dev override: ``HERMES_DEV_BILLING_FIXTURE`` short-circuits to a fixture so the
card-on-file / admin / scope states are testable offline (mirrors
``HERMES_DEV_CREDITS_FIXTURE`` for the usage model).
``HERMES_DEV_BILLING_FIXTURE`` short-circuits to a fixture so card-on-file /
admin / scope states are testable offline.
"""
fixture = _dev_fixture_billing_state()
if fixture is not None:
return fixture
try:
from hermes_cli.nous_billing import (
BillingAuthError,
BillingError,
_absolutize_portal_url,
get_billing_state,
resolve_portal_base_url,
)
except Exception:
return BillingState(logged_in=False, error="billing client unavailable")
try:
payload = get_billing_state(timeout=timeout)
except BillingAuthError:
return BillingState(logged_in=False)
except BillingError as exc:
logger.debug("billing ▸ /state fetch failed (fail-open)", exc_info=True)
return BillingState(logged_in=False, error=str(exc))
except Exception:
logger.debug("billing ▸ /state unexpected error (fail-open)", exc_info=True)
return BillingState(logged_in=False, error="could not load billing state")
# Prefer a server-supplied portalUrl if present (resolved to absolute in case
# it's relative); else build the standard one.
raw_portal = payload.get("portalUrl") if isinstance(payload, dict) else None
portal_url = _absolutize_portal_url(raw_portal) if raw_portal else None
if not portal_url:
try:
portal_url = _fallback_portal_url(resolve_portal_base_url())
except Exception:
portal_url = None
return billing_state_from_payload(payload, portal_url=portal_url)
return fetch_portal_state(
"get_billing_state",
"billing",
failed=lambda **kw: BillingState(logged_in=False, **kw),
parse=lambda payload, portal_url: billing_state_from_payload(payload, portal_url=portal_url),
portal_fallback=lambda base: f"{base.rstrip('/')}/billing?topup=open",
timeout=timeout,
log=logger,
)
def _fallback_portal_url(base: str) -> str:
"""Standard billing deep-link when the server omits ``portalUrl``."""
return f"{base.rstrip('/')}/billing?topup=open"
# =============================================================================
# Dev fixtures (throwaway scaffolding — env-var driven, no live portal)
# =============================================================================
# ── Dev fixtures (throwaway scaffolding — env-var driven, no live portal) ────
def _dev_fixture_billing_state() -> Optional[BillingState]:
"""Map ``HERMES_DEV_BILLING_FIXTURE`` to a :class:`BillingState` for offline UX.
"""``HERMES_DEV_BILLING_FIXTURE`` -> :class:`BillingState` for offline UX; None when unset.
Recognized names::
nocard logged in · billing on · admin · NO card on file
card card on file · auto-reload off
card-autoreload card on file · auto-reload on
notadmin logged in · MEMBER role (billing actions disabled)
billing-off logged in · admin · per-org kill-switch OFF
logged-out not logged in
Returns ``None`` when the env var is unset (the real portal path runs).
Mirrors ``HERMES_DEV_CREDITS_FIXTURE``; the usage *bar* still comes from
``HERMES_DEV_CREDITS_FIXTURE`` (set both to pair a bar with a billing state).
nocard · card · card-sub (provenance label) · card-autoreload · notadmin (MEMBER)
· billing-off (per-org kill-switch) · logged-out. Unknown name → logged-out with
``error`` so the misconfiguration is visible. Pair with ``HERMES_DEV_CREDITS_FIXTURE``
for the usage bar.
"""
name = (os.getenv("HERMES_DEV_BILLING_FIXTURE") or "").strip().lower()
if not name:
return None
aliases = {
"logged_out": "logged-out", "loggedout": "logged-out",
"card_sub": "card-sub",
"card_autoreload": "card-autoreload", "autoreload": "card-autoreload",
"not-admin": "notadmin", "member": "notadmin",
"billing_off": "billing-off", "off": "billing-off",
}
name = aliases.get(name, name)
if name == "logged-out":
return BillingState(logged_in=False)
# Shared fixture portal host (matches subscription_view._DEV_FIXTURE_PORTAL —
# prod host, not staging; the ?topup=open suffix is the /topup deep-link).
portal = "https://portal.nousresearch.com/billing?topup=open"
# Prod portal host (matches subscription_view._DEV_FIXTURE_PORTAL) + the /topup deep-link suffix.
common: dict[str, Any] = dict(
logged_in=True,
org_id="org_acme",
org_slug="acme",
org_name="Acme Inc",
@@ -431,53 +390,35 @@ def _dev_fixture_billing_state() -> Optional[BillingState]:
charge_presets=(Decimal("10"), Decimal("25"), Decimal("50")),
min_usd=Decimal("5"),
max_usd=Decimal("500"),
portal_url=portal,
portal_url="https://portal.nousresearch.com/billing?topup=open",
)
card = CardInfo(brand="Visa", last4="4242")
autoreload_on = AutoReload(enabled=True, threshold_usd=Decimal("5"), reload_to_usd=Decimal("25"))
if name in ("logged-out", "logged_out", "loggedout"):
return BillingState(logged_in=False)
if name == "nocard":
return BillingState(logged_in=True, card=None, **common)
if name == "card":
return BillingState(logged_in=True, card=card, **common)
if name in ("card-sub", "card_sub"):
# Post-resolver: the card came from the subscription (provenance label).
_sub_card = CardInfo(brand="Visa", last4="4242", resolved_via="subPin")
return BillingState(logged_in=True, card=_sub_card, **common)
if name in ("card-autoreload", "card_autoreload", "autoreload"):
return BillingState(logged_in=True, card=card, auto_reload=autoreload_on, **common)
if name in ("notadmin", "not-admin", "member"):
opts = {**common, "role": "MEMBER"}
return BillingState(logged_in=True, card=card, **opts)
if name in ("billing-off", "billing_off", "off"):
opts = {**common, "cli_billing_enabled": False}
return BillingState(logged_in=True, card=None, **opts)
# Unknown name → logged-out so the misconfiguration is visible.
return BillingState(logged_in=False, error=f"unknown HERMES_DEV_BILLING_FIXTURE: {name}")
overrides: dict[str, dict[str, Any]] = {
"nocard": dict(card=None),
"card": dict(card=card),
"card-sub": dict(card=CardInfo(brand="Visa", last4="4242", resolved_via="subPin")),
"card-autoreload": dict(
card=card, auto_reload=AutoReload(enabled=True, threshold_usd=Decimal("5"), reload_to_usd=Decimal("25"))
),
"notadmin": dict(card=card, role="MEMBER"),
"billing-off": dict(card=None, cli_billing_enabled=False),
}
if name not in overrides:
return BillingState(logged_in=False, error=f"unknown HERMES_DEV_BILLING_FIXTURE: {name}")
return BillingState(**{**common, **overrides[name]})
# =============================================================================
# Idempotency
# =============================================================================
# ── Idempotency ──────────────────────────────────────────────────────────────
def new_idempotency_key() -> str:
"""Fresh UUID for a user-confirmed purchase (reuse on retry of the SAME buy).
The ``Idempotency-Key`` header is mandatory on ``POST /charge``; generate one
per confirmed purchase and reuse it across retries so a double-submit collapses
to a single charge. Never reuse a key across different amounts (the server
returns 409 idempotency_conflict).
"""
"""Fresh UUID for a user-confirmed purchase. ``Idempotency-Key`` is mandatory on
``POST /charge``: reuse the key across retries of the SAME buy so a double-submit
collapses to one charge; never reuse across amounts (server 409 idempotency_conflict)."""
return str(uuid.uuid4())
# =============================================================================
# Amount validation (Screen 3 custom input)
# =============================================================================
# ── Amount validation (custom charge input) ──────────────────────────────────
@dataclass(frozen=True)
@@ -490,18 +431,13 @@ class AmountValidation:
def validate_charge_amount(
raw: str, *, min_usd: Optional[Decimal], max_usd: Optional[Decimal]
) -> AmountValidation:
"""Validate a custom charge amount against bounds + 2dp (multipleOf 0.01).
Mirrors the server's accept/reject so the UI can give instant feedback rather
than round-tripping a sure-to-fail charge. The server is still authoritative.
"""
cleaned = (raw or "").strip().lstrip("$").strip()
amount = parse_money(cleaned)
"""Mirror the server's accept/reject (bounds + multipleOf 0.01) for instant UI
feedback; the server is still authoritative."""
amount = parse_money((raw or "").strip().lstrip("$").strip())
if amount is None:
return AmountValidation(ok=False, error="Enter a dollar amount, e.g. 100")
if amount <= 0:
return AmountValidation(ok=False, error="Amount must be greater than $0")
# multipleOf 0.01 — reject sub-cent precision.
if amount != amount.quantize(Decimal("0.01")):
return AmountValidation(ok=False, error="Amount can't be smaller than a cent")
if min_usd is not None and amount < min_usd:

View File

@@ -1,12 +1,10 @@
"""Per-agent iteration budget — thread-safe consume/refund counter.
Extracted from ``run_agent.py``. Each ``AIAgent`` instance (parent or
subagent) holds an :class:`IterationBudget`; the parent's cap comes from
``max_iterations`` (default 500), each subagent's cap comes from
``delegation.max_iterations`` (default 50).
``run_agent`` re-exports ``IterationBudget`` so existing
``from run_agent import IterationBudget`` imports keep working unchanged.
Each ``AIAgent`` (parent or subagent) holds its own :class:`IterationBudget`:
the parent's cap is ``max_iterations`` (default 500), each subagent's is
``delegation.max_iterations`` (default 50), so total iterations across parent
+ subagents can exceed the parent's cap. ``run_agent`` re-exports the class so
``from run_agent import IterationBudget`` keeps working.
"""
from __future__ import annotations
@@ -15,19 +13,9 @@ import threading
class IterationBudget:
"""Thread-safe iteration counter for an agent.
Each agent (parent or subagent) gets its own ``IterationBudget``.
The parent's budget is capped at ``max_iterations`` (default 500).
Each subagent gets an independent budget capped at
``delegation.max_iterations`` (default 50) — this means total
iterations across parent + subagents can exceed the parent's cap.
Users control the per-subagent limit via ``delegation.max_iterations``
in config.yaml.
``execute_code`` (programmatic tool calling) iterations are refunded via
:meth:`refund` so they don't eat into the budget.
"""
"""Thread-safe iteration counter. ``execute_code`` (programmatic tool
calling) iterations are refunded via :meth:`refund` so they don't eat
into the budget."""
def __init__(self, max_total: int):
self.max_total = max_total

View File

@@ -1,34 +1,27 @@
"""LM Studio reasoning-effort resolution shared by the chat-completions
transport and run_agent's iteration-limit summary path.
LM Studio publishes per-model ``capabilities.reasoning.allowed_options`` (e.g.
``["off","on"]`` for toggle-style models, ``["off","minimal","low"]`` for
graduated models). We map the user's ``reasoning_config`` onto LM Studio's
OpenAI-compatible vocabulary, then clamp against the model's allowed set so
the server doesn't 400 on an unsupported effort.
LM Studio publishes per-model ``capabilities.reasoning.allowed_options``
(``["off","on"]`` for toggle models, ``["off","minimal","low"]`` for graduated
ones). We map the user's ``reasoning_config`` onto LM Studio's OpenAI-compatible
vocabulary, then clamp against the model's allowed set so the server doesn't 400.
"""
from __future__ import annotations
from typing import List, Optional
# LM Studio accepts these top-level reasoning_effort values via its
# OpenAI-compatible chat.completions endpoint.
# Top-level reasoning_effort values LM Studio's OpenAI-compatible endpoint accepts.
_LM_VALID_EFFORTS = {"none", "minimal", "low", "medium", "high", "xhigh"}
# Toggle-style models publish allowed_options as ["off","on"] in /api/v1/models.
# Map them onto the OpenAI-compatible request vocabulary.
# Toggle-style models publish allowed_options as ["off","on"]; map onto the
# request vocabulary. Also applied to the published allowed_options themselves.
_LM_EFFORT_ALIASES = {"off": "none", "on": "medium"}
# Hermes' generic effort ladder grew past LM Studio's vocabulary ("max",
# "ultra"). Clamp the stronger generic levels onto LM Studio's ceiling: left
# alone they miss _LM_VALID_EFFORTS, keep the initialized "medium" default and
# are thereby conflated with unparseable input, so asking for more reasoning
# yields less than "xhigh". Mirrors the ceiling clamp every other provider
# applies (see agent/transports/codex.py).
#
# Deliberately separate from _LM_EFFORT_ALIASES: that mapping is also applied
# to the model's published allowed_options, which must not be rewritten.
# Hermes' ladder grew past LM Studio's vocabulary ("max", "ultra"). Without this
# ceiling clamp they miss _LM_VALID_EFFORTS, keep the "medium" default and are
# conflated with unparseable input — asking for more yields less than "xhigh".
# Kept separate from _LM_EFFORT_ALIASES, which must not rewrite allowed_options.
_LM_EFFORT_CLAMP = {"max": "xhigh", "ultra": "xhigh"}
@@ -36,12 +29,12 @@ def resolve_lmstudio_effort(
reasoning_config: Optional[dict],
allowed_options: Optional[List[str]],
) -> Optional[str]:
"""Return the ``reasoning_effort`` string to send to LM Studio, or ``None``.
"""Return the ``reasoning_effort`` to send to LM Studio, or ``None``.
``None`` means "omit the field": the user picked a level the model can't
honor, so let LM Studio fall back to the model's declared default rather
than silently substituting a different effort. When ``allowed_options`` is
falsy (probe failed), skip clamping and send the resolved effort anyway.
honor, so LM Studio falls back to the model's declared default rather than
a silently substituted effort. Falsy ``allowed_options`` (probe failed)
skips clamping and sends the resolved effort anyway.
"""
effort = "medium"
if reasoning_config and isinstance(reasoning_config, dict):

File diff suppressed because it is too large Load Diff

View File

@@ -1,32 +1,16 @@
"""Centralized Nous Portal request tags.
Every Hermes request that hits the Nous Portal — main agent loop, auxiliary
client (compression / titles / vision / web_extract / session_search / etc.),
and any future code path — must carry the same product-attribution tags so
Nous can attribute usage to Hermes Agent and bucket it by client release.
Every Hermes request to the Nous Portal (main loop, auxiliary client, fallback
paths) must carry the same product-attribution tags, sent in OpenAI-compatible
``extra_body['tags']``::
Tag shape (sent in OpenAI-compatible ``extra_body['tags']``):
["product=hermes-agent", "client=hermes-client-v<__version__>"]
[
"product=hermes-agent",
"client=hermes-client-v<__version__>",
]
The version is sourced live from ``hermes_cli.__version__`` so it auto-aligns
to whatever release is installed; the release script
(``scripts/release.py``) regex-bumps that single string, and every Portal
request picks up the new tag on the next process start.
Why one helper instead of inlining the literal at each site:
* Four call sites (main loop profile, aux client, run_agent compression
fallback, web_tools fallback) used to drift apart — see PR #24194 which
only got the aux site, leaving the main loop sending a different tag set.
* Tests should assert the same tag list everywhere; centralizing makes that
assertion a one-liner against this module.
Do NOT pre-compute these as module-level constants in the consumers. The
version can change at runtime (editable installs, hot-reload tooling), and
``hermes_cli.__version__`` is the canonical source of truth.
One helper instead of inlined literals: the call sites drifted apart before,
and tests can assert one tag list everywhere. The version is read live from
``hermes_cli.__version__`` (the release script bumps that single string) — do
NOT pre-compute it as a module constant in consumers; it can change at runtime
(editable installs, hot reload).
"""
from __future__ import annotations
@@ -34,63 +18,47 @@ from __future__ import annotations
from contextvars import ContextVar
from typing import List, Optional
# ── Ambient conversation context ─────────────────────────────────────────────
#
# The main agent loop knows its ``session_id``; the dozens of auxiliary call
# sites (compression, title generation, vision, web_extract, session_search,
# MoA reference/aggregator slots, curator, kanban helpers, ...) do not — they
# funnel through ``agent.auxiliary_client.call_llm`` which has no session
# handle. Rather than threading a ``session_id`` parameter through every one
# of those call sites (and every future one), the agent loop publishes the
# active conversation id here and ``nous_portal_tags()`` picks it up as a
# fallback whenever no explicit ``session_id`` is passed.
#
# ContextVar (not a module global) so concurrent agents in one process —
# gateway sessions, delegate_task subagents, batch runners — never see each
# other's conversation id. Worker threads spawned via
# ``tools.thread_context.propagate_context_to_thread`` (background review,
# MoA fan-out, tool executor) inherit it through the copied Context; bare
# threads (title generator) capture it explicitly at spawn time.
# Ambient conversation id (ATTRIBUTION value, sent as ``conversation=<id>``).
# The agent loop publishes it at turn entry; the dozens of auxiliary call
# sites funnelling through ``auxiliary_client.call_llm`` (no session handle)
# pick it up via ``nous_portal_tags()`` instead of threading a session_id
# parameter everywhere. A ContextVar, not a module global, so concurrent agents
# in one process (gateway sessions, delegate subagents) never see each other's
# id; ``tools.thread_context.propagate_context_to_thread`` workers inherit it,
# bare threads capture it at spawn time.
_conversation_id: ContextVar[Optional[str]] = ContextVar(
"nous_portal_conversation_id", default=None
)
# ── Ambient routing/affinity scope ───────────────────────────────────────────
#
# Separate from the conversation id above, which is an ATTRIBUTION value: it
# names the conversation a request belongs to and is sent to the Portal as
# ``conversation=<id>``. The affinity scope is a ROUTING value — OpenRouter's
# sticky ``session_id``, Nous Portal's sticky key and xAI's ``x-grok-conv-id``
# use it to pin one conversation to one backend/prompt cache.
#
# The two agree for every host that keeps one session id per conversation, so
# the providers historically read the attribution id for both. They diverge
# for a host that mints one physical session per RESPONSE: attribution still
# resolves per row, while routing must follow the key the host declared for
# the whole chat (``agent.prompt_cache_scope.declared_conversation_scope``,
# issue #96811). Only that declared value is published here — unset means
# "no declaration", and consumers fall back to the conversation id exactly as
# before, so delegate trees keep sharing their parent's sticky key.
# Ambient affinity scope (ROUTING value): OpenRouter's sticky ``session_id``,
# Nous Portal's sticky key and xAI's ``x-grok-conv-id`` pin a conversation to
# one backend/prompt cache. Usually equal to the conversation id, but a host
# that mints one physical session per RESPONSE must route on the key it
# declared for the whole chat (``prompt_cache_scope.declared_conversation_scope``).
# Only that declared value is published; unset means consumers fall back to the
# conversation id, so delegate trees keep sharing their parent's sticky key.
_affinity_scope: ContextVar[Optional[str]] = ContextVar(
"hermes_affinity_scope", default=None
)
def set_affinity_scope(scope: Optional[str]):
"""Publish the declared routing/affinity scope for this turn.
def _reset_var(var: ContextVar, token) -> None:
"""Reset ``var``; a token from another Context (reset on a different
thread) falls back to clearing rather than raising in cleanup paths."""
try:
var.reset(token)
except Exception:
var.set(None)
Returns the ContextVar token; pair with :func:`reset_affinity_scope`.
"""
def set_affinity_scope(scope: Optional[str]):
"""Publish the declared routing/affinity scope; returns the ContextVar token."""
return _affinity_scope.set(scope or None)
def reset_affinity_scope(token) -> None:
"""Restore the previous affinity scope (pair with ``set_affinity_scope``)."""
try:
_affinity_scope.reset(token)
except Exception:
_affinity_scope.set(None)
_reset_var(_affinity_scope, token)
def get_affinity_scope() -> Optional[str]:
@@ -101,22 +69,16 @@ def get_affinity_scope() -> Optional[str]:
def set_conversation_context(conversation_id: Optional[str]):
"""Publish the active conversation id for ambient Portal tagging.
Called by the agent loop at turn entry with the conversation's stable
id (the session-lineage ROOT id, so the tag survives context-compression
session rotation). Pass ``None`` to clear. Returns the ContextVar token
so callers can ``reset_conversation_context(token)`` on turn exit.
Called by the agent loop at turn entry with the session-lineage ROOT id
(so the tag survives context-compression rotation). ``None`` clears.
Returns the ContextVar token for ``reset_conversation_context``.
"""
return _conversation_id.set(conversation_id or None)
def reset_conversation_context(token) -> None:
"""Restore the previous conversation context (pair with ``set_...``)."""
try:
_conversation_id.reset(token)
except Exception:
# Token from another Context (e.g. reset on a different thread) —
# fall back to clearing rather than raising in cleanup paths.
_conversation_id.set(None)
_reset_var(_conversation_id, token)
def get_conversation_context() -> Optional[str]:
@@ -125,11 +87,7 @@ def get_conversation_context() -> Optional[str]:
def _hermes_version() -> str:
"""Return the current Hermes release version, e.g. ``"0.13.0"``.
Falls back to ``"unknown"`` if ``hermes_cli`` cannot be imported (should
never happen in a real install — guarded for defensive testing).
"""
"""Current Hermes release version; ``"unknown"`` if hermes_cli is unimportable."""
try:
from hermes_cli import __version__
return __version__
@@ -138,48 +96,24 @@ def _hermes_version() -> str:
def hermes_client_tag() -> str:
"""Return the ``client=...`` tag for Nous Portal requests.
Format: ``client=hermes-client-v<MAJOR>.<MINOR>.<PATCH>``.
"""
"""``client=hermes-client-v<MAJOR>.<MINOR>.<PATCH>``."""
return f"client=hermes-client-v{_hermes_version()}"
def conversation_tag(session_id: str) -> str:
"""Return the ``conversation=...`` tag for a Hermes session/conversation.
Format: ``conversation=<session_id>``. ``session_id`` is the canonical
Hermes conversation identifier (``AIAgent.session_id``) — the same value
used for ``~/.hermes/sessions/`` storage, session logs, and lineage.
Unlike the product/client tags this is high-cardinality (one value per
conversation), so it is only appended when a session id is actually
available — never as part of the always-on base tag set.
"""
"""``conversation=<session_id>`` — high-cardinality, so only appended when
a session id is actually available, never in the always-on base set."""
return f"conversation={session_id}"
def nous_portal_tags(session_id: str | None = None) -> List[str]:
"""Return the canonical list of Nous Portal product tags.
"""Return a fresh list of the canonical Nous Portal tags.
Always returns a fresh list so callers can mutate it freely
(e.g. ``merged_extra.setdefault("tags", []).extend(nous_portal_tags())``).
When ``session_id`` is provided, a ``conversation=<session_id>`` tag is
appended so Portal usage can be attributed to a specific Hermes
conversation. When it is omitted, the ambient conversation context
(``set_conversation_context``, published by the agent loop at turn
entry) is used instead — this is how auxiliary calls (compression,
titles, vision, MoA slots, ...) inherit the conversation tag without
per-call-site plumbing. Callers outside any conversation (e.g. the
auxiliary client's import-time base tags) get the canonical two-tag set.
The ambient conversation context (lineage ROOT id published by the agent
loop) wins over the explicit ``session_id``, which remains a fallback for
callers outside any agent turn; with neither, the two-tag base set.
"""
tags = ["product=hermes-agent", hermes_client_tag()]
# Ambient context first: the agent loop publishes the lineage ROOT id
# (stable across context-compression rotation and delegate subagent
# trees), which is the better conversation key than a per-segment
# session_id passed explicitly. The explicit argument remains as a
# fallback for callers running outside any agent turn.
effective = get_conversation_context() or session_id
if effective:
tags.append(conversation_tag(effective))

View File

@@ -1,37 +1,26 @@
"""Canonical reasoning-effort vocabulary and wire clamping.
Hermes' internal effort ladder (``hermes_constants.VALID_REASONING_EFFORTS``
plus the ``none`` disable level) is wider than what any single provider wire
accepts. Historically every transport and provider profile hand-rolled its own
translation map, and the class of bugs that produced was constant: a new
internal level (``ultra``) leaking to a wire that rejects it with HTTP 400
(#89503, #70058), or an unknown level being dropped to a weak default so the
strongest ask resolved *weaker* than an explicit ``high`` — a ladder
inversion (#74295, #87279).
This module is the single source of truth both kinds of code use instead:
plus ``none``) is wider than any single provider wire accepts. Hand-rolled
per-transport translation maps produced two recurring bugs: a new internal
level (``ultra``) leaking to a wire that 400s on it, and an unknown level
dropped to a weak default so the strongest ask resolved *weaker* than an
explicit ``high`` (ladder inversion). This module is the single source of
truth instead:
- :data:`EFFORT_LADDER` — canonical low→high ordering.
- :func:`clamp_effort` — the one clamping policy: keep a supported level
verbatim, otherwise take the **nearest weaker** supported level (never
silently escalate cost above what was asked), and only when nothing weaker
exists take the weakest supported level (a provider whose minimum thinking
level is ``high`` serves ``high`` for a ``low`` ask — GLM-5.2's shape).
- Named wire-vocabulary constants for the common OpenAI-compatible surfaces,
so call sites declare *data* ("this route accepts these levels") rather
than logic.
- :func:`clamp_effort` — keep a supported level verbatim, else the **nearest
weaker** supported level (never silently escalate cost); only when nothing
weaker exists take the weakest supported level (GLM-5.2's floor is ``high``).
- Named wire-vocabulary constants so call sites declare *data*, not logic.
Rules for call sites:
1. **Wire shape stays local.** Whether a route wants ``extra_body.reasoning``,
a top-level ``reasoning_effort`` string, or a ``thinking`` toggle is the
caller's business. Only the *vocabulary math* lives here.
2. **Unset stays unset.** ``clamp_effort`` translates an explicit request; it
does not invent one. When the user expressed no effort, prefer omitting
the field so the server default applies.
3. **Never patch a predicate.** When a provider rejects a level, fix its
declared supported set (data), never add another vendor-name special case
at the call site.
1. Wire shape (``extra_body.reasoning`` vs top-level ``reasoning_effort`` vs
a ``thinking`` toggle) stays local; only the vocabulary math lives here.
2. Unset stays unset: ``clamp_effort`` translates an explicit request, never
invents one — omit the field so the server default applies.
3. Never patch a predicate: when a provider rejects a level, fix its declared
supported set (data), not the call site.
"""
from __future__ import annotations
@@ -39,32 +28,26 @@ from __future__ import annotations
import re
from typing import Optional, Sequence
#: K3 slug detector — matches ``k3`` as a delimited token (``k3``,
#: ``k3-256k``, ``kimi-k3``, ``kimi-k3-cot``) without matching K2-era names
#: (``kimi-k2.6``). From #76427 by @ruizanthony.
#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``)
#: without matching K2-era names (``kimi-k2.6``).
_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)")
# Canonical low→high ordering used for nearest-level clamping. Superset of
# hermes_constants.VALID_REASONING_EFFORTS ("none" included so an explicit
# disable can be clamped too when a provider publishes it as a level).
# Canonical low→high ordering for nearest-level clamping. Includes "none" so an
# explicit disable can be clamped when a provider publishes it as a level.
EFFORT_LADDER: tuple[str, ...] = (
"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra",
)
# ``ultra`` is Hermes-internal ladder vocabulary (the Codex product tier); no
# provider wire accepts it verbatim anywhere. Every declared wire set below
# therefore stops at ``max`` — ``ultra`` always clamps down.
# ``ultra`` is Hermes-internal (the Codex product tier); no wire accepts it, so
# every declared set below stops at ``max`` and ``ultra`` always clamps down.
#: The widest OpenAI-compatible wire vocabulary (OpenRouter, Nous Portal):
#: exactly max|xhigh|high|medium|low|minimal|none.
#: Widest OpenAI-compatible wire vocabulary (OpenRouter, Nous Portal).
OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = (
"none", "minimal", "low", "medium", "high", "xhigh", "max",
)
#: OpenAI/Codex Responses backend — per-model vocabulary, live-verified
#: (Aug 2026): ``minimal`` is rejected by both generations (clamps to low);
#: ``max`` is gpt-5.6-only — gpt-5.5 rejects it with "Supported values are:
#: 'none', 'low', 'medium', 'high', 'xhigh'" (#68365's premise, confirmed).
#: OpenAI/Codex Responses, per model generation (live-verified): ``minimal``
#: is rejected by both (clamps to low); ``max`` is gpt-5.6-only.
CODEX_GPT56_EFFORTS: tuple[str, ...] = (
"none", "low", "medium", "high", "xhigh", "max",
)
@@ -80,78 +63,64 @@ def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
return CODEX_LEGACY_EFFORTS
#: Backward-compat alias (pre-#68365-verification name).
CODEX_RESPONSES_EFFORTS: tuple[str, ...] = CODEX_GPT56_EFFORTS
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Actual Computer relays (SGLang/vLLM): none/low/medium/high/max.
#: Actual Computer relays (SGLang/vLLM).
ACTUAL_RELAY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max")
#: Moonshot/Kimi K3: low/high/max (server default high).
#: Moonshot/Kimi K3 (server default high) vs K2-era models.
KIMI_K3_EFFORTS: tuple[str, ...] = ("low", "high", "max")
#: Moonshot/Kimi K2-era models: low/medium/high.
KIMI_K2_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: OpenCode "Ox Alpha" stealth model (x-preview-f-free): thinking is always
#: on and the wire accepts exactly low/high/max — medium/none/xhigh 400 with
#: "This model always engages in thinking and cannot be disabled; please use
#: low, high, or max" (verified live 2026-08-21). xhigh rounds up to max.
#: OpenCode "Ox Alpha" (x-preview-f-free): thinking cannot be disabled and the
#: wire accepts exactly low/high/max (medium/none/xhigh 400); xhigh rounds up.
OX_ALPHA_EFFORTS: tuple[str, ...] = ("low", "high", "max")
OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Tencent TokenHub: low/medium/high.
#: Tencent TokenHub.
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Nebius Token Factory: low/medium/high (top-level reasoning_effort knob).
#: Nebius Token Factory (top-level reasoning_effort knob).
NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Kimi K3's vendor-documented translation quirks (platform.kimi.ai
#: thinking-model guide): ``high`` is K3's positional middle AND server
#: default, so ``medium`` rounds to it rather than down to ``low``; ``xhigh``
#: rounds up to ``max`` (K3's top tier), matching the kimi-coding plugin.
#: Kimi K3 vendor-documented quirks: ``high`` is K3's positional middle AND
#: server default, so ``medium`` rounds to it rather than down to ``low``;
#: ``xhigh`` rounds up to ``max`` (K3's top tier).
KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"}
#: GLM-5.2 native reasoning_effort knob: exactly two enabled levels,
#: ``high`` (its minimum thinking level) and ``max`` (per Z.AI/BigModel
#: docs). ``xhigh`` requests the top tier, not the floor.
#: GLM-5.2 native knob: exactly ``high`` (its minimum thinking level) and
#: ``max``; ``xhigh`` requests the top tier, not the floor.
GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified
#: live on api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every
#: level accepted with monotonic reasoning-token scaling (low=4, medium=11,
#: high=98, max=125 on the probe prompt). ``xhigh`` requests the top tier.
#: GLM-5.3 widens the knob to a graded scale (live-verified, monotonic
#: reasoning-token scaling); ``xhigh`` requests the top tier.
GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh``
#: requests the top tier (matches the shipped profile mapping).
#: DeepSeek V4 OpenAI-compat endpoint; ``xhigh`` requests the top tier.
DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
DEEPSEEK_V4_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Ollama Cloud /v1/chat/completions: accepts {none, low, medium, high, max};
#: rejects ``minimal`` with HTTP 400. ``xhigh`` requests the top tier.
#: Ollama Cloud /v1/chat/completions: rejects ``minimal`` with HTTP 400.
OLLAMA_CLOUD_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max")
OLLAMA_CLOUD_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Meta Model API (Muse): minimal..xhigh; rejects ``none``.
#: Meta Model API (Muse): rejects ``none``.
META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh")
#: Upstage Solar Pro/Open: low/medium/high.
#: Upstage Solar Pro/Open.
SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
"""Supported effort set for a Moonshot/Kimi model slug.
"""Supported effort set for a Moonshot/Kimi slug.
K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``,
and the ``kimi-k3*`` aliases; its documented set is low/high/max.
Everything earlier speaks low/medium/high. Boundary-matched so K2-era
names (``kimi-k2.6``) never match (detection regex from #76427 by
@ruizanthony).
K3 is served as bare ``k3``, plan variants (``k3-256k``) and ``kimi-k3*``
aliases; everything earlier speaks low/medium/high. Boundary-matched so
K2-era names (``kimi-k2.6``) never match.
"""
m = (model or "").strip().lower().split("/")[-1]
if _KIMI_K3_SLUG_RE.search(m):
@@ -166,22 +135,14 @@ def clamp_effort(
) -> Optional[str]:
"""Clamp a requested reasoning effort onto a wire's supported levels.
``overrides`` is an optional declared mapping consulted first, for routes
whose vendor documents a translation that differs from nearest-weaker
(Kimi K3 documents ``medium → high``: high is its positional middle and
server default). Overrides are data, not logic — a call site never adds
vendor ``if``\\ s around this function.
Otherwise: returns the requested effort unchanged when it is supported,
when the supported set is unknown (``None``/empty), or when the effort
isn't a recognized ladder level (custom providers may use bespoke names —
pass through rather than guess). Otherwise returns the **nearest weaker**
supported level, so a clamp never silently escalates cost; when nothing
weaker exists, the weakest supported level is returned (the caller asked
for *some* thinking and the provider's floor is the closest honest match).
The policy is monotonic: a stronger request never resolves to a weaker
wire level than a weaker request would.
``overrides`` (a declared vendor mapping, e.g. Kimi K3 ``medium → high``)
is consulted first. Otherwise the request passes through unchanged when it
is supported, when the supported set is unknown/empty, or when it isn't a
recognized ladder level (custom providers may use bespoke names). Else the
**nearest weaker** supported level is returned so a clamp never escalates
cost; when nothing weaker exists, the weakest supported level is (the
provider's floor is the closest honest match). Monotonic: a stronger
request never resolves weaker than a weaker request would.
"""
requested = str(effort or "").strip().lower()
if not requested or not supported:
@@ -199,9 +160,8 @@ def clamp_effort(
return mapped
if requested not in EFFORT_LADDER:
return effort
# "none" disables reasoning — it is never a *degradation target* for an
# enabled ask (clamping "minimal" to "none" would silently switch
# thinking off). It still passes through verbatim when requested.
# "none" disables reasoning — never a degradation target for an enabled
# ask (clamping "minimal" to "none" would silently switch thinking off).
candidates = [level for level in supported_norm if level != "none"]
if not candidates:
return effort
@@ -218,9 +178,8 @@ def clamp_effort(
def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
"""Extract the user's explicit effort from a reasoning config, or None.
Returns ``None`` when the config is absent, malformed, carries no effort,
or reasoning is explicitly disabled — callers should then omit the wire
field entirely so the server default applies (rule 2 above).
None when the config is absent/malformed, carries no effort, or reasoning
is explicitly disabled — callers then omit the wire field (rule 2 above).
"""
if not isinstance(reasoning_config, dict):
return None

View File

@@ -1,32 +1,13 @@
"""Boundary repair for providers that stream reasoning as discrete summary parts.
Reasoning-summary models (OpenAI's gpt-5.x family, and anything relaying the
Responses API onto the OpenAI chat wire) do not stream a chain of thought token
by token. They emit one ``reasoning_content`` delta per *completed* summary
part, each opening with a bold markdown heading::
{"delta": {"reasoning_content": "**Investigating likely culprit PRs**"}}
{"delta": {"reasoning_content": "**Inspecting message schema**"}}
On the Responses API those parts are delimited by ``summary_index``
(``response.reasoning_summary_part.added`` / ``.done``). The OpenAI chat wire
carries no such field — verified live against Nous Portal's
``openai/gpt-5.6-sol``, whose reasoning chunks contain nothing but
``delta.reasoning_content`` — so the boundary cannot be recovered from
metadata, and consumers that concatenate deltas glue the parts together:
**Investigating likely culprit PRs****Inspecting message schema**
That ``****`` run is neither a bold close nor a bold open to a markdown parser,
so the whole trace renders as one unbroken, unspaced, half-bold paragraph.
The AI SDK hit exactly this (vercel/ai#6742) and fixed it upstream by starting
a new reasoning part per ``summary_index``. That route needs the index, which
this wire does not give us, so we re-derive the boundary from the one signal it
does carry: a delta opening a bold heading. Hermes' own Responses adapter
already joins its summary parts with a blank line
(``agent/codex_responses_adapter.py``), so this brings the chat-completions
stream in line with the path that keeps the structure.
Reasoning-summary models (OpenAI gpt-5.x and anything relaying the Responses
API onto the OpenAI chat wire) emit one ``reasoning_content`` delta per
*completed* summary part, each opening with a bold heading. The Responses API
delimits parts by ``summary_index``; the chat wire carries no such field
(verified live on Nous Portal ``openai/gpt-5.6-sol``), so concatenating deltas
glues ``**One****Two**`` into one half-bold paragraph. We re-derive the
boundary from the one signal the wire keeps — a delta opening a bold heading —
matching the blank-line join Hermes' own Responses adapter already does.
"""
from __future__ import annotations
@@ -37,31 +18,20 @@ __all__ = ["separate_glued_reasoning_blocks"]
def separate_glued_reasoning_blocks(previous: str, delta: str) -> str:
"""Return *delta*, prefixed with a paragraph break when it glues onto *previous*.
*previous* is the reasoning text accumulated so far (only its tail matters).
A break is inserted when *delta* opens a bold heading and *previous* is
mid-line, which is the summary-part boundary the chat wire drops. Both
shapes the upstream issue reports are covered: a heading-only part butting
against the next heading (``**One****Two**``), and a part whose prose body
butts against the next heading (``...interaction!**Next**``).
Token-streamed reasoning is left alone: its deltas carry their own leading
whitespace, so *previous* ends mid-line only when the model really did run
two parts together.
A break is inserted when *delta* opens a *closed* bold heading and
*previous* (the accumulated reasoning; only its tail matters) is mid-line.
Covers a heading butting a heading (``**One****Two**``) and prose butting a
heading (``...interaction!**Next**``). Token-streamed reasoning is left
alone: its deltas carry their own whitespace, and a fragment that merely
opens emphasis (``**`` alone) is not a part boundary — summary parts always
carry the whole heading in one delta.
"""
if not previous or not delta:
return delta
if not delta.startswith("**"):
return delta
# Already separated — the provider (or an earlier part) ended the line.
if previous[-1].isspace():
return delta
# Require a *closed* heading. A token-streamed fragment that merely opens
# emphasis ("**" then "bold" then "**" across three deltas) is not a part
# boundary; a summary part always carries its whole heading in one delta.
if "**" not in delta[2:]:
return delta
return f"\n\n{delta}"

View File

@@ -1,53 +1,23 @@
"""Per-reasoning-model stale-timeout floor for known reasoning models.
"""Per-model stale-timeout FLOOR for known reasoning models.
Reasoning models (those that emit extended thinking blocks before their
first content token) routinely exceed Hermes's default chat-model
stale detectors:
Reasoning models (extended thinking before the first content token) routinely
exceed the default chat-model stale detectors (stream ``HERMES_STREAM_STALE_TIMEOUT``
180s, non-stream ``HERMES_API_CALL_STALE_TIMEOUT`` 90s): upstream proxies /
load-balancers idle-kill the stream mid-think, surfacing as
``BrokenPipeError``/``RemoteProtocolError`` on the next read. The existing
stale-detector scaling consults :func:`get_reasoning_stale_timeout_floor` and
applies ``max(default, floor)``. Being a floor it:
* Stream stale detector: ``HERMES_STREAM_STALE_TIMEOUT`` default 180s
``agent/chat_completion_helpers.py:2544``
* Non-stream stale detector: ``HERMES_API_CALL_STALE_TIMEOUT`` default 90s
``run_agent.py:1140``
* never overrides explicit user config (``providers.<id>.models.<model>.
stale_timeout_seconds`` / ``request_timeout_seconds`` win — this never runs
in that branch);
* never lowers an existing threshold;
* has zero effect on non-allowlisted models (resolver returns ``None``).
For NVIDIA Nemotron 3 Ultra on the hosted NIM gateway the empirical
upstream idle kill is ~120s (first-party reproduction at
NVIDIA/NemoClaw#4846 — TTFB ~31s, stream dies at 120s). The same
failure mode exists on OpenAI o1/o3, Anthropic Opus 4.x thinking,
DeepSeek R1, Qwen QwQ, xAI Grok reasoning — every cloud reasoning
model hits upstream-proxies / load-balancers with idle timeouts
shorter than the model's thinking phase. Result: the stale detector
kills the connection mid-think, surfacing as
``BrokenPipeError``/``RemoteProtocolError`` on the next read.
This module provides a floor that the existing stale-detector scaling
blocks consult via :func:`get_reasoning_stale_timeout_floor` and
apply as ``max(default, floor)``. It is a FLOOR:
* Never overrides explicit user config (``providers.<id>.models.<model>.stale_timeout_seconds``
or ``request_timeout_seconds`` already wins — this code never runs
in that branch).
* Never lowers an existing threshold.
* Has zero effect on non-reasoning models — they are not in the
allowlist and the resolver returns ``None``.
Matching uses start-anchored regex on the slug-only component of
the model name (after stripping any aggregator prefix like
``openai/``, ``x-ai/``, ``anthropic/``). The right-anchor matches
end-of-string or a ``-``/``.``/``_`` slug separator, so ``qwen3-235b``
matches the ``qwen3`` family entry (a future model slug would be
``qwen3-235b-instruct`` and would also match) but ``some-other-qwen3``
does NOT match ``qwen3`` (the ``-qwen3`` is not at start of slug).
The ``o1`` case is the most delicate: a model named
``llama-4-70b-o1-preview`` is a hypothetical community derivative that
should NOT trigger the reasoning-model floor for the user (the user
chose a non-OpenAI model, not a reasoning model). The start-of-slug
anchor naturally excludes this — the matched ``o1-preview`` is at
position 11 of the slug, not at position 0. The previous substring-
with-trailing-hyphen design would have over-matched here, which is
why start-of-slug anchoring is the right shape.
Fixes #52217.
Matching is start-anchored on the slug after any aggregator prefix
(``openai/``, ``x-ai/``) with an end-or-separator right anchor, so
``qwen3-235b`` matches ``qwen3`` but ``some-other-qwen3`` and a hypothetical
``llama-4-70b-o1-preview`` do not trigger the ``o1`` floor.
"""
from __future__ import annotations
@@ -56,41 +26,25 @@ import re
from typing import Optional
# (slug, floor_seconds). Each slug is matched as a discrete
# word-boundary component via the wrapper regex in ``_match_any``
# below. Order is irrelevant — the first regex match wins.
# (slug, floor_seconds). Order irrelevant — longest slug wins at match time.
_REASONING_STALE_TIMEOUT_FLOORS: tuple[tuple[str, int], ...] = (
# NVIDIA Nemotron — reasoning models behind hosted NIM with
# documented 60-180s upstream idle kill (NVIDIA/NemoClaw#4846:
# 120s measured).
# NVIDIA Nemotron behind hosted NIM: documented 60-180s upstream idle kill.
("nemotron-3-ultra", 600),
("nemotron-3-super", 600),
("nemotron-3-nano", 300),
("nemotron-3.5-lightning", 300),
# DeepSeek — R1 and V4 reasoning models on hosted NIM / DeepSeek direct.
# V4 series emits reasoning_content in a separate delta field before
# final content, requiring the same extended stale timeout floor.
# DeepSeek R1 / V4 (reasoning_content streamed before final content).
("deepseek-r1", 600),
("deepseek-reasoner", 600),
("deepseek-v4-flash", 600),
("deepseek-v4-pro", 600),
# Qwen — QwQ reasoning + Qwen3 thinking variants. QwQ-32B
# preview is the stable slug; ``qwen3`` covers the family of
# thinking-mode Qwen3 models (qwen3-235b-a22b, qwen3-32b, etc.)
# without over-matching every Qwen3 instruct variant — the
# right-anchor requires the slug to be at the start of the
# remaining model name, so ``qwen3-235b-instruct`` (instruct is
# NOT a thinking variant) would still match. Acceptable
# trade-off: instruct variants of qwen3 get the 180s floor
# even though they don't reason. The cost is a slightly longer
# wait on a hung provider; the alternative (matching only
# ``qwen3-.*-thinking``) breaks the moment NVIDIA or Alibaba
# ships a slightly different naming shape.
# Qwen QwQ + the qwen3 family. Instruct variants also match ``qwen3`` —
# accepted: a slightly longer wait on a hung provider beats a pattern
# (``qwen3-.*-thinking``) that breaks on the next naming shape.
("qwq-32b", 300),
("qwen3", 180),
# OpenAI o-series — known multi-minute TTFB. Each variant
# enumerated explicitly so bare ``o1`` doesn't over-match
# ``olmo-1`` or hypothetical future community derivatives.
# OpenAI o-series: each variant enumerated so bare ``o1`` cannot
# over-match ``olmo-1`` or community derivatives.
("o1", 600),
("o1-mini", 600),
("o1-pro", 600),
@@ -99,80 +53,36 @@ _REASONING_STALE_TIMEOUT_FLOORS: tuple[tuple[str, int], ...] = (
("o3-pro", 600),
("o3-mini", 300),
("o4-mini", 300),
# Anthropic Claude 4.x thinking variants. Anchored at
# ``claude-opus-4`` so non-thinking Claude 3.x or future
# non-reasoning Claude variants don't match.
# Anthropic Claude 4.x+ thinking variants (anchored so 3.x never matches).
("claude-opus-4", 240),
("claude-opus-5", 240),
("claude-sonnet-5", 180),
("claude-sonnet-4.5", 180),
("claude-sonnet-4.6", 180),
# Anthropic Mythos-class named reasoning models (claude-fable-5, …).
# 1M context + 128K output — heavier thinking phase than the
# numbered Claude line, so the floor is in the deep-reasoning tier
# alongside o1 / deepseek-r1 / nemotron-3-ultra. Without this
# entry the stale-stream detector kills fable-5's thinking phase
# at the default 180s (300s with context scaling), tripping the
# cross-turn circuit breaker after 5 consecutive stale kills.
# Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a
# heavier thinking phase than the numbered line — deep-reasoning tier,
# otherwise the stale detector trips the cross-turn circuit breaker.
("claude-fable", 600),
# xAI Grok reasoning variants. Explicit reasoning-only keys
# plus one for the ``non-reasoning`` variant so users picking
# the fast variant don't get the 300s floor. Bare ``grok-3``,
# ``grok-4`` etc. don't match — only the explicit reasoning /
# non-reasoning pairs.
# xAI Grok: explicit reasoning / non-reasoning pairs only, so bare
# ``grok-3``/``grok-4`` fast variants don't inherit the 300s floor.
("grok-4-fast-reasoning", 300),
("grok-4.20-reasoning", 300),
("grok-4.5", 300),
("grok-4.6", 300),
("grok-4-fast-non-reasoning", 180),
# "Ox Alpha" stealth reasoning model (stealth/ox-alpha on OpenRouter,
# x-preview-f-free on OpenCode Zen). Marketed as a reasoning model for
# long-horizon coding/agentic work; 1M context — same tier as the Grok
# reasoning variants.
# "Ox Alpha" stealth reasoning model (OpenRouter / OpenCode Zen slugs).
("ox-alpha", 300),
("x-preview-f-free", 300),
# Thinking Machines Inkling (thinkingmachines/inkling[-small][:free]
# on OpenRouter). Reasoning model (OpenRouter supported_parameters
# includes "reasoning"); 1M context — same tier as the Grok
# reasoning variants and Ox Alpha. "inkling" left-anchors on the
# slug after the aggregator prefix and the right anchor accepts the
# "-" separator, so inkling-small and the :free SKUs all match.
# Thinking Machines Inkling; covers inkling-small and :free SKUs.
("inkling", 300),
)
# Pre-compile each pattern. Wrapper = start-of-slug + slug + end-or-
# separator, where ``start-of-slug`` means start-of-string OR
# immediately after the last ``/`` (aggregator separator) and
# ``end-or-separator`` means end-of-string OR a ``-``/``.``/``_``.
#
# Why start-of-slug and not start-of-string: aggregator prefixes
# like ``openai/`` should not affect matching — the slug identity is
# the part after the last ``/``. Stripping the aggregator prefix in
# :func:`get_reasoning_stale_timeout_floor` before regex matching
# gives the wrapper a clean start-of-string anchor.
#
# Why end-or-separator on the right: ``openai/o3-mini`` must match
# the ``o3-mini`` slug (the right anchor is end-of-string). And
# ``openai/o3-mini-2025-01-31`` must also match ``o3-mini`` (the right
# anchor is the ``-`` separator). But ``openai/o3-mini-fork`` should
# NOT match ``o3-mini`` if we wanted to exclude forks — though the
# pattern ``o3-mini-fork`` would be matched as a derivative anyway,
# so we accept that community forks inheriting the same prefix are
# treated as reasoning models (a reasonable default — the upstream
# gateway timing is the same).
# Pre-compile all patterns at module load time to avoid per-call regex
# compilation and thread-safety issues with the mutable _PATTERN_CACHE.
# The list is built once at import and never mutated afterwards, so it is
# safe for free-threaded Python 3.13+ without any locking. The slug is kept
# in each entry for debuggability (log/inspection), even though _match_any
# only consumes floor + pattern.
# Pre-compiled once at import (immutable afterwards — safe under free-threaded
# Python). Right anchor: end-of-string or a slug separator; ``:`` is included
# because OpenRouter routing suffixes (``:free``, ``:nitro``) attach directly
# to the slug. Sorted longest-first so ``o3-mini`` beats ``o3``.
_SORTED_REASONING_FLOORS: list[tuple[str, float, re.Pattern[str]]] = [
# Right anchor: end-of-string or a slug separator. ``:`` is in the
# separator class because OpenRouter SKU/routing suffixes
# (``:free``, ``:batch``, ``:nitro``, ``:floor``) attach directly to
# the slug — ``thinkingmachines/inkling:free`` must match the
# ``inkling`` entry the same way ``inkling-small`` does.
(slug, floor, re.compile(r"^" + re.escape(slug) + r"(?:$|[\-._:])"))
for slug, floor in sorted(
_REASONING_STALE_TIMEOUT_FLOORS, key=lambda kv: -len(kv[0])
@@ -180,38 +90,14 @@ _SORTED_REASONING_FLOORS: list[tuple[str, float, re.Pattern[str]]] = [
]
def _match_any(model_lower: str) -> Optional[float]:
"""Return the floor for the first matching slug, else None.
Each table entry is matched as a start-of-slug prefix with the
slug-separator-or-end-of-string right-anchor. Table iteration
order is irrelevant: longest slug wins (so ``o3-mini`` beats
``o3`` on a model like ``openai/o3-mini``).
"""
for _slug, floor, pattern in _SORTED_REASONING_FLOORS:
if pattern.search(model_lower):
return float(floor)
return None
def get_reasoning_stale_timeout_floor(model: object) -> Optional[float]:
"""Return the stale-timeout floor (seconds) for a known reasoning model.
Returns ``None`` when the model is not in the allowlist or the
argument is empty / not a string. Matching uses
word-boundary-anchored regex on the lowercased model name, so
``openai/o3-mini`` matches the ``o3-mini`` slug but
``olmo-1`` does NOT match ``o1`` (the ``o1`` substring is not
at a word boundary inside ``olmo-1``).
Aggregator prefixes (``openai/``, ``x-ai/``, ``anthropic/`` etc.)
are preserved through matching — the ``/`` is itself a word
boundary, so ``openai/o3-mini`` matches ``o3-mini`` because the
``/`` before ``o3-mini`` satisfies the left-anchor alternation.
This is a FLOOR — callers must apply it as ``max(default, floor)``
and only when no explicit user-configured per-model
``stale_timeout_seconds`` exists.
``None`` when the model is not allowlisted or the argument is empty / not
a string. The aggregator prefix (everything up to the last ``/``) is
stripped so the slug is matched start-anchored. Callers apply this as
``max(default, floor)`` and only when no explicit per-model
``stale_timeout_seconds`` is configured.
>>> get_reasoning_stale_timeout_floor("nvidia/nemotron-3-ultra-550b-a55b")
600.0
@@ -243,9 +129,9 @@ def get_reasoning_stale_timeout_floor(model: object) -> Optional[float]:
name = model.strip().lower()
if not name:
return None
# Strip aggregator prefix (everything before and including the
# last ``/``). The wrapper regex anchors at start-of-string, so
# the slug identity is the bare model name.
if "/" in name:
name = name.rsplit("/", 1)[1]
return _match_any(name)
for _slug, floor, pattern in _SORTED_REASONING_FLOORS:
if pattern.search(name):
return float(floor)
return None

View File

@@ -1,19 +1,13 @@
"""Surface-agnostic core for the ``/subscription`` TUI screen.
Companion to :mod:`agent.billing_view` — same fail-open philosophy: when not
logged in or the portal is unreachable, return a struct with ``logged_in=False``
and let the surface degrade gracefully (never crash). Money is decimal end-to-end
(server emits decimal strings); we only format for display.
Companion to :mod:`agent.billing_view` — same fail-open philosophy (``logged_in=False``
when not logged in / portal unreachable; never crash) and decimal money end-to-end.
The TUI ``SubscriptionOverlay`` drives the plan change in-terminal (V3): it
previews the effect, then schedules a downgrade / cancellation / resume
(chargeless) or applies an upgrade (charges the card on the subscription). The
portal deep-link (built locally from ``portal_url`` + ``org_id``) remains the
fallback for an upgrade that needs 3DS / was declined.
WS1 dependency: ``GET /api/billing/subscription`` is a NAS endpoint (WS1 Phase A).
Until it ships, the fail-open contract handles 404s — the builder returns
``logged_in=False`` and the surface degrades gracefully.
The TUI ``SubscriptionOverlay`` drives the plan change in-terminal: preview, then
schedule a downgrade / cancellation / resume (chargeless) or apply an upgrade
(charges the subscription card). The portal deep-link (``portal_url`` + ``org_id``)
remains the fallback for an upgrade that needs 3DS / was declined. Until the NAS
``GET /api/billing/subscription`` endpoint ships, 404s take the fail-open path.
"""
from __future__ import annotations
@@ -24,24 +18,21 @@ from dataclasses import dataclass
from decimal import Decimal
from typing import Any, Optional
from agent.billing_view import parse_money
from agent.billing_view import OrgRoleCapability, fetch_portal_state, format_money, parse_money, parse_org_fields
logger = logging.getLogger(__name__)
# =============================================================================
# Parsed sub-structures
# =============================================================================
# ── Parsed sub-structures ────────────────────────────────────────────────────
@dataclass(frozen=True)
class CurrentSubscription:
"""The user's active subscription. ``None`` (not this object) = no plan.
When present, ``tier_id`` / ``tier_name`` / ``monthly_credits`` /
``cycle_ends_at`` are always set (NAS guarantees a present ``current`` is a
fully-populated plan). Only ``credits_remaining`` and the cancel/downgrade
fields are optional.
NAS guarantees a present ``current`` is fully populated: ``tier_id`` /
``tier_name`` / ``monthly_credits`` / ``cycle_ends_at`` are always set; only
``credits_remaining`` and the cancel/downgrade fields are optional.
"""
tier_id: Optional[str] = None
@@ -57,12 +48,11 @@ class CurrentSubscription:
@dataclass(frozen=True)
class SubscriptionTier:
"""A selectable plan in the catalog — one row of the in-terminal tier picker.
"""One row of the tier picker (mirrors NAS ``SubscriptionTierOption``).
Mirrors NAS's ``SubscriptionTierOption``. ``is_current`` marks the active plan
(shown but not selectable); ``is_enabled=False`` is a grandfathered tier the
user is on but that can no longer be selected. ``tier_order`` sorts the picker
and drives the upgrade-vs-downgrade direction hint.
``is_current`` = active plan (shown, not selectable); ``is_enabled=False`` = a
grandfathered tier the user is on but can no longer select. ``tier_order`` sorts
the picker and drives the upgrade-vs-downgrade hint.
"""
tier_id: str
@@ -76,13 +66,11 @@ class SubscriptionTier:
@dataclass(frozen=True)
class SubscriptionChangePreview:
"""Parsed ``POST /api/billing/subscription/preview`` — what a change would do.
"""Parsed ``POST /api/billing/subscription/preview``.
``effect`` is the disposition the commit would take:
- ``charge_now`` → an upgrade; ``amount_due_now_cents`` is the prorated charge.
- ``scheduled`` → a downgrade / same-price change at ``effective_at`` (period end).
- ``no_op`` → already on the target tier.
- ``blocked`` → the commit would be refused; ``reason`` says why.
``effect``: ``charge_now`` (upgrade; ``amount_due_now_cents`` is the prorated
charge) · ``scheduled`` (downgrade / same-price change at ``effective_at``) ·
``no_op`` (already on target) · ``blocked`` (commit refused; ``reason`` says why).
"""
effect: str
@@ -97,12 +85,9 @@ class SubscriptionChangePreview:
@dataclass(frozen=True)
class SubscriptionState:
"""Parsed ``GET /api/billing/subscription`` — the overview screen's data.
Fail-open: ``logged_in=False`` (and empty fields) when not logged in or the
portal is unreachable.
"""
class SubscriptionState(OrgRoleCapability):
"""Parsed ``GET /api/billing/subscription``. Fail-open: ``logged_in=False``
(empty fields) when not logged in or the portal is unreachable."""
logged_in: bool
org_name: Optional[str] = None
@@ -113,35 +98,15 @@ class SubscriptionState:
current: Optional[CurrentSubscription] = None
tiers: tuple[SubscriptionTier, ...] = () # selectable catalog (picker)
portal_url: Optional[str] = None
# When the fetch failed (vs cleanly not-logged-in), the message for the surface.
error: Optional[str] = None
@property
def is_admin(self) -> bool:
"""Deprecated/display only — a legacy OWNER/ADMIN check.
NOT a capability check; use :attr:`can_change_plan` for gating billing
plan-change actions.
"""
return (self.role or "").upper() in ("OWNER", "ADMIN")
@property
def can_change_plan(self) -> bool:
"""Server capability when supplied; otherwise the legacy role fallback."""
if self.can_change_plan_raw is not None:
return self.can_change_plan_raw
return self.is_admin
error: Optional[str] = None # set when the fetch failed (vs cleanly not-logged-in)
# =============================================================================
# Payload parsing
# =============================================================================
# ── Payload parsing ──────────────────────────────────────────────────────────
def _parse_current(raw: Any) -> Optional[CurrentSubscription]:
# "No plan" is wire-represented as current:null (free personal OR team) —
# the old all-null-object shape is gone. A present current is a real plan,
# so guard on a real tier id and return None otherwise.
# "No plan" is wire-represented as current:null; a present current is a real
# plan, so guard on a real tier id and return None otherwise.
if not isinstance(raw, dict):
return None
tier_id = raw.get("tierId") or raw.get("id")
@@ -161,11 +126,8 @@ def _parse_current(raw: Any) -> Optional[CurrentSubscription]:
def _coalesce(*vals: Any) -> Any:
"""First non-``None`` value (preserves a legit ``0``/``0.0``, unlike ``or``).
NAS sends ``0`` for the free tier's ``tierOrder`` / ``dollarsPerMonth``; a plain
``x or default`` would drop those, so coalesce on ``None`` specifically.
"""
"""First non-``None`` value. NAS sends ``0`` for the free tier's ``tierOrder`` /
``dollarsPerMonth``, which a plain ``x or default`` would drop."""
for v in vals:
if v is not None:
return v
@@ -197,8 +159,7 @@ def subscription_change_preview_from_payload(
effect = payload.get("effect")
cents = payload.get("amountDueNowCents")
return SubscriptionChangePreview(
# An unrecognized/missing effect is treated as ``blocked`` — fail safe, never
# charge on a malformed quote.
# Unrecognized/missing effect → ``blocked``: fail safe, never charge on a malformed quote.
effect=effect if isinstance(effect, str) else "blocked",
reason=payload.get("reason") or None,
current_tier_id=payload.get("currentTierId"),
@@ -215,89 +176,49 @@ def subscription_state_from_payload(
payload: dict[str, Any], *, portal_url: Optional[str] = None
) -> SubscriptionState:
"""Map a raw ``/api/billing/subscription`` JSON dict into :class:`SubscriptionState`."""
raw_org = payload.get("org")
org: dict[str, Any] = raw_org if isinstance(raw_org, dict) else {}
org, can_change_plan_raw = parse_org_fields(payload)
raw_context = payload.get("context")
context = raw_context if raw_context in ("personal", "team") else "personal"
raw_tiers = payload.get("tiers")
tiers = (
tuple(t for t in (_parse_tier(x) for x in raw_tiers) if t is not None)
if isinstance(raw_tiers, list)
else ()
)
return SubscriptionState(
logged_in=True,
org_name=org.get("name"),
org_id=org.get("id") or None,
role=org.get("role"),
can_change_plan_raw=(
payload.get("canChangePlan")
if isinstance(payload.get("canChangePlan"), bool)
else None
),
context=context,
can_change_plan_raw=can_change_plan_raw,
context=raw_context if raw_context in ("personal", "team") else "personal",
current=_parse_current(payload.get("current")),
tiers=tiers,
portal_url=portal_url,
)
# =============================================================================
# Fail-open builders (the surface front doors)
# =============================================================================
# ── Fail-open builders (the surface front doors) ─────────────────────────────
def build_subscription_state(*, timeout: float = 15.0) -> SubscriptionState:
"""Fetch + parse ``GET /api/billing/subscription``. Fail-open.
"""Fetch + parse ``GET /api/billing/subscription``. Fail-open (see
:func:`agent.billing_view.fetch_portal_state`).
Returns ``SubscriptionState(logged_in=False)`` when not logged in. On a
portal/HTTP failure, returns ``logged_in=False`` with ``error`` set so the
surface can show a clear message rather than crashing.
Dev override: when ``HERMES_DEV_SUBSCRIPTION_FIXTURE`` names a fixture state,
``/subscription`` renders from that fixture instead of the real portal — so
every plan/cancel/downgrade/team/not-admin state is testable on both
the CLI and TUI without a live account. Throwaway scaffolding; see
:func:`dev_fixture_subscription_state`.
``HERMES_DEV_SUBSCRIPTION_FIXTURE`` short-circuits to a fixture so every
plan/cancel/downgrade/team/not-admin state is testable on CLI and TUI offline.
"""
fixture = dev_fixture_subscription_state()
if fixture is not None:
return fixture
try:
from hermes_cli.nous_billing import (
BillingAuthError,
BillingError,
_absolutize_portal_url,
get_subscription_state,
resolve_portal_base_url,
)
except Exception:
return SubscriptionState(logged_in=False, error="billing client unavailable")
try:
payload = get_subscription_state(timeout=timeout)
except BillingAuthError:
return SubscriptionState(logged_in=False)
except BillingError as exc:
logger.debug("subscription ▸ /state fetch failed (fail-open)", exc_info=True)
return SubscriptionState(logged_in=False, error=str(exc))
except Exception:
logger.debug("subscription ▸ /state unexpected error (fail-open)", exc_info=True)
return SubscriptionState(logged_in=False, error="could not load subscription state")
raw_portal = payload.get("portalUrl") if isinstance(payload, dict) else None
portal_url = _absolutize_portal_url(raw_portal) if raw_portal else None
if not portal_url:
try:
portal_url = resolve_portal_base_url()
except Exception:
portal_url = None
return subscription_state_from_payload(payload, portal_url=portal_url)
return fetch_portal_state(
"get_subscription_state",
"subscription",
failed=lambda **kw: SubscriptionState(logged_in=False, **kw),
parse=lambda payload, portal_url: subscription_state_from_payload(payload, portal_url=portal_url),
portal_fallback=lambda base: base,
timeout=timeout,
log=logger,
)
def subscription_manage_url(
@@ -305,33 +226,24 @@ def subscription_manage_url(
) -> Optional[str]:
"""Build ``{portal_origin}/manage-subscription?org_id=<id>[&plan=<tier_id>]``.
Mirrors the TUI's ``buildManageUrl`` (``subscription.ts``): the deep-link
target is NAS's OWN ``/manage-subscription`` page (NOT the Stripe Billing
Portal — decided Jun 23), which routes upgrade→Checkout / downgrade→scheduled
internally. ``org_id`` pins the page to the right account in multi-org
situations. Returns ``None`` when no portal URL is resolvable.
``tier_id`` (the stable ``tiers[]`` id, never a name/slug) is appended as
``plan=`` so the portal preselects the picked plan — only for a NEW
subscription / upgrade the user chose. The portal validates it and simply
ignores an unknown tier, so the CLI appends unconditionally when a tier was
picked (parity with the TUI's ``?plan=``).
Mirrors the TUI's ``buildManageUrl``: the target is NAS's OWN ``/manage-subscription``
page (NOT the Stripe Billing Portal), which routes upgrade→Checkout /
downgrade→scheduled internally. ``org_id`` pins the right account in multi-org
situations. ``tier_id`` (the stable ``tiers[]`` id, never a name/slug) preselects
the picked plan; the portal ignores an unknown tier, so it's appended
unconditionally when picked. None when no portal URL is resolvable.
"""
from urllib.parse import urlencode, urlsplit, urlunsplit
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
if not state.portal_url:
return None
try:
parts = urlsplit(state.portal_url)
except Exception:
return None
if parts.scheme not in ("http", "https") or not parts.netloc:
return None
from urllib.parse import parse_qsl
# Preserve unrelated portal query params; org_id / plan are contract-owned
# (org_id before plan — insertion order is the emitted query order).
params = dict(parse_qsl(parts.query, keep_blank_values=True))
@@ -341,68 +253,34 @@ def subscription_manage_url(
params["org_id"] = state.org_id
if tier_id:
params["plan"] = tier_id
query = urlencode(params)
return urlunsplit((parts.scheme, parts.netloc, "/manage-subscription", query, ""))
return urlunsplit((parts.scheme, parts.netloc, "/manage-subscription", urlencode(params), ""))
# =============================================================================
# Shared plan-catalog helpers (consumed by the CLI Free catalog + paid picker)
# =============================================================================
def _format_dollars_grouped(value: Optional[Decimal]) -> str:
"""``$1,000`` / ``$1,234.50`` — the whole-vs-fractional rule of
``billing_view.format_money`` but thousands-grouped, matching the TUI's
``toLocaleString('en-US')``.
The shared ``format_money`` is intentionally ungrouped (and asserted so across
other surfaces), so plan-catalog rows group locally to mirror the TUI.
"""
if value is None:
return "—"
if value == value.to_integral_value():
return f"${format(value.to_integral_value(), ',f')}"
return f"${format(value.quantize(Decimal('0.01')), ',f')}"
# ── Shared plan-catalog helpers (CLI Free catalog + paid picker) ─────────────
def selectable_tiers(state: SubscriptionState) -> list[SubscriptionTier]:
"""Enabled paid tiers other than the current plan, cheapest first.
One derivation shared by the CLI Free catalog and the paid change picker:
``is_enabled and not is_current and tier_order > 0`` (free / no-sub excluded —
dropping to free is a cancellation), sorted by ``tier_order``.
"""
"""Enabled paid tiers other than the current plan, cheapest first (``tier_order > 0``
— dropping to free is a cancellation, not a plan pick)."""
return sorted(
(
t
for t in (state.tiers or ())
if t.is_enabled and not t.is_current and (t.tier_order or 0) > 0
),
(t for t in (state.tiers or ()) if t.is_enabled and not t.is_current and (t.tier_order or 0) > 0),
key=lambda t: t.tier_order or 0,
)
def format_tier_row(tier: SubscriptionTier) -> str:
"""``name · $X/mo[ · $Y credits/mo]`` — the shared plan-catalog row.
Mirrors the TUI Free rows (``subscriptionOverlay.tsx``): thousands-grouped
money, and the ``$Y credits/mo`` suffix ONLY when monthly credits are present
and > 0 (a ``None`` / zero-credits tier hides it — never ``· — credits/mo`` or
``· $0 credits/mo``).
"""
row = f"{tier.name} · {_format_dollars_grouped(tier.dollars_per_month)}/mo"
"""``name · $X/mo[ · $Y credits/mo]`` — thousands-grouped money (mirrors the TUI
Free rows); the credits suffix appears ONLY when monthly credits are present and > 0."""
row = f"{tier.name} · {format_money(tier.dollars_per_month, grouped=True)}/mo"
mc = tier.monthly_credits
if mc is not None and mc > 0:
row += f" · {_format_dollars_grouped(mc)} credits/mo"
row += f" · {format_money(mc, grouped=True)} credits/mo"
return row
def is_upgrade(state: SubscriptionState, tier_id: str) -> bool:
"""True when ``tier_id`` ranks above the current plan by ``tier_order``.
Prefers the active subscription's tier; falls back to the ``tiers[]``
``is_current`` marker (what the picker derives from), else 0 (free).
"""
"""True when ``tier_id`` ranks above the current plan by ``tier_order``. Prefers the
active subscription's tier; falls back to the ``tiers[]`` ``is_current`` marker, else 0."""
orders = {t.tier_id: (t.tier_order or 0) for t in (state.tiers or ())}
cur_id = state.current.tier_id if state.current else None
if cur_id is not None and cur_id in orders:
@@ -412,96 +290,63 @@ def is_upgrade(state: SubscriptionState, tier_id: str) -> bool:
return orders.get(tier_id, 0) > cur_order
# =============================================================================
# Dev fixtures (throwaway scaffolding — env-var driven, no live portal)
# =============================================================================
# ── Dev fixtures (throwaway scaffolding — env-var driven, no live portal) ────
_DEV_FIXTURE_PORTAL = "https://portal.nousresearch.com/billing"
def _dev_current(**over: Any) -> CurrentSubscription:
base: dict[str, Any] = dict(
tier_id="plus",
tier_name="Plus",
monthly_credits=Decimal("1000"),
credits_remaining=Decimal("420"),
cycle_ends_at="2026-07-01",
tier_id="plus", tier_name="Plus", monthly_credits=Decimal("1000"),
credits_remaining=Decimal("420"), cycle_ends_at="2026-07-01",
)
base.update(over)
return CurrentSubscription(**base)
return CurrentSubscription(**{**base, **over})
def _dev_tiers(current_id: Optional[str]) -> tuple[SubscriptionTier, ...]:
"""A sample plan catalog for fixtures (marks ``current_id`` as the active tier)."""
specs = (
("free", "Free", 0, "0", "0"),
("plus", "Plus", 1, "20", "1000"),
("super", "Super", 2, "40", "3000"),
("ultra", "Ultra", 3, "80", "7000"),
)
specs = (("free", "Free", 0, "0", "0"), ("plus", "Plus", 1, "20", "1000"),
("super", "Super", 2, "40", "3000"), ("ultra", "Ultra", 3, "80", "7000"))
return tuple(
SubscriptionTier(
tier_id=tid,
name=name,
tier_order=order,
dollars_per_month=parse_money(dpm),
monthly_credits=parse_money(mc),
is_current=(tid == current_id),
is_enabled=True,
tier_id=tid, name=name, tier_order=order, dollars_per_month=parse_money(dpm),
monthly_credits=parse_money(mc), is_current=(tid == current_id), is_enabled=True,
)
for tid, name, order, dpm, mc in specs
)
def dev_fixture_subscription_state() -> Optional[SubscriptionState]:
"""Return a fixture :class:`SubscriptionState` for ``HERMES_DEV_SUBSCRIPTION_FIXTURE``.
"""``HERMES_DEV_SUBSCRIPTION_FIXTURE`` -> fixture :class:`SubscriptionState`; None when unset.
Lets every CLI/TUI subscription state be exercised without a live portal:
free | mid | top | not-admin | downgrade | cancel | team |
logged-out
Returns ``None`` when the env var is unset/empty (the real portal path runs).
Throwaway scaffolding — mirrors ``HERMES_DEV_CREDITS_FIXTURE``.
``free | mid | top | not-admin | downgrade | cancel | team | logged-out``. Unknown
name → logged-out with ``error`` so the misconfiguration is visible.
"""
name = (os.getenv("HERMES_DEV_SUBSCRIPTION_FIXTURE") or "").strip().lower()
if not name:
return None
common = dict(org_name="Acme Inc", org_id="org_acme", role="OWNER", portal_url=_DEV_FIXTURE_PORTAL)
if name in ("logged-out", "logged_out", "loggedout"):
name = {"logged_out": "logged-out", "loggedout": "logged-out", "mid-tier": "mid",
"top-tier": "top", "member": "not-admin"}.get(name, name)
if name == "logged-out":
return SubscriptionState(logged_in=False)
if name == "free":
return SubscriptionState(logged_in=True, current=None, tiers=_dev_tiers(None), **common)
if name in ("mid", "mid-tier"):
return SubscriptionState(logged_in=True, current=_dev_current(), tiers=_dev_tiers("plus"), **common)
if name in ("top", "top-tier"):
return SubscriptionState(
logged_in=True,
common = dict(logged_in=True, org_name="Acme Inc", org_id="org_acme", role="OWNER", portal_url=_DEV_FIXTURE_PORTAL)
plus = dict(current=_dev_current(), tiers=_dev_tiers("plus"))
states: dict[str, dict[str, Any]] = {
"free": dict(current=None, tiers=_dev_tiers(None)),
"mid": plus,
"top": dict(
current=_dev_current(tier_id="ultra", tier_name="Ultra", monthly_credits=Decimal("7000"), credits_remaining=Decimal("5000")),
tiers=_dev_tiers("ultra"),
**common,
)
if name in ("not-admin", "member"):
return SubscriptionState(logged_in=True, current=_dev_current(), tiers=_dev_tiers("plus"), **{**common, "role": "MEMBER"})
if name == "downgrade":
return SubscriptionState(
logged_in=True,
),
"not-admin": {**plus, "role": "MEMBER"},
"downgrade": dict(
current=_dev_current(tier_id="super", tier_name="Super", monthly_credits=Decimal("3000"), credits_remaining=Decimal("1500"), pending_downgrade_tier_name="Plus", pending_downgrade_at="2026-07-15"),
tiers=_dev_tiers("super"),
**common,
)
if name == "cancel":
return SubscriptionState(
logged_in=True,
current=_dev_current(cancel_at_period_end=True, cancellation_effective_at="2026-07-01"),
tiers=_dev_tiers("plus"),
**common,
)
if name == "team":
return SubscriptionState(logged_in=True, context="team", current=None, org_name="Acme Engineering", org_id="org_eng", role="OWNER", portal_url=_DEV_FIXTURE_PORTAL)
# Unknown name → behave as logged-out so the misconfiguration is visible.
return SubscriptionState(logged_in=False, error=f"unknown HERMES_DEV_SUBSCRIPTION_FIXTURE: {name}")
),
"cancel": dict(current=_dev_current(cancel_at_period_end=True, cancellation_effective_at="2026-07-01"), tiers=_dev_tiers("plus")),
"team": dict(context="team", current=None, org_name="Acme Engineering", org_id="org_eng"),
}
if name not in states:
return SubscriptionState(logged_in=False, error=f"unknown HERMES_DEV_SUBSCRIPTION_FIXTURE: {name}")
return SubscriptionState(**{**common, **states[name]})

View File

@@ -1,30 +1,11 @@
"""Thinking-timeout detection and user-facing guidance for reasoning models.
When a known reasoning model (NVIDIA Nemotron 3 Ultra, OpenAI o1/o3,
Anthropic Opus 4.x thinking, DeepSeek R1, Qwen QwQ, xAI Grok reasoning)
hits a transport-layer error before the first content token arrives, the
upstream proxy has almost certainly idle-killed a long thinking stream —
not a true context overflow or a configuration error. The user needs
distinct guidance for this case:
"The model's thinking phase exceeded the upstream proxy's idle
timeout before the first content token arrived. This is a known
issue with reasoning models behind cloud gateways (NVIDIA NIM,
OpenAI, Anthropic, DeepSeek). Workarounds in priority order:
1. Set `providers.<provider>.models.<model>.stale_timeout_seconds: 900`
in `~/.hermes/config.yaml` to extend the per-call timeout...
2. Lower `reasoning_budget` or set `reasoning_effort: medium`...
3. Use a smaller / faster reasoning model..."
The existing `_is_stream_drop` guidance at
``agent/conversation_loop.py:3464-3486`` fires for large-file-write
stream drops ("try execute_code with Python's open() for large files")
which is the WRONG advice for the thinking-timeout case. This module
provides the detection and the message as standalone helpers so the
detection logic is unit-testable without driving the full retry loop,
and the message text can be regression-tested for spelling and accuracy.
Part 2 of Fixes #52310.
When a known reasoning model hits a transport-layer error before the first
content token, the upstream proxy has almost certainly idle-killed a long
thinking stream — not a context overflow or configuration error. The generic
stream-drop guidance in conversation_loop ("use execute_code for large files")
is wrong for that case, so detection and message live here as standalone,
unit-testable helpers.
"""
from __future__ import annotations
@@ -32,12 +13,9 @@ from __future__ import annotations
from typing import Optional
# Substring set that identifies a transport-layer failure on the
# response stream. Same shape as the existing
# ``_SERVER_DISCONNECT_PATTERNS`` in ``agent/error_classifier.py:394``
# but extended to also catch the OSS-level error signature
# (``broken pipe`` / ``errno 32``) that the upstream kill surfaces
# to the OpenAI SDK wrapper.
# Transport-layer failure signatures on the response stream — the classifier's
# server-disconnect set plus the OS-level ``broken pipe`` / ``errno 32`` the
# upstream kill surfaces through the OpenAI SDK wrapper.
_THINKING_TIMEOUT_SUBSTRINGS: tuple[str, ...] = (
"broken pipe",
"errno 32",
@@ -50,53 +28,22 @@ _THINKING_TIMEOUT_SUBSTRINGS: tuple[str, ...] = (
def is_thinking_timeout(classified: object, model: str, error_msg: str) -> bool:
"""Return True when a reasoning model's thinking phase hit a transport kill.
"""True when a reasoning model's thinking phase hit a transport kill.
Args:
classified: a :class:`agent.error_classifier.ClassifiedError` instance
(duck-typed here to avoid an import cycle in unit tests).
model: the model slug at failure time (e.g.
``"nvidia/nemotron-3-ultra-550b-a55b"``).
error_msg: lowercased string representation of the underlying
exception (typically ``str(api_error).lower()``).
Returns True when ALL conditions hold:
1. ``classified.reason == FailoverReason.timeout`` (the classifier
override at ``agent/error_classifier.py:720-738`` ensures this
is the case for reasoning models even on large sessions).
2. ``api_error`` has no ``.status_code`` attribute set (transport
disconnect, not an HTTP error).
3. ``model`` is in the reasoning-model allowlist (reuses
``agent.reasoning_timeouts.get_reasoning_stale_timeout_floor``).
4. ``error_msg`` contains one of the transport-kill substrings.
Non-reasoning models always return False. Non-transport errors
(billing / rate_limit / auth / context_overflow / format_error)
always return False. HTTP-status errors always return False.
All must hold: ``classified.reason`` is the ``timeout`` FailoverReason
(duck-typed via ``.value`` to avoid importing error_classifier), ``model``
is in the reasoning allowlist (``reasoning_timeouts``), and ``error_msg``
carries a transport-kill substring. The caller gates on the error having no
HTTP ``status_code`` before calling. Non-reasoning models and non-transport
errors (billing / rate_limit / auth / context_overflow) return False.
"""
# Import here (not at module top) to keep this helper cheap to
# import even from callers that don't need it. ``agent.reasoning_timeouts``
# is small and dependency-free.
from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor
# Condition 1: classifier says timeout. Use a string/value check
# rather than importing FailoverReason so this module has zero
# import cycles from the error_classifier package.
reason = getattr(classified, "reason", None)
reason_value = getattr(reason, "value", None)
if reason_value != "timeout":
if getattr(reason, "value", None) != "timeout":
return False
# Condition 2: no HTTP status code (transport, not API error).
# Caller is expected to gate on ``getattr(api_error, "status_code", None) is None``
# before calling this helper; the surface here is just the post-gate
# boolean so the caller can pass an already-prepped error_msg.
# Condition 3: reasoning model allowlist.
if get_reasoning_stale_timeout_floor(model) is None:
return False
# Condition 4: transport-kill substring in the error message.
error_msg_lower = (error_msg or "").lower()
return any(p in error_msg_lower for p in _THINKING_TIMEOUT_SUBSTRINGS)
@@ -104,18 +51,11 @@ def is_thinking_timeout(classified: object, model: str, error_msg: str) -> bool:
def build_thinking_timeout_guidance(
provider: str, model: str, model_label: Optional[str] = None,
) -> str:
"""Return the user-facing guidance string appended to ``_final_response``.
"""User-facing guidance appended to the final response.
Args:
provider: provider slug (e.g. ``"nvidia"``, ``"openai"``).
model: bare model slug the user would put in their config
(e.g. ``"nemotron-3-ultra-550b-a55b"`` if the user uses
NVIDIA direct, or the full ``"nvidia/nemotron-3-ultra-550b-a55b"``
if they go through an aggregator). Used verbatim in the
config snippet so the user can copy-paste.
model_label: optional short label for the model name in the
prose (e.g. ``"Nemotron 3 Ultra"``). Falls back to the
slug if not provided.
``model`` is used verbatim in the config snippet so it is copy-pasteable
(bare slug for direct providers, ``vendor/slug`` through aggregators);
``model_label`` is the optional prose name, defaulting to the slug.
"""
label = model_label or model
return (

File diff suppressed because it is too large Load Diff

View File

@@ -16,7 +16,7 @@ supported vocabulary. The policy under test:
import pytest
from agent.reasoning_effort import (
CODEX_RESPONSES_EFFORTS,
CODEX_GPT56_EFFORTS,
EFFORT_LADDER,
GLM52_EFFORTS,
GLM52_OVERRIDES,
@@ -137,8 +137,8 @@ class TestGlm52Vocabulary:
class TestCodexVocabulary:
def test_minimal_and_ultra(self):
assert clamp_effort("minimal", CODEX_RESPONSES_EFFORTS) == "low"
assert clamp_effort("ultra", CODEX_RESPONSES_EFFORTS) == "max"
assert clamp_effort("minimal", CODEX_GPT56_EFFORTS) == "low"
assert clamp_effort("ultra", CODEX_GPT56_EFFORTS) == "max"
def test_per_model_max_support(self):
"""Live-verified (Aug 2026, #68365): 'max' is gpt-5.6-only — gpt-5.5