diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index 5c7f232dc5..bc71be7eb2 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -1,26 +1,13 @@ """Canonical reasoning-effort vocabulary and wire clamping. -Hermes' internal effort ladder (``hermes_constants.VALID_REASONING_EFFORTS`` -plus ``none``) is wider than any single provider wire accepts. Hand-rolled -per-transport translation maps produced two recurring bugs: a new internal -level (``ultra``) leaking to a wire that 400s on it, and an unknown level -dropped to a weak default so the strongest ask resolved *weaker* than an -explicit ``high`` (ladder inversion). This module is the single source of -truth instead: - -- :data:`EFFORT_LADDER` — canonical low→high ordering. -- :func:`clamp_effort` — keep a supported level verbatim, else the **nearest - weaker** supported level (never silently escalate cost); only when nothing - weaker exists take the weakest supported level (GLM-5.2's floor is ``high``). -- Named wire-vocabulary constants so call sites declare *data*, not logic. - -Rules for call sites: -1. Wire shape (``extra_body.reasoning`` vs top-level ``reasoning_effort`` vs - a ``thinking`` toggle) stays local; only the vocabulary math lives here. -2. Unset stays unset: ``clamp_effort`` translates an explicit request, never - invents one — omit the field so the server default applies. -3. Never patch a predicate: when a provider rejects a level, fix its declared - supported set (data), not the call site. +Hermes' internal effort ladder (``VALID_REASONING_EFFORTS`` plus ``none``) is wider than any +single provider wire accepts; hand-rolled per-transport maps leaked new levels (``ultra``) to +wires that 400 and inverted the ladder (unknown → weak default). Single source of truth: +:data:`EFFORT_LADDER` (low→high), :func:`clamp_effort` (verbatim if supported, else the +nearest WEAKER level; only when nothing weaker exists the weakest supported), and named +wire-vocabulary constants so call sites declare data. Rules: wire shape stays local, only the +vocabulary math lives here; unset stays unset (never invent an effort); when a provider +rejects a level fix its declared set, never a predicate. """ from __future__ import annotations @@ -28,40 +15,21 @@ from __future__ import annotations import re from typing import Optional, Sequence -#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``) -#: without matching K2-era names (``kimi-k2.6``). +#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``). _KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)") -# Canonical low→high ordering for nearest-level clamping. Includes "none" so an -# explicit disable can be clamped when a provider publishes it as a level. -EFFORT_LADDER: tuple[str, ...] = ( - "none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", -) - -# ``ultra`` is Hermes-internal (the Codex product tier); no wire accepts it, so -# every declared set below stops at ``max`` and ``ultra`` always clamps down. +# Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit +# disable can be clamped when a provider publishes it as a level. ``ultra`` is Hermes-internal +# (the Codex product tier): no wire accepts it, every declared set stops at ``max``. +EFFORT_LADDER: tuple[str, ...] = ("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra") #: Widest OpenAI-compatible wire vocabulary (OpenRouter, Nous Portal). -OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ( - "none", "minimal", "low", "medium", "high", "xhigh", "max", -) - -#: OpenAI/Codex Responses, per model generation (live-verified): ``minimal`` -#: is rejected by both (clamps to low); ``max`` is gpt-5.6-only. -CODEX_GPT56_EFFORTS: tuple[str, ...] = ( - "none", "low", "medium", "high", "xhigh", "max", -) -CODEX_LEGACY_EFFORTS: tuple[str, ...] = ( - "none", "low", "medium", "high", "xhigh", -) - - -def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]: - """Supported effort set for an OpenAI/Codex Responses model.""" - if "gpt-5.6" in (model or "").lower(): - return CODEX_GPT56_EFFORTS - return CODEX_LEGACY_EFFORTS +OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ("none", "minimal", "low", "medium", "high", "xhigh", "max") +#: OpenAI/Codex Responses per model generation (live-verified): ``minimal`` is rejected by +#: both (clamps to low); ``max`` is gpt-5.6-only. +CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh", "max") +CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh") #: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high. XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh") @@ -70,33 +38,27 @@ XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high") #: Actual Computer relays (SGLang/vLLM). ACTUAL_RELAY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max") -#: Moonshot/Kimi K3 (server default high) vs K2-era models. +#: Moonshot/Kimi K3 (server default high) vs K2-era models. K3 quirks: ``high`` is K3's +#: positional middle AND server default, so ``medium`` rounds to it rather than down to +#: ``low``; ``xhigh`` rounds up to ``max`` (K3's top tier). KIMI_K3_EFFORTS: tuple[str, ...] = ("low", "high", "max") KIMI_K2_EFFORTS: tuple[str, ...] = ("low", "medium", "high") +KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"} -#: OpenCode "Ox Alpha" (x-preview-f-free): thinking cannot be disabled and the -#: wire accepts exactly low/high/max (medium/none/xhigh 400); xhigh rounds up. +#: OpenCode "Ox Alpha" (x-preview-f-free): thinking cannot be disabled and the wire accepts +#: exactly low/high/max (medium/none/xhigh 400); xhigh rounds up. OX_ALPHA_EFFORTS: tuple[str, ...] = ("low", "high", "max") OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"} -#: Tencent TokenHub. +#: Tencent TokenHub / Nebius Token Factory / Upstage Solar: plain three-level knobs. TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high") - -#: Nebius Token Factory (top-level reasoning_effort knob). NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high") +SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high") -#: Kimi K3 vendor-documented quirks: ``high`` is K3's positional middle AND -#: server default, so ``medium`` rounds to it rather than down to ``low``; -#: ``xhigh`` rounds up to ``max`` (K3's top tier). -KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"} - -#: GLM-5.2 native knob: exactly ``high`` (its minimum thinking level) and -#: ``max``; ``xhigh`` requests the top tier, not the floor. +#: GLM-5.2 native knob: exactly ``high`` (its minimum thinking level) and ``max``; GLM-5.3 +#: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier. GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} - -#: GLM-5.3 widens the knob to a graded scale (live-verified, monotonic -#: reasoning-token scaling); ``xhigh`` requests the top tier. GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} @@ -111,21 +73,16 @@ OLLAMA_CLOUD_OVERRIDES: dict[str, str] = {"xhigh": "max"} #: Meta Model API (Muse): rejects ``none``. META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh") -#: Upstage Solar Pro/Open. -SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high") + +def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]: + """Supported effort set for an OpenAI/Codex Responses model.""" + return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]: - """Supported effort set for a Moonshot/Kimi slug. - - K3 is served as bare ``k3``, plan variants (``k3-256k``) and ``kimi-k3*`` - aliases; everything earlier speaks low/medium/high. Boundary-matched so - K2-era names (``kimi-k2.6``) never match. - """ + """Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3).""" m = (model or "").strip().lower().split("/")[-1] - if _KIMI_K3_SLUG_RE.search(m): - return KIMI_K3_EFFORTS - return KIMI_K2_EFFORTS + return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS def clamp_effort( @@ -135,55 +92,37 @@ def clamp_effort( ) -> Optional[str]: """Clamp a requested reasoning effort onto a wire's supported levels. - ``overrides`` (a declared vendor mapping, e.g. Kimi K3 ``medium → high``) - is consulted first. Otherwise the request passes through unchanged when it - is supported, when the supported set is unknown/empty, or when it isn't a - recognized ladder level (custom providers may use bespoke names). Else the - **nearest weaker** supported level is returned so a clamp never escalates - cost; when nothing weaker exists, the weakest supported level is (the - provider's floor is the closest honest match). Monotonic: a stronger - request never resolves weaker than a weaker request would. + ``overrides`` (a declared vendor mapping, e.g. Kimi K3 ``medium → high``) is consulted + first. Otherwise the request passes through unchanged when it is supported, when the + supported set is unknown/empty, or when it isn't a recognized ladder level (custom + providers may use bespoke names). Else the **nearest weaker** supported level is returned + so a clamp never escalates cost; when nothing weaker exists, the weakest supported level + (the provider's floor is the closest honest match). Monotonic: a stronger request never + resolves weaker than a weaker request would. """ requested = str(effort or "").strip().lower() if not requested or not supported: return effort - supported_norm = [ - str(level).strip().lower() - for level in supported - if str(level).strip().lower() in EFFORT_LADDER - ] + supported_norm = [lvl for lvl in (str(s).strip().lower() for s in supported) if lvl in EFFORT_LADDER] if not supported_norm or requested in supported_norm: return effort - if overrides: - mapped = overrides.get(requested) - if mapped in supported_norm: - return mapped + if overrides and overrides.get(requested) in supported_norm: + return overrides[requested] if requested not in EFFORT_LADDER: return effort - # "none" disables reasoning — never a degradation target for an enabled - # ask (clamping "minimal" to "none" would silently switch thinking off). + # "none" disables reasoning — never a degradation target for an enabled ask + # (clamping "minimal" to "none" would silently switch thinking off). candidates = [level for level in supported_norm if level != "none"] if not candidates: return effort requested_idx = EFFORT_LADDER.index(requested) - below = [ - level for level in candidates - if EFFORT_LADDER.index(level) < requested_idx - ] - if below: - return max(below, key=EFFORT_LADDER.index) - return min(candidates, key=EFFORT_LADDER.index) + below = [level for level in candidates if EFFORT_LADDER.index(level) < requested_idx] + return max(below, key=EFFORT_LADDER.index) if below else min(candidates, key=EFFORT_LADDER.index) def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]: - """Extract the user's explicit effort from a reasoning config, or None. - - None when the config is absent/malformed, carries no effort, or reasoning - is explicitly disabled — callers then omit the wire field (rule 2 above). - """ - if not isinstance(reasoning_config, dict): + """The user's explicit effort, or None (absent/malformed config, no effort, or reasoning + disabled) — callers then omit the wire field.""" + if not isinstance(reasoning_config, dict) or reasoning_config.get("enabled") is False: return None - if reasoning_config.get("enabled") is False: - return None - effort = str(reasoning_config.get("effort") or "").strip().lower() - return effort or None + return str(reasoning_config.get("effort") or "").strip().lower() or None diff --git a/agent/reasoning_timeouts.py b/agent/reasoning_timeouts.py index 322b803878..76ae3511e1 100644 --- a/agent/reasoning_timeouts.py +++ b/agent/reasoning_timeouts.py @@ -1,23 +1,11 @@ """Per-model stale-timeout FLOOR for known reasoning models. -Reasoning models (extended thinking before the first content token) routinely -exceed the default chat-model stale detectors (stream ``HERMES_STREAM_STALE_TIMEOUT`` -180s, non-stream ``HERMES_API_CALL_STALE_TIMEOUT`` 90s): upstream proxies / -load-balancers idle-kill the stream mid-think, surfacing as -``BrokenPipeError``/``RemoteProtocolError`` on the next read. The existing -stale-detector scaling consults :func:`get_reasoning_stale_timeout_floor` and -applies ``max(default, floor)``. Being a floor it: - -* never overrides explicit user config (``providers..models.. - stale_timeout_seconds`` / ``request_timeout_seconds`` win — this never runs - in that branch); -* never lowers an existing threshold; -* has zero effect on non-allowlisted models (resolver returns ``None``). - -Matching is start-anchored on the slug after any aggregator prefix -(``openai/``, ``x-ai/``) with an end-or-separator right anchor, so -``qwen3-235b`` matches ``qwen3`` but ``some-other-qwen3`` and a hypothetical -``llama-4-70b-o1-preview`` do not trigger the ``o1`` floor. +Reasoning models routinely exceed the default chat-model stale detectors (stream 180s, +non-stream 90s): upstream proxies idle-kill the stream mid-think, surfacing as +``BrokenPipeError``/``RemoteProtocolError``. The stale-detector scaling applies +``max(default, floor)`` from :func:`get_reasoning_stale_timeout_floor`, so this never +overrides explicit per-model ``stale_timeout_seconds``/``request_timeout_seconds`` (that +branch never calls it), never lowers a threshold, and is ``None`` for non-allowlisted models. """ from __future__ import annotations @@ -26,111 +14,58 @@ import re from typing import Optional -# (slug, floor_seconds). Order irrelevant — longest slug wins at match time. -_REASONING_STALE_TIMEOUT_FLOORS: tuple[tuple[str, int], ...] = ( - # NVIDIA Nemotron behind hosted NIM: documented 60-180s upstream idle kill. - ("nemotron-3-ultra", 600), - ("nemotron-3-super", 600), - ("nemotron-3-nano", 300), - ("nemotron-3.5-lightning", 300), - # DeepSeek R1 / V4 (reasoning_content streamed before final content). - ("deepseek-r1", 600), - ("deepseek-reasoner", 600), - ("deepseek-v4-flash", 600), - ("deepseek-v4-pro", 600), - # Qwen QwQ + the qwen3 family. Instruct variants also match ``qwen3`` — - # accepted: a slightly longer wait on a hung provider beats a pattern - # (``qwen3-.*-thinking``) that breaks on the next naming shape. - ("qwq-32b", 300), - ("qwen3", 180), - # OpenAI o-series: each variant enumerated so bare ``o1`` cannot - # over-match ``olmo-1`` or community derivatives. - ("o1", 600), - ("o1-mini", 600), - ("o1-pro", 600), - ("o1-preview", 600), - ("o3", 600), - ("o3-pro", 600), - ("o3-mini", 300), - ("o4-mini", 300), +# floor_seconds -> slugs. Order irrelevant — longest slug wins at match time. +_REASONING_STALE_TIMEOUT_FLOORS: dict[int, tuple[str, ...]] = { + 600: ( + # NVIDIA Nemotron behind hosted NIM: documented 60-180s upstream idle kill. + "nemotron-3-ultra", "nemotron-3-super", + # DeepSeek R1 / V4 (reasoning_content streamed before final content). + "deepseek-r1", "deepseek-reasoner", "deepseek-v4-flash", "deepseek-v4-pro", + # OpenAI o-series: each variant enumerated so bare ``o1`` cannot over-match ``olmo-1``. + "o1", "o1-mini", "o1-pro", "o1-preview", "o3", "o3-pro", + # Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a heavier thinking + # phase than the numbered line — otherwise the stale detector trips the circuit breaker. + "claude-fable", + ), + 300: ( + "nemotron-3-nano", "nemotron-3.5-lightning", "qwq-32b", "o3-mini", "o4-mini", + # xAI Grok: explicit reasoning pairs only, so bare ``grok-3``/``grok-4`` fast variants + # don't inherit the floor. + "grok-4-fast-reasoning", "grok-4.20-reasoning", "grok-4.5", "grok-4.6", + # "Ox Alpha" stealth reasoning model (OpenRouter / OpenCode Zen slugs); Thinking + # Machines Inkling (covers inkling-small and :free SKUs). + "ox-alpha", "x-preview-f-free", "inkling", + ), # Anthropic Claude 4.x+ thinking variants (anchored so 3.x never matches). - ("claude-opus-4", 240), - ("claude-opus-5", 240), - ("claude-sonnet-5", 180), - ("claude-sonnet-4.5", 180), - ("claude-sonnet-4.6", 180), - # Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a - # heavier thinking phase than the numbered line — deep-reasoning tier, - # otherwise the stale detector trips the cross-turn circuit breaker. - ("claude-fable", 600), - # xAI Grok: explicit reasoning / non-reasoning pairs only, so bare - # ``grok-3``/``grok-4`` fast variants don't inherit the 300s floor. - ("grok-4-fast-reasoning", 300), - ("grok-4.20-reasoning", 300), - ("grok-4.5", 300), - ("grok-4.6", 300), - ("grok-4-fast-non-reasoning", 180), - # "Ox Alpha" stealth reasoning model (OpenRouter / OpenCode Zen slugs). - ("ox-alpha", 300), - ("x-preview-f-free", 300), - # Thinking Machines Inkling; covers inkling-small and :free SKUs. - ("inkling", 300), -) + 240: ("claude-opus-4", "claude-opus-5"), + # qwen3 family: instruct variants also match — a slightly longer wait on a hung provider + # beats a pattern (``qwen3-.*-thinking``) that breaks on the next naming shape. + 180: ("qwen3", "claude-sonnet-5", "claude-sonnet-4.5", "claude-sonnet-4.6", "grok-4-fast-non-reasoning"), +} -# Pre-compiled once at import (immutable afterwards — safe under free-threaded -# Python). Right anchor: end-of-string or a slug separator; ``:`` is included -# because OpenRouter routing suffixes (``:free``, ``:nitro``) attach directly -# to the slug. Sorted longest-first so ``o3-mini`` beats ``o3``. +# Pre-compiled once at import (immutable afterwards — safe under free-threaded Python). +# Right anchor: end-of-string or a slug separator; ``:`` because OpenRouter routing suffixes +# (``:free``, ``:nitro``) attach directly to the slug. Longest-first so ``o3-mini`` beats ``o3``. _SORTED_REASONING_FLOORS: list[tuple[str, float, re.Pattern[str]]] = [ (slug, floor, re.compile(r"^" + re.escape(slug) + r"(?:$|[\-._:])")) for slug, floor in sorted( - _REASONING_STALE_TIMEOUT_FLOORS, key=lambda kv: -len(kv[0]) + ((slug, floor) for floor, slugs in _REASONING_STALE_TIMEOUT_FLOORS.items() for slug in slugs), + key=lambda kv: -len(kv[0]), ) ] def get_reasoning_stale_timeout_floor(model: object) -> Optional[float]: - """Return the stale-timeout floor (seconds) for a known reasoning model. + """Stale-timeout floor (seconds) for a known reasoning model, else ``None``. - ``None`` when the model is not allowlisted or the argument is empty / not - a string. The aggregator prefix (everything up to the last ``/``) is - stripped so the slug is matched start-anchored. Callers apply this as - ``max(default, floor)`` and only when no explicit per-model - ``stale_timeout_seconds`` is configured. - - >>> get_reasoning_stale_timeout_floor("nvidia/nemotron-3-ultra-550b-a55b") - 600.0 - >>> get_reasoning_stale_timeout_floor("openai/o3-mini") - 300.0 - >>> get_reasoning_stale_timeout_floor("deepseek/deepseek-r1") - 600.0 - >>> get_reasoning_stale_timeout_floor("deepseek/deepseek-v4-flash") - 600.0 - >>> get_reasoning_stale_timeout_floor("deepseek/deepseek-v4-pro") - 600.0 - >>> get_reasoning_stale_timeout_floor("qwen/qwen3-235b-a22b-thinking") - 180.0 - >>> get_reasoning_stale_timeout_floor("x-ai/grok-4-fast-reasoning") - 300.0 - >>> get_reasoning_stale_timeout_floor("anthropic/claude-opus-4-6") - 240.0 - >>> get_reasoning_stale_timeout_floor("anthropic/claude-fable-5") - 600.0 - >>> get_reasoning_stale_timeout_floor("gpt-4o") is None - True - >>> get_reasoning_stale_timeout_floor("olmo-1") is None - True - >>> get_reasoning_stale_timeout_floor(None) is None - True + The aggregator prefix (up to the last ``/``) is stripped and the slug matched + start-anchored with an end-or-separator right anchor, so ``qwen3-235b`` matches ``qwen3`` + but ``some-other-qwen3`` and ``llama-4-70b-o1-preview`` do not. """ if not model or not isinstance(model, str): return None - name = model.strip().lower() - if not name: - return None - if "/" in name: - name = name.rsplit("/", 1)[1] + name = model.strip().lower().rsplit("/", 1)[-1] for _slug, floor, pattern in _SORTED_REASONING_FLOORS: if pattern.search(name): return float(floor)