refactor(agent/reasoning_effort,reasoning_timeouts): floor table keyed by seconds, compact clamp, docstrings
This commit is contained in:
@@ -1,26 +1,13 @@
|
||||
"""Canonical reasoning-effort vocabulary and wire clamping.
|
||||
|
||||
Hermes' internal effort ladder (``hermes_constants.VALID_REASONING_EFFORTS``
|
||||
plus ``none``) is wider than any single provider wire accepts. Hand-rolled
|
||||
per-transport translation maps produced two recurring bugs: a new internal
|
||||
level (``ultra``) leaking to a wire that 400s on it, and an unknown level
|
||||
dropped to a weak default so the strongest ask resolved *weaker* than an
|
||||
explicit ``high`` (ladder inversion). This module is the single source of
|
||||
truth instead:
|
||||
|
||||
- :data:`EFFORT_LADDER` — canonical low→high ordering.
|
||||
- :func:`clamp_effort` — keep a supported level verbatim, else the **nearest
|
||||
weaker** supported level (never silently escalate cost); only when nothing
|
||||
weaker exists take the weakest supported level (GLM-5.2's floor is ``high``).
|
||||
- Named wire-vocabulary constants so call sites declare *data*, not logic.
|
||||
|
||||
Rules for call sites:
|
||||
1. Wire shape (``extra_body.reasoning`` vs top-level ``reasoning_effort`` vs
|
||||
a ``thinking`` toggle) stays local; only the vocabulary math lives here.
|
||||
2. Unset stays unset: ``clamp_effort`` translates an explicit request, never
|
||||
invents one — omit the field so the server default applies.
|
||||
3. Never patch a predicate: when a provider rejects a level, fix its declared
|
||||
supported set (data), not the call site.
|
||||
Hermes' internal effort ladder (``VALID_REASONING_EFFORTS`` plus ``none``) is wider than any
|
||||
single provider wire accepts; hand-rolled per-transport maps leaked new levels (``ultra``) to
|
||||
wires that 400 and inverted the ladder (unknown → weak default). Single source of truth:
|
||||
:data:`EFFORT_LADDER` (low→high), :func:`clamp_effort` (verbatim if supported, else the
|
||||
nearest WEAKER level; only when nothing weaker exists the weakest supported), and named
|
||||
wire-vocabulary constants so call sites declare data. Rules: wire shape stays local, only the
|
||||
vocabulary math lives here; unset stays unset (never invent an effort); when a provider
|
||||
rejects a level fix its declared set, never a predicate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -28,40 +15,21 @@ from __future__ import annotations
|
||||
import re
|
||||
from typing import Optional, Sequence
|
||||
|
||||
#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``)
|
||||
#: without matching K2-era names (``kimi-k2.6``).
|
||||
#: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``).
|
||||
_KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)")
|
||||
|
||||
# Canonical low→high ordering for nearest-level clamping. Includes "none" so an
|
||||
# explicit disable can be clamped when a provider publishes it as a level.
|
||||
EFFORT_LADDER: tuple[str, ...] = (
|
||||
"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra",
|
||||
)
|
||||
|
||||
# ``ultra`` is Hermes-internal (the Codex product tier); no wire accepts it, so
|
||||
# every declared set below stops at ``max`` and ``ultra`` always clamps down.
|
||||
# Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit
|
||||
# disable can be clamped when a provider publishes it as a level. ``ultra`` is Hermes-internal
|
||||
# (the Codex product tier): no wire accepts it, every declared set stops at ``max``.
|
||||
EFFORT_LADDER: tuple[str, ...] = ("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra")
|
||||
|
||||
#: Widest OpenAI-compatible wire vocabulary (OpenRouter, Nous Portal).
|
||||
OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = (
|
||||
"none", "minimal", "low", "medium", "high", "xhigh", "max",
|
||||
)
|
||||
|
||||
#: OpenAI/Codex Responses, per model generation (live-verified): ``minimal``
|
||||
#: is rejected by both (clamps to low); ``max`` is gpt-5.6-only.
|
||||
CODEX_GPT56_EFFORTS: tuple[str, ...] = (
|
||||
"none", "low", "medium", "high", "xhigh", "max",
|
||||
)
|
||||
CODEX_LEGACY_EFFORTS: tuple[str, ...] = (
|
||||
"none", "low", "medium", "high", "xhigh",
|
||||
)
|
||||
|
||||
|
||||
def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
"""Supported effort set for an OpenAI/Codex Responses model."""
|
||||
if "gpt-5.6" in (model or "").lower():
|
||||
return CODEX_GPT56_EFFORTS
|
||||
return CODEX_LEGACY_EFFORTS
|
||||
OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ("none", "minimal", "low", "medium", "high", "xhigh", "max")
|
||||
|
||||
#: OpenAI/Codex Responses per model generation (live-verified): ``minimal`` is rejected by
|
||||
#: both (clamps to low); ``max`` is gpt-5.6-only.
|
||||
CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh", "max")
|
||||
CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh")
|
||||
|
||||
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
|
||||
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
|
||||
@@ -70,33 +38,27 @@ XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
#: Actual Computer relays (SGLang/vLLM).
|
||||
ACTUAL_RELAY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "max")
|
||||
|
||||
#: Moonshot/Kimi K3 (server default high) vs K2-era models.
|
||||
#: Moonshot/Kimi K3 (server default high) vs K2-era models. K3 quirks: ``high`` is K3's
|
||||
#: positional middle AND server default, so ``medium`` rounds to it rather than down to
|
||||
#: ``low``; ``xhigh`` rounds up to ``max`` (K3's top tier).
|
||||
KIMI_K3_EFFORTS: tuple[str, ...] = ("low", "high", "max")
|
||||
KIMI_K2_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"}
|
||||
|
||||
#: OpenCode "Ox Alpha" (x-preview-f-free): thinking cannot be disabled and the
|
||||
#: wire accepts exactly low/high/max (medium/none/xhigh 400); xhigh rounds up.
|
||||
#: OpenCode "Ox Alpha" (x-preview-f-free): thinking cannot be disabled and the wire accepts
|
||||
#: exactly low/high/max (medium/none/xhigh 400); xhigh rounds up.
|
||||
OX_ALPHA_EFFORTS: tuple[str, ...] = ("low", "high", "max")
|
||||
OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
|
||||
#: Tencent TokenHub.
|
||||
#: Tencent TokenHub / Nebius Token Factory / Upstage Solar: plain three-level knobs.
|
||||
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
#: Nebius Token Factory (top-level reasoning_effort knob).
|
||||
NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
#: Kimi K3 vendor-documented quirks: ``high`` is K3's positional middle AND
|
||||
#: server default, so ``medium`` rounds to it rather than down to ``low``;
|
||||
#: ``xhigh`` rounds up to ``max`` (K3's top tier).
|
||||
KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"}
|
||||
|
||||
#: GLM-5.2 native knob: exactly ``high`` (its minimum thinking level) and
|
||||
#: ``max``; ``xhigh`` requests the top tier, not the floor.
|
||||
#: GLM-5.2 native knob: exactly ``high`` (its minimum thinking level) and ``max``; GLM-5.3
|
||||
#: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier.
|
||||
GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
|
||||
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
|
||||
#: GLM-5.3 widens the knob to a graded scale (live-verified, monotonic
|
||||
#: reasoning-token scaling); ``xhigh`` requests the top tier.
|
||||
GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
|
||||
GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
|
||||
@@ -111,21 +73,16 @@ OLLAMA_CLOUD_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
#: Meta Model API (Muse): rejects ``none``.
|
||||
META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh")
|
||||
|
||||
#: Upstage Solar Pro/Open.
|
||||
SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
"""Supported effort set for an OpenAI/Codex Responses model."""
|
||||
return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS
|
||||
|
||||
|
||||
def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
"""Supported effort set for a Moonshot/Kimi slug.
|
||||
|
||||
K3 is served as bare ``k3``, plan variants (``k3-256k``) and ``kimi-k3*``
|
||||
aliases; everything earlier speaks low/medium/high. Boundary-matched so
|
||||
K2-era names (``kimi-k2.6``) never match.
|
||||
"""
|
||||
"""Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3)."""
|
||||
m = (model or "").strip().lower().split("/")[-1]
|
||||
if _KIMI_K3_SLUG_RE.search(m):
|
||||
return KIMI_K3_EFFORTS
|
||||
return KIMI_K2_EFFORTS
|
||||
return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS
|
||||
|
||||
|
||||
def clamp_effort(
|
||||
@@ -135,55 +92,37 @@ def clamp_effort(
|
||||
) -> Optional[str]:
|
||||
"""Clamp a requested reasoning effort onto a wire's supported levels.
|
||||
|
||||
``overrides`` (a declared vendor mapping, e.g. Kimi K3 ``medium → high``)
|
||||
is consulted first. Otherwise the request passes through unchanged when it
|
||||
is supported, when the supported set is unknown/empty, or when it isn't a
|
||||
recognized ladder level (custom providers may use bespoke names). Else the
|
||||
**nearest weaker** supported level is returned so a clamp never escalates
|
||||
cost; when nothing weaker exists, the weakest supported level is (the
|
||||
provider's floor is the closest honest match). Monotonic: a stronger
|
||||
request never resolves weaker than a weaker request would.
|
||||
``overrides`` (a declared vendor mapping, e.g. Kimi K3 ``medium → high``) is consulted
|
||||
first. Otherwise the request passes through unchanged when it is supported, when the
|
||||
supported set is unknown/empty, or when it isn't a recognized ladder level (custom
|
||||
providers may use bespoke names). Else the **nearest weaker** supported level is returned
|
||||
so a clamp never escalates cost; when nothing weaker exists, the weakest supported level
|
||||
(the provider's floor is the closest honest match). Monotonic: a stronger request never
|
||||
resolves weaker than a weaker request would.
|
||||
"""
|
||||
requested = str(effort or "").strip().lower()
|
||||
if not requested or not supported:
|
||||
return effort
|
||||
supported_norm = [
|
||||
str(level).strip().lower()
|
||||
for level in supported
|
||||
if str(level).strip().lower() in EFFORT_LADDER
|
||||
]
|
||||
supported_norm = [lvl for lvl in (str(s).strip().lower() for s in supported) if lvl in EFFORT_LADDER]
|
||||
if not supported_norm or requested in supported_norm:
|
||||
return effort
|
||||
if overrides:
|
||||
mapped = overrides.get(requested)
|
||||
if mapped in supported_norm:
|
||||
return mapped
|
||||
if overrides and overrides.get(requested) in supported_norm:
|
||||
return overrides[requested]
|
||||
if requested not in EFFORT_LADDER:
|
||||
return effort
|
||||
# "none" disables reasoning — never a degradation target for an enabled
|
||||
# ask (clamping "minimal" to "none" would silently switch thinking off).
|
||||
# "none" disables reasoning — never a degradation target for an enabled ask
|
||||
# (clamping "minimal" to "none" would silently switch thinking off).
|
||||
candidates = [level for level in supported_norm if level != "none"]
|
||||
if not candidates:
|
||||
return effort
|
||||
requested_idx = EFFORT_LADDER.index(requested)
|
||||
below = [
|
||||
level for level in candidates
|
||||
if EFFORT_LADDER.index(level) < requested_idx
|
||||
]
|
||||
if below:
|
||||
return max(below, key=EFFORT_LADDER.index)
|
||||
return min(candidates, key=EFFORT_LADDER.index)
|
||||
below = [level for level in candidates if EFFORT_LADDER.index(level) < requested_idx]
|
||||
return max(below, key=EFFORT_LADDER.index) if below else min(candidates, key=EFFORT_LADDER.index)
|
||||
|
||||
|
||||
def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
|
||||
"""Extract the user's explicit effort from a reasoning config, or None.
|
||||
|
||||
None when the config is absent/malformed, carries no effort, or reasoning
|
||||
is explicitly disabled — callers then omit the wire field (rule 2 above).
|
||||
"""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
"""The user's explicit effort, or None (absent/malformed config, no effort, or reasoning
|
||||
disabled) — callers then omit the wire field."""
|
||||
if not isinstance(reasoning_config, dict) or reasoning_config.get("enabled") is False:
|
||||
return None
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return None
|
||||
effort = str(reasoning_config.get("effort") or "").strip().lower()
|
||||
return effort or None
|
||||
return str(reasoning_config.get("effort") or "").strip().lower() or None
|
||||
|
||||
@@ -1,23 +1,11 @@
|
||||
"""Per-model stale-timeout FLOOR for known reasoning models.
|
||||
|
||||
Reasoning models (extended thinking before the first content token) routinely
|
||||
exceed the default chat-model stale detectors (stream ``HERMES_STREAM_STALE_TIMEOUT``
|
||||
180s, non-stream ``HERMES_API_CALL_STALE_TIMEOUT`` 90s): upstream proxies /
|
||||
load-balancers idle-kill the stream mid-think, surfacing as
|
||||
``BrokenPipeError``/``RemoteProtocolError`` on the next read. The existing
|
||||
stale-detector scaling consults :func:`get_reasoning_stale_timeout_floor` and
|
||||
applies ``max(default, floor)``. Being a floor it:
|
||||
|
||||
* never overrides explicit user config (``providers.<id>.models.<model>.
|
||||
stale_timeout_seconds`` / ``request_timeout_seconds`` win — this never runs
|
||||
in that branch);
|
||||
* never lowers an existing threshold;
|
||||
* has zero effect on non-allowlisted models (resolver returns ``None``).
|
||||
|
||||
Matching is start-anchored on the slug after any aggregator prefix
|
||||
(``openai/``, ``x-ai/``) with an end-or-separator right anchor, so
|
||||
``qwen3-235b`` matches ``qwen3`` but ``some-other-qwen3`` and a hypothetical
|
||||
``llama-4-70b-o1-preview`` do not trigger the ``o1`` floor.
|
||||
Reasoning models routinely exceed the default chat-model stale detectors (stream 180s,
|
||||
non-stream 90s): upstream proxies idle-kill the stream mid-think, surfacing as
|
||||
``BrokenPipeError``/``RemoteProtocolError``. The stale-detector scaling applies
|
||||
``max(default, floor)`` from :func:`get_reasoning_stale_timeout_floor`, so this never
|
||||
overrides explicit per-model ``stale_timeout_seconds``/``request_timeout_seconds`` (that
|
||||
branch never calls it), never lowers a threshold, and is ``None`` for non-allowlisted models.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -26,111 +14,58 @@ import re
|
||||
from typing import Optional
|
||||
|
||||
|
||||
# (slug, floor_seconds). Order irrelevant — longest slug wins at match time.
|
||||
_REASONING_STALE_TIMEOUT_FLOORS: tuple[tuple[str, int], ...] = (
|
||||
# NVIDIA Nemotron behind hosted NIM: documented 60-180s upstream idle kill.
|
||||
("nemotron-3-ultra", 600),
|
||||
("nemotron-3-super", 600),
|
||||
("nemotron-3-nano", 300),
|
||||
("nemotron-3.5-lightning", 300),
|
||||
# DeepSeek R1 / V4 (reasoning_content streamed before final content).
|
||||
("deepseek-r1", 600),
|
||||
("deepseek-reasoner", 600),
|
||||
("deepseek-v4-flash", 600),
|
||||
("deepseek-v4-pro", 600),
|
||||
# Qwen QwQ + the qwen3 family. Instruct variants also match ``qwen3`` —
|
||||
# accepted: a slightly longer wait on a hung provider beats a pattern
|
||||
# (``qwen3-.*-thinking``) that breaks on the next naming shape.
|
||||
("qwq-32b", 300),
|
||||
("qwen3", 180),
|
||||
# OpenAI o-series: each variant enumerated so bare ``o1`` cannot
|
||||
# over-match ``olmo-1`` or community derivatives.
|
||||
("o1", 600),
|
||||
("o1-mini", 600),
|
||||
("o1-pro", 600),
|
||||
("o1-preview", 600),
|
||||
("o3", 600),
|
||||
("o3-pro", 600),
|
||||
("o3-mini", 300),
|
||||
("o4-mini", 300),
|
||||
# floor_seconds -> slugs. Order irrelevant — longest slug wins at match time.
|
||||
_REASONING_STALE_TIMEOUT_FLOORS: dict[int, tuple[str, ...]] = {
|
||||
600: (
|
||||
# NVIDIA Nemotron behind hosted NIM: documented 60-180s upstream idle kill.
|
||||
"nemotron-3-ultra", "nemotron-3-super",
|
||||
# DeepSeek R1 / V4 (reasoning_content streamed before final content).
|
||||
"deepseek-r1", "deepseek-reasoner", "deepseek-v4-flash", "deepseek-v4-pro",
|
||||
# OpenAI o-series: each variant enumerated so bare ``o1`` cannot over-match ``olmo-1``.
|
||||
"o1", "o1-mini", "o1-pro", "o1-preview", "o3", "o3-pro",
|
||||
# Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a heavier thinking
|
||||
# phase than the numbered line — otherwise the stale detector trips the circuit breaker.
|
||||
"claude-fable",
|
||||
),
|
||||
300: (
|
||||
"nemotron-3-nano", "nemotron-3.5-lightning", "qwq-32b", "o3-mini", "o4-mini",
|
||||
# xAI Grok: explicit reasoning pairs only, so bare ``grok-3``/``grok-4`` fast variants
|
||||
# don't inherit the floor.
|
||||
"grok-4-fast-reasoning", "grok-4.20-reasoning", "grok-4.5", "grok-4.6",
|
||||
# "Ox Alpha" stealth reasoning model (OpenRouter / OpenCode Zen slugs); Thinking
|
||||
# Machines Inkling (covers inkling-small and :free SKUs).
|
||||
"ox-alpha", "x-preview-f-free", "inkling",
|
||||
),
|
||||
# Anthropic Claude 4.x+ thinking variants (anchored so 3.x never matches).
|
||||
("claude-opus-4", 240),
|
||||
("claude-opus-5", 240),
|
||||
("claude-sonnet-5", 180),
|
||||
("claude-sonnet-4.5", 180),
|
||||
("claude-sonnet-4.6", 180),
|
||||
# Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a
|
||||
# heavier thinking phase than the numbered line — deep-reasoning tier,
|
||||
# otherwise the stale detector trips the cross-turn circuit breaker.
|
||||
("claude-fable", 600),
|
||||
# xAI Grok: explicit reasoning / non-reasoning pairs only, so bare
|
||||
# ``grok-3``/``grok-4`` fast variants don't inherit the 300s floor.
|
||||
("grok-4-fast-reasoning", 300),
|
||||
("grok-4.20-reasoning", 300),
|
||||
("grok-4.5", 300),
|
||||
("grok-4.6", 300),
|
||||
("grok-4-fast-non-reasoning", 180),
|
||||
# "Ox Alpha" stealth reasoning model (OpenRouter / OpenCode Zen slugs).
|
||||
("ox-alpha", 300),
|
||||
("x-preview-f-free", 300),
|
||||
# Thinking Machines Inkling; covers inkling-small and :free SKUs.
|
||||
("inkling", 300),
|
||||
)
|
||||
240: ("claude-opus-4", "claude-opus-5"),
|
||||
# qwen3 family: instruct variants also match — a slightly longer wait on a hung provider
|
||||
# beats a pattern (``qwen3-.*-thinking``) that breaks on the next naming shape.
|
||||
180: ("qwen3", "claude-sonnet-5", "claude-sonnet-4.5", "claude-sonnet-4.6", "grok-4-fast-non-reasoning"),
|
||||
}
|
||||
|
||||
|
||||
# Pre-compiled once at import (immutable afterwards — safe under free-threaded
|
||||
# Python). Right anchor: end-of-string or a slug separator; ``:`` is included
|
||||
# because OpenRouter routing suffixes (``:free``, ``:nitro``) attach directly
|
||||
# to the slug. Sorted longest-first so ``o3-mini`` beats ``o3``.
|
||||
# Pre-compiled once at import (immutable afterwards — safe under free-threaded Python).
|
||||
# Right anchor: end-of-string or a slug separator; ``:`` because OpenRouter routing suffixes
|
||||
# (``:free``, ``:nitro``) attach directly to the slug. Longest-first so ``o3-mini`` beats ``o3``.
|
||||
_SORTED_REASONING_FLOORS: list[tuple[str, float, re.Pattern[str]]] = [
|
||||
(slug, floor, re.compile(r"^" + re.escape(slug) + r"(?:$|[\-._:])"))
|
||||
for slug, floor in sorted(
|
||||
_REASONING_STALE_TIMEOUT_FLOORS, key=lambda kv: -len(kv[0])
|
||||
((slug, floor) for floor, slugs in _REASONING_STALE_TIMEOUT_FLOORS.items() for slug in slugs),
|
||||
key=lambda kv: -len(kv[0]),
|
||||
)
|
||||
]
|
||||
|
||||
|
||||
def get_reasoning_stale_timeout_floor(model: object) -> Optional[float]:
|
||||
"""Return the stale-timeout floor (seconds) for a known reasoning model.
|
||||
"""Stale-timeout floor (seconds) for a known reasoning model, else ``None``.
|
||||
|
||||
``None`` when the model is not allowlisted or the argument is empty / not
|
||||
a string. The aggregator prefix (everything up to the last ``/``) is
|
||||
stripped so the slug is matched start-anchored. Callers apply this as
|
||||
``max(default, floor)`` and only when no explicit per-model
|
||||
``stale_timeout_seconds`` is configured.
|
||||
|
||||
>>> get_reasoning_stale_timeout_floor("nvidia/nemotron-3-ultra-550b-a55b")
|
||||
600.0
|
||||
>>> get_reasoning_stale_timeout_floor("openai/o3-mini")
|
||||
300.0
|
||||
>>> get_reasoning_stale_timeout_floor("deepseek/deepseek-r1")
|
||||
600.0
|
||||
>>> get_reasoning_stale_timeout_floor("deepseek/deepseek-v4-flash")
|
||||
600.0
|
||||
>>> get_reasoning_stale_timeout_floor("deepseek/deepseek-v4-pro")
|
||||
600.0
|
||||
>>> get_reasoning_stale_timeout_floor("qwen/qwen3-235b-a22b-thinking")
|
||||
180.0
|
||||
>>> get_reasoning_stale_timeout_floor("x-ai/grok-4-fast-reasoning")
|
||||
300.0
|
||||
>>> get_reasoning_stale_timeout_floor("anthropic/claude-opus-4-6")
|
||||
240.0
|
||||
>>> get_reasoning_stale_timeout_floor("anthropic/claude-fable-5")
|
||||
600.0
|
||||
>>> get_reasoning_stale_timeout_floor("gpt-4o") is None
|
||||
True
|
||||
>>> get_reasoning_stale_timeout_floor("olmo-1") is None
|
||||
True
|
||||
>>> get_reasoning_stale_timeout_floor(None) is None
|
||||
True
|
||||
The aggregator prefix (up to the last ``/``) is stripped and the slug matched
|
||||
start-anchored with an end-or-separator right anchor, so ``qwen3-235b`` matches ``qwen3``
|
||||
but ``some-other-qwen3`` and ``llama-4-70b-o1-preview`` do not.
|
||||
"""
|
||||
if not model or not isinstance(model, str):
|
||||
return None
|
||||
name = model.strip().lower()
|
||||
if not name:
|
||||
return None
|
||||
if "/" in name:
|
||||
name = name.rsplit("/", 1)[1]
|
||||
name = model.strip().lower().rsplit("/", 1)[-1]
|
||||
for _slug, floor, pattern in _SORTED_REASONING_FLOORS:
|
||||
if pattern.search(name):
|
||||
return float(floor)
|
||||
|
||||
Reference in New Issue
Block a user