fix(models): resolve routed OpenRouter ids in models.dev catalog lookups

Extends the routing-variant fix to the sibling catalog paths, so the whole
bug class is covered rather than just context length (#97820).

_find_model_entry(), lookup_models_dev_context(), get_model_info(), and
get_model_capabilities() all keyed on the full suffixed id, so a routed id
such as z-ai/glm-5.3-flash:floor missed models.dev and fell through to the
generic "glm" family default (202,752 instead of 1,310,720).

The retry with the base id runs LAST — after exact and case-insensitive
matching — so a real catalog SKU always wins over its base.

:free and :batch are deliberately NOT stripped. They are real catalog SKUs,
not routing modifiers: OpenRouter's /models currently lists 18 :free and 65
:batch entries, and 13 of those carry a context window different from their
base (z-ai/glm-5.2:free is 256K vs the base's 1.05M). Stripping them would
report a window LARGER than the model has, so the request fails at the API
instead of merely compacting early — a worse failure than the under-report
this fixes. An absent SKU must also miss rather than inherit the base, so
model_overrides _default fill-gap semantics keep working.

The suffix set is shared from hermes_constants, so validation, context
resolution, and catalog lookup all agree on one definition.
This commit is contained in:
jakobdylanc
2026-09-01 16:53:04 -04:00
committed by Teknium
parent 284ee03df1
commit f28c52845a
2 changed files with 196 additions and 7 deletions

View File

@@ -19,6 +19,8 @@ from typing import Any, Dict, List, Optional, Tuple
from utils import atomic_json_write, atomic_write_text
from hermes_constants import openrouter_variant_base
import requests
logger = logging.getLogger(__name__)
@@ -462,12 +464,46 @@ def _get_provider_models(provider: str, *, allow_network: bool = False) -> Optio
return _registry_models(mdev_id, allow_network=allow_network) if mdev_id else None
def _iter_model_entries(models: Dict[str, Any], model: str, *, suffix_fallback: bool = True):
_OPENROUTER_CATALOG_PROVIDERS = frozenset({"openrouter"})
def _openrouter_catalog_lookup_base(provider: str, model: str) -> Optional[str]:
"""Return the base id to retry a catalog lookup with, or ``None``.
OpenRouter's ``:nitro`` / ``:floor`` / ``:exacto`` / ``:online`` are
request-time routing modifiers: they change which endpoint serves the
request, never which model runs. ``/models`` and models.dev list only the
base id, so a routed id must resolve to the base model's metadata.
Scoped to OpenRouter so a genuine ``model:tag`` on another provider (an
Ollama tag, a ``:cloud`` catalog key) is never rewritten.
Deliberately excludes ``:free``, ``:batch``, ``:extended``, and
``:thinking``: those are REAL catalog SKUs with their own entries and
their own — sometimes different — context windows. Stripping them would
report a window LARGER than the model actually has. Their real entries
are found by the exact/case-insensitive passes above, and a genuinely
absent SKU must miss so ``model_overrides`` ``_default`` fill-gap
semantics still apply.
"""
if provider not in _OPENROUTER_CATALOG_PROVIDERS:
return None
return openrouter_variant_base(model)
def _iter_model_entries(
models: Dict[str, Any], model: str, *, suffix_fallback: bool = True, provider: str = ""
):
"""Yield ``(model_id, entry)`` candidates: exact, case-insensitive, then (optionally)
``:cloud``/``-cloud`` suffixed forms. Suffix fallback: some providers (ollama-cloud) store
``kimi-k2.6:cloud`` while the live API returns the bare name; without it context lookup falls to
stale OpenRouter metadata and trips the 64k minimum-context guard. Every consumer shares this
order so a suffix-keyed catalog model counts as KNOWN for ``model_overrides`` fill-gap ``_default``."""
order so a suffix-keyed catalog model counts as KNOWN for ``model_overrides`` fill-gap ``_default``.
``provider`` enables the OpenRouter routing-variant fallback as a LAST
resort — after exact, case-insensitive, and ``:cloud`` matching — so a
real catalog SKU always wins over its base.
"""
for name in ([model] + [model + suffix for suffix in (":cloud", "-cloud")] if suffix_fallback else [model]):
entry = models.get(name)
if isinstance(entry, dict):
@@ -476,11 +512,20 @@ def _iter_model_entries(models: Dict[str, Any], model: str, *, suffix_fallback:
for mid, mdata in models.items():
if mid.lower() == name_lower and isinstance(mdata, dict):
yield mid, mdata
routed_base = _openrouter_catalog_lookup_base(provider, model)
if routed_base is not None:
# Recursion is bounded: the base never carries a recognized variant suffix,
# and provider is cleared so the retry cannot loop.
yield from _iter_model_entries(
models, routed_base, suffix_fallback=suffix_fallback, provider=""
)
def _find_model_entry(models: Dict[str, Any], model: str) -> Optional[Dict[str, Any]]:
def _find_model_entry(
models: Dict[str, Any], model: str, provider: str = ""
) -> Optional[Dict[str, Any]]:
"""First catalog entry for *model* (exact, case-insensitive, suffix), or None."""
return next((entry for _mid, entry in _iter_model_entries(models, model)), None)
return next((entry for _mid, entry in _iter_model_entries(models, model, provider=provider)), None)
def _extract_limit(entry: Any, key: str) -> Optional[int]:
@@ -506,7 +551,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
if override_ctx is not None:
return override_ctx
models = _get_provider_models(provider, allow_network=allow_network)
catalog_ctx = next((ctx for _mid, entry in _iter_model_entries(models, model) if (ctx := _extract_context(entry))), None) if models is not None else None
catalog_ctx = next((ctx for _mid, entry in _iter_model_entries(models, model, provider=provider) if (ctx := _extract_context(entry))), None) if models is not None else None
return catalog_ctx if catalog_ctx is not None else _default_override_context(provider)
@@ -700,7 +745,7 @@ def get_model_capabilities(provider: str, model: str, *, allow_network: bool = F
(#84482).
"""
models = _get_provider_models(provider, allow_network=allow_network)
entry = _find_model_entry(models, model) if models is not None else None
entry = _find_model_entry(models, model, provider) if models is not None else None
raw = _apply_overrides(provider, model, entry)
if raw is None:
return None
@@ -803,7 +848,7 @@ def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = Fal
"""
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
models = _registry_models(mdev_id, allow_network=allow_network)
mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False), (model_id, None)) if models is not None else (model_id, None)
mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False, provider=provider_id), (model_id, None)) if models is not None else (model_id, None)
# Not in catalog — an override (explicit or _default) may still provide it.
raw = _apply_overrides(provider_id, model_id, entry)
return _parse_model_info(mid, raw, mdev_id) if raw is not None else None

View File

@@ -1335,3 +1335,147 @@ class TestModelOverrides:
assert info is not None
assert "image" in info.input_modalities
assert info.attachment is True
# =========================================================================
# OpenRouter routing-variant suffixes — catalog lookup across consumers
# =========================================================================
class TestOpenRouterRoutingVariantCatalogLookup:
"""OpenRouter's `:nitro`/`:floor`/`:exacto`/`:online` are request-time
routing modifiers, not catalog models. models.dev (like OpenRouter's own
/models) lists only the base id, so every catalog consumer must resolve a
routed id to the base model's metadata while the suffixed id stays on the
wire. See issue #97820.
The counterpart to the context-length half of this fix lives in
tests/agent/test_model_metadata.py::TestOpenRouterRoutingVariantContextLength.
Critically, `:free` and `:batch` are NOT routing variants — they are real
catalog SKUs with their own entries and sometimes their own context
windows. Stripping them would report a window LARGER than the model has,
which fails at the API rather than merely compacting early.
"""
#: Base model plus a `:free` SKU whose window deliberately DIFFERS from
#: it — mirrors real catalog entries such as z-ai/glm-5.2 (1.05M) vs
#: z-ai/glm-5.2:free (256K).
REGISTRY = {
"openrouter": {
"id": "openrouter",
"models": {
"z-ai/glm-5.3-flash": {
"id": "z-ai/glm-5.3-flash",
"limit": {"context": 1310720, "output": 131072},
"tool_call": True,
"reasoning": True,
},
"z-ai/glm-5.2": {
"id": "z-ai/glm-5.2",
"limit": {"context": 1048576, "output": 131072},
},
"z-ai/glm-5.2:free": {
"id": "z-ai/glm-5.2:free",
"limit": {"context": 256000, "output": 131072},
},
},
},
}
ROUTING_SUFFIXES = ["nitro", "floor", "exacto", "online"]
def _patch(self):
return patch(
"agent.models_dev.fetch_models_dev", return_value=self.REGISTRY
)
@pytest.mark.parametrize("suffix", ROUTING_SUFFIXES)
def test_context_lookup_matches_base(self, suffix):
"""A routed id resolves to exactly its base model's context."""
with self._patch():
base = lookup_models_dev_context("openrouter", "z-ai/glm-5.3-flash")
routed = lookup_models_dev_context(
"openrouter", f"z-ai/glm-5.3-flash:{suffix}"
)
assert routed == base == 1310720
@pytest.mark.parametrize("suffix", ROUTING_SUFFIXES)
def test_capabilities_match_base(self, suffix):
"""get_model_capabilities resolves routed ids (not just context)."""
with self._patch():
base = get_model_capabilities("openrouter", "z-ai/glm-5.3-flash")
routed = get_model_capabilities(
"openrouter", f"z-ai/glm-5.3-flash:{suffix}"
)
assert routed is not None
assert routed.context_window == base.context_window == 1310720
assert routed.supports_tools == base.supports_tools
assert routed.supports_reasoning == base.supports_reasoning
@pytest.mark.parametrize("suffix", ROUTING_SUFFIXES)
def test_model_info_matches_base(self, suffix):
"""get_model_info resolves routed ids."""
with self._patch():
base = get_model_info("openrouter", "z-ai/glm-5.3-flash")
routed = get_model_info("openrouter", f"z-ai/glm-5.3-flash:{suffix}")
assert routed is not None
assert routed.context_window == base.context_window == 1310720
def test_suffix_match_is_case_insensitive(self):
with self._patch():
assert (
lookup_models_dev_context("openrouter", "z-ai/glm-5.3-flash:FLOOR")
== 1310720
)
def test_real_free_sku_keeps_its_own_window(self):
"""REGRESSION GUARD: `:free` is a real SKU, not a routing variant.
z-ai/glm-5.2:free is 256K while its base is 1.05M. Treating `:free`
as strippable would report 1.05M — a window the model does not have,
so the request fails at the API instead of compacting early. The SKU's
own entry must win.
"""
with self._patch():
assert (
lookup_models_dev_context("openrouter", "z-ai/glm-5.2:free") == 256000
)
assert lookup_models_dev_context("openrouter", "z-ai/glm-5.2") == 1048576
def test_absent_sku_suffix_still_misses(self):
"""A `:free` id with no catalog entry must MISS rather than fall back
to its base — otherwise model_overrides `_default` fill-gap semantics
break, and an absent free tier would inherit the paid tier's window."""
with self._patch():
assert (
lookup_models_dev_context("openrouter", "z-ai/glm-5.3-flash:free")
is None
)
def test_non_openrouter_provider_is_never_stripped(self):
"""The carve-out is OpenRouter-only; another provider's colon-suffixed
id (an Ollama tag, a `:cloud` key) keeps exact-match semantics."""
registry = {
"anthropic": {
"id": "anthropic",
"models": {"claude-x": {"limit": {"context": 200000}}},
},
}
with patch("agent.models_dev.fetch_models_dev", return_value=registry):
assert lookup_models_dev_context("anthropic", "claude-x:floor") is None
assert lookup_models_dev_context("anthropic", "claude-x") == 200000
def test_unknown_suffix_is_not_stripped(self):
with self._patch():
assert (
lookup_models_dev_context("openrouter", "z-ai/glm-5.3-flash:bogus")
is None
)
def test_bare_model_lookup_unaffected(self):
"""No suffix — behavior is byte-identical to before the change."""
with self._patch():
assert (
lookup_models_dev_context("openrouter", "z-ai/glm-5.3-flash") == 1310720
)
assert lookup_models_dev_context("openrouter", "nope/missing") is None