fix(web): bound the /api/model/info context-length probe

get_model_info called get_model_context_length inline. The resolver
chain runs several sequential provider probes, each with its own
multi-second timeout, so an unreachable model.base_url held the
response for tens of seconds. Bound the whole chain to a 5s budget on
a throwaway worker thread; on timeout the response degrades to
auto_context_length=0 while the abandoned probe finishes on its own.

Fixes https://github.com/NousResearch/hermes-agent/issues/63214 (backend half)
This commit is contained in:
Hermes Agent
2026-09-25 12:31:55 -05:00
committed by brooklyn!
parent ffa40b07f0
commit 43c1eee1fa
2 changed files with 107 additions and 4 deletions

View File

@@ -5,6 +5,7 @@ Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch o
"""
import asyncio
import concurrent.futures
import logging
from typing import Optional
@@ -50,6 +51,41 @@ def _load_config_scoped(profile: Optional[str]) -> dict:
return load_config()
# Blocking budget for /api/model/info's context-length resolution. The resolver
# chain (agent.model_metadata.get_model_context_length) runs several sequential
# provider probes, each with its own multi-second timeout, so an unreachable or
# blackholed model.base_url can hold this response for tens of seconds — and the
# Desktop Model Settings page waits on it (#63214).
_MODEL_INFO_PROBE_BUDGET_S = 5.0
def _bounded_context_length_probe(model: str, base_url: str, provider: str) -> int:
"""``get_model_context_length`` with the route's blocking budget.
On timeout the abandoned probe keeps running in its worker thread (bounded
by its own per-request timeouts) while the response degrades to
``auto_context_length = 0`` ("auto-detected: unknown").
"""
from agent.model_metadata import get_model_context_length
pool = concurrent.futures.ThreadPoolExecutor(max_workers=1, thread_name_prefix="model-info-probe")
try:
return pool.submit(
get_model_context_length, model=model, base_url=base_url, provider=provider,
config_context_length=None
).result(timeout=_MODEL_INFO_PROBE_BUDGET_S)
except concurrent.futures.TimeoutError:
_log.warning(
"GET /api/model/info: context-length probe for %r at %s exceeded %.1fs — returning unknown",
model, base_url or "<default>", _MODEL_INFO_PROBE_BUDGET_S,
)
return 0
finally:
# wait=False: never block the response (or interpreter exit) on the
# abandoned probe.
pool.shutdown(wait=False)
@router.get("/api/model/info")
def get_model_info(profile: Optional[str] = None):
"""Resolved metadata for the configured model: auto-detected vs configured
@@ -65,10 +101,10 @@ def get_model_info(profile: Optional[str] = None):
return dict(_EMPTY_MODEL_INFO, provider=provider)
try:
from agent.model_metadata import get_model_context_length
# config_context_length=None: ignore the override — we want the auto value
auto_ctx = get_model_context_length(model=model_name, base_url=base_url, provider=provider,
config_context_length=None)
# config_context_length=None: ignore the override — we want the auto value.
# Bounded: the resolver's provider probes can hang for tens of seconds
# when model.base_url is unreachable (#63214).
auto_ctx = _bounded_context_length_probe(model_name, base_url, provider)
except Exception:
auto_ctx = 0

View File

@@ -0,0 +1,67 @@
"""/api/model/info must bound its context-length probe (#63214).
The resolver chain (``agent.model_metadata.get_model_context_length``) runs
several sequential provider probes, each with its own multi-second timeout, so
an unreachable ``model.base_url`` held the response for tens of seconds — and
the Desktop Model Settings page, which awaits this endpoint inside one
``Promise.all``, showed loading skeletons indefinitely.
"""
from __future__ import annotations
import time
def _config(model: str, provider: str, base_url: str) -> dict:
return {"model": {"default": model, "provider": provider, "base_url": base_url}}
def test_model_info_degrades_when_the_context_probe_exceeds_its_budget(monkeypatch):
from hermes_cli.web_routers import models as router
monkeypatch.setattr(
router, "_load_config_scoped",
lambda profile: _config("some-model", "custom-proxy", "http://localhost:9/v1"),
)
monkeypatch.setattr(router, "_MODEL_INFO_PROBE_BUDGET_S", 0.2)
import agent.model_metadata as metadata
def _hanging_probe(model, base_url="", api_key="", config_context_length=None, provider="", custom_providers=None):
time.sleep(1.0)
raise AssertionError("the abandoned probe should never be awaited")
monkeypatch.setattr(metadata, "get_model_context_length", _hanging_probe)
started = time.monotonic()
info = router.get_model_info()
elapsed = time.monotonic() - started
# The response degrades to "auto context unknown" instead of hanging…
assert info["model"] == "some-model"
assert info["provider"] == "custom-proxy"
assert info["auto_context_length"] == 0
# …and it returns within the (shrunk) budget, not the probe's 1s sleep.
assert elapsed < 0.9
def test_model_info_surfaces_the_context_value_when_the_probe_is_fast(monkeypatch):
from hermes_cli.web_routers import models as router
monkeypatch.setattr(
router, "_load_config_scoped",
lambda profile: _config("some-model", "custom-proxy", "http://localhost:9/v1"),
)
import agent.model_metadata as metadata
def _fast_probe(model, base_url="", api_key="", config_context_length=None, provider="", custom_providers=None):
return 131072
monkeypatch.setattr(metadata, "get_model_context_length", _fast_probe)
info = router.get_model_info()
assert info["model"] == "some-model"
assert info["auto_context_length"] == 131072
assert info["effective_context_length"] == 131072