From 43c1eee1fa6f58a79f54b0770ffe8e479b9fce3a Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Fri, 25 Sep 2026 12:31:55 -0500 Subject: [PATCH] fix(web): bound the /api/model/info context-length probe get_model_info called get_model_context_length inline. The resolver chain runs several sequential provider probes, each with its own multi-second timeout, so an unreachable model.base_url held the response for tens of seconds. Bound the whole chain to a 5s budget on a throwaway worker thread; on timeout the response degrades to auto_context_length=0 while the abandoned probe finishes on its own. Fixes https://github.com/NousResearch/hermes-agent/issues/63214 (backend half) --- hermes_cli/web_routers/models.py | 44 ++++++++++-- .../test_model_info_probe_timeout.py | 67 +++++++++++++++++++ 2 files changed, 107 insertions(+), 4 deletions(-) create mode 100644 tests/hermes_cli/test_model_info_probe_timeout.py diff --git a/hermes_cli/web_routers/models.py b/hermes_cli/web_routers/models.py index e7911dfd7f..fbb516a837 100644 --- a/hermes_cli/web_routers/models.py +++ b/hermes_cli/web_routers/models.py @@ -5,6 +5,7 @@ Extracted from ``hermes_cli.web_server``; helpers/state that tests monkeypatch o """ import asyncio +import concurrent.futures import logging from typing import Optional @@ -50,6 +51,41 @@ def _load_config_scoped(profile: Optional[str]) -> dict: return load_config() +# Blocking budget for /api/model/info's context-length resolution. The resolver +# chain (agent.model_metadata.get_model_context_length) runs several sequential +# provider probes, each with its own multi-second timeout, so an unreachable or +# blackholed model.base_url can hold this response for tens of seconds — and the +# Desktop Model Settings page waits on it (#63214). +_MODEL_INFO_PROBE_BUDGET_S = 5.0 + + +def _bounded_context_length_probe(model: str, base_url: str, provider: str) -> int: + """``get_model_context_length`` with the route's blocking budget. + + On timeout the abandoned probe keeps running in its worker thread (bounded + by its own per-request timeouts) while the response degrades to + ``auto_context_length = 0`` ("auto-detected: unknown"). + """ + from agent.model_metadata import get_model_context_length + + pool = concurrent.futures.ThreadPoolExecutor(max_workers=1, thread_name_prefix="model-info-probe") + try: + return pool.submit( + get_model_context_length, model=model, base_url=base_url, provider=provider, + config_context_length=None + ).result(timeout=_MODEL_INFO_PROBE_BUDGET_S) + except concurrent.futures.TimeoutError: + _log.warning( + "GET /api/model/info: context-length probe for %r at %s exceeded %.1fs — returning unknown", + model, base_url or "", _MODEL_INFO_PROBE_BUDGET_S, + ) + return 0 + finally: + # wait=False: never block the response (or interpreter exit) on the + # abandoned probe. + pool.shutdown(wait=False) + + @router.get("/api/model/info") def get_model_info(profile: Optional[str] = None): """Resolved metadata for the configured model: auto-detected vs configured @@ -65,10 +101,10 @@ def get_model_info(profile: Optional[str] = None): return dict(_EMPTY_MODEL_INFO, provider=provider) try: - from agent.model_metadata import get_model_context_length - # config_context_length=None: ignore the override — we want the auto value - auto_ctx = get_model_context_length(model=model_name, base_url=base_url, provider=provider, - config_context_length=None) + # config_context_length=None: ignore the override — we want the auto value. + # Bounded: the resolver's provider probes can hang for tens of seconds + # when model.base_url is unreachable (#63214). + auto_ctx = _bounded_context_length_probe(model_name, base_url, provider) except Exception: auto_ctx = 0 diff --git a/tests/hermes_cli/test_model_info_probe_timeout.py b/tests/hermes_cli/test_model_info_probe_timeout.py new file mode 100644 index 0000000000..9d40a8c06e --- /dev/null +++ b/tests/hermes_cli/test_model_info_probe_timeout.py @@ -0,0 +1,67 @@ +"""/api/model/info must bound its context-length probe (#63214). + +The resolver chain (``agent.model_metadata.get_model_context_length``) runs +several sequential provider probes, each with its own multi-second timeout, so +an unreachable ``model.base_url`` held the response for tens of seconds — and +the Desktop Model Settings page, which awaits this endpoint inside one +``Promise.all``, showed loading skeletons indefinitely. +""" + +from __future__ import annotations + +import time + + +def _config(model: str, provider: str, base_url: str) -> dict: + return {"model": {"default": model, "provider": provider, "base_url": base_url}} + + +def test_model_info_degrades_when_the_context_probe_exceeds_its_budget(monkeypatch): + from hermes_cli.web_routers import models as router + + monkeypatch.setattr( + router, "_load_config_scoped", + lambda profile: _config("some-model", "custom-proxy", "http://localhost:9/v1"), + ) + monkeypatch.setattr(router, "_MODEL_INFO_PROBE_BUDGET_S", 0.2) + + import agent.model_metadata as metadata + + def _hanging_probe(model, base_url="", api_key="", config_context_length=None, provider="", custom_providers=None): + time.sleep(1.0) + raise AssertionError("the abandoned probe should never be awaited") + + monkeypatch.setattr(metadata, "get_model_context_length", _hanging_probe) + + started = time.monotonic() + info = router.get_model_info() + elapsed = time.monotonic() - started + + # The response degrades to "auto context unknown" instead of hanging… + assert info["model"] == "some-model" + assert info["provider"] == "custom-proxy" + assert info["auto_context_length"] == 0 + # …and it returns within the (shrunk) budget, not the probe's 1s sleep. + assert elapsed < 0.9 + + +def test_model_info_surfaces_the_context_value_when_the_probe_is_fast(monkeypatch): + from hermes_cli.web_routers import models as router + + monkeypatch.setattr( + router, "_load_config_scoped", + lambda profile: _config("some-model", "custom-proxy", "http://localhost:9/v1"), + ) + + import agent.model_metadata as metadata + + def _fast_probe(model, base_url="", api_key="", config_context_length=None, provider="", custom_providers=None): + return 131072 + + monkeypatch.setattr(metadata, "get_model_context_length", _fast_probe) + + info = router.get_model_info() + + assert info["model"] == "some-model" + assert info["auto_context_length"] == 131072 + assert info["effective_context_length"] == 131072