fix(local-runtime): resumed llamacpp sessions follow the live managed port on every surface (#114336)

* test(local-runtime): pin resume to the live managed llama.cpp port

A session that stored last boot's loopback URL must not keep the client
on a dead ephemeral port after the supervisor moves.

* fix(local-runtime): follow the live managed llama.cpp port on resume

Sessions persist last boot's loopback URL, so a supervisor port change
left the client on a dead endpoint. Drop that snapshot for llamacpp
and keep the live supervisor URL.

* fix(tui_gateway): drop the llamacpp snapshot URL at one seam

_resolve_agent_model_runtime already discards a persisted base_url when the
resolution came from the local runtime; the second blank in
_stored_session_runtime_overrides (wrapped in a try/except around an import
and a string compare) duplicated it.

* fix(cli): keep a launch-time --base-url on a same-provider llamacpp resume

Re-resolving the managed endpoint is for the snapshot URL a session persisted;
an explicit --base-url for the provider the session already ran on is user
intent and stays in charge. Also hand target_model to the resolver like the
provider-changed branch does.

* fix(gateway): rehydrated llamacpp overrides follow the live managed port

Same bug class as the CLI and TUI resume paths: after a gateway restart the
persisted /model override kept last boot's loopback URL over the freshly
resolved managed endpoint, so a supervisor that came back on an ephemeral port
(18434 busy) left the session on connection errors.

* docs(local-runtime): port-fallback warning no longer asks for a model re-pick

---------

Co-authored-by: xxxigm <tuancanhnguyen706@gmail.com>
This commit is contained in:
Austin Pickett
2026-09-17 14:24:53 -04:00
committed by GitHub
parent 27dfbe397f
commit 1792e8bf5f
7 changed files with 124 additions and 7 deletions

View File

@@ -17,6 +17,7 @@ from gateway.config import Platform
from gateway.session import SessionSource, build_session_context_prompt
from gateway.run_shutdown import _log_suppressed
from hermes_cli.config import cfg_get
from hermes_cli.local_runtime.endpoint import LLAMACPP_ALIASES
if TYPE_CHECKING: # string annotations only; never imported at runtime (cycle)
from gateway.run import GatewayRunner # noqa: F401
@@ -166,7 +167,9 @@ class GatewayAgentCacheMixin:
override[k] = runtime.get(k)
override["request_overrides"] = dict(runtime.get("request_overrides") or {})
override["capabilities"] = dict(runtime.get("capabilities") or {})
if not override.get("base_url"):
if not override.get("base_url") or provider.strip().lower() in LLAMACPP_ALIASES:
# The managed llama.cpp supervisor owns its live port; a persisted loopback URL from a
# boot that fell back to an ephemeral port would strand the session on a dead endpoint.
override["base_url"] = runtime.get("base_url")
except Exception:
logger.debug(

View File

@@ -420,15 +420,40 @@ class CLIModelSwitchMixin:
if route is None:
return
stored_model, stored_provider, stored_base_url, stored_api_mode, provider_changed = route
from hermes_cli.local_runtime.endpoint import LLAMACPP_ALIASES
managed = str(stored_provider or "").strip().lower() in LLAMACPP_ALIASES
self.model = stored_model
if stored_provider:
self.provider = stored_provider
self.requested_provider = stored_provider
if stored_base_url:
if stored_base_url and not managed:
self.base_url = stored_base_url
if stored_api_mode:
self.api_mode = stored_api_mode
if provider_changed:
if managed and not (getattr(self, "_explicit_base_url", None) and not provider_changed):
# The supervisor owns the live port: last boot's loopback URL (an ephemeral fallback when
# 18434 was busy) must not pin the resume onto a dead endpoint. A launch-time --base-url
# for this same provider is user intent and keeps winning.
self._explicit_api_key = None
self._explicit_base_url = None
try:
from hermes_cli.runtime_provider import resolve_runtime_provider
resolved = resolve_runtime_provider(requested=stored_provider, target_model=self.model or None)
if resolved.get("api_key"):
self.api_key = resolved["api_key"]
self._credential_pool = resolved.get("credential_pool")
if resolved.get("base_url"):
self.base_url = resolved["base_url"]
if not stored_api_mode and resolved.get("api_mode"):
self.api_mode = resolved["api_mode"]
except Exception:
if stored_base_url:
self.base_url = stored_base_url
logger.debug(
"Credential re-resolution for resumed session provider "
"%s failed; keeping ambient credentials",
stored_provider, exc_info=True)
elif provider_changed:
# Launch-time explicit overrides belong to the AMBIENT provider and would poison
# _ensure_runtime_credentials for the restored one. api_key is never persisted to
# the session DB — runtime provider resolution owns credentials.

View File

@@ -33,9 +33,10 @@ TOUCH_EXPECT = "paris"
_RESTART_BACKOFF_S = (1, 5, 15, 60)
_RESIDENT = ("loaded", "ready")
# Chosen once and reused across restarts: sessions persist the resolved base_url, so an ephemeral
# port would strand every resumed session after each restart. Deliberately NOT 8080 so we never
# collide with a user's own llama-server/Ollama-adjacent stack.
# Chosen once and reused across restarts: sessions persist the resolved base_url as a snapshot, and
# every resume path re-resolves llamacpp-alias sessions to the live endpoint (a stale port is
# recoverable, but a stable one keeps external tooling pointed at the right place). Deliberately NOT
# 8080 so we never collide with a user's own llama-server/Ollama-adjacent stack.
_DEFAULT_PORT = 18434
@@ -67,7 +68,7 @@ def _stable_port() -> int:
except OSError:
logger.warning(
"port %d busy; managed llama-server falling back to an ephemeral "
"port — existing sessions may need a model re-pick", _DEFAULT_PORT)
"port — resumed sessions follow the live endpoint", _DEFAULT_PORT)
return _free_port()

View File

@@ -141,6 +141,27 @@ def test_runner_rehydrates_override_after_restart(store_factory):
assert route["runtime"]["capabilities"] == {"openai_native_compaction": True}
def test_rehydrate_llamacpp_override_follows_live_managed_port(store_factory):
"""The managed llama.cpp supervisor may come back on an ephemeral port (18434 busy). The persisted
loopback URL is a snapshot of the previous boot, so rehydration must take the live endpoint."""
store = store_factory()
session_key = store.get_or_create_session(_make_source()).session_key
store.set_model_override(session_key, {
"model": "Local.Model-Q4_K_M", "provider": "llamacpp", "base_url": "http://127.0.0.1:51489/v1"})
runner = _make_runner(store_factory())
with patch(
"gateway.run._resolve_runtime_agent_kwargs_for_provider",
return_value={"api_key": "local-key", "base_url": "http://127.0.0.1:18434/v1",
"provider": "custom", "requested_provider": "llamacpp"},
):
runner._rehydrate_session_model_override(session_key)
override = runner._session_model_overrides[session_key]
assert override["base_url"] == "http://127.0.0.1:18434/v1"
assert override["api_key"] == "local-key"
def test_sanitize_model_override():
assert sanitize_model_override(None) is None
assert sanitize_model_override({}) is None

View File

@@ -82,6 +82,52 @@ def test_restore_session_model_restores_model_and_provider():
assert stub._explicit_base_url == "https://f/v1"
def test_restore_llamacpp_session_follows_live_managed_endpoint(monkeypatch):
"""Managed llama.cpp owns its live port. A snapshot of last boot's loopback
URL must not pin the resumed client to a dead ephemeral endpoint."""
live = {
"provider": "custom",
"base_url": "http://127.0.0.1:18434/v1",
"api_key": "sk-live-managed",
"api_mode": "chat_completions",
}
monkeypatch.setattr(
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda **_kw: live)
stub = _make_stub()
stub._restore_session_model(_row(
model="Qwen3.8-27B-UD-IQ3_XXS",
model_config={
"provider": "llamacpp",
"base_url": "http://127.0.0.1:51489/v1",
"api_mode": "chat_completions",
}))
assert stub.provider == "llamacpp"
assert stub.requested_provider == "llamacpp"
assert stub.base_url == live["base_url"]
assert stub._explicit_base_url in (None, "")
assert stub.api_key == live["api_key"]
def test_restore_llamacpp_session_keeps_launch_base_url_for_same_provider(monkeypatch):
"""`hermes --provider llamacpp --base-url X --resume` is user intent for the SAME provider the
session ran on; the live-endpoint re-resolution must not re-point it at the local supervisor."""
calls = []
monkeypatch.setattr(
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda **kw: calls.append(kw) or {"base_url": "http://127.0.0.1:18434/v1"})
user_url = "http://gpu-box:8080/v1"
stub = _make_stub(provider="llamacpp", requested_provider="llamacpp", base_url=user_url,
_explicit_base_url=user_url)
stub._restore_session_model(_row(
model="Qwen3.8-27B-UD-IQ3_XXS",
model_config={"provider": "llamacpp", "base_url": "http://127.0.0.1:51489/v1"}))
assert stub.model == "Qwen3.8-27B-UD-IQ3_XXS"
assert stub.base_url == user_url
assert stub._explicit_base_url == user_url
assert calls == []
def test_restore_session_model_explicit_cli_flag_wins():
stub = _make_stub(model="cli-flag-model", _explicit_model_override=True)
stub._restore_session_model(_row())

View File

@@ -57,6 +57,24 @@ def test_live_local_identity_survives_new_chat_and_resume(local_route):
assert server._session_info(agent, session)["provider"] == "anthropic"
def test_resume_follows_live_managed_port_not_snapshot(local_route):
"""A llamacpp session that stored last boot's loopback URL must follow the
live supervisor after the managed server moves to another port."""
route, _session = local_route
model = "Local.Model-Q4_K_M"
stale = "http://127.0.0.1:51489/v1"
row = {"model": model, "model_config": {
"model": model, "provider": "llamacpp", "base_url": stale,
"api_mode": "chat_completions"}}
overrides = server._stored_session_runtime_overrides(row)
restored_model, restored = server._resolve_agent_model_runtime(
overrides["model_override"], overrides.get("provider_override"))
assert restored_model == model
assert restored["base_url"] == route["base_url"]
assert restored["base_url"] != stale
assert restored["api_key"] == route["api_key"]
def test_session_info_recovers_identity_from_the_owning_profile(tmp_path, monkeypatch):
import json
from pathlib import Path

View File

@@ -2241,6 +2241,9 @@ def _resolve_agent_model_runtime(model_override, provider_override) -> tuple[str
if not resolution.selected_model:
raise RuntimeError("Auth fallback resolved without a model")
return resolution.selected_model, resolution.runtime
if resolution.runtime.get("source") == "local-runtime":
# Live supervisor beat any persisted loopback URL for this identity.
overrides.pop("base_url", None)
resolution.runtime.update({k: v for k, v in overrides.items() if v})
return model, resolution.runtime