fix: send the title model call after the turn on a shared custom endpoint

On a `custom` main route (llama.cpp, Ollama, vLLM, ...) whose
auxiliary.title_generation is not pinned elsewhere, the turn prologue fired the
`response_format: json_schema` title request on a daemon thread at the same
instant as the turn's own streaming request, against the same self-hosted
server. A single-slot server can decode the title grammar/completion into the
main reply: the user then receives `{"title": ...}` as the assistant turn, the
main loop persists it as a genuine assistant row, replays it, and the model
adopts the format (#117296). No Hermes writer routes the aux response into the
transcript; the leaked JSON is the main completion itself.

`maybe_auto_title` now returns the upgrade thread and leaves it UNSTARTED when
`title_upgrade_must_wait_for_turn(main_runtime)`; the prologue parks it on
`agent._deferred_title_upgrade` and `finalize_turn` starts it once the model
has answered. Hosted providers keep the turn-start timing. Usage accounting
(`task='title_generation'`) and `sessions.title` are unchanged.
This commit is contained in:
teknium1
2026-09-20 13:17:01 -07:00
committed by Teknium
parent 3e579ee7af
commit efc947d72a
6 changed files with 119 additions and 11 deletions

View File

@@ -177,6 +177,40 @@ def _model_title_upgrade_enabled() -> bool:
return True
def title_upgrade_must_wait_for_turn(main_runtime: Optional[dict]) -> bool:
"""True when the model title call would hit the SAME self-hosted endpoint as the turn's own request.
A ``custom`` main route (llama.cpp, Ollama, vLLM, LM Studio…) whose ``auxiliary.title_generation``
is not pinned elsewhere shares one local server between the streaming main request and the
concurrent ``response_format: json_schema`` title request. Single-slot servers then serve the
title grammar/completion into the main turn: the user's reply arrives as ``{"title": ...}``, is
persisted as a genuine assistant row and replayed, and the model adopts the format (#117296).
Running the title call after the turn settles keeps the two requests off the wire at once.
Hosted providers multiplex requests independently and keep the turn-start timing.
"""
provider = str((main_runtime or {}).get("provider") or "").strip().lower()
if provider != "custom":
return False
try:
cfg = _title_config()
except Exception:
return True
pinned_provider = str(cfg.get("provider") or "").strip().lower()
pinned_base_url = str(cfg.get("base_url") or "").strip().rstrip("/")
main_base_url = str((main_runtime or {}).get("base_url") or "").strip().rstrip("/")
if pinned_provider and pinned_provider not in ("", "auto", "custom"):
return False
return not pinned_base_url or pinned_base_url == main_base_url
def start_title_upgrade(upgrade: Optional[threading.Thread]) -> None:
"""Start a (deferred) title upgrade thread; joinable via ``wait_for_title_upgrades`` only once started."""
if upgrade is None or upgrade.ident is not None:
return
_UPGRADE_THREADS.add(upgrade)
upgrade.start()
def strip_control_wrappers(text: str) -> str:
"""Remove leading control wrappers (nested too) so a slash-command turn reduces to the prose the user typed."""
current = (text or "").strip()
@@ -628,10 +662,13 @@ def maybe_auto_title(
title_callback: Optional[TitleCallback] = None,
runtime_validator: Optional[RuntimeValidator] = None,
title_preview: str | None = None,
) -> None:
"""Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model."""
) -> Optional[threading.Thread]:
"""Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model.
Returns the upgrade thread: already started, or — when ``title_upgrade_must_wait_for_turn`` — left
UNSTARTED for the caller to hand to ``start_title_upgrade`` once the turn's model request settled."""
if not session_db or not session_id or not user_message:
return
return None
# History may be pre- or post-message. Past the opening turn, skip once the session holds an
# ``llm``/``user`` name: count alone left a machinery-opened session nameless, and a ``derived``
# name is still a placeholder (instant slice, or the model's greeting title for a bare "hi") that
@@ -642,7 +679,7 @@ def maybe_auto_title(
_has_upgraded_title(session_db, session_id)
or (user_msg_count > 3 and not _session_is_untitled(session_db, session_id))
):
return
return None
kanban_title = _kanban_task_title()
if kanban_title:
# The card already carries a human-written name; an auxiliary model call per spawned worker
@@ -652,16 +689,16 @@ def maybe_auto_title(
persisted = _persist_session_title(session_db, session_id, kanban_title, source="llm")
if persisted:
_notify_title(title_callback, persisted, "llm", "Kanban task title")
return
return None
if not is_titleable_user_message(user_message):
return
return None
if not _auto_title_enabled(): # config read after the cheap guards so the file isn't touched every turn
logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false")
return
return None
apply_instant_title(session_db, session_id, user_message, title_callback, title_preview=title_preview)
if not _model_title_upgrade_enabled():
logger.debug("Instant title persisted; model upgrade disabled by auxiliary.title_generation.model_upgrade_enabled=false")
return
return None
# The thread must resolve auxiliary.title_generation (config, provider key, language) for the
# profile whose turn this is: a bare Thread starts with an empty context and lands on the launch
# profile under multiplex, titling X's session with the default profile's model and billing its key.
@@ -675,5 +712,8 @@ def maybe_auto_title(
args=(session_db, session_id, user_message),
kwargs=upgrade_kwargs,
)
_UPGRADE_THREADS.add(upgrade)
upgrade.start()
if title_upgrade_must_wait_for_turn(main_runtime):
logger.debug("Auto-title upgrade deferred past the turn: shares the custom endpoint with the main request")
return upgrade
start_title_upgrade(upgrade)
return upgrade

View File

@@ -193,7 +193,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
for k in ("model", "provider", "base_url", "api_key", "api_mode", "session_id")
}
# See #19027.
maybe_auto_title(
upgrade = maybe_auto_title(
session_db,
session_id,
user_text,
@@ -210,10 +210,24 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
),
title_preview=title_preview,
)
# Unstarted = the title call would share a self-hosted endpoint with this turn's request
# (#117296); ``finalize_turn`` starts it once the model has answered.
if upgrade is not None and upgrade.ident is None:
agent._deferred_title_upgrade = upgrade
except Exception:
logger.debug("Turn-start auto-title dispatch failed", exc_info=True)
def start_deferred_title_upgrade(agent: Any) -> None:
"""Fire the title upgrade ``_maybe_title_session_at_turn_start`` held back; no-op when none."""
upgrade = getattr(agent, "_deferred_title_upgrade", None)
if upgrade is None:
return
agent._deferred_title_upgrade = None
from agent.title_generator import start_title_upgrade
start_title_upgrade(upgrade)
def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> int:
"""Locate this turn's user message after compaction rebuilt ``messages``.

View File

@@ -483,6 +483,10 @@ def finalize_turn(
_rollback_interrupted_preflight_display(agent, interrupted)
_cleanup_errors: List[str] = []
# The model has answered (or the loop gave up): a title upgrade held back because it shares a
# self-hosted endpoint with the main request (#117296) may go out now.
from agent.turn_context import start_deferred_title_upgrade
_guarded_cleanup("start_deferred_title_upgrade", lambda: start_deferred_title_upgrade(agent), _cleanup_errors, logger)
# ``user_message`` may be a multimodal list of parts; the trajectory format wants a string.
_guarded_cleanup(
"save_trajectory",

View File

@@ -448,6 +448,36 @@ class TestMaybeAutoTitle:
runtime_validator=None,
)
@pytest.mark.parametrize(
"main_runtime, title_cfg, deferred",
[
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {}, True),
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://127.0.0.1:8080/v1/"}, True),
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"provider": "openrouter"}, False),
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://10.0.0.2:8080/v1"}, False),
({"provider": "openrouter", "base_url": "https://openrouter.ai/api/v1"}, {}, False),
],
)
def test_title_call_waits_for_the_turn_when_it_shares_a_custom_endpoint(self, main_runtime, title_cfg, deferred):
"""#117296: a self-hosted server serving the main turn and the concurrent json_schema title request
can decode the title into the main reply. The upgrade must not go on the wire until the caller starts
it after the turn; every other route keeps the turn-start timing."""
import threading
from agent import title_generator as tg
db = MagicMock()
db.get_session_title.return_value = None
started = threading.Event()
with patch.object(tg, "_title_config", return_value=title_cfg), \
patch.object(tg, "auto_title_session", side_effect=lambda *a, **k: started.set()):
upgrade = maybe_auto_title(db, "sess-1", "hello", [{"role": "user", "content": "hello"}], main_runtime=main_runtime)
assert isinstance(upgrade, threading.Thread)
if deferred:
assert upgrade.ident is None and not started.wait(0.3), "title request went out during the turn"
assert upgrade not in tg._UPGRADE_THREADS # join-before-start would raise in wait_for_title_upgrades
tg.start_title_upgrade(upgrade)
assert started.wait(timeout=10), "auto_title thread never ran"
assert upgrade in tg._UPGRADE_THREADS
def test_kanban_worker_is_named_after_its_card_without_the_llm_thread(self, tmp_path, monkeypatch):
"""A worker's session takes the board card's title synchronously; no auxiliary model call (#111166)."""
from hermes_cli import kanban_db, kanban_db_connect

View File

@@ -407,3 +407,16 @@ def test_budget_exhausted_child_does_not_record_parent_kanban_timeout(monkeypatc
)
record.assert_not_called()
def test_finalize_turn_starts_the_title_upgrade_the_prologue_held_back():
"""#117296: the turn prologue leaves a same-endpoint title upgrade unstarted on the agent; the finalizer
is the only place that may start it, and only once the model request is done."""
import threading
ran = threading.Event()
agent = _LimitAgent()
agent._deferred_title_upgrade = threading.Thread(target=ran.set, daemon=True)
_finalize(agent, final_response="done", exit_reason="text_response(1)", api_call_count=1)
assert ran.wait(timeout=5), "deferred title upgrade never started"
assert agent._deferred_title_upgrade is None

View File

@@ -1407,6 +1407,13 @@ No background `auto-title` thread starts and no automatic title-model request is
explicit repair command `hermes sessions retitle-skills` still calls the model. `enabled: false`
still disables both stages.
On a `custom` main provider (llama.cpp, Ollama, vLLM, LM Studio and other self-hosted
OpenAI-compatible servers) the title model call is sent **after** the turn's reply has
arrived, not concurrently with it, unless `auxiliary.title_generation` is pinned to another
provider or `base_url`. A single-slot local server that receives the `json_schema` title
request while decoding the reply can otherwise answer the reply with `{"title": ...}`,
which is then stored and replayed as the assistant's turn.
In Hermes Desktop, a plain-text paste over 3,000 characters becomes a generated `.txt`
attachment. The first ~1,000 characters of that paste are handed to the title stages as a
title-only hint (the agent turn still sees only the attachment reference), so a "summarize