From efc947d72a26a01476e1a19ecb99d8013b8595f3 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:17:01 -0700 Subject: [PATCH] fix: send the title model call after the turn on a shared custom endpoint On a `custom` main route (llama.cpp, Ollama, vLLM, ...) whose auxiliary.title_generation is not pinned elsewhere, the turn prologue fired the `response_format: json_schema` title request on a daemon thread at the same instant as the turn's own streaming request, against the same self-hosted server. A single-slot server can decode the title grammar/completion into the main reply: the user then receives `{"title": ...}` as the assistant turn, the main loop persists it as a genuine assistant row, replays it, and the model adopts the format (#117296). No Hermes writer routes the aux response into the transcript; the leaked JSON is the main completion itself. `maybe_auto_title` now returns the upgrade thread and leaves it UNSTARTED when `title_upgrade_must_wait_for_turn(main_runtime)`; the prologue parks it on `agent._deferred_title_upgrade` and `finalize_turn` starts it once the model has answered. Hosted providers keep the turn-start timing. Usage accounting (`task='title_generation'`) and `sessions.title` are unchanged. --- agent/title_generator.py | 60 +++++++++++++++---- agent/turn_context.py | 16 ++++- agent/turn_finalizer.py | 4 ++ tests/agent/test_title_generator.py | 30 ++++++++++ ...est_turn_finalizer_iteration_limit_exit.py | 13 ++++ website/docs/user-guide/configuration.md | 7 +++ 6 files changed, 119 insertions(+), 11 deletions(-) diff --git a/agent/title_generator.py b/agent/title_generator.py index e2303ca03a..9e5073cbb8 100644 --- a/agent/title_generator.py +++ b/agent/title_generator.py @@ -177,6 +177,40 @@ def _model_title_upgrade_enabled() -> bool: return True +def title_upgrade_must_wait_for_turn(main_runtime: Optional[dict]) -> bool: + """True when the model title call would hit the SAME self-hosted endpoint as the turn's own request. + + A ``custom`` main route (llama.cpp, Ollama, vLLM, LM Studio…) whose ``auxiliary.title_generation`` + is not pinned elsewhere shares one local server between the streaming main request and the + concurrent ``response_format: json_schema`` title request. Single-slot servers then serve the + title grammar/completion into the main turn: the user's reply arrives as ``{"title": ...}``, is + persisted as a genuine assistant row and replayed, and the model adopts the format (#117296). + Running the title call after the turn settles keeps the two requests off the wire at once. + Hosted providers multiplex requests independently and keep the turn-start timing. + """ + provider = str((main_runtime or {}).get("provider") or "").strip().lower() + if provider != "custom": + return False + try: + cfg = _title_config() + except Exception: + return True + pinned_provider = str(cfg.get("provider") or "").strip().lower() + pinned_base_url = str(cfg.get("base_url") or "").strip().rstrip("/") + main_base_url = str((main_runtime or {}).get("base_url") or "").strip().rstrip("/") + if pinned_provider and pinned_provider not in ("", "auto", "custom"): + return False + return not pinned_base_url or pinned_base_url == main_base_url + + +def start_title_upgrade(upgrade: Optional[threading.Thread]) -> None: + """Start a (deferred) title upgrade thread; joinable via ``wait_for_title_upgrades`` only once started.""" + if upgrade is None or upgrade.ident is not None: + return + _UPGRADE_THREADS.add(upgrade) + upgrade.start() + + def strip_control_wrappers(text: str) -> str: """Remove leading control wrappers (nested too) so a slash-command turn reduces to the prose the user typed.""" current = (text or "").strip() @@ -628,10 +662,13 @@ def maybe_auto_title( title_callback: Optional[TitleCallback] = None, runtime_validator: Optional[RuntimeValidator] = None, title_preview: str | None = None, -) -> None: - """Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model.""" +) -> Optional[threading.Thread]: + """Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model. + + Returns the upgrade thread: already started, or — when ``title_upgrade_must_wait_for_turn`` — left + UNSTARTED for the caller to hand to ``start_title_upgrade`` once the turn's model request settled.""" if not session_db or not session_id or not user_message: - return + return None # History may be pre- or post-message. Past the opening turn, skip once the session holds an # ``llm``/``user`` name: count alone left a machinery-opened session nameless, and a ``derived`` # name is still a placeholder (instant slice, or the model's greeting title for a bare "hi") that @@ -642,7 +679,7 @@ def maybe_auto_title( _has_upgraded_title(session_db, session_id) or (user_msg_count > 3 and not _session_is_untitled(session_db, session_id)) ): - return + return None kanban_title = _kanban_task_title() if kanban_title: # The card already carries a human-written name; an auxiliary model call per spawned worker @@ -652,16 +689,16 @@ def maybe_auto_title( persisted = _persist_session_title(session_db, session_id, kanban_title, source="llm") if persisted: _notify_title(title_callback, persisted, "llm", "Kanban task title") - return + return None if not is_titleable_user_message(user_message): - return + return None if not _auto_title_enabled(): # config read after the cheap guards so the file isn't touched every turn logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false") - return + return None apply_instant_title(session_db, session_id, user_message, title_callback, title_preview=title_preview) if not _model_title_upgrade_enabled(): logger.debug("Instant title persisted; model upgrade disabled by auxiliary.title_generation.model_upgrade_enabled=false") - return + return None # The thread must resolve auxiliary.title_generation (config, provider key, language) for the # profile whose turn this is: a bare Thread starts with an empty context and lands on the launch # profile under multiplex, titling X's session with the default profile's model and billing its key. @@ -675,5 +712,8 @@ def maybe_auto_title( args=(session_db, session_id, user_message), kwargs=upgrade_kwargs, ) - _UPGRADE_THREADS.add(upgrade) - upgrade.start() + if title_upgrade_must_wait_for_turn(main_runtime): + logger.debug("Auto-title upgrade deferred past the turn: shares the custom endpoint with the main request") + return upgrade + start_title_upgrade(upgrade) + return upgrade diff --git a/agent/turn_context.py b/agent/turn_context.py index 616edceba6..64f02e7c9a 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -193,7 +193,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: for k in ("model", "provider", "base_url", "api_key", "api_mode", "session_id") } # See #19027. - maybe_auto_title( + upgrade = maybe_auto_title( session_db, session_id, user_text, @@ -210,10 +210,24 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: ), title_preview=title_preview, ) + # Unstarted = the title call would share a self-hosted endpoint with this turn's request + # (#117296); ``finalize_turn`` starts it once the model has answered. + if upgrade is not None and upgrade.ident is None: + agent._deferred_title_upgrade = upgrade except Exception: logger.debug("Turn-start auto-title dispatch failed", exc_info=True) +def start_deferred_title_upgrade(agent: Any) -> None: + """Fire the title upgrade ``_maybe_title_session_at_turn_start`` held back; no-op when none.""" + upgrade = getattr(agent, "_deferred_title_upgrade", None) + if upgrade is None: + return + agent._deferred_title_upgrade = None + from agent.title_generator import start_title_upgrade + start_title_upgrade(upgrade) + + def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> int: """Locate this turn's user message after compaction rebuilt ``messages``. diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index 92663289b8..e1c2de569e 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -483,6 +483,10 @@ def finalize_turn( _rollback_interrupted_preflight_display(agent, interrupted) _cleanup_errors: List[str] = [] + # The model has answered (or the loop gave up): a title upgrade held back because it shares a + # self-hosted endpoint with the main request (#117296) may go out now. + from agent.turn_context import start_deferred_title_upgrade + _guarded_cleanup("start_deferred_title_upgrade", lambda: start_deferred_title_upgrade(agent), _cleanup_errors, logger) # ``user_message`` may be a multimodal list of parts; the trajectory format wants a string. _guarded_cleanup( "save_trajectory", diff --git a/tests/agent/test_title_generator.py b/tests/agent/test_title_generator.py index 890bb32197..8c6e1783ab 100644 --- a/tests/agent/test_title_generator.py +++ b/tests/agent/test_title_generator.py @@ -448,6 +448,36 @@ class TestMaybeAutoTitle: runtime_validator=None, ) + @pytest.mark.parametrize( + "main_runtime, title_cfg, deferred", + [ + ({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {}, True), + ({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://127.0.0.1:8080/v1/"}, True), + ({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"provider": "openrouter"}, False), + ({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://10.0.0.2:8080/v1"}, False), + ({"provider": "openrouter", "base_url": "https://openrouter.ai/api/v1"}, {}, False), + ], + ) + def test_title_call_waits_for_the_turn_when_it_shares_a_custom_endpoint(self, main_runtime, title_cfg, deferred): + """#117296: a self-hosted server serving the main turn and the concurrent json_schema title request + can decode the title into the main reply. The upgrade must not go on the wire until the caller starts + it after the turn; every other route keeps the turn-start timing.""" + import threading + from agent import title_generator as tg + db = MagicMock() + db.get_session_title.return_value = None + started = threading.Event() + with patch.object(tg, "_title_config", return_value=title_cfg), \ + patch.object(tg, "auto_title_session", side_effect=lambda *a, **k: started.set()): + upgrade = maybe_auto_title(db, "sess-1", "hello", [{"role": "user", "content": "hello"}], main_runtime=main_runtime) + assert isinstance(upgrade, threading.Thread) + if deferred: + assert upgrade.ident is None and not started.wait(0.3), "title request went out during the turn" + assert upgrade not in tg._UPGRADE_THREADS # join-before-start would raise in wait_for_title_upgrades + tg.start_title_upgrade(upgrade) + assert started.wait(timeout=10), "auto_title thread never ran" + assert upgrade in tg._UPGRADE_THREADS + def test_kanban_worker_is_named_after_its_card_without_the_llm_thread(self, tmp_path, monkeypatch): """A worker's session takes the board card's title synchronously; no auxiliary model call (#111166).""" from hermes_cli import kanban_db, kanban_db_connect diff --git a/tests/agent/test_turn_finalizer_iteration_limit_exit.py b/tests/agent/test_turn_finalizer_iteration_limit_exit.py index caa76e7eb9..494fc9d49c 100644 --- a/tests/agent/test_turn_finalizer_iteration_limit_exit.py +++ b/tests/agent/test_turn_finalizer_iteration_limit_exit.py @@ -407,3 +407,16 @@ def test_budget_exhausted_child_does_not_record_parent_kanban_timeout(monkeypatc ) record.assert_not_called() + + +def test_finalize_turn_starts_the_title_upgrade_the_prologue_held_back(): + """#117296: the turn prologue leaves a same-endpoint title upgrade unstarted on the agent; the finalizer + is the only place that may start it, and only once the model request is done.""" + import threading + + ran = threading.Event() + agent = _LimitAgent() + agent._deferred_title_upgrade = threading.Thread(target=ran.set, daemon=True) + _finalize(agent, final_response="done", exit_reason="text_response(1)", api_call_count=1) + assert ran.wait(timeout=5), "deferred title upgrade never started" + assert agent._deferred_title_upgrade is None diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 222ce31b48..c36d55a482 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1407,6 +1407,13 @@ No background `auto-title` thread starts and no automatic title-model request is explicit repair command `hermes sessions retitle-skills` still calls the model. `enabled: false` still disables both stages. +On a `custom` main provider (llama.cpp, Ollama, vLLM, LM Studio and other self-hosted +OpenAI-compatible servers) the title model call is sent **after** the turn's reply has +arrived, not concurrently with it, unless `auxiliary.title_generation` is pinned to another +provider or `base_url`. A single-slot local server that receives the `json_schema` title +request while decoding the reply can otherwise answer the reply with `{"title": ...}`, +which is then stored and replayed as the assistant's turn. + In Hermes Desktop, a plain-text paste over 3,000 characters becomes a generated `.txt` attachment. The first ~1,000 characters of that paste are handed to the title stages as a title-only hint (the agent turn still sees only the attachment reference), so a "summarize