fix: send the title model call after the turn on a shared custom endpoint
On a `custom` main route (llama.cpp, Ollama, vLLM, ...) whose
auxiliary.title_generation is not pinned elsewhere, the turn prologue fired the
`response_format: json_schema` title request on a daemon thread at the same
instant as the turn's own streaming request, against the same self-hosted
server. A single-slot server can decode the title grammar/completion into the
main reply: the user then receives `{"title": ...}` as the assistant turn, the
main loop persists it as a genuine assistant row, replays it, and the model
adopts the format (#117296). No Hermes writer routes the aux response into the
transcript; the leaked JSON is the main completion itself.
`maybe_auto_title` now returns the upgrade thread and leaves it UNSTARTED when
`title_upgrade_must_wait_for_turn(main_runtime)`; the prologue parks it on
`agent._deferred_title_upgrade` and `finalize_turn` starts it once the model
has answered. Hosted providers keep the turn-start timing. Usage accounting
(`task='title_generation'`) and `sessions.title` are unchanged.
This commit is contained in:
@@ -177,6 +177,40 @@ def _model_title_upgrade_enabled() -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def title_upgrade_must_wait_for_turn(main_runtime: Optional[dict]) -> bool:
|
||||
"""True when the model title call would hit the SAME self-hosted endpoint as the turn's own request.
|
||||
|
||||
A ``custom`` main route (llama.cpp, Ollama, vLLM, LM Studio…) whose ``auxiliary.title_generation``
|
||||
is not pinned elsewhere shares one local server between the streaming main request and the
|
||||
concurrent ``response_format: json_schema`` title request. Single-slot servers then serve the
|
||||
title grammar/completion into the main turn: the user's reply arrives as ``{"title": ...}``, is
|
||||
persisted as a genuine assistant row and replayed, and the model adopts the format (#117296).
|
||||
Running the title call after the turn settles keeps the two requests off the wire at once.
|
||||
Hosted providers multiplex requests independently and keep the turn-start timing.
|
||||
"""
|
||||
provider = str((main_runtime or {}).get("provider") or "").strip().lower()
|
||||
if provider != "custom":
|
||||
return False
|
||||
try:
|
||||
cfg = _title_config()
|
||||
except Exception:
|
||||
return True
|
||||
pinned_provider = str(cfg.get("provider") or "").strip().lower()
|
||||
pinned_base_url = str(cfg.get("base_url") or "").strip().rstrip("/")
|
||||
main_base_url = str((main_runtime or {}).get("base_url") or "").strip().rstrip("/")
|
||||
if pinned_provider and pinned_provider not in ("", "auto", "custom"):
|
||||
return False
|
||||
return not pinned_base_url or pinned_base_url == main_base_url
|
||||
|
||||
|
||||
def start_title_upgrade(upgrade: Optional[threading.Thread]) -> None:
|
||||
"""Start a (deferred) title upgrade thread; joinable via ``wait_for_title_upgrades`` only once started."""
|
||||
if upgrade is None or upgrade.ident is not None:
|
||||
return
|
||||
_UPGRADE_THREADS.add(upgrade)
|
||||
upgrade.start()
|
||||
|
||||
|
||||
def strip_control_wrappers(text: str) -> str:
|
||||
"""Remove leading control wrappers (nested too) so a slash-command turn reduces to the prose the user typed."""
|
||||
current = (text or "").strip()
|
||||
@@ -628,10 +662,13 @@ def maybe_auto_title(
|
||||
title_callback: Optional[TitleCallback] = None,
|
||||
runtime_validator: Optional[RuntimeValidator] = None,
|
||||
title_preview: str | None = None,
|
||||
) -> None:
|
||||
"""Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model."""
|
||||
) -> Optional[threading.Thread]:
|
||||
"""Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model.
|
||||
|
||||
Returns the upgrade thread: already started, or — when ``title_upgrade_must_wait_for_turn`` — left
|
||||
UNSTARTED for the caller to hand to ``start_title_upgrade`` once the turn's model request settled."""
|
||||
if not session_db or not session_id or not user_message:
|
||||
return
|
||||
return None
|
||||
# History may be pre- or post-message. Past the opening turn, skip once the session holds an
|
||||
# ``llm``/``user`` name: count alone left a machinery-opened session nameless, and a ``derived``
|
||||
# name is still a placeholder (instant slice, or the model's greeting title for a bare "hi") that
|
||||
@@ -642,7 +679,7 @@ def maybe_auto_title(
|
||||
_has_upgraded_title(session_db, session_id)
|
||||
or (user_msg_count > 3 and not _session_is_untitled(session_db, session_id))
|
||||
):
|
||||
return
|
||||
return None
|
||||
kanban_title = _kanban_task_title()
|
||||
if kanban_title:
|
||||
# The card already carries a human-written name; an auxiliary model call per spawned worker
|
||||
@@ -652,16 +689,16 @@ def maybe_auto_title(
|
||||
persisted = _persist_session_title(session_db, session_id, kanban_title, source="llm")
|
||||
if persisted:
|
||||
_notify_title(title_callback, persisted, "llm", "Kanban task title")
|
||||
return
|
||||
return None
|
||||
if not is_titleable_user_message(user_message):
|
||||
return
|
||||
return None
|
||||
if not _auto_title_enabled(): # config read after the cheap guards so the file isn't touched every turn
|
||||
logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false")
|
||||
return
|
||||
return None
|
||||
apply_instant_title(session_db, session_id, user_message, title_callback, title_preview=title_preview)
|
||||
if not _model_title_upgrade_enabled():
|
||||
logger.debug("Instant title persisted; model upgrade disabled by auxiliary.title_generation.model_upgrade_enabled=false")
|
||||
return
|
||||
return None
|
||||
# The thread must resolve auxiliary.title_generation (config, provider key, language) for the
|
||||
# profile whose turn this is: a bare Thread starts with an empty context and lands on the launch
|
||||
# profile under multiplex, titling X's session with the default profile's model and billing its key.
|
||||
@@ -675,5 +712,8 @@ def maybe_auto_title(
|
||||
args=(session_db, session_id, user_message),
|
||||
kwargs=upgrade_kwargs,
|
||||
)
|
||||
_UPGRADE_THREADS.add(upgrade)
|
||||
upgrade.start()
|
||||
if title_upgrade_must_wait_for_turn(main_runtime):
|
||||
logger.debug("Auto-title upgrade deferred past the turn: shares the custom endpoint with the main request")
|
||||
return upgrade
|
||||
start_title_upgrade(upgrade)
|
||||
return upgrade
|
||||
|
||||
@@ -193,7 +193,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
|
||||
for k in ("model", "provider", "base_url", "api_key", "api_mode", "session_id")
|
||||
}
|
||||
# See #19027.
|
||||
maybe_auto_title(
|
||||
upgrade = maybe_auto_title(
|
||||
session_db,
|
||||
session_id,
|
||||
user_text,
|
||||
@@ -210,10 +210,24 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None:
|
||||
),
|
||||
title_preview=title_preview,
|
||||
)
|
||||
# Unstarted = the title call would share a self-hosted endpoint with this turn's request
|
||||
# (#117296); ``finalize_turn`` starts it once the model has answered.
|
||||
if upgrade is not None and upgrade.ident is None:
|
||||
agent._deferred_title_upgrade = upgrade
|
||||
except Exception:
|
||||
logger.debug("Turn-start auto-title dispatch failed", exc_info=True)
|
||||
|
||||
|
||||
def start_deferred_title_upgrade(agent: Any) -> None:
|
||||
"""Fire the title upgrade ``_maybe_title_session_at_turn_start`` held back; no-op when none."""
|
||||
upgrade = getattr(agent, "_deferred_title_upgrade", None)
|
||||
if upgrade is None:
|
||||
return
|
||||
agent._deferred_title_upgrade = None
|
||||
from agent.title_generator import start_title_upgrade
|
||||
start_title_upgrade(upgrade)
|
||||
|
||||
|
||||
def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> int:
|
||||
"""Locate this turn's user message after compaction rebuilt ``messages``.
|
||||
|
||||
|
||||
@@ -483,6 +483,10 @@ def finalize_turn(
|
||||
_rollback_interrupted_preflight_display(agent, interrupted)
|
||||
|
||||
_cleanup_errors: List[str] = []
|
||||
# The model has answered (or the loop gave up): a title upgrade held back because it shares a
|
||||
# self-hosted endpoint with the main request (#117296) may go out now.
|
||||
from agent.turn_context import start_deferred_title_upgrade
|
||||
_guarded_cleanup("start_deferred_title_upgrade", lambda: start_deferred_title_upgrade(agent), _cleanup_errors, logger)
|
||||
# ``user_message`` may be a multimodal list of parts; the trajectory format wants a string.
|
||||
_guarded_cleanup(
|
||||
"save_trajectory",
|
||||
|
||||
@@ -448,6 +448,36 @@ class TestMaybeAutoTitle:
|
||||
runtime_validator=None,
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"main_runtime, title_cfg, deferred",
|
||||
[
|
||||
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {}, True),
|
||||
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://127.0.0.1:8080/v1/"}, True),
|
||||
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"provider": "openrouter"}, False),
|
||||
({"provider": "custom", "base_url": "http://127.0.0.1:8080/v1"}, {"base_url": "http://10.0.0.2:8080/v1"}, False),
|
||||
({"provider": "openrouter", "base_url": "https://openrouter.ai/api/v1"}, {}, False),
|
||||
],
|
||||
)
|
||||
def test_title_call_waits_for_the_turn_when_it_shares_a_custom_endpoint(self, main_runtime, title_cfg, deferred):
|
||||
"""#117296: a self-hosted server serving the main turn and the concurrent json_schema title request
|
||||
can decode the title into the main reply. The upgrade must not go on the wire until the caller starts
|
||||
it after the turn; every other route keeps the turn-start timing."""
|
||||
import threading
|
||||
from agent import title_generator as tg
|
||||
db = MagicMock()
|
||||
db.get_session_title.return_value = None
|
||||
started = threading.Event()
|
||||
with patch.object(tg, "_title_config", return_value=title_cfg), \
|
||||
patch.object(tg, "auto_title_session", side_effect=lambda *a, **k: started.set()):
|
||||
upgrade = maybe_auto_title(db, "sess-1", "hello", [{"role": "user", "content": "hello"}], main_runtime=main_runtime)
|
||||
assert isinstance(upgrade, threading.Thread)
|
||||
if deferred:
|
||||
assert upgrade.ident is None and not started.wait(0.3), "title request went out during the turn"
|
||||
assert upgrade not in tg._UPGRADE_THREADS # join-before-start would raise in wait_for_title_upgrades
|
||||
tg.start_title_upgrade(upgrade)
|
||||
assert started.wait(timeout=10), "auto_title thread never ran"
|
||||
assert upgrade in tg._UPGRADE_THREADS
|
||||
|
||||
def test_kanban_worker_is_named_after_its_card_without_the_llm_thread(self, tmp_path, monkeypatch):
|
||||
"""A worker's session takes the board card's title synchronously; no auxiliary model call (#111166)."""
|
||||
from hermes_cli import kanban_db, kanban_db_connect
|
||||
|
||||
@@ -407,3 +407,16 @@ def test_budget_exhausted_child_does_not_record_parent_kanban_timeout(monkeypatc
|
||||
)
|
||||
|
||||
record.assert_not_called()
|
||||
|
||||
|
||||
def test_finalize_turn_starts_the_title_upgrade_the_prologue_held_back():
|
||||
"""#117296: the turn prologue leaves a same-endpoint title upgrade unstarted on the agent; the finalizer
|
||||
is the only place that may start it, and only once the model request is done."""
|
||||
import threading
|
||||
|
||||
ran = threading.Event()
|
||||
agent = _LimitAgent()
|
||||
agent._deferred_title_upgrade = threading.Thread(target=ran.set, daemon=True)
|
||||
_finalize(agent, final_response="done", exit_reason="text_response(1)", api_call_count=1)
|
||||
assert ran.wait(timeout=5), "deferred title upgrade never started"
|
||||
assert agent._deferred_title_upgrade is None
|
||||
|
||||
@@ -1407,6 +1407,13 @@ No background `auto-title` thread starts and no automatic title-model request is
|
||||
explicit repair command `hermes sessions retitle-skills` still calls the model. `enabled: false`
|
||||
still disables both stages.
|
||||
|
||||
On a `custom` main provider (llama.cpp, Ollama, vLLM, LM Studio and other self-hosted
|
||||
OpenAI-compatible servers) the title model call is sent **after** the turn's reply has
|
||||
arrived, not concurrently with it, unless `auxiliary.title_generation` is pinned to another
|
||||
provider or `base_url`. A single-slot local server that receives the `json_schema` title
|
||||
request while decoding the reply can otherwise answer the reply with `{"title": ...}`,
|
||||
which is then stored and replayed as the assistant's turn.
|
||||
|
||||
In Hermes Desktop, a plain-text paste over 3,000 characters becomes a generated `.txt`
|
||||
attachment. The first ~1,000 characters of that paste are handed to the title stages as a
|
||||
title-only hint (the agent turn still sees only the attachment reference), so a "summarize
|
||||
|
||||
Reference in New Issue
Block a user