"""Auto-generate short session titles from the user's opening message. Two stages, both off the critical path: an **instant** deterministic title (written before the model is called, cannot fail), then an **upgrade** from one small-model call (cheap tier, thinking off, JSON-constrained). Storage enforces provenance ``derived < llm < user``: stage 2 only replaces stage 1 and neither replaces a name the user typed.""" import json import logging import os import re import threading import time import weakref from contextlib import suppress from typing import Any, Callable, Optional from agent.auxiliary_client import call_llm from agent.context_compressor import LEGACY_SUMMARY_PREFIX from agent.delegation_context import is_dispatcher_owned_worker_context from agent.message_content import flatten_message_text logger = logging.getLogger(__name__) # In-flight stage-2 upgrade threads. They bill their aux usage to the session from a daemon thread, # so a process that reads the ledger right before exit (``-z --usage-file``) must be able to join # them (bounded) instead of racing the write (#112848). _UPGRADE_THREADS: "weakref.WeakSet[threading.Thread]" = weakref.WeakSet() def wait_for_title_upgrades(timeout: float = 10.0) -> None: """Bounded join of the auto-title threads still running; never raises.""" deadline = time.monotonic() + timeout for thread in list(_UPGRADE_THREADS): thread.join(max(0.0, deadline - time.monotonic())) # (task_name, exception) -> None; surfaces auxiliary failures so silent drops don't pile up as NULL titles. FailureCallback = Callable[[str, BaseException], None] # (title, source) -> None; source is the persisted provenance (``derived`` / ``llm``). Consumers paying a # rate-limited remote rename per title (Discord thread, Telegram topic) should act on ``llm`` only. TitleCallback = Callable[[str, str], None] # () -> bool, called right before the LLM request; False skips (e.g. the user switched models and # the request would reload one the runtime already evicted). # Validation callback: () -> bool. See #19027. RuntimeValidator = Callable[[], bool] # Text budget handed to the model (Claude Code / OpenClaw converged on 1000). MAX_TITLE_INPUT_CHARS = 1000 _PASTE_PREVIEW_LABEL = "\n\nPasted content:\n" _ATTACHMENT_REF_RE = re.compile(r"@(?:file|folder):\S+") # Footers the @-reference expander appends below the typed text (agent/context_references.py). _CONTEXT_FOOTER_RE = re.compile(r"\n+--- (?:Context Warnings|Attached Context) ---\n.*", re.DOTALL) # Cap on the instant derived title; a raw fragment reads worse the longer it runs. MAX_DERIVED_TITLE_CHARS = 48 # Answer-shaped guard: a tiny model sometimes answers instead of titling; longer is rejected, not truncated. # Upper bound on accepted title word count. Titling is a 3-7 word task; a small tiny-model sometimes ignores # the task and answers the user's message instead — that answer must never become the session title (see the # answer-shaped output guard in generate_title; port of can1357/oh-my-pi#7306). 12 leaves headroom for # legitimate wordy titles while excluding full-sentence answers. _MAX_TITLE_WORDS = 12 # Output budget for the title call: room for a fenced/prefixed JSON reply and for a reasoning model that # thinks despite the thinking-disabled request, without letting a runaway reply burn minutes. TITLE_MAX_TOKENS = 512 # The example titles shown to the model in the prompt, and the echo-guard # set: when the opening message carries little topical signal, a small model # sometimes takes the cheapest schema-valid answer and parrots one of these # back verbatim — most visibly "Fix login button on mobile" naming sessions # that have nothing to do with a login button. The prompt's example lines are # rendered from these constants so the guard set and the prompt cannot drift # apart. Port of QwenLM/qwen-code#9709. _PROMPT_GOOD_EXAMPLES = ( "Fix login button on mobile", "Postgres connection pool exhaustion", "Friendly greeting", ) _PROMPT_VAGUE_EXAMPLE = "Code changes" # "Friendly greeting" is deliberately NOT in the reject set: the prompt # instructs the model to produce it for bare greetings, so it is a legitimate # output, not an echo failure. The too-vague example is rejected too — a model # repeating the counter-example says nothing about the session, and the # derived title the guard falls back to is strictly more informative. _EXAMPLE_ECHO_REJECT = frozenset( t.lower() for t in _PROMPT_GOOD_EXAMPLES if t != "Friendly greeting" ) | {_PROMPT_VAGUE_EXAMPLE.lower()} # "Friendly greeting" is what the prompt asks for when the opener has no topic yet, so # it is the one model title that must NOT settle the session: it is persisted at # ``derived`` authority (a placeholder, like the instant title) and the next substantive # turn upgrades it. Kept as a prompt example on purpose — a predictable placeholder is # detectable, an improvised one ("Casual check-in chat") would lock the title as ``llm``. _PROVISIONAL_GREETING_TITLE = "friendly greeting" _TITLE_PROMPT_TEMPLATE = ( "You name chat sessions. Given the user's opening message, write a title " "that lets them find this conversation again in a list.\n\n" "Rules:\n" "- 3 to 7 words, sentence case (capitalize only the first word and proper nouns).\n" "- Name what the user wants DONE, not that they asked a question.\n" "- Keep technical terms, filenames, numbers, and error codes exact.\n" "- Drop filler words: the, this, my, a, an.\n" "- No trailing punctuation, no quotes, no tool names, no 'Title:' prefix.\n" "- Never answer the message. Name it.\n" "- Always produce something, even for a bare greeting.\n" "__LANGUAGE_RULE__\n" + "".join(f'Good: {{"title": "{t}"}}\n' for t in _PROMPT_GOOD_EXAMPLES) + f'Too vague: {{"title": "{_PROMPT_VAGUE_EXAMPLE}"}}\n' 'Too long: {"title": "Investigate and fix the issue where the login button ' 'does not respond on mobile devices"}\n\n' 'Reply with JSON only: {"title": "..."}' ) _LANGUAGE_RULE_MATCH_USER = "- Write the title in the same language as the user's message." _LANGUAGE_RULE_PINNED = "- Write the title in {language}." # Constrains the response to a single title field ("model answered instead of titling" failure class). _TITLE_RESPONSE_FORMAT = { "type": "json_schema", "json_schema": {"name": "session_title", "strict": True, "schema": { "type": "object", "properties": {"title": {"type": "string"}}, "required": ["title"], "additionalProperties": False}}, } # Control-tag wrappers around machine-authored content inside a nominal "user" message (Codex CLI's # RECOGNIZED_CONTROL_WRAPPERS): stripped, titling continues on what remains. _CONTROL_WRAPPERS = tuple( (f"<{tag}>", f"") for tag in ("command-message", "command-name", "command-args", "local-command-caveat", "local-command-stderr", "local-command-stdout", "task-notification", "system-reminder", "ide_opened_file", "ide_selection") ) # Hermes' own machine-authored openers: a compaction handoff or resumed session must not be titled after them. _MACHINE_PREFIXES = ( "[CONTEXT COMPACTION", LEGACY_SUMMARY_PREFIX, "[Runtime note:", "[System note:", "[SYSTEM]", # tui_gateway.server._MODEL_SWITCH_MARKER_PREFIX (keep in sync); persisted as role="user" because # strict providers reject a non-first system message. # Model-switch marker from tui_gateway.server._append_model_switch_marker. It is persisted with # role="user" (strict OpenAI-compatible providers reject a system message that is not first — #48338), # so without this entry it looks like a real opening turn: switching models before the first real # message titled the session "[System: The active model for this chat has…" instead of the user's actual # question. "[System: The active model for this chat has changed to ", ) def _title_config() -> dict: """``auxiliary.title_generation`` (lazy read-only import: no hermes_cli cycle, no migration writes).""" from hermes_cli.config import load_config_readonly return ((load_config_readonly() or {}).get("auxiliary") or {}).get("title_generation") or {} def _title_language() -> str: """Configured title language, or "" to match the user.""" try: return str(_title_config().get("language", "")).strip() except Exception: return "" def _auto_title_enabled() -> bool: try: from utils import is_truthy_value return is_truthy_value(_title_config().get("enabled"), default=True) except Exception: logger.debug("Failed to read title_generation.enabled", exc_info=True) return True def _model_title_upgrade_enabled() -> bool: """Distinct from ``enabled``: keep the instant derived title, skip the background model call (#85194).""" try: from utils import is_truthy_value return is_truthy_value(_title_config().get("model_upgrade_enabled"), default=True) except Exception: logger.debug("Failed to read title_generation.model_upgrade_enabled", exc_info=True) return True def title_upgrade_must_wait_for_turn(main_runtime: Optional[dict]) -> bool: """True when the model title call would hit the SAME self-hosted endpoint as the turn's own request. A self-hosted main route (``_is_self_hosted_provider``: custom, lmstudio, local and their aliases) whose ``auxiliary.title_generation`` is not pinned elsewhere (a pin naming the same custom route — ``custom:``, bare ```` or its display name — is not "elsewhere", #120558) shares one local server between the streaming main request and the concurrent ``response_format: json_schema`` title request. Single-slot servers then serve the title grammar/completion into the main turn: the user's reply arrives as ``{"title": ...}``, is persisted as a genuine assistant row and replayed, and the model adopts the format (#117296). Running the title call after the turn settles keeps the two requests off the wire at once. Hosted providers multiplex requests independently and keep the turn-start timing. """ provider = str((main_runtime or {}).get("provider") or "").strip().lower() if not _is_self_hosted_provider(provider): return False try: cfg = _title_config() pinned_provider = str(cfg.get("provider") or "").strip().lower() main_base_url = str((main_runtime or {}).get("base_url") or "").strip().rstrip("/") if pinned_provider not in ("", "auto") and not _title_pin_may_share_endpoint( pinned_provider, provider, main_base_url): return False except Exception: return True pinned_base_url = str(cfg.get("base_url") or "").strip().rstrip("/") return not pinned_base_url or pinned_base_url == main_base_url def _is_self_hosted_provider(provider: str) -> bool: """``custom``, a named ``custom:`` route, LM Studio or ``local`` (vllm/llama.cpp): one local server per route. Normalised here so the main route and the title pin resolve aliases (``ollama``, ``lm-studio``…) the same way. """ from hermes_cli.providers import normalize_provider provider = normalize_provider(provider) return provider in ("custom", "lmstudio", "local") or provider.startswith("custom:") def _title_pin_may_share_endpoint(pinned_provider: str, main_provider: str, main_base_url: str) -> bool: """A title pin that can land on the turn's own self-hosted server (only its ``base_url`` can prove otherwise). Hosted pins (``openrouter``…) multiplex and never share the slot. A pin to ``custom``/``lmstudio``/``local``/any ``custom:`` is assumed to share until the caller compares ``base_url``, and a bare ```` / display-name pin is the same endpoint when it aliases the main ``custom:`` route (``hermes_cli.providers.custom_provider_aliases`` — the resolver's own identity set) or resolves to a configured custom entry serving ``main_base_url`` (a keyed ``providers:`` entry's display name does not alias its ``custom:`` id). """ from hermes_cli.config import get_compatible_custom_providers, load_config_readonly from hermes_cli.providers import custom_provider_aliases, resolve_custom_provider if _is_self_hosted_provider(pinned_provider): return True if custom_provider_aliases(pinned_provider) & custom_provider_aliases(main_provider): return True pdef = resolve_custom_provider(pinned_provider, get_compatible_custom_providers(load_config_readonly())) return bool(pdef and main_base_url and pdef.base_url.strip().rstrip("/") == main_base_url) def start_title_upgrade(upgrade: Optional[threading.Thread]) -> None: """Start a (deferred) title upgrade thread; joinable via ``wait_for_title_upgrades`` only once started.""" if upgrade is None or upgrade.ident is not None: return _UPGRADE_THREADS.add(upgrade) upgrade.start() def strip_control_wrappers(text: str) -> str: """Remove leading control wrappers (nested too) so a slash-command turn reduces to the prose the user typed.""" current = (text or "").strip() for _ in range(len(_CONTROL_WRAPPERS) * 2): # bounded: each pass must remove a wrapper or we stop stripped = _strip_one_wrapper(current) if stripped == current: break current = stripped return current def _strip_one_wrapper(text: str) -> str: lowered = text.lower() for open_tag, close_tag in _CONTROL_WRAPPERS: if not lowered.startswith(open_tag): continue end = lowered.find(close_tag) if end == -1: # unterminated wrapper: drop the opening tag and keep the body return text[len(open_tag):].strip() # Prefer trailing prose; otherwise the wrapper body is all we have. return (text[end + len(close_tag):].strip() or text[len(open_tag):end].strip()).strip() return text # Matches an ``@file:``/``@folder:`` context reference the way the canonical # ``agent.context_references`` reference pattern does: an unquoted ``\S+`` # value, or a backtick/double/single-quoted value (a space-bearing path is # written backtick-quoted). Kept local so the titler doesn't import the # context-reference machinery (circularity / startup cost) just to detect the # attachment-only shape (#92068). _QUOTED = r"(?:`[^`\n]+`|\"[^\"\n]+\"|'[^'\n]+')" # Mirrors the canonical agent.context_references value shape: a quoted # (space-bearing) path may carry a ``:start[-end]`` line-range suffix, and a # bare path swallows its own range via ``\S+``. _CONTEXT_REFERENCE_TOKEN_RE = re.compile( rf"(? bool: """True when nothing but context references (and their expansion footer) remain — a manual-attach opener with no typed request and no paste preview has no topic of its own to title (#92068).""" residual = _CONTEXT_REFERENCE_TOKEN_RE.sub("", _CONTEXT_FOOTER_RE.sub("", message)) return not residual.strip() def _summarize_user_message(user_message: str) -> str: """Text worth titling: describe a ``/skill`` invocation (it embeds the whole skill body), then strip wrappers.""" if not user_message: return "" described = None try: from agent.skill_commands import describe_skill_invocation described = describe_skill_invocation(user_message) except Exception: logger.debug("Skill-scaffolding summary failed; titling raw", exc_info=True) return strip_control_wrappers(user_message if described is None else described) def build_title_input(user_message: str, title_preview: str | None = None) -> str: """Combine the opening text with a bounded Desktop-generated paste preview. ``title_preview`` is deliberately an explicit, auxiliary-only value: ordinary attachments never populate it, and it is never returned to the main turn. Keep enough of a separately typed request to preserve a useful instruction, then spend the remaining title budget on the beginning of the pasted topic. """ message = _summarize_user_message(user_message) preview = title_preview.strip() if isinstance(title_preview, str) else "" if not preview: return message[:MAX_TITLE_INPUT_CHARS] # The titler sees the message AFTER @-reference expansion, so the generated ref drags a # warnings/attached-context footer along; the preview already carries the topic, so drop it. message = _CONTEXT_FOOTER_RE.sub("", message).strip() # A paste-only opener is just the generated `@file:` ref: the preview IS the topic, so it # leads (derive_title takes the first line, and a file path is not a title). if not _ATTACHMENT_REF_RE.sub("", message).strip(): return preview[:MAX_TITLE_INPUT_CHARS] message_budget = min(len(message), MAX_TITLE_INPUT_CHARS // 2) preview_budget = MAX_TITLE_INPUT_CHARS - message_budget - len(_PASTE_PREVIEW_LABEL) if preview_budget <= 0: return message[:MAX_TITLE_INPUT_CHARS] return message[:message_budget] + _PASTE_PREVIEW_LABEL + preview[:preview_budget] def is_titleable_user_message(user_message: str) -> bool: """False for machine-authored openers and turns that reduce to nothing once scaffolding is stripped.""" return (isinstance(user_message, str) and bool(user_message.strip()) and not user_message.lstrip().startswith(_MACHINE_PREFIXES) and bool(_summarize_user_message(user_message).strip()) # An attachment-only opener (manual attach, no paste preview) is a # file drop, not a request: deriving its "title" from the message # names the session after the truncated file path (#92068). and not _attachment_only_opener(user_message)) def derive_title(user_message: str, title_preview: str | None = None) -> Optional[str]: """Instant title: first meaningful line trimmed to a word boundary. No model, never fails.""" # Attachment-only opener, no paste preview: a file drop has no topic — # refuse rather than name the session after the truncated path (#92068). if not title_preview and _attachment_only_opener(user_message): return None line = " ".join(_first_line(build_title_input(user_message, title_preview)).split()) if len(line) > MAX_DERIVED_TITLE_CHARS: cut = line[:MAX_DERIVED_TITLE_CHARS] space = cut.rfind(" ") line = (cut[:space] if space > MAX_DERIVED_TITLE_CHARS // 2 else cut).rstrip(" ,.;:—-") + "…" return line or None def _strip_title_prefix(text: str) -> str: return text[6:].strip() if text.lower().startswith("title:") else text def _first_line(text: str) -> str: return next((ln.strip() for ln in text.splitlines() if ln.strip()), "") def _extract_json_title(raw: str) -> Optional[str]: """Title from a ``{"title": ...}`` payload — strict parse, then a loose ``"title": "..."`` scan; None when absent.""" try: parsed = json.loads(raw) if isinstance(parsed, dict) and isinstance(parsed.get("title"), str): return parsed["title"].strip() except (ValueError, TypeError): pass match = re.search(r'"title\"\s*:\s*"((?:[^"\\]|\\.)*)"', raw) if match: with suppress(ValueError): return json.loads(f'"{match.group(1)}"').strip() return match.group(1).strip() return None def _is_truncated_structured_output(raw: str) -> bool: """Structured output the token cap cut before its closing quote/brace/fence (``{"title``, a bare fence opener). Checked only after the JSON paths failed, and on structure alone (a JSON-shaped opener, a fence opener that is never closed) so quoted, *emphasized* or ``[WIP]``-prefixed prose titles are untouched (#83903).""" return raw.startswith(('{"', '["', "[{")) or (raw.startswith("```") and raw.count("```") % 2 == 1) def _extract_title_text(content: str) -> str: """Strict JSON, then a loose JSON scan, then first-line prose (a provider ignoring ``response_format`` still titles). A truncated structured payload is dropped rather than handed to the prose fallback: the fragment would otherwise be persisted as the session title.""" if not content: return "" raw = content.strip() fenced = re.match(r"^```(?:json)?\s*(.*?)\s*```$", raw, re.DOTALL) if fenced: raw = fenced.group(1).strip() title = _extract_json_title(raw) if title is not None: return title if _is_truncated_structured_output(raw): return "" # Prose fallback: scrub blocks so reasoning can't leak into a title. try: from agent.agent_runtime_helpers import strip_think_blocks raw = strip_think_blocks(None, raw).strip() except Exception: logger.debug("strip_think_blocks unavailable for title output", exc_info=True) return _strip_title_prefix(_first_line(raw)).strip("\"'").strip() def _title_from_reasoning(message: Any) -> str: """The ``{"title": ...}`` payload when a reasoning model put it in ``reasoning_content`` / ``reasoning`` and left ``content`` empty (glm-5 / minimax under ``json_schema``, #82291). Structured extraction only: chain-of-thought prose is never a title, so there is no prose fallback here.""" for field in ("reasoning_content", "reasoning"): text = getattr(message, field, None) if isinstance(text, str) and text.strip(): title = _extract_json_title(text.strip()) if title: return title return "" def _clean_title(text: str) -> Optional[str]: """Normalize a model-produced title, or None when nothing usable remains.""" title = _strip_title_prefix(" ".join((text or "").split()).strip("\"'").strip()).rstrip(".!,;:") if len(title) > 80: title = title[:77].rstrip() + "..." return title or None def _safe_callback(callback: Optional[Callable], args: tuple, log_fmt: str, label: str) -> None: """Invoke an optional consumer callback, never raising.""" try: if callback is not None: callback(*args) except Exception: logger.debug(log_fmt, label, exc_info=True) def _report_failure(failure_callback: Optional[FailureCallback], exc: BaseException, label: str) -> None: _safe_callback(failure_callback, ("title generation", exc), "%s failure_callback raised", label) def _notify_title(title_callback: Optional[TitleCallback], title: str, source: str, label: str) -> None: _safe_callback(title_callback, (title, source), "%s callback failed", label) def _is_provisional_greeting_title(title: str) -> bool: """The prompt's greeting placeholder (also "Friendly greeting in chat" and quoted/bracketed variants).""" normalized = re.sub(r"^[\W_]+|[\W_]+$", "", title.strip(), flags=re.UNICODE).lower() return normalized in (_PROVISIONAL_GREETING_TITLE, _PROVISIONAL_GREETING_TITLE + " in chat") def _is_prompt_example_echo(title: str) -> bool: """Return True when *title* is one of the prompt's own example titles. Comparison is case-insensitive after stripping any leading/trailing run of non-letter/non-digit characters, so bracket/quote wrappers cannot bypass the guard — while ``_clean_title`` keeps brackets for real titles like "(WIP) Fix build". Unicode-aware so full-width wrappers are covered too. """ normalized = re.sub(r"^[\W_]+|[\W_]+$", "", title.strip(), flags=re.UNICODE).lower() return normalized in _EXAMPLE_ECHO_REJECT def generate_title( user_message: str, timeout: Optional[float] = None, failure_callback: Optional[FailureCallback] = None, main_runtime: dict = None, runtime_validator: Optional[RuntimeValidator] = None, title_preview: str | None = None, ) -> Optional[str]: """Title from the opening message alone (waiting for the assistant made this slow and bought nothing). ``runtime_validator`` runs right before the request; False skips silently. If it returns False (e.g. the user's model was switched since the background thread captured its runtime snapshot), the call is skipped silently — no request is sent, so a stale title request can't reload a model the runtime already unloaded (#19027). """ if not _auto_title_enabled(): logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false") return None try: if runtime_validator is not None and not runtime_validator(): logger.debug("Title generation skipped: runtime validator returned False") return None except Exception: # fail open: a broken validator must not disable titling logger.debug("Title runtime validator raised; proceeding", exc_info=True) user_snippet = build_title_input(user_message, title_preview) if not user_snippet.strip() or ( # An attachment-only opener (manual attach, no paste preview) reaches # here via auto_title_session, which lacks the instant-title guard: # refuse it rather than titling the session after the file path (#92068). not title_preview and _attachment_only_opener(user_snippet) ): return None language = _title_language() # str.replace, not str.format: the prompt embeds literal JSON braces. prompt = _TITLE_PROMPT_TEMPLATE.replace( "__LANGUAGE_RULE__", _LANGUAGE_RULE_PINNED.format(language=language) if language else _LANGUAGE_RULE_MATCH_USER, ) try: # Use the provider's default temperature instead of forcing 0.3. # Some models (e.g. GPT-5.6) only accept their server-side default # and reject explicit temperature values, causing the daemon title # thread to fail with "Unsupported value: 'temperature'". # See: #72351, #51083, #51157 response = call_llm( task="title_generation", messages=[{"role": "system", "content": prompt}, {"role": "user", "content": user_snippet}], # A title is a handful of tokens, but 64 was cut mid-JSON by fenced/prefixed replies and by # reasoning models whose thinking survives the disable below (#83903, #82291). A model that # honours the JSON contract stops after ~15 tokens regardless, so the ceiling only costs on # replies that would have been garbage anyway. temperature=None: omitted from the wire so # default-only reasoning models accept the first request (#72351). max_tokens=TITLE_MAX_TOKENS, temperature=None, timeout=timeout, main_runtime=main_runtime, extra_body={"response_format": _TITLE_RESPONSE_FORMAT}, # The module contract above promises thinking-disabled operation, # but nothing enforced it: with the aux default reasoning_effort # "" (provider default), Gemini enables internal thinking and # bills thought tokens against max_tokens=64 — the JSON payload # never lands, and the prose fallback stores the opening fence # ("```json") as the session title (#91927). reasoning_config={"enabled": False}, ) message = response.choices[0].message title = _clean_title(_extract_title_text(message.content or "") or _title_from_reasoning(message)) # Answer-shaped output guard: titling is a 3-7 word task, so a title with many words is a model that # ignored the task and answered the user's message instead ("I don't have context on X — that's not # something I recognize..."). Truncating would store half an assistant blob as the session title, # which is still an assistant blob — reject instead so the caller retries on the next exchange # (maybe_auto_title retries a placeholder title through the third exchange). Port of can1357/oh-my-pi#7306. if title is not None and len(title.split()) > _MAX_TITLE_WORDS: # Answer-shaped output: reject (not truncate) so the caller retries next exchange. logger.debug("Rejecting answer-shaped title output (%d words > %d)", len(title.split()), _MAX_TITLE_WORDS) return None # Example-echo guard: a title that parrots one of the prompt's own # examples back verbatim says nothing about the session — reject it so # the instant derived title (a slice of the user's actual words) # survives instead. Exact match after wrapper-stripping, deliberately # not fuzzy, so a genuinely topical title that merely resembles an # example still passes. Wrappers are stripped for the comparison only # ("(Fix login button on mobile)" is the same canned echo as the bare # example). Port of QwenLM/qwen-code#9709. if title is not None and _is_prompt_example_echo(title): logger.debug("Rejecting prompt-example echo title: %r", title) return None return title except Exception as e: # WARNING so it shows in agent.log without debug mode; stack at debug. logger.warning("Title generation failed: %s", e) logger.debug("Title generation traceback", exc_info=True) _report_failure(failure_callback, e, "Title generation") return None def _has_upgraded_title(session_db, session_id: str) -> bool: """True when the session already carries an ``llm``/``user`` title (or the check fails).""" try: source_fn = getattr(session_db, "get_session_title_source", None) if source_fn is not None: return source_fn(session_id) not in (None, "derived") return bool(session_db.get_session_title(session_id)) except Exception: return True def _persist_session_title(session_db, session_id, title, *, source, dedupe=True): """Persist at *source* authority via ``set_auto_title`` (precedence check + write in one transaction, so a manual ``/title`` is never overwritten); None when a higher authority held the row. ``ValueError`` = unique-title index collision → append ``#N`` via ``get_next_title_in_lineage``; ``dedupe=False`` re-raises instead (the derived title is on the critical path, collides constantly on "hi", and the model replaces it a second later anyway). ``ValueError`` means the name is taken by an unrelated session (the unique-title index); rather than leave the session untitled (#50537), append a ``#N`` suffix via ``get_next_title_in_lineage``. """ auto_fn = getattr(session_db, "set_auto_title", None) def _set(candidate): if auto_fn is not None: if auto_fn(session_id, candidate, source=source): return candidate logger.debug("Skipping %s title: a higher-authority title already holds session %s", source, session_id) return None legacy_fn = getattr(session_db, "set_auto_title_if_empty", None) # older store without provenance if legacy_fn is not None: return candidate if legacy_fn(session_id, candidate) else None if session_db.set_session_title(session_id, candidate) is False: raise RuntimeError(f"session {session_id} not found when storing title") return candidate try: return _set(title) except ValueError: next_title_fn = getattr(session_db, "get_next_title_in_lineage", None) deduped = next_title_fn(title) if dedupe and next_title_fn is not None else None if not deduped or deduped == title: raise return _set(deduped) def apply_subagent_title(session_db, session_id: str, goal: str) -> Optional[str]: """Title a delegate run ``Subagent: `` at ``derived`` authority. No model call: runs fan out in bulk, and the prefix alone is what tells them apart from conversations wherever ``sessions.show_subagents`` lists them (#97202). Collisions get ``#N``. Never raises.""" try: derived = derive_title(goal) if is_titleable_user_message(goal) else None return _persist_session_title(session_db, session_id, f"Subagent: {derived}", source="derived") if derived else None except Exception: logger.debug("Subagent title failed for %s", session_id, exc_info=True) return None def apply_instant_title( session_db, session_id: str, user_message: str, title_callback: Optional[TitleCallback] = None, title_preview: str | None = None, ) -> Optional[str]: """Write the derived title inline. Returns it, or None (no usable text, or a ``derived``+ title exists). Never raises. ``title_preview`` must reach this stage too: the model upgrade's own ``derive_title`` fallback writes ``derived`` provenance, which never replaces the ``derived`` title written here.""" if not session_db or not session_id: return None try: title = derive_title(user_message, title_preview) if is_titleable_user_message(user_message) else None persisted = _persist_session_title(session_db, session_id, title, source="derived", dedupe=False) if title else None if persisted: _notify_title(title_callback, persisted, "derived", "Instant-title") return persisted except Exception: logger.debug("Instant title failed", exc_info=True) return None def auto_title_session( session_db, session_id: str, user_message: str, failure_callback: Optional[FailureCallback] = None, main_runtime: dict = None, title_callback: Optional[TitleCallback] = None, runtime_validator: Optional[RuntimeValidator] = None, title_preview: str | None = None, ) -> None: """Generate and store the model title (daemon-thread target); skips sessions already carrying an ``llm``/``user`` title (a ``derived`` one is expected — upgrading it is the point). Never lets an exception escape (the threading excepthook would spray a traceback into the terminal); the canonical trigger is the post-``hermes update`` window where lazy imports read NEW source against OLD modules.""" try: if not session_db or not session_id or _has_upgraded_title(session_db, session_id): return # This thread starts AFTER the turn's ambient context was reset; republish it so the call carries # the same Portal ``conversation=`` tag (root-of-lineage) and bills usage to this session. from agent.aux_accounting import set_accounting_context from agent.portal_tags import set_conversation_context conversation_id = session_id with suppress(Exception): conversation_id = session_db.get_conversation_root(session_id) or session_id set_conversation_context(conversation_id) # Same for the accounting context, so the title call's token usage is recorded against this session # (task='title_generation', #23270). set_accounting_context(session_db, session_id) title, source = generate_title( user_message, failure_callback=failure_callback, main_runtime=main_runtime, runtime_validator=runtime_validator, title_preview=title_preview, ), "llm" if title and _is_provisional_greeting_title(title): source = "derived" if not title: # the inline attempt declined collisions; off the critical path the lineage scan is affordable title, source = derive_title(user_message, title_preview), "derived" if not title: return try: persisted = _persist_session_title(session_db, session_id, title, source=source) except Exception as e: logger.debug("Failed to set auto-generated title: %s", e) return if persisted is not None: logger.debug("Auto-generated session title: %s", persisted) _notify_title(title_callback, persisted, source, "Auto-title") except Exception as e: # WARNING so operators see it in agent.log; names the likely cause. logger.warning("Auto-title failed (harmless; if this started after an update, restart the running Hermes process): %s", e) logger.debug("Auto-title traceback", exc_info=True) _report_failure(failure_callback, e, "Auto-title") def _is_real_user_turn(message: Any) -> bool: """A question a person actually asked (Hermes persists machinery under ``role="user"``).""" if not isinstance(message, dict) or message.get("role") != "user": return False content = message.get("content") return is_titleable_user_message(content if isinstance(content, str) else flatten_message_text(content)) def _session_is_untitled(session_db, session_id: str) -> bool: """No title of any provenance; False when it can't tell (no model call per turn for an unreadable title).""" getter = getattr(session_db, "get_session_title", None) try: return callable(getter) and not str(getter(session_id) or "").strip() except Exception: logger.debug("Untitled check failed for %s", session_id, exc_info=True) return False def _kanban_task_title() -> Optional[str]: """Kanban worker: the card's title, or ``Kanban task `` when the board can't be read; None elsewhere (including delegate_task children of the worker, which inherit the env var but are not the card).""" task_id = (os.environ.get("HERMES_KANBAN_TASK") or "").strip() if not task_id or not is_dispatcher_owned_worker_context(): return None try: from hermes_cli import kanban_db, kanban_db_connect from hermes_state import SessionDB with kanban_db_connect.connect_closing() as conn: task = kanban_db.get_task(conn, task_id) title = " ".join((task.title or "").split()) if task is not None else "" # Cards have no length cap; the title store rejects past MAX_TITLE_LENGTH (and the ``#N`` # retry suffix needs room), which would leave the worker nameless. cap = SessionDB.MAX_TITLE_LENGTH - 4 if len(title) > cap: title = title[: cap - 1].rstrip() + "…" except Exception: logger.debug("Kanban task %s unreadable; naming the session after its id", task_id, exc_info=True) title = "" return title or f"Kanban task {task_id}" def maybe_auto_title( session_db, session_id: str, user_message: str, conversation_history: Optional[list] = None, failure_callback: Optional[FailureCallback] = None, main_runtime: dict = None, title_callback: Optional[TitleCallback] = None, runtime_validator: Optional[RuntimeValidator] = None, title_preview: str | None = None, ) -> Optional[threading.Thread]: """Instant inline title, then a daemon-thread upgrade. Call at the START of a turn, before the model. Returns the upgrade thread: already started, or — when ``title_upgrade_must_wait_for_turn`` — left UNSTARTED for the caller to hand to ``start_title_upgrade`` once the turn's model request settled.""" if not session_db or not session_id or not user_message: return None # History may be pre- or post-message. Past the opening turn, skip once the session holds an # ``llm``/``user`` name: count alone left a machinery-opened session nameless, and a ``derived`` # name is still a placeholder (instant slice, or the model's greeting title for a bare "hi") that # the first substantive turn should replace. Untitled sessions always get another shot; a # placeholder gets turns 2-3, so a failing title model costs at most three calls, not one per turn. user_msg_count = sum(1 for m in (conversation_history or []) if _is_real_user_turn(m)) if user_msg_count > 1 and ( _has_upgraded_title(session_db, session_id) or (user_msg_count > 3 and not _session_is_untitled(session_db, session_id)) ): return None kanban_title = _kanban_task_title() if kanban_title: # The card already carries a human-written name; an auxiliary model call per spawned worker # only competes with the worker for capacity (#111166). Final (``llm``) authority: nothing # upgrades it later, and a manual ``/title`` still wins inside ``set_auto_title``. with suppress(Exception): persisted = _persist_session_title(session_db, session_id, kanban_title, source="llm") if persisted: _notify_title(title_callback, persisted, "llm", "Kanban task title") return None if not is_titleable_user_message(user_message): return None if not _auto_title_enabled(): # config read after the cheap guards so the file isn't touched every turn logger.debug("Auto-title skipped: auxiliary.title_generation.enabled=false") return None apply_instant_title(session_db, session_id, user_message, title_callback, title_preview=title_preview) if not _model_title_upgrade_enabled(): logger.debug("Instant title persisted; model upgrade disabled by auxiliary.title_generation.model_upgrade_enabled=false") return None # The thread must resolve auxiliary.title_generation (config, provider key, language) for the # profile whose turn this is: a bare Thread starts with an empty context and lands on the launch # profile under multiplex, titling X's session with the default profile's model and billing its key. from agent.memory_provider import spawn_context_thread upgrade_kwargs = dict(failure_callback=failure_callback, main_runtime=main_runtime, title_callback=title_callback, runtime_validator=runtime_validator) if isinstance(title_preview, str) and title_preview.strip(): upgrade_kwargs["title_preview"] = title_preview upgrade = spawn_context_thread( auto_title_session, name="auto-title", args=(session_db, session_id, user_message), kwargs=upgrade_kwargs, ) if title_upgrade_must_wait_for_turn(main_runtime): logger.debug("Auto-title upgrade deferred past the turn: shares the self-hosted endpoint with the main request") return upgrade start_title_upgrade(upgrade) return upgrade