fix(tools): rebuild Bot Chat prompt when model capability overrides flip

capability_fingerprint ignored model.supports_vision and model.context_length,
so an eternal Bot Chat kept its stored prompt after those overrides changed.
A NULL stored prompt already rebuilds and no longer takes the stale probe.
This commit is contained in:
brooklyn!
2026-09-24 05:19:03 -05:00
parent 687ca6ed88
commit 50ff26bdc0
4 changed files with 132 additions and 8 deletions

View File

@@ -606,14 +606,18 @@ def _print_billing_or_entitlement_guidance(
))
def _bot_chat_prompt_stale(agent, stored_prompt: str) -> bool:
def _bot_chat_prompt_stale(agent, stored_prompt: str | None) -> bool:
"""Bot Chat capability epoch check for a stored prompt.
The stored prompt embeds a capability fingerprint; a mismatch is a deliberate
once-per-change rebuild. Unstamped prompts never match; probe failures fail closed
to "reuse" so the cache is kept. Legacy upgrade: a Bot Chat prompt predating the
epoch mechanism gets ONE title-gated migration rebuild; the stamped result cannot
re-fire."""
re-fire. A NULL or empty stored prompt already rebuilds every turn, so this probe
is not a gate there and must not run.
"""
if not stored_prompt:
return False
try:
from tools.bot_mode_probe import (
BOT_CHAT_TITLE,
@@ -702,6 +706,7 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history)
)
if stored_prompt and _stored_prompt_matches_runtime(agent, stored_prompt):
# NULL/empty rows never reach this probe: they already rebuild below.
if _bot_chat_prompt_stale(agent, stored_prompt):
logger.info(
"Bot Chat capability epoch changed for session %s; rebuilding system prompt to "

View File

@@ -608,5 +608,37 @@ class TestPerResponseSessionWritePath:
assert "is null" not in caplog.text
def test_null_stored_prompt_does_not_take_the_stale_probe_path(tmp_path):
"""A NULL system_prompt row already rebuilds. The capability probe must not gate it."""
from hermes_state import SessionDB
from agent.conversation_loop import _bot_chat_prompt_stale
agent = SimpleNamespace(
_bot_mode_protocol=True,
_session_title_hint="Bot Chat",
_session_db=None,
session_id="test-session-id",
)
with patch(
"tools.bot_mode_probe.stored_prompt_capability_stale", return_value=False
) as probe:
assert _bot_chat_prompt_stale(agent, None) is False
probe.assert_not_called()
with SessionDB(db_path=tmp_path / "state.db") as db:
db.create_session("test-session-id", source="tui")
row = db.get_session("test-session-id")
assert row is not None and row["system_prompt"] is None
restoring = _make_agent(session_db=db, prebuilt_prompt="BUILT")
with patch(
"tools.bot_mode_probe.stored_prompt_capability_stale", return_value=False
) as probe:
_restore_or_build_system_prompt(
restoring, None, [{"role": "user", "content": "hi"}]
)
probe.assert_not_called()
restoring._build_system_prompt.assert_called_once()
if __name__ == "__main__":
pytest.main([__file__, "-v"])

View File

@@ -209,6 +209,60 @@ def test_fingerprint_changes_on_each_capability_axis(tmp_path):
assert bot_mode_probe.capability_fingerprint(home) != after_soul
def test_fingerprint_changes_when_model_vision_override_flips(tmp_path):
"""A Bot Chat prompt must rebuild when model.supports_vision flips."""
home = tmp_path / ".hermes"
home.mkdir()
_make_bot_profile(home, "researcher", managed=True)
config = home / "config.yaml"
config.write_text("model:\n supports_vision: true\n", encoding="utf-8")
vision_enabled = bot_mode_probe.capability_fingerprint(home)
stamped = "system stuff\n\n" + bot_mode_probe.epoch_line(home)
assert vision_enabled == bot_mode_probe.capability_fingerprint(home)
assert not bot_mode_probe.stored_prompt_capability_stale(stamped, home)
config.write_text("model:\n supports_vision: false\n", encoding="utf-8")
vision_disabled = bot_mode_probe.capability_fingerprint(home)
assert vision_disabled != vision_enabled
assert bot_mode_probe.stored_prompt_capability_stale(stamped, home)
restamped = "system stuff\n\n" + bot_mode_probe.epoch_line(home)
assert not bot_mode_probe.stored_prompt_capability_stale(restamped, home)
config.write_text("model:\n supports_vision: true\n", encoding="utf-8")
assert bot_mode_probe.capability_fingerprint(home) != vision_disabled
def test_vision_override_spellings_share_one_fingerprint(tmp_path):
"""YAML boolean tokens that image routing treats as the same override share an epoch."""
home = tmp_path / ".hermes"
home.mkdir()
_make_bot_profile(home, "researcher", managed=True)
config = home / "config.yaml"
config.write_text("model:\n supports_vision: true\n", encoding="utf-8")
enabled = bot_mode_probe.capability_fingerprint(home)
config.write_text("model:\n supports_vision: yes\n", encoding="utf-8")
assert bot_mode_probe.capability_fingerprint(home) == enabled
config.write_text("model:\n supports_vision: false\n", encoding="utf-8")
assert bot_mode_probe.capability_fingerprint(home) != enabled
def test_fingerprint_changes_when_model_context_length_override_changes(tmp_path):
"""context_length truncates context files in the rebuilt prompt, so it is part of the epoch."""
home = tmp_path / ".hermes"
home.mkdir()
_make_bot_profile(home, "researcher", managed=True)
config = home / "config.yaml"
config.write_text("model:\n context_length: 32768\n", encoding="utf-8")
narrow = bot_mode_probe.capability_fingerprint(home)
config.write_text("model:\n context_length: 131072\n", encoding="utf-8")
assert bot_mode_probe.capability_fingerprint(home) != narrow
config.write_text("model:\n context_length: 32768\n", encoding="utf-8")
assert bot_mode_probe.capability_fingerprint(home) == narrow
def test_stored_prompt_staleness(tmp_path):
home = tmp_path / ".hermes"
home.mkdir()

View File

@@ -318,19 +318,50 @@ def get_bot_mode_protocol_section(home: str | os.PathLike | None = None, *, forc
# ── capability epoch ─────────────────────────────────────────────────────────
# Bot Chat sessions are effectively eternal, so "build the prompt once" would strand
# capability changes (skills, toolsets, MCP, SOUL, roster, peers) forever. The fingerprint
# hashes exactly that surface; the built prompt embeds it and agent/conversation_loop.py
# rebuilds only when the stored epoch differs from disk — once per change, never per-turn drift.
# capability changes (skills, toolsets, MCP, SOUL, roster, peers, model capability
# overrides that change the prompt) forever. The fingerprint hashes exactly that
# surface; the built prompt embeds it and agent/conversation_loop.py rebuilds only
# when the stored epoch differs from disk — once per change, never per-turn drift.
_EPOCH_PREFIX = "Capability epoch: "
_EPOCH_RE_TEXT = r"Capability epoch: ([0-9a-f]{12})"
def _model_prompt_capability_surface(model_cfg: object) -> dict:
"""``model.*`` overrides whose flip changes a rebuilt prompt, coerced like their consumers.
``supports_vision`` uses image routing's strict bool so YAML ``yes`` and ``true`` share
one epoch. ``context_length`` is the cap ``build_system_prompt_parts`` uses to truncate
context files. Routing keys (provider, default, base_url) are identity lines, not this
surface — and no model id is special-cased.
"""
from agent.image_routing import _coerce_capability_bool
if not isinstance(model_cfg, dict):
model_cfg = {}
raw_ctx = model_cfg.get("context_length")
ctx = None
# bool is an int subclass; ``context_length: true`` is not a window.
if isinstance(raw_ctx, bool):
ctx = None
elif isinstance(raw_ctx, int):
ctx = raw_ctx if raw_ctx > 0 else None
elif isinstance(raw_ctx, str) and raw_ctx.strip().isdigit():
parsed = int(raw_ctx.strip())
ctx = parsed if parsed > 0 else None
return {
"supports_vision": _coerce_capability_bool(model_cfg.get("supports_vision")),
"context_length": ctx,
}
def capability_fingerprint(home: str | os.PathLike | None = None) -> str:
"""12-hex digest of the capability surface for ``home``'s profile: disabled skills +
enabled toolsets + MCP config, SOUL.md bytes, installed skill names, the Bot-Mode roster
(+ roles), peers and the relay roster. Deliberately NOT cached — the point is detecting
on-disk drift against a stored prompt's epoch. Never raises ("unavailable" on failure)."""
enabled toolsets + MCP config, model capability overrides that change the prompt
(``supports_vision``, ``context_length``), SOUL.md bytes, installed skill names, the
Bot-Mode roster (+ roles), peers and the relay roster. Deliberately NOT cached — the
point is detecting on-disk drift against a stored prompt's epoch. Never raises
("unavailable" on failure)."""
import hashlib
import json
@@ -350,6 +381,8 @@ def capability_fingerprint(home: str | os.PathLike | None = None) -> str:
reset_hermes_home_override(token)
skills_cfg = cfg.get("skills") if isinstance(cfg.get("skills"), dict) else {}
tools_cfg = cfg.get("tools") if isinstance(cfg.get("tools"), dict) else {}
model_cfg = cfg.get("model") if isinstance(cfg.get("model"), dict) else {}
surface["model_capabilities"] = _model_prompt_capability_surface(model_cfg)
surface["disabled_skills"] = sorted(str(s).lower() for s in (skills_cfg.get("disabled") or []))
surface["enabled_toolsets"] = sorted(str(t) for t in (tools_cfg.get("enabled_toolsets") or []))
mcp = cfg.get("mcp_servers")