The Honcho provider resolved per-repo/per-directory session names from os.getcwd(), which is the backend process launch directory on Desktop/gateway hosts (typically $HOME), not the user's workspace. With a manual sessions map entry for the home directory, every Desktop conversation landed in that fallback bucket instead of the project's per-repo session. Use agent.runtime_cwd.resolve_agent_cwd() — the same single source of truth already used for system-prompt and context-file discovery — so Honcho session routing agrees with everything else about where the agent logically lives. It honors the pinned session cwd, then TERMINAL_CWD, then the launch directory; CLI sessions launched inside a project resolve identically either way. Adds a regression covering the Desktop-style case: backend launched in $HOME, workspace elsewhere, home-directory manual map present. Refs #24740 (cherry picked from commit cfb32757de2509dd404acf98311e357d73fa741e)
1286 lines
66 KiB
Python
1286 lines
66 KiB
Python
"""Honcho memory plugin — MemoryProvider for Honcho AI-native memory.
|
||
|
||
Cross-session user modeling with dialectic Q&A, semantic search, peer cards and
|
||
persistent conclusions; five tools (profile, search, reasoning, context, conclude).
|
||
Config chain: $HERMES_HOME/honcho.json -> ~/.honcho/config.json -> env vars.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import contextlib
|
||
import hashlib
|
||
import json
|
||
import logging
|
||
import os
|
||
import re
|
||
import threading
|
||
import time
|
||
from typing import Any, Callable, Dict, List, Optional
|
||
|
||
from agent.memory_manager import sanitize_context
|
||
from agent.memory_provider import MemoryProvider, is_trivial_prompt
|
||
from agent.coding_context import INTERACTIVE_CODING_PLATFORMS as _LOCAL_PLATFORMS
|
||
from agent.turn_author import a2a_key
|
||
from plugins.memory.honcho.client import HonchoClientConfig, resolve_config_path
|
||
from plugins.memory.honcho.client import _host_block, _HostLookup
|
||
from plugins.memory.honcho.client import join_plugin_threads, spawn_context_thread
|
||
from plugins.memory.honcho.dialectic import DialecticMixin
|
||
from plugins.memory.honcho.session_peers import assistant_peer_id_for, sanitize_peer_id
|
||
from plugins.memory.honcho.session_context import usable_honcho_summary
|
||
from plugins.memory.honcho.tool_schemas import ALL_TOOL_SCHEMAS
|
||
from tools.registry import tool_error
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
|
||
# Gateway-internal notifications arrive through the same user-role channel as genuine
|
||
# user messages; they are execution metadata and must never become durable memory.
|
||
# Deliberately anchored: a human discussing one of these strings mid-message is valid input.
|
||
_INTERNAL_GATEWAY_TURN_RE = re.compile(
|
||
r"^\s*(?:"
|
||
r"\[ASYNC (?:DELEGATION )?(?:BATCH )?COMPLETE[^\]]*\]|"
|
||
r"\[CONTEXT COMPACTION[^\]]*\]|"
|
||
r"\[CONTEXT SUMMARY\]:?|"
|
||
r"\[PRIOR CONTEXT[^\]]*\]|"
|
||
r"\[Your active task list was preserved across context compression\]|"
|
||
r"\[IMPORTANT: Background process \d+ matched watch pattern[^\n]*|"
|
||
r"A background fan-out of \d+ subagent\(s\) you dispatched earlier has finished\.|"
|
||
r"A background subagent you dispatched earlier has finished\."
|
||
r")",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
|
||
def _is_internal_gateway_turn(text: str) -> bool:
|
||
"""Return True for machine-generated gateway/delegation notifications."""
|
||
return bool(_INTERNAL_GATEWAY_TURN_RE.match(text or ""))
|
||
|
||
|
||
def _cfg_usable(cfg) -> bool:
|
||
"""Enabled with a credential or a self-hosted URL to talk to."""
|
||
return bool(cfg.enabled and (cfg.api_key or cfg.base_url))
|
||
|
||
|
||
# Static per-mode system prompt text (prompt-cache friendly: never changes between turns).
|
||
_TOOL_GUIDE = (
|
||
"Use honcho_profile for a quick factual snapshot, "
|
||
"honcho_search for raw excerpts, honcho_context for raw peer context, "
|
||
"honcho_reasoning for synthesized answers (pass reasoning_level "
|
||
"minimal/low/medium/high/max — you pick the depth per call), "
|
||
"honcho_conclude to save facts about the user."
|
||
)
|
||
_PROMPT_HEADERS = {
|
||
"context": (
|
||
"# Honcho Memory\nActive (context-injection mode). Relevant user context is automatically "
|
||
"injected before each turn. No memory tools are available — context is managed automatically."
|
||
),
|
||
"tools": (
|
||
f"# Honcho Memory\nActive (tools-only mode). {_TOOL_GUIDE} "
|
||
"No automatic context injection — you must use tools to access memory."
|
||
),
|
||
"hybrid": (
|
||
"# Honcho Memory\nActive (hybrid mode). Relevant context is auto-injected AND memory tools "
|
||
f"are available. {_TOOL_GUIDE}"
|
||
),
|
||
}
|
||
|
||
|
||
|
||
_FLAG_WORDS = {"1": True, "true": True, "yes": True, "on": True,
|
||
"0": False, "false": False, "no": False, "off": False, "": False}
|
||
|
||
|
||
def _as_flag(raw: Any, default: Optional[bool]) -> Optional[bool]:
|
||
"""A config or env value read as a boolean. Unrecognized strings keep ``default``."""
|
||
if isinstance(raw, str):
|
||
return _FLAG_WORDS.get(raw.strip().lower(), default)
|
||
return default if raw is None else bool(raw)
|
||
|
||
|
||
# (injection.sessionStart name, context key, heading). Render order is fixed here, not by config order.
|
||
_CONTEXT_SECTIONS = (
|
||
("summary", "summary", "Session Summary"),
|
||
("peerRepresentation", "representation", "User Representation"),
|
||
("peerCard", "card", "User Peer Card"),
|
||
("aiRepresentation", "ai_representation", "AI Self-Representation"),
|
||
("aiCard", "ai_card", "AI Identity Card"),
|
||
)
|
||
|
||
_PREWARM_QUERY = "Summarize what you know about this user. Focus on preferences, current projects, and working style."
|
||
|
||
|
||
class HonchoMemoryProvider(DialecticMixin, MemoryProvider):
|
||
"""Honcho AI-native memory with dialectic Q&A and persistent user modeling."""
|
||
|
||
def backup_paths(self) -> List[str]:
|
||
"""Whole ~/.honcho dir (peer/session config when no profile-local honcho.json exists)."""
|
||
try:
|
||
from .client import resolve_global_config_path
|
||
return [str(resolve_global_config_path().parent)]
|
||
except Exception:
|
||
return []
|
||
|
||
def __init__(self, query_rewriter: Optional[Callable[[str], str]] = None):
|
||
self._manager = None # HonchoSessionManager
|
||
self._config = None # HonchoClientConfig
|
||
self._session_key = ""
|
||
self._query_rewriter = query_rewriter
|
||
self._prefetch_result = ""
|
||
self._prefetch_lock = threading.Lock()
|
||
self._prefetch_thread: Optional[threading.Thread] = None
|
||
self._sync_thread: Optional[threading.Thread] = None
|
||
self._memwrite_thread: Optional[threading.Thread] = None
|
||
self._recall_mode = "hybrid" # "context", "tools", or "hybrid"
|
||
self._recall_sync = False
|
||
self._recall_generation = object()
|
||
self._recall_sync_thread: Optional[threading.Thread] = None
|
||
self._recall_sync_lock = threading.Lock()
|
||
# Base context cache — refreshed on context_cadence, not frozen.
|
||
self._base_context_cache: Optional[str] = None
|
||
self._base_context_lock = threading.Lock()
|
||
|
||
# Recall cadence state (overwritten from config in initialize()).
|
||
self._turn_count = 0
|
||
# Author of the turn in flight, refreshed by on_turn_start.
|
||
self._turn_author: dict[str, Any] = {}
|
||
# (config path, mtime_ns, size) -> identity_signature() values.
|
||
self._identity_signature_memo: dict[tuple, dict[str, Any]] = {}
|
||
# Injection audit. Off unless the logging key enables it: the record holds the user's representation.
|
||
self._injection_log_path: Optional[str] = None
|
||
self._injection_log_lock = threading.Lock()
|
||
# Pinned injection.sessionStart names; None means unpinned and everything renders.
|
||
self._session_start_components: Optional[frozenset] = None
|
||
self._query_rewrite_enabled = False
|
||
self._injection_frequency = "every-turn" # or "first-turn"
|
||
self._context_cadence = 1 # minimum turns between context API calls
|
||
self._dialectic_cadence = 1 # backwards-compat fallback; wizard writes 2 on new configs
|
||
self._dialectic_depth = 1 # .chat() calls per dialectic cycle (1-3)
|
||
self._dialectic_depth_levels: list[str] | None = None # per-pass reasoning levels
|
||
self._reasoning_heuristic: bool = True # scale base level by query length
|
||
self._reasoning_level_cap: str = "high" # ceiling for auto-selected level
|
||
self._last_context_turn = self._last_dialectic_turn = -999
|
||
# Liveness: monotonic start of the current prefetch thread, the turn the pending
|
||
# result was fired at, and consecutive empty dialectic returns (drives backoff).
|
||
self._prefetch_thread_started_at: float = 0.0
|
||
self._prefetch_result_fired_at: int = -999
|
||
self._dialectic_empty_streak: int = 0
|
||
|
||
# Tools-only mode may defer session initialization until a tool call.
|
||
self._session_initialized = False
|
||
self._lazy_init_kwargs: Optional[dict] = None
|
||
self._lazy_init_session_id: Optional[str] = None
|
||
self._init_thread: Optional[threading.Thread] = None
|
||
self._init_lock = threading.Lock()
|
||
# Init auth failures live here because the failed manager is discarded.
|
||
self._init_auth_failure: Optional[str] = None
|
||
self._init_auth_notice_emitted = False
|
||
# Set when no user peer could be named (no runtime identity, no peerName). Init is not retried.
|
||
self._init_peer_failure: Optional[str] = None
|
||
self._init_peer_platform: str = "cli"
|
||
self._init_peer_notice_emitted = False
|
||
self._cron_skipped = False # cron and flush contexts disable the plugin entirely
|
||
|
||
@property
|
||
def name(self) -> str:
|
||
return "honcho"
|
||
|
||
def is_available(self) -> bool:
|
||
"""Check if Honcho is configured. No network calls."""
|
||
try:
|
||
from plugins.memory.honcho.client import HonchoClientConfig
|
||
return _cfg_usable(HonchoClientConfig.from_global_config())
|
||
except Exception:
|
||
return False
|
||
|
||
def save_config(self, values, hermes_home):
|
||
"""Merge ``values`` into $HERMES_HOME/honcho.json (Honcho SDK native format); a file that does not parse raises.
|
||
Holds the token refresh locks so a rotation cannot land between the read and the write."""
|
||
from pathlib import Path
|
||
from utils import atomic_json_write
|
||
from plugins.memory.honcho.oauth import _config_refresh_lock, _read_config_strict, _refresh_lock
|
||
config_path = Path(hermes_home) / "honcho.json"
|
||
with _refresh_lock, _config_refresh_lock(config_path):
|
||
existing = _read_config_strict(config_path)
|
||
atomic_json_write(config_path, {**existing, **values}, mode=0o600)
|
||
|
||
def get_config_schema(self):
|
||
return [
|
||
{"key": "api_key", "description": "Honcho API key", "secret": True, "env_var": "HONCHO_API_KEY", "url": "https://app.honcho.dev"},
|
||
{"key": "baseUrl", "description": "Honcho base URL (for self-hosted)"},
|
||
]
|
||
|
||
def post_setup(self, hermes_home: str, config: dict) -> None:
|
||
"""Run the full Honcho setup wizard after provider selection."""
|
||
import types
|
||
from plugins.memory.honcho.cli import cmd_setup
|
||
cmd_setup(types.SimpleNamespace())
|
||
|
||
# ----- Session lifecycle -----
|
||
|
||
def initialize(self, session_id: str, **kwargs) -> None:
|
||
"""Configure recall settings and start (or defer) Honcho session creation."""
|
||
self._recall_generation = object()
|
||
try:
|
||
agent_context, platform = kwargs.get("agent_context", ""), kwargs.get("platform", "cli")
|
||
if agent_context in {"cron", "flush"} or platform == "cron":
|
||
logger.debug("Honcho skipped: cron/flush context (agent_context=%s, platform=%s)",
|
||
agent_context, platform)
|
||
self._cron_skipped = True
|
||
return
|
||
|
||
from plugins.memory.honcho.client import HonchoClientConfig, get_honcho_client # noqa: F401 — ImportError probe
|
||
from plugins.memory.honcho.session import HonchoSessionManager # noqa: F401
|
||
|
||
cfg = HonchoClientConfig.from_global_config()
|
||
if not _cfg_usable(cfg):
|
||
logger.debug("Honcho not configured — plugin inactive")
|
||
return
|
||
|
||
self._config = cfg
|
||
self._recall_mode = cfg.recall_mode
|
||
self._recall_sync = getattr(cfg, "recall_sync", False)
|
||
look = _HostLookup(_host_block(cfg.raw, cfg.host or ""), cfg.raw)
|
||
self._injection_log_path = self._resolve_injection_log_path(look)
|
||
self._session_start_components = self._resolve_session_start(look)
|
||
logger.debug("Honcho recall_mode: %s", self._recall_mode)
|
||
for name in ("injection_frequency", "context_cadence", "dialectic_cadence",
|
||
"dialectic_depth_levels", "reasoning_heuristic"):
|
||
setattr(self, f"_{name}", getattr(cfg, name))
|
||
self._query_rewrite_enabled = cfg.query_rewrite
|
||
self._FIRST_TURN_BASE_TIMEOUT = cfg.first_turn_base_wait
|
||
self._FIRST_TURN_DIALECTIC_CAP = cfg.first_turn_dialectic_wait
|
||
self._dialectic_depth = max(1, min(cfg.dialectic_depth, 3))
|
||
if cfg.reasoning_level_cap in self._LEVEL_ORDER:
|
||
self._reasoning_level_cap = cfg.reasoning_level_cap
|
||
|
||
# aiPeer comes from honcho.json only; SOUL.md is persona content, not identity config.
|
||
self._lazy_init_kwargs = dict(kwargs)
|
||
self._lazy_init_session_id = session_id
|
||
self._session_key = self._resolve_session_key(cfg, session_id, **kwargs)
|
||
|
||
# Session creation can block on Honcho/DB outages, so context/hybrid startup
|
||
# fails open in a background thread. Tools-only mode has an explicit contract:
|
||
# init_on_session_start=False stays lazy until the first tool call, True is eager.
|
||
if self._recall_mode != "tools":
|
||
self._start_session_init_background(wait_timeout=0.1)
|
||
elif cfg.init_on_session_start:
|
||
self._ensure_session()
|
||
else:
|
||
logger.debug("Honcho tools-only mode — deferring session init until first tool call")
|
||
except ImportError:
|
||
logger.debug("honcho-ai package not installed — plugin inactive")
|
||
except Exception as e:
|
||
logger.warning("Honcho init failed: %s", e)
|
||
self._manager = None
|
||
|
||
def _resolve_session_key(self, cfg, session_id: str, **kwargs) -> str:
|
||
"""Resolve the Honcho session key without touching the network."""
|
||
from agent.runtime_cwd import resolve_agent_cwd
|
||
|
||
cwd = kwargs.get("cwd") or str(resolve_agent_cwd())
|
||
return cfg.resolve_session_name(
|
||
cwd=cwd,
|
||
session_title=kwargs.get("session_title"), session_id=session_id,
|
||
session_title_source=kwargs.get("session_title_source"),
|
||
gateway_session_key=kwargs.get("gateway_session_key"),
|
||
) or session_id or "hermes-default"
|
||
|
||
def _can_start_init(self) -> bool:
|
||
return not (self._cron_skipped or self._session_initialized) and bool(self._config) and self._lazy_init_kwargs is not None
|
||
|
||
def _run_session_init(self, label: str) -> bool:
|
||
"""Run _do_session_init with the deferred kwargs; on failure discard the manager
|
||
and (for auth or unresolved-peer failures) keep the detail for the one-time notice."""
|
||
from plugins.memory.honcho.session import HonchoAuthError
|
||
from plugins.memory.honcho.session_peers import HonchoPeerUnresolvedError
|
||
|
||
init_kwargs = self._lazy_init_kwargs
|
||
if init_kwargs is None: # another init path already consumed the deferred kwargs
|
||
return self._manager is not None
|
||
try:
|
||
self._do_session_init(self._config, self._lazy_init_session_id or "hermes-default", **dict(init_kwargs))
|
||
except Exception as e:
|
||
self._manager = None
|
||
self._session_initialized = False
|
||
detail: object = e
|
||
if isinstance(e, HonchoAuthError):
|
||
# Keep the auth detail so the one-time notice survives the manager discard.
|
||
self._init_auth_failure = str(e)
|
||
detail = "authentication rejected"
|
||
elif isinstance(e, HonchoPeerUnresolvedError):
|
||
# A missing peerName does not heal mid-session, so drop the deferred kwargs and stop retrying.
|
||
self._init_peer_failure = str(e)
|
||
self._init_peer_platform = str(dict(init_kwargs).get("platform") or "cli")
|
||
self._lazy_init_kwargs = self._lazy_init_session_id = None
|
||
logger.warning("Honcho %s session init failed: %s", label, detail)
|
||
return False
|
||
self._lazy_init_kwargs = self._lazy_init_session_id = None
|
||
if self._init_auth_failure is not None:
|
||
self._init_auth_failure = None
|
||
self._init_auth_notice_emitted = False
|
||
return True
|
||
|
||
def _start_session_init_background(self, *, wait_timeout: float = 0.0, blocking: bool = True) -> None:
|
||
"""Start session initialization in a daemon thread so a slow/down Honcho can't
|
||
block agent construction or first prompt assembly. ``wait_timeout`` lets fast
|
||
(mock) initializations finish before returning."""
|
||
if not self._can_start_init():
|
||
return
|
||
if not blocking and not self._init_lock.acquire(blocking=False):
|
||
return
|
||
try:
|
||
with self._init_lock if blocking else contextlib.nullcontext():
|
||
if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()):
|
||
return
|
||
self._init_thread = spawn_context_thread(lambda: self._run_session_init("background"),
|
||
name="honcho-session-init", owner=self)
|
||
self._init_thread.start()
|
||
if wait_timeout > 0:
|
||
self._init_thread.join(timeout=wait_timeout)
|
||
finally:
|
||
if not blocking:
|
||
self._init_lock.release()
|
||
|
||
def _ensure_session(self) -> bool:
|
||
"""Lazily initialize the Honcho session (tools-only mode). True when the manager is ready."""
|
||
if self._manager and self._session_initialized:
|
||
return True
|
||
if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()):
|
||
return False
|
||
return self._run_session_init("lazy") and self._manager is not None
|
||
|
||
def _do_session_init(self, cfg, session_id: str, **kwargs) -> None:
|
||
"""Shared session initialization for both eager and lazy paths."""
|
||
from plugins.memory.honcho.client import get_honcho_client
|
||
from plugins.memory.honcho.session import HonchoSessionManager
|
||
|
||
self._manager = HonchoSessionManager(
|
||
honcho=get_honcho_client(cfg), config=cfg, context_tokens=cfg.context_tokens,
|
||
runtime_user_peer_name=kwargs.get("user_id") or None,
|
||
runtime_user_peer_name_alt=kwargs.get("user_id_alt") or None,
|
||
)
|
||
self._session_key = self._resolve_session_key(cfg, session_id, **kwargs)
|
||
logger.debug("Honcho session key resolved: %s", self._session_key)
|
||
|
||
# The provider is not "ready" until this method returns: background startup sets
|
||
# _manager before get_or_create/migration/prewarm finish, and lifecycle hooks must
|
||
# not treat that partially initialized state as usable.
|
||
session = self._manager.get_or_create(self._session_key)
|
||
|
||
# Per-session strategy creates a fresh Honcho session every run, so a per-run
|
||
# MEMORY.md/USER.md/SOUL.md upload would flood the backend with duplicates.
|
||
if cfg.session_strategy == "per-session":
|
||
logger.debug("Honcho memory file migration skipped: per-session strategy creates a fresh session per run (%s)",
|
||
self._session_key)
|
||
elif not session.messages:
|
||
try:
|
||
from hermes_constants import get_hermes_home
|
||
self._manager.migrate_memory_files(self._session_key, str(get_hermes_home() / "memories"))
|
||
logger.debug("Honcho memory file migration attempted for new session: %s", self._session_key)
|
||
except Exception as e:
|
||
logger.debug("Honcho memory file migration skipped: %s", e)
|
||
|
||
# Generic dialectic prewarm is incompatible with latest-message query rewriting,
|
||
# which needs the first substantive user message.
|
||
if self._recall_mode in {"context", "hybrid"} and not self._recall_sync:
|
||
if self._query_rewriter is None or not self._query_rewrite_enabled:
|
||
self._spawn_dialectic(_PREWARM_QUERY, thread_name="honcho-prewarm-dialectic", fired_at=0,
|
||
log_label="dialectic prewarm", use_query_rewrite=False)
|
||
logger.debug("Honcho dialectic prewarm started for session: %s", self._session_key)
|
||
else:
|
||
logger.debug("Honcho generic dialectic prewarm skipped: awaiting first user message")
|
||
|
||
self._session_initialized = True
|
||
|
||
def _session_ready(self) -> bool:
|
||
"""Whether the manager/session key can be used safely. Background init sets
|
||
``_manager`` before get-or-create completes, so ``_session_initialized`` is the real
|
||
guard; tests/legacy construction may inject a ready manager without the flag —
|
||
allowed only with no init thread in flight."""
|
||
if not self._manager or not self._session_key:
|
||
return False
|
||
return self._session_initialized or not (self._init_thread and self._init_thread.is_alive())
|
||
|
||
def _writes_enabled(self) -> bool:
|
||
"""``saveMessages`` is the operator's hard write gate for every Honcho mutation path."""
|
||
return not self._cron_skipped and getattr(self._config, "save_messages", True)
|
||
|
||
def _ready_or_kick_init(self) -> bool:
|
||
"""True when writes may proceed; otherwise (outside tools mode) start background init."""
|
||
if self._session_ready():
|
||
return True
|
||
if self._recall_mode != "tools":
|
||
self._start_session_init_background()
|
||
return False
|
||
|
||
# ----- Prompt / prefetch -----
|
||
|
||
@staticmethod
|
||
def _resolve_session_start(look: _HostLookup) -> Optional[frozenset]:
|
||
"""The pinned ``injection.sessionStart`` list as a set, or None when unpinned.
|
||
An explicit empty list means inject nothing and stays distinct from unset."""
|
||
injection = look.present("injection")
|
||
if not isinstance(injection, dict):
|
||
return None
|
||
listed = injection.get("sessionStart")
|
||
if not isinstance(listed, (list, tuple)):
|
||
return None
|
||
return frozenset(str(x) for x in listed)
|
||
|
||
def _format_first_turn_context(self, ctx: dict) -> str:
|
||
"""Render the prefetch context, keeping only the ``injection.sessionStart`` components when pinned.
|
||
The summary passes usable_honcho_summary here, so a contaminated one never reaches _base_context_cache."""
|
||
ctx = {**ctx, "summary": usable_honcho_summary(ctx.get("summary")) or ""}
|
||
allowed = self._session_start_components
|
||
parts, suppressed = [], []
|
||
for name, key, header in _CONTEXT_SECTIONS:
|
||
value = ctx.get(key, "")
|
||
if not value:
|
||
continue
|
||
if allowed is not None and name not in allowed:
|
||
suppressed.append(f"{name} ({len(value)}B)")
|
||
continue
|
||
parts.append(f"## {header}\n{value}")
|
||
if suppressed:
|
||
logger.debug("Honcho session-start injection filtered by config: kept %s, suppressed %s",
|
||
[n for n, k, _ in _CONTEXT_SECTIONS if ctx.get(k) and (allowed is None or n in allowed)],
|
||
suppressed)
|
||
return "\n\n".join(parts)
|
||
|
||
def system_prompt_block(self) -> str:
|
||
"""Static mode header + tool instructions (prompt-cache friendly).
|
||
Live context (representation, card) is injected via prefetch()."""
|
||
if self._cron_skipped or not (self._config or (self._manager and self._session_key)):
|
||
return ""
|
||
return _PROMPT_HEADERS.get(self._recall_mode, _PROMPT_HEADERS["hybrid"])
|
||
|
||
@staticmethod
|
||
def _resolve_injection_log_path(look: _HostLookup) -> Optional[str]:
|
||
"""Where to append the injection audit, or None to keep it off.
|
||
The ``logging`` key or HONCHO_LOGGING switches it on. HONCHO_INJECTION_LOG overrides the destination."""
|
||
explicit = os.environ.get("HONCHO_INJECTION_LOG")
|
||
if explicit:
|
||
return explicit
|
||
enabled = _as_flag(look.pick_set("logging"), default=None)
|
||
if enabled is None:
|
||
enabled = _as_flag(os.environ.get("HONCHO_LOGGING"), default=False)
|
||
if not enabled:
|
||
return None
|
||
return os.path.join(os.path.expanduser("~"), ".honcho", "injection.log")
|
||
|
||
def _log_injection(self, reason: str, payload: str = "") -> str:
|
||
"""Append one record of what this turn injected and why, then return ``payload`` unchanged. Never raises.
|
||
The reason matters because prefetch has several ways to return nothing and each needs a different fix."""
|
||
path = self._injection_log_path
|
||
if not path:
|
||
return payload
|
||
try:
|
||
record = json.dumps({
|
||
"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "turn": self._turn_count,
|
||
"session_key": self._session_key or "", "recall_mode": self._recall_mode,
|
||
"reason": reason, "bytes": len(payload.encode("utf-8")), "payload": payload,
|
||
}, ensure_ascii=False)
|
||
with self._injection_log_lock:
|
||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||
# The record holds the user's representation verbatim, so the file is owner-only.
|
||
fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600)
|
||
with os.fdopen(fd, "a", encoding="utf-8") as fh:
|
||
fh.write(record + "\n")
|
||
except Exception as e:
|
||
logger.debug("Honcho injection log write failed: %s", e)
|
||
return payload
|
||
|
||
def _first_turn_wait(self, base: float) -> float:
|
||
"""Turn-1 wait budget: a short request timeout may tighten, but never expand, it."""
|
||
request_timeout = getattr(self._config, "timeout", None)
|
||
return max(0.0, base if request_timeout is None else min(base, max(0.0, request_timeout)))
|
||
|
||
def _fetch_base_context_layer(self, query: str, first_turn_base_deadline: float | None) -> str:
|
||
"""Layer 1: representation + card. The first fetch gets the remaining turn-1 budget;
|
||
later turns consume the refresh queued by the previous turn."""
|
||
with self._base_context_lock:
|
||
first_base_fetch = self._base_context_cache is None
|
||
if first_base_fetch:
|
||
self._base_context_cache = ""
|
||
self._last_context_turn = self._turn_count
|
||
base_context = self._base_context_cache
|
||
|
||
if not self._manager:
|
||
return base_context
|
||
|
||
def _adopt(ctx: dict) -> str:
|
||
"""Cache a fresh context dict's formatted block; keep the old text if it formats empty."""
|
||
formatted = self._format_first_turn_context(ctx)
|
||
if formatted:
|
||
with self._base_context_lock:
|
||
self._base_context_cache = formatted
|
||
return formatted or base_context
|
||
|
||
if not first_base_fetch:
|
||
fresh_ctx = self._manager.pop_context_result(self._session_key)
|
||
return _adopt(fresh_ctx) if fresh_ctx else base_context
|
||
|
||
ctx_holder: dict[str, dict] = {}
|
||
|
||
def _fetch_base() -> None:
|
||
ctx_holder["ctx"] = ctx = self._manager.get_prefetch_context(self._session_key, query or None) or {}
|
||
if ctx:
|
||
self._manager.set_context_result(self._session_key, ctx)
|
||
|
||
bt = self._spawn_write(_fetch_base, "honcho-base-first", "Honcho first-turn base context failed: %s")
|
||
base_wait = max(0.0, first_turn_base_deadline - time.monotonic()) if first_turn_base_deadline is not None else 0.0
|
||
bt.join(timeout=base_wait)
|
||
if ctx := ctx_holder.get("ctx"):
|
||
self._manager.pop_context_result(self._session_key)
|
||
return _adopt(ctx)
|
||
if bt.is_alive():
|
||
logger.debug("Honcho first-turn base context still running after %.1fs — will surface on next turn", base_wait)
|
||
return base_context
|
||
|
||
def _first_turn_dialectic_wait(self, query: str) -> None:
|
||
"""Turn 1 only: reuse an in-flight prewarm or start one dialectic, then wait briefly.
|
||
Unfinished work stays async and surfaces on a later turn."""
|
||
with self._prefetch_lock:
|
||
prewarm_landed = bool(self._prefetch_result)
|
||
if prewarm_landed and self._last_dialectic_turn == -999:
|
||
self._last_dialectic_turn = self._turn_count
|
||
if self._last_dialectic_turn != -999 or not query:
|
||
return
|
||
|
||
dia_wait = self._first_turn_wait(self._FIRST_TURN_DIALECTIC_CAP)
|
||
if not self._thread_is_live():
|
||
self._spawn_dialectic(query, thread_name="honcho-prefetch-first", fired_at=self._turn_count,
|
||
log_label="first-turn dialectic")
|
||
if (live := self._prefetch_thread) is not None:
|
||
live.join(timeout=dia_wait)
|
||
if self._prefetch_thread and self._prefetch_thread.is_alive():
|
||
logger.debug("Honcho first-turn dialectic still running after %.1fs — will surface on next turn", dia_wait)
|
||
|
||
def prefetch(self, query: str, *, session_id: str = "") -> str:
|
||
"""Base context (representation + card, refreshed on context_cadence) plus the
|
||
dialectic supplement (refreshed on dialectic_cadence), within the context budget.
|
||
Empty in tools-only mode."""
|
||
if self._cron_skipped or self._recall_mode == "tools":
|
||
return self._log_injection("cron-or-tools-mode")
|
||
|
||
if self._recall_sync:
|
||
from plugins.memory.honcho.recall_sync import prefetch_sync
|
||
notice = self._pop_auth_notice() or self._pop_peer_notice()
|
||
payload = "\n\n".join(part for part in (notice, prefetch_sync(self, query)) if part)
|
||
return self._log_injection("injected" if payload else "recall-sync-empty", payload)
|
||
|
||
first_turn_base_deadline = (time.monotonic() + self._first_turn_wait(self._FIRST_TURN_BASE_TIMEOUT)
|
||
if self._turn_count <= 1 else None)
|
||
|
||
if not self._session_ready():
|
||
# Only turn 1 may wait for session init; later turns fail open.
|
||
self._start_session_init_background()
|
||
if first_turn_base_deadline is not None and self._init_thread is not None:
|
||
self._init_thread.join(timeout=max(0.0, first_turn_base_deadline - time.monotonic()))
|
||
if not self._session_ready():
|
||
# A failed init still owes the user its one-time notice.
|
||
return self._log_injection("session-not-ready", self._pop_auth_notice() or self._pop_peer_notice())
|
||
|
||
# Trivial turns start no work, but may consume a ready pending result.
|
||
if self._is_trivial_prompt(query):
|
||
ready = self._consume_pending_dialectic()
|
||
return self._log_injection("trivial-prompt", self._truncate_to_budget(ready) if ready else "")
|
||
|
||
# One-time notice, relayed by the model, that auth is dead and memory is paused.
|
||
parts = [self._pop_auth_notice()]
|
||
# First-turn mode suppresses only the base layer; dialectic is independent.
|
||
if not (self._injection_frequency == "first-turn" and self._turn_count > 1):
|
||
parts.append(self._fetch_base_context_layer(query, first_turn_base_deadline))
|
||
self._first_turn_dialectic_wait(query)
|
||
# Consume only results that are already ready; later turns never wait.
|
||
parts.append(self._consume_pending_dialectic())
|
||
parts = [p for p in parts if p and p.strip()]
|
||
if not parts:
|
||
return self._log_injection("fetched-but-empty")
|
||
return self._log_injection("injected", self._truncate_to_budget("\n\n".join(parts)))
|
||
|
||
def _pop_auth_notice(self) -> str:
|
||
"""One-time model-facing notice that Honcho auth expired and memory is paused."""
|
||
# getattr (not a direct call): test fakes install minimal managers without pop_auth_notice.
|
||
pop = getattr(self._manager, "pop_auth_notice", None)
|
||
msg = pop() if callable(pop) else None
|
||
if not isinstance(msg, str) or not msg:
|
||
# Init failures discard the manager; the provider kept the detail.
|
||
if self._init_auth_failure is None or self._init_auth_notice_emitted:
|
||
return ""
|
||
self._init_auth_notice_emitted = True
|
||
msg = self._init_auth_failure
|
||
return ("[Honcho memory status] Authentication with the Honcho memory backend has expired and automatic "
|
||
f"token refresh failed, so memory sync and recall are paused. Reason: {msg}\n"
|
||
"Tell the user (once) that Honcho memory is paused and that running 'hermes honcho setup' "
|
||
"to re-authenticate will restore it.")
|
||
|
||
def _peer_failure_text(self) -> str:
|
||
"""The stored peer failure plus the fix that fits the session's platform. On a gateway the fix is a
|
||
user id from the transport, never peerName: a shared peerName would merge every user onto one peer."""
|
||
text = self._init_peer_failure or ""
|
||
if self._init_peer_platform in _LOCAL_PLATFORMS:
|
||
return f"{text} Set one with 'hermes honcho peer --user <name>'."
|
||
return f"{text} This platform supplied no user id for the chat, so memory stays off here."
|
||
|
||
def _pop_peer_notice(self) -> str:
|
||
"""One-time model-facing notice that no user peer could be named and memory is off."""
|
||
if self._init_peer_failure is None or self._init_peer_notice_emitted:
|
||
return ""
|
||
self._init_peer_notice_emitted = True
|
||
if self._init_peer_platform in _LOCAL_PLATFORMS:
|
||
advice = "Tell the user (once) that Honcho memory is off until honcho.json names a user peer."
|
||
else:
|
||
advice = ("Tell the user (once) that Honcho memory is off for this chat. Do not suggest peerName: "
|
||
"on a shared gateway it would merge every user onto one peer.")
|
||
return f"[Honcho memory status] Honcho memory is off for this session. {self._peer_failure_text()}\n{advice}"
|
||
|
||
def _truncate_to_budget(self, text: str) -> str:
|
||
"""Truncate text to the context_tokens budget (≈4 chars/token) at a word boundary."""
|
||
if not self._config or not self._config.context_tokens:
|
||
return text
|
||
budget_chars = self._config.context_tokens * 4
|
||
if len(text) <= budget_chars:
|
||
return text
|
||
truncated = text[:budget_chars]
|
||
last_space = truncated.rfind(" ")
|
||
return (truncated[:last_space] if last_space > budget_chars * 0.8 else truncated) + " …"
|
||
|
||
def queue_prefetch(self, query: str, *, session_id: str = "") -> None:
|
||
"""Fire background prefetch threads for the upcoming turn.
|
||
Context and dialectic refreshes have independent cadence controls."""
|
||
if self._cron_skipped or self._recall_mode == "tools" or self._recall_sync:
|
||
return
|
||
if not self._session_ready() or not query:
|
||
self._start_session_init_background()
|
||
return
|
||
# Trivial prompts don't warrant either a context refresh or a dialectic call.
|
||
if self._is_trivial_prompt(query):
|
||
return
|
||
|
||
# First-turn-only base context never needs a later refresh.
|
||
context_due = self._context_cadence <= 1 or (self._turn_count - self._last_context_turn) >= self._context_cadence
|
||
if self._injection_frequency != "first-turn" and context_due:
|
||
self._last_context_turn = self._turn_count
|
||
try:
|
||
self._manager.prefetch_context(self._session_key, query)
|
||
except Exception as e:
|
||
logger.debug("Honcho context prefetch failed: %s", e)
|
||
|
||
# Dialectic layer: a hung call older than timeout × multiplier counts as dead.
|
||
if self._thread_is_live():
|
||
logger.debug("Honcho dialectic prefetch skipped: prior thread still running")
|
||
return
|
||
# Cadence gate, widened by the empty-streak backoff so a persistently silent
|
||
# backend doesn't retry every turn forever.
|
||
effective = self._effective_cadence()
|
||
if (self._turn_count - self._last_dialectic_turn) < effective:
|
||
logger.debug("Honcho dialectic prefetch skipped: effective cadence %d (base %d, empty streak %d), turns since last: %d",
|
||
effective, self._dialectic_cadence, self._dialectic_empty_streak,
|
||
self._turn_count - self._last_dialectic_turn)
|
||
return
|
||
self._spawn_dialectic(query, thread_name="honcho-prefetch", fired_at=self._turn_count, log_label="prefetch")
|
||
|
||
# Shared with the core prefetch gate so the two classifiers can never drift apart.
|
||
_is_trivial_prompt = staticmethod(is_trivial_prompt)
|
||
|
||
def identity_signature(self) -> Dict[str, Any]:
|
||
"""Identity-mapping values from honcho.json that bust a cached gateway agent when they change.
|
||
|
||
Memoized on the file's mtime and size, so the per-message call is one stat. ``{}`` when the
|
||
config cannot be read."""
|
||
try:
|
||
path = resolve_config_path()
|
||
try:
|
||
stat = path.stat()
|
||
memo_key = (str(path), stat.st_mtime_ns, stat.st_size)
|
||
except OSError:
|
||
memo_key = (str(path), None, None)
|
||
cached = self._identity_signature_memo.get(memo_key)
|
||
if cached is not None:
|
||
return dict(cached)
|
||
cfg = HonchoClientConfig.from_global_config(config_path=path)
|
||
aliases = cfg.user_peer_aliases if isinstance(cfg.user_peer_aliases, dict) else {}
|
||
values = {
|
||
"workspace": cfg.workspace_id,
|
||
"user_identity": cfg.peer_name,
|
||
"agent_identity": cfg.ai_peer,
|
||
"pin_user_identity": bool(cfg.pin_peer_name),
|
||
"runtime_identity_prefix": cfg.runtime_peer_prefix or "",
|
||
"user_identity_aliases": sorted(aliases.items()),
|
||
"session_prefixing": [bool(cfg.session_peer_prefix), bool(cfg.session_ai_peer_prefix)],
|
||
"a2a_sessions": bool(cfg.a2a_sessions),
|
||
}
|
||
self._identity_signature_memo = {memo_key: values}
|
||
return dict(values)
|
||
except Exception:
|
||
return {}
|
||
|
||
def on_turn_start(self, turn_number: int, message: str, **kwargs) -> None:
|
||
"""Track turn count for cadence, and record who wrote this turn: a shared session carries
|
||
several participants, and the peer resolved at session init only names whoever opened it."""
|
||
self._recall_generation = object()
|
||
self._turn_count = turn_number
|
||
self._turn_author = {"id": kwargs.get("author_id") or None, "name": kwargs.get("author_name") or None,
|
||
"is_bot": bool(kwargs.get("author_is_bot"))}
|
||
|
||
def on_session_switch(self, new_session_id: str, **kwargs) -> None:
|
||
"""Discard in-flight recall even when the configured backend session is pinned."""
|
||
self._recall_generation = object()
|
||
|
||
# ----- Writes -----
|
||
|
||
@staticmethod
|
||
def _chunk_message(content: str, limit: int) -> list[str]:
|
||
"""Split content to fit the Honcho message limit, cutting at paragraph, then
|
||
sentence, then word boundaries; continuation chunks get a "[continued] " prefix
|
||
so Honcho's representation engine can reconstruct the full message."""
|
||
if len(content) <= limit:
|
||
return [content]
|
||
|
||
prefix = "[continued] "
|
||
chunks: list[str] = []
|
||
remaining, first = content, True
|
||
while remaining:
|
||
effective = limit if first else limit - len(prefix)
|
||
if len(remaining) <= effective:
|
||
chunks.append(remaining if first else prefix + remaining)
|
||
break
|
||
|
||
segment = remaining[:effective]
|
||
# Paragraph, then sentence (keeping ". "), then word boundary; else a hard cut.
|
||
for sep in ("\n\n", ". ", " "):
|
||
cut = segment.rfind(sep)
|
||
if cut >= 0 and sep == ". ":
|
||
cut += 2
|
||
if cut >= effective * 0.3:
|
||
break
|
||
else:
|
||
cut = effective
|
||
|
||
chunk = remaining[:cut].rstrip()
|
||
remaining = remaining[cut:].lstrip()
|
||
chunks.append(chunk if first else prefix + chunk)
|
||
first = False
|
||
|
||
return chunks
|
||
|
||
def sync_turn(
|
||
self, user_content: str, assistant_content: str, *, session_id: str = "",
|
||
turn_author: Optional[Dict[str, Any]] = None,
|
||
) -> None:
|
||
"""Record the conversation turn in Honcho (non-blocking), chunking messages that
|
||
exceed the Honcho API limit. Honors saveMessages: false. ``turn_author`` names who wrote
|
||
the user side. The ``on_turn_start`` stash is the fallback for callers that never pass it.
|
||
A bot author's turn is written into that bot's own a2a session, never the human's."""
|
||
if not self._writes_enabled():
|
||
return
|
||
if _is_internal_gateway_turn(user_content):
|
||
logger.debug("Honcho sync skipped machine-generated gateway turn")
|
||
return
|
||
if not self._ready_or_kick_init():
|
||
return
|
||
|
||
msg_limit = self._config.message_max_chars if self._config else 25000
|
||
clean_user_content = sanitize_context(user_content or "").strip()
|
||
clean_assistant_content = sanitize_context(assistant_content or "").strip()
|
||
# Skip only when the whole turn is empty: an interrupted or tool-only turn can have
|
||
# an empty assistant side, and the user's message must still be persisted.
|
||
if not clean_user_content and not clean_assistant_content:
|
||
return
|
||
|
||
author = turn_author if isinstance(turn_author, dict) else self._turn_author
|
||
session_kwargs: dict[str, str] = {}
|
||
if author.get("is_bot"):
|
||
# A bot's turn never lands in the human's session: its own a2a session or nothing.
|
||
if not getattr(self._config, "a2a_sessions", True):
|
||
logger.debug("Honcho sync skipped a bot-authored turn because a2aSessions is off")
|
||
return
|
||
author_id = str(author.get("id") or "").strip()
|
||
if not author_id:
|
||
logger.debug("Honcho sync skipped a bot-authored turn that named no author id")
|
||
return
|
||
bot_peer_id = self._manager.resolve_author_peer_id(
|
||
self._session_key, author_id, author.get("name"), is_bot=True)
|
||
if not bot_peer_id or bot_peer_id == self._manager.assistant_peer_id():
|
||
logger.debug("Honcho sync skipped a bot-authored turn: author %s has no peer of its own", author_id)
|
||
return
|
||
session_key = self._a2a_session_key({**author, "id": author_id})
|
||
# The bot is the a2a session's own user peer, so its messages need no per-message author.
|
||
session_kwargs["user_peer_id"] = bot_peer_id
|
||
author_peer_id = None
|
||
else:
|
||
session_key = self._session_key
|
||
# Resolved before the thread starts so a following turn cannot retag a queued write.
|
||
author_peer_id = self._manager.resolve_author_peer_id(session_key, author.get("id"), author.get("name"))
|
||
|
||
def _sync():
|
||
session = self._manager.get_or_create(session_key, **session_kwargs)
|
||
for chunk in self._chunk_message(clean_user_content, msg_limit) if clean_user_content else ():
|
||
session.add_message("user", chunk, author_peer_id=author_peer_id)
|
||
for chunk in self._chunk_message(clean_assistant_content, msg_limit) if clean_assistant_content else ():
|
||
session.add_message("assistant", chunk)
|
||
# save() (not _flush_session) so writeFrequency batching is honored.
|
||
self._manager.save(session)
|
||
|
||
if self._sync_thread and self._sync_thread.is_alive():
|
||
self._sync_thread.join(timeout=5.0)
|
||
self._sync_thread = self._spawn_write(_sync, "honcho-sync", "Honcho sync_turn failed: %s")
|
||
|
||
def _a2a_session_key(self, author: Dict[str, Any]) -> str:
|
||
"""Honcho session for one sender bot's turns into this agent, named from core's ``a2a_key``.
|
||
|
||
This agent's ``aiPeer`` is in the key because two profiles can share a workspace and a session
|
||
key. The digest keeps two ids apart when sanitizing would make them equal."""
|
||
prefix, _, ident = (a2a_key(author) or "").partition(":")
|
||
digest = hashlib.sha256(ident.encode("utf-8")).hexdigest()[:8]
|
||
recipient = assistant_peer_id_for(self._config)
|
||
key = f"{self._session_key}:{prefix}:{recipient}:{sanitize_peer_id(ident)}-{digest}"
|
||
return HonchoClientConfig._enforce_session_id_limit(key, key)
|
||
|
||
def _bot_turn_write_refusal(self) -> Optional[str]:
|
||
"""Refusal for memory writes while a bot-authored turn runs. Conclusions and cards describe the human."""
|
||
if self._turn_author.get("is_bot"):
|
||
return tool_error("Honcho memory writes are off during a bot-to-bot turn. Conclusions and profile edits describe the human.")
|
||
return None
|
||
|
||
def _spawn_write(self, fn: Callable[[], None], name: str, fail_msg: str) -> threading.Thread:
|
||
"""Run a Honcho write off-thread; failures are debug-logged, never raised into the turn."""
|
||
def _run():
|
||
try:
|
||
fn()
|
||
except Exception as e:
|
||
logger.debug(fail_msg, e)
|
||
|
||
thread = spawn_context_thread(_run, name=name, owner=self)
|
||
thread.start()
|
||
return thread
|
||
|
||
def on_memory_write(
|
||
self, action: str, target: str, content: str, metadata: Optional[Dict[str, Any]] = None,
|
||
) -> None:
|
||
"""Mirror built-in user-profile writes as Honcho conclusions (``metadata`` accepted
|
||
for interface compatibility, not yet threaded into the conclusion payload)."""
|
||
if action != "add" or target != "user" or not content:
|
||
return
|
||
if self._turn_author.get("is_bot"):
|
||
logger.debug("Honcho memory mirror skipped during a bot-authored turn")
|
||
return
|
||
if not self._writes_enabled() or not self._ready_or_kick_init():
|
||
return
|
||
self._memwrite_thread = self._spawn_write(lambda: self._manager.create_conclusion(self._session_key, content),
|
||
"honcho-memwrite", "Honcho memory mirror failed: %s")
|
||
|
||
def on_session_end(self, messages: List[Dict[str, Any]]) -> None:
|
||
"""Flush all pending messages to Honcho on session end."""
|
||
if not self._writes_enabled() or not self._manager:
|
||
return
|
||
if not self._session_initialized and self._init_thread and self._init_thread.is_alive():
|
||
return
|
||
if self._sync_thread and self._sync_thread.is_alive():
|
||
self._sync_thread.join(timeout=10.0)
|
||
try:
|
||
self._manager.flush_all()
|
||
except Exception as e:
|
||
logger.debug("Honcho session-end flush failed: %s", e)
|
||
|
||
# ----- Tools -----
|
||
|
||
def get_tool_schemas(self) -> List[Dict[str, Any]]:
|
||
"""Tool schemas by recall_mode; context-only mode exposes no Honcho tools."""
|
||
if self._cron_skipped or self._recall_mode == "context":
|
||
return []
|
||
return list(ALL_TOOL_SCHEMAS)
|
||
|
||
def _empty_profile_hint(self, peer: str) -> Dict[str, Any]:
|
||
"""Diagnostic hint for an empty honcho_profile card, so the model can explain WHY
|
||
instead of surfacing a cryptic "no facts" to the user. Likely causes, in order:
|
||
observation disabled for the peer; card not accumulated yet (fresh peer / few
|
||
dialectic cycles); self-hosted Honcho < 3.x without peer-card support."""
|
||
cfg = self._config
|
||
reasons: List[str] = []
|
||
kind = "user" if peer == "user" else "ai"
|
||
if cfg is not None and not (getattr(cfg, f"{kind}_observe_me", True) or getattr(cfg, f"{kind}_observe_others", True)):
|
||
reasons.append(f"observation is disabled for peer '{peer}' (user_observe_me/ai_observe_me in config)")
|
||
cadence, turn = self._dialectic_cadence, self._turn_count
|
||
if turn < max(2, cadence):
|
||
reasons.append(f"this session has only {turn} turn(s); peer cards accumulate as the dialectic "
|
||
f"layer reasons over conversation history (cadence every {cadence} turn(s))")
|
||
if not reasons:
|
||
reasons.append("peer card has no facts yet — Honcho's dialectic layer builds this over time from "
|
||
"observed turns; self-hosted Honcho < 3.x does not support peer cards at all")
|
||
return {
|
||
"result": "No profile facts available yet.",
|
||
"hint": ("This is not an error. " + "; ".join(reasons)
|
||
+ ". Try honcho_reasoning for a synthesized answer, or honcho_search to query raw conversation excerpts."),
|
||
}
|
||
|
||
def _tool_profile(self, args: dict) -> str:
|
||
peer = args.get("peer", "user")
|
||
if card_update := args.get("card"):
|
||
if refusal := self._bot_turn_write_refusal():
|
||
return refusal
|
||
result = self._manager.set_peer_card(self._session_key, card_update, peer=peer)
|
||
if result is None:
|
||
return tool_error("Failed to update peer card.")
|
||
return json.dumps({"result": f"Peer card updated ({len(result)} facts).", "card": result})
|
||
card = self._manager.get_peer_card(self._session_key, peer=peer)
|
||
return json.dumps({"result": card} if card else self._empty_profile_hint(peer))
|
||
|
||
def _tool_search(self, args: dict) -> str:
|
||
if not (query := (args.get("query") or "").strip()):
|
||
return tool_error("Missing required parameter: query")
|
||
max_tokens = min(int(args.get("max_tokens", 800)), 2000)
|
||
result = self._manager.search_context(self._session_key, query, max_tokens=max_tokens, peer=args.get("peer", "user"))
|
||
return json.dumps({"result": result or "No relevant context found."})
|
||
|
||
def _tool_reasoning(self, args: dict) -> str:
|
||
from plugins.memory.honcho.session import HonchoAuthError
|
||
|
||
if not (query := (args.get("query") or "").strip()):
|
||
return tool_error("Missing required parameter: query")
|
||
try:
|
||
# Explicit reasoning bypasses the automatic-injection cap, and surfaces
|
||
# timeouts/server errors as errors rather than an indistinguishable "no result".
|
||
result = self._manager.dialectic_query(
|
||
self._session_key, query, reasoning_level=args.get("reasoning_level"),
|
||
peer=args.get("peer", "user"), apply_injection_cap=False, raise_errors=True,
|
||
)
|
||
except HonchoAuthError:
|
||
raise # rendered by handle_tool_call's auth-specific handler
|
||
except Exception as e:
|
||
logger.warning("honcho_reasoning failed: %s", e)
|
||
return tool_error(
|
||
f"Honcho reasoning query failed ({e}). This is a backend error, not an empty result — "
|
||
"the peer may still have relevant context. Slow dialectic calls at higher reasoning levels "
|
||
"can exceed the configured timeout; consider a lower reasoning_level or raising the "
|
||
"'timeout' value in honcho.json."
|
||
)
|
||
# Auto-injection respects the cadence gap after an explicit call.
|
||
self._last_dialectic_turn = self._turn_count
|
||
return json.dumps({"result": result or "No result from Honcho."})
|
||
|
||
def _tool_context(self, args: dict) -> str:
|
||
ctx = self._manager.get_session_context(self._session_key, peer=args.get("peer", "user"))
|
||
if not ctx:
|
||
return json.dumps({"result": "No context available yet."})
|
||
sections = (("Summary", usable_honcho_summary(ctx.get("summary"))),
|
||
("Representation", ctx.get("representation")), ("Card", ctx.get("card")))
|
||
parts = [f"## {header}\n{value}" for header, value in sections if value]
|
||
if recent := ctx.get("recent_messages"):
|
||
parts.append("## Recent messages\n" + "\n".join(f" [{m['role']}] {m['content'][:200]}" for m in recent[-5:]))
|
||
return json.dumps({"result": "\n\n".join(parts) or "No context available."})
|
||
|
||
def _tool_conclude(self, args: dict) -> str:
|
||
delete_id = (args.get("delete_id") or "").strip()
|
||
conclusion = args.get("conclusion", "").strip()
|
||
list_mode = bool(args.get("list"))
|
||
peer = args.get("peer", "user")
|
||
if sum([bool(delete_id), bool(conclusion), list_mode]) != 1:
|
||
return tool_error("Exactly one of conclusion, delete_id, or list must be provided.")
|
||
query = (args.get("query") or "").strip()
|
||
if query and not list_mode:
|
||
return tool_error("query is only valid when list is true.")
|
||
|
||
if list_mode:
|
||
return json.dumps({"conclusions": self._manager.list_conclusions(self._session_key, query=query or None, peer=peer)})
|
||
if refusal := self._bot_turn_write_refusal():
|
||
return refusal
|
||
if delete_id:
|
||
if self._manager.delete_conclusion(self._session_key, delete_id, peer=peer):
|
||
return json.dumps({"result": f"Conclusion {delete_id} deleted."})
|
||
return tool_error(f"Failed to delete conclusion {delete_id}.")
|
||
if self._manager.create_conclusion(self._session_key, conclusion, peer=peer):
|
||
return json.dumps({"result": f"Conclusion saved for {peer}: {conclusion}"})
|
||
return tool_error("Failed to save conclusion.")
|
||
|
||
_TOOL_HANDLERS = {
|
||
"honcho_profile": _tool_profile,
|
||
"honcho_search": _tool_search,
|
||
"honcho_reasoning": _tool_reasoning,
|
||
"honcho_context": _tool_context,
|
||
"honcho_conclude": _tool_conclude,
|
||
}
|
||
|
||
def handle_tool_call(self, tool_name: str, args: dict, **kwargs) -> str:
|
||
"""Dispatch a Honcho tool call, lazily initializing the session in tools-only mode."""
|
||
from plugins.memory.honcho.session import HonchoAuthError
|
||
|
||
if self._cron_skipped:
|
||
return tool_error("Honcho is not active (cron context).")
|
||
if not self._session_initialized:
|
||
if self._init_thread and self._init_thread.is_alive():
|
||
return tool_error("Honcho session is still initializing; try again shortly.")
|
||
if not self._ensure_session():
|
||
if self._init_auth_failure:
|
||
return tool_error(f"Honcho memory authentication failed: {self._init_auth_failure}")
|
||
if self._init_peer_failure:
|
||
return tool_error(self._peer_failure_text())
|
||
return tool_error("Honcho session could not be initialized.")
|
||
if not self._manager or not self._session_key:
|
||
return tool_error("Honcho is not active for this session.")
|
||
if (handler := self._TOOL_HANDLERS.get(tool_name)) is None:
|
||
return tool_error(f"Unknown tool: {tool_name}")
|
||
try:
|
||
return handler(self, args)
|
||
except HonchoAuthError as e:
|
||
# Never report an auth failure as an empty result; the model would read it as "no memory".
|
||
logger.error("Honcho tool %s failed: authentication rejected", tool_name)
|
||
return tool_error(f"Honcho memory authentication failed: {e}")
|
||
except Exception as e:
|
||
logger.error("Honcho tool %s failed: %s", tool_name, e)
|
||
return tool_error(f"Honcho {tool_name} failed: {e}")
|
||
|
||
# Shutdown never joins for less than this. A thread blocked in httpx can hold the HTTP timeout.
|
||
_SHUTDOWN_JOIN_FLOOR = 5.0
|
||
|
||
def _shutdown_join_budget(self) -> float:
|
||
"""The floor, or the configured HTTP timeout when longer, so a thread blocked in a Honcho call can finish."""
|
||
from plugins.memory.honcho.client_cache import _resolve_timeout_from_sources
|
||
return max(self._SHUTDOWN_JOIN_FLOOR, _resolve_timeout_from_sources(self._config))
|
||
|
||
def shutdown(self) -> None:
|
||
"""Join the write threads, flush and stop the manager, then join every other thread this provider or its
|
||
manager spawned, all within one budget. A daemon thread still blocked in httpx I/O at exit aborts the process."""
|
||
self._recall_generation = object()
|
||
budget = self._shutdown_join_budget()
|
||
deadline = time.monotonic() + budget
|
||
for t in (self._sync_thread, self._memwrite_thread):
|
||
if t and t.is_alive():
|
||
t.join(timeout=max(0.0, deadline - time.monotonic()))
|
||
manager = self._manager
|
||
if manager and not (self._init_thread and self._init_thread.is_alive() and not self._session_initialized):
|
||
# saveMessages: false skips persistence, but the async-writer thread must still be joined.
|
||
with contextlib.suppress(Exception):
|
||
remaining = max(0.0, deadline - time.monotonic())
|
||
if getattr(self._config, "save_messages", True):
|
||
manager.shutdown(timeout=remaining) # flush_all() + join the writer
|
||
else:
|
||
manager.stop_async_writer(timeout=remaining)
|
||
left = join_plugin_threads((self, manager), timeout=max(0.0, deadline - time.monotonic()))
|
||
if left:
|
||
logger.warning("Honcho shutdown timed out after %.1fs with %d thread(s) still running: %s",
|
||
budget, len(left), ", ".join(left))
|
||
|
||
|
||
def register(ctx) -> None:
|
||
"""Register Honcho as a memory provider plugin."""
|
||
from plugins.memory.query_rewrite import rewrite_memory_query
|
||
|
||
ctx.register_memory_provider(HonchoMemoryProvider(query_rewriter=rewrite_memory_query))
|
||
|
||
|
||
# ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ----
|
||
# Names external plugins imported from this module before the Sep 2026 decomposition.
|
||
# Internal code MUST NOT use these (scripts/check_compat_pointers.py fails CI if it does).
|
||
# The whole block is removed by reverting the commit that added it.
|
||
|
||
CONCLUDE_SCHEMA = {
|
||
"name": "honcho_conclude",
|
||
"description": (
|
||
"Write, delete, or list CONCLUSIONS — persistent, derived facts about a peer that "
|
||
"feeds their long-term profile (card + representation). Use this to record "
|
||
"something durable you've learned about the peer (a stable preference, a "
|
||
"correction, a standing constraint) so future sessions carry it forward. "
|
||
"You MUST pass exactly one of `conclusion` (to create), `delete_id` (to "
|
||
"delete), or `list` (to list/search); any other combination is an error. "
|
||
"A deletion ID is an opaque server-generated string: first call with `list=true` "
|
||
"and optionally `query`, then pass the returned ID as `delete_id`. "
|
||
"Deletion exists only for "
|
||
"PII removal — for merely wrong facts, write a corrected conclusion instead; "
|
||
"Honcho self-heals contradictions over time. This is a WRITE tool: to read "
|
||
"the profile use honcho_profile / honcho_context, and to search what was "
|
||
"said use honcho_search."
|
||
),
|
||
"parameters": {
|
||
"type": "object",
|
||
"properties": {
|
||
"conclusion": {
|
||
"type": "string",
|
||
"description": "A factual statement to persist. Provide this when creating a conclusion. Do not send it together with delete_id or list.",
|
||
},
|
||
"delete_id": {
|
||
"type": "string",
|
||
"description": "Conclusion ID to delete for PII removal. Provide this when deleting a conclusion. Do not send it together with conclusion or list. Get this id from a prior `list` call — never guess it.",
|
||
},
|
||
"list": {
|
||
"type": "boolean",
|
||
"description": "Set to true to list or search stored conclusions (with their ids) instead of creating or deleting one. Do not send together with conclusion or delete_id.",
|
||
},
|
||
"query": {
|
||
"type": "string",
|
||
"description": "Optional semantic search query, used only when `list` is true. Omit to list the most recent conclusions instead of searching.",
|
||
},
|
||
"peer": {
|
||
"type": "string",
|
||
"description": "The peer the conclusion is ABOUT. Built-in aliases: 'user' (default), 'ai'. Or pass any peer ID from this workspace.",
|
||
},
|
||
},
|
||
"required": [],
|
||
},
|
||
}
|
||
|
||
CONTEXT_SCHEMA = {
|
||
"name": "honcho_context",
|
||
"description": (
|
||
"Retrieve the standing SNAPSHOT Honcho holds for the current session — "
|
||
"session summary, the peer's representation, the peer card, and the most "
|
||
"recent messages — in one call. No query, no LLM synthesis (cheaper than "
|
||
"honcho_reasoning). Use it to orient yourself on what Honcho currently "
|
||
"knows about this conversation and peer. This is a fixed snapshot, not a "
|
||
"search: to look up a specific past fact use honcho_search; to ask a "
|
||
"question and get a synthesized answer use honcho_reasoning; for just the "
|
||
"compact card use honcho_profile."
|
||
),
|
||
"parameters": {
|
||
"type": "object",
|
||
"properties": {
|
||
"peer": {
|
||
"type": "string",
|
||
"description": "Peer to query. Built-in aliases: 'user' (default), 'ai'. Or pass any peer ID from this workspace.",
|
||
},
|
||
},
|
||
"required": [],
|
||
},
|
||
}
|
||
|
||
PROFILE_SCHEMA = {
|
||
"name": "honcho_profile",
|
||
"description": (
|
||
"Read or write a peer's CARD — a short, curated list of standing facts "
|
||
"about that peer (name, role, preferences, communication style, recurring "
|
||
"patterns). This is the cheapest, fastest Honcho call: no query, no LLM, "
|
||
"just the current card. Pass `card` to overwrite it; omit `card` to read. "
|
||
"An empty read returns a `hint` explaining why (observation disabled, fresh "
|
||
"peer, representation still warming up) — that is NOT an error; the card "
|
||
"accumulates over time from observed conversation. "
|
||
"Related tools: honcho_context for the fuller standing snapshot (card + "
|
||
"representation + summary + recent messages); honcho_search to find "
|
||
"specific things that were actually said; honcho_reasoning for a "
|
||
"synthesized answer to a question."
|
||
),
|
||
"parameters": {
|
||
"type": "object",
|
||
"properties": {
|
||
"peer": {
|
||
"type": "string",
|
||
"description": "Peer to query. Built-in aliases: 'user' (default), 'ai'. Or pass any peer ID from this workspace.",
|
||
},
|
||
"card": {
|
||
"type": "array",
|
||
"items": {"type": "string"},
|
||
"description": "New peer card as a list of fact strings. Omit to read the current card.",
|
||
},
|
||
},
|
||
"required": [],
|
||
},
|
||
}
|
||
|
||
REASONING_SCHEMA = {
|
||
"name": "honcho_reasoning",
|
||
"description": (
|
||
"Ask Honcho's dialectic agent a natural-language question about a peer and "
|
||
"get back a SYNTHESIZED answer. This is the only Honcho tool that runs an "
|
||
"LLM: it agentically searches both raw messages and derived conclusions, "
|
||
"reasons over them, and writes a prose answer — so it is the slowest and "
|
||
"most expensive call (seconds + tokens). Reach for it for nuanced or "
|
||
"open-ended questions ('how does this person prefer to receive feedback?', "
|
||
"'what's their relationship to project X?') where you want Honcho to do the "
|
||
"synthesis. For a specific fact that was stated, prefer honcho_search "
|
||
"(cheap, raw excerpts, you synthesize). For standing profile facts, prefer "
|
||
"honcho_profile / honcho_context (no LLM). "
|
||
"Pass reasoning_level to control depth: minimal (fast/cheap), low (default), "
|
||
"medium, high, max (deep/expensive). Omit for the configured default."
|
||
),
|
||
"parameters": {
|
||
"type": "object",
|
||
"properties": {
|
||
"query": {
|
||
"type": "string",
|
||
"description": "A natural language question.",
|
||
},
|
||
"reasoning_level": {
|
||
"type": "string",
|
||
"description": (
|
||
"Override the default reasoning depth. "
|
||
"Omit to use the configured default (typically low).\n"
|
||
"reasoning_level parameter guide:\n"
|
||
"- minimal: use ONLY for a single quick factual lookup (e.g. "
|
||
"'what is the user's name'). Honcho hard-caps this tier's output "
|
||
"at 250 tokens combined with the model's own hidden reasoning "
|
||
"tokens — a multi-part answer can get cut off mid-thought before "
|
||
"it even reaches the final-answer phase, especially on models "
|
||
"with reasoning/thinking enabled.\n"
|
||
"- low/medium/high/max: use for anything requiring a synthesized, "
|
||
"multi-fact, or summary-style answer (e.g. 'summarize known facts "
|
||
"about this peer', 'what are their communication preferences'). "
|
||
"These tiers have no output-token cap of their own (fall back to "
|
||
"Honcho's 8192-token global default), so they don't have "
|
||
"minimal's cutoff failure mode.\n"
|
||
" - low: straightforward questions with clear answers\n"
|
||
" - medium: multi-aspect questions requiring synthesis across observations\n"
|
||
" - high: complex behavioral patterns, contradictions, deep analysis\n"
|
||
" - max: thorough audit-level analysis, leave no stone unturned\n"
|
||
"Default to at least 'low' unless the query is genuinely a single "
|
||
"fact lookup."
|
||
),
|
||
"enum": ["minimal", "low", "medium", "high", "max"],
|
||
},
|
||
"peer": {
|
||
"type": "string",
|
||
"description": "Peer to query. Built-in aliases: 'user' (default), 'ai'. Or pass any peer ID from this workspace.",
|
||
},
|
||
},
|
||
"required": ["query"],
|
||
},
|
||
}
|
||
|
||
SEARCH_SCHEMA = {
|
||
"name": "honcho_search",
|
||
"description": (
|
||
"Hybrid (semantic + keyword) search over a peer's actual message "
|
||
"history across ALL past sessions they took part in — not just the "
|
||
"current one. Returns RRF-ranked raw message excerpts (what was "
|
||
"literally said, including the assistant's own messages about the "
|
||
"peer), no LLM synthesis. Cheaper and faster than honcho_reasoning. "
|
||
"Use this to recall specific past facts — 'what did I say about X', "
|
||
"'what was the regimen/decision/config we settled on' — and reason "
|
||
"over the excerpts yourself. For nuanced questions needing synthesis, "
|
||
"use honcho_reasoning instead."
|
||
),
|
||
"parameters": {
|
||
"type": "object",
|
||
"properties": {
|
||
"query": {
|
||
"type": "string",
|
||
"description": "What to look for — a topic, keyword, name, or natural-language description of the fact you're trying to recall.",
|
||
},
|
||
"max_tokens": {
|
||
"type": "integer",
|
||
"description": "Approximate budget for returned excerpts (default 800, max 2000). Larger budgets return more/longer ranked snippets.",
|
||
},
|
||
"peer": {
|
||
"type": "string",
|
||
"description": "Whose history to search. Built-in aliases: 'user' (default), 'ai'. Or pass any peer ID from this workspace. Spans every session that peer took part in.",
|
||
},
|
||
},
|
||
"required": ["query"],
|
||
},
|
||
}
|
||
|
||
|
||
_PLUGIN_COMPAT_LAZY = {
|
||
'TRIVIAL_PROMPT_RE': ('agent.memory_provider', 'TRIVIAL_PROMPT_RE'),
|
||
}
|
||
|
||
|
||
def __getattr__(name): # PEP 562 — lazy so no import cycles
|
||
target = _PLUGIN_COMPAT_LAZY.get(name)
|
||
if target is None:
|
||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||
import importlib
|
||
from hermes_cli.plugin_compat import warn_once
|
||
warn_once(__name__, name, *target)
|
||
return getattr(importlib.import_module(target[0]), target[1])
|
||
# ---- END PLUGIN-COMPAT ----
|