Files
hermes-agent/plugins/memory/honcho/__init__.py

887 lines
42 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Honcho memory plugin — MemoryProvider for Honcho AI-native memory.
Cross-session user modeling with dialectic Q&A, semantic search, peer cards and
persistent conclusions via the Honcho SDK. Five tools (profile, search, reasoning,
context, conclude) are exposed through the MemoryProvider interface.
Config chain: $HERMES_HOME/honcho.json (profile-scoped) -> ~/.honcho/config.json
(legacy global) -> environment variables.
"""
from __future__ import annotations
import json
import logging
import re
import threading
import time
from typing import Any, Callable, Dict, List, Optional
from agent.memory_manager import sanitize_context
from agent.memory_provider import MemoryProvider, is_trivial_prompt
from plugins.memory.honcho.client import spawn_context_thread
from plugins.memory.honcho.dialectic import DialecticMixin
from plugins.memory.honcho.tool_schemas import ( # noqa: F401 — re-exported
ALL_TOOL_SCHEMAS,
CONCLUDE_SCHEMA,
CONTEXT_SCHEMA,
PROFILE_SCHEMA,
REASONING_SCHEMA,
SEARCH_SCHEMA,
)
from tools.registry import tool_error
logger = logging.getLogger(__name__)
# Gateway-internal notifications arrive through the same user-role channel as genuine
# user messages; they are execution metadata and must never become durable memory.
# Deliberately anchored: a human discussing one of these strings mid-message is valid input.
_INTERNAL_GATEWAY_TURN_RE = re.compile(
r"^\s*(?:"
r"\[ASYNC (?:DELEGATION )?(?:BATCH )?COMPLETE[^\]]*\]|"
r"\[CONTEXT COMPACTION[^\]]*\]|"
r"\[CONTEXT SUMMARY\]:?|"
r"\[PRIOR CONTEXT[^\]]*\]|"
r"\[Your active task list was preserved across context compression\]|"
r"\[IMPORTANT: Background process \d+ matched watch pattern[^\n]*|"
r"A background fan-out of \d+ subagent\(s\) you dispatched earlier has finished\.|"
r"A background subagent you dispatched earlier has finished\."
r")",
re.IGNORECASE,
)
def _is_internal_gateway_turn(text: str) -> bool:
"""Return True for machine-generated gateway/delegation notifications."""
return bool(_INTERNAL_GATEWAY_TURN_RE.match(text or ""))
def _cfg_usable(cfg) -> bool:
"""Enabled with a credential or a self-hosted URL to talk to."""
return bool(cfg.enabled and (cfg.api_key or cfg.base_url))
# Static per-mode system prompt text (prompt-cache friendly: never changes between turns).
_TOOL_GUIDE = (
"Use honcho_profile for a quick factual snapshot, "
"honcho_search for raw excerpts, honcho_context for raw peer context, "
"honcho_reasoning for synthesized answers (pass reasoning_level "
"minimal/low/medium/high/max — you pick the depth per call), "
"honcho_conclude to save facts about the user."
)
_PROMPT_HEADERS = {
"context": (
"# Honcho Memory\nActive (context-injection mode). Relevant user context is automatically "
"injected before each turn. No memory tools are available — context is managed automatically."
),
"tools": (
f"# Honcho Memory\nActive (tools-only mode). {_TOOL_GUIDE} "
"No automatic context injection — you must use tools to access memory."
),
"hybrid": (
"# Honcho Memory\nActive (hybrid mode). Relevant context is auto-injected AND memory tools "
f"are available. {_TOOL_GUIDE}"
),
}
# (context key, section header) for the injected base-context block, in display order.
_CONTEXT_SECTIONS = (
("summary", "Session Summary"),
("representation", "User Representation"),
("card", "User Peer Card"),
("ai_representation", "AI Self-Representation"),
("ai_card", "AI Identity Card"),
)
_PREWARM_QUERY = "Summarize what you know about this user. Focus on preferences, current projects, and working style."
class HonchoMemoryProvider(DialecticMixin, MemoryProvider):
"""Honcho AI-native memory with dialectic Q&A and persistent user modeling."""
def backup_paths(self) -> List[str]:
"""Whole ~/.honcho dir (peer/session config when no profile-local honcho.json exists)."""
try:
from .client import resolve_global_config_path
return [str(resolve_global_config_path().parent)]
except Exception:
return []
def __init__(self, query_rewriter: Optional[Callable[[str], str]] = None):
self._manager = None # HonchoSessionManager
self._config = None # HonchoClientConfig
self._session_key = ""
self._query_rewriter = query_rewriter
self._prefetch_result = ""
self._prefetch_lock = threading.Lock()
self._prefetch_thread: Optional[threading.Thread] = None
self._sync_thread: Optional[threading.Thread] = None
self._memwrite_thread: Optional[threading.Thread] = None
self._recall_mode = "hybrid" # "context", "tools", or "hybrid"
# Base context cache — refreshed on context_cadence, not frozen
self._base_context_cache: Optional[str] = None
self._base_context_lock = threading.Lock()
# Recall cadence state (overwritten from config in initialize()).
self._turn_count = 0
self._query_rewrite_enabled = False
self._injection_frequency = "every-turn" # or "first-turn"
self._context_cadence = 1 # minimum turns between context API calls
self._dialectic_cadence = 1 # backwards-compat fallback; wizard writes 2 on new configs
self._dialectic_depth = 1 # .chat() calls per dialectic cycle (1-3)
self._dialectic_depth_levels: list[str] | None = None # per-pass reasoning levels
self._reasoning_heuristic: bool = True # scale base level by query length
self._reasoning_level_cap: str = "high" # ceiling for auto-selected level
self._last_context_turn = self._last_dialectic_turn = -999
# Liveness state
self._prefetch_thread_started_at: float = 0.0 # monotonic ts of current thread
self._prefetch_result_fired_at: int = -999 # turn the pending result was fired at
self._dialectic_empty_streak: int = 0 # consecutive empty returns
# Tools-only mode may defer session initialization until a tool call.
self._session_initialized = False
self._lazy_init_kwargs: Optional[dict] = None
self._lazy_init_session_id: Optional[str] = None
self._init_thread: Optional[threading.Thread] = None
self._init_lock = threading.Lock()
# Init auth failures live here because the failed manager is discarded.
self._init_auth_failure: Optional[str] = None
self._init_auth_notice_emitted = False
# Cron and flush contexts disable the plugin entirely.
self._cron_skipped = False
@property
def name(self) -> str:
return "honcho"
def is_available(self) -> bool:
"""Check if Honcho is configured. No network calls."""
try:
from plugins.memory.honcho.client import HonchoClientConfig
return _cfg_usable(HonchoClientConfig.from_global_config())
except Exception:
return False
def save_config(self, values, hermes_home):
"""Write config to $HERMES_HOME/honcho.json (Honcho SDK native format)."""
from pathlib import Path
from utils import atomic_json_write
from plugins.memory.honcho.client import _read_config
config_path = Path(hermes_home) / "honcho.json"
try:
existing = _read_config(config_path)
except Exception:
existing = {}
atomic_json_write(config_path, {**existing, **values}, mode=0o600)
def get_config_schema(self):
return [
{"key": "api_key", "description": "Honcho API key", "secret": True, "env_var": "HONCHO_API_KEY", "url": "https://app.honcho.dev"},
{"key": "baseUrl", "description": "Honcho base URL (for self-hosted)"},
]
def post_setup(self, hermes_home: str, config: dict) -> None:
"""Run the full Honcho setup wizard after provider selection."""
import types
from plugins.memory.honcho.cli import cmd_setup
cmd_setup(types.SimpleNamespace())
# ----- Session lifecycle -----
def initialize(self, session_id: str, **kwargs) -> None:
"""Configure recall settings and start (or defer) Honcho session creation."""
try:
agent_context, platform = kwargs.get("agent_context", ""), kwargs.get("platform", "cli")
if agent_context in {"cron", "flush"} or platform == "cron":
logger.debug("Honcho skipped: cron/flush context (agent_context=%s, platform=%s)",
agent_context, platform)
self._cron_skipped = True
return
from plugins.memory.honcho.client import HonchoClientConfig, get_honcho_client # noqa: F401 — ImportError probe
from plugins.memory.honcho.session import HonchoSessionManager # noqa: F401
cfg = HonchoClientConfig.from_global_config()
if not _cfg_usable(cfg):
logger.debug("Honcho not configured — plugin inactive")
return
self._config = cfg
self._recall_mode = cfg.recall_mode
logger.debug("Honcho recall_mode: %s", self._recall_mode)
for name in ("injection_frequency", "context_cadence", "dialectic_cadence",
"dialectic_depth_levels", "reasoning_heuristic"):
setattr(self, f"_{name}", getattr(cfg, name))
self._query_rewrite_enabled = cfg.query_rewrite
self._FIRST_TURN_BASE_TIMEOUT = cfg.first_turn_base_wait
self._FIRST_TURN_DIALECTIC_CAP = cfg.first_turn_dialectic_wait
self._dialectic_depth = max(1, min(cfg.dialectic_depth, 3))
if cfg.reasoning_level_cap in self._LEVEL_ORDER:
self._reasoning_level_cap = cfg.reasoning_level_cap
# aiPeer comes from honcho.json only; SOUL.md is persona content, not identity config.
self._lazy_init_kwargs = dict(kwargs)
self._lazy_init_session_id = session_id
self._session_key = self._resolve_session_key(cfg, session_id, **kwargs)
# Session creation can block on Honcho/DB outages, so context/hybrid startup
# fails open in a background thread. Tools-only mode has an explicit contract:
# init_on_session_start=False stays lazy until the first tool call, True is eager.
if self._recall_mode == "tools":
if cfg.init_on_session_start:
self._ensure_session()
else:
logger.debug("Honcho tools-only mode — deferring session init until first tool call")
return
self._start_session_init_background(wait_timeout=0.1)
except ImportError:
logger.debug("honcho-ai package not installed — plugin inactive")
except Exception as e:
logger.warning("Honcho init failed: %s", e)
self._manager = None
def _resolve_session_key(self, cfg, session_id: str, **kwargs) -> str:
"""Resolve the Honcho session key without touching the network."""
return (
cfg.resolve_session_name(
session_title=kwargs.get("session_title"),
session_id=session_id,
gateway_session_key=kwargs.get("gateway_session_key"),
)
or session_id
or "hermes-default"
)
def _can_start_init(self) -> bool:
return not (self._cron_skipped or self._session_initialized) and bool(self._config) and self._lazy_init_kwargs is not None
def _run_session_init(self, label: str) -> bool:
"""Run _do_session_init with the deferred kwargs; on failure discard the manager
and (for auth failures) keep the detail for the one-time notice."""
from plugins.memory.honcho.session import HonchoAuthError
init_kwargs = self._lazy_init_kwargs
if init_kwargs is None: # another init path already consumed the deferred kwargs
return self._manager is not None
try:
self._do_session_init(self._config, self._lazy_init_session_id or "hermes-default", **dict(init_kwargs))
self._lazy_init_kwargs = None
self._lazy_init_session_id = None
if self._init_auth_failure is not None:
self._init_auth_failure = None
self._init_auth_notice_emitted = False
return True
except Exception as e:
self._manager = None
self._session_initialized = False
detail: object = e
if isinstance(e, HonchoAuthError):
# Keep the auth detail so the one-time notice survives the manager discard.
self._init_auth_failure = str(e)
detail = "authentication rejected"
logger.warning("Honcho %s session init failed: %s", label, detail)
return False
def _start_session_init_background(self, *, wait_timeout: float = 0.0) -> None:
"""Start session initialization in a daemon thread so a slow/down Honcho can't
block agent construction or first prompt assembly. ``wait_timeout`` lets fast
(mock) initializations finish before returning."""
if not self._can_start_init():
return
with self._init_lock:
if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()):
return
self._init_thread = spawn_context_thread(
lambda: self._run_session_init("background"), name="honcho-session-init",
)
self._init_thread.start()
if wait_timeout > 0:
self._init_thread.join(timeout=wait_timeout)
def _ensure_session(self) -> bool:
"""Lazily initialize the Honcho session (tools-only mode). True when the manager is ready."""
if self._manager and self._session_initialized:
return True
if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()):
return False
return self._run_session_init("lazy") and self._manager is not None
def _do_session_init(self, cfg, session_id: str, **kwargs) -> None:
"""Shared session initialization for both eager and lazy paths."""
from plugins.memory.honcho.client import get_honcho_client
from plugins.memory.honcho.session import HonchoSessionManager
self._manager = HonchoSessionManager(
honcho=get_honcho_client(cfg),
config=cfg,
context_tokens=cfg.context_tokens,
runtime_user_peer_name=kwargs.get("user_id") or None,
runtime_user_peer_name_alt=kwargs.get("user_id_alt") or None,
)
self._session_key = self._resolve_session_key(cfg, session_id, **kwargs)
logger.debug("Honcho session key resolved: %s", self._session_key)
# The provider is not "ready" until this method returns: background startup sets
# _manager before get_or_create/migration/prewarm finish, and lifecycle hooks must
# not treat that partially initialized state as usable.
session = self._manager.get_or_create(self._session_key)
# Per-session strategy creates a fresh Honcho session every run, so a per-run
# MEMORY.md/USER.md/SOUL.md upload would flood the backend with duplicates.
if cfg.session_strategy == "per-session":
logger.debug(
"Honcho memory file migration skipped: per-session strategy creates a fresh session per run (%s)",
self._session_key,
)
elif not session.messages:
try:
from hermes_constants import get_hermes_home
self._manager.migrate_memory_files(self._session_key, str(get_hermes_home() / "memories"))
logger.debug("Honcho memory file migration attempted for new session: %s", self._session_key)
except Exception as e:
logger.debug("Honcho memory file migration skipped: %s", e)
# Generic dialectic prewarm is incompatible with latest-message query rewriting,
# which needs the first substantive user message.
if self._recall_mode in {"context", "hybrid"}:
if self._query_rewriter is None or not self._query_rewrite_enabled:
self._spawn_dialectic(
_PREWARM_QUERY, thread_name="honcho-prewarm-dialectic", fired_at=0,
log_label="dialectic prewarm", use_query_rewrite=False,
)
logger.debug("Honcho dialectic prewarm started for session: %s", self._session_key)
else:
logger.debug("Honcho generic dialectic prewarm skipped: awaiting first user message")
self._session_initialized = True
def _session_ready(self) -> bool:
"""Whether the manager/session key can be used safely.
Background init sets ``_manager`` before get-or-create completes, so
``_session_initialized`` is the real guard; tests/legacy construction may inject
a ready manager without the flag — allow that only with no init thread in flight.
"""
if not self._manager or not self._session_key:
return False
if self._session_initialized:
return True
return not (self._init_thread and self._init_thread.is_alive())
def _writes_enabled(self) -> bool:
"""``saveMessages`` is the operator's hard write gate for every Honcho mutation path."""
return not self._cron_skipped and getattr(self._config, "save_messages", True)
def _ready_or_kick_init(self) -> bool:
"""True when writes may proceed; otherwise (outside tools mode) start background init."""
if self._session_ready():
return True
if self._recall_mode != "tools":
self._start_session_init_background()
return False
# ----- Prompt / prefetch -----
def _format_first_turn_context(self, ctx: dict) -> str:
"""Format the prefetch context dict into a readable system prompt block."""
parts = [f"## {header}\n{ctx.get(key, '')}" for key, header in _CONTEXT_SECTIONS if ctx.get(key, "")]
return "\n\n".join(parts)
def system_prompt_block(self) -> str:
"""Static mode header + tool instructions (prompt-cache friendly).
Live context (representation, card) is injected via prefetch()."""
if self._cron_skipped or not (self._config or (self._manager and self._session_key)):
return ""
return _PROMPT_HEADERS.get(self._recall_mode, _PROMPT_HEADERS["hybrid"])
def _first_turn_wait(self, base: float) -> float:
"""Turn-1 wait budget: a short request timeout may tighten, but never expand, it."""
request_timeout = getattr(self._config, "timeout", None)
if request_timeout is not None:
base = min(base, max(0.0, request_timeout))
return max(0.0, base)
def _fetch_base_context_layer(self, query: str, first_turn_base_deadline: float | None) -> str:
"""Layer 1: representation + card. The first fetch gets the remaining turn-1 budget;
later turns consume the refresh queued by the previous turn."""
with self._base_context_lock:
first_base_fetch = self._base_context_cache is None
if first_base_fetch:
self._base_context_cache = ""
self._last_context_turn = self._turn_count
base_context = self._base_context_cache
if not self._manager:
return base_context
def _adopt(ctx: dict) -> str:
"""Cache a fresh context dict's formatted block; keep the old text if it formats empty."""
formatted = self._format_first_turn_context(ctx)
if formatted:
with self._base_context_lock:
self._base_context_cache = formatted
return formatted or base_context
if not first_base_fetch:
fresh_ctx = self._manager.pop_context_result(self._session_key)
return _adopt(fresh_ctx) if fresh_ctx else base_context
ctx_holder: dict[str, dict] = {}
def _fetch_base() -> None:
ctx = self._manager.get_prefetch_context(self._session_key, query or None) or {}
ctx_holder["ctx"] = ctx
if ctx:
self._manager.set_context_result(self._session_key, ctx)
bt = self._spawn_write(_fetch_base, "honcho-base-first", "Honcho first-turn base context failed: %s")
base_wait = max(0.0, first_turn_base_deadline - time.monotonic()) if first_turn_base_deadline is not None else 0.0
bt.join(timeout=base_wait)
ctx = ctx_holder.get("ctx")
if ctx:
self._manager.pop_context_result(self._session_key)
return _adopt(ctx)
if bt.is_alive():
logger.debug("Honcho first-turn base context still running after %.1fs — will surface on next turn", base_wait)
return base_context
def _first_turn_dialectic_wait(self, query: str) -> None:
"""Turn 1 only: reuse an in-flight prewarm or start one dialectic, then wait briefly.
Unfinished work stays async and surfaces on a later turn."""
with self._prefetch_lock:
prewarm_landed = bool(self._prefetch_result)
if prewarm_landed and self._last_dialectic_turn == -999:
self._last_dialectic_turn = self._turn_count
if self._last_dialectic_turn != -999 or not query:
return
dia_wait = self._first_turn_wait(self._FIRST_TURN_DIALECTIC_CAP)
if not self._thread_is_live():
self._spawn_dialectic(
query, thread_name="honcho-prefetch-first", fired_at=self._turn_count,
log_label="first-turn dialectic",
)
live = self._prefetch_thread
if live is not None:
live.join(timeout=dia_wait)
if self._prefetch_thread and self._prefetch_thread.is_alive():
logger.debug("Honcho first-turn dialectic still running after %.1fs — will surface on next turn", dia_wait)
def prefetch(self, query: str, *, session_id: str = "") -> str:
"""Base context (representation + card, refreshed on context_cadence) plus the
dialectic supplement (refreshed on dialectic_cadence), within the context budget.
Empty in tools-only mode."""
if self._cron_skipped or self._recall_mode == "tools":
return ""
first_turn_base_deadline = (
time.monotonic() + self._first_turn_wait(self._FIRST_TURN_BASE_TIMEOUT) if self._turn_count <= 1 else None
)
if not self._session_ready():
# Only turn 1 may wait for session init; later turns fail open.
self._start_session_init_background()
if first_turn_base_deadline is not None and self._init_thread is not None:
self._init_thread.join(timeout=max(0.0, first_turn_base_deadline - time.monotonic()))
if not self._session_ready():
# A failed auth init still owes the user the one-time notice.
return self._pop_auth_notice()
# Trivial turns start no work, but may consume a ready pending result.
if self._is_trivial_prompt(query):
ready = self._consume_pending_dialectic()
return self._truncate_to_budget(ready) if ready else ""
# One-time notice, relayed by the model, that auth is dead and memory is paused.
parts = [self._pop_auth_notice()]
# First-turn mode suppresses only the base layer; dialectic is independent.
if not (self._injection_frequency == "first-turn" and self._turn_count > 1):
parts.append(self._fetch_base_context_layer(query, first_turn_base_deadline))
self._first_turn_dialectic_wait(query)
# Consume only results that are already ready; later turns never wait.
parts.append(self._consume_pending_dialectic())
parts = [p for p in parts if p and p.strip()]
return self._truncate_to_budget("\n\n".join(parts)) if parts else ""
def _pop_auth_notice(self) -> str:
"""One-time model-facing notice that Honcho auth expired and memory is paused."""
# getattr (not a direct call): test fakes install minimal managers without pop_auth_notice.
pop = getattr(self._manager, "pop_auth_notice", None)
msg = pop() if callable(pop) else None
if not isinstance(msg, str) or not msg:
# Init failures discard the manager; the provider kept the detail.
if self._init_auth_failure is None or self._init_auth_notice_emitted:
return ""
self._init_auth_notice_emitted = True
msg = self._init_auth_failure
return (
"[Honcho memory status] Authentication with the Honcho memory backend has expired and automatic "
f"token refresh failed, so memory sync and recall are paused. Reason: {msg}\n"
"Tell the user (once) that Honcho memory is paused and that running 'hermes honcho setup' "
"to re-authenticate will restore it."
)
def _truncate_to_budget(self, text: str) -> str:
"""Truncate text to the context_tokens budget (≈4 chars/token) at a word boundary."""
if not self._config or not self._config.context_tokens:
return text
budget_chars = self._config.context_tokens * 4
if len(text) <= budget_chars:
return text
truncated = text[:budget_chars]
last_space = truncated.rfind(" ")
if last_space > budget_chars * 0.8:
truncated = truncated[:last_space]
return truncated + " …"
def queue_prefetch(self, query: str, *, session_id: str = "") -> None:
"""Fire background prefetch threads for the upcoming turn.
Context and dialectic refreshes have independent cadence controls."""
if self._cron_skipped or self._recall_mode == "tools":
return
if not self._session_ready() or not query:
self._start_session_init_background()
return
# Trivial prompts don't warrant either a context refresh or a dialectic call.
if self._is_trivial_prompt(query):
return
# First-turn-only base context never needs a later refresh.
context_due = self._context_cadence <= 1 or (self._turn_count - self._last_context_turn) >= self._context_cadence
if self._injection_frequency != "first-turn" and context_due:
self._last_context_turn = self._turn_count
try:
self._manager.prefetch_context(self._session_key, query)
except Exception as e:
logger.debug("Honcho context prefetch failed: %s", e)
# Dialectic layer: a hung call older than timeout × multiplier counts as dead.
if self._thread_is_live():
logger.debug("Honcho dialectic prefetch skipped: prior thread still running")
return
# Cadence gate, widened by the empty-streak backoff so a persistently silent
# backend doesn't retry every turn forever.
effective = self._effective_cadence()
if (self._turn_count - self._last_dialectic_turn) < effective:
logger.debug(
"Honcho dialectic prefetch skipped: effective cadence %d "
"(base %d, empty streak %d), turns since last: %d",
effective, self._dialectic_cadence, self._dialectic_empty_streak,
self._turn_count - self._last_dialectic_turn,
)
return
self._spawn_dialectic(
query, thread_name="honcho-prefetch", fired_at=self._turn_count, log_label="prefetch",
)
# Shared with the core prefetch gate so the two classifiers can never drift apart.
_is_trivial_prompt = staticmethod(is_trivial_prompt)
def on_turn_start(self, turn_number: int, message: str, **kwargs) -> None:
"""Track turn count for cadence and injection_frequency logic."""
self._turn_count = turn_number
# ----- Writes -----
@staticmethod
def _chunk_message(content: str, limit: int) -> list[str]:
"""Split content to fit the Honcho message limit, cutting at paragraph, then
sentence, then word boundaries; continuation chunks get a "[continued] " prefix
so Honcho's representation engine can reconstruct the full message."""
if len(content) <= limit:
return [content]
prefix = "[continued] "
chunks = []
remaining = content
first = True
while remaining:
effective = limit if first else limit - len(prefix)
if len(remaining) <= effective:
chunks.append(remaining if first else prefix + remaining)
break
segment = remaining[:effective]
# Paragraph, then sentence (keeping ". "), then word boundary; else a hard cut.
for sep in ("\n\n", ". ", " "):
cut = segment.rfind(sep)
if cut >= 0 and sep == ". ":
cut += 2
if cut >= effective * 0.3:
break
else:
cut = effective
chunk = remaining[:cut].rstrip()
remaining = remaining[cut:].lstrip()
chunks.append(chunk if first else prefix + chunk)
first = False
return chunks
def sync_turn(self, user_content: str, assistant_content: str, *, session_id: str = "") -> None:
"""Record the conversation turn in Honcho (non-blocking), chunking messages that
exceed the Honcho API limit. Honors saveMessages: false."""
if not self._writes_enabled():
return
if _is_internal_gateway_turn(user_content):
logger.debug("Honcho sync skipped machine-generated gateway turn")
return
if not self._ready_or_kick_init():
return
msg_limit = self._config.message_max_chars if self._config else 25000
clean_user_content = sanitize_context(user_content or "").strip()
clean_assistant_content = sanitize_context(assistant_content or "").strip()
# Skip only when the whole turn is empty: an interrupted or tool-only turn can have
# an empty assistant side, and the user's message must still be persisted.
if not clean_user_content and not clean_assistant_content:
return
def _sync():
session = self._manager.get_or_create(self._session_key)
for role, content in (("user", clean_user_content), ("assistant", clean_assistant_content)):
for chunk in self._chunk_message(content, msg_limit) if content else ():
session.add_message(role, chunk)
# save() (not _flush_session) so writeFrequency batching is honored.
self._manager.save(session)
if self._sync_thread and self._sync_thread.is_alive():
self._sync_thread.join(timeout=5.0)
self._sync_thread = self._spawn_write(_sync, "honcho-sync", "Honcho sync_turn failed: %s")
@staticmethod
def _spawn_write(fn: Callable[[], None], name: str, fail_msg: str) -> threading.Thread:
"""Run a Honcho write off-thread; failures are debug-logged, never raised into the turn."""
def _run():
try:
fn()
except Exception as e:
logger.debug(fail_msg, e)
thread = spawn_context_thread(_run, name=name)
thread.start()
return thread
def on_memory_write(
self, action: str, target: str, content: str, metadata: Optional[Dict[str, Any]] = None,
) -> None:
"""Mirror built-in user-profile writes as Honcho conclusions (``metadata`` accepted
for interface compatibility, not yet threaded into the conclusion payload)."""
if action != "add" or target != "user" or not content:
return
if not self._writes_enabled() or not self._ready_or_kick_init():
return
self._memwrite_thread = self._spawn_write(
lambda: self._manager.create_conclusion(self._session_key, content),
"honcho-memwrite", "Honcho memory mirror failed: %s",
)
def on_session_end(self, messages: List[Dict[str, Any]]) -> None:
"""Flush all pending messages to Honcho on session end."""
if not self._writes_enabled() or not self._manager:
return
if not self._session_initialized and self._init_thread and self._init_thread.is_alive():
return
if self._sync_thread and self._sync_thread.is_alive():
self._sync_thread.join(timeout=10.0)
try:
self._manager.flush_all()
except Exception as e:
logger.debug("Honcho session-end flush failed: %s", e)
# ----- Tools -----
def get_tool_schemas(self) -> List[Dict[str, Any]]:
"""Tool schemas by recall_mode; context-only mode exposes no Honcho tools."""
if self._cron_skipped or self._recall_mode == "context":
return []
return list(ALL_TOOL_SCHEMAS)
def _empty_profile_hint(self, peer: str) -> Dict[str, Any]:
"""Diagnostic hint for an empty honcho_profile card, so the model can explain WHY
instead of surfacing a cryptic "no facts" to the user. Likely causes, in order:
observation disabled for the peer; card not accumulated yet (fresh peer / few
dialectic cycles); self-hosted Honcho < 3.x without peer-card support."""
cfg = self._config
reasons: List[str] = []
kind = "user" if peer == "user" else "ai"
if cfg is not None and not (
getattr(cfg, f"{kind}_observe_me", True) or getattr(cfg, f"{kind}_observe_others", True)
):
reasons.append(f"observation is disabled for peer '{peer}' (user_observe_me/ai_observe_me in config)")
cadence, turn = self._dialectic_cadence, self._turn_count
if turn < max(2, cadence):
reasons.append(
f"this session has only {turn} turn(s); peer cards accumulate as the dialectic "
f"layer reasons over conversation history (cadence every {cadence} turn(s))"
)
if not reasons:
reasons.append(
"peer card has no facts yet — Honcho's dialectic layer builds this over time from "
"observed turns; self-hosted Honcho < 3.x does not support peer cards at all"
)
return {
"result": "No profile facts available yet.",
"hint": (
"This is not an error. " + "; ".join(reasons)
+ ". Try honcho_reasoning for a synthesized answer, or honcho_search to query raw conversation excerpts."
),
}
def _tool_profile(self, args: dict) -> str:
peer = args.get("peer", "user")
card_update = args.get("card")
if card_update:
result = self._manager.set_peer_card(self._session_key, card_update, peer=peer)
if result is None:
return tool_error("Failed to update peer card.")
return json.dumps({"result": f"Peer card updated ({len(result)} facts).", "card": result})
card = self._manager.get_peer_card(self._session_key, peer=peer)
if not card:
return json.dumps(self._empty_profile_hint(peer))
return json.dumps({"result": card})
def _tool_search(self, args: dict) -> str:
query = (args.get("query") or "").strip()
if not query:
return tool_error("Missing required parameter: query")
max_tokens = min(int(args.get("max_tokens", 800)), 2000)
result = self._manager.search_context(
self._session_key, query, max_tokens=max_tokens, peer=args.get("peer", "user"),
)
return json.dumps({"result": result or "No relevant context found."})
def _tool_reasoning(self, args: dict) -> str:
from plugins.memory.honcho.session import HonchoAuthError
query = (args.get("query") or "").strip()
if not query:
return tool_error("Missing required parameter: query")
try:
result = self._manager.dialectic_query(
self._session_key, query,
reasoning_level=args.get("reasoning_level"),
peer=args.get("peer", "user"),
# Explicit reasoning bypasses the automatic-injection cap, and surfaces
# timeouts/server errors as errors rather than an indistinguishable "no result".
apply_injection_cap=False,
raise_errors=True,
)
except HonchoAuthError:
raise # rendered by handle_tool_call's auth-specific handler
except Exception as e:
logger.warning("honcho_reasoning failed: %s", e)
return tool_error(
f"Honcho reasoning query failed ({e}). This is a backend error, not an empty result — "
"the peer may still have relevant context. Slow dialectic calls at higher reasoning levels "
"can exceed the configured timeout; consider a lower reasoning_level or raising the "
"'timeout' value in honcho.json."
)
# Auto-injection respects the cadence gap after an explicit call.
self._last_dialectic_turn = self._turn_count
return json.dumps({"result": result or "No result from Honcho."})
def _tool_context(self, args: dict) -> str:
ctx = self._manager.get_session_context(self._session_key, peer=args.get("peer", "user"))
if not ctx:
return json.dumps({"result": "No context available yet."})
parts = [
f"## {header}\n{ctx[key]}"
for key, header in (("summary", "Summary"), ("representation", "Representation"), ("card", "Card"))
if ctx.get(key)
]
if ctx.get("recent_messages"):
msg_str = "\n".join(f" [{m['role']}] {m['content'][:200]}" for m in ctx["recent_messages"][-5:])
parts.append(f"## Recent messages\n{msg_str}")
return json.dumps({"result": "\n\n".join(parts) or "No context available."})
def _tool_conclude(self, args: dict) -> str:
delete_id = (args.get("delete_id") or "").strip()
conclusion = args.get("conclusion", "").strip()
list_mode = bool(args.get("list"))
peer = args.get("peer", "user")
if sum([bool(delete_id), bool(conclusion), list_mode]) != 1:
return tool_error("Exactly one of conclusion, delete_id, or list must be provided.")
query = (args.get("query") or "").strip()
if query and not list_mode:
return tool_error("query is only valid when list is true.")
if list_mode:
conclusions = self._manager.list_conclusions(self._session_key, query=query or None, peer=peer)
return json.dumps({"conclusions": conclusions})
if delete_id:
if self._manager.delete_conclusion(self._session_key, delete_id, peer=peer):
return json.dumps({"result": f"Conclusion {delete_id} deleted."})
return tool_error(f"Failed to delete conclusion {delete_id}.")
if self._manager.create_conclusion(self._session_key, conclusion, peer=peer):
return json.dumps({"result": f"Conclusion saved for {peer}: {conclusion}"})
return tool_error("Failed to save conclusion.")
_TOOL_HANDLERS = {
"honcho_profile": _tool_profile,
"honcho_search": _tool_search,
"honcho_reasoning": _tool_reasoning,
"honcho_context": _tool_context,
"honcho_conclude": _tool_conclude,
}
def handle_tool_call(self, tool_name: str, args: dict, **kwargs) -> str:
"""Dispatch a Honcho tool call, lazily initializing the session in tools-only mode."""
from plugins.memory.honcho.session import HonchoAuthError
if self._cron_skipped:
return tool_error("Honcho is not active (cron context).")
if not self._session_initialized:
if self._init_thread and self._init_thread.is_alive():
return tool_error("Honcho session is still initializing; try again shortly.")
if not self._ensure_session():
if self._init_auth_failure:
return tool_error(f"Honcho memory authentication failed: {self._init_auth_failure}")
return tool_error("Honcho session could not be initialized.")
if not self._manager or not self._session_key:
return tool_error("Honcho is not active for this session.")
handler = self._TOOL_HANDLERS.get(tool_name)
if handler is None:
return tool_error(f"Unknown tool: {tool_name}")
try:
return handler(self, args)
except HonchoAuthError as e:
# Never report an auth failure as an empty result; the model would read it as "no memory".
logger.error("Honcho tool %s failed: authentication rejected", tool_name)
return tool_error(f"Honcho memory authentication failed: {e}")
except Exception as e:
logger.error("Honcho tool %s failed: %s", tool_name, e)
return tool_error(f"Honcho {tool_name} failed: {e}")
def shutdown(self) -> None:
for t in (self._prefetch_thread, self._sync_thread, self._memwrite_thread):
if t and t.is_alive():
t.join(timeout=5.0)
manager = self._manager
if not manager or (self._init_thread and self._init_thread.is_alive() and not self._session_initialized):
return
try:
# saveMessages: false skips persistence, but the async-writer thread must still
# be joined so daemon threads aren't left blocked in httpx I/O at interpreter exit.
if getattr(self._config, "save_messages", True):
manager.shutdown() # flush_all() + join the writer
else:
manager.stop_async_writer()
except Exception:
pass
def register(ctx) -> None:
"""Register Honcho as a memory provider plugin."""
from plugins.memory.query_rewrite import rewrite_memory_query
ctx.register_memory_provider(
HonchoMemoryProvider(query_rewriter=rewrite_memory_query)
)