- delete _cached_prompt_reflects_builtin_memory + _builtin_memory_prompt_snapshot (zero callers since the commit site moved to byte-equality; the only test reference asserts their ABSENCE from the commit window) - delete CompressionExecutorSaturatedError (never raised or caught anywhere) - _compress_context_via_codex_app_server: three near-identical skip branches collapse into one skip_reason + single log line (same message text)
5092 lines
209 KiB
Python
5092 lines
209 KiB
Python
"""Context compression: feasibility probe, warning replay, compress, image fix.
|
||
|
||
Thread-safety contract for extension points
|
||
--------------------------------------------
|
||
With ``compression.context_timeout_seconds > 0`` (default) the whole pass,
|
||
context engines and memory providers included, runs on a pooled daemon thread.
|
||
* Calls may arrive on any pooled thread; never rely on thread-affinity/locals.
|
||
* The message list is a private deep snapshot; in-place mutation is allowed
|
||
but invisible to the live conversation unless the pass commits.
|
||
* State is published ONLY on an admitted :class:`CompressionCommitFence`
|
||
commit; work of an engine still running after a host timeout is discarded.
|
||
* One pass per session at a time (durable lock), but different sessions may
|
||
run concurrently, so shared engine/provider instances must be thread-safe.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import concurrent.futures
|
||
import copy
|
||
import inspect
|
||
import json
|
||
import logging
|
||
import math
|
||
import os
|
||
import tempfile
|
||
import time
|
||
import uuid
|
||
import threading
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Any, Callable, Dict, List, Literal, Optional, Tuple
|
||
|
||
from agent.auxiliary_client import AuxiliaryExplicitCancellation
|
||
from agent.context_engine import (
|
||
automatic_compaction_status_message,
|
||
sanitize_memory_context,
|
||
)
|
||
from agent.memory_provider import PRE_COMPRESS_CHECKPOINT_API_VERSION
|
||
from agent.model_metadata import (
|
||
estimate_messages_tokens_rough,
|
||
estimate_request_tokens_rough,
|
||
)
|
||
from agent.session_activity import ActivityProvenance, normalize_activity_provenance
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# Terminal outcomes from host/hygiene timeout or cooldown writers. Detached
|
||
# heartbeat workers must not clobber these (timeout unobservable). Seeing one
|
||
# latches the heartbeat silent so a later UNKNOWN rewrite can't re-arm a zombie.
|
||
_TERMINAL_COMPRESSION_PROVENANCES = frozenset(
|
||
{
|
||
ActivityProvenance.AGENT_COMPRESSION_TIMEOUT,
|
||
ActivityProvenance.AGENT_COMPRESSION_COOLDOWN,
|
||
}
|
||
)
|
||
|
||
# Split failures are usually transient lease/DB conditions, so use the FIRST
|
||
# timeout-ladder rung (60s), not the 600s summary-provider cooldown.
|
||
_SPLIT_FAILURE_COOLDOWN_SECONDS = 60
|
||
|
||
# Marker tui_gateway/server.py::_status_update matches to tag kind="compacting"
|
||
# for drivers' "Summarizing…" UI. Keep the phrase intact when rewording. Idle/
|
||
# preflight/retry lines lack it; is_compaction_progress_status covers those.
|
||
COMPACTION_STATUS_MARKER = "Compacting context"
|
||
COMPACTION_STATUS = (
|
||
f"🗜️ {COMPACTION_STATUS_MARKER} — summarizing earlier conversation so I can continue..."
|
||
)
|
||
|
||
COMPACTION_DONE_STATUS = "✓ Context compaction complete — continuing turn..."
|
||
|
||
|
||
def _strip_marker_for_comparison(msgs: Any) -> Any:
|
||
"""Copy ``msgs`` with the ``_db_persisted`` marker removed for no-op comparison.
|
||
|
||
Live dicts carry the marker while ``compress()`` output is swept, so a raw
|
||
``==`` would misclassify an identical no-op copy as progress. Non-list inputs
|
||
and non-dict entries pass through unchanged.
|
||
"""
|
||
from agent.context_compressor import _DB_PERSISTED_MARKER
|
||
|
||
if not isinstance(msgs, list):
|
||
return msgs
|
||
return [
|
||
{k: v for k, v in m.items() if k != _DB_PERSISTED_MARKER}
|
||
if isinstance(m, dict)
|
||
else m
|
||
for m in msgs
|
||
]
|
||
|
||
|
||
def _emit_compaction_done(agent: Any) -> None:
|
||
"""Emit the structured terminal edge for a started compaction."""
|
||
status_callback = getattr(agent, "status_callback", None)
|
||
if not status_callback:
|
||
return
|
||
try:
|
||
status_callback("compacted", COMPACTION_DONE_STATUS)
|
||
except Exception:
|
||
logger.debug("status_callback error in compaction completion", exc_info=True)
|
||
|
||
|
||
# Every ROUTINE compression status line lives here: suppressed on chat platforms
|
||
# by _TELEGRAM_NOISY_STATUS_RE (gateway/run.py); update that regex + telegram
|
||
# noise test when rewording. Failure notices and /compress feedback: NOT here.
|
||
PRE_API_COMPRESSION_STATUS_TEMPLATE = (
|
||
"📦 Pre-API compression: ~{tokens:,} tokens "
|
||
"near the context/output limit. Compacting before the next model call."
|
||
)
|
||
PREFLIGHT_COMPRESSION_STATUS_TEMPLATE = (
|
||
"📦 Preflight compression: ~{tokens:,} tokens "
|
||
">= {threshold:,} threshold. This may take a moment."
|
||
)
|
||
IDLE_COMPACTION_STATUS_TEMPLATE = (
|
||
"💤 Resumed after {idle_seconds}s idle — compacting "
|
||
"~{tokens:,} tokens before continuing."
|
||
)
|
||
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE = (
|
||
"🗜️ Context too large (~{tokens:,} tokens) — compressing ({attempt}/{cap})..."
|
||
)
|
||
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE = (
|
||
"🗜️ Compressed {before} → {after} messages, retrying..."
|
||
)
|
||
COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE = (
|
||
"🗜️ Compressed ~{before:,} → ~{after:,} tokens, retrying..."
|
||
)
|
||
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE = (
|
||
"🗜️ Context reduced to {new_ctx:,} tokens (was {old_ctx:,}), retrying..."
|
||
)
|
||
|
||
# FAILURE-class notice: compression blocked, so the session grows until the
|
||
# provider limit kills it. Must stay visible on gateways: never add it to
|
||
# ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE.
|
||
CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = (
|
||
"⚠ Context is over the compression threshold "
|
||
"(~{tokens:,} tokens >= {threshold:,}) "
|
||
"but compression is currently blocked ({reason}). "
|
||
"The model may stop responding. Run /new to start a fresh "
|
||
"session or /compress to retry immediately."
|
||
)
|
||
|
||
# Formatted from the same constants the emission sites use, so noise-filter
|
||
# tests exercise the ACTUAL wording.
|
||
ROUTINE_COMPRESSION_STATUS_SAMPLES = (
|
||
COMPACTION_STATUS,
|
||
COMPACTION_DONE_STATUS,
|
||
PRE_API_COMPRESSION_STATUS_TEMPLATE.format(tokens=123456),
|
||
PREFLIGHT_COMPRESSION_STATUS_TEMPLATE.format(tokens=120000, threshold=100000),
|
||
IDLE_COMPACTION_STATUS_TEMPLATE.format(idle_seconds=3600, tokens=120000),
|
||
COMPRESSION_RETRY_TOO_LARGE_STATUS_TEMPLATE.format(tokens=250000, attempt=1, cap=3),
|
||
COMPRESSION_RETRY_MESSAGES_STATUS_TEMPLATE.format(before=30, after=12),
|
||
COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=250000, after=120000),
|
||
COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE.format(
|
||
new_ctx=120000, old_ctx=250000
|
||
),
|
||
)
|
||
|
||
|
||
def is_compaction_progress_status(text: str | None) -> bool:
|
||
"""True for in-progress auto-compaction lifecycle lines (not the done edge).
|
||
|
||
The gateway re-tags matches as ``kind="compacting"`` for the whole pause;
|
||
matching only the marker left idle/preflight/retry lines looking hung.
|
||
``COMPACTION_DONE_STATUS`` is emitted as ``kind="compacted"`` and must not
|
||
match here.
|
||
"""
|
||
if not isinstance(text, str):
|
||
return False
|
||
body = text.strip()
|
||
if not body:
|
||
return False
|
||
if COMPACTION_STATUS_MARKER in body:
|
||
return True
|
||
if body == COMPACTION_DONE_STATUS:
|
||
return False
|
||
lowered = body.lower()
|
||
if "compaction complete" in lowered:
|
||
return False
|
||
# Failure-class overflow warning mentions compression but is a blocked
|
||
# notice, not progress — keep it lifecycle so chat gateways stay loud.
|
||
if "compression is currently blocked" in lowered:
|
||
return False
|
||
return (
|
||
"compact" in lowered
|
||
or "compress" in lowered
|
||
or "context reduced to" in lowered
|
||
)
|
||
|
||
|
||
def _refresh_agent_tool_definitions(agent) -> bool:
|
||
"""Rebuild agent.tools at the compaction commit boundary.
|
||
|
||
Forever-sessions never restart, so this is the only moment config changes
|
||
reach the frozen dynamic tool schemas; the prompt cache is already invalid.
|
||
Delegates to refresh_agent_mcp_tools in content_aware mode (swaps on schema
|
||
CONTENT change). Returns True when tools were added. Never raises.
|
||
"""
|
||
from tools.mcp_tool import refresh_agent_mcp_tools
|
||
|
||
added = refresh_agent_mcp_tools(agent, content_aware=True)
|
||
if added:
|
||
logger.info(
|
||
"Compaction tool refresh added tools: %s", sorted(added),
|
||
)
|
||
return bool(added)
|
||
|
||
|
||
_COMPRESSOR_ATTEMPT_STATE_FIELDS = (
|
||
"_previous_summary",
|
||
"_summary_has_user_turn",
|
||
"compression_count",
|
||
"_last_compression_savings_pct",
|
||
"_ineffective_compression_count",
|
||
"_anti_thrash_recovery_deadline",
|
||
"_fallback_compression_streak",
|
||
"_verify_compaction_cleared_threshold",
|
||
"_last_compression_made_progress",
|
||
"_summary_failure_cooldown_until",
|
||
"_cooldown_persist_failed",
|
||
"_last_summary_error",
|
||
"_consecutive_timeout_failures",
|
||
"_last_summary_dropped_count",
|
||
"_last_summary_fallback_used",
|
||
"_last_compress_aborted",
|
||
"_last_summary_auth_failure",
|
||
"_last_summary_network_failure",
|
||
"_last_summary_empty_content_failure",
|
||
"_last_summary_truncated_failure",
|
||
"_last_aux_model_failure_error",
|
||
"_last_aux_model_failure_model",
|
||
"_summary_model_fallen_back",
|
||
"summary_model",
|
||
"_last_compression_telemetry",
|
||
"_active_compression_telemetry",
|
||
"_compression_telemetry_seed",
|
||
"_proactive_prune_rearm_tokens",
|
||
)
|
||
|
||
_COMPRESSOR_COOLDOWN_STATE_FIELDS = (
|
||
"_summary_failure_cooldown_until",
|
||
"_last_summary_error",
|
||
"_cooldown_persist_failed",
|
||
)
|
||
|
||
|
||
def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]:
|
||
"""Copy only the mutable bookkeeping owned by one compression attempt.
|
||
|
||
The allow-list avoids copying clients, DB handles, locks and plugin resources;
|
||
missing fields are ignored so legacy/third-party compressors keep working.
|
||
"""
|
||
try:
|
||
values = vars(compressor)
|
||
except TypeError:
|
||
return {}
|
||
selected = {
|
||
name: values[name]
|
||
for name in _COMPRESSOR_ATTEMPT_STATE_FIELDS
|
||
if name in values
|
||
}
|
||
# Copy the collection as one object so aliases between fields (notably
|
||
# _active_compression_telemetry and _last_compression_telemetry) survive.
|
||
return copy.deepcopy(selected)
|
||
|
||
|
||
# Attempt ownership: stall-fallback detaches a timed-out worker and reuses the
|
||
# compressor, so its late unwind could restore a stale snapshot or clear the
|
||
# fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs.
|
||
|
||
_COMPRESSOR_ATTEMPT_LOCK = threading.Lock()
|
||
|
||
|
||
def _claim_compressor_attempt(compressor: Any) -> int:
|
||
"""Claim the compressor for a new attempt; return its monotonic generation id.
|
||
|
||
Restores or cancelled-check mutations stamped with an OLDER generation no-op,
|
||
so a detached late attempt cannot clobber its successor's state.
|
||
"""
|
||
with _COMPRESSOR_ATTEMPT_LOCK:
|
||
generation = int(getattr(compressor, "_compression_attempt_generation", 0) or 0) + 1
|
||
try:
|
||
compressor._compression_attempt_generation = generation
|
||
except Exception:
|
||
# Slotted/frozen compressor: gen 0 disables the guard. Per-compressor, so gen-0
|
||
# and gen>0 attempts can never coexist on one instance.
|
||
return 0
|
||
return generation
|
||
|
||
|
||
def _compressor_attempt_is_current(compressor: Any, generation: int) -> bool:
|
||
"""True when *generation* still owns the compressor (or guard disabled)."""
|
||
if not generation:
|
||
return True
|
||
with _COMPRESSOR_ATTEMPT_LOCK:
|
||
return (
|
||
int(getattr(compressor, "_compression_attempt_generation", 0) or 0)
|
||
== generation
|
||
)
|
||
|
||
|
||
def _install_compression_cancelled_check(
|
||
compressor: Any, check: Any, generation: int
|
||
) -> None:
|
||
"""Install the F4 cancellation consult, stamped with its owner attempt."""
|
||
with _COMPRESSOR_ATTEMPT_LOCK:
|
||
try:
|
||
compressor._compression_cancelled_check = check
|
||
compressor._compression_cancelled_check_owner = generation
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def _clear_compression_cancelled_check_if_owner(
|
||
compressor: Any, generation: int
|
||
) -> bool:
|
||
"""Clear the cancellation consult only when *generation* installed it.
|
||
|
||
Prevents a detached late primary from tearing down a newer fallback's
|
||
callback. Returns True when cleared.
|
||
"""
|
||
with _COMPRESSOR_ATTEMPT_LOCK:
|
||
owner = getattr(compressor, "_compression_cancelled_check_owner", None)
|
||
if owner is not None and generation and owner != generation:
|
||
return False
|
||
try:
|
||
compressor._compression_cancelled_check = None
|
||
compressor._compression_cancelled_check_owner = None
|
||
except Exception:
|
||
pass
|
||
return True
|
||
|
||
|
||
def _restore_compressor_attempt_state(
|
||
compressor: Any,
|
||
snapshot: dict[str, Any],
|
||
*,
|
||
durable_cooldown_authoritative: Optional[bool] = None,
|
||
durable_cooldown_state: Optional[dict[str, Any]] = None,
|
||
attempt_generation: Optional[int] = None,
|
||
) -> None:
|
||
"""Restore the per-attempt snapshot after a pre-commit hard cancel.
|
||
|
||
A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out
|
||
primary's late unwind cannot roll back state owned by the fallback attempt.
|
||
"""
|
||
if attempt_generation is not None and not _compressor_attempt_is_current(
|
||
compressor, attempt_generation
|
||
):
|
||
logger.warning(
|
||
"Skipping stale compressor attempt-state restore: attempt "
|
||
"generation %s no longer owns the compressor (current: %s). A "
|
||
"newer (stall-fallback) attempt's state is preserved.",
|
||
attempt_generation,
|
||
getattr(compressor, "_compression_attempt_generation", None),
|
||
)
|
||
return
|
||
# Success clears the durable cooldown pre-commit; recreate/clear that row BEFORE
|
||
# restoring in-memory values or the next refresh overwrites the rollback. Never
|
||
# turn unknown durable state / unpersisted local cooldowns into DB writes.
|
||
if (
|
||
"_summary_failure_cooldown_until" in snapshot
|
||
and durable_cooldown_authoritative is not False
|
||
and (
|
||
durable_cooldown_authoritative is True
|
||
or not bool(snapshot.get("_cooldown_persist_failed", False))
|
||
)
|
||
):
|
||
session_db = vars(compressor).get("_session_db")
|
||
session_id = vars(compressor).get("_session_id")
|
||
if session_db is not None and session_id:
|
||
if durable_cooldown_authoritative is True:
|
||
restorer = getattr(
|
||
type(session_db),
|
||
"restore_compression_failure_cooldown_row",
|
||
None,
|
||
)
|
||
if not callable(restorer) or durable_cooldown_state is None:
|
||
raise RuntimeError(
|
||
"exact compression cooldown rollback API is unavailable"
|
||
)
|
||
# This API restores raw columns (including expired and null
|
||
# combinations), verifies the read-back, and propagates failure.
|
||
restorer(
|
||
session_db,
|
||
session_id,
|
||
copy.deepcopy(durable_cooldown_state),
|
||
)
|
||
else:
|
||
try:
|
||
deadline = float(
|
||
snapshot["_summary_failure_cooldown_until"] or 0.0
|
||
)
|
||
remaining = max(0.0, deadline - time.monotonic())
|
||
durable_deadline = time.time() + remaining
|
||
durable_error = snapshot.get("_last_summary_error")
|
||
if remaining > 0:
|
||
recorder = getattr(
|
||
type(session_db),
|
||
"record_compression_failure_cooldown",
|
||
None,
|
||
)
|
||
if callable(recorder):
|
||
recorder(
|
||
session_db,
|
||
session_id,
|
||
durable_deadline,
|
||
durable_error,
|
||
)
|
||
else:
|
||
clearer = getattr(
|
||
type(session_db),
|
||
"clear_compression_failure_cooldown",
|
||
None,
|
||
)
|
||
if callable(clearer):
|
||
clearer(session_db, session_id)
|
||
except Exception:
|
||
# Legacy/third-party compatibility path: its existing APIs
|
||
# do not provide a verifiable transaction contract.
|
||
logger.debug(
|
||
"compression cooldown persistence rollback failed",
|
||
exc_info=True,
|
||
)
|
||
restored = copy.deepcopy(snapshot)
|
||
# Re-validate under the claim lock: the slow durable rollback above leaves a
|
||
# window where a fallback may have claimed; stale writes must not interleave.
|
||
# The rollback itself is safe: landing after a fallback needs a prior claim.
|
||
with _COMPRESSOR_ATTEMPT_LOCK:
|
||
if attempt_generation is not None and attempt_generation and (
|
||
int(getattr(compressor, "_compression_attempt_generation", 0) or 0)
|
||
!= attempt_generation
|
||
):
|
||
logger.warning(
|
||
"Skipping stale compressor attempt-state restore at write "
|
||
"time: attempt generation %s lost the compressor mid-restore.",
|
||
attempt_generation,
|
||
)
|
||
return
|
||
for name, value in restored.items():
|
||
setattr(compressor, name, value)
|
||
|
||
|
||
def _capture_authoritative_cooldown_under_lease(
|
||
compressor: Any,
|
||
attempt_snapshot: dict[str, Any],
|
||
) -> tuple[Optional[bool], Optional[dict[str, Any]]]:
|
||
"""Refresh and snapshot built-in durable cooldown state under the lease.
|
||
|
||
Third-party compressors are not invoked: plugin code must not run under the
|
||
lease. Returns ``False`` on durable read failure (rollback must not mistake
|
||
unknown state for an empty row) and ``None`` when the legacy API is absent.
|
||
"""
|
||
try:
|
||
from agent.context_compressor import ContextCompressor
|
||
|
||
if not isinstance(compressor, ContextCompressor):
|
||
return None, None
|
||
values = vars(compressor)
|
||
session_db = values.get("_session_db")
|
||
session_id = values.get("_session_id")
|
||
raw_reader = (
|
||
getattr(
|
||
type(session_db), "get_compression_failure_cooldown_row", None
|
||
)
|
||
if session_db is not None
|
||
else None
|
||
)
|
||
if session_db is None or not session_id:
|
||
# Unbound compressors have no durable row to mutate or restore.
|
||
return None, None
|
||
if not callable(raw_reader):
|
||
return False, None
|
||
# Read the raw persisted row: the active getter filters expired rows and is not
|
||
# a lossless rollback snapshot.
|
||
durable_state = raw_reader(session_db, session_id)
|
||
if not isinstance(durable_state, dict):
|
||
raise TypeError("raw compression cooldown snapshot must be a mapping")
|
||
ContextCompressor.get_active_compression_failure_cooldown(
|
||
compressor,
|
||
refresh=True,
|
||
)
|
||
except Exception as exc:
|
||
logger.debug("authoritative compression cooldown capture failed: %s", exc)
|
||
return False, None
|
||
authoritative = getattr(
|
||
compressor, "_last_cooldown_refresh_was_authoritative", None
|
||
)
|
||
if authoritative is not True:
|
||
return authoritative, None
|
||
|
||
values = vars(compressor)
|
||
for name in _COMPRESSOR_COOLDOWN_STATE_FIELDS:
|
||
if name in values:
|
||
attempt_snapshot[name] = copy.deepcopy(values[name])
|
||
return True, copy.deepcopy(durable_state)
|
||
|
||
|
||
class CompressionCommitFence:
|
||
"""Fence timeout cancellation against post-summary session mutation.
|
||
|
||
The sync worker thread cannot be killed; the fence makes the commit boundary
|
||
deterministic: cancellation wins before mutation starts, or waits for an
|
||
already-started commit to finish completely.
|
||
"""
|
||
|
||
def __init__(self, total_ceiling_seconds: float | None = None) -> None:
|
||
self._lock = threading.Lock()
|
||
self._cancelled = False
|
||
self._commit_started = False
|
||
# begin_commit holds self._lock until finish_commit, so this Event is readable
|
||
# WITHOUT the lock: hosts can see a hung commit and fire the overrun warning.
|
||
self._commit_phase = threading.Event()
|
||
# Set on ANY host unwind without the fence lock, so a host that cannot block
|
||
# behind an in-flight commit still blocks FUTURE commits. bool store is atomic.
|
||
self._admission_revoked = False
|
||
# Worker publishes a holder-scoped release once it owns the durable lock; a
|
||
# timed-out host frees the lease without racing a NEW holder (no ABA).
|
||
self._lock_release_guard = threading.Lock()
|
||
self._cancelled_lock_release: Optional[Callable[[], None]] = None
|
||
self._cancelled_lock_release_requested = False
|
||
# Touched per streamed summary token; waiters distinguish SLOW-but-alive from
|
||
# HUNG so slow models are not killed by a fixed wall-clock deadline.
|
||
self._last_progress = time.monotonic()
|
||
self._progress_observed = False
|
||
self._deadline: float | None = None
|
||
self._retain_cancelled_lock_until_worker_done = False
|
||
# Set once the commit path captured the active-row watermark: later rows survive
|
||
# as concurrent tail, so hosts may KEEP a detached worker's commit admission.
|
||
self._commit_watermark_fenced = False
|
||
if total_ceiling_seconds is not None:
|
||
self.set_total_ceiling_seconds(total_ceiling_seconds)
|
||
|
||
def set_total_ceiling_seconds(self, seconds: float) -> None:
|
||
"""Arm the wall-clock deadline shared by the host and worker."""
|
||
seconds = float(seconds)
|
||
if seconds <= 0:
|
||
raise ValueError("total compression ceiling must be positive")
|
||
self._deadline = time.monotonic() + seconds
|
||
|
||
def touch_progress(self) -> None:
|
||
"""Record forward progress (e.g. a streamed summary token arriving).
|
||
|
||
Called from the worker thread, read by waiters via ``seconds_since_progress``;
|
||
a bare float store is atomic in CPython so no lock is needed.
|
||
"""
|
||
self._last_progress = time.monotonic()
|
||
self._progress_observed = True
|
||
|
||
@property
|
||
def progress_observed(self) -> bool:
|
||
"""Whether semantic provider progress was reported for this attempt."""
|
||
return self._progress_observed
|
||
|
||
@property
|
||
def deadline_exceeded(self) -> bool:
|
||
deadline = self._deadline
|
||
return deadline is not None and time.monotonic() >= deadline
|
||
|
||
@property
|
||
def deadline_monotonic(self) -> float | None:
|
||
"""The armed deadline as an absolute ``time.monotonic()`` instant.
|
||
|
||
Published so the worker's stream consumer can stop exactly when the host
|
||
stops waiting (see ``auxiliary_client.aux_stream_deadline``).
|
||
"""
|
||
return self._deadline
|
||
|
||
def seconds_since_progress(self) -> float:
|
||
"""Seconds since the worker last reported forward progress."""
|
||
return max(0.0, time.monotonic() - self._last_progress)
|
||
|
||
def cancel_before_commit(self, cancel_event: Any = None) -> bool:
|
||
"""Cancel a pending commit, or wait for an active commit to finish.
|
||
|
||
Returns ``True`` when cancellation won before the commit boundary; ``False``
|
||
after blocking until an already-started commit fully completed.
|
||
"""
|
||
with self._lock:
|
||
if self._commit_started:
|
||
if cancel_event is not None:
|
||
cancel_event.set()
|
||
return False
|
||
self._cancelled = True
|
||
if cancel_event is not None:
|
||
cancel_event.set()
|
||
return True
|
||
|
||
def try_cancel_before_commit(self) -> Optional[bool]:
|
||
"""Non-blocking form of :meth:`cancel_before_commit`.
|
||
|
||
Returns ``None`` while an active commit owns the fence so an async caller can
|
||
yield instead of blocking its event loop.
|
||
"""
|
||
if not self._lock.acquire(blocking=False):
|
||
return None
|
||
try:
|
||
if self._commit_started:
|
||
return False
|
||
self._cancelled = True
|
||
return True
|
||
finally:
|
||
self._lock.release()
|
||
|
||
def begin_commit(self, cancel_event: Any = None) -> bool:
|
||
"""Atomically admit commit unless a hard cancellation already won."""
|
||
self._lock.acquire()
|
||
if (
|
||
self.is_cancelled
|
||
or self._admission_revoked
|
||
or (cancel_event is not None and bool(cancel_event.is_set()))
|
||
):
|
||
self._cancelled = True
|
||
self._lock.release()
|
||
if self._admission_revoked:
|
||
# A revoke that lost the fence-lock race deferred its lease release; commit was
|
||
# refused, so releasing now is safe (idempotent with holder-qualified cleanup).
|
||
self.release_cancelled_compression_lock()
|
||
return False
|
||
self._commit_started = True
|
||
# Set while the fence lock is held so observers can never see
|
||
# commit_in_flight=True for a commit that lost to cancellation.
|
||
self._commit_phase.set()
|
||
return True
|
||
|
||
def finish_commit(self) -> None:
|
||
"""Leave a commit boundary entered by :meth:`begin_commit`."""
|
||
self._commit_phase.clear()
|
||
self._lock.release()
|
||
if self._admission_revoked:
|
||
# A revoke during THIS commit deferred its lease release (no freeing under an
|
||
# active SessionDB mutation); commit is done, release now. Holder-qualified.
|
||
self.release_cancelled_compression_lock()
|
||
|
||
@property
|
||
def commit_in_flight(self) -> bool:
|
||
"""Lock-free read: an admitted commit has begun and not yet finished.
|
||
|
||
Safe while the worker holds the fence lock for a hung commit; lets hosts reach
|
||
the overrun-warning loop instead of spinning on ``try_cancel_before_commit``.
|
||
"""
|
||
return self._commit_phase.is_set()
|
||
|
||
@property
|
||
def is_cancelled(self) -> bool:
|
||
"""True after cancellation won before the commit boundary."""
|
||
return self._cancelled or self._admission_revoked or self.deadline_exceeded
|
||
|
||
def retain_compression_lock_until_worker_done(self) -> None:
|
||
"""Prevent a timed-out live worker from overlapping a retry."""
|
||
self._retain_cancelled_lock_until_worker_done = True
|
||
|
||
def mark_commit_watermark_fenced(self) -> None:
|
||
"""Record that this attempt's commit is bounded by a start watermark.
|
||
|
||
A watermark-fenced commit archives only rows at or below the watermark and
|
||
clones later rows as live tail, so a detached worker may keep its admission.
|
||
"""
|
||
self._commit_watermark_fenced = True
|
||
|
||
@property
|
||
def commit_watermark_fenced(self) -> bool:
|
||
"""Lock-free read: the worker's commit is watermark-bounded."""
|
||
return self._commit_watermark_fenced
|
||
|
||
def allow_cancelled_lock_release(self) -> None:
|
||
"""Undo :meth:`retain_compression_lock_until_worker_done`.
|
||
|
||
Called after a bounded join confirmed the timed-out worker exited, so the
|
||
durable lease may be released and a fallback attempt can proceed.
|
||
"""
|
||
self._retain_cancelled_lock_until_worker_done = False
|
||
|
||
def revoke_commit_admission(self) -> None:
|
||
"""Revoke FUTURE commit admission without blocking on the fence lock.
|
||
|
||
An in-flight commit is never abandoned, but ``begin_commit`` re-checks the
|
||
flag under the lock so no new commit is admitted. The lease release must not
|
||
run mid-commit (a second compressor could interleave): released now if the
|
||
lock is free, else deferred to ``finish_commit``/refusal (holder-qualified).
|
||
"""
|
||
self._admission_revoked = True
|
||
if self._lock.acquire(blocking=False):
|
||
try:
|
||
self.release_cancelled_compression_lock()
|
||
finally:
|
||
self._lock.release()
|
||
# else: deferred — finish_commit()/begin_commit() re-check _admission_revoked
|
||
# and release once no commit can be mid-mutation.
|
||
|
||
# ── Holder-qualified durable-lease cancellation: release is DELETE WHERE
|
||
# holder = ?, so a stale release can never free a NEW holder's lease (no ABA).
|
||
|
||
def begin_lock_setup(self) -> bool:
|
||
"""Fence durable-lock acquisition and release-hook publication.
|
||
|
||
The caller holds the fence until the holder-qualified release hook is
|
||
published (or no lock was taken), so a timeout cannot win in that gap.
|
||
"""
|
||
self._lock.acquire()
|
||
if self.is_cancelled or self._admission_revoked:
|
||
self._lock.release()
|
||
return False
|
||
return True
|
||
|
||
def finish_lock_setup(self) -> None:
|
||
"""Leave a lock setup boundary entered by :meth:`begin_lock_setup`."""
|
||
self._lock.release()
|
||
|
||
def register_cancelled_lock_release(
|
||
self, release: Callable[[], None]
|
||
) -> bool:
|
||
"""Publish the timed-out worker's holder-qualified lock release.
|
||
|
||
Returns whether cleanup was already requested; in that race the release runs
|
||
synchronously before returning.
|
||
"""
|
||
with self._lock_release_guard:
|
||
self._cancelled_lock_release = release
|
||
requested = self._cancelled_lock_release_requested
|
||
if requested:
|
||
release()
|
||
return requested
|
||
|
||
def clear_cancelled_lock_release(self, release: Callable[[], None]) -> None:
|
||
"""Forget ``release`` after the worker's normal cleanup finishes."""
|
||
with self._lock_release_guard:
|
||
if self._cancelled_lock_release is release:
|
||
self._cancelled_lock_release = None
|
||
|
||
def release_cancelled_compression_lock(self) -> None:
|
||
"""Release the cancelled worker's lock without finalizing its clients.
|
||
|
||
Only valid after cancellation won. A request racing ahead of hook publication
|
||
is retained and fulfilled when the worker publishes the hook.
|
||
"""
|
||
if self._retain_cancelled_lock_until_worker_done:
|
||
return
|
||
with self._lock_release_guard:
|
||
self._cancelled_lock_release_requested = True
|
||
release = self._cancelled_lock_release
|
||
if release is not None:
|
||
release()
|
||
|
||
|
||
# Defaults for the in-agent (non-hygiene) progress-aware compress_context wrap.
|
||
# Mirror hermes_cli.config.DEFAULT_CONFIG["compression"] keys of the same name.
|
||
DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0
|
||
DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0
|
||
|
||
# Unlike explicit_interrupt: a /stop after the stall window arms the durable
|
||
# backoff so the next automatic turn does not re-enter the stalled strategy.
|
||
STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted"
|
||
|
||
# Daemon pool so a fence-cancelled hung worker cannot block interpreter exit
|
||
# via the atexit join. Never shut down per call (workers may still be winding).
|
||
_compress_timeout_executor = None
|
||
_compress_timeout_executor_lock = threading.Lock()
|
||
|
||
# Overrun waits proceed in bounded slices so each window logs (escalating)
|
||
# instead of one silent future.result(). Clamped to ceiling for tiny test values
|
||
_COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0
|
||
|
||
# A worker exiting within the grace proves no provider call is in flight, so the
|
||
# lease can be released even on the total-ceiling path. One that doesn't exit is
|
||
# orphaned behind the poison fence and keeps its lease so no attempt overlaps.
|
||
_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0
|
||
|
||
|
||
def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool:
|
||
"""Best-effort bounded join of a fence-cancelled compression worker.
|
||
|
||
Returns True when the future settled within ``grace_seconds`` (thread provably
|
||
exited); False for a still-running worker, which the caller must treat as an
|
||
orphan behind the poison fence.
|
||
"""
|
||
try:
|
||
grace = max(float(grace_seconds), 0.0)
|
||
except (TypeError, ValueError):
|
||
grace = 0.0
|
||
try:
|
||
future.result(timeout=grace)
|
||
return True
|
||
except concurrent.futures.TimeoutError:
|
||
return False
|
||
except concurrent.futures.CancelledError:
|
||
# Never started; nothing can be in flight.
|
||
return True
|
||
except Exception:
|
||
# Exception swallowed: the host already chose the fallback result and the fence
|
||
# keeps the failed attempt from touching session state.
|
||
logger.debug(
|
||
"cancelled compression worker exited with an exception",
|
||
exc_info=True,
|
||
)
|
||
return True
|
||
|
||
|
||
# Executor queue is unbounded: a queued job would wait out its timeout unstarted
|
||
# and run stale later. Cap admission at worker count; fail fast (warn, continue
|
||
# uncompressed). Slots free via done-callback; a never-returning worker loses 1.
|
||
_COMPRESS_EXECUTOR_MAX_WORKERS = 4
|
||
_compress_admission_lock = threading.Lock()
|
||
_compress_admitted_count = 0
|
||
|
||
|
||
def _try_admit_compression_job() -> bool:
|
||
"""Reserve one bounded compression-pool admission slot (F6)."""
|
||
global _compress_admitted_count
|
||
with _compress_admission_lock:
|
||
if _compress_admitted_count >= _COMPRESS_EXECUTOR_MAX_WORKERS:
|
||
return False
|
||
_compress_admitted_count += 1
|
||
return True
|
||
|
||
|
||
def _release_compression_admission(_future=None) -> None:
|
||
"""Free an admission slot (future done-callback or failed submit)."""
|
||
global _compress_admitted_count
|
||
with _compress_admission_lock:
|
||
if _compress_admitted_count > 0:
|
||
_compress_admitted_count -= 1
|
||
|
||
|
||
def _get_compress_timeout_executor():
|
||
"""Return the process-wide compress-timeout DaemonThreadPoolExecutor."""
|
||
global _compress_timeout_executor
|
||
executor = _compress_timeout_executor
|
||
if executor is not None:
|
||
return executor
|
||
from tools.daemon_pool import DaemonThreadPoolExecutor
|
||
|
||
with _compress_timeout_executor_lock:
|
||
if _compress_timeout_executor is None:
|
||
# Small pool: compress is rare/heavy; sized for live compress + cancelled
|
||
# workers still winding down, not asyncio's min(32, cpu+4).
|
||
_compress_timeout_executor = DaemonThreadPoolExecutor(
|
||
max_workers=_COMPRESS_EXECUTOR_MAX_WORKERS,
|
||
thread_name_prefix="compress-ctx-timeout",
|
||
)
|
||
return _compress_timeout_executor
|
||
|
||
|
||
def resolve_context_compression_timeouts(
|
||
compression_cfg: Optional[dict] = None,
|
||
) -> Tuple[float, float]:
|
||
"""Return ``(idle_timeout_seconds, total_ceiling_seconds)``.
|
||
|
||
``idle_timeout_seconds <= 0`` disables the progress-aware wrapper. The ceiling
|
||
is clamped to at least one idle window when the idle budget is positive.
|
||
"""
|
||
idle = DEFAULT_CONTEXT_TIMEOUT_SECONDS
|
||
ceiling = DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS
|
||
cfg = compression_cfg
|
||
if cfg is None:
|
||
try:
|
||
from hermes_cli.config import load_config
|
||
|
||
raw = load_config()
|
||
maybe = raw.get("compression", {}) if isinstance(raw, dict) else {}
|
||
cfg = maybe if isinstance(maybe, dict) else {}
|
||
except Exception:
|
||
cfg = {}
|
||
if isinstance(cfg, dict):
|
||
raw_idle = cfg.get("context_timeout_seconds")
|
||
if raw_idle is not None:
|
||
try:
|
||
parsed = float(raw_idle)
|
||
# Explicit 0/negative disables; positive values win.
|
||
idle = parsed
|
||
except (TypeError, ValueError):
|
||
pass
|
||
raw_ceiling = cfg.get("context_total_ceiling_seconds")
|
||
if raw_ceiling is not None:
|
||
try:
|
||
parsed = float(raw_ceiling)
|
||
if parsed > 0:
|
||
ceiling = parsed
|
||
except (TypeError, ValueError):
|
||
pass
|
||
if idle > 0:
|
||
ceiling = max(ceiling, idle)
|
||
return idle, ceiling
|
||
|
||
|
||
def compression_attempt_stalled(
|
||
*,
|
||
commit_fence: Optional[CompressionCommitFence],
|
||
started_at: float,
|
||
idle_timeout_seconds: Optional[float] = None,
|
||
) -> bool:
|
||
"""Return whether a pre-commit cancel landed after the stall window.
|
||
|
||
An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity
|
||
budget counts as a stall so the next automatic turn does not blindly retry.
|
||
"""
|
||
idle = idle_timeout_seconds
|
||
if idle is None:
|
||
idle, _ceiling = resolve_context_compression_timeouts()
|
||
try:
|
||
idle = float(idle)
|
||
except (TypeError, ValueError):
|
||
return False
|
||
if idle <= 0:
|
||
return False
|
||
if commit_fence is not None:
|
||
try:
|
||
return float(commit_fence.seconds_since_progress()) >= idle
|
||
except Exception:
|
||
return False
|
||
try:
|
||
return (time.monotonic() - float(started_at)) >= idle
|
||
except (TypeError, ValueError):
|
||
return False
|
||
|
||
|
||
def _stall_source_fingerprint(
|
||
agent: Any,
|
||
messages: Any,
|
||
approx_tokens: Optional[int],
|
||
) -> str:
|
||
"""Identity of the stalled source context + summary strategy."""
|
||
compressor = getattr(agent, "context_compressor", None)
|
||
model = (
|
||
getattr(compressor, "summary_model", None)
|
||
or getattr(agent, "model", None)
|
||
or ""
|
||
)
|
||
n_messages = len(messages) if isinstance(messages, list) else 0
|
||
try:
|
||
tokens = int(approx_tokens or 0)
|
||
except (TypeError, ValueError):
|
||
tokens = 0
|
||
return f"msgs={n_messages}:tokens={tokens}:model={model}"
|
||
|
||
|
||
def _record_stall_interrupted_backoff(
|
||
agent: Any,
|
||
*,
|
||
commit_fence: Optional[CompressionCommitFence],
|
||
started_at: float,
|
||
messages: Any,
|
||
approx_tokens: Optional[int],
|
||
) -> bool:
|
||
"""Persist a stall-interrupted cooldown after snapshot restore.
|
||
|
||
Must run *after* ``_restore_compressor_attempt_state`` so rollback cannot wipe
|
||
the new row. Returns True when the backoff was recorded.
|
||
"""
|
||
if not compression_attempt_stalled(
|
||
commit_fence=commit_fence, started_at=started_at
|
||
):
|
||
return False
|
||
compressor = getattr(agent, "context_compressor", None)
|
||
record = getattr(compressor, "record_timeout_failure", None)
|
||
if not callable(record):
|
||
return False
|
||
error = (
|
||
f"{STALL_INTERRUPTED_FAILURE_CLASS}:"
|
||
f"{_stall_source_fingerprint(agent, messages, approx_tokens)}"
|
||
)
|
||
try:
|
||
record(error, failure_kind="stall_interrupted")
|
||
except Exception:
|
||
logger.debug(
|
||
"stall-interrupted compression cooldown persist failed",
|
||
exc_info=True,
|
||
)
|
||
return False
|
||
logger.info(
|
||
"Recorded stall-interrupted compression backoff (session=%s, %s)",
|
||
getattr(agent, "session_id", None) or "none",
|
||
error,
|
||
)
|
||
return True
|
||
|
||
|
||
def resolve_compression_fallback_route() -> Optional[dict]:
|
||
"""Return the first usable ``auxiliary.compression.fallback_chain`` entry.
|
||
|
||
The aux client applies the chain only from its exception handler, so a silent
|
||
stall never reaches it; this pins the route onto one bounded retry instead.
|
||
Only the first complete entry: if it errors, the aux client's own exception
|
||
path walks the rest. ``None`` when none is usable (skip compression).
|
||
"""
|
||
try:
|
||
from agent.auxiliary_client import (
|
||
_fallback_entry_api_key,
|
||
_get_auxiliary_task_config,
|
||
)
|
||
|
||
chain = _get_auxiliary_task_config("compression").get("fallback_chain")
|
||
except Exception:
|
||
logger.debug("compression fallback_chain lookup failed", exc_info=True)
|
||
return None
|
||
if not isinstance(chain, list):
|
||
return None
|
||
|
||
for index, entry in enumerate(chain):
|
||
if not isinstance(entry, dict):
|
||
continue
|
||
provider = str(entry.get("provider") or "").strip()
|
||
model = str(entry.get("model") or "").strip()
|
||
# Both are required to name a route. _resolve_fallback_entry applies
|
||
# the same rule when the aux client walks this chain itself.
|
||
if not provider or not model:
|
||
continue
|
||
try:
|
||
api_key = _fallback_entry_api_key(entry)
|
||
except Exception:
|
||
logger.debug(
|
||
"compression fallback_chain[%d] api key resolution failed",
|
||
index,
|
||
exc_info=True,
|
||
)
|
||
api_key = None
|
||
from agent.auxiliary_client import _coerce_positive_timeout
|
||
|
||
timeout = _coerce_positive_timeout(entry.get("timeout"))
|
||
return {
|
||
"label": f"fallback_chain[{index}]({provider})",
|
||
"provider": provider,
|
||
"model": model,
|
||
"base_url": str(entry.get("base_url") or "").strip() or None,
|
||
"api_key": api_key or None,
|
||
"api_mode": str(
|
||
entry.get("api_mode") or entry.get("transport") or ""
|
||
).strip() or None,
|
||
"timeout": timeout,
|
||
}
|
||
return None
|
||
|
||
|
||
def _retry_compression_on_fallback_chain(
|
||
*,
|
||
worker: Callable[[CompressionCommitFence], Tuple[list, str]],
|
||
messages: list,
|
||
system_prompt_fallback: Any,
|
||
idle_timeout_seconds: float,
|
||
total_ceiling_seconds: float,
|
||
on_commit_overrun: Optional[Callable[[float, float], None]] = None,
|
||
on_timeout_cause: Optional[Callable[[bool, bool], None]] = None,
|
||
telemetry_agent: Any = None,
|
||
new_fence: Optional[Callable[[], CompressionCommitFence]] = None,
|
||
) -> Optional[Tuple[list, str]]:
|
||
"""Re-run an aborted compression once with the summary route pinned.
|
||
|
||
Returns ``(messages, system_prompt)`` on real compression, else ``None`` and
|
||
the caller degrades as before. The entry's ``timeout`` sets the idle window.
|
||
Re-runs the whole worker, so pre-compression callbacks must be idempotent.
|
||
"""
|
||
# An explicit stop is not a stalled route. The retry worker would abort on
|
||
# the same event anyway, but starting one at all makes /stop look ignored.
|
||
hard_cancel = getattr(telemetry_agent, "_hard_interrupt_requested", None)
|
||
if callable(getattr(hard_cancel, "is_set", None)) and hard_cancel.is_set():
|
||
return None
|
||
|
||
route = resolve_compression_fallback_route()
|
||
if route is None:
|
||
return None
|
||
|
||
# The aborted fence refuses all commits; mint a fresh one via the host factory
|
||
# so a /stop during the retry serializes against THIS attempt's commit boundary.
|
||
retry_fence = None
|
||
if new_fence is not None:
|
||
try:
|
||
retry_fence = new_fence()
|
||
except Exception:
|
||
logger.warning(
|
||
"compression stall-fallback fence factory failed; the retry "
|
||
"will run on an unpublished fence (a /stop mid-retry cannot "
|
||
"serialize against its commit boundary)",
|
||
exc_info=True,
|
||
)
|
||
if not isinstance(retry_fence, CompressionCommitFence):
|
||
logger.warning(
|
||
"compression stall-fallback retry running on an unpublished fence; "
|
||
"hard-interrupt admission will read the aborted attempt's fence "
|
||
"rather than the retry's commit boundary",
|
||
)
|
||
retry_fence = CompressionCommitFence()
|
||
idle = float(route.get("timeout") or idle_timeout_seconds)
|
||
ceiling = max(float(total_ceiling_seconds), idle)
|
||
logger.warning(
|
||
"Context compression stalled on the configured summary route — "
|
||
"retrying once on %s (%s) before continuing without compression",
|
||
route["label"],
|
||
route["model"],
|
||
)
|
||
try:
|
||
from agent.context_compressor import pin_summary_route
|
||
|
||
with pin_summary_route(route):
|
||
result_msgs, result_prompt = run_compress_context_with_progress_timeout(
|
||
worker=worker,
|
||
messages=messages,
|
||
system_prompt_fallback=system_prompt_fallback,
|
||
idle_timeout_seconds=idle,
|
||
total_ceiling_seconds=ceiling,
|
||
on_commit_overrun=on_commit_overrun,
|
||
on_timeout_cause=on_timeout_cause,
|
||
fence=retry_fence,
|
||
telemetry_agent=telemetry_agent,
|
||
stall_fallback=False,
|
||
)
|
||
except Exception:
|
||
# The primary already failed; a failing fallback must degrade, never
|
||
# turn "continue without compression" into a raised turn.
|
||
logger.warning(
|
||
"Context compression fallback attempt on %s failed",
|
||
route["label"],
|
||
exc_info=True,
|
||
)
|
||
return None
|
||
if result_msgs is messages:
|
||
# Aborted or no-op: the worker hands back the caller's own list.
|
||
logger.warning(
|
||
"Context compression fallback attempt on %s produced no "
|
||
"compression; continuing without compression",
|
||
route["label"],
|
||
)
|
||
return None
|
||
logger.info(
|
||
"Context compression recovered on %s after the primary summary route "
|
||
"stalled",
|
||
route["label"],
|
||
)
|
||
return result_msgs, result_prompt
|
||
|
||
|
||
def run_compress_context_with_progress_timeout(
|
||
*,
|
||
worker: Callable[[CompressionCommitFence], Tuple[list, str]],
|
||
messages: list,
|
||
system_prompt_fallback: Any,
|
||
idle_timeout_seconds: float,
|
||
total_ceiling_seconds: float,
|
||
on_timeout: Optional[Callable[[float, float, float], None]] = None,
|
||
on_timeout_cause: Optional[Callable[[bool, bool], None]] = None,
|
||
on_commit_overrun: Optional[Callable[[float, float], None]] = None,
|
||
fence: Optional[CompressionCommitFence] = None,
|
||
telemetry_agent: Any = None,
|
||
stall_fallback: bool = True,
|
||
new_fence: Optional[Callable[[], CompressionCommitFence]] = None,
|
||
) -> Tuple[list, str]:
|
||
"""Run ``worker(fence)`` under a sync progress-aware (idle + ceiling) timeout.
|
||
|
||
Budgets bound the PRE-commit phase only: an admitted commit always completes
|
||
(overrun logged, surfaced once via ``on_commit_overrun``). A pre-commit cancel
|
||
returns ``(messages, system_prompt_fallback)`` (lazy callable), detaching the
|
||
worker; a stall first retries the chain once on ``new_fence``, then on_timeout
|
||
"""
|
||
if idle_timeout_seconds <= 0:
|
||
raise ValueError(
|
||
"run_compress_context_with_progress_timeout requires "
|
||
"idle_timeout_seconds > 0; call compress_context directly to disable"
|
||
)
|
||
|
||
def _resolve_fallback_prompt() -> str:
|
||
if callable(system_prompt_fallback):
|
||
return system_prompt_fallback()
|
||
return system_prompt_fallback
|
||
|
||
ceiling = max(float(total_ceiling_seconds), float(idle_timeout_seconds))
|
||
idle = float(idle_timeout_seconds)
|
||
fence = fence if fence is not None else CompressionCommitFence()
|
||
fence.set_total_ceiling_seconds(ceiling)
|
||
# Sync mirror of gateway hygiene's run_in_executor + wait_for loop: offload,
|
||
# poll idle budget + ceiling, fence-cancel on timeout so no late commit lands.
|
||
from tools.thread_context import propagate_context_to_thread
|
||
|
||
executor = _get_compress_timeout_executor()
|
||
# Refuse rather than queue when the pool is full: a queued job would wait out
|
||
# its budget unstarted and run stale later. Skip compression this cycle.
|
||
if not _try_admit_compression_job():
|
||
logger.warning(
|
||
"Context compression pool saturated (%d workers busy) — "
|
||
"refusing new compression this cycle and continuing without "
|
||
"compression. Wedged workers are fence-cancelled and free their "
|
||
"slot when they return; if this persists, check the summary "
|
||
"provider health.",
|
||
_COMPRESS_EXECUTOR_MAX_WORKERS,
|
||
)
|
||
# Saturation refusals must hit the same telemetry stream as other failures, or
|
||
# a wedged pool looks like compression simply stopped being attempted.
|
||
if telemetry_agent is not None:
|
||
_emit_compression_attempt_telemetry(
|
||
telemetry_agent,
|
||
started_at=time.monotonic(),
|
||
commit_status="aborted",
|
||
split_status="aborted",
|
||
failure_class="pool_saturated",
|
||
)
|
||
return messages, _resolve_fallback_prompt()
|
||
|
||
def _fence_gated_worker(worker_fence: CompressionCommitFence):
|
||
# An admitted job may start after the host stopped waiting; check the fence
|
||
# BEFORE summary work so a stale job never burns an LLM call.
|
||
if worker_fence.deadline_exceeded:
|
||
raise concurrent.futures.TimeoutError(
|
||
"compression deadline expired before worker start"
|
||
)
|
||
if worker_fence.is_cancelled:
|
||
logger.info(
|
||
"Skipping stale compression job: fence cancelled before start"
|
||
)
|
||
return messages, ""
|
||
return worker(worker_fence)
|
||
|
||
# Bare pool workers start with an empty ContextVar map; propagate the
|
||
# parent conversation/approval context into the worker.
|
||
try:
|
||
future = executor.submit(
|
||
propagate_context_to_thread(_fence_gated_worker), fence
|
||
)
|
||
except BaseException:
|
||
_release_compression_admission()
|
||
raise
|
||
future.add_done_callback(_release_compression_admission)
|
||
wait_started = time.monotonic()
|
||
# EVERY host unwind must revoke commit admission or a detached worker could
|
||
# later mutate durable state; handled_exit marks paths that settle it themselves
|
||
handled_exit = False
|
||
try:
|
||
while True:
|
||
waited = time.monotonic() - wait_started
|
||
remaining_ceiling = ceiling - waited
|
||
if remaining_ceiling <= 0:
|
||
break
|
||
# Charge idle budget from LAST PROGRESS, not slice start, or silence could
|
||
# approach 2x the budget.
|
||
since_progress = fence.seconds_since_progress()
|
||
wait_slice = min(
|
||
max(idle - since_progress, 0.005), remaining_ceiling
|
||
)
|
||
try:
|
||
result = future.result(timeout=wait_slice)
|
||
handled_exit = True
|
||
return result
|
||
except concurrent.futures.TimeoutError:
|
||
waited = time.monotonic() - wait_started
|
||
since_progress = fence.seconds_since_progress()
|
||
if (
|
||
not fence.deadline_exceeded
|
||
and since_progress < idle
|
||
and waited < ceiling
|
||
):
|
||
logger.info(
|
||
"Context compression still streaming after %.0fs "
|
||
"(last progress %.1fs ago) — extending wait "
|
||
"(ceiling %.0fs)",
|
||
waited,
|
||
since_progress,
|
||
ceiling,
|
||
)
|
||
continue
|
||
break
|
||
|
||
# F6: a not-yet-started future must not linger as a stale queued job.
|
||
# cancel() is a no-op for a running worker (fence handles that path).
|
||
future.cancel()
|
||
|
||
total_exhausted = (
|
||
time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded
|
||
)
|
||
if total_exhausted:
|
||
# A total-ceiling candidate may be unwinding a healthy provider call; keep its
|
||
# lease until it exits so no other attempt overlaps the unchanged source.
|
||
fence.retain_compression_lock_until_worker_done()
|
||
|
||
if on_timeout_cause is not None:
|
||
try:
|
||
on_timeout_cause(total_exhausted, fence.progress_observed)
|
||
except Exception:
|
||
logger.debug(
|
||
"compress_context timeout-cause callback failed",
|
||
exc_info=True,
|
||
)
|
||
|
||
cancelled: Optional[bool] = None
|
||
while cancelled is None:
|
||
# begin_commit holds the fence lock until finish_commit, so try_cancel spins
|
||
# forever on a hung commit; lock-free marker makes the overrun loop reachable.
|
||
if fence.commit_in_flight:
|
||
cancelled = False
|
||
break
|
||
cancelled = fence.try_cancel_before_commit()
|
||
if cancelled is None:
|
||
# Fence is held only transiently here, but that window rides SessionDB write
|
||
# patience (seconds). 25ms keeps sub-tick latency without a 1kHz spin.
|
||
time.sleep(0.025)
|
||
if not cancelled:
|
||
# begin_commit won the race: SessionDB mutation cannot be fence-cancelled, so
|
||
# wait in bounded slices, logging (escalating) + surfacing once via
|
||
# on_commit_overrun WHILE the commit hangs. Never silently hung or abandoned.
|
||
overrun_surfaced = False
|
||
overrun_reports = 0
|
||
while True:
|
||
waited = time.monotonic() - wait_started
|
||
remaining = ceiling - waited
|
||
if remaining <= 0:
|
||
# Bounded increments so each overrun window is visible in logs rather than one
|
||
# silent unbounded block.
|
||
remaining = min(
|
||
_COMMIT_OVERRUN_WAIT_SLICE_SECONDS,
|
||
max(ceiling, 0.05),
|
||
)
|
||
overrun_reports += 1
|
||
log = (
|
||
logger.warning if overrun_reports <= 2 else logger.error
|
||
)
|
||
log(
|
||
"Context compression SessionDB commit still running "
|
||
"%.1fs past the total ceiling (waited %.1fs, ceiling "
|
||
"%.1fs); commit cannot be abandoned mid-flight — "
|
||
"continuing to wait (check SessionDB health if this "
|
||
"persists)",
|
||
waited - ceiling,
|
||
waited,
|
||
ceiling,
|
||
)
|
||
if not overrun_surfaced and on_commit_overrun is not None:
|
||
overrun_surfaced = True
|
||
try:
|
||
on_commit_overrun(waited, ceiling)
|
||
except Exception:
|
||
logger.debug(
|
||
"compress_context commit-overrun callback "
|
||
"failed",
|
||
exc_info=True,
|
||
)
|
||
try:
|
||
result = future.result(timeout=remaining)
|
||
handled_exit = True
|
||
return result
|
||
except concurrent.futures.TimeoutError:
|
||
# Commit-phase progress is informative only — the commit must complete; loop
|
||
# and re-report with the updated overrun window.
|
||
continue
|
||
|
||
# Idle-timeout: cancel won pre-commit. Also free the worker's durable lease via
|
||
# the holder-qualified hook so a NEW compressor can acquire at once (no ABA).
|
||
handled_exit = True
|
||
# Total-ceiling only: bounded grace for the worker to exit (it checks the fence
|
||
# between provider phases; an uninterruptible call is orphaned). Idle-stall
|
||
# skips the join: worker is hung, fallback needs a prompt return, fence guards.
|
||
if total_exhausted:
|
||
worker_exited = _join_cancelled_worker(
|
||
future,
|
||
min(_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS, ceiling),
|
||
)
|
||
if worker_exited:
|
||
# Worker provably exited: no provider call can outlive this attempt, so lease
|
||
# retention is unneeded and a retry cannot overlap.
|
||
fence.allow_cancelled_lock_release()
|
||
else:
|
||
logger.warning(
|
||
"Cancelled compression worker did not exit within %.1fs "
|
||
"grace — orphaning it behind the poison fence (late "
|
||
"result will be discarded); retaining the session "
|
||
"compression lease until it exits so no new attempt "
|
||
"overlaps it",
|
||
min(_CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS, ceiling),
|
||
)
|
||
fence.release_cancelled_compression_lock()
|
||
waited = time.monotonic() - wait_started
|
||
since_progress = fence.seconds_since_progress()
|
||
# Lease is free, so run the fallback BEFORE on_timeout: that callback records
|
||
# the summary-failure cooldown, which would no-op the retry's summary call.
|
||
if stall_fallback:
|
||
recovered = _retry_compression_on_fallback_chain(
|
||
worker=worker,
|
||
messages=messages,
|
||
system_prompt_fallback=system_prompt_fallback,
|
||
idle_timeout_seconds=idle,
|
||
total_ceiling_seconds=ceiling,
|
||
on_commit_overrun=on_commit_overrun,
|
||
on_timeout_cause=on_timeout_cause,
|
||
telemetry_agent=telemetry_agent,
|
||
new_fence=new_fence,
|
||
)
|
||
if recovered is not None:
|
||
return recovered
|
||
if on_timeout is not None:
|
||
try:
|
||
on_timeout(idle, waited, since_progress)
|
||
except Exception:
|
||
logger.debug(
|
||
"compress_context timeout callback failed",
|
||
exc_info=True,
|
||
)
|
||
else:
|
||
logger.warning(
|
||
"Context compression made no progress for %.1fs "
|
||
"(total wait %.1fs, ceiling %.1fs); continuing without "
|
||
"compression",
|
||
since_progress,
|
||
waited,
|
||
ceiling,
|
||
)
|
||
# Leave the future on the shared pool: fence cancel won, so a late
|
||
# commit cannot land (same detachment model as gateway hygiene).
|
||
return messages, _resolve_fallback_prompt()
|
||
finally:
|
||
if not handled_exit:
|
||
# Any unwind while waiting: revoke commit admission and release the worker's
|
||
# lease before the host unwinds, so the detached worker can never publish.
|
||
fence.revoke_commit_admission()
|
||
|
||
|
||
class CompressionCheckpointUnavailable(RuntimeError):
|
||
"""Raised when required durable pre-compress checkpointing is unavailable."""
|
||
|
||
|
||
def _checkpoint_blocked(reason: str) -> CompressionCheckpointUnavailable:
|
||
return CompressionCheckpointUnavailable(
|
||
"BLOCKED_MISSING_PREREQUISITE: required pre-compress checkpoint "
|
||
f"unavailable: {reason}"
|
||
)
|
||
|
||
|
||
def _lock_api_is_absent_on_session_db(lock_db: Any) -> bool:
|
||
"""Whether the live in-memory SessionDB class structurally predates locks.
|
||
|
||
Only the exact old ``hermes_state.SessionDB`` class (hot-reload skew) may fail
|
||
open; proxies, lookalikes, non-callables and descriptor failures fail closed.
|
||
"""
|
||
try:
|
||
from hermes_state import SessionDB
|
||
|
||
missing = object()
|
||
return (
|
||
type(lock_db) is SessionDB
|
||
and inspect.getattr_static(
|
||
SessionDB, "try_acquire_compression_lock", missing
|
||
) is missing
|
||
)
|
||
except Exception:
|
||
return False
|
||
|
||
|
||
def _refresh_persisted_compression_guards(
|
||
compressor: Any,
|
||
*,
|
||
include_cooldown: bool = True,
|
||
) -> None:
|
||
"""Refresh durable automatic-compression guards on a built-in compressor."""
|
||
method_calls = [
|
||
("_load_fallback_compression_streak", {}),
|
||
("_load_ineffective_compression_count", {}),
|
||
]
|
||
if include_cooldown:
|
||
method_calls.insert(
|
||
0,
|
||
("get_active_compression_failure_cooldown", {"refresh": True}),
|
||
)
|
||
for method_name, kwargs in method_calls:
|
||
method = getattr(type(compressor), method_name, None)
|
||
if not callable(method):
|
||
continue
|
||
try:
|
||
method(compressor, **kwargs)
|
||
except Exception as exc:
|
||
logger.debug("compression guard refresh failed (%s): %s", method_name, exc)
|
||
|
||
|
||
def _session_was_rotated_by_compression(session_db: Any, session_id: str) -> bool:
|
||
"""Return whether another path already rotated this compression parent."""
|
||
getter = getattr(type(session_db), "get_session", None)
|
||
if not callable(getter):
|
||
return False
|
||
session = getter(session_db, session_id)
|
||
return bool(
|
||
session
|
||
and session.get("ended_at") is not None
|
||
and session.get("end_reason") == "compression"
|
||
)
|
||
|
||
|
||
def _emit_compression_attempt_telemetry(
|
||
agent: Any,
|
||
*,
|
||
started_at: float,
|
||
commit_status: str,
|
||
split_status: str,
|
||
failure_class: str | None = None,
|
||
commit_started_at: float | None = None,
|
||
) -> None:
|
||
"""Emit one content-free JSON log line for a compression attempt."""
|
||
try:
|
||
telemetry = getattr(agent.context_compressor, "_last_compression_telemetry", None)
|
||
if not isinstance(telemetry, dict):
|
||
telemetry = {}
|
||
payload = dict(telemetry)
|
||
payload.setdefault("event", "compression_attempt")
|
||
payload.setdefault("attempt_id", getattr(agent, "_compression_attempt_id", "") or uuid.uuid4().hex)
|
||
payload.setdefault("session_id", getattr(agent, "session_id", "") or "")
|
||
payload["total_duration_ms"] = int((time.monotonic() - started_at) * 1000)
|
||
payload["commit_status"] = commit_status
|
||
payload["split_status"] = split_status
|
||
if commit_started_at is not None:
|
||
commit_ms = max(0, int((time.monotonic() - commit_started_at) * 1000))
|
||
telemetry["commit_ms"] = commit_ms
|
||
payload["commit_ms"] = commit_ms
|
||
if failure_class:
|
||
payload["failure_class"] = failure_class
|
||
payload.setdefault("chunking", False)
|
||
payload.setdefault("chunk_count", 0)
|
||
payload["fallback_used"] = bool(
|
||
payload.get("fallback_used")
|
||
or getattr(agent.context_compressor, "_last_summary_fallback_used", False)
|
||
or getattr(agent.context_compressor, "_last_aux_model_failure_model", None)
|
||
)
|
||
logger.info(
|
||
"context compression attempt telemetry: %s",
|
||
json.dumps(payload, sort_keys=True, separators=(",", ":")),
|
||
)
|
||
except Exception as exc:
|
||
logger.debug("failed to emit compression attempt telemetry: %s", exc)
|
||
|
||
|
||
def _existing_system_prompt(agent: Any, system_message: str) -> str:
|
||
"""Cached system prompt, or a fresh build when nothing is cached (abort paths)."""
|
||
existing = getattr(agent, "_cached_system_prompt", None)
|
||
if not existing:
|
||
existing = agent._build_system_prompt(system_message)
|
||
return existing
|
||
|
||
|
||
def _emit_aborted_attempt_telemetry(
|
||
agent: Any, started_at: float, failure_class: str | None
|
||
) -> None:
|
||
_emit_compression_attempt_telemetry(
|
||
agent,
|
||
started_at=started_at,
|
||
commit_status="aborted",
|
||
split_status="aborted",
|
||
failure_class=failure_class,
|
||
)
|
||
|
||
|
||
def _restore_messages_snapshot(messages: list, snapshot: Optional[list]) -> None:
|
||
"""Put the pre-compression deep snapshot back into the live list if it drifted."""
|
||
if snapshot is not None and messages != snapshot:
|
||
messages[:] = copy.deepcopy(snapshot)
|
||
|
||
|
||
def _restore_prune_rearm_tokens(compressor: Any, snapshot: dict) -> None:
|
||
"""Restore ONLY the prune runway from the attempt snapshot.
|
||
|
||
compress() zeroes it in memory while the durable copy only clears on a
|
||
successful commit; a kept transcript keeps its cached prefix, and 0 would let
|
||
the next prune break that cache.
|
||
"""
|
||
if "_proactive_prune_rearm_tokens" in snapshot:
|
||
compressor._proactive_prune_rearm_tokens = snapshot["_proactive_prune_rearm_tokens"]
|
||
|
||
|
||
def compression_skipped_due_to_lock(agent: Any) -> bool:
|
||
"""Type-pinned read of the per-session lock-skip signal.
|
||
|
||
``agent._compression_skipped_due_to_lock`` is a holder string or ``True`` when
|
||
a pass no-oped because the lock was held, ``None`` otherwise. Pinning avoids
|
||
MagicMock auto-attributes hijacking mocked agents into the lock-skip branch.
|
||
"""
|
||
_sig = getattr(agent, "_compression_skipped_due_to_lock", None)
|
||
return _sig is True or isinstance(_sig, str)
|
||
|
||
|
||
def _get_context_compression_timeout_state(
|
||
agent: Any,
|
||
*,
|
||
create: bool,
|
||
) -> Optional[Tuple[Any, Optional[threading.local]]]:
|
||
"""Return the stable lock and thread-local timeout state for an agent."""
|
||
try:
|
||
attributes = vars(agent)
|
||
except TypeError:
|
||
return None
|
||
|
||
lock = attributes.setdefault(
|
||
"_context_compression_timeout_state_lock",
|
||
threading.Lock(),
|
||
)
|
||
with lock:
|
||
state = attributes.get("_context_compression_timeout_state")
|
||
if create and not isinstance(state, threading.local):
|
||
state = threading.local()
|
||
attributes["_context_compression_timeout_state"] = state
|
||
return lock, state if isinstance(state, threading.local) else None
|
||
|
||
|
||
def reset_context_compression_timeout_outcome(agent: Any) -> None:
|
||
"""Clear the current thread's owned-compression timeout outcome.
|
||
|
||
The ``agent._last_compression_timed_out`` mirror stays authoritative for
|
||
minimal agent doubles that do not support ``vars()``.
|
||
"""
|
||
locked_state = _get_context_compression_timeout_state(agent, create=True)
|
||
if locked_state is None or locked_state[1] is None:
|
||
agent._last_compression_timed_out = False
|
||
return
|
||
lock, state = locked_state
|
||
with lock:
|
||
state.timed_out = False
|
||
agent._last_compression_timed_out = False
|
||
|
||
|
||
def mark_context_compression_timed_out(agent: Any) -> None:
|
||
"""Mark the current owned compression as host-timed-out."""
|
||
locked_state = _get_context_compression_timeout_state(agent, create=True)
|
||
if locked_state is None or locked_state[1] is None:
|
||
agent._last_compression_timed_out = True
|
||
return
|
||
lock, state = locked_state
|
||
with lock:
|
||
state.timed_out = True
|
||
agent._last_compression_timed_out = True
|
||
|
||
|
||
def context_compression_timed_out(agent: Any) -> bool:
|
||
"""Return whether this thread's owned compression hit its host timeout.
|
||
|
||
Thread-local so overlapping automatic/manual entrypoints cannot hide each
|
||
other's timeout; attribute fallback for minimal doubles; reads type-pinned.
|
||
"""
|
||
locked_state = _get_context_compression_timeout_state(agent, create=False)
|
||
if locked_state is not None:
|
||
lock, state = locked_state
|
||
with lock:
|
||
if isinstance(state, threading.local):
|
||
return getattr(state, "timed_out", None) is True
|
||
return getattr(agent, "_last_compression_timed_out", None) is True
|
||
|
||
|
||
def _automatic_gate_blocked(
|
||
blocked: Any, compressor: Any, bypass_cooldown: bool
|
||
) -> bool:
|
||
"""Evaluate the automatic breaker gate, optionally ignoring the cooldown.
|
||
|
||
Engines whose gate predates ``bypass_cooldown`` are called with the legacy
|
||
no-argument shape.
|
||
"""
|
||
if bypass_cooldown:
|
||
try:
|
||
accepts = "ignore_cooldown" in inspect.signature(blocked).parameters
|
||
except (TypeError, ValueError):
|
||
accepts = False
|
||
if accepts:
|
||
return bool(blocked(compressor, ignore_cooldown=True))
|
||
return bool(blocked(compressor))
|
||
|
||
|
||
def compression_blocked_transiently(agent: Any) -> bool:
|
||
"""Type-pinned read of the transient-block signal.
|
||
|
||
Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure
|
||
cooldown or structural backoff). Consumers must defer, not count it toward
|
||
``compression_exhausted``, or an overflow auto-reset wipes a session that was
|
||
merely cooling down. The permanent ``ineffective`` breaker never sets it.
|
||
"""
|
||
_sig = getattr(agent, "_compression_blocked_transient", None)
|
||
return isinstance(_sig, str) and bool(_sig)
|
||
|
||
|
||
def _mark_compression_blocked_transient(agent: Any, compressor: Any) -> None:
|
||
"""Publish the transient-block signal when the active guard is transient.
|
||
|
||
Classification comes from ``_compression_block_reason``: ``cooldown:*`` and
|
||
``structural_backoff:*`` are transient; ``ineffective`` stays unmarked.
|
||
"""
|
||
reason_fn = getattr(compressor, "_compression_block_reason", None)
|
||
reason = None
|
||
if callable(reason_fn):
|
||
try:
|
||
reason = reason_fn()
|
||
except Exception:
|
||
logger.debug("compression block-reason read failed", exc_info=True)
|
||
if isinstance(reason, str) and (
|
||
reason.startswith("cooldown") or reason.startswith("structural_backoff")
|
||
):
|
||
logger.info(
|
||
"Skipping automatic compression re-entry: transient guard "
|
||
"active (%s, session=%s, last failure: %s) — will retry after "
|
||
"the backoff lapses; /compress forces an immediate retry",
|
||
reason,
|
||
getattr(agent, "session_id", None) or "none",
|
||
getattr(compressor, "_last_summary_error", None) or "unknown",
|
||
)
|
||
try:
|
||
agent._compression_blocked_transient = reason
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def _adopt_live_compression_child(
|
||
agent: Any,
|
||
session_db: Any,
|
||
parent_session_id: str,
|
||
) -> Optional[List[Dict[str, Any]]]:
|
||
"""Move a stale compression contender onto the live continuation tip.
|
||
|
||
Resolve and load first, then mutate the agent, so ambiguous lineage or an
|
||
unreadable handoff fails closed. Uses the transitive ``get_compression_tip``
|
||
walk; a tip is adopted only while its row is still live.
|
||
"""
|
||
resolver = getattr(type(session_db), "get_compression_tip", None)
|
||
row_getter = getattr(type(session_db), "get_session", None)
|
||
loader = getattr(type(session_db), "get_messages_as_conversation", None)
|
||
if not callable(resolver) or not callable(row_getter) or not callable(loader):
|
||
return None
|
||
tip = resolver(session_db, parent_session_id)
|
||
if not tip or str(tip) == str(parent_session_id):
|
||
return None
|
||
child_session_id = str(tip)
|
||
child = row_getter(session_db, child_session_id)
|
||
if not isinstance(child, dict) or child.get("ended_at") is not None:
|
||
return None
|
||
recovered = loader(session_db, child_session_id)
|
||
if not isinstance(recovered, list) or not recovered:
|
||
return None
|
||
# Revalidate after loading: the tip may have rotated or a competing
|
||
# continuation may have appeared between the two DB reads.
|
||
confirmed = resolver(session_db, parent_session_id)
|
||
if not confirmed or str(confirmed) != child_session_id:
|
||
return None
|
||
|
||
agent.session_id = child_session_id
|
||
try:
|
||
from gateway.session_context import set_current_session_id
|
||
|
||
set_current_session_id(child_session_id)
|
||
except Exception:
|
||
os.environ["HERMES_SESSION_ID"] = child_session_id
|
||
try:
|
||
from hermes_logging import set_session_context
|
||
|
||
set_session_context(child_session_id)
|
||
except Exception:
|
||
pass
|
||
|
||
agent._session_db_created = True
|
||
if child.get("system_prompt"):
|
||
agent._cached_system_prompt = child["system_prompt"]
|
||
agent._last_flushed_db_idx = len(recovered)
|
||
agent._flushed_db_message_session_id = child_session_id
|
||
agent._flushed_db_message_ids = {
|
||
id(message) for message in recovered if isinstance(message, dict)
|
||
}
|
||
|
||
on_session_start = getattr(agent.context_compressor, "on_session_start", None)
|
||
if callable(on_session_start):
|
||
try:
|
||
on_session_start(
|
||
child_session_id,
|
||
boundary_reason="compression",
|
||
old_session_id=parent_session_id,
|
||
session_db=session_db,
|
||
platform=getattr(agent, "platform", None) or "cli",
|
||
conversation_id=getattr(agent, "_gateway_session_key", None),
|
||
)
|
||
except Exception as exc:
|
||
logger.debug("context engine compression-child adoption failed: %s", exc)
|
||
else:
|
||
bind_state = getattr(agent.context_compressor, "bind_session_state", None)
|
||
if callable(bind_state):
|
||
try:
|
||
bind_state(session_db=session_db, session_id=child_session_id)
|
||
except Exception:
|
||
pass
|
||
try:
|
||
if agent._memory_manager:
|
||
agent._memory_manager.on_session_switch(
|
||
child_session_id,
|
||
parent_session_id=parent_session_id,
|
||
reset=False,
|
||
reason="compression",
|
||
)
|
||
except Exception as exc:
|
||
logger.debug("memory manager compression-child adoption failed: %s", exc)
|
||
|
||
return recovered
|
||
|
||
|
||
def recover_rotated_compression_session(
|
||
agent: Any,
|
||
) -> Optional[List[Dict[str, Any]]]:
|
||
"""Recover a stale live agent before a new turn writes to its old parent."""
|
||
session_db = getattr(agent, "_session_db", None)
|
||
session_id = getattr(agent, "session_id", None) or ""
|
||
if session_db is None or not session_id:
|
||
return None
|
||
try:
|
||
if not _session_was_rotated_by_compression(session_db, session_id):
|
||
return None
|
||
# Rotation holds the parent lease until the child handoff is durable; wait
|
||
# briefly rather than observe the parent-ended/child-empty intermediate state.
|
||
holder_getter = getattr(session_db, "get_compression_lock_holder", None)
|
||
for attempt in range(21):
|
||
recovered = _adopt_live_compression_child(agent, session_db, session_id)
|
||
if recovered is not None:
|
||
return recovered
|
||
holder = holder_getter(session_id) if callable(holder_getter) else None
|
||
if not holder or attempt == 20:
|
||
if not holder:
|
||
orphan_reopener = getattr(
|
||
type(session_db),
|
||
"reopen_orphaned_compression_session",
|
||
None,
|
||
)
|
||
if callable(orphan_reopener):
|
||
try:
|
||
if orphan_reopener(session_db, session_id):
|
||
logger.warning(
|
||
"compression recovery: reopened orphaned "
|
||
"session=%s with no continuation",
|
||
session_id,
|
||
)
|
||
except Exception as exc:
|
||
logger.warning(
|
||
"orphaned compression session reopen failed "
|
||
"for %s: %s",
|
||
session_id,
|
||
exc,
|
||
)
|
||
return None
|
||
time.sleep(0.05)
|
||
return None
|
||
except Exception as exc:
|
||
logger.warning(
|
||
"compression session recovery failed for session=%s (%s: %s)",
|
||
session_id,
|
||
type(exc).__name__,
|
||
exc,
|
||
)
|
||
return None
|
||
|
||
|
||
def _compression_lock_holder(agent: Any) -> str:
|
||
"""Build a unique lock holder id: ``pid:tid:agent-instance:uuid``.
|
||
|
||
pid+tid tell crashed holders apart in diagnostics; instance id and per-acquire
|
||
uuid disambiguate co-resident agents on one thread or pooled compressions.
|
||
"""
|
||
import threading
|
||
return (
|
||
f"pid={os.getpid()}"
|
||
f":tid={threading.get_ident()}"
|
||
f":agent={id(agent):x}"
|
||
f":nonce={uuid.uuid4().hex[:8]}"
|
||
)
|
||
|
||
|
||
def _supported_compression_kwargs(
|
||
compress_fn: Any,
|
||
*,
|
||
current_tokens: Optional[int],
|
||
focus_topic: Optional[str],
|
||
force: bool,
|
||
memory_context: str,
|
||
bypass_cooldown: bool = False,
|
||
) -> dict:
|
||
"""Return only compression kwargs accepted by an engine callable.
|
||
|
||
Inspecting first keeps older plugin signatures compatible without catching
|
||
``TypeError`` and running a stateful compressor twice.
|
||
"""
|
||
candidates = {
|
||
"current_tokens": current_tokens,
|
||
"focus_topic": focus_topic,
|
||
"force": force,
|
||
}
|
||
if bypass_cooldown:
|
||
candidates["bypass_cooldown"] = True
|
||
if memory_context:
|
||
candidates["memory_context"] = memory_context
|
||
try:
|
||
parameters = inspect.signature(compress_fn).parameters
|
||
except (TypeError, ValueError):
|
||
# current_tokens has always been in the ContextEngine ABC; use the oldest call
|
||
# shape when the callable has no inspectable signature.
|
||
return {"current_tokens": current_tokens}
|
||
|
||
accepts_kwargs = any(
|
||
parameter.kind is inspect.Parameter.VAR_KEYWORD
|
||
for parameter in parameters.values()
|
||
)
|
||
if accepts_kwargs:
|
||
return candidates
|
||
return {name: value for name, value in candidates.items() if name in parameters}
|
||
|
||
|
||
class _CompressionActivityHeartbeat:
|
||
"""Refresh the agent inactivity tracker while compression blocks in an aux call."""
|
||
|
||
def __init__(
|
||
self,
|
||
agent: Any,
|
||
interval_seconds: float | None = None,
|
||
commit_fence: Optional[CompressionCommitFence] = None,
|
||
) -> None:
|
||
self._agent = agent
|
||
self._commit_fence = commit_fence
|
||
# Latched once host cancel/timeout wins or a terminal stamp is observed,
|
||
# so a later UNKNOWN rewrite cannot re-arm a detached zombie heartbeat.
|
||
self._suppressed = False
|
||
if interval_seconds is None:
|
||
interval_seconds = getattr(agent, "_compression_activity_heartbeat_interval", 60.0)
|
||
try:
|
||
interval_seconds = float(interval_seconds or 60.0)
|
||
except (TypeError, ValueError):
|
||
interval_seconds = 60.0
|
||
if not math.isfinite(interval_seconds):
|
||
interval_seconds = 60.0
|
||
self._interval_seconds = max(0.1, interval_seconds)
|
||
self._stop = threading.Event()
|
||
self._thread = threading.Thread(
|
||
target=self._run,
|
||
name="compression-activity-heartbeat",
|
||
daemon=True,
|
||
)
|
||
|
||
def start(self) -> "_CompressionActivityHeartbeat":
|
||
# A new compression episode always republishes agent.compression even
|
||
# if a prior timeout/cooldown stamp is still on the agent.
|
||
self._suppressed = False
|
||
self._touch("context compression started", allow_terminal_overwrite=True)
|
||
self._thread.start()
|
||
return self
|
||
|
||
def stop(self, desc: str = "context compression completed") -> None:
|
||
self._stop.set()
|
||
if self._thread.is_alive() and threading.current_thread() is not self._thread:
|
||
self._thread.join(timeout=1.0)
|
||
# Host timeout already owns the terminal stamp; a detached worker's
|
||
# late stop must not republish agent.compression / "completed".
|
||
if self._should_suppress():
|
||
return
|
||
# Force persist: /compress never hits run_conversation's turn-end clear, so
|
||
# durable labels would stay "in progress" for the 60s persist window.
|
||
self._touch(desc, force_persist=True)
|
||
|
||
def _fence_cancelled(self) -> bool:
|
||
fence = self._commit_fence
|
||
return fence is not None and fence.is_cancelled
|
||
|
||
def _should_suppress(self) -> bool:
|
||
if self._suppressed:
|
||
return True
|
||
if self._fence_cancelled():
|
||
self._suppressed = True
|
||
return True
|
||
return False
|
||
|
||
def _touch(
|
||
self,
|
||
desc: str,
|
||
*,
|
||
allow_terminal_overwrite: bool = False,
|
||
force_persist: bool = False,
|
||
) -> None:
|
||
try:
|
||
if not allow_terminal_overwrite:
|
||
if self._should_suppress():
|
||
return
|
||
current = normalize_activity_provenance(
|
||
getattr(self._agent, "_last_activity_provenance", None)
|
||
)
|
||
if current in _TERMINAL_COMPRESSION_PROVENANCES:
|
||
self._suppressed = True
|
||
return
|
||
touch = getattr(self._agent, "_touch_activity", None)
|
||
if callable(touch):
|
||
# Re-check after reading provenance: host may cancel/stamp
|
||
# TIMEOUT between the earlier guard and the write.
|
||
if not allow_terminal_overwrite and self._should_suppress():
|
||
return
|
||
touch(
|
||
desc,
|
||
provenance=ActivityProvenance.AGENT_COMPRESSION,
|
||
force_persist=force_persist,
|
||
)
|
||
except Exception:
|
||
logger.debug("compression activity heartbeat touch failed", exc_info=True)
|
||
|
||
def _run(self) -> None:
|
||
while not self._stop.wait(self._interval_seconds):
|
||
if self._should_suppress():
|
||
return
|
||
self._touch("context compression in progress")
|
||
|
||
|
||
def _direct_messages_for_pre_compress_memory(messages: Any) -> list[dict[str, Any]]:
|
||
"""Return direct user/assistant evidence safe for memory checkpointing.
|
||
|
||
Summaries, tool rows and system messages are omitted; assistant prose is kept
|
||
with ``tool_calls`` stripped, and pure tool-call wrappers are dropped.
|
||
"""
|
||
# Deferred import: context_compressor → turn_context → this module would form
|
||
# an import cycle.
|
||
from agent.context_compressor import COMPRESSED_SUMMARY_METADATA_KEY
|
||
|
||
direct_messages: list[dict[str, Any]] = []
|
||
for message in messages or []:
|
||
if not isinstance(message, dict):
|
||
continue
|
||
role = message.get("role")
|
||
if role not in {"user", "assistant"}:
|
||
continue
|
||
if message.get(COMPRESSED_SUMMARY_METADATA_KEY):
|
||
continue
|
||
if role == "assistant" and message.get("tool_calls"):
|
||
content = message.get("content")
|
||
has_prose = bool(
|
||
content.strip() if isinstance(content, str) else content
|
||
)
|
||
if not has_prose:
|
||
continue
|
||
message = {k: v for k, v in message.items() if k != "tool_calls"}
|
||
direct_messages.append(message)
|
||
return direct_messages
|
||
|
||
|
||
class _CompressionLockLeaseRefresher:
|
||
def __init__(
|
||
self,
|
||
db: Any,
|
||
session_id: str,
|
||
holder: str,
|
||
ttl_seconds: float,
|
||
refresh_interval_seconds: float | None = None,
|
||
) -> None:
|
||
self._db = db
|
||
self._session_id = session_id
|
||
self._holder = holder
|
||
self._ttl_seconds = ttl_seconds
|
||
if refresh_interval_seconds is None:
|
||
refresh_interval_seconds = max(1.0, min(60.0, ttl_seconds / 2.0))
|
||
self._refresh_interval_seconds = max(0.1, float(refresh_interval_seconds))
|
||
# Tolerate transient refresh failures for at most one TTL so the lease cannot
|
||
# outlive its TTL; floor 1 so interval >= ttl still tolerates one blip.
|
||
self._max_consecutive_failures = max(
|
||
1, int(self._ttl_seconds / self._refresh_interval_seconds)
|
||
)
|
||
self._stop = threading.Event()
|
||
self._thread = threading.Thread(
|
||
target=self._run,
|
||
name="compression-lock-refresh",
|
||
daemon=True,
|
||
)
|
||
|
||
def start(self) -> "_CompressionLockLeaseRefresher":
|
||
self._thread.start()
|
||
return self
|
||
|
||
def stop(self) -> None:
|
||
self._stop.set()
|
||
# join() timing out mid-UPDATE is safe: daemon thread, and a late refresh on a
|
||
# released lock is a rowcount-0 no-op. stop() does not guarantee quiescence.
|
||
if self._thread.is_alive() and threading.current_thread() is not self._thread:
|
||
self._thread.join(timeout=1.0)
|
||
|
||
def _run(self) -> None:
|
||
# A single falsy refresh (transient DB blip) must not kill the lease; only
|
||
# ttl/interval consecutive failures do, so a stuck refresher never outlives TTL.
|
||
consecutive_failures = 0
|
||
# Refresh immediately: work between try_acquire() and start() is charged to the
|
||
# first lease, so on a short TTL it could expire before tick #1.
|
||
first = True
|
||
while first or not self._stop.wait(self._refresh_interval_seconds):
|
||
if first:
|
||
first = False
|
||
if self._stop.is_set():
|
||
break
|
||
try:
|
||
refreshed = self._db.refresh_compression_lock(
|
||
self._session_id,
|
||
self._holder,
|
||
ttl_seconds=self._ttl_seconds,
|
||
)
|
||
except Exception as exc:
|
||
logger.debug("compression lock refresh raised: %s", exc)
|
||
refreshed = False
|
||
if refreshed:
|
||
consecutive_failures = 0
|
||
continue
|
||
consecutive_failures += 1
|
||
if consecutive_failures >= self._max_consecutive_failures:
|
||
logger.debug(
|
||
"compression lock refresh failed %d times in a row; "
|
||
"stopping lease refresher for session %s",
|
||
consecutive_failures, self._session_id,
|
||
)
|
||
break
|
||
|
||
|
||
def check_compression_model_feasibility(agent: Any) -> None:
|
||
"""Warn at session start if the aux compression context is below the threshold.
|
||
|
||
Called from ``AIAgent.__init__`` (CLI sees it via ``_vprint``); the gateway
|
||
wires ``status_callback`` later, so ``replay_compression_warning`` resends it.
|
||
"""
|
||
if not agent.compression_enabled:
|
||
return
|
||
try:
|
||
from agent.auxiliary_client import (
|
||
_resolve_task_provider_model,
|
||
_try_configured_fallback_for_unavailable_client,
|
||
get_text_auxiliary_client,
|
||
)
|
||
from agent.model_metadata import (
|
||
MINIMUM_CONTEXT_LENGTH,
|
||
get_model_context_length,
|
||
)
|
||
|
||
# Provider may be "auto"; fall back to the client's base_url hostname so the
|
||
# user can tell where the compression model is actually called.
|
||
try:
|
||
_aux_cfg_provider, _, _, _, _ = _resolve_task_provider_model("compression")
|
||
except Exception:
|
||
_aux_cfg_provider = ""
|
||
client, aux_model = get_text_auxiliary_client(
|
||
"compression",
|
||
main_runtime=agent._current_main_runtime(),
|
||
)
|
||
if client is None or not aux_model:
|
||
fb_client, fb_model, fb_label = _try_configured_fallback_for_unavailable_client(
|
||
"compression",
|
||
_aux_cfg_provider,
|
||
)
|
||
if fb_client is not None and fb_model:
|
||
client, aux_model = fb_client, fb_model
|
||
if "(" in fb_label and fb_label.endswith(")"):
|
||
_aux_cfg_provider = fb_label.rsplit("(", 1)[1][:-1]
|
||
if client is None or not aux_model:
|
||
if _aux_cfg_provider and _aux_cfg_provider != "auto":
|
||
msg = (
|
||
"⚠ Configured auxiliary compression provider "
|
||
f"'{_aux_cfg_provider}' is unavailable — context "
|
||
"compression will drop middle turns without a summary. "
|
||
"Check auxiliary.compression in config.yaml and "
|
||
"reauthenticate that provider."
|
||
)
|
||
else:
|
||
msg = (
|
||
"⚠ No auxiliary LLM provider configured — context "
|
||
"compression will drop middle turns without a summary. "
|
||
"Run `hermes setup` or set OPENROUTER_API_KEY."
|
||
)
|
||
agent._compression_warning = msg
|
||
agent._emit_status(msg)
|
||
logger.warning(
|
||
"No auxiliary LLM provider for compression — "
|
||
"summaries will be unavailable."
|
||
)
|
||
return
|
||
|
||
aux_base_url = str(getattr(client, "base_url", ""))
|
||
# client.api_key may be a callable (Entra bearer); the resolver only needs a key
|
||
# for live catalogue probes, so pass "" rather than mint a JWT for a lookup.
|
||
_raw_aux_key = getattr(client, "api_key", "")
|
||
aux_api_key = "" if (callable(_raw_aux_key) and not isinstance(_raw_aux_key, str)) else str(_raw_aux_key or "")
|
||
|
||
aux_context = get_model_context_length(
|
||
aux_model,
|
||
base_url=aux_base_url,
|
||
api_key=aux_api_key,
|
||
config_context_length=getattr(agent, "_aux_compression_context_length_config", None),
|
||
# Resolve each model with its own provider so provider-specific paths (Bedrock
|
||
# table, OpenRouter API) hit the correct client, not the main model's.
|
||
provider=(_aux_cfg_provider if _aux_cfg_provider and _aux_cfg_provider != "auto" else getattr(agent, "provider", "")),
|
||
custom_providers=agent._custom_providers,
|
||
)
|
||
|
||
# Aux model must meet MINIMUM_CONTEXT_LENGTH like the main model, else it cannot
|
||
# summarise a full threshold-sized window.
|
||
if aux_context and aux_context < MINIMUM_CONTEXT_LENGTH:
|
||
raise ValueError(
|
||
f"Auxiliary compression model {aux_model} has a context "
|
||
f"window of {aux_context:,} tokens, which is below the "
|
||
f"minimum {MINIMUM_CONTEXT_LENGTH:,} required by Hermes "
|
||
f"Agent. Choose a compression model with at least "
|
||
f"{MINIMUM_CONTEXT_LENGTH // 1000}K context (set "
|
||
f"auxiliary.compression.model in config.yaml), or set "
|
||
f"auxiliary.compression.context_length to override the "
|
||
f"detected value if it is wrong."
|
||
)
|
||
|
||
threshold = agent.context_compressor.threshold_tokens
|
||
if aux_context < threshold:
|
||
# Lower the live threshold so compression works this session. The summariser
|
||
# sends one user prompt (no system/tools), so threshold == aux_context is safe.
|
||
old_threshold = threshold
|
||
new_threshold = aux_context
|
||
agent.context_compressor.threshold_tokens = new_threshold
|
||
# tail_token_budget derives from the threshold; keep it in lockstep (as
|
||
# update_model does) or the 1.5x tail ceiling exceeds the trigger and re-fires.
|
||
summary_target_ratio = getattr(
|
||
agent.context_compressor, "summary_target_ratio", None
|
||
)
|
||
if isinstance(summary_target_ratio, (int, float)):
|
||
agent.context_compressor.tail_token_budget = int(
|
||
new_threshold * summary_target_ratio
|
||
)
|
||
# Keep threshold_percent in sync so update_model re-derives from a sensible
|
||
# value rather than the original too-high one.
|
||
main_ctx = agent.context_compressor.context_length
|
||
if main_ctx:
|
||
agent.context_compressor.threshold_percent = (
|
||
new_threshold / main_ctx
|
||
)
|
||
safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50
|
||
# Mirror the compressor's threshold math (percent floor, output reservation,
|
||
# 64K floor): a suggestion it would override is silently ignored and this
|
||
# warning reappears every session. External engines own policy: keep it plain.
|
||
from agent.context_compressor import ContextCompressor as _CC
|
||
|
||
recomputed_threshold = None
|
||
if main_ctx and isinstance(agent.context_compressor, _CC):
|
||
recomputed_threshold = _CC._compute_threshold_tokens(
|
||
main_ctx,
|
||
_CC._effective_threshold_percent(main_ctx, safe_pct / 100),
|
||
getattr(agent.context_compressor, "max_tokens", None),
|
||
)
|
||
threshold_suggestion_viable = (
|
||
recomputed_threshold is None or recomputed_threshold <= aux_context
|
||
)
|
||
# "model (provider)" labels for both sides; empty/"auto" provider falls back to
|
||
# the client's base_url hostname.
|
||
_main_model = getattr(agent, "model", "") or "?"
|
||
_main_provider = getattr(agent, "provider", "") or ""
|
||
_aux_provider_label = (
|
||
_aux_cfg_provider
|
||
if _aux_cfg_provider and _aux_cfg_provider != "auto"
|
||
else ""
|
||
)
|
||
if not _aux_provider_label:
|
||
try:
|
||
from urllib.parse import urlparse
|
||
_aux_provider_label = (
|
||
urlparse(aux_base_url).hostname or aux_base_url
|
||
)
|
||
except Exception:
|
||
_aux_provider_label = aux_base_url or "auto"
|
||
_main_label = (
|
||
f"{_main_model} ({_main_provider})"
|
||
if _main_provider
|
||
else _main_model
|
||
)
|
||
_aux_label = f"{aux_model} ({_aux_provider_label})"
|
||
msg = (
|
||
f"⚠ Compression model {_aux_label} context is "
|
||
f"{aux_context:,} tokens, but the main model "
|
||
f"{_main_label}'s compression threshold was "
|
||
f"{old_threshold:,} tokens. "
|
||
f"Auto-lowered this session's threshold to "
|
||
f"{new_threshold:,} tokens so compression can run.\n"
|
||
)
|
||
if threshold_suggestion_viable:
|
||
msg += (
|
||
f" To make this permanent, edit config.yaml — either:\n"
|
||
f" 1. Use a larger compression model:\n"
|
||
f" auxiliary:\n"
|
||
f" compression:\n"
|
||
f" model: <model-with-{old_threshold:,}+-context>\n"
|
||
f" 2. Lower the compression threshold:\n"
|
||
f" compression:\n"
|
||
f" threshold: 0.{safe_pct:02d}"
|
||
)
|
||
else:
|
||
msg += (
|
||
f" To make this permanent, use a larger compression "
|
||
f"model in config.yaml:\n"
|
||
f" auxiliary:\n"
|
||
f" compression:\n"
|
||
f" model: <model-with-{old_threshold:,}+-context>\n"
|
||
f" (Lowering compression.threshold cannot help here — "
|
||
f"with {_main_label}'s {main_ctx:,}-token window, "
|
||
f"Hermes's small-context floor and output reservation "
|
||
f"would recompute the trigger to "
|
||
f"{recomputed_threshold:,} tokens, still above the "
|
||
f"compression model's {aux_context:,}.)"
|
||
)
|
||
agent._compression_warning = msg
|
||
agent._emit_status(msg)
|
||
logger.warning(
|
||
"Auxiliary compression model %s has %d token context, "
|
||
"below the main model's compression threshold of %d "
|
||
"tokens — auto-lowered session threshold to %d to "
|
||
"keep compression working.",
|
||
aux_model,
|
||
aux_context,
|
||
old_threshold,
|
||
new_threshold,
|
||
)
|
||
except ValueError:
|
||
# Hard rejections (aux below minimum context) must propagate
|
||
# so the session refuses to start.
|
||
raise
|
||
except Exception as exc:
|
||
logger.debug(
|
||
"Compression feasibility check failed (non-fatal): %s", exc
|
||
)
|
||
|
||
|
||
def replay_compression_warning(agent: Any) -> None:
|
||
"""Re-send the stored compression warning through ``status_callback``.
|
||
|
||
Called once at the start of the first ``run_conversation()``, when the gateway
|
||
callback (absent during ``__init__``) is finally wired.
|
||
"""
|
||
msg = getattr(agent, "_compression_warning", None)
|
||
if msg and agent.status_callback:
|
||
try:
|
||
agent.status_callback("lifecycle", msg)
|
||
except Exception:
|
||
pass
|
||
|
||
|
||
def conversation_history_after_compression(
|
||
agent: Any,
|
||
messages: list,
|
||
previous_history: Optional[list] = None,
|
||
) -> Optional[list]:
|
||
"""Return the correct flush baseline after a compression boundary.
|
||
|
||
Session rotation returns ``None`` so the child gets the full compacted list.
|
||
In-place compaction returns a shallow copy of the already-persisted rows (else
|
||
the identity flush re-appends them). Aborted/no-op attempts keep the baseline:
|
||
marking all persisted drops unflushed turns; clearing re-appends rows.
|
||
"""
|
||
if bool(getattr(agent, "_last_compression_attempt_recorded", False)):
|
||
attempt_in_place = getattr(agent, "_last_compression_attempt_in_place", None)
|
||
if attempt_in_place is True:
|
||
return list(messages)
|
||
if attempt_in_place is False:
|
||
return None
|
||
return previous_history
|
||
if bool(getattr(agent, "_last_compaction_in_place", False)):
|
||
return list(messages)
|
||
return None
|
||
|
||
|
||
_SYNTHETIC_USER_PREFIXES = (
|
||
"[System: Your previous response was truncated",
|
||
"[System: The previous response was cut off",
|
||
"[System: Your previous tool call",
|
||
"[Your active task list was preserved across context compression]",
|
||
"[IMPORTANT: Background process ",
|
||
)
|
||
|
||
|
||
def _message_text(message: Any) -> str:
|
||
content = message.get("content") if isinstance(message, dict) else None
|
||
if isinstance(content, str):
|
||
return content
|
||
if isinstance(content, list):
|
||
return "\n".join(
|
||
str(part.get("text") or part.get("content") or "")
|
||
for part in content
|
||
if isinstance(part, dict)
|
||
)
|
||
return ""
|
||
|
||
|
||
_SYNTHETIC_USER_FLAGS = (
|
||
"_todo_snapshot_synthetic",
|
||
"_empty_recovery_synthetic",
|
||
"_verification_stop_synthetic",
|
||
"_pre_verify_synthetic",
|
||
"_dropped_toolcall_nudge",
|
||
)
|
||
|
||
|
||
def _is_real_user_message(message: Any) -> bool:
|
||
"""Distinguish human intent from user-role runtime scaffolding.
|
||
|
||
A compaction summary flipped to ``role="user"`` for alternation is scaffolding
|
||
and must not short-circuit anchor restoration.
|
||
"""
|
||
if not isinstance(message, dict) or message.get("role") != "user":
|
||
return False
|
||
if any(message.get(flag) for flag in _SYNTHETIC_USER_FLAGS):
|
||
return False
|
||
text = _message_text(message).strip()
|
||
if not text:
|
||
return False
|
||
if text.startswith(_SYNTHETIC_USER_PREFIXES):
|
||
return False
|
||
from agent.context_compressor import ContextCompressor
|
||
|
||
return not ContextCompressor._is_synthetic_compression_user_turn(message)
|
||
|
||
|
||
def _message_contains_busy_steer(message: Any) -> bool:
|
||
"""Return whether *message* carries a busy-steer marker.
|
||
|
||
Steer follow-ups live as markers inside ``role=tool`` results, so they carry
|
||
user intent that ``_is_real_user_message`` alone would miss.
|
||
"""
|
||
text = _message_text(message)
|
||
if not text:
|
||
return False
|
||
try:
|
||
from agent.prompt_builder import STEER_MARKER_CLOSE, STEER_MARKER_OPEN
|
||
|
||
return STEER_MARKER_OPEN in text and STEER_MARKER_CLOSE in text
|
||
except Exception:
|
||
return "[OUT-OF-BAND USER MESSAGE" in text and "[/OUT-OF-BAND USER MESSAGE]" in text
|
||
|
||
|
||
def _extract_steer_text_from_message(message: Any) -> Optional[str]:
|
||
"""Extract the inner user text from a steer marker, or None."""
|
||
text = _message_text(message)
|
||
if not text:
|
||
return None
|
||
try:
|
||
from agent.prompt_builder import STEER_MARKER_CLOSE, STEER_MARKER_OPEN
|
||
|
||
open_marker = STEER_MARKER_OPEN
|
||
close_marker = STEER_MARKER_CLOSE
|
||
except Exception:
|
||
open_marker = "[OUT-OF-BAND USER MESSAGE"
|
||
close_marker = "[/OUT-OF-BAND USER MESSAGE]"
|
||
start = text.find(open_marker)
|
||
if start == -1:
|
||
# Fallback: marker wording may evolve; look for the stable prefix.
|
||
fallback_open = "[OUT-OF-BAND USER MESSAGE"
|
||
start = text.find(fallback_open)
|
||
if start == -1:
|
||
return None
|
||
# Skip to end of the opening line.
|
||
nl = text.find("\n", start)
|
||
if nl != -1:
|
||
start = nl + 1
|
||
else:
|
||
start += len(fallback_open)
|
||
else:
|
||
start += len(open_marker)
|
||
end = text.find(close_marker, start)
|
||
if end == -1:
|
||
end = text.find("[/OUT-OF-BAND USER MESSAGE]", start)
|
||
if end == -1:
|
||
return None
|
||
extracted = text[start:end].strip()
|
||
return extracted if extracted else None
|
||
|
||
|
||
def _compressed_has_busy_steer(messages: list) -> bool:
|
||
"""Whether *messages* already carries a steer marker in a ``role=tool`` row.
|
||
|
||
Only tool rows count, so a summary merely quoting the marker text is not
|
||
mistaken for live intent.
|
||
"""
|
||
for msg in messages:
|
||
if not isinstance(msg, dict) or msg.get("role") != "tool":
|
||
continue
|
||
if _message_contains_busy_steer(msg):
|
||
return True
|
||
return False
|
||
|
||
|
||
def _strip_stale_todo_snapshot(content: Any) -> Any:
|
||
"""Remove a previously merged todo-snapshot block from message content.
|
||
|
||
Snapshots are appended to the trailing user turn, so a surviving header is
|
||
stale; stripping before re-injection prevents accumulation across boundaries.
|
||
"""
|
||
from tools.todo_tool import TODO_INJECTION_HEADER
|
||
|
||
if isinstance(content, str):
|
||
idx = content.find(TODO_INJECTION_HEADER)
|
||
if idx == -1:
|
||
return content
|
||
return content[:idx].rstrip()
|
||
if isinstance(content, list):
|
||
cleaned = []
|
||
for part in content:
|
||
if not isinstance(part, dict):
|
||
cleaned.append(part)
|
||
continue
|
||
if part.get("type") == "text":
|
||
text = str(part.get("text") or "")
|
||
idx = text.find(TODO_INJECTION_HEADER)
|
||
if idx != -1:
|
||
stripped = text[:idx].rstrip()
|
||
if stripped:
|
||
p = dict(part)
|
||
p["text"] = stripped
|
||
cleaned.append(p)
|
||
else:
|
||
cleaned.append(part)
|
||
else:
|
||
cleaned.append(part)
|
||
return cleaned
|
||
return content
|
||
|
||
|
||
def _todo_snapshot_is_only_content(content: Any, stripped: Any) -> bool:
|
||
"""Return whether stripping the snapshot leaves no structured content.
|
||
|
||
Text snapshots trail a string; structured ones occupy their own text part, so
|
||
only an empty remainder proves the row was scaffolding alone. Text extraction
|
||
is deliberately not used: image, audio and other non-text parts must survive.
|
||
"""
|
||
if isinstance(content, str) and isinstance(stripped, str):
|
||
return not stripped.strip()
|
||
if isinstance(content, list) and isinstance(stripped, list):
|
||
return not stripped
|
||
return False
|
||
|
||
|
||
def _replace_message_content(message: dict, content: Any) -> None:
|
||
"""Rewrite message content without allowing an old API sidecar to replay."""
|
||
from agent.turn_context import drop_stale_api_content
|
||
|
||
message["content"] = content
|
||
drop_stale_api_content(message)
|
||
|
||
|
||
# Compaction re-injects the todo list verbatim but prunes skills to markers, so
|
||
# couple them: tell the model to reload pruned skills BEFORE acting on tasks.
|
||
# Lives after TODO_INJECTION_HEADER so it strips with the snapshot next time.
|
||
_PRUNED_SKILL_RELOAD_NOTICE_HEADER = (
|
||
"[Skills pruned during compression — reload before acting on these tasks]"
|
||
)
|
||
|
||
|
||
def _pruned_skill_reload_notice(compressed: list) -> str:
|
||
"""Reload notice for skills whose bodies were pruned, or ``""``.
|
||
|
||
Scans ``[SKILL_PRUNED: ...]`` markers in the post-compression transcript;
|
||
first-seen order, deduplicated, capped at ``_MAX_PRUNED_SKILL_MARKERS``.
|
||
"""
|
||
from agent.context_compressor import (
|
||
_MAX_PRUNED_SKILL_MARKERS,
|
||
_extract_pruned_skill_names,
|
||
)
|
||
|
||
names: list = []
|
||
for message in compressed:
|
||
if not isinstance(message, dict):
|
||
continue
|
||
for name in _extract_pruned_skill_names(_message_text(message)):
|
||
if name not in names:
|
||
names.append(name)
|
||
del names[_MAX_PRUNED_SKILL_MARKERS:]
|
||
if not names:
|
||
return ""
|
||
calls = "; ".join(f"skill_view(name='{name}')" for name in names)
|
||
return (
|
||
f"{_PRUNED_SKILL_RELOAD_NOTICE_HEADER}\n"
|
||
"The task list above crossed the compression boundary verbatim, but "
|
||
"the skill instructions that governed it were pruned. Before "
|
||
f"executing any preserved task that depends on these skills, reload "
|
||
f"them first: {calls}. After reloading, re-check that each pending "
|
||
"task is still justified — findings recorded before the boundary may "
|
||
"have invalidated it."
|
||
)
|
||
|
||
|
||
def _merge_anchor_into_user_message(target: dict, anchor: dict) -> None:
|
||
"""Fold the human anchor into an existing user-role scaffolding turn.
|
||
|
||
Used only when any insertion would create consecutive user turns. Anchor text
|
||
leads, scaffolding follows, and synthetic flags are cleared.
|
||
"""
|
||
anchor_content = anchor.get("content")
|
||
target_content = target.get("content")
|
||
if isinstance(anchor_content, list) or isinstance(target_content, list):
|
||
anchor_parts = (
|
||
list(anchor_content)
|
||
if isinstance(anchor_content, list)
|
||
else [{"type": "text", "text": str(anchor_content or "")}]
|
||
)
|
||
target_parts = (
|
||
list(target_content)
|
||
if isinstance(target_content, list)
|
||
else [{"type": "text", "text": str(target_content or "")}]
|
||
)
|
||
_replace_message_content(target, anchor_parts + target_parts)
|
||
else:
|
||
merged = f"{anchor_content or ''}\n\n{target_content or ''}".strip()
|
||
_replace_message_content(target, merged)
|
||
for flag in _SYNTHETIC_USER_FLAGS:
|
||
target.pop(flag, None)
|
||
|
||
|
||
CompressedUserTurnOutcome = Literal[
|
||
"inserted",
|
||
"merged",
|
||
"already_present",
|
||
"placeholder_appended",
|
||
]
|
||
|
||
|
||
def _insert_real_user_anchor(messages: list, anchor: dict) -> CompressedUserTurnOutcome:
|
||
"""Insert the latest human turn without breaking role alternation."""
|
||
from agent.context_compressor import _DB_PERSISTED_MARKER
|
||
|
||
def _role(msg: Any) -> Optional[str]:
|
||
return msg.get("role") if isinstance(msg, dict) else None
|
||
|
||
# Preferred anchor: the summary boundary — first assistant message not preceded
|
||
# by a user turn. Left neighbour is then non-user, right is an assistant.
|
||
for index, message in enumerate(messages):
|
||
if _role(message) != "assistant":
|
||
continue
|
||
previous_role = _role(messages[index - 1]) if index > 0 else None
|
||
if previous_role != "user":
|
||
anchor[_DB_PERSISTED_MARKER] = True
|
||
messages.insert(index, anchor)
|
||
return "inserted"
|
||
# Every assistant is user-preceded (or there are none). Appending is
|
||
# safe whenever the transcript does not already end with a user turn.
|
||
if not messages or _role(messages[-1]) != "user":
|
||
anchor[_DB_PERSISTED_MARKER] = True
|
||
messages.append(anchor)
|
||
return "inserted"
|
||
# The transcript ends with a user-role message and no slot avoids
|
||
# user/user adjacency.
|
||
from agent.context_compressor import ContextCompressor
|
||
|
||
if ContextCompressor._is_context_summary_content(
|
||
_message_text(messages[-1])
|
||
):
|
||
# Never merge into a summary: its prefix must stay at message start for summary
|
||
# detection; repair_message_sequence merges adjacent user turns summary-first.
|
||
anchor[_DB_PERSISTED_MARKER] = True
|
||
messages.append(anchor)
|
||
return "inserted"
|
||
# Trailing user-role scaffolding (e.g. the todo snapshot): merge instead
|
||
# of inserting a consecutive same-role message (#55677 strict templates).
|
||
_merge_anchor_into_user_message(messages[-1], anchor)
|
||
messages[-1][_DB_PERSISTED_MARKER] = True
|
||
return "merged"
|
||
|
||
|
||
def _ensure_compressed_has_user_turn(
|
||
original_messages: list, compressed: list
|
||
) -> CompressedUserTurnOutcome:
|
||
"""Preserve human intent, not merely a synthetic user-role placeholder."""
|
||
if any(_is_real_user_message(message) for message in compressed):
|
||
return "already_present"
|
||
if _compressed_has_busy_steer(compressed):
|
||
return "already_present"
|
||
from agent.context_compressor import (
|
||
COMPRESSION_CONTINUATION_USER_CONTENT,
|
||
_fresh_compaction_message_copy,
|
||
)
|
||
|
||
# One reversed scan over BOTH kinds: scanning steer then user would let an older
|
||
# consumed steer outrank a newer real user request and replay it.
|
||
for message in reversed(original_messages):
|
||
if _is_real_user_message(message):
|
||
return _insert_real_user_anchor(
|
||
compressed,
|
||
_fresh_compaction_message_copy(message),
|
||
)
|
||
if not isinstance(message, dict) or message.get("role") != "tool":
|
||
continue
|
||
steer_text = _extract_steer_text_from_message(message)
|
||
if steer_text:
|
||
return _insert_real_user_anchor(
|
||
compressed,
|
||
{"role": "user", "content": steer_text},
|
||
)
|
||
from agent.message_metadata import append_message
|
||
|
||
append_message(
|
||
compressed,
|
||
{
|
||
"role": "user",
|
||
"content": COMPRESSION_CONTINUATION_USER_CONTENT,
|
||
},
|
||
)
|
||
return "placeholder_appended"
|
||
|
||
|
||
def _messages_match_scoped_identity(left: Any, right: Any) -> bool:
|
||
"""Compare the live turn identity we care about for rotation stamping."""
|
||
if not isinstance(left, dict) or not isinstance(right, dict):
|
||
return False
|
||
if left.get("role") != right.get("role"):
|
||
return False
|
||
if left.get("content") != right.get("content"):
|
||
return False
|
||
left_timestamp = left.get("timestamp")
|
||
right_timestamp = right.get("timestamp")
|
||
if left_timestamp is not None and right_timestamp is not None:
|
||
return left_timestamp == right_timestamp
|
||
return True
|
||
|
||
|
||
_PENDING_CONTEXT_ENGINE_NOTIFICATION = (
|
||
"_pending_context_engine_compression_notification"
|
||
)
|
||
|
||
|
||
def _notify_context_engine_compression_complete(
|
||
agent: Any,
|
||
*,
|
||
new_session_id: str,
|
||
old_session_id: str,
|
||
) -> bool:
|
||
"""Notify the active context engine after a durable compression commit."""
|
||
# Opt-in relay session-span segmentation. Observer semantics — failure must
|
||
# never undo or delay the committed compression.
|
||
try:
|
||
from agent import relay_runtime
|
||
|
||
relay_runtime.SESSION_COORDINATOR.notify_session_compacted(
|
||
profile_key=relay_runtime.current_profile_key(),
|
||
session_id=new_session_id,
|
||
old_session_id=old_session_id,
|
||
)
|
||
except Exception:
|
||
logger.debug("relay segment rotation notification failed", exc_info=True)
|
||
callback = getattr(agent.context_compressor, "on_session_start", None)
|
||
if not callable(callback):
|
||
return False
|
||
try:
|
||
callback(
|
||
new_session_id,
|
||
boundary_reason="compression",
|
||
old_session_id=old_session_id,
|
||
platform=getattr(agent, "platform", None) or "cli",
|
||
conversation_id=getattr(agent, "_gateway_session_key", None),
|
||
)
|
||
except Exception:
|
||
# Context-engine hooks are observers. A callback failure must not undo
|
||
# history that the core or an outer host transaction already committed.
|
||
logger.debug(
|
||
"context engine on_session_start (compression) failed",
|
||
exc_info=True,
|
||
)
|
||
return False
|
||
return True
|
||
|
||
|
||
def _queue_context_engine_compression_notification(
|
||
agent: Any,
|
||
*,
|
||
new_session_id: str,
|
||
old_session_id: str,
|
||
) -> None:
|
||
"""Stage exactly one existing hook call for an outer host transaction."""
|
||
if callable(getattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, None)):
|
||
raise RuntimeError("a compression notification is already pending")
|
||
|
||
def _notify() -> bool:
|
||
return _notify_context_engine_compression_complete(
|
||
agent,
|
||
new_session_id=new_session_id,
|
||
old_session_id=old_session_id,
|
||
)
|
||
|
||
setattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, _notify)
|
||
|
||
|
||
def finalize_context_engine_compression_notification(
|
||
agent: Any,
|
||
*,
|
||
committed: bool,
|
||
) -> bool:
|
||
"""Emit or discard a deferred notification; repeated calls are no-ops."""
|
||
pending = getattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, None)
|
||
setattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, None)
|
||
if not committed or not callable(pending):
|
||
return False
|
||
return bool(pending())
|
||
|
||
|
||
class _CompactionLifecycle:
|
||
"""Owns the one-shot terminal edge of the compaction status lifecycle.
|
||
|
||
``commit_status`` is rebound to "committed" only on success and read at
|
||
``complete()`` time, so abort paths keep the terminal edge suppressed.
|
||
"""
|
||
|
||
def __init__(self, agent: Any, status_emitted: bool) -> None:
|
||
self._agent = agent
|
||
self._status_emitted = status_emitted
|
||
self._done_emitted = False
|
||
self.commit_status = "aborted"
|
||
|
||
def complete(self, *, force_terminal: bool = False) -> None:
|
||
if self._done_emitted:
|
||
return
|
||
self._done_emitted = True
|
||
# Suppressed start → no terminal edge. Non-compacting aborts (lock contender,
|
||
# cancelled fence) opt in via force_terminal so clients can retire their phase.
|
||
# Failure warnings go through _emit_warning and are never suppressed here.
|
||
if self._status_emitted and (
|
||
self.commit_status == "committed" or force_terminal
|
||
):
|
||
_emit_compaction_done(self._agent)
|
||
|
||
|
||
class _CompressionLease:
|
||
"""The per-attempt durable compression lock plus its lifecycle plumbing.
|
||
|
||
``holder`` is None when no durable lock is owned (legacy DB, no session db);
|
||
``watermark`` is MAX(id) of active rows at lease start (None = archive
|
||
everything, no concurrent-tail preservation this cycle).
|
||
"""
|
||
|
||
def __init__(
|
||
self,
|
||
agent: Any,
|
||
*,
|
||
db: Any,
|
||
sid: str,
|
||
ttl: float,
|
||
refresh_interval: Any,
|
||
commit_fence: Optional[CompressionCommitFence],
|
||
lifecycle: _CompactionLifecycle,
|
||
) -> None:
|
||
self._agent = agent
|
||
self.db = db
|
||
self.sid = sid
|
||
self.ttl = ttl
|
||
self._refresh_interval = refresh_interval
|
||
self._commit_fence = commit_fence
|
||
self._lifecycle = lifecycle
|
||
self.holder: Optional[str] = None
|
||
self.watermark: Optional[int] = None
|
||
self._refresher: Optional[_CompressionLockLeaseRefresher] = None
|
||
self._released = False
|
||
self._release_guard = threading.Lock()
|
||
# Fence lock acquisition + release-hook publication together so a host timeout
|
||
# cannot win between acquiring the lock and having a way to release it.
|
||
self._lock_setup_entered = False
|
||
|
||
def begin_lock_setup(self) -> bool:
|
||
if self._commit_fence is None:
|
||
return True
|
||
self._lock_setup_entered = self._commit_fence.begin_lock_setup()
|
||
return self._lock_setup_entered
|
||
|
||
def finish_lock_setup(self) -> None:
|
||
if not self._lock_setup_entered or self._commit_fence is None:
|
||
return
|
||
self._lock_setup_entered = False
|
||
self._commit_fence.finish_lock_setup()
|
||
|
||
def start_refresher(self) -> None:
|
||
if self.holder is None:
|
||
return
|
||
candidate = _CompressionLockLeaseRefresher(
|
||
self.db, self.sid, self.holder, self.ttl, self._refresh_interval
|
||
)
|
||
# Cancellation may release the holder between hook publication and this
|
||
# start; serialize with the release path so no refresher starts on a freed lock.
|
||
with self._release_guard:
|
||
if not self._released:
|
||
self._refresher = candidate
|
||
self._refresher.start()
|
||
|
||
def release_holder_only(self) -> None:
|
||
"""Stop this holder's refresher and release only its durable lock.
|
||
|
||
Holder-qualified and idempotent: safe for the host after a timeout because a
|
||
newer holder's lease can never be deleted by this stale release.
|
||
"""
|
||
with self._release_guard:
|
||
if self._released:
|
||
return
|
||
self._released = True
|
||
if getattr(self._agent, "_active_compression_lock_holder", None) == self.holder:
|
||
self._agent._active_compression_lock_holder = None
|
||
if self._refresher is not None:
|
||
try:
|
||
self._refresher.stop()
|
||
except Exception as _stop_err:
|
||
logger.debug("compression lock refresher stop failed: %s", _stop_err)
|
||
if self.db is not None and self.sid and self.holder:
|
||
try:
|
||
self.db.release_compression_lock(self.sid, self.holder)
|
||
except Exception as _rel_err:
|
||
logger.debug("compression lock release failed: %s", _rel_err)
|
||
|
||
def release(self) -> None:
|
||
"""Finish lifecycle cleanup and release the OLD session lock once."""
|
||
try:
|
||
self._lifecycle.complete()
|
||
finally:
|
||
try:
|
||
self.release_holder_only()
|
||
finally:
|
||
try:
|
||
if self._commit_fence is not None:
|
||
self._commit_fence.clear_cancelled_lock_release(
|
||
self.release_holder_only
|
||
)
|
||
finally:
|
||
self.finish_lock_setup()
|
||
|
||
|
||
def _acquire_compression_lease(
|
||
agent: Any,
|
||
*,
|
||
commit_fence: Optional[CompressionCommitFence],
|
||
lifecycle: _CompactionLifecycle,
|
||
system_message: str,
|
||
approx_tokens: Optional[int],
|
||
attempt_started_at: float,
|
||
) -> Tuple[Optional[_CompressionLease], Optional[str]]:
|
||
"""Take the per-session compression lock; ``(None, prompt)`` means sit out.
|
||
|
||
Two AIAgents sharing a session_id (e.g. background review fork) would both
|
||
rotate and orphan a child. Keyed on the OLD id (what rivals read from
|
||
SessionEntry). Loser sits out: messages unchanged, caller sees no-op.
|
||
Only structural absence of the lock API (version skew) fails open; once
|
||
resolved, any exception fails closed since unlocked runs can fork lineage.
|
||
"""
|
||
_lock_db = getattr(agent, "_session_db", None)
|
||
_lock_sid = agent.session_id or ""
|
||
_try_acquire_lock = None
|
||
_lock_lookup_error: Optional[Exception] = None
|
||
_legacy_session_db_without_lock_api = False
|
||
# Clear stale lock-skip so this call's outcome alone is visible; else a manual
|
||
# /compress after an auto lock-skip falsely reports "already in progress".
|
||
agent._compression_skipped_due_to_lock = None
|
||
if _lock_db is not None:
|
||
try:
|
||
_legacy_session_db_without_lock_api = _lock_api_is_absent_on_session_db(
|
||
_lock_db
|
||
)
|
||
except Exception as exc:
|
||
_lock_lookup_error = exc
|
||
if _lock_lookup_error is None and not _legacy_session_db_without_lock_api:
|
||
try:
|
||
_try_acquire_lock = _lock_db.try_acquire_compression_lock
|
||
if not callable(_try_acquire_lock):
|
||
_lock_lookup_error = TypeError(
|
||
"compression lock API is present but not callable"
|
||
)
|
||
except Exception as exc:
|
||
_lock_lookup_error = exc
|
||
try:
|
||
_lock_ttl = float(getattr(agent, "_compression_lock_ttl_seconds", 300.0) or 300.0)
|
||
except (TypeError, ValueError):
|
||
_lock_ttl = 300.0
|
||
lease = _CompressionLease(
|
||
agent,
|
||
db=_lock_db,
|
||
sid=_lock_sid,
|
||
ttl=_lock_ttl,
|
||
refresh_interval=getattr(agent, "_compression_lock_refresh_interval", None),
|
||
commit_fence=commit_fence,
|
||
lifecycle=lifecycle,
|
||
)
|
||
|
||
if _lock_db is not None and _lock_sid:
|
||
lease.holder = _compression_lock_holder(agent)
|
||
if _lock_lookup_error is not None:
|
||
# Attribute lookup itself failed for a reason other than a missing
|
||
# lock API. It is unsafe to proceed without a lock in that case.
|
||
lease.holder = None
|
||
logger.warning(
|
||
"compression lock lookup raised unexpectedly for session=%s "
|
||
"(%s: %s) — skipping compression this cycle",
|
||
_lock_sid, type(_lock_lookup_error).__name__, _lock_lookup_error,
|
||
)
|
||
_lock_acquired = False
|
||
elif _try_acquire_lock is None:
|
||
# Lock API absent on this instance: log once, proceed unlocked so version skew
|
||
# cannot stall the outer auto-compression loop forever.
|
||
lease.holder = None
|
||
if getattr(agent, "_last_compression_lock_error_sid", None) != _lock_sid:
|
||
agent._last_compression_lock_error_sid = _lock_sid
|
||
logger.warning(
|
||
"compression lock subsystem unavailable for session=%s "
|
||
"— proceeding without lock. This usually means a stale "
|
||
"in-memory module after an update; restart the process "
|
||
"(or `hermes update`) to resync.",
|
||
_lock_sid,
|
||
)
|
||
_lock_acquired = True # acquired-but-unlocked compatibility path
|
||
else:
|
||
if not lease.begin_lock_setup():
|
||
logger.info(
|
||
"Compression commit cancelled before lock acquisition "
|
||
"(session=%s).",
|
||
agent.session_id or "none",
|
||
)
|
||
agent._last_compaction_in_place = False
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "commit_fence_cancelled")
|
||
lifecycle.complete(force_terminal=True)
|
||
return None, _existing_sp
|
||
try:
|
||
_lock_acquired = _try_acquire_lock(
|
||
_lock_sid, lease.holder, ttl_seconds=_lock_ttl
|
||
)
|
||
if _lock_acquired:
|
||
# Watermark = MAX(id) of active rows at START. Appends aren't blocked during
|
||
# summary; later rows are concurrent tail that archive_and_compact re-sequences.
|
||
try:
|
||
lease.watermark = _lock_db.get_active_message_watermark(
|
||
_lock_sid
|
||
)
|
||
# A captured watermark makes the commit safe against later rows on BOTH commit
|
||
# paths; tell the fence so a host may keep this attempt's admission.
|
||
if commit_fence is not None:
|
||
try:
|
||
commit_fence.mark_commit_watermark_fenced()
|
||
except AttributeError:
|
||
pass # test doubles without the method
|
||
except Exception as _wm_err:
|
||
# Watermark capture is safety-additive (fallback archives everything), so
|
||
# failure here must not abort compression.
|
||
logger.warning(
|
||
"compression watermark capture failed for "
|
||
"session=%s (%s) — concurrent appends this cycle "
|
||
"will be archived with the snapshot",
|
||
_lock_sid, _wm_err,
|
||
)
|
||
lease.watermark = None
|
||
except Exception as _lock_err:
|
||
# Method entered but failed: not version skew, fail closed. Acquire may have
|
||
# committed, so release holder-qualified best-effort (safe if never acquired).
|
||
try:
|
||
_lock_db.release_compression_lock(_lock_sid, lease.holder)
|
||
except Exception as _release_err:
|
||
logger.debug(
|
||
"compression lock cleanup after failed acquire failed: %s",
|
||
_release_err,
|
||
)
|
||
lease.holder = None
|
||
logger.warning(
|
||
"compression lock acquisition raised unexpectedly for "
|
||
"session=%s (%s: %s) — skipping compression this cycle",
|
||
_lock_sid, type(_lock_err).__name__, _lock_err,
|
||
)
|
||
_lock_acquired = False
|
||
if not _lock_acquired:
|
||
lease.finish_lock_setup()
|
||
try:
|
||
existing = _lock_db.get_compression_lock_holder(_lock_sid)
|
||
except Exception:
|
||
existing = None
|
||
logger.warning(
|
||
"compression skipped: another path is compressing session=%s "
|
||
"(holder=%s) — returning messages unchanged to avoid session fork",
|
||
_lock_sid, existing,
|
||
)
|
||
lease.holder = None # don't release a lock we don't own
|
||
# Distinguish lock-contention no-op from "nothing to compress" so manual
|
||
# /compress can show a clear status instead of "No changes".
|
||
agent._compression_skipped_due_to_lock = existing or True
|
||
# Surface to the user once — quiet for downstream auto-compress loops
|
||
if getattr(agent, "_last_compression_lock_warning_sid", None) != _lock_sid:
|
||
agent._last_compression_lock_warning_sid = _lock_sid
|
||
try:
|
||
agent._emit_warning(
|
||
"⚠ Skipping concurrent compression — another path "
|
||
"is already compressing this session. Will retry "
|
||
"after it finishes."
|
||
)
|
||
except Exception:
|
||
pass
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
try:
|
||
if hasattr(agent.context_compressor, "_begin_compression_telemetry"):
|
||
agent.context_compressor._begin_compression_telemetry(current_tokens=approx_tokens)
|
||
except Exception:
|
||
pass
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "lock_contended")
|
||
lifecycle.complete(force_terminal=True)
|
||
return None, _existing_sp
|
||
|
||
if lease.holder is not None:
|
||
agent._active_compression_lock_holder = lease.holder
|
||
if (
|
||
commit_fence is not None
|
||
and commit_fence.register_cancelled_lock_release(
|
||
lease.release_holder_only
|
||
)
|
||
):
|
||
# Cancellation won during lock setup (hook ran synchronously, lease gone):
|
||
# abort before any summary work.
|
||
logger.info(
|
||
"Compression commit cancelled before summary dispatch "
|
||
"(session=%s).",
|
||
agent.session_id or "none",
|
||
)
|
||
agent._last_compaction_in_place = False
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "commit_fence_cancelled")
|
||
lease.release()
|
||
return None, _existing_sp
|
||
return lease, None
|
||
|
||
|
||
def _adopt_if_parent_rotated(
|
||
agent: Any, lease: _CompressionLease, messages: list, system_message: str
|
||
) -> Optional[Tuple[list, str]]:
|
||
"""Sit out (or adopt the live child) when the parent was already rotated.
|
||
|
||
A late contender can take the parent lock after the winner released it and
|
||
rotated; holding the lock does not prove this agent still owns a live parent.
|
||
Returns the ``compress_context`` result to hand back, or None to proceed.
|
||
"""
|
||
if lease.db is None or not lease.sid:
|
||
return None
|
||
try:
|
||
_parent_already_rotated = _session_was_rotated_by_compression(
|
||
lease.db, lease.sid
|
||
)
|
||
except Exception as _session_err:
|
||
logger.warning(
|
||
"compression session ownership lookup failed for session=%s "
|
||
"(%s: %s) - skipping compression this cycle",
|
||
lease.sid,
|
||
type(_session_err).__name__,
|
||
_session_err,
|
||
)
|
||
lease.release()
|
||
return messages, _existing_system_prompt(agent, system_message)
|
||
if not _parent_already_rotated:
|
||
return None
|
||
recovered_messages = _adopt_live_compression_child(agent, lease.db, lease.sid)
|
||
lease.release()
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
if recovered_messages is not None:
|
||
logger.warning(
|
||
"compression recovery: stale session=%s adopted live child=%s",
|
||
lease.sid,
|
||
agent.session_id,
|
||
)
|
||
return recovered_messages, _existing_sp
|
||
logger.warning(
|
||
"compression skipped: session=%s was already rotated by "
|
||
"another compression path, but no unique live child could be adopted",
|
||
lease.sid,
|
||
)
|
||
return messages, _existing_sp
|
||
|
||
|
||
def _adopt_grown_durable_parent(
|
||
agent: Any, lease: _CompressionLease, messages: list
|
||
) -> Optional[list]:
|
||
"""Return the durable parent transcript when it outgrew the in-memory snapshot.
|
||
|
||
Rotation only (in-place never loses rows). The snapshot predates the lease: if
|
||
durable grew, a writer committed a turn — ADOPT it (aborting wedged busy
|
||
sessions forever). Length check only: in-memory edits of past turns are legal.
|
||
"""
|
||
if lease.db is None or not lease.sid:
|
||
return None
|
||
durable_loader = getattr(type(lease.db), "get_messages_as_conversation", None)
|
||
if not callable(durable_loader):
|
||
return None
|
||
durable_parent = durable_loader(lease.db, lease.sid)
|
||
if not (isinstance(durable_parent, list) and len(durable_parent) > len(messages)):
|
||
return None
|
||
# In-memory carries this turn's un-persisted user tail; flush it via the normal
|
||
# rotation-boundary path before adopting, else skip adoption (would drop input).
|
||
_preflush_idx = getattr(agent, "_persist_user_message_idx", None)
|
||
_preflush_ok = False
|
||
if isinstance(_preflush_idx, int) and 0 <= _preflush_idx < len(messages):
|
||
try:
|
||
_preflush_ok = agent._flush_messages_to_session_db(
|
||
messages,
|
||
conversation_history=messages[:_preflush_idx],
|
||
)
|
||
except Exception:
|
||
_preflush_ok = False
|
||
else:
|
||
# No un-persisted tail: transcript is fully durable, so adopting the longer
|
||
# parent cannot drop live input — adopt directly.
|
||
_preflush_ok = True
|
||
if not _preflush_ok:
|
||
logger.warning(
|
||
"compression: session=%s grew before lease "
|
||
"(%d → %d msgs) but the pre-adoption flush of the "
|
||
"live tail failed; skipping durable-snapshot "
|
||
"adoption so un-persisted user input is kept",
|
||
lease.sid,
|
||
len(messages),
|
||
len(durable_parent),
|
||
)
|
||
return None
|
||
# Re-read after the flush so the adopted snapshot carries the just-persisted tail.
|
||
durable_parent = durable_loader(lease.db, lease.sid)
|
||
if not (isinstance(durable_parent, list) and len(durable_parent) > len(messages)):
|
||
return None
|
||
logger.info(
|
||
"compression: session=%s grew before lease "
|
||
"(%d → %d msgs); adopting durable snapshot",
|
||
lease.sid,
|
||
len(messages),
|
||
len(durable_parent),
|
||
)
|
||
return durable_parent
|
||
|
||
|
||
def _pre_compress_memory_context(
|
||
agent: Any, messages: list, checkpoint_required: bool
|
||
) -> str:
|
||
"""Provider ``on_pre_compress()`` insights to surface in the summary ("" if none).
|
||
|
||
Raw messages stay the API v1 provider contract; normalized evidence goes only
|
||
to API v2+ checkpoint providers inside MemoryManager.on_pre_compress().
|
||
Raises :class:`CompressionCheckpointUnavailable` when a required checkpoint
|
||
cannot be taken.
|
||
"""
|
||
memory_context = ""
|
||
memory_manager = getattr(agent, "_memory_manager", None)
|
||
evidence_messages = _direct_messages_for_pre_compress_memory(messages)
|
||
if checkpoint_required:
|
||
supports_checkpoint = getattr(
|
||
memory_manager, "supports_pre_compress_checkpoint", None
|
||
)
|
||
if memory_manager is None or not callable(supports_checkpoint):
|
||
raise _checkpoint_blocked(
|
||
f"no active provider implements checkpoint API "
|
||
f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}"
|
||
)
|
||
try:
|
||
compatible = bool(
|
||
supports_checkpoint(PRE_COMPRESS_CHECKPOINT_API_VERSION)
|
||
)
|
||
except Exception as exc:
|
||
raise _checkpoint_blocked("provider capability probe failed") from exc
|
||
if not compatible:
|
||
raise _checkpoint_blocked(
|
||
f"active provider does not implement checkpoint API "
|
||
f"v{PRE_COMPRESS_CHECKPOINT_API_VERSION}"
|
||
)
|
||
try:
|
||
_maybe_ctx = memory_manager.on_pre_compress(
|
||
messages,
|
||
evidence_messages=evidence_messages,
|
||
require_checkpoint=True,
|
||
checkpoint_api_version=PRE_COMPRESS_CHECKPOINT_API_VERSION,
|
||
)
|
||
except Exception as exc:
|
||
logger.warning(
|
||
"Required pre-compress checkpoint failed (%s)",
|
||
type(exc).__name__,
|
||
)
|
||
raise _checkpoint_blocked(
|
||
f"provider checkpoint API v{PRE_COMPRESS_CHECKPOINT_API_VERSION} failed"
|
||
) from exc
|
||
if isinstance(_maybe_ctx, str):
|
||
memory_context = sanitize_memory_context(_maybe_ctx)
|
||
elif memory_manager:
|
||
try:
|
||
_maybe_ctx = memory_manager.on_pre_compress(
|
||
messages, evidence_messages=evidence_messages
|
||
)
|
||
if isinstance(_maybe_ctx, str):
|
||
memory_context = sanitize_memory_context(_maybe_ctx)
|
||
except Exception:
|
||
pass
|
||
return memory_context
|
||
|
||
|
||
def _resolve_compress_call(
|
||
agent: Any,
|
||
*,
|
||
approx_tokens: Optional[int],
|
||
focus_topic: Optional[str],
|
||
force: bool,
|
||
memory_context: str,
|
||
bypass_cooldown: bool,
|
||
) -> Tuple[Callable[..., Any], dict[str, Any]]:
|
||
"""Bind ``compress()`` and only the kwargs its signature accepts."""
|
||
compress_fn = agent.context_compressor.compress
|
||
compress_kwargs = _supported_compression_kwargs(
|
||
compress_fn,
|
||
current_tokens=approx_tokens,
|
||
focus_topic=focus_topic,
|
||
force=force,
|
||
memory_context=memory_context,
|
||
bypass_cooldown=bypass_cooldown,
|
||
)
|
||
if memory_context.strip() and "memory_context" not in compress_kwargs:
|
||
engine_name = getattr(
|
||
agent.context_compressor,
|
||
"name",
|
||
type(agent.context_compressor).__name__,
|
||
)
|
||
if (
|
||
getattr(agent, "_last_memory_context_unsupported_engine", None)
|
||
!= engine_name
|
||
):
|
||
agent._last_memory_context_unsupported_engine = engine_name
|
||
logger.warning(
|
||
"context engine %s does not accept memory_context; continuing "
|
||
"without provider-supplied summary context",
|
||
engine_name,
|
||
)
|
||
return compress_fn, compress_kwargs
|
||
|
||
|
||
def _run_summary_dispatch(
|
||
agent: Any,
|
||
messages: list,
|
||
compress_fn: Callable[..., Any],
|
||
compress_kwargs: dict[str, Any],
|
||
*,
|
||
commit_fence: Optional[CompressionCommitFence],
|
||
attempt_generation: Any,
|
||
hard_cancel_event: Any,
|
||
) -> list:
|
||
"""Run the compressor under the fence's progress hook, deadline and interrupt guard."""
|
||
# Publish progress to the commit fence so hosts extend deadlines while tokens
|
||
# flow. Any active hook (even no-op) selects the streamed path: the timeout is
|
||
# inactivity-based and a byte-trickling provider hits the stream total ceiling.
|
||
from agent.auxiliary_client import (
|
||
aux_interrupt_protection,
|
||
aux_progress_hook,
|
||
aux_stream_deadline,
|
||
)
|
||
_progress_hook = (
|
||
commit_fence.touch_progress if commit_fence is not None
|
||
else (lambda: None)
|
||
)
|
||
# Return leg: cancel frees the owner but the provider daemon streams on to its
|
||
# own larger ceiling; share the host deadline so orphan streams stop with it.
|
||
_host_stream_deadline = (
|
||
commit_fence.deadline_monotonic if commit_fence is not None else None
|
||
)
|
||
# A LATE successful summary must not undo the host's timeout cooldown: the
|
||
# compressor checks cancellation before clearing; removed in finally (no leak).
|
||
if commit_fence is not None:
|
||
_install_compression_cancelled_check(
|
||
agent.context_compressor,
|
||
lambda: commit_fence.is_cancelled,
|
||
attempt_generation,
|
||
)
|
||
|
||
def _compression_cancel_requested() -> bool:
|
||
return bool(
|
||
(
|
||
hard_cancel_event is not None
|
||
and hard_cancel_event.is_set()
|
||
)
|
||
or (
|
||
commit_fence is not None
|
||
and commit_fence.is_cancelled
|
||
)
|
||
)
|
||
|
||
try:
|
||
# F6: never start expensive summary work for an already-cancelled
|
||
# fence (a stale queued job admitted after host departure).
|
||
if commit_fence is not None and commit_fence.is_cancelled:
|
||
logger.info(
|
||
"Compression cancelled before summary dispatch "
|
||
"(session=%s) — skipping summary work.",
|
||
agent.session_id or "none",
|
||
)
|
||
compressed = messages
|
||
else:
|
||
with aux_progress_hook(_progress_hook), aux_stream_deadline(
|
||
_host_stream_deadline
|
||
), aux_interrupt_protection(
|
||
cancel_check=_compression_cancel_requested
|
||
):
|
||
compressed = compress_fn(messages, **compress_kwargs)
|
||
# Freeze a hard stop that arrived after the last provider attempt but before
|
||
# session state rotates.
|
||
if (
|
||
hard_cancel_event is not None
|
||
and hard_cancel_event.is_set()
|
||
):
|
||
raise AuxiliaryExplicitCancellation()
|
||
finally:
|
||
if commit_fence is not None:
|
||
_clear_compression_cancelled_check_if_owner(
|
||
agent.context_compressor, attempt_generation
|
||
)
|
||
return compressed
|
||
|
||
|
||
def _fold_todo_snapshot(agent: Any, compressed: list) -> None:
|
||
"""Strip stale todo snapshots from ``compressed`` and fold the live one in (in place)."""
|
||
todo_snapshot = agent._todo_store.format_for_injection()
|
||
# Non-empty store (even all done) is authoritative: drop the old snapshot. A
|
||
# truly empty store may be un-rehydrated post-compaction: keep the snapshot.
|
||
_todo_has_items = getattr(agent._todo_store, "has_items", None)
|
||
try:
|
||
_todo_store_is_authoritative = bool(
|
||
_todo_has_items()
|
||
) if callable(_todo_has_items) else False
|
||
except Exception:
|
||
# Store may implement only format_for_injection(); unknown authority must
|
||
# preserve the pending snapshot rather than risk deleting it.
|
||
_todo_store_is_authoritative = False
|
||
if _todo_store_is_authoritative:
|
||
for _todo_idx in range(len(compressed) - 1, -1, -1):
|
||
_todo_message = compressed[_todo_idx]
|
||
if not isinstance(_todo_message, dict) or _todo_message.get("role") != "user":
|
||
continue
|
||
_todo_content = _todo_message.get("content")
|
||
_todo_stripped = _strip_stale_todo_snapshot(_todo_content)
|
||
if _todo_stripped == _todo_content:
|
||
continue
|
||
if (
|
||
_todo_message.get("_todo_snapshot_synthetic")
|
||
and _todo_snapshot_is_only_content(
|
||
_todo_content, _todo_stripped
|
||
)
|
||
):
|
||
compressed.pop(_todo_idx)
|
||
if _todo_idx < len(compressed):
|
||
# A standalone snapshot can drift from the tail; deleting it may expose two
|
||
# assistant rows, so use the normal replay repair to keep metadata consistent.
|
||
agent._repair_message_sequence(compressed)
|
||
else:
|
||
_replace_message_content(_todo_message, _todo_stripped)
|
||
# No longer todo-only scaffolding; other synthetic flags stay authoritative and
|
||
# _is_real_user_message() recomputes provenance from content + flags.
|
||
_todo_message.pop("_todo_snapshot_synthetic", None)
|
||
break
|
||
if todo_snapshot:
|
||
# If this boundary pruned skill bodies, the policy behind the todos is gone:
|
||
# add a reload notice after TODO_INJECTION_HEADER so both strip together.
|
||
_reload_notice = _pruned_skill_reload_notice(compressed)
|
||
if _reload_notice:
|
||
todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}"
|
||
# Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair);
|
||
# strip old snapshots first. Scaffolding tails must not absorb it (provenance).
|
||
from agent.context_compressor import _append_text_to_content
|
||
|
||
merged = False
|
||
_tail = (
|
||
compressed[-1]
|
||
if compressed and isinstance(compressed[-1], dict)
|
||
else None
|
||
)
|
||
if _tail is not None and _tail.get("role") == "user":
|
||
_stripped = _strip_stale_todo_snapshot(_tail.get("content"))
|
||
_probe = {
|
||
key: value for key, value in _tail.items() if key != "content"
|
||
}
|
||
_probe["content"] = _stripped
|
||
if _is_real_user_message(_probe):
|
||
_snapshot_text = (
|
||
f"\n\n{todo_snapshot}"
|
||
if isinstance(_stripped, str) and _stripped
|
||
else todo_snapshot
|
||
)
|
||
_replace_message_content(
|
||
_tail,
|
||
_append_text_to_content(_stripped, _snapshot_text),
|
||
)
|
||
merged = True
|
||
elif _stripped != _tail.get("content") and not _message_text(
|
||
{"role": "user", "content": _stripped}
|
||
).strip():
|
||
# The tail was nothing but an earlier snapshot row —
|
||
# refresh it in place instead of stacking a duplicate.
|
||
_replace_message_content(_tail, todo_snapshot)
|
||
_tail["_todo_snapshot_synthetic"] = True
|
||
merged = True
|
||
if not merged:
|
||
compressed.append({
|
||
"role": "user",
|
||
"content": todo_snapshot,
|
||
"_todo_snapshot_synthetic": True,
|
||
})
|
||
|
||
|
||
def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str:
|
||
"""Refresh tool schemas and rebuild the system prompt at the commit boundary."""
|
||
cached_system_prompt = agent._cached_system_prompt
|
||
agent._invalidate_system_prompt()
|
||
|
||
# Refresh tool schemas at the commit boundary: forever-sessions never restart,
|
||
# so config reaches agent.tools here. Keep list identity if byte-equal (cache).
|
||
try:
|
||
_refresh_agent_tool_definitions(agent)
|
||
except Exception: # noqa: BLE001
|
||
logger.warning(
|
||
"compaction tool-definition refresh failed; keeping the "
|
||
"session's existing tool snapshot",
|
||
exc_info=True,
|
||
)
|
||
|
||
# ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes
|
||
# never reached long sessions. Equal bytes keep KV; preserve object identity.
|
||
rebuilt_system_prompt = agent._build_system_prompt(system_message)
|
||
if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt:
|
||
new_system_prompt = cached_system_prompt
|
||
agent._cached_system_prompt = cached_system_prompt
|
||
from agent.system_prompt import reconstruct_static_prefix
|
||
|
||
reconstruct_static_prefix(
|
||
agent,
|
||
system_message=system_message,
|
||
log_label="compression keep-prompt",
|
||
)
|
||
else:
|
||
new_system_prompt = rebuilt_system_prompt
|
||
agent._cached_system_prompt = new_system_prompt
|
||
if cached_system_prompt is not None:
|
||
logger.info(
|
||
"Compaction rebuilt a drifted system prompt "
|
||
"(session=%s, %d -> %d chars): builder output changed "
|
||
"since the stored snapshot (update, config change, or "
|
||
"memory/skills growth)",
|
||
agent.session_id or "none",
|
||
len(cached_system_prompt),
|
||
len(new_system_prompt),
|
||
)
|
||
return new_system_prompt
|
||
|
||
|
||
def _salvage_or_refuse_grown_transcript(
|
||
agent: Any,
|
||
messages: list,
|
||
compressed: list,
|
||
*,
|
||
system_message: str,
|
||
attempt_started_at: float,
|
||
attempt_snapshot: dict,
|
||
) -> Tuple[Optional[list], Optional[str]]:
|
||
"""Anti-growth guard at the COMMIT SITE (in-place commits before the gateway can inspect).
|
||
|
||
Compares like-for-like rough estimates; on growth tries one mechanical salvage
|
||
pass, else treats the attempt as a refused no-op. Returns ``(compressed, None)``
|
||
to proceed or ``(None, prompt)`` when refused (caller releases the lease).
|
||
"""
|
||
# Anti-growth guard at the COMMIT SITE: in-place commits here before the gateway
|
||
# can inspect. Compare like-for-like rough estimates; on growth treat as no-op.
|
||
_rough_in = estimate_messages_tokens_rough(messages)
|
||
_rough_out = estimate_messages_tokens_rough(compressed)
|
||
if _rough_out > _rough_in:
|
||
# Todo refresh and user-turn anchoring run after the compressor's own size check
|
||
# and can tip a break-even candidate; give it one mechanical salvage pass.
|
||
from agent.context_compressor import salvage_grown_transcript
|
||
|
||
_salvaged = salvage_grown_transcript(
|
||
messages, compressed, budget=_rough_in
|
||
)
|
||
if _salvaged is not None:
|
||
_salv_est = estimate_messages_tokens_rough(_salvaged)
|
||
if _salv_est < _rough_in:
|
||
logger.info(
|
||
"Compression salvage recovered a shrinking "
|
||
"transcript (session=%s, ~%s -> ~%s tokens)",
|
||
agent.session_id or "none",
|
||
f"{_rough_in:,}",
|
||
f"{_salv_est:,}",
|
||
)
|
||
compressed = _salvaged
|
||
_rough_out = _salv_est
|
||
if _rough_out > _rough_in:
|
||
logger.warning(
|
||
"Compression refused: compressed transcript would be "
|
||
"larger than the original (session=%s, ~%s -> ~%s "
|
||
"tokens); keeping the original transcript unchanged",
|
||
agent.session_id or "none",
|
||
f"{_rough_in:,}",
|
||
f"{_rough_out:,}",
|
||
)
|
||
# Flag the refusal on compressor state so /compress feedback reports it instead
|
||
# of comparing list lengths (adoption can change the count), claiming success.
|
||
try:
|
||
agent.context_compressor._last_compress_refused_would_grow = True
|
||
except Exception:
|
||
pass
|
||
try:
|
||
agent._emit_warning(
|
||
"⚠️ Compression refused: the generated summary "
|
||
"would have GROWN the conversation instead of "
|
||
"shrinking it. No messages were dropped — "
|
||
"conversation continues unchanged."
|
||
)
|
||
except Exception:
|
||
pass
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "would_grow")
|
||
# Count the refusal as an ineffective-compaction strike so the anti-thrash
|
||
# breaker latches; otherwise auto-compress retries the same summary every turn.
|
||
try:
|
||
agent.context_compressor.record_rejected_compaction()
|
||
except Exception:
|
||
logger.debug(
|
||
"could not record rejected-compaction strike",
|
||
exc_info=True,
|
||
)
|
||
_restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot)
|
||
return None, _existing_sp
|
||
return compressed, None
|
||
|
||
|
||
def _publish_rotated_compaction(
|
||
agent: Any,
|
||
messages: list,
|
||
compressed: list,
|
||
*,
|
||
new_system_prompt: str,
|
||
lease: _CompressionLease,
|
||
old_session_id: str,
|
||
compressed_user_turn_outcome: str,
|
||
) -> None:
|
||
"""Rotate the session: flush the parent, publish the child, re-point the agent.
|
||
|
||
Flushes current-turn msgs to the OLD session, passing the durable prefix
|
||
(messages[:persist idx]) so preflight, which runs before rows are
|
||
marker-stamped, can't re-append them.
|
||
"""
|
||
current_idx = getattr(agent, "_persist_user_message_idx", None)
|
||
persisted_history = (
|
||
messages[:current_idx]
|
||
if isinstance(current_idx, int)
|
||
and 0 <= current_idx <= len(messages)
|
||
else None
|
||
)
|
||
# The flush is durable and NOT rolled back on abort: a deliberately-ended parent
|
||
# fails publish forever, so check that before writing. Automatic end stamps are
|
||
# healed by publish (don't abort); the lease is re-acquirable (don't check it).
|
||
_parent_row_reader = getattr(agent._session_db, "get_session", None)
|
||
_parent_already_ended = False
|
||
if callable(_parent_row_reader):
|
||
try:
|
||
from hermes_state_common import is_automatic_end_reason
|
||
|
||
_parent_row = _parent_row_reader(old_session_id) or {}
|
||
_parent_already_ended = (
|
||
_parent_row.get("ended_at") is not None
|
||
and not is_automatic_end_reason(
|
||
_parent_row.get("end_reason")
|
||
)
|
||
)
|
||
except Exception:
|
||
# Fail OPEN: an unreadable row must not turn a cheap
|
||
# guard into a new way to lose compression.
|
||
_parent_already_ended = False
|
||
if _parent_already_ended:
|
||
raise RuntimeError(
|
||
f"Compression parent already ended: {old_session_id}"
|
||
)
|
||
# Foreign-tail ceiling: the flush below writes OUR rows (already in handoff);
|
||
# rows above the start watermark up to this MAX(id) are foreign appends.
|
||
try:
|
||
_foreign_tail_ceiling = (
|
||
agent._session_db.get_active_message_watermark(
|
||
agent.session_id
|
||
)
|
||
)
|
||
except Exception:
|
||
# No trustworthy ceiling: the clone could duplicate the handoff, so skip tail
|
||
# preservation this rotation.
|
||
_foreign_tail_ceiling = None
|
||
try:
|
||
agent._flush_messages_to_session_db(
|
||
messages,
|
||
conversation_history=persisted_history,
|
||
)
|
||
except Exception:
|
||
pass # best-effort — don't block compression on a flush error
|
||
# Publish closure + child + handoff in one transaction so no reader sees an
|
||
# empty child. Child stays on the parent's profile ("default" persists as NULL);
|
||
# publish also COALESCEs from the parent row for threads lacking HERMES_HOME.
|
||
try:
|
||
from hermes_cli.profiles import get_active_profile_name
|
||
|
||
_profile_for_child = get_active_profile_name()
|
||
if _profile_for_child == "default":
|
||
_profile_for_child = None
|
||
except Exception:
|
||
_profile_for_child = None
|
||
old_title = agent._session_db.get_session_title(agent.session_id)
|
||
new_session_id = (
|
||
f"{datetime.now().strftime('%Y%m%d_%H%M%S')}_"
|
||
f"{uuid.uuid4().hex[:6]}"
|
||
)
|
||
from agent.context_compressor import _DB_PERSISTED_MARKER
|
||
agent._session_db.publish_compression_child(
|
||
parent_session_id=old_session_id,
|
||
child_session_id=new_session_id,
|
||
source=agent.platform
|
||
or os.environ.get("HERMES_SESSION_SOURCE", "cli"),
|
||
model=agent.model,
|
||
model_config=agent._session_init_model_config,
|
||
system_prompt=new_system_prompt,
|
||
messages=compressed,
|
||
cwd=getattr(agent, "working_directory", None),
|
||
profile_name=_profile_for_child,
|
||
compression_lock_holder=lease.holder,
|
||
require_compression_lease=lease.holder is not None,
|
||
require_lease_refresh=lease.holder is not None,
|
||
lease_ttl_seconds=lease.ttl,
|
||
watermark=(
|
||
lease.watermark
|
||
if _foreign_tail_ceiling is not None
|
||
else None
|
||
),
|
||
watermark_ceiling=_foreign_tail_ceiling,
|
||
)
|
||
# `already_present` stamping is done by run_agent's _sync_persisted_markers;
|
||
# this branch covers inserted/merged only; direct callers must use that wrapper.
|
||
if compressed_user_turn_outcome in {"inserted", "merged"}:
|
||
# Stamp the anchor source row itself, not the (drifted, possibly out-of-range)
|
||
# persist index; don't match the HANDOFF row — for `merged` it is a superset.
|
||
_compressed_anchor_source = None
|
||
for _reversed_message in reversed(messages):
|
||
if _is_real_user_message(_reversed_message):
|
||
_compressed_anchor_source = _reversed_message
|
||
break
|
||
if isinstance(_compressed_anchor_source, dict):
|
||
_compressed_anchor_source[_DB_PERSISTED_MARKER] = True
|
||
_session_messages = getattr(
|
||
agent, "_session_messages", None
|
||
)
|
||
if (
|
||
isinstance(_session_messages, list)
|
||
and _session_messages is not messages
|
||
):
|
||
# Adoption may leave _session_messages on the pre-adoption list with an out-of-
|
||
# range idx; stamp every scoped twin against the ANCHOR SOURCE, as the wrapper.
|
||
_anchor_timestamp = _compressed_anchor_source.get(
|
||
"timestamp"
|
||
)
|
||
_found_exact_timestamp_candidate = False
|
||
if _anchor_timestamp is not None:
|
||
for _twin_message in _session_messages:
|
||
if (
|
||
isinstance(_twin_message, dict)
|
||
and _twin_message.get("timestamp")
|
||
== _anchor_timestamp
|
||
and _messages_match_scoped_identity(
|
||
_twin_message,
|
||
_compressed_anchor_source,
|
||
)
|
||
):
|
||
# Count an exact scoped twin REGARDLESS of marker: an already-stamped twin must
|
||
# still suppress the broad fallback or a content-equal old dup gets stamped.
|
||
_found_exact_timestamp_candidate = True
|
||
if not _twin_message.get(
|
||
_DB_PERSISTED_MARKER
|
||
):
|
||
_twin_message[
|
||
_DB_PERSISTED_MARKER
|
||
] = True
|
||
if not _found_exact_timestamp_candidate:
|
||
# No exact twin anywhere (or timestamp-less anchor): stamp every scoped match.
|
||
# An already-stamped exact hit never opens this branch.
|
||
for _twin_message in _session_messages:
|
||
if (
|
||
isinstance(_twin_message, dict)
|
||
and not _twin_message.get(
|
||
_DB_PERSISTED_MARKER
|
||
)
|
||
and _messages_match_scoped_identity(
|
||
_twin_message,
|
||
_compressed_anchor_source,
|
||
)
|
||
):
|
||
_twin_message[
|
||
_DB_PERSISTED_MARKER
|
||
] = True
|
||
for _handoff_message in compressed:
|
||
if isinstance(_handoff_message, dict):
|
||
_handoff_message[_DB_PERSISTED_MARKER] = True
|
||
agent.session_id = new_session_id
|
||
agent._db_flush_scan_prefix = None
|
||
try:
|
||
from gateway.session_context import set_current_session_id
|
||
|
||
set_current_session_id(agent.session_id)
|
||
except Exception:
|
||
os.environ["HERMES_SESSION_ID"] = agent.session_id
|
||
try:
|
||
from hermes_logging import set_session_context
|
||
|
||
set_session_context(agent.session_id)
|
||
except Exception:
|
||
pass
|
||
agent._session_db_created = True
|
||
# Carry /goal to the child: load_goal is a flat per-session lookup with no
|
||
# parent walk, so the goal would silently die at the boundary.
|
||
try:
|
||
from hermes_cli.goals import migrate_goal_to_session
|
||
migrate_goal_to_session(old_session_id, agent.session_id, reason="compression")
|
||
except Exception as _goal_err:
|
||
logger.debug("Could not migrate goal on compression: %s", _goal_err)
|
||
# Same boundary hazard for /heartbeat state — carry it too.
|
||
try:
|
||
from hermes_cli.heartbeat import migrate_heartbeat_to_session
|
||
migrate_heartbeat_to_session(old_session_id, agent.session_id)
|
||
except Exception as _hb_err:
|
||
logger.debug("Could not migrate heartbeat on compression: %s", _hb_err)
|
||
# Same hazard for a persistent /loop: carry it so recurring wakeups survive.
|
||
try:
|
||
from hermes_cli.loops import migrate_loop_to_session
|
||
migrate_loop_to_session(old_session_id, agent.session_id, reason="compression")
|
||
except Exception as _loop_err:
|
||
logger.debug("Could not migrate loop on compression: %s", _loop_err)
|
||
# Carry the title unchanged: renumbering per rotation made one session look
|
||
# like many. Uniqueness holds: _set_session_title transfers off the ancestor.
|
||
if old_title:
|
||
# Read provenance BEFORE the write: the transfer clears the ancestor's row, so
|
||
# a later read is None and the child would be frozen as "user".
|
||
_src = None
|
||
try:
|
||
_src = agent._session_db.get_session_title_source(
|
||
old_session_id
|
||
)
|
||
except Exception as _src_err:
|
||
logger.debug(
|
||
"Could not read title provenance: %s", _src_err
|
||
)
|
||
try:
|
||
agent._session_db.set_session_title(
|
||
agent.session_id, old_title
|
||
)
|
||
except (ValueError, Exception) as e:
|
||
logger.debug("Could not propagate title on compression: %s", e)
|
||
else:
|
||
# set_session_title() records "user"; restore the original authority so an
|
||
# inherited auto-title stays upgradeable and a manual one stays pinned.
|
||
if _src is not None:
|
||
try:
|
||
agent._session_db.set_session_title_source(
|
||
agent.session_id, _src
|
||
)
|
||
except Exception as _src_err:
|
||
logger.debug(
|
||
"Could not propagate title provenance: %s",
|
||
_src_err,
|
||
)
|
||
|
||
|
||
def _warn_summary_or_aux_fallback(agent: Any) -> None:
|
||
"""Surface a failed summary, or a recovered-but-broken aux compression model, once."""
|
||
summary_error = getattr(agent.context_compressor, "_last_summary_error", None)
|
||
if summary_error:
|
||
if getattr(agent, "_last_compression_summary_warning", None) != summary_error:
|
||
agent._last_compression_summary_warning = summary_error
|
||
agent._emit_warning(
|
||
f"⚠ Compression summary failed: {summary_error}. "
|
||
"Inserted a fallback context marker."
|
||
)
|
||
else:
|
||
# Aux model may have errored and been recovered on main; tell the user their
|
||
# auxiliary.compression.model is broken even though compression succeeded.
|
||
_aux_fail_model = getattr(agent.context_compressor, "_last_aux_model_failure_model", None)
|
||
_aux_fail_err = getattr(agent.context_compressor, "_last_aux_model_failure_error", None)
|
||
if _aux_fail_model:
|
||
# Dedup on (model, error) so we don't spam on every compaction
|
||
_aux_key = (_aux_fail_model, _aux_fail_err)
|
||
if getattr(agent, "_last_aux_fallback_warning_key", None) != _aux_key:
|
||
agent._last_aux_fallback_warning_key = _aux_key
|
||
agent._emit_warning(
|
||
f"ℹ Configured compression model '{_aux_fail_model}' failed "
|
||
f"({_aux_fail_err or 'unknown error'}). Recovered using main model — "
|
||
"check auxiliary.compression.model in config.yaml."
|
||
)
|
||
|
||
|
||
def _finish_compaction_boundary(
|
||
agent: Any,
|
||
compressed: list,
|
||
*,
|
||
new_system_prompt: str,
|
||
old_session_id: Optional[str],
|
||
in_place: bool,
|
||
compacted_in_place: bool,
|
||
session_commit_succeeded: bool,
|
||
defer_context_engine_notification: bool,
|
||
compression_made_progress: bool,
|
||
compression_used_fallback: bool,
|
||
compression_feasibility_skip: bool,
|
||
task_id: str,
|
||
) -> int:
|
||
"""Post-commit bookkeeping: notify engines/providers/hooks, re-arm usage tracking.
|
||
|
||
Returns the rough post-compression token estimate (diagnostics only).
|
||
"""
|
||
# old_session_id is bound only on rotation; _boundary_parent is the id the
|
||
# boundary notifications attribute prior state to (old id, or same id in-place).
|
||
_old_sid = old_session_id
|
||
_is_boundary = bool(_old_sid) or in_place
|
||
_context_engine_boundary_committed = session_commit_succeeded and (
|
||
bool(_old_sid) or compacted_in_place
|
||
)
|
||
_boundary_parent = _old_sid or agent.session_id or ""
|
||
|
||
# The heartbeat's terminal stamp landed on the PARENT before the id re-pointed;
|
||
# clear labels (keep last_activity_at) so the archived row isn't falsely fresh.
|
||
if _old_sid and session_commit_succeeded:
|
||
try:
|
||
_labels_db = getattr(agent, "_session_db", None)
|
||
_clear_labels = getattr(
|
||
type(_labels_db) if _labels_db is not None else None,
|
||
"clear_session_activity_labels",
|
||
None,
|
||
)
|
||
if callable(_clear_labels):
|
||
_clear_labels(_labels_db, _old_sid)
|
||
except Exception:
|
||
logger.debug(
|
||
"failed to clear archived compression parent's activity "
|
||
"labels (ignored)",
|
||
exc_info=True,
|
||
)
|
||
|
||
# Plugin engines use boundary_reason="compression" to keep lineage/checkpoint
|
||
# state. Fires in BOTH modes: in-place passes the same id, the boundary is real.
|
||
if _context_engine_boundary_committed:
|
||
if defer_context_engine_notification:
|
||
_queue_context_engine_compression_notification(
|
||
agent,
|
||
new_session_id=agent.session_id or "",
|
||
old_session_id=_boundary_parent,
|
||
)
|
||
else:
|
||
_notify_context_engine_compression_complete(
|
||
agent,
|
||
new_session_id=agent.session_id or "",
|
||
old_session_id=_boundary_parent,
|
||
)
|
||
|
||
# Providers refresh cached per-session state; reset=False, conversation goes on.
|
||
# Fires in BOTH modes so buffers don't double-count dropped turns in-place.
|
||
try:
|
||
if _is_boundary and agent._memory_manager:
|
||
agent._memory_manager.on_session_switch(
|
||
agent.session_id or "",
|
||
parent_session_id=_boundary_parent,
|
||
reset=False,
|
||
reason="compression",
|
||
)
|
||
except Exception as _me_err:
|
||
logger.debug("memory manager on_session_switch (compression): %s", _me_err)
|
||
|
||
# Route via _emit_status so the warning reaches gateway platforms; store it on
|
||
# _compression_warning so a late-bound status_callback can replay it.
|
||
_cc = agent.context_compressor.compression_count
|
||
if _cc >= 2:
|
||
_cc_msg = (
|
||
f"{agent.log_prefix}⚠️ Session compressed {_cc} times — "
|
||
f"accuracy may degrade. Consider /new to start fresh."
|
||
)
|
||
agent._compression_warning = _cc_msg
|
||
agent._emit_status(_cc_msg)
|
||
|
||
# session:compress lets hooks ingest the old session before it's lost;
|
||
# in_place=True tells them the same id was compacted rather than rotated.
|
||
if getattr(agent, "event_callback", None):
|
||
try:
|
||
agent.event_callback("session:compress", {
|
||
"platform": agent.platform or "",
|
||
"session_id": agent.session_id,
|
||
"old_session_id": _old_sid or "",
|
||
"in_place": in_place,
|
||
"compression_count": agent.context_compressor.compression_count,
|
||
})
|
||
except Exception as e:
|
||
logger.debug("event_callback error on session:compress: %s", e)
|
||
|
||
# Rotation-independent flag: the gateway uses it (not an id diff) to re-baseline
|
||
# transcript handling (history_offset=0 + rewrite on the same id) in-place.
|
||
agent._last_compression_attempt_in_place = compacted_in_place
|
||
agent._last_compaction_in_place = compacted_in_place
|
||
|
||
# Diagnostics only, not provider usage: schema-heavy rough estimates can stay
|
||
# above threshold even after the next real request fits.
|
||
_compressed_est = estimate_request_tokens_rough(
|
||
compressed,
|
||
system_prompt=new_system_prompt or "",
|
||
tools=agent.tools or None,
|
||
)
|
||
agent.context_compressor.last_compression_rough_tokens = _compressed_est
|
||
agent.context_compressor.last_prompt_tokens = -1
|
||
agent.context_compressor.last_completion_tokens = 0
|
||
agent.context_compressor.awaiting_real_usage_after_compression = True
|
||
# Transcript rewritten: invalidate the usage anchor's base snapshot explicitly
|
||
# (its structural check would fail closed anyway); estimate until re-anchored.
|
||
agent._usage_anchor = None
|
||
agent._turn_base_usage_anchor = None
|
||
# Arm the effectiveness verdict only after a completed rewrite crosses the
|
||
# boundary so later usage isn't charged to an attempt that changed nothing.
|
||
if compression_made_progress:
|
||
record_boundary = getattr(
|
||
type(agent.context_compressor),
|
||
"record_completed_compaction",
|
||
None,
|
||
)
|
||
if callable(record_boundary):
|
||
record_boundary(
|
||
agent.context_compressor,
|
||
used_fallback=compression_used_fallback,
|
||
feasibility_skip=compression_feasibility_skip,
|
||
)
|
||
else:
|
||
agent.context_compressor._verify_compaction_cleared_threshold = True
|
||
|
||
# Clear file-read dedup cache: original read content was summarized away, so a
|
||
# re-read needs full content, not a "file unchanged" stub.
|
||
try:
|
||
from tools.file_tools import reset_file_dedup
|
||
reset_file_dedup(task_id)
|
||
except Exception:
|
||
pass
|
||
# Same for the skill_view repeat-view dedup: a post-compression
|
||
# re-view must return the full skill content again.
|
||
try:
|
||
from tools.skills_tool import reset_skill_view_dedup
|
||
reset_skill_view_dedup(task_id)
|
||
except Exception:
|
||
pass
|
||
return _compressed_est
|
||
|
||
|
||
def _candidate_rejected(
|
||
agent: Any,
|
||
compressed: Any,
|
||
messages: list,
|
||
messages_before_compression: list,
|
||
*,
|
||
attempt_generation: Any,
|
||
attempt_started_at: float,
|
||
) -> bool:
|
||
"""Reject an unusable compression candidate before any session mutation.
|
||
|
||
Order matters: compressor-reported abort, no progress, empty transcript,
|
||
superseded attempt. Each branch surfaces its own warning/telemetry; the
|
||
caller releases the lease and returns the input unchanged when True.
|
||
"""
|
||
# Aborted compression returns input unchanged: surface the error, skip rotation
|
||
# (no session ended); auto-compress callers detect no-op via equal lengths.
|
||
if getattr(agent.context_compressor, "_last_compress_aborted", False):
|
||
_err = getattr(agent.context_compressor, "_last_summary_error", None) or "unknown error"
|
||
if getattr(agent, "_last_compression_summary_warning", None) != _err:
|
||
agent._last_compression_summary_warning = _err
|
||
agent._emit_warning(
|
||
f"⚠ Compression aborted: {_err}. "
|
||
"No messages were dropped — conversation continues unchanged. "
|
||
"Run /compress to retry, or /new to start a fresh session."
|
||
)
|
||
_emit_aborted_attempt_telemetry(
|
||
agent,
|
||
attempt_started_at,
|
||
(
|
||
getattr(agent.context_compressor, "_last_summary_error", None)
|
||
and "summary_generation_aborted"
|
||
),
|
||
)
|
||
return True
|
||
|
||
# Compare semantic state, not identity: engines may return an equal copy or
|
||
# mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive.
|
||
if compressed == messages_before_compression or (
|
||
_strip_marker_for_comparison(compressed)
|
||
== _strip_marker_for_comparison(messages_before_compression)
|
||
):
|
||
if messages != messages_before_compression:
|
||
messages[:] = copy.deepcopy(messages_before_compression)
|
||
logger.info(
|
||
"Compression made no progress (session=%s) — skipping boundary rewrite.",
|
||
agent.session_id or "none",
|
||
)
|
||
# Unchanged output would fail identically next turn; arm structural backoff so
|
||
# auto-compress stops re-firing each turn (success lifts it, force overrides).
|
||
try:
|
||
_no_progress_recorder = getattr(
|
||
agent.context_compressor, "_record_structural_no_op", None
|
||
)
|
||
if callable(_no_progress_recorder):
|
||
_no_progress_recorder(
|
||
"compaction returned the transcript unchanged "
|
||
"(no_progress)"
|
||
)
|
||
except Exception:
|
||
logger.debug(
|
||
"no-progress backoff arm failed", exc_info=True
|
||
)
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "no_progress")
|
||
return True
|
||
|
||
if not compressed:
|
||
logger.error(
|
||
"context compression returned an empty transcript; refusing to "
|
||
"rotate session=%s so the parent remains resumable",
|
||
agent.session_id or "none",
|
||
)
|
||
try:
|
||
agent._emit_warning(
|
||
"⚠ Compression returned an empty transcript. "
|
||
"No session split was performed; conversation continues unchanged."
|
||
)
|
||
except Exception:
|
||
pass
|
||
return True
|
||
|
||
# A newer attempt claiming this compressor supersedes us; discard the late
|
||
# candidate. Fence poison alone misses a successor that minted its own fence.
|
||
if not _compressor_attempt_is_current(agent.context_compressor, attempt_generation):
|
||
logger.warning(
|
||
"Discarding late compression candidate: attempt generation "
|
||
"%s was superseded by a newer attempt (current: %s) "
|
||
"(session=%s).",
|
||
attempt_generation,
|
||
getattr(
|
||
agent.context_compressor,
|
||
"_compression_attempt_generation",
|
||
None,
|
||
),
|
||
agent.session_id or "none",
|
||
)
|
||
_restore_messages_snapshot(messages, messages_before_compression)
|
||
agent._last_compaction_in_place = False
|
||
_emit_aborted_attempt_telemetry(agent, attempt_started_at, "attempt_superseded")
|
||
return True
|
||
return False
|
||
|
||
|
||
def compress_context(
|
||
agent: Any,
|
||
messages: list,
|
||
system_message: str,
|
||
*,
|
||
approx_tokens: Optional[int] = None,
|
||
task_id: str = "default",
|
||
focus_topic: Optional[str] = None,
|
||
force: bool = False,
|
||
bypass_cooldown: bool = False,
|
||
defer_context_engine_notification: bool = False,
|
||
commit_fence: Optional[CompressionCommitFence] = None,
|
||
) -> Tuple[list, str]:
|
||
"""Compress conversation context and split the session in SQLite.
|
||
|
||
``force`` (manual /compress) clears the summary-failure cooldown;
|
||
``bypass_cooldown`` (provider-proven overflow) skips it once, breakers still
|
||
apply. ``commit_fence`` stops a timed-out worker mutating session state.
|
||
Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split.
|
||
"""
|
||
_compressor_attempt_snapshot = _snapshot_compressor_attempt_state(
|
||
agent.context_compressor
|
||
)
|
||
# Claim attempt ownership so a late-unwinding sibling (stall-fallback overlap)
|
||
# cannot restore its snapshot over ours or clear our cancellation consult.
|
||
_attempt_generation = _claim_compressor_attempt(agent.context_compressor)
|
||
_durable_cooldown_authoritative: Optional[bool] = None
|
||
_durable_cooldown_state: Optional[dict[str, Any]] = None
|
||
if (
|
||
defer_context_engine_notification
|
||
and callable(getattr(agent, _PENDING_CONTEXT_ENGINE_NOTIFICATION, None))
|
||
):
|
||
raise RuntimeError("a compression notification is already pending")
|
||
|
||
# Per-attempt outcome for conversation_history_after_compression(); None means
|
||
# aborted/no boundary, so the previous flush baseline stays authoritative.
|
||
agent._last_compression_attempt_recorded = True
|
||
agent._last_compression_attempt_in_place = None
|
||
# Clear at the VERY TOP, before codex/breaker early-returns: a stale value must
|
||
# not make a later no-op look like lock contention to automatic-path consumers.
|
||
agent._compression_skipped_due_to_lock = None
|
||
# Per-attempt transient-block signal, set when a cooldown/backoff guard no-ops
|
||
# this pass.
|
||
agent._compression_blocked_transient = None
|
||
|
||
_attempt_started_at = time.monotonic()
|
||
_attempt_id = uuid.uuid4().hex
|
||
_trigger_source = "manual" if force else "auto"
|
||
try:
|
||
agent._compression_attempt_id = _attempt_id
|
||
setattr(agent.context_compressor, "_compression_telemetry_seed", {
|
||
"attempt_id": _attempt_id,
|
||
"session_id": agent.session_id or "",
|
||
"trigger_source": _trigger_source,
|
||
})
|
||
except Exception:
|
||
pass
|
||
|
||
# Codex owns the real thread; route compaction to its own compact (config
|
||
# compression.codex_app_server_auto). Memory handoff is Hermes-only: no native
|
||
# summary prompt to inject into. `is True`: MagicMock attributes are truthy.
|
||
checkpoint_required = (
|
||
getattr(agent, "compression_checkpoint_required", False) is True
|
||
)
|
||
if getattr(agent, "api_mode", None) == "codex_app_server":
|
||
if checkpoint_required:
|
||
raise _checkpoint_blocked(
|
||
"codex_app_server owns the authoritative thread and does not "
|
||
"expose a truthful pre-compaction transcript boundary"
|
||
)
|
||
_codex_fence_entered = False
|
||
if commit_fence is not None:
|
||
_codex_fence_entered = commit_fence.begin_commit(
|
||
getattr(agent, "_hard_interrupt_requested", None)
|
||
)
|
||
if not _codex_fence_entered:
|
||
_restore_compressor_attempt_state(
|
||
agent.context_compressor, _compressor_attempt_snapshot,
|
||
attempt_generation=_attempt_generation,
|
||
)
|
||
existing_prompt = _existing_system_prompt(agent, system_message)
|
||
return messages, existing_prompt
|
||
try:
|
||
return _compress_context_via_codex_app_server(
|
||
agent,
|
||
messages,
|
||
system_message,
|
||
approx_tokens=approx_tokens,
|
||
task_id=task_id,
|
||
force=force,
|
||
)
|
||
finally:
|
||
if _codex_fence_entered:
|
||
commit_fence.finish_commit()
|
||
|
||
# All automatic entrypoints honor compressor cooldown/breaker state; hygiene's
|
||
# fresh AIAgent loads the persisted streak via bind_session_state() first.
|
||
if not force:
|
||
_refresh_persisted_compression_guards(agent.context_compressor)
|
||
blocked = getattr(
|
||
type(agent.context_compressor),
|
||
"_automatic_compression_blocked",
|
||
None,
|
||
)
|
||
if callable(blocked) and _automatic_gate_blocked(
|
||
blocked, agent.context_compressor, bypass_cooldown
|
||
):
|
||
_mark_compression_blocked_transient(agent, agent.context_compressor)
|
||
existing_prompt = _existing_system_prompt(agent, system_message)
|
||
return messages, existing_prompt
|
||
|
||
# Lazy feasibility probe (~400ms cold) on first attempt, not __init__; it sets
|
||
# _compression_warning so status replay still surfaces the warning.
|
||
if not getattr(agent, "_compression_feasibility_checked", False):
|
||
# Mark checked only after the probe completes; a raise leaves it unset
|
||
# harmlessly, transient failures are swallowed inside so it sets next pass.
|
||
check_compression_model_feasibility(agent)
|
||
agent._compression_feasibility_checked = True
|
||
|
||
_pre_msg_count = len(messages)
|
||
# In-place keeps the SAME session_id (no rotation/child/renumber/re-sync). A
|
||
# missing attribute must default True, not rotation, which can wedge sessions.
|
||
in_place = bool(getattr(agent, "compression_in_place", True))
|
||
# Set True once the in-place DB write actually completes (the DB block can
|
||
# raise and skip it). Surfaced to the gateway via agent._last_compaction_in_place.
|
||
compacted_in_place = False
|
||
logger.info(
|
||
"context compression started: session=%s messages=%d tokens=~%s model=%s focus=%r",
|
||
agent.session_id or "none", _pre_msg_count,
|
||
f"{approx_tokens:,}" if approx_tokens else "unknown", agent.model,
|
||
focus_topic,
|
||
)
|
||
_compaction_status = COMPACTION_STATUS
|
||
if not force:
|
||
_compaction_status = automatic_compaction_status_message(
|
||
agent.context_compressor,
|
||
phase="compress",
|
||
default_message=_compaction_status,
|
||
approx_tokens=approx_tokens,
|
||
message_count=_pre_msg_count,
|
||
model=agent.model,
|
||
focus_topic=focus_topic,
|
||
)
|
||
_compaction_status_emitted = bool(_compaction_status)
|
||
if _compaction_status:
|
||
agent._emit_status(_compaction_status)
|
||
lifecycle = _CompactionLifecycle(agent, _compaction_status_emitted)
|
||
|
||
lease, _abort_prompt = _acquire_compression_lease(
|
||
agent,
|
||
commit_fence=commit_fence,
|
||
lifecycle=lifecycle,
|
||
system_message=system_message,
|
||
approx_tokens=approx_tokens,
|
||
attempt_started_at=_attempt_started_at,
|
||
)
|
||
if lease is None:
|
||
return messages, _abort_prompt
|
||
|
||
# Publish the holder-qualified release hook before a timeout can win the
|
||
# fence. If no durable lock was acquired there is no hook to publish.
|
||
lease.finish_lock_setup()
|
||
|
||
_adopted = _adopt_if_parent_rotated(agent, lease, messages, system_message)
|
||
if _adopted is not None:
|
||
return _adopted
|
||
|
||
# Snapshot durable cooldown only once we own the lease. Runs for force=True
|
||
# too but skips the automatic breaker gate: manual compression retries now.
|
||
_durable_cooldown_authoritative, _durable_cooldown_state = (
|
||
_capture_authoritative_cooldown_under_lease(
|
||
agent.context_compressor,
|
||
_compressor_attempt_snapshot,
|
||
)
|
||
)
|
||
if _durable_cooldown_authoritative is False:
|
||
# Durable cooldown read failed under a built-in compressor: force=True could
|
||
# clear an unknown newer row before cancellation could restore it. Abort.
|
||
lease.release()
|
||
existing_prompt = _existing_system_prompt(agent, system_message)
|
||
return messages, existing_prompt
|
||
|
||
# Another path may have compacted this session in place since construction;
|
||
# re-read breaker state under the lock, not the bind_session_state() snapshot.
|
||
if not force:
|
||
compressor = agent.context_compressor
|
||
_refresh_persisted_compression_guards(
|
||
compressor,
|
||
include_cooldown=False,
|
||
)
|
||
blocked = getattr(
|
||
type(compressor),
|
||
"_automatic_compression_blocked",
|
||
None,
|
||
)
|
||
if callable(blocked) and _automatic_gate_blocked(
|
||
blocked, compressor, bypass_cooldown
|
||
):
|
||
_mark_compression_blocked_transient(agent, compressor)
|
||
lease.release()
|
||
existing_prompt = _existing_system_prompt(agent, system_message)
|
||
return messages, existing_prompt
|
||
|
||
_activity_heartbeat: Optional[_CompressionActivityHeartbeat] = None
|
||
messages_before_compression = None
|
||
try:
|
||
lease.start_refresher()
|
||
|
||
if not in_place:
|
||
_adopted_parent = _adopt_grown_durable_parent(agent, lease, messages)
|
||
if _adopted_parent is not None:
|
||
messages = _adopted_parent
|
||
_pre_msg_count = len(messages)
|
||
# Estimate was for the stale snapshot; force re-derivation from adopted rows.
|
||
approx_tokens = 0
|
||
# Adopted list is fully durable: re-anchor persist idx at the end so the post-
|
||
# compression flush skips it; run_agent marker sync realigns _session_messages.
|
||
agent._persist_user_message_idx = len(messages)
|
||
|
||
memory_context = _pre_compress_memory_context(agent, messages, checkpoint_required)
|
||
|
||
compress_fn, compress_kwargs = _resolve_compress_call(
|
||
agent,
|
||
approx_tokens=approx_tokens,
|
||
focus_topic=focus_topic,
|
||
force=force,
|
||
memory_context=memory_context,
|
||
bypass_cooldown=bypass_cooldown,
|
||
)
|
||
|
||
messages_before_compression = copy.deepcopy(messages)
|
||
_activity_heartbeat = _CompressionActivityHeartbeat(
|
||
agent, commit_fence=commit_fence
|
||
).start()
|
||
# Interrupts/redirects must not tear a summary in half. Use the explicit stop
|
||
# Event (message fields race) + fence timeout so pool slots free promptly.
|
||
_hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None)
|
||
compressed = _run_summary_dispatch(
|
||
agent,
|
||
messages,
|
||
compress_fn,
|
||
compress_kwargs,
|
||
commit_fence=commit_fence,
|
||
attempt_generation=_attempt_generation,
|
||
hard_cancel_event=_hard_cancel_event,
|
||
)
|
||
except AuxiliaryExplicitCancellation:
|
||
try:
|
||
_restore_compressor_attempt_state(
|
||
agent.context_compressor,
|
||
_compressor_attempt_snapshot,
|
||
durable_cooldown_authoritative=_durable_cooldown_authoritative,
|
||
durable_cooldown_state=_durable_cooldown_state,
|
||
attempt_generation=_attempt_generation,
|
||
)
|
||
except BaseException as _rollback_exc:
|
||
# Compensation failure must surface, but it must not strand the
|
||
# session lease or retain an in-memory transcript mutation.
|
||
_restore_messages_snapshot(messages, messages_before_compression)
|
||
if _activity_heartbeat is not None:
|
||
_activity_heartbeat.stop("context compression rollback failed")
|
||
_activity_heartbeat = None
|
||
lease.release()
|
||
_emit_aborted_attempt_telemetry(agent, _attempt_started_at, f"rollback:{type(_rollback_exc).__name__}")
|
||
raise
|
||
_restore_messages_snapshot(messages, messages_before_compression)
|
||
# Record after restore so rollback cannot wipe a stall backoff, and
|
||
# while the lease is still held so the next turn cannot race it.
|
||
_stall_backoff = _record_stall_interrupted_backoff(
|
||
agent,
|
||
commit_fence=commit_fence,
|
||
started_at=_attempt_started_at,
|
||
messages=messages,
|
||
approx_tokens=approx_tokens,
|
||
)
|
||
if _activity_heartbeat is not None:
|
||
_activity_heartbeat.stop("context compression cancelled")
|
||
_activity_heartbeat = None
|
||
lease.release()
|
||
_emit_aborted_attempt_telemetry(
|
||
agent,
|
||
_attempt_started_at,
|
||
(
|
||
STALL_INTERRUPTED_FAILURE_CLASS
|
||
if _stall_backoff
|
||
else "explicit_interrupt"
|
||
),
|
||
)
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
return messages, _existing_sp
|
||
except BaseException as _compress_exc:
|
||
# Any failure after lock acquisition must release it or the session is
|
||
# permanently blocked from compression.
|
||
if _activity_heartbeat is not None:
|
||
_activity_heartbeat.stop("context compression failed")
|
||
_activity_heartbeat = None
|
||
lease.release()
|
||
_emit_aborted_attempt_telemetry(agent, _attempt_started_at, f"exception:{type(_compress_exc).__name__}")
|
||
raise
|
||
finally:
|
||
if _activity_heartbeat is not None:
|
||
_activity_heartbeat.stop("context compression completed")
|
||
|
||
_commit_fence_entered = False
|
||
try:
|
||
# Capture the verdict before rotation callbacks: lifecycle hooks may reset
|
||
# compressor fields on rebind; record only after the full boundary commits.
|
||
_compression_made_progress = bool(
|
||
getattr(agent.context_compressor, "_last_compression_made_progress", False)
|
||
)
|
||
_compression_used_fallback = bool(
|
||
getattr(agent.context_compressor, "_last_summary_fallback_used", False)
|
||
)
|
||
_compression_feasibility_skip = bool(
|
||
getattr(agent.context_compressor, "_last_feasibility_skip", False)
|
||
)
|
||
|
||
if _candidate_rejected(
|
||
agent,
|
||
compressed,
|
||
messages,
|
||
messages_before_compression,
|
||
attempt_generation=_attempt_generation,
|
||
attempt_started_at=_attempt_started_at,
|
||
):
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
lease.release()
|
||
return messages, _existing_sp
|
||
|
||
if commit_fence is not None:
|
||
_commit_fence_entered = commit_fence.begin_commit(_hard_cancel_event)
|
||
if not _commit_fence_entered:
|
||
_restore_compressor_attempt_state(
|
||
agent.context_compressor,
|
||
_compressor_attempt_snapshot,
|
||
durable_cooldown_authoritative=_durable_cooldown_authoritative,
|
||
durable_cooldown_state=_durable_cooldown_state,
|
||
attempt_generation=_attempt_generation,
|
||
)
|
||
_restore_messages_snapshot(messages, messages_before_compression)
|
||
logger.info(
|
||
"Compression commit cancelled before session mutation "
|
||
"(session=%s).",
|
||
agent.session_id or "none",
|
||
)
|
||
agent._last_compaction_in_place = False
|
||
_stall_backoff = _record_stall_interrupted_backoff(
|
||
agent,
|
||
commit_fence=commit_fence,
|
||
started_at=_attempt_started_at,
|
||
messages=messages,
|
||
approx_tokens=approx_tokens,
|
||
)
|
||
_existing_sp = _existing_system_prompt(agent, system_message)
|
||
_emit_aborted_attempt_telemetry(
|
||
agent,
|
||
_attempt_started_at,
|
||
(
|
||
STALL_INTERRUPTED_FAILURE_CLASS
|
||
if _stall_backoff
|
||
else "commit_fence_cancelled"
|
||
),
|
||
)
|
||
lease.release()
|
||
return messages, _existing_sp
|
||
|
||
_warn_summary_or_aux_fallback(agent)
|
||
|
||
_fold_todo_snapshot(agent, compressed)
|
||
compressed_user_turn_outcome = _ensure_compressed_has_user_turn(
|
||
messages, compressed
|
||
)
|
||
|
||
new_system_prompt = _rebuild_system_prompt_at_boundary(agent, system_message)
|
||
|
||
_session_commit_succeeded = False
|
||
_commit_started_at = time.monotonic()
|
||
split_status = "not_applicable"
|
||
old_session_id: Optional[str] = None # bound only once rotation begins
|
||
if agent._session_db:
|
||
split_status = "pending"
|
||
try:
|
||
# Memory extraction runs in BOTH modes: pre-compaction turns are summarized
|
||
# away whether or not the id rotates.
|
||
agent.commit_memory_session(messages)
|
||
|
||
# Pop _compaction_tail tags before the size estimate / rotation: they must not
|
||
# inflate anti-growth or reach the provider. Track ids: salvage may subset list.
|
||
_tail_tagged_ids = {
|
||
id(m)
|
||
for m in compressed
|
||
if isinstance(m, dict) and m.pop("_compaction_tail", None)
|
||
}
|
||
|
||
compressed, _refused_sp = _salvage_or_refuse_grown_transcript(
|
||
agent,
|
||
messages,
|
||
compressed,
|
||
system_message=system_message,
|
||
attempt_started_at=_attempt_started_at,
|
||
attempt_snapshot=_compressor_attempt_snapshot,
|
||
)
|
||
if compressed is None:
|
||
lease.release()
|
||
return messages, _refused_sp
|
||
|
||
if in_place:
|
||
# In-place compaction: same session_id; soft-archive old turns (active=0, still
|
||
# searchable) + insert `compressed` atomically; no pre-flush (tail already in).
|
||
from agent.context_compressor import (
|
||
PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY,
|
||
)
|
||
|
||
# Tail rows tagged by compress() are archived as superseded duplicates, not
|
||
# compacted=1. Count against the FINAL list — salvage may have dropped rows.
|
||
_tail_count = sum(
|
||
1 for m in compressed if id(m) in _tail_tagged_ids
|
||
)
|
||
agent._session_db.archive_and_compact(
|
||
agent.session_id,
|
||
compressed,
|
||
model_config_patch={
|
||
PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY: None,
|
||
},
|
||
watermark=lease.watermark,
|
||
lock_holder=lease.holder,
|
||
tail_count=_tail_count,
|
||
)
|
||
split_status = "in_place_committed"
|
||
# compress() returned marker-swept copies; stamp them as persisted or the next
|
||
# flush re-INSERTs the whole compacted transcript, doubling the live set.
|
||
from agent.context_compressor import (
|
||
stamp_db_persisted_markers,
|
||
)
|
||
|
||
stamp_db_persisted_markers(compressed)
|
||
# Reset flush identity set so next turn diffs against the COMPACTED transcript:
|
||
# only genuinely new messages append (no summary dup, no resurrected turns).
|
||
agent._flushed_db_message_ids = set()
|
||
# Rotation-independent signal; the gateway reads this (not an id diff) to
|
||
# re-baseline transcript handling.
|
||
compacted_in_place = True
|
||
else:
|
||
# Bind old_session_id first: it is the rollback key in the handler below.
|
||
old_session_id = agent.session_id
|
||
_publish_rotated_compaction(
|
||
agent,
|
||
messages,
|
||
compressed,
|
||
new_system_prompt=new_system_prompt,
|
||
lease=lease,
|
||
old_session_id=old_session_id,
|
||
compressed_user_turn_outcome=compressed_user_turn_outcome,
|
||
)
|
||
split_status = "rotated_committed"
|
||
|
||
# In-place mode still updates/replaces the current row here.
|
||
# Rotation already published prompt + compacted handoff atomically.
|
||
if in_place:
|
||
agent._session_db.update_system_prompt(
|
||
agent.session_id, new_system_prompt
|
||
)
|
||
agent._last_flushed_db_idx = 0
|
||
else:
|
||
agent._last_flushed_db_idx = len(compressed)
|
||
agent._flushed_db_message_session_id = agent.session_id
|
||
_session_commit_succeeded = True
|
||
except Exception as e:
|
||
if (
|
||
not in_place
|
||
and old_session_id
|
||
and agent.session_id == old_session_id
|
||
):
|
||
# Atomic publication failed (including lease loss): keep the
|
||
# parent live and discard the stale compacted snapshot.
|
||
old_session_id = None
|
||
# _db_flush_scan_prefix is intentionally NOT cleared: the scan is identity-based
|
||
# and the deepcopy replaces every row. A failed parent flush clears its own; the
|
||
# snapshot path leaves the live list untouched. Recheck both before adding one.
|
||
messages[:] = copy.deepcopy(messages_before_compression)
|
||
compressed = messages
|
||
_compression_made_progress = False
|
||
# Only the runway rolls back: the full snapshot restore is for pre-commit
|
||
# cancels (telemetry keeps failed values).
|
||
_restore_prune_rearm_tokens(agent.context_compressor, _compressor_attempt_snapshot)
|
||
elif (
|
||
in_place
|
||
and split_status != "in_place_committed"
|
||
and messages_before_compression is not None
|
||
):
|
||
# In-place rollback: archive_and_compact is atomic so old rows stay active, but
|
||
# marker-swept `compressed` would re-INSERT on top of them (doubling each try).
|
||
# Gate on split_status (set right after commit); deepcopy keeps markers/identity
|
||
messages[:] = copy.deepcopy(messages_before_compression)
|
||
compressed = messages
|
||
_compression_made_progress = False
|
||
_restore_prune_rearm_tokens(agent.context_compressor, _compressor_attempt_snapshot)
|
||
split_status = (
|
||
"aborted"
|
||
if old_session_id is None and not in_place
|
||
else "failed_not_indexed"
|
||
)
|
||
# If rotation rolled back to the parent, agent.session_id is the indexed parent
|
||
# and old_session_id was cleared: recovery, not an un-indexed orphan.
|
||
if old_session_id is None and not in_place:
|
||
logger.warning(
|
||
"Compression rotation aborted and rolled back to the "
|
||
"parent session (%s): %s", agent.session_id or "?", e,
|
||
)
|
||
else:
|
||
logger.warning("Session DB compression split failed — new session will NOT be indexed: %s", e)
|
||
# Arm the failure cooldown so the next turn can't rerun the doomed compression;
|
||
# try/except so a stub compressor can't mask the original error in this handler.
|
||
try:
|
||
agent.context_compressor._record_compression_failure_cooldown(
|
||
_SPLIT_FAILURE_COOLDOWN_SECONDS,
|
||
f"session_split_failed: {e}",
|
||
)
|
||
except Exception:
|
||
logger.debug(
|
||
"could not record split-failure cooldown",
|
||
exc_info=True,
|
||
)
|
||
|
||
_compressed_est = _finish_compaction_boundary(
|
||
agent,
|
||
compressed,
|
||
new_system_prompt=new_system_prompt,
|
||
old_session_id=old_session_id,
|
||
in_place=in_place,
|
||
compacted_in_place=compacted_in_place,
|
||
session_commit_succeeded=_session_commit_succeeded,
|
||
defer_context_engine_notification=defer_context_engine_notification,
|
||
compression_made_progress=_compression_made_progress,
|
||
compression_used_fallback=_compression_used_fallback,
|
||
compression_feasibility_skip=_compression_feasibility_skip,
|
||
task_id=task_id,
|
||
)
|
||
|
||
logger.info(
|
||
"context compression done: session=%s messages=%d->%d rough_tokens=~%s awaiting_real_usage=true",
|
||
agent.session_id or "none", _pre_msg_count, len(compressed),
|
||
f"{_compressed_est:,}",
|
||
)
|
||
lifecycle.commit_status = "committed" if split_status in {"not_applicable", "in_place_committed", "rotated_committed"} else "aborted"
|
||
_emit_compression_attempt_telemetry(
|
||
agent,
|
||
started_at=_attempt_started_at,
|
||
commit_status=lifecycle.commit_status,
|
||
split_status=split_status,
|
||
failure_class=(
|
||
"session_split_failed"
|
||
if split_status in {"failed_not_indexed", "aborted"}
|
||
else None
|
||
),
|
||
commit_started_at=_commit_started_at,
|
||
)
|
||
return compressed, new_system_prompt
|
||
finally:
|
||
# Release the OLD session's lock only after rotation and all post-rotation
|
||
# bookkeeping; a waking contender then sees the NEW id and acquires on that.
|
||
try:
|
||
lease.release()
|
||
finally:
|
||
if _commit_fence_entered:
|
||
commit_fence.finish_commit()
|
||
|
||
|
||
def _codex_compaction_cooldown_remaining(agent: Any) -> float:
|
||
"""Seconds left on this session's compaction-failure cooldown (0 = clear)."""
|
||
compressor = getattr(agent, "context_compressor", None)
|
||
getter = getattr(compressor, "get_active_compression_failure_cooldown", None)
|
||
if not callable(getter):
|
||
return 0.0
|
||
try:
|
||
state = getter(refresh=True)
|
||
except Exception:
|
||
logger.debug("codex compaction cooldown lookup failed", exc_info=True)
|
||
return 0.0
|
||
if not state:
|
||
return 0.0
|
||
try:
|
||
return max(0.0, float(state.get("remaining_seconds") or 0.0))
|
||
except (TypeError, ValueError):
|
||
return 0.0
|
||
|
||
|
||
def _record_codex_compaction_failure(agent: Any, error: str) -> None:
|
||
"""Arm the shared compression-failure cooldown after a failed codex compaction.
|
||
|
||
The codex path returns the transcript unchanged, so without a cooldown the
|
||
still-over-threshold session would retry every turn.
|
||
"""
|
||
from agent.context_compressor import _SUMMARY_FAILURE_COOLDOWN_SECONDS
|
||
|
||
compressor = getattr(agent, "context_compressor", None)
|
||
recorder = getattr(compressor, "_record_compression_failure_cooldown", None)
|
||
if not callable(recorder):
|
||
return
|
||
try:
|
||
recorder(_SUMMARY_FAILURE_COOLDOWN_SECONDS, error)
|
||
except Exception:
|
||
logger.debug("codex compaction cooldown persist failed", exc_info=True)
|
||
|
||
|
||
def _compress_context_via_codex_app_server(
|
||
agent: Any,
|
||
messages: list,
|
||
system_message: Optional[str],
|
||
*,
|
||
approx_tokens: Optional[int] = None,
|
||
task_id: str = "default",
|
||
force: bool = False,
|
||
) -> Tuple[list, str]:
|
||
"""Route compaction to Codex app-server for Codex-owned threads.
|
||
|
||
Rewriting the local transcript would not shrink the Codex thread, so Codex
|
||
compacts its own thread and Hermes' transcript is left unchanged.
|
||
"""
|
||
_sid = getattr(agent, "session_id", None) or "none"
|
||
_tokens = f"{approx_tokens:,}" if approx_tokens else "unknown"
|
||
auto_mode = str(
|
||
getattr(agent, "codex_app_server_auto_compaction", "native") or "native"
|
||
).lower()
|
||
if auto_mode not in {"native", "hermes", "off"}:
|
||
auto_mode = "native"
|
||
skip_reason = None
|
||
if not force and auto_mode != "hermes":
|
||
skip_reason = f"mode={auto_mode} force=false"
|
||
elif not force:
|
||
# Automatic entrypoints honor the compressor-owned cooldown: a recent compaction
|
||
# failed, and retrying every turn is what thrashes.
|
||
_cooldown_remaining = _codex_compaction_cooldown_remaining(agent)
|
||
if _cooldown_remaining > 0:
|
||
skip_reason = f"failure cooldown active for {_cooldown_remaining:.0f}s"
|
||
codex_session = getattr(agent, "_codex_session", None)
|
||
if skip_reason is None and codex_session is None:
|
||
skip_reason = "no active codex thread"
|
||
if skip_reason is not None:
|
||
logger.info(
|
||
"codex app-server compaction skipped: %s (session=%s messages=%d tokens=~%s)",
|
||
skip_reason, _sid, len(messages), _tokens,
|
||
)
|
||
return messages, _existing_system_prompt(agent, system_message)
|
||
|
||
logger.info(
|
||
"codex app-server compaction started: session=%s messages=%d tokens=~%s",
|
||
_sid, len(messages), _tokens,
|
||
)
|
||
try:
|
||
agent._emit_status(COMPACTION_STATUS)
|
||
except Exception:
|
||
pass
|
||
|
||
_activity_heartbeat: Optional[_CompressionActivityHeartbeat] = None
|
||
try:
|
||
_activity_heartbeat = _CompressionActivityHeartbeat(agent).start()
|
||
result = codex_session.compact_thread()
|
||
except BaseException:
|
||
if _activity_heartbeat is not None:
|
||
_activity_heartbeat.stop("context compression failed")
|
||
raise
|
||
|
||
if getattr(result, "interrupted", False) or getattr(result, "error", None):
|
||
_activity_heartbeat.stop("context compression failed")
|
||
else:
|
||
_activity_heartbeat.stop("context compression completed")
|
||
|
||
if getattr(result, "should_retire", False):
|
||
try:
|
||
codex_session.close()
|
||
except Exception:
|
||
pass
|
||
agent._codex_session = None
|
||
|
||
if getattr(result, "interrupted", False) or getattr(result, "error", None):
|
||
try:
|
||
agent._emit_warning(
|
||
f"⚠ Codex app-server compaction failed: {result.error}"
|
||
)
|
||
except Exception:
|
||
pass
|
||
# The transcript is returned unchanged, so the session is still over
|
||
# threshold. Without a brake the next turn retries immediately.
|
||
_record_codex_compaction_failure(
|
||
agent,
|
||
str(getattr(result, "error", None) or "compaction interrupted"),
|
||
)
|
||
return messages, _existing_system_prompt(agent, system_message)
|
||
|
||
try:
|
||
from agent.codex_runtime import (
|
||
_record_codex_app_server_compaction,
|
||
_record_codex_app_server_usage,
|
||
)
|
||
|
||
_record_codex_app_server_compaction(
|
||
agent,
|
||
result,
|
||
approx_tokens=approx_tokens,
|
||
force=True,
|
||
)
|
||
# An empty usage report must consume the pending verdict, not leave deferral
|
||
# armed until a later turn; minimal test engines may lack update_from_response.
|
||
if hasattr(agent.context_compressor, "update_from_response"):
|
||
_record_codex_app_server_usage(agent, result)
|
||
except Exception:
|
||
logger.debug("codex compaction bookkeeping failed", exc_info=True)
|
||
|
||
try:
|
||
from tools.file_tools import reset_file_dedup
|
||
|
||
reset_file_dedup(task_id)
|
||
except Exception:
|
||
pass
|
||
|
||
logger.info(
|
||
"codex app-server compaction done: session=%s thread=%s turn=%s",
|
||
_sid,
|
||
getattr(result, "thread_id", None) or "",
|
||
getattr(result, "turn_id", None) or "",
|
||
)
|
||
existing_prompt = _existing_system_prompt(agent, system_message)
|
||
# Terminal edge only on success — failure/interrupt paths above return
|
||
# without it, matching the main compress_context() gating.
|
||
_emit_compaction_done(agent)
|
||
return messages, existing_prompt
|
||
|
||
|
||
def try_shrink_image_parts_in_messages(
|
||
api_messages: list,
|
||
*,
|
||
max_dimension: int = 8000,
|
||
) -> bool:
|
||
"""Re-encode oversized native image parts to recover from image-too-large errors.
|
||
|
||
Mutates ``api_messages`` in place. Returns True if any part was replaced,
|
||
False if nothing to shrink or Pillow could not help. Targets data-URL parts
|
||
over 4 MB or ``max_dimension``; http(s) image URLs are left untouched.
|
||
"""
|
||
if not api_messages:
|
||
return False
|
||
|
||
try:
|
||
from tools.vision_tools import _resize_image_for_vision
|
||
except Exception as exc:
|
||
logger.warning("image-shrink recovery: vision_tools unavailable — %s", exc)
|
||
return False
|
||
|
||
# 4 MB leaves headroom under Anthropic's 5 MB; shrinking loses quality but only
|
||
# runs after a confirmed provider rejection, so the alternative is failure.
|
||
target_bytes = 4 * 1024 * 1024
|
||
# Anthropic also caps per-side pixels (8000, or lower in many-image requests);
|
||
# the caller passes the parsed ceiling when the rejection includes it.
|
||
changed_count = 0
|
||
# Track over-target parts that could not be shrunk: if any remain, a retry
|
||
# re-sends the same payload and wastes the single retry budget.
|
||
unshrinkable_oversized = 0
|
||
|
||
def _decode_pixels(data_url: str) -> Optional[tuple]:
|
||
"""Return ``(width, height)`` of a base64 data URL, or None on failure.
|
||
|
||
None when Pillow is missing or the payload is corrupt; caller falls back to a
|
||
bytes-only check.
|
||
"""
|
||
try:
|
||
import base64 as _b64_dim
|
||
import io as _io_dim
|
||
header_d, _, data_d = data_url.partition(",")
|
||
if not data_d or not data_url.startswith("data:"):
|
||
return None
|
||
from PIL import Image as _PILImage
|
||
with _PILImage.open(_io_dim.BytesIO(_b64_dim.b64decode(data_d))) as _img:
|
||
return _img.size
|
||
except Exception:
|
||
return None
|
||
|
||
def _shrink_data_url(url: str) -> tuple:
|
||
"""Return ``(resized_url, unshrinkable)`` for a data URL.
|
||
|
||
``resized_url`` is None when no rewrite applied. ``unshrinkable`` is True only
|
||
when the image violated a constraint and resizing failed to satisfy that same
|
||
constraint, so the caller knows a retry is pointless.
|
||
"""
|
||
if not isinstance(url, str) or not url.startswith("data:"):
|
||
return None, False
|
||
|
||
# The accept gate MUST use the axis that triggered the shrink: a pixel downscale
|
||
# can re-encode to MORE bytes (PNG non-monotonic); byte-only reject wedges.
|
||
needs_shrink = len(url) > target_bytes # over byte budget
|
||
triggered_by = "bytes" if needs_shrink else None
|
||
if not needs_shrink:
|
||
# Bytes fine; check pixels against the provider cap (tiny bytes, huge pixels).
|
||
dims = _decode_pixels(url)
|
||
if dims is None:
|
||
# Pillow missing or corrupt data — fall back to byte-only.
|
||
return None, False
|
||
if max(dims) <= max_dimension:
|
||
return None, False # both bytes and pixels are within limits
|
||
needs_shrink = True
|
||
triggered_by = "dimension"
|
||
|
||
try:
|
||
header, _, data = url.partition(",")
|
||
mime = "image/jpeg"
|
||
if header.startswith("data:"):
|
||
mime_part = header[len("data:"):].split(";", 1)[0].strip()
|
||
if mime_part.startswith("image/"):
|
||
mime = mime_part
|
||
import base64 as _b64
|
||
raw = _b64.b64decode(data)
|
||
suffix = {
|
||
"image/png": ".png", "image/gif": ".gif", "image/webp": ".webp",
|
||
"image/jpeg": ".jpg", "image/jpg": ".jpg", "image/bmp": ".bmp",
|
||
}.get(mime, ".jpg")
|
||
tmp = tempfile.NamedTemporaryFile(
|
||
prefix="hermes_shrink_", suffix=suffix, delete=False,
|
||
)
|
||
try:
|
||
tmp.write(raw)
|
||
tmp.close()
|
||
resized = _resize_image_for_vision(
|
||
Path(tmp.name),
|
||
mime_type=mime,
|
||
max_base64_bytes=target_bytes,
|
||
max_dimension=max_dimension,
|
||
)
|
||
finally:
|
||
try:
|
||
Path(tmp.name).unlink(missing_ok=True)
|
||
except Exception:
|
||
pass
|
||
if not resized:
|
||
# Resize returned nothing — Pillow couldn't help.
|
||
return None, True
|
||
if triggered_by == "bytes":
|
||
# Byte budget is the binding constraint — bytes must shrink.
|
||
if len(resized) >= len(url):
|
||
return None, True # re-encode made it bigger
|
||
# The resizer may return an over-cap blob (long side freezes at the 64px short-
|
||
# side floor); still over cap → re-400, so unshrinkable. Undecodable dims: skip.
|
||
new_dims = _decode_pixels(resized)
|
||
if new_dims is not None and max(new_dims) > max_dimension:
|
||
return None, True
|
||
return resized, False
|
||
# Dimension cap is binding: accept a byte-larger re-encode if now within cap.
|
||
new_dims = _decode_pixels(resized)
|
||
if new_dims is not None:
|
||
if max(new_dims) <= max_dimension:
|
||
return resized, False
|
||
# Still over the per-side cap — the resize didn't satisfy it.
|
||
return None, True
|
||
# Can't verify dimensions: fall back to the bytes-must-shrink gate so we never
|
||
# accept an unverifiable byte-larger blob.
|
||
if len(resized) >= len(url):
|
||
return None, True
|
||
return resized, False
|
||
except Exception as exc:
|
||
logger.warning("image-shrink recovery: re-encode failed — %s", exc)
|
||
return None, triggered_by is not None
|
||
|
||
def _source_to_data_url(source: Any) -> Optional[str]:
|
||
if not isinstance(source, dict) or source.get("type") != "base64":
|
||
return None
|
||
data = source.get("data")
|
||
if not isinstance(data, str) or not data:
|
||
return None
|
||
media_type = str(source.get("media_type") or "image/jpeg").strip()
|
||
if not media_type.startswith("image/"):
|
||
media_type = "image/jpeg"
|
||
return f"data:{media_type};base64,{data}"
|
||
|
||
def _write_data_url_to_source(source: dict, data_url: str) -> dict:
|
||
"""Return a NEW source dict carrying the re-encoded payload.
|
||
|
||
Copy-on-write: parts may be shared with the persistent history, so mutating
|
||
in place would store the degraded image; the caller replaces the part.
|
||
"""
|
||
header, _, data = data_url.partition(",")
|
||
media_type = "image/jpeg"
|
||
if header.startswith("data:"):
|
||
candidate = header[len("data:"):].split(";", 1)[0].strip()
|
||
if candidate.startswith("image/"):
|
||
media_type = candidate
|
||
return {
|
||
**source,
|
||
"type": "base64",
|
||
"media_type": media_type,
|
||
"data": data,
|
||
}
|
||
|
||
for msg in api_messages:
|
||
if not isinstance(msg, dict):
|
||
continue
|
||
content = msg.get("content")
|
||
if not isinstance(content, list):
|
||
continue
|
||
# Copy-on-write: part/source dicts can alias stored history, so build a new
|
||
# content list and reassign msg["content"] on the per-call copy.
|
||
new_content: list | None = None
|
||
for part_idx, part in enumerate(content):
|
||
if not isinstance(part, dict):
|
||
continue
|
||
ptype = part.get("type")
|
||
if ptype == "image":
|
||
source = part.get("source")
|
||
url = _source_to_data_url(source)
|
||
resized, unshrinkable = _shrink_data_url(url or "")
|
||
if resized and isinstance(source, dict):
|
||
if new_content is None:
|
||
new_content = list(content)
|
||
new_content[part_idx] = {
|
||
**part,
|
||
"source": _write_data_url_to_source(source, resized),
|
||
}
|
||
changed_count += 1
|
||
elif unshrinkable:
|
||
unshrinkable_oversized += 1
|
||
continue
|
||
if ptype not in {"image_url", "input_image"}:
|
||
continue
|
||
image_value = part.get("image_url")
|
||
# OpenAI chat.completions: {"image_url": {"url": "data:..."}}
|
||
# OpenAI Responses: {"image_url": "data:..."}
|
||
if isinstance(image_value, dict):
|
||
url = image_value.get("url", "")
|
||
resized, unshrinkable = _shrink_data_url(url)
|
||
if resized:
|
||
if new_content is None:
|
||
new_content = list(content)
|
||
new_content[part_idx] = {
|
||
**part,
|
||
"image_url": {**image_value, "url": resized},
|
||
}
|
||
changed_count += 1
|
||
elif unshrinkable:
|
||
unshrinkable_oversized += 1
|
||
elif isinstance(image_value, str):
|
||
resized, unshrinkable = _shrink_data_url(image_value)
|
||
if resized:
|
||
if new_content is None:
|
||
new_content = list(content)
|
||
new_content[part_idx] = {**part, "image_url": resized}
|
||
changed_count += 1
|
||
elif unshrinkable:
|
||
unshrinkable_oversized += 1
|
||
if new_content is not None:
|
||
msg["content"] = new_content
|
||
|
||
if changed_count:
|
||
logger.info(
|
||
"image-shrink recovery: re-encoded %d image part(s) to fit under %.0f MB",
|
||
changed_count, target_bytes / (1024 * 1024),
|
||
)
|
||
if unshrinkable_oversized:
|
||
# An unshrinkable oversized image makes retry pointless; signal no progress even
|
||
# if others shrank so the caller surfaces the original error.
|
||
logger.warning(
|
||
"image-shrink recovery: %d oversized image part(s) could not be "
|
||
"shrunk under %.0f MB — not retrying (would re-send rejected payload)",
|
||
unshrinkable_oversized, target_bytes / (1024 * 1024),
|
||
)
|
||
return False
|
||
return changed_count > 0
|
||
|
||
|
||
__all__ = [
|
||
"COMPACTION_STATUS",
|
||
"COMPACTION_DONE_STATUS",
|
||
"COMPACTION_STATUS_MARKER",
|
||
"is_compaction_progress_status",
|
||
"check_compression_model_feasibility",
|
||
"replay_compression_warning",
|
||
"compress_context",
|
||
"try_shrink_image_parts_in_messages",
|
||
]
|