fix(native_compaction): preserve compression summary messages during pre-checkpoint pruning
prune_pre_checkpoint_items() had a hardcoded role=='user' filter that discarded all non-user messages before a checkpoint — including Hermes' own compression summaries (role='assistant'), causing total context amnesia about past conversation summaries. The fix: - _is_summary_item delegates to the canonical agent.context_compressor.is_compaction_summary_message provenance check (not an ad-hoc heuristic) - Summaries are retained whole (never byte-sliced) within a 32k token budget - Idempotent across repeated checkpoints (dedup by identical text) - _chat_messages_to_responses_input threads item_sources (raw chat messages) through to the pruner, so it can read summary content directly from the source when the Responses conversion shape is lossy (tool-result carrier becomes function_call_output, or stale codex_message_items replay shadows merged content) Fixes #90975. Salvage of #90976 by @JoaoMarcos44.
This commit is contained in:
@@ -479,6 +479,12 @@ def _chat_messages_to_responses_input(
|
||||
conversation is still on the wire.
|
||||
"""
|
||||
items: List[Dict[str, Any]] = []
|
||||
# Parallel to `items`: the raw chat message each converted item came
|
||||
# from. Pruning needs this to read a canonical summary carrier's
|
||||
# up-to-date, provenance-tagged content directly — the converted `item`
|
||||
# can be a lossy shape (stale exact-replay, or a typed
|
||||
# `function_call_output` wrapper) that no longer carries it (#90976).
|
||||
item_sources: List[Optional[Dict[str, Any]]] = []
|
||||
seen_item_ids: set = set()
|
||||
|
||||
for msg in messages:
|
||||
@@ -567,6 +573,7 @@ def _chat_messages_to_responses_input(
|
||||
if k not in ("id", "_issuer_kind")
|
||||
}
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
if item_id:
|
||||
seen_item_ids.add(item_id)
|
||||
has_codex_reasoning = True
|
||||
@@ -623,14 +630,17 @@ def _chat_messages_to_responses_input(
|
||||
if isinstance(phase, str) and phase.strip():
|
||||
replay_item["phase"] = phase.strip()
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
replayed_message_items += 1
|
||||
|
||||
if replayed_message_items > 0:
|
||||
pass
|
||||
elif content_parts:
|
||||
items.append({"role": "assistant", "content": content_parts})
|
||||
item_sources.append(msg)
|
||||
elif content_text.strip():
|
||||
items.append({"role": "assistant", "content": content_text})
|
||||
item_sources.append(msg)
|
||||
elif has_codex_reasoning:
|
||||
# The Responses API requires a following item after each
|
||||
# reasoning item (otherwise: missing_following_item error).
|
||||
@@ -638,6 +648,7 @@ def _chat_messages_to_responses_input(
|
||||
# content, emit an empty assistant message as the required
|
||||
# following item.
|
||||
items.append({"role": "assistant", "content": ""})
|
||||
item_sources.append(msg)
|
||||
|
||||
tool_calls = msg.get("tool_calls")
|
||||
if isinstance(tool_calls, list):
|
||||
@@ -680,6 +691,7 @@ def _chat_messages_to_responses_input(
|
||||
"name": fn_name,
|
||||
"arguments": arguments,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
# Non-assistant (user) role: emit multimodal parts when present,
|
||||
@@ -688,6 +700,7 @@ def _chat_messages_to_responses_input(
|
||||
items.append({"role": role, "content": content_parts})
|
||||
else:
|
||||
items.append({"role": role, "content": content_text})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
if role == "tool":
|
||||
@@ -722,24 +735,38 @@ def _chat_messages_to_responses_input(
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"output": output_value,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
|
||||
# Native server-side compaction: when a replayed checkpoint is present,
|
||||
# restructure the wire around it. The server renders nothing placed
|
||||
# before a compaction item (live-verified Aug 2026), so pre-checkpoint
|
||||
# history is dead upload weight and — worse — the user's plaintext asks
|
||||
# from before the boundary silently vanish from the model's view. Keep
|
||||
# the newest checkpoint first, retain pre-checkpoint USER messages
|
||||
# verbatim within a token budget (Codex CLI parity), and leave the
|
||||
# history is dead upload weight and — worse — the user's plaintext asks,
|
||||
# and any local-compression summary already merged into that history,
|
||||
# silently vanish from the model's view. Keep the newest checkpoint
|
||||
# first, retain pre-checkpoint USER messages and compression-SUMMARY
|
||||
# messages (whole, never byte-sliced) verbatim within a token budget
|
||||
# each (Codex CLI parity for the user side), and leave the
|
||||
# post-checkpoint tail untouched. Gated on the CURRENT request's native
|
||||
# eligibility, not merely on the presence of a checkpoint: a persisted
|
||||
# checkpoint outlives the gate, and pruning for a request that carries no
|
||||
# ``context_management`` deletes history the server never compacted.
|
||||
#
|
||||
# ``item_sources`` (parallel to ``items``) carries the raw chat message
|
||||
# each converted item came from. A canonical summary carrier's content
|
||||
# can be lost or gone stale by the time it becomes a Responses item — a
|
||||
# merge-into-tail tool-result carrier becomes a typed
|
||||
# ``function_call_output`` (no ``content``/``role`` at all), and a
|
||||
# merge-into-tail assistant carrier can be shadowed by a stale exact
|
||||
# ``codex_message_items`` replay from before the merge rewrote its
|
||||
# content. Pruning reads the source message's own up-to-date,
|
||||
# provenance-tagged content directly instead of trying to recover it
|
||||
# from whatever shape the conversion produced (#90976).
|
||||
if not native_compaction_eligible:
|
||||
return items
|
||||
|
||||
from agent.native_compaction import prune_pre_checkpoint_items
|
||||
|
||||
return prune_pre_checkpoint_items(items)
|
||||
return prune_pre_checkpoint_items(items, item_sources=item_sources)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -32,15 +32,25 @@ captured compaction items ride the existing ``codex_reasoning_items``
|
||||
sidecar, which already handles persistence (state.db), gateway session
|
||||
replay, cross-issuer stamping, and the encrypted-replay kill switch.
|
||||
|
||||
This module is dependency-free on purpose so the transport, adapter, and
|
||||
conversation loop can share the gate without import cycles.
|
||||
This module stays free of transport/adapter dependencies so the transport,
|
||||
adapter, and conversation loop can share the gate without import cycles. The
|
||||
two exceptions — ``agent.context_compressor`` and ``agent.message_content`` —
|
||||
sit below this module in the dependency graph (neither imports
|
||||
``native_compaction``), so importing their provenance/text primitives here
|
||||
introduces no cycle.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from agent.context_compressor import is_compaction_summary_message
|
||||
from agent.message_content import flatten_message_text
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Native compaction fires this many tokens below the local compressor's
|
||||
# trigger so the server always gets the first shot at compaction.
|
||||
LOCAL_TRIGGER_SAFETY_MARGIN = 8_192
|
||||
@@ -147,73 +157,130 @@ def native_compaction_context_management(
|
||||
# Retention budget for plaintext user messages carried across a native
|
||||
# compaction boundary (mirrors Codex CLI's RETAINED_MESSAGE_TOKEN_BUDGET).
|
||||
# Live verification (Aug 2026, gpt-5.6 @ api.openai.com): the server renders
|
||||
# NOTHING placed before a replayed compaction checkpoint — a fact stated in a
|
||||
# pre-checkpoint input item is invisible to the model ("NONE" recall), while
|
||||
# the same item placed after the checkpoint recalls perfectly. Without
|
||||
# retention, every plaintext user ask from before the compaction survives
|
||||
# only as whatever the opaque server summary kept — the goal-drift failure
|
||||
# mode. Codex CLI solves this by rebuilding history with user messages
|
||||
# retained verbatim; ``prune_pre_checkpoint_items`` is our wire-level
|
||||
# equivalent.
|
||||
RETAINED_USER_MESSAGE_TOKEN_BUDGET = 64_000
|
||||
|
||||
# Retention budget for local compression summary messages carried across a native
|
||||
# compaction boundary to prevent summary token inflation.
|
||||
RETAINED_SUMMARY_TOKEN_BUDGET = 32_000
|
||||
|
||||
|
||||
def _approx_tokens(text: str) -> int:
|
||||
"""Cheap chars//4 token estimate — same shape Codex uses for retention."""
|
||||
return max(1, len(text) // 4)
|
||||
|
||||
|
||||
def _user_item_text(item: Dict[str, Any]) -> Optional[str]:
|
||||
"""Extract the retained-budget text of a user-role input item.
|
||||
def _extract_item_text(item: Any) -> Optional[str]:
|
||||
"""Extract measurable text from string, list content, output_text, or nested metadata text.
|
||||
|
||||
Returns None when the item carries no measurable text (empty message).
|
||||
Multimodal list content is measured by its ``input_text`` parts; images
|
||||
count as zero, matching Codex's retention accounting.
|
||||
Returns None when the item carries no measurable text.
|
||||
Handles string content, multipart lists (input_text/text/output_text), and fallback keys.
|
||||
"""
|
||||
if not isinstance(item, dict):
|
||||
return None
|
||||
|
||||
content = item.get("content")
|
||||
if content is None and "output_text" in item:
|
||||
content = item.get("output_text")
|
||||
|
||||
if isinstance(content, str):
|
||||
return content if content.strip() else None
|
||||
|
||||
if isinstance(content, list):
|
||||
text = "".join(
|
||||
part.get("text", "")
|
||||
for part in content
|
||||
if isinstance(part, dict) and part.get("type") == "input_text"
|
||||
)
|
||||
return text if text.strip() or content else None
|
||||
parts = []
|
||||
for part in content:
|
||||
if isinstance(part, str):
|
||||
if part.strip():
|
||||
parts.append(part.strip())
|
||||
elif isinstance(part, dict):
|
||||
part_text = part.get("text") or part.get("input_text") or part.get("output_text")
|
||||
if isinstance(part_text, str) and part_text.strip():
|
||||
parts.append(part_text.strip())
|
||||
part_meta = part.get("metadata")
|
||||
if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str):
|
||||
if part_meta["text"].strip():
|
||||
parts.append(part_meta["text"].strip())
|
||||
text = " ".join(parts)
|
||||
return text if text.strip() else None
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _is_summary_item(item: Any) -> bool:
|
||||
"""True when *item* is a canonical Hermes compression-summary message.
|
||||
|
||||
Delegates entirely to
|
||||
``agent.context_compressor.is_compaction_summary_message`` — the single
|
||||
authoritative provenance check already used by every other summary
|
||||
consumer (memory providers, frontends, the compactor itself). It prefers
|
||||
the exact, truthy ``COMPRESSED_SUMMARY_METADATA_KEY`` marker and falls
|
||||
back to the canonical prefix classifier (``SUMMARY_PREFIX`` /
|
||||
``LEGACY_SUMMARY_PREFIX`` / historical prefixes, including the
|
||||
merge-into-tail shape) for the case where the underscore-prefixed key
|
||||
was already stripped by a wire sanitizer.
|
||||
|
||||
Deliberately NOT a second heuristic: no arbitrary underscore-key scan, no
|
||||
inference from a falsy or unrelated metadata key, and no matching on
|
||||
ad-hoc content headings like ``"## Summary"`` in ordinary text — any of
|
||||
those can promote a normal user/assistant message (or adversarial
|
||||
content) to durable retained history (#90975 review).
|
||||
"""
|
||||
return is_compaction_summary_message(item)
|
||||
|
||||
|
||||
def prune_pre_checkpoint_items(
|
||||
items: List[Dict[str, Any]],
|
||||
retained_user_token_budget: int = RETAINED_USER_MESSAGE_TOKEN_BUDGET,
|
||||
retained_summary_token_budget: int = RETAINED_SUMMARY_TOKEN_BUDGET,
|
||||
enable_summary_retention: bool = True,
|
||||
item_sources: Optional[List[Any]] = None,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Restructure Responses input around the newest compaction checkpoint.
|
||||
|
||||
The server drops every input item that precedes a replayed ``compaction``
|
||||
item (live-verified Aug 2026), so sending pre-checkpoint history is dead
|
||||
weight AND silently erases the user's plaintext asks. When a checkpoint
|
||||
is present, rebuild the wire as::
|
||||
weight AND silently erases the user's plaintext asks — including any
|
||||
local-compression summary the agent already produced, which previously
|
||||
vanished here because it carries ``role="assistant"``, not ``"user"``
|
||||
(#90975). When a checkpoint is present, rebuild the wire as::
|
||||
|
||||
[checkpoint run] + [retained user messages (newest-first budget)] + [post]
|
||||
[checkpoint run] + [retained user & summary messages (newest-first budget)] + [post]
|
||||
|
||||
- The NEWEST contiguous run of checkpoints wins (the server can emit
|
||||
more than one compaction item in a single response — live-observed
|
||||
Aug 2026 — and they arrive adjacent; a run from a newer response
|
||||
cumulatively carries prior windows, so older runs are dropped).
|
||||
- Retained user messages are the user-role items from before the
|
||||
checkpoint, kept verbatim newest-first within
|
||||
- The NEWEST contiguous run of checkpoints wins.
|
||||
- Retained user messages are kept verbatim within
|
||||
``retained_user_token_budget``; the boundary message is head-truncated
|
||||
when it only partially fits (string content only).
|
||||
- Everything after the checkpoint is untouched, so function_call /
|
||||
function_call_output pairing is preserved (a checkpoint is captured on
|
||||
an assistant response, and that response's own calls and their outputs
|
||||
are all emitted after its reasoning items).
|
||||
- No checkpoint in ``items`` → returned unchanged (self-gating: non-native
|
||||
routes and kill-switched sessions never see a restructured wire).
|
||||
|
||||
Deterministic for a given history, so the request prefix stays stable
|
||||
across turns and server-side prompt caching keeps working.
|
||||
when it only partially fits (string content only) — goals are usually
|
||||
stated up front, so the head is the valuable end.
|
||||
- Compression summary messages (``_is_summary_item``, the canonical
|
||||
``agent.context_compressor`` provenance check) are retained whole
|
||||
within ``retained_summary_token_budget``. A summary is never
|
||||
byte/character-sliced: Hermes summaries carry structural framing
|
||||
(handoff prefix, end marker, merge-into-tail delimiters) that a blind
|
||||
slice can corrupt, so one that doesn't fit whole is dropped instead.
|
||||
A summary already retained once (identical text) is never duplicated,
|
||||
so repeated checkpoints stay idempotent.
|
||||
- ``enable_summary_retention`` is a function-level override (used by
|
||||
tests and callers that need the pre-#90975 behavior back); it is not
|
||||
wired to a user-facing config surface.
|
||||
- Original relative chronological order between user messages and
|
||||
summaries is preserved.
|
||||
- ``item_sources`` (optional, parallel to ``items``) is the raw chat
|
||||
message each Responses item was converted from. By the time a summary
|
||||
reaches this function as a converted ``item`` it can already be lossy:
|
||||
a merge-into-tail tool-result carrier becomes a typed
|
||||
``function_call_output`` (no ``content``/``role`` survives the
|
||||
conversion at all), and a merge-into-tail assistant carrier can be
|
||||
shadowed by a stale exact ``codex_message_items`` replay captured
|
||||
before the merge rewrote its content. When a source is provided and is
|
||||
itself a canonical summary carrier (``is_compaction_summary_message``),
|
||||
its content is read directly from the source — never from the
|
||||
converted item — and it is retained as a synthesized
|
||||
``role="assistant"`` message regardless of what shape the original
|
||||
item took. Without ``item_sources`` (default), retention only sees
|
||||
what survived conversion, matching pre-#90976 behavior (#90976).
|
||||
"""
|
||||
if not isinstance(items, list) or not items:
|
||||
return items
|
||||
|
||||
last_cp = None
|
||||
for i, item in enumerate(items):
|
||||
if isinstance(item, dict) and item.get("type") == "compaction":
|
||||
@@ -234,36 +301,105 @@ def prune_pre_checkpoint_items(
|
||||
checkpoint_run = items[first_cp : last_cp + 1]
|
||||
post = items[last_cp + 1 :]
|
||||
|
||||
if isinstance(item_sources, list) and len(item_sources) == len(items):
|
||||
pre_sources: List[Any] = item_sources[:first_cp]
|
||||
else:
|
||||
pre_sources = [None] * len(pre)
|
||||
|
||||
retained_reversed: List[Dict[str, Any]] = []
|
||||
remaining = max(0, int(retained_user_token_budget))
|
||||
for item in reversed(pre):
|
||||
if not isinstance(item, dict) or item.get("role") != "user":
|
||||
user_remaining = max(0, int(retained_user_token_budget))
|
||||
summary_remaining = max(0, int(retained_summary_token_budget))
|
||||
seen_summary_texts: set = set()
|
||||
|
||||
def _try_retain_summary(text: Optional[str]) -> Optional[Dict[str, Any]]:
|
||||
"""Check budget/dedup/cost for a summary; return cost info or None."""
|
||||
if not text or summary_remaining <= 0 or text in seen_summary_texts:
|
||||
return None
|
||||
cost = _approx_tokens(text)
|
||||
if cost > summary_remaining:
|
||||
# Never byte-slice a summary's structural framing — drop it
|
||||
# whole rather than corrupt the handoff prefix / end marker.
|
||||
return None
|
||||
seen_summary_texts.add(text)
|
||||
return {"cost": cost}
|
||||
|
||||
for item, source in zip(reversed(pre), reversed(pre_sources)):
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
# Skip typed items (function_call_output etc. never carry role=user,
|
||||
# but stay defensive about future shapes).
|
||||
|
||||
# Canonical source-based summary detection: reads the ORIGINAL chat
|
||||
# message's own content, so it sees past a lossy conversion (a
|
||||
# typed `function_call_output` wrapper, or a stale exact-replay
|
||||
# message) that erased the summary from `item` itself (#90976).
|
||||
# This is never a heuristic promotion of arbitrary item content —
|
||||
# it only fires when the source message itself is a canonical,
|
||||
# provenance-tagged summary carrier.
|
||||
if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source):
|
||||
text = flatten_message_text(source.get("content")) if isinstance(source, dict) else ""
|
||||
text = text if text.strip() else None
|
||||
result = _try_retain_summary(text)
|
||||
if result:
|
||||
_src_role = source.get("role")
|
||||
retained_reversed.append({
|
||||
"role": _src_role if _src_role in ("user", "assistant") else "assistant",
|
||||
"content": text,
|
||||
})
|
||||
summary_remaining -= result["cost"]
|
||||
continue
|
||||
|
||||
# Skip typed non-message items (function_call_output etc. never
|
||||
# carry role=user or a summary flag, but stay defensive about
|
||||
# future shapes).
|
||||
if "type" in item and item.get("type") != "message":
|
||||
continue
|
||||
if remaining <= 0:
|
||||
break
|
||||
text = _user_item_text(item)
|
||||
|
||||
is_summary = enable_summary_retention and _is_summary_item(item)
|
||||
is_user = item.get("role") == "user"
|
||||
|
||||
if not is_user and not is_summary:
|
||||
continue
|
||||
|
||||
text = _extract_item_text(item)
|
||||
if text is None:
|
||||
continue
|
||||
cost = _approx_tokens(text)
|
||||
if cost <= remaining:
|
||||
retained_reversed.append(item)
|
||||
remaining -= cost
|
||||
elif isinstance(item.get("content"), str):
|
||||
# Head-truncate the boundary message: goals are usually stated
|
||||
# up front, so the head is the valuable end.
|
||||
truncated = dict(item)
|
||||
truncated["content"] = item["content"][: remaining * 4]
|
||||
if truncated["content"].strip():
|
||||
retained_reversed.append(truncated)
|
||||
remaining = 0
|
||||
# Multimodal boundary message that doesn't fit whole: skip rather
|
||||
# than rewrite parts.
|
||||
# Image-only user messages have empty text but non-empty content —
|
||||
# main retains them at 1-token cost (images count as zero, matching
|
||||
# Codex's retention accounting). Don't skip them just because text
|
||||
# is falsy.
|
||||
if not text and not is_user:
|
||||
continue
|
||||
|
||||
return checkpoint_run + list(reversed(retained_reversed)) + post
|
||||
if is_summary:
|
||||
result = _try_retain_summary(text)
|
||||
if result:
|
||||
retained_reversed.append(item)
|
||||
summary_remaining -= result["cost"]
|
||||
elif is_user:
|
||||
if user_remaining <= 0:
|
||||
continue
|
||||
cost = _approx_tokens(text)
|
||||
if cost <= user_remaining:
|
||||
retained_reversed.append(item)
|
||||
user_remaining -= cost
|
||||
elif isinstance(item.get("content"), str):
|
||||
truncated = dict(item)
|
||||
truncated["content"] = item["content"][: user_remaining * 4]
|
||||
if truncated["content"].strip():
|
||||
retained_reversed.append(truncated)
|
||||
user_remaining = 0
|
||||
|
||||
retained_ordered = list(reversed(retained_reversed))
|
||||
result = checkpoint_run + retained_ordered + post
|
||||
|
||||
logger.debug(
|
||||
"Pruned pre-checkpoint items: %d input -> %d retained (user_rem=%d, summary_rem=%d)",
|
||||
len(items),
|
||||
len(result),
|
||||
user_remaining,
|
||||
summary_remaining,
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool:
|
||||
|
||||
394
tests/run_agent/test_native_compaction_summary_retention.py
Normal file
394
tests/run_agent/test_native_compaction_summary_retention.py
Normal file
@@ -0,0 +1,394 @@
|
||||
"""Tests for native compaction summary retention during pre-checkpoint pruning (#90975).
|
||||
|
||||
``prune_pre_checkpoint_items`` previously dropped every pre-checkpoint item
|
||||
whose ``role`` was not ``"user"`` — which silently deleted Hermes' own local
|
||||
compression summaries (``role="assistant"``) from the wire on every native
|
||||
compaction turn. These tests cover the fix's summary retention path, its
|
||||
reliance on the canonical ``agent.context_compressor`` provenance check (not
|
||||
an ad-hoc heuristic), whole-or-drop truncation, and idempotency.
|
||||
"""
|
||||
|
||||
from agent.context_compressor import (
|
||||
COMPRESSED_SUMMARY_METADATA_KEY,
|
||||
ContextCompressor,
|
||||
SUMMARY_PREFIX,
|
||||
_MERGED_PRIOR_CONTEXT_HEADER,
|
||||
_MERGED_SUMMARY_DELIMITER,
|
||||
_SUMMARY_END_MARKER,
|
||||
)
|
||||
from agent.native_compaction import (
|
||||
_extract_item_text,
|
||||
_is_summary_item,
|
||||
prune_pre_checkpoint_items,
|
||||
)
|
||||
|
||||
|
||||
def _standalone_summary_content(body: str = "## Active Task\nstuff") -> str:
|
||||
return f"{SUMMARY_PREFIX}\n{body}\n\n{_SUMMARY_END_MARKER}"
|
||||
|
||||
|
||||
def _merged_summary_content(tail: str = "preserved prior turn") -> str:
|
||||
return (
|
||||
f"{_MERGED_PRIOR_CONTEXT_HEADER}\n{tail}\n\n"
|
||||
f"{_MERGED_SUMMARY_DELIMITER}\n\n"
|
||||
f"{SUMMARY_PREFIX}\nbody\n\n{_SUMMARY_END_MARKER}"
|
||||
)
|
||||
|
||||
|
||||
class TestIsSummaryItemCanonical:
|
||||
"""`_is_summary_item` must delegate to the canonical provenance check —
|
||||
exact metadata flag or the canonical prefix classifier — never an
|
||||
ad-hoc heuristic (#90975 blocking review)."""
|
||||
|
||||
def test_truthy_metadata_flag_detected(self):
|
||||
assert _is_summary_item({COMPRESSED_SUMMARY_METADATA_KEY: True}) is True
|
||||
|
||||
def test_standalone_content_detected_without_metadata(self):
|
||||
# The wire sanitizers strip underscore keys, so content-only
|
||||
# detection must still work on the canonical prefix.
|
||||
assert _is_summary_item({"role": "assistant", "content": _standalone_summary_content()}) is True
|
||||
|
||||
def test_merged_content_detected_without_metadata(self):
|
||||
assert _is_summary_item({"role": "assistant", "content": _merged_summary_content()}) is True
|
||||
|
||||
def test_malformed_inputs_are_not_summaries(self):
|
||||
assert _is_summary_item(None) is False
|
||||
assert _is_summary_item(123) is False
|
||||
assert _is_summary_item({}) is False
|
||||
|
||||
|
||||
class TestIsSummaryItemNegativeWitnesses:
|
||||
"""Content that merely resembles a summary must never be promoted to
|
||||
durable retained history — that is authority drift (#90975 blocking
|
||||
review, required item 4)."""
|
||||
|
||||
def test_summary_heading_in_ordinary_user_text_is_not_a_summary(self):
|
||||
item = {"role": "user", "content": "## Summary\nplease summarize the PR for me"}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_false_valued_metadata_flag_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "hi", COMPRESSED_SUMMARY_METADATA_KEY: False}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_arbitrary_underscore_summary_key_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "hi", "_my_custom_summary_flag": True}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_non_hermes_assistant_content_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "Conversation Summary: I finished the task."}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
|
||||
class TestExtractItemTextVariations:
|
||||
def test_string_content(self):
|
||||
assert _extract_item_text({"content": "Hello world"}) == "Hello world"
|
||||
|
||||
def test_multipart_list_content(self):
|
||||
item = {
|
||||
"content": [
|
||||
{"type": "input_text", "text": "Part 1"},
|
||||
{"type": "text", "text": "Part 2"},
|
||||
{"type": "other", "output_text": "Part 3"},
|
||||
]
|
||||
}
|
||||
assert _extract_item_text(item) == "Part 1 Part 2 Part 3"
|
||||
|
||||
def test_output_text_fallback(self):
|
||||
assert _extract_item_text({"output_text": "Output fallback"}) == "Output fallback"
|
||||
|
||||
def test_malformed_or_empty(self):
|
||||
assert _extract_item_text({"content": None}) is None
|
||||
assert _extract_item_text({"content": []}) is None
|
||||
assert _extract_item_text(None) is None
|
||||
assert _extract_item_text("string_item") is None
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsRetainsSummaries:
|
||||
def test_retains_summary_and_user_in_original_order(self):
|
||||
summary_content = _standalone_summary_content("Step 1 complete")
|
||||
items = [
|
||||
{"role": "user", "content": "User Ask 1"},
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "User Ask 2"},
|
||||
{"role": "assistant", "content": "Normal chatter to prune"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "User Ask 3"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_user_token_budget=1000)
|
||||
|
||||
assert pruned[0]["type"] == "compaction"
|
||||
contents = [m.get("content") for m in pruned[1:]]
|
||||
assert contents == [
|
||||
"User Ask 1",
|
||||
summary_content,
|
||||
"User Ask 2",
|
||||
"User Ask 3",
|
||||
]
|
||||
|
||||
def test_role_agnostic_retention_does_not_touch_user_budget(self):
|
||||
summary_content = _standalone_summary_content("x" * 2000)
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "short ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(
|
||||
items, retained_user_token_budget=10, retained_summary_token_budget=10_000
|
||||
)
|
||||
|
||||
contents = [m.get("content") for m in pruned]
|
||||
assert summary_content in contents
|
||||
assert "short ask" in contents
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsSummaryBudget:
|
||||
def test_oversized_summary_is_dropped_whole_not_sliced(self):
|
||||
"""A summary that cannot fit the remaining budget is dropped
|
||||
entirely rather than character-sliced (#90975 blocking review,
|
||||
required item 3): slicing can corrupt the handoff prefix / end
|
||||
marker that keeps the summary non-active."""
|
||||
long_summary = _standalone_summary_content("Summary line " * 500)
|
||||
items = [
|
||||
{"role": "assistant", "content": long_summary, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_summary_token_budget=100)
|
||||
|
||||
assert not any(m.get(COMPRESSED_SUMMARY_METADATA_KEY) for m in pruned)
|
||||
|
||||
def test_summary_that_fits_budget_is_retained_whole(self):
|
||||
summary_content = _standalone_summary_content("short body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_summary_token_budget=10_000)
|
||||
|
||||
retained = [m for m in pruned if m.get(COMPRESSED_SUMMARY_METADATA_KEY)]
|
||||
assert len(retained) == 1
|
||||
assert retained[0]["content"] == summary_content
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsIdempotency:
|
||||
def test_duplicate_summary_text_is_not_retained_twice(self):
|
||||
"""A repeated checkpoint sequence can leave the same summary text
|
||||
present at more than one pre-checkpoint position; retention must
|
||||
stay idempotent rather than duplicate it (#90975 blocking review,
|
||||
required item 5)."""
|
||||
summary_content = _standalone_summary_content("same body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "mid ask"},
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
|
||||
matches = [m for m in pruned if m.get("content") == summary_content]
|
||||
assert len(matches) == 1
|
||||
|
||||
def test_re_pruning_an_already_pruned_result_is_stable(self):
|
||||
summary_content = _standalone_summary_content("stable body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
]
|
||||
|
||||
once = prune_pre_checkpoint_items(items)
|
||||
twice = prune_pre_checkpoint_items(once)
|
||||
assert once == twice
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsLiveCompressorEmissions:
|
||||
"""Exercise the real ``ContextCompressor`` marker renderer instead of a
|
||||
hand-built stand-in, for both standalone and merge-into-tail shapes
|
||||
(#90975 blocking review, required item 5)."""
|
||||
|
||||
def test_standalone_live_marker_is_retained(self):
|
||||
rendered = ContextCompressor._render_micro_marker_content("Live handoff body")
|
||||
assert ContextCompressor.classify_summary_content(rendered) == "standalone"
|
||||
|
||||
items = [
|
||||
{"role": "assistant", "content": rendered, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert any(m.get("content") == rendered for m in pruned)
|
||||
|
||||
def test_merged_tail_summary_is_retained_and_classified_merged(self):
|
||||
merged = _merged_summary_content("earlier preserved turn text")
|
||||
assert ContextCompressor.classify_summary_content(merged) == "merged"
|
||||
|
||||
items = [
|
||||
{"role": "assistant", "content": merged, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert any(m.get("content") == merged for m in pruned)
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsEnableSummaryRetentionToggle:
|
||||
def test_disabling_summary_retention_drops_pre_checkpoint_summaries(self):
|
||||
summary_content = _standalone_summary_content("Old")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob"},
|
||||
{"role": "user", "content": "New ask"},
|
||||
]
|
||||
|
||||
pruned_disabled = prune_pre_checkpoint_items(items, enable_summary_retention=False)
|
||||
contents = [m.get("content") for m in pruned_disabled]
|
||||
assert summary_content not in contents
|
||||
|
||||
|
||||
def _checkpoint_message(item_id: str = "rs_cp1", blob: str = "cp_blob_1"):
|
||||
"""An assistant message carrying a replayable native-compaction checkpoint."""
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"codex_reasoning_items": [
|
||||
{"type": "compaction", "encrypted_content": blob, "id": item_id},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class TestChatMessagesToResponsesInputSummaryCarrierLoss:
|
||||
"""Adapter-level witnesses for the second blocking review (#90976):
|
||||
``prune_pre_checkpoint_items`` only ever saw whatever ``_is_summary_item``
|
||||
could recover from an already-converted Responses ``item`` — but two
|
||||
real merge-into-tail carrier shapes lose or shadow the summary content
|
||||
during ``_chat_messages_to_responses_input`` itself, *before* pruning
|
||||
ever runs:
|
||||
|
||||
* a tool-result carrier becomes a typed ``function_call_output`` (no
|
||||
``content``/``role`` survive the conversion at all), and
|
||||
* an assistant carrier with a stale ``codex_message_items`` sidecar
|
||||
replays the pre-merge exact message item instead of the rewritten
|
||||
(summary-bearing) ``content``.
|
||||
|
||||
These feed real chat messages, shaped exactly the way
|
||||
``ContextCompressor.compress()`` merge-into-tail produces them (same
|
||||
``COMPRESSED_SUMMARY_METADATA_KEY`` stamp, same merge delimiters/end
|
||||
marker), through the real ``_chat_messages_to_responses_input`` with a
|
||||
replayed checkpoint — not a hand-built Responses item passed straight
|
||||
to the pruner.
|
||||
"""
|
||||
|
||||
def test_tool_result_merge_carrier_summary_survives_the_adapter(self):
|
||||
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
||||
|
||||
merged = _merged_summary_content("preserved tool context")
|
||||
messages = [
|
||||
{"role": "user", "content": "please do the thing"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "do_thing", "arguments": "{}"},
|
||||
}],
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "call_1",
|
||||
"content": merged,
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
},
|
||||
_checkpoint_message(),
|
||||
{"role": "user", "content": "next ask after checkpoint"},
|
||||
]
|
||||
|
||||
items = _chat_messages_to_responses_input(
|
||||
messages, native_compaction_eligible=True,
|
||||
)
|
||||
|
||||
# The summary survives, exactly once, as a plain message item —
|
||||
# never as a `function_call_output` (which the pruner cannot see,
|
||||
# and which would orphan the dropped `function_call` it used to
|
||||
# pair with).
|
||||
assert not any(
|
||||
isinstance(it, dict) and it.get("type") == "function_call_output"
|
||||
for it in items
|
||||
)
|
||||
matches = [
|
||||
it for it in items
|
||||
if isinstance(it, dict) and _extract_item_text(it) == merged
|
||||
]
|
||||
assert len(matches) == 1
|
||||
assert matches[0].get("type") != "function_call_output"
|
||||
|
||||
# And the newest checkpoint still leads the wire.
|
||||
assert items[0].get("type") == "compaction"
|
||||
|
||||
def test_assistant_merge_carrier_with_stale_replay_summary_survives(self):
|
||||
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
||||
|
||||
merged = _merged_summary_content("preserved assistant context")
|
||||
messages = [
|
||||
{"role": "user", "content": "question"},
|
||||
{
|
||||
"role": "assistant",
|
||||
# Rewritten by the compressor merge — this is what must
|
||||
# reach the wire.
|
||||
"content": merged,
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
# Stale sidecar captured BEFORE the merge rewrote the
|
||||
# content above. The exact-replay path prefers this over
|
||||
# `content` for prefix-cache continuity, which is exactly
|
||||
# what shadows the summary (#90976).
|
||||
"codex_message_items": [{
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"id": "msg_stale_1",
|
||||
"status": "completed",
|
||||
"content": [{"type": "output_text", "text": "stale pre-merge answer"}],
|
||||
}],
|
||||
},
|
||||
_checkpoint_message(),
|
||||
{"role": "user", "content": "next ask"},
|
||||
]
|
||||
|
||||
items = _chat_messages_to_responses_input(
|
||||
messages, native_compaction_eligible=True,
|
||||
)
|
||||
|
||||
assert not any(
|
||||
isinstance(it, dict) and _extract_item_text(it) == "stale pre-merge answer"
|
||||
for it in items
|
||||
)
|
||||
matches = [
|
||||
it for it in items
|
||||
if isinstance(it, dict) and _extract_item_text(it) == merged
|
||||
]
|
||||
assert len(matches) == 1
|
||||
assert items[0].get("type") == "compaction"
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsMalformedInputs:
|
||||
def test_handles_none_non_dict_and_empty_items_safely(self):
|
||||
assert prune_pre_checkpoint_items(None) is None
|
||||
assert prune_pre_checkpoint_items([]) == []
|
||||
|
||||
items = [
|
||||
None,
|
||||
123,
|
||||
"raw_string",
|
||||
{"role": "user", "content": "Valid user ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert len(pruned) == 2
|
||||
assert pruned[0]["type"] == "compaction"
|
||||
assert pruned[1]["content"] == "Valid user ask"
|
||||
Reference in New Issue
Block a user