fix(compression): opaque encrypted_content costs 0 in every local token estimate

Codex Responses reasoning and compaction items carry ciphertext the provider prices by its own
token count, never by bytes; a single native compaction checkpoint is ~5M chars, which the
bytes/4 estimator turned into ~1.29M "tokens" against a 204K threshold (#100611). #104192 deferred
that decision for one request; this removes the mis-pricing at the source so the preflight
estimator and the tail-budget walk agree (a mismatched size class protects blob-heavy rows as
"small" and compaction re-fires). Only real usage prices these items, and the usage anchor carries
that price forward.

evals/native_compaction/ab_checkpoint_preflight.py: preflight estimate after checkpoint
1,292,413 -> 58 rough tokens; the over-threshold negative arm still compresses.
This commit is contained in:
Teknium
2026-09-06 12:41:52 -07:00
parent 0f4587e336
commit 562e6e4824
4 changed files with 38 additions and 18 deletions

View File

@@ -25,7 +25,8 @@ from agent.context_engine import ContextEngine, sanitize_memory_context
from agent.error_classifier import FailoverReason, classify_api_error
from agent.micro_compaction import MicroCompactionMixin
from agent.model_metadata import (
MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough
MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough,
strip_opaque_replay_items,
)
from agent.redact import redact_sensitive_text
from agent.turn_context import drop_stale_api_content
@@ -1132,7 +1133,8 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
tokens = text_tokens + 10 # +10 for role/key overhead
tokens += sum(estimate_tokens_rough(str(tc)) for tc in msg.get("tool_calls") or [] if isinstance(tc, dict))
for key in _ALWAYS_REPLAYED_BUDGET_KEYS:
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
# Opaque ciphertext is priced only by real usage (same rule as the preflight estimator).
tokens += _serialized_length_for_budget(strip_opaque_replay_items(msg.get(key))) // _CHARS_PER_TOKEN
if not charge_stale_thinking:
return tokens
# Wire ships at most ONE generic thinking key (reasoning_content wins);

View File

@@ -2079,6 +2079,18 @@ def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
return count * cost_per_image
def strip_opaque_replay_items(items: Any) -> Any:
"""``codex_reasoning_items`` with ``encrypted_content`` blanked for local token estimation.
The ciphertext is priced by the provider's own count, never by its bytes (a compaction
checkpoint alone can be 5M chars, #100611); only real usage prices it."""
if not isinstance(items, list):
return items
return [
{k: ("" if k == "encrypted_content" else v) for k, v in item.items()} if isinstance(item, dict) else item
for item in items
]
def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
"""Shadow of a message holding only what the provider actually receives.
* ``api_content`` SUBSTITUTES ``content`` (mirrors ``turn_context.substitute_api_content`` exactly):
@@ -2086,7 +2098,11 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
other shape would UNDERcount — the dangerous direction.
* Base64 images become a placeholder; ``_count_image_tokens`` charges them flat.
* ``reasoning`` never ships as-is (request builds pop it after optionally promoting it into
``reasoning_content``); counting both inflated estimates up to +53%."""
``reasoning_content``); counting both inflated estimates up to +53%.
* Opaque provider blobs (``encrypted_content`` on codex reasoning / compaction items) are
ciphertext the provider prices by its OWN token count, never by bytes; a native compaction
checkpoint alone can be 5M chars (#100611). They contribute 0 here: only real usage ever
prices them, and the usage anchor carries that price forward."""
sidecar = msg.get("api_content")
sidecar_wins = isinstance(sidecar, str) and bool(sidecar) and msg.get("role") in ("user", "assistant")
_rc = msg.get("reasoning_content")
@@ -2108,6 +2124,10 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
]
elif k == "content" and isinstance(v, dict) and v.get("_multimodal"):
shadow[k] = v.get("text_summary", "")
elif k == "codex_reasoning_items":
shadow[k] = strip_opaque_replay_items(v)
elif k == "encrypted_content": # a Responses reasoning/compaction item passed as a row
shadow[k] = ""
else:
shadow[k] = v
return shadow

View File

@@ -73,18 +73,16 @@ class TestBudgetExcludesReasoningDetailsEnvelope:
f"thinking prose charged twice (delta {both - once} tokens)"
)
def test_codex_reasoning_items_still_charged(self):
"""Codex Responses genuinely replays these on every request (#55572);
the exclusion is scoped to reasoning_details only."""
def test_codex_encrypted_content_is_opaque_to_local_estimates(self):
"""Ciphertext is priced by the provider's own count, never by its bytes (a native
compaction checkpoint alone is ~5M chars, #100611): the budget walk and the trigger
estimator both charge it ~0, so only real usage ever prices it (#104462)."""
from agent.model_metadata import estimate_messages_tokens_rough
base = {"role": "assistant", "content": "hi"}
loaded = dict(
base,
codex_reasoning_items=[{"encrypted_content": "e" * 20_000}],
)
assert (
_estimate_msg_budget_tokens(loaded) - _estimate_msg_budget_tokens(base)
> 4_000
)
loaded = dict(base, codex_reasoning_items=[{"type": "reasoning", "encrypted_content": "e" * 20_000}])
assert _estimate_msg_budget_tokens(loaded) - _estimate_msg_budget_tokens(base) < 50
assert estimate_messages_tokens_rough([loaded]) - estimate_messages_tokens_rough([base]) < 50
class TestReasoningDetailsTextChars:

View File

@@ -44,9 +44,9 @@ class TestChargeStaleThinking:
# The delta is the thinking text — a substantial chunk, not noise.
assert full - stale > 300
def test_codex_sidecar_always_charged(self):
"""Wire-replayed Codex blobs (incl. native compaction checkpoints)
must stay in the budget even for stale turns — #55572's invariant."""
def test_codex_sidecar_is_not_thinking_text(self):
"""The Codex sidecar is not stale thinking: the stale path drops nothing from it. Its
ciphertext is priced only by real usage (#104462), so the row costs the same as a bare one."""
msg = _assistant(codex=True)
full = _estimate_msg_budget_tokens(msg, charge_stale_thinking=True)
stale = _estimate_msg_budget_tokens(msg, charge_stale_thinking=False)
@@ -54,7 +54,7 @@ class TestChargeStaleThinking:
bare = _estimate_msg_budget_tokens(
{"role": "assistant", "content": "done"}, charge_stale_thinking=False
)
assert stale > bare + 500 # blob still charged on the stale path
assert stale < bare + 100
def test_default_is_conservative_full_charge(self):
msg = _assistant(thinking=True)