fix(compression): opaque encrypted_content costs 0 in every local token estimate
Codex Responses reasoning and compaction items carry ciphertext the provider prices by its own token count, never by bytes; a single native compaction checkpoint is ~5M chars, which the bytes/4 estimator turned into ~1.29M "tokens" against a 204K threshold (#100611). #104192 deferred that decision for one request; this removes the mis-pricing at the source so the preflight estimator and the tail-budget walk agree (a mismatched size class protects blob-heavy rows as "small" and compaction re-fires). Only real usage prices these items, and the usage anchor carries that price forward. evals/native_compaction/ab_checkpoint_preflight.py: preflight estimate after checkpoint 1,292,413 -> 58 rough tokens; the over-threshold negative arm still compresses.
This commit is contained in:
@@ -25,7 +25,8 @@ from agent.context_engine import ContextEngine, sanitize_memory_context
|
||||
from agent.error_classifier import FailoverReason, classify_api_error
|
||||
from agent.micro_compaction import MicroCompactionMixin
|
||||
from agent.model_metadata import (
|
||||
MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough
|
||||
MINIMUM_CONTEXT_LENGTH, get_model_context_length, estimate_messages_tokens_rough, estimate_tokens_rough,
|
||||
strip_opaque_replay_items,
|
||||
)
|
||||
from agent.redact import redact_sensitive_text
|
||||
from agent.turn_context import drop_stale_api_content
|
||||
@@ -1132,7 +1133,8 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
|
||||
tokens = text_tokens + 10 # +10 for role/key overhead
|
||||
tokens += sum(estimate_tokens_rough(str(tc)) for tc in msg.get("tool_calls") or [] if isinstance(tc, dict))
|
||||
for key in _ALWAYS_REPLAYED_BUDGET_KEYS:
|
||||
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
|
||||
# Opaque ciphertext is priced only by real usage (same rule as the preflight estimator).
|
||||
tokens += _serialized_length_for_budget(strip_opaque_replay_items(msg.get(key))) // _CHARS_PER_TOKEN
|
||||
if not charge_stale_thinking:
|
||||
return tokens
|
||||
# Wire ships at most ONE generic thinking key (reasoning_content wins);
|
||||
|
||||
@@ -2079,6 +2079,18 @@ def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
|
||||
return count * cost_per_image
|
||||
|
||||
|
||||
def strip_opaque_replay_items(items: Any) -> Any:
|
||||
"""``codex_reasoning_items`` with ``encrypted_content`` blanked for local token estimation.
|
||||
The ciphertext is priced by the provider's own count, never by its bytes (a compaction
|
||||
checkpoint alone can be 5M chars, #100611); only real usage prices it."""
|
||||
if not isinstance(items, list):
|
||||
return items
|
||||
return [
|
||||
{k: ("" if k == "encrypted_content" else v) for k, v in item.items()} if isinstance(item, dict) else item
|
||||
for item in items
|
||||
]
|
||||
|
||||
|
||||
def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Shadow of a message holding only what the provider actually receives.
|
||||
* ``api_content`` SUBSTITUTES ``content`` (mirrors ``turn_context.substitute_api_content`` exactly):
|
||||
@@ -2086,7 +2098,11 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
|
||||
other shape would UNDERcount — the dangerous direction.
|
||||
* Base64 images become a placeholder; ``_count_image_tokens`` charges them flat.
|
||||
* ``reasoning`` never ships as-is (request builds pop it after optionally promoting it into
|
||||
``reasoning_content``); counting both inflated estimates up to +53%."""
|
||||
``reasoning_content``); counting both inflated estimates up to +53%.
|
||||
* Opaque provider blobs (``encrypted_content`` on codex reasoning / compaction items) are
|
||||
ciphertext the provider prices by its OWN token count, never by bytes; a native compaction
|
||||
checkpoint alone can be 5M chars (#100611). They contribute 0 here: only real usage ever
|
||||
prices them, and the usage anchor carries that price forward."""
|
||||
sidecar = msg.get("api_content")
|
||||
sidecar_wins = isinstance(sidecar, str) and bool(sidecar) and msg.get("role") in ("user", "assistant")
|
||||
_rc = msg.get("reasoning_content")
|
||||
@@ -2108,6 +2124,10 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
|
||||
]
|
||||
elif k == "content" and isinstance(v, dict) and v.get("_multimodal"):
|
||||
shadow[k] = v.get("text_summary", "")
|
||||
elif k == "codex_reasoning_items":
|
||||
shadow[k] = strip_opaque_replay_items(v)
|
||||
elif k == "encrypted_content": # a Responses reasoning/compaction item passed as a row
|
||||
shadow[k] = ""
|
||||
else:
|
||||
shadow[k] = v
|
||||
return shadow
|
||||
|
||||
@@ -73,18 +73,16 @@ class TestBudgetExcludesReasoningDetailsEnvelope:
|
||||
f"thinking prose charged twice (delta {both - once} tokens)"
|
||||
)
|
||||
|
||||
def test_codex_reasoning_items_still_charged(self):
|
||||
"""Codex Responses genuinely replays these on every request (#55572);
|
||||
the exclusion is scoped to reasoning_details only."""
|
||||
def test_codex_encrypted_content_is_opaque_to_local_estimates(self):
|
||||
"""Ciphertext is priced by the provider's own count, never by its bytes (a native
|
||||
compaction checkpoint alone is ~5M chars, #100611): the budget walk and the trigger
|
||||
estimator both charge it ~0, so only real usage ever prices it (#104462)."""
|
||||
from agent.model_metadata import estimate_messages_tokens_rough
|
||||
|
||||
base = {"role": "assistant", "content": "hi"}
|
||||
loaded = dict(
|
||||
base,
|
||||
codex_reasoning_items=[{"encrypted_content": "e" * 20_000}],
|
||||
)
|
||||
assert (
|
||||
_estimate_msg_budget_tokens(loaded) - _estimate_msg_budget_tokens(base)
|
||||
> 4_000
|
||||
)
|
||||
loaded = dict(base, codex_reasoning_items=[{"type": "reasoning", "encrypted_content": "e" * 20_000}])
|
||||
assert _estimate_msg_budget_tokens(loaded) - _estimate_msg_budget_tokens(base) < 50
|
||||
assert estimate_messages_tokens_rough([loaded]) - estimate_messages_tokens_rough([base]) < 50
|
||||
|
||||
|
||||
class TestReasoningDetailsTextChars:
|
||||
|
||||
@@ -44,9 +44,9 @@ class TestChargeStaleThinking:
|
||||
# The delta is the thinking text — a substantial chunk, not noise.
|
||||
assert full - stale > 300
|
||||
|
||||
def test_codex_sidecar_always_charged(self):
|
||||
"""Wire-replayed Codex blobs (incl. native compaction checkpoints)
|
||||
must stay in the budget even for stale turns — #55572's invariant."""
|
||||
def test_codex_sidecar_is_not_thinking_text(self):
|
||||
"""The Codex sidecar is not stale thinking: the stale path drops nothing from it. Its
|
||||
ciphertext is priced only by real usage (#104462), so the row costs the same as a bare one."""
|
||||
msg = _assistant(codex=True)
|
||||
full = _estimate_msg_budget_tokens(msg, charge_stale_thinking=True)
|
||||
stale = _estimate_msg_budget_tokens(msg, charge_stale_thinking=False)
|
||||
@@ -54,7 +54,7 @@ class TestChargeStaleThinking:
|
||||
bare = _estimate_msg_budget_tokens(
|
||||
{"role": "assistant", "content": "done"}, charge_stale_thinking=False
|
||||
)
|
||||
assert stale > bare + 500 # blob still charged on the stale path
|
||||
assert stale < bare + 100
|
||||
|
||||
def test_default_is_conservative_full_charge(self):
|
||||
msg = _assistant(thinking=True)
|
||||
|
||||
Reference in New Issue
Block a user