fix: exclude display-only message fields from token estimates

Edit tool rows now carry display_metadata.tool_result_metadata.inline_diff
(~9KB of ANSI per edit) on the live message dict. build_api_messages strips
display_kind/display_metadata/_row_id before the wire, but the local
estimator's wire shadow only dropped PERSISTENCE_ONLY_MESSAGE_FIELDS
({"timestamp"}), so estimate_messages_tokens_rough and
estimate_request_tokens_rough priced the diff: one edit row went from 32 to
2527 estimated tokens (100 edits: 3100 -> 252600). That inflates compaction
preflight, post-tool checks and turn-overflow scoring, compacts early and
breaks the prompt cache.

PERSISTENCE_ONLY_MESSAGE_FIELDS (agent/message_metadata.py) is now the single
set of local-only fields: timestamp, display_kind, display_metadata, _row_id.
build_api_messages pops exactly that set and the estimator shadow drops it,
so the two can no longer drift. The iteration-limit summary path
(_iteration_summary_api_messages) hand-builds its wire messages and stripped
timestamp but not display_*; its key set now unions the same constant, so
the inline diff never reaches the provider there either (strict gateways
reject unknown keys). The tail-budget walk (_estimate_msg_budget_tokens) is
allowlist-based and already ignored these fields.
This commit is contained in:
teknium1
2026-09-23 22:37:51 -07:00
committed by Teknium
parent f27a0a937d
commit 444066b82c
4 changed files with 27 additions and 11 deletions

View File

@@ -39,7 +39,7 @@ from agent.gemini_native_adapter import is_native_gemini_base_url
# misidentify and, without an api_key, return 401 on every leg (issue #89863).
from agent.model_metadata import is_local_endpoint
from agent.message_content import flatten_message_text
from agent.message_metadata import append_message, stamp_message_timestamp
from agent.message_metadata import PERSISTENCE_ONLY_MESSAGE_FIELDS, append_message, stamp_message_timestamp
from agent.message_sanitization import (
_sanitize_surrogates, _repair_tool_call_arguments, normalize_finish_reason as _normalize_finish_reason,
sanitize_outbound_kwargs, strip_images_for_rejecting_model,
@@ -2129,8 +2129,8 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None, reset_a
# Keys outside the Chat Completions schema that strict gateways (Fireworks-backed OpenCode
# Go, Mistral, Moonshot/Kimi) reject with 422. The transport's convert_messages() drops them
# in the main loop; the summary path calls chat.completions.create() directly, so mirror it.
_SUMMARY_FOREIGN_MESSAGE_KEYS = ("reasoning", "finish_reason", "tool_name", "codex_reasoning_items",
"codex_message_items", "timestamp", "platform_message_id")
_SUMMARY_FOREIGN_MESSAGE_KEYS = PERSISTENCE_ONLY_MESSAGE_FIELDS | {"reasoning", "finish_reason", "tool_name",
"codex_reasoning_items", "codex_message_items", "platform_message_id"}
_EMPTY_SUMMARY_RESPONSE = "I reached the iteration limit and couldn't generate a summary."

View File

@@ -6,9 +6,12 @@ from time import time as wall_time
from typing import Any, MutableMapping, Optional, TypeVar
# These fields describe Hermes' durable record, not provider-visible message
# content. They must not influence context-pressure decisions.
PERSISTENCE_ONLY_MESSAGE_FIELDS = frozenset({"timestamp"})
# These fields describe Hermes' durable record and timeline display, not
# provider-visible message content. The request builder strips them from every
# outgoing copy and the token estimator ignores them: one set, so an estimate
# never prices bytes the provider never receives (an edit's inline_diff in
# display_metadata is ~9KB and would trigger premature compaction).
PERSISTENCE_ONLY_MESSAGE_FIELDS = frozenset({"timestamp", "display_kind", "display_metadata", "_row_id"})
_Message = TypeVar("_Message", bound=MutableMapping[str, Any])

View File

@@ -22,7 +22,7 @@ from agent.iteration_budget import IterationBudget
from agent.memory_manager import build_memory_context_block
from agent.memory_provider import is_trivial_prompt
from agent.message_content import flatten_message_text
from agent.message_metadata import append_message, stamp_message_timestamp
from agent.message_metadata import PERSISTENCE_ONLY_MESSAGE_FIELDS, append_message, stamp_message_timestamp
from agent.model_metadata import estimate_messages_tokens_rough, estimate_request_tokens_rough
from agent.image_token_cost import bind_image_token_cost
from agent.usage_anchor import anchored_context_tokens, restore_usage_anchor
@@ -1220,11 +1220,12 @@ def build_api_messages(
# persisted history via nested containers; see _clone_message_for_send.
api_msg = _clone_message_for_send(msg)
# api_content is bookkeeping (exact bytes sent), never a provider field — pop
# it from EVERY outgoing copy. display_* is display-only timeline metadata
# (strict OpenAI backends reject unknown keys); _row_id is the durable row id
# from _rows_to_conversation and only chat-completions strips underscore keys.
# it from EVERY outgoing copy. Persistence/display fields (display_*, _row_id,
# timestamp) are local bookkeeping: strict OpenAI backends reject unknown keys
# and only chat-completions strips underscore keys. The token estimator drops
# the same set, so it never prices bytes the provider never receives.
_api_content = api_msg.pop("api_content", None)
for key in ("display_kind", "display_metadata", "_row_id"):
for key in PERSISTENCE_ONLY_MESSAGE_FIELDS:
api_msg.pop(key, None)
# Inject ephemeral context (memory prefetch + pre_llm_call user hooks)

View File

@@ -23,6 +23,7 @@ from agent.model_metadata import (
_strip_provider_prefix,
estimate_tokens_rough,
estimate_messages_tokens_rough,
estimate_request_tokens_rough,
get_model_context_length,
get_next_probe_tier,
get_cached_context_length,
@@ -70,6 +71,17 @@ class TestEstimateMessagesTokensRough:
estimate_messages_tokens_rough([msg])
)
def test_display_only_fields_do_not_change_estimate(self):
"""An edit row's inline_diff rides display_metadata, which the request builder strips;
pricing it would compact early and break the prompt cache."""
wire = {"role": "tool", "tool_call_id": "call-1", "name": "write_file",
"content": '{"bytes_written": 18000}'}
rich = {**wire, "_row_id": 42, "display_kind": "tool_result",
"display_metadata": {"tool_result_metadata": {"inline_diff": "\x1b[32m+ line\x1b[0m\n" * 600}}}
assert estimate_messages_tokens_rough([rich]) == estimate_messages_tokens_rough([wire])
assert estimate_request_tokens_rough([rich]) == estimate_request_tokens_rough([wire])
def test_message_with_list_content(self):
"""Vision messages with multimodal content arrays.