# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
277 lines
12 KiB
Python
277 lines
12 KiB
Python
"""Regression tests for #84371 — compaction dead-loop on reasoning-heavy
|
|
codex_responses sessions.
|
|
|
|
Root cause: the compaction TRIGGER (``estimate_messages_tokens_rough`` /
|
|
``estimate_request_tokens_rough``) charged stale ``reasoning`` /
|
|
``reasoning_content`` on EVERY assistant message, while the tail-protection
|
|
walk (``_find_tail_cut_by_tokens``) charged them on the newest turn only
|
|
(#73624). On a session where most tokens live in stale reasoning replay the
|
|
trigger fires above threshold while the walk protects everything —
|
|
``middle_window_tokens == 0`` / "insufficient progress" — and the same
|
|
compaction re-fires every turn, each attempt burning a full aux
|
|
summarization.
|
|
|
|
Wire truth (``_chat_messages_to_responses_input``): the codex_responses
|
|
input builder never reads the text thinking keys; reasoning continuity rides
|
|
the encrypted ``codex_reasoning_items`` sidecar, which both estimators charge
|
|
unconditionally. So the TRIGGER overcounted reality and the fix makes the
|
|
trigger route-aware (charge stale thinking only when the route echoes it),
|
|
while echo-back chat-completions routes (DeepSeek/Kimi/MiMo thinking mode)
|
|
now charge it in the walk too — one policy per session shape, chosen by
|
|
``message_sanitization.stale_thinking_reaches_wire``.
|
|
"""
|
|
|
|
from agent.context_compressor import (
|
|
ContextCompressor,
|
|
_estimate_msg_budget_tokens,
|
|
)
|
|
from agent.message_sanitization import stale_thinking_reaches_wire
|
|
from agent.model_metadata import (
|
|
estimate_messages_tokens_rough,
|
|
)
|
|
|
|
STALE_THINKING = "considering the next move carefully... " * 200 # ~2K tok
|
|
|
|
def _reasoning_heavy_session(n_turns: int = 40) -> list:
|
|
"""Transcript whose bulk is stale reasoning replay (the #84371 shape)."""
|
|
msgs = [{"role": "system", "content": "You are Hermes."}]
|
|
msgs.append({"role": "user", "content": "do the big task"})
|
|
for i in range(n_turns):
|
|
msgs.append(
|
|
{
|
|
"role": "assistant",
|
|
"content": f"step {i}",
|
|
"reasoning_content": STALE_THINKING,
|
|
"tool_calls": [
|
|
{
|
|
"id": f"c{i}",
|
|
"type": "function",
|
|
"function": {"name": "t", "arguments": "{}"},
|
|
}
|
|
],
|
|
}
|
|
)
|
|
msgs.append({"role": "tool", "tool_call_id": f"c{i}", "content": f"r{i}"})
|
|
return msgs
|
|
|
|
class TestWireTruthPredicate:
|
|
def test_codex_responses_never_ships_stale_thinking_text(self):
|
|
assert stale_thinking_reaches_wire(
|
|
"codex_responses", "deepseek", "deepseek-v4-flash", ""
|
|
) is False
|
|
|
|
def test_chat_completions_echo_family_ships_it(self):
|
|
# DeepSeek thinking mode over chat_completions echoes stored
|
|
# reasoning_content back on every assistant turn.
|
|
assert stale_thinking_reaches_wire(
|
|
"", "deepseek", "deepseek-reasoner", "https://api.deepseek.com"
|
|
) is True
|
|
|
|
def test_strict_chat_completions_strips_it(self):
|
|
assert stale_thinking_reaches_wire(
|
|
"", "mistral", "mistral-large", "https://api.mistral.ai"
|
|
) is False
|
|
|
|
class TestEstimatorParity:
|
|
"""Trigger-fires must imply the walk finds a compactable middle."""
|
|
|
|
def test_trigger_fires_implies_walk_finds_middle(self):
|
|
msgs = _reasoning_heavy_session()
|
|
wire = stale_thinking_reaches_wire(
|
|
"codex_responses", "deepseek", "deepseek-v4-flash", ""
|
|
)
|
|
trigger = estimate_messages_tokens_rough(
|
|
msgs, charge_stale_thinking=wire
|
|
)
|
|
|
|
cc = ContextCompressor(
|
|
model="deepseek-v4-flash",
|
|
provider="deepseek",
|
|
api_mode="codex_responses",
|
|
quiet_mode=True,
|
|
config_context_length=200_000,
|
|
)
|
|
# THE RELATION, not literals: whenever the route-aware trigger says
|
|
# the session is over threshold, the tail walk must leave a real
|
|
# middle region so the fired compaction can actually make progress.
|
|
if trigger >= cc.threshold_tokens:
|
|
start = cc._protect_head_size(msgs)
|
|
end = cc._find_tail_cut_by_tokens(msgs, start)
|
|
assert end > start, (
|
|
"trigger fired but the tail walk protected everything — "
|
|
"the #84371 dead-loop shape"
|
|
)
|
|
middle_tokens = estimate_messages_tokens_rough(msgs[start:end])
|
|
assert middle_tokens > 0
|
|
|
|
def test_route_aware_trigger_matches_walk_size_class(self):
|
|
"""On codex_responses the trigger no longer counts stale thinking
|
|
the walk excludes: both figures land in the same size class."""
|
|
msgs = _reasoning_heavy_session()
|
|
legacy = estimate_messages_tokens_rough(msgs)
|
|
route_aware = estimate_messages_tokens_rough(
|
|
msgs, charge_stale_thinking=False
|
|
)
|
|
from agent.context_compressor import _last_assistant_index
|
|
|
|
newest = _last_assistant_index(msgs)
|
|
walk = sum(
|
|
_estimate_msg_budget_tokens(m, charge_stale_thinking=(i == newest))
|
|
for i, m in enumerate(msgs)
|
|
)
|
|
# Stale thinking dominates this transcript, so the legacy figure is
|
|
# several times the walk's; the route-aware figure must not be.
|
|
assert legacy > 3 * walk
|
|
assert route_aware < 2 * walk
|
|
|
|
def test_newest_turn_thinking_still_charged(self):
|
|
msgs = _reasoning_heavy_session(n_turns=2)
|
|
stripped = estimate_messages_tokens_rough(
|
|
msgs, charge_stale_thinking=False
|
|
)
|
|
no_thinking = estimate_messages_tokens_rough(
|
|
[
|
|
{k: v for k, v in m.items()
|
|
if k not in ("reasoning", "reasoning_content")}
|
|
for m in msgs
|
|
]
|
|
)
|
|
# The newest assistant turn's thinking survives the stale strip.
|
|
assert stripped > no_thinking
|
|
|
|
def test_walk_charges_stale_thinking_on_echo_route(self):
|
|
"""Echo-back chat_completions route: the walk now charges stale
|
|
thinking on every turn, matching the trigger's full charge; the
|
|
codex_responses route keeps newest-turn-only. Assert the per-route
|
|
charge policy directly — for each route, the walk's per-message sum
|
|
must land in the same size class as that route's trigger estimate."""
|
|
msgs = _reasoning_heavy_session()
|
|
from agent.context_compressor import _last_assistant_index
|
|
|
|
cc_echo = ContextCompressor(
|
|
model="deepseek-reasoner",
|
|
provider="deepseek",
|
|
api_mode="",
|
|
base_url="https://api.deepseek.com",
|
|
quiet_mode=True,
|
|
config_context_length=200_000,
|
|
)
|
|
cc_codex = ContextCompressor(
|
|
model="deepseek-v4-flash",
|
|
provider="deepseek",
|
|
api_mode="codex_responses",
|
|
quiet_mode=True,
|
|
config_context_length=200_000,
|
|
)
|
|
assert cc_echo._stale_thinking_on_wire() is True
|
|
assert cc_codex._stale_thinking_on_wire() is False
|
|
|
|
newest = _last_assistant_index(msgs)
|
|
|
|
def walk_sum(cc):
|
|
charge_all = cc._stale_thinking_on_wire()
|
|
return sum(
|
|
_estimate_msg_budget_tokens(
|
|
m, charge_stale_thinking=(charge_all or i == newest)
|
|
)
|
|
for i, m in enumerate(msgs)
|
|
)
|
|
|
|
trigger_echo = estimate_messages_tokens_rough(
|
|
msgs, charge_stale_thinking=True
|
|
)
|
|
trigger_codex = estimate_messages_tokens_rough(
|
|
msgs, charge_stale_thinking=False
|
|
)
|
|
walk_echo = walk_sum(cc_echo)
|
|
walk_codex = walk_sum(cc_codex)
|
|
|
|
# Per route, trigger and walk agree within a small rough-estimator
|
|
# factor — the pre-fix codex disagreement was >3x.
|
|
assert walk_echo <= trigger_echo * 2 and trigger_echo <= walk_echo * 2
|
|
assert walk_codex <= trigger_codex * 2 and trigger_codex <= walk_codex * 2
|
|
# And the echo route genuinely charges the stale thinking bulk.
|
|
assert walk_echo > 3 * walk_codex
|
|
|
|
class TestReasoningDoubleCount:
|
|
"""``reasoning`` and ``reasoning_content`` carrying the same text must be
|
|
charged once — the wire ships at most one of them."""
|
|
|
|
def test_trigger_counts_identical_pair_once(self):
|
|
base = [{"role": "assistant", "content": "x"}]
|
|
rc_only = [{"role": "assistant", "content": "x",
|
|
"reasoning_content": "y" * 4000}]
|
|
both = [{"role": "assistant", "content": "x",
|
|
"reasoning": "y" * 4000, "reasoning_content": "y" * 4000}]
|
|
e0 = estimate_messages_tokens_rough(base)
|
|
e1 = estimate_messages_tokens_rough(rc_only)
|
|
e2 = estimate_messages_tokens_rough(both)
|
|
# Adding a duplicate `reasoning` key must not (materially) grow the
|
|
# estimate: the increment over rc_only stays far below a second copy.
|
|
assert (e2 - e0) < 1.5 * (e1 - e0)
|
|
|
|
def test_walk_counts_identical_pair_once(self):
|
|
base = {"role": "assistant", "content": "x"}
|
|
rc_only = {"role": "assistant", "content": "x",
|
|
"reasoning_content": "y" * 4000}
|
|
both = {"role": "assistant", "content": "x",
|
|
"reasoning": "y" * 4000, "reasoning_content": "y" * 4000}
|
|
w0 = _estimate_msg_budget_tokens(base, charge_stale_thinking=True)
|
|
w1 = _estimate_msg_budget_tokens(rc_only, charge_stale_thinking=True)
|
|
w2 = _estimate_msg_budget_tokens(both, charge_stale_thinking=True)
|
|
assert (w2 - w0) < 1.5 * (w1 - w0)
|
|
|
|
def test_reasoning_alone_still_charged(self):
|
|
"""No reasoning_content to displace it → `reasoning` is the
|
|
promotion proxy and must stay counted."""
|
|
base = [{"role": "assistant", "content": "x"}]
|
|
r_only = [{"role": "assistant", "content": "x",
|
|
"reasoning": "y" * 4000}]
|
|
assert (
|
|
estimate_messages_tokens_rough(r_only)
|
|
> estimate_messages_tokens_rough(base) + 500
|
|
)
|
|
w_base = _estimate_msg_budget_tokens(
|
|
{"role": "assistant", "content": "x"}, charge_stale_thinking=True
|
|
)
|
|
w_r = _estimate_msg_budget_tokens(
|
|
{"role": "assistant", "content": "x", "reasoning": "y" * 4000},
|
|
charge_stale_thinking=True,
|
|
)
|
|
assert w_r > w_base + 500
|
|
|
|
def test_pad_reasoning_content_does_not_displace(self):
|
|
"""A one-space echo pad is not real content; `reasoning` still
|
|
carries the chargeable text."""
|
|
msg = {"role": "assistant", "content": "x",
|
|
"reasoning": "y" * 4000, "reasoning_content": " "}
|
|
w = _estimate_msg_budget_tokens(msg, charge_stale_thinking=True)
|
|
w_base = _estimate_msg_budget_tokens(
|
|
{"role": "assistant", "content": "x", "reasoning_content": " "},
|
|
charge_stale_thinking=True,
|
|
)
|
|
assert w > w_base + 500
|
|
|
|
class TestNoProgressDeadLoopBreaker:
|
|
"""A fired compaction that returns the transcript unchanged must arm the
|
|
structural backoff so it cannot re-fire (and re-summarize) every turn."""
|
|
|
|
def test_no_progress_arms_structural_backoff(self):
|
|
import time
|
|
|
|
cc = ContextCompressor(
|
|
model="deepseek-v4-flash",
|
|
provider="deepseek",
|
|
api_mode="codex_responses",
|
|
quiet_mode=True,
|
|
config_context_length=200_000,
|
|
)
|
|
assert cc._structural_no_op_backoff_until <= time.monotonic()
|
|
cc._record_structural_no_op("compaction returned unchanged")
|
|
assert cc._structural_no_op_backoff_until > time.monotonic()
|
|
# Over-threshold but blocked: no second aux summarization this turn.
|
|
over = cc.threshold_tokens + 1
|
|
assert cc.should_compress(over) is False
|
|
reason = cc._compression_block_reason() or ""
|
|
assert reason.startswith("structural_backoff")
|