# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
177 lines
6.7 KiB
Python
177 lines
6.7 KiB
Python
"""Proactive-prune rearm must not lock out an over-threshold session (#101889).
|
|
|
|
``_proactive_prune_rearm_tokens`` is armed from a message-bodies-only estimate,
|
|
but the provider bills the system prompt and tool schemas too. On a schema-heavy
|
|
session the message-only estimate can sit permanently just below the rearm mark
|
|
while the real request rides *above* ``threshold_tokens`` — the prune declines
|
|
every iteration, full compression never gets there, and nothing is logged. The
|
|
session then grows until the provider rejects the request.
|
|
|
|
Pinned here as invariants (no frozen config literals): the gates are evaluated
|
|
against this compressor's own ``threshold_tokens`` / ``proactive_prune_tokens``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import Any, Dict, List
|
|
from unittest.mock import patch
|
|
|
|
from agent.context_compressor import ContextCompressor, _estimate_msg_budget_tokens
|
|
|
|
LARGE_WINDOW = 1_000_000
|
|
|
|
def _compressor(**kw: Any) -> ContextCompressor:
|
|
defaults = dict(
|
|
model="test",
|
|
quiet_mode=True,
|
|
threshold_percent=0.50,
|
|
protect_first_n=2,
|
|
protect_last_n=4,
|
|
proactive_prune_tokens=48_000,
|
|
proactive_prune_min_result_chars=8_000,
|
|
)
|
|
defaults.update(kw)
|
|
with patch(
|
|
"agent.context_compressor.get_model_context_length",
|
|
return_value=LARGE_WINDOW,
|
|
):
|
|
return ContextCompressor(**defaults)
|
|
|
|
def _history(n_pairs: int = 8, big: int = 9_000) -> List[Dict[str, Any]]:
|
|
msgs: List[Dict[str, Any]] = [{"role": "system", "content": "sys"}]
|
|
for i in range(n_pairs):
|
|
cid = f"call_{i}"
|
|
msgs.append({
|
|
"role": "assistant",
|
|
"content": "",
|
|
"tool_calls": [{
|
|
"id": cid,
|
|
"type": "function",
|
|
"function": {"name": "terminal", "arguments": '{"cmd":"ls"}'},
|
|
}],
|
|
})
|
|
msgs.append({
|
|
"role": "tool",
|
|
"tool_call_id": cid,
|
|
"content": chr(65 + i) * big if i < 3 else "ok",
|
|
})
|
|
return msgs
|
|
|
|
def _park_rearm_just_above_messages(
|
|
compressor: ContextCompressor, messages: List[Dict[str, Any]]
|
|
) -> int:
|
|
"""Reproduce the reporter's state: message-only estimate stuck 913 tokens
|
|
below the rearm mark (schema overhead makes up the rest of the request)."""
|
|
before = sum(_estimate_msg_budget_tokens(m) for m in messages)
|
|
compressor._proactive_prune_rearm_tokens = before + 913
|
|
assert before < compressor._proactive_prune_rearm_tokens
|
|
return before
|
|
|
|
def _over_threshold_warnings(caplog) -> list:
|
|
return [
|
|
r for r in caplog.records
|
|
if r.levelno >= logging.WARNING
|
|
and "over the compression threshold" in r.getMessage()
|
|
]
|
|
|
|
def test_billed_basis_over_threshold_defeats_message_only_rearm_lockout() -> None:
|
|
"""Over ``threshold_tokens`` on the provider-billed basis, the rearm gate
|
|
must not short-circuit the prune on the message-only estimate alone."""
|
|
c = _compressor()
|
|
msgs = _history()
|
|
_park_rearm_just_above_messages(c, msgs)
|
|
billed = c.threshold_tokens + 1 # provider says: over threshold, now
|
|
|
|
scans: List[int] = []
|
|
# Stand in for the real multi-pass scan: a NEW list whose old tool outputs
|
|
# are reclaimed, so the (untouched) reclaim gate can commit it.
|
|
reclaimed = [dict(m) for m in msgs]
|
|
for m in reclaimed[:-2]:
|
|
if m.get("role") == "tool":
|
|
m["content"] = "[pruned]"
|
|
|
|
def _scan(*args: Any, **kwargs: Any) -> tuple[List[Dict[str, Any]], int]:
|
|
scans.append(1)
|
|
return reclaimed, 3
|
|
|
|
with patch.object(c, "_prune_old_tool_results", _scan):
|
|
result, pruned = c.prune_tool_results_only(msgs, current_tokens=billed)
|
|
|
|
assert scans, "rearm gate short-circuited on the message-only estimate"
|
|
assert pruned == 3
|
|
assert result is not msgs
|
|
|
|
def test_message_only_rearm_still_holds_below_threshold() -> None:
|
|
"""Prompt-cache hysteresis is intact while the real request is under the
|
|
compression threshold — the rearm bypass is an overflow escape hatch only."""
|
|
c = _compressor()
|
|
msgs = _history()
|
|
_park_rearm_just_above_messages(c, msgs)
|
|
under = c.threshold_tokens - 1
|
|
assert under >= c.proactive_prune_tokens # above the prune trigger
|
|
|
|
with patch.object(
|
|
c,
|
|
"_prune_old_tool_results",
|
|
side_effect=AssertionError("scan must not run below threshold"),
|
|
):
|
|
result, pruned = c.prune_tool_results_only(msgs, current_tokens=under)
|
|
|
|
assert result is msgs
|
|
assert pruned == 0
|
|
|
|
def test_no_op_below_the_prune_trigger() -> None:
|
|
"""Under ``proactive_prune_tokens`` nothing is reclaimed, rearm or not —
|
|
the bypass must not turn into over-pruning of small sessions."""
|
|
c = _compressor()
|
|
msgs = _history()
|
|
c.on_session_reset() # fully rearmed; only the trigger gates
|
|
|
|
with patch.object(
|
|
c,
|
|
"_prune_old_tool_results",
|
|
side_effect=AssertionError("scan must not run below the trigger"),
|
|
):
|
|
result, pruned = c.prune_tool_results_only(
|
|
msgs, current_tokens=c.proactive_prune_tokens - 1
|
|
)
|
|
|
|
assert result is msgs
|
|
assert pruned == 0
|
|
|
|
def test_over_threshold_reclamation_no_op_warns_once(caplog) -> None:
|
|
"""A session riding above the threshold with every reclamation path
|
|
declining must be distinguishable in the log — and must not spam the same
|
|
reason on every tool iteration."""
|
|
# Reclaim floor above anything this transcript can free: the scan runs,
|
|
# finds candidates, and the commit gate rejects it — a silent no-op today.
|
|
c = _compressor(proactive_prune_min_reclaim_tokens=10_000_000)
|
|
msgs = _history()
|
|
billed = c.threshold_tokens + 5_000
|
|
|
|
with caplog.at_level(logging.WARNING, logger="agent.context_compressor"):
|
|
result, pruned = c.prune_tool_results_only(msgs, current_tokens=billed)
|
|
assert (result, pruned) == (msgs, 0)
|
|
|
|
warnings = _over_threshold_warnings(caplog)
|
|
assert warnings, "over-threshold reclamation no-op was silent"
|
|
|
|
# Same state on the next tool iteration: deduped, not re-logged.
|
|
with caplog.at_level(logging.WARNING, logger="agent.context_compressor"):
|
|
c.prune_tool_results_only(msgs, current_tokens=billed)
|
|
assert len(_over_threshold_warnings(caplog)) == len(warnings)
|
|
|
|
def test_under_threshold_no_op_is_not_warned(caplog) -> None:
|
|
"""Ordinary hysteresis below the threshold stays quiet."""
|
|
c = _compressor(proactive_prune_min_reclaim_tokens=10_000_000)
|
|
msgs = _history()
|
|
|
|
with caplog.at_level(logging.WARNING, logger="agent.context_compressor"):
|
|
result, pruned = c.prune_tool_results_only(
|
|
msgs, current_tokens=c.threshold_tokens - 1
|
|
)
|
|
|
|
assert (result, pruned) == (msgs, 0)
|
|
assert not _over_threshold_warnings(caplog)
|