# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
454 lines
20 KiB
Python
454 lines
20 KiB
Python
"""Regression coverage for #29824 — the WebUI session viewer (and TUI
|
|
chat panel) was showing the ``[CONTEXT COMPACTION — REFERENCE ONLY]``
|
|
handoff block in the slot where the user had just been reading the
|
|
assistant's actual reply, because the previously-visible reply got
|
|
rolled into the compaction summary by the token-budget tail walk.
|
|
|
|
The fix adds ``_ensure_last_assistant_message_in_tail`` — a mirror of
|
|
the existing ``_ensure_last_user_message_in_tail`` (#10896 anchor) —
|
|
that pulls ``cut_idx`` back to include the most recent assistant
|
|
message with non-empty text content, with the standard tool-group
|
|
realignment so we don't orphan a ``tool_call`` / ``tool_result`` pair.
|
|
|
|
Pinned here:
|
|
|
|
* ``TestFindLastAssistantMessageIdx`` — pure helper contract:
|
|
finds the most recent **content-bearing** assistant message,
|
|
skips tool-call-only stubs, falls back to "any assistant" only
|
|
when no content-bearing reply exists in the compressible region,
|
|
honours ``head_end``, returns -1 when there's no assistant at all.
|
|
|
|
* ``TestEnsureLastAssistantMessageInTail`` — direct: walks
|
|
``cut_idx`` back when the last reply is in the compressed middle,
|
|
is a no-op when it's already in the tail, never crosses
|
|
``head_end``, re-aligns through tool groups.
|
|
|
|
* ``TestFindTailCutByTokensAnchorsAssistant`` — integration with
|
|
the existing tail-cut path: the exact reporter scenario (long
|
|
tool-output run after the previously-visible reply) preserves
|
|
the reply; combines with the user anchor for the same-turn
|
|
preservation; soft-ceiling overrun no longer hides the reply.
|
|
|
|
* ``TestCompactionRollupReproduction`` — end-to-end through
|
|
``compress()`` with a stubbed summariser: pre-fix the reply
|
|
text is absorbed into the summary (regression demonstrated by
|
|
asserting on the OLD behaviour fails); post-fix the reply text
|
|
is still present in the compressed transcript as a regular
|
|
assistant message.
|
|
|
|
* ``TestSourceGuardrail`` — static asserts on
|
|
``agent/context_compressor.py`` so a future refactor can't
|
|
silently drop the anchor.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
@pytest.fixture()
|
|
def compressor():
|
|
"""ContextCompressor with mocked deps and a tight tail budget so
|
|
the helpers' anchor behaviour is observable."""
|
|
from agent.context_compressor import ContextCompressor
|
|
with patch(
|
|
"agent.context_compressor.get_model_context_length",
|
|
return_value=100_000,
|
|
):
|
|
c = ContextCompressor(
|
|
model="test/model",
|
|
threshold_percent=0.85,
|
|
protect_first_n=2,
|
|
protect_last_n=2,
|
|
quiet_mode=True,
|
|
)
|
|
c.tail_token_budget = 50
|
|
return c
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helper: _find_last_assistant_message_idx
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestFindLastAssistantMessageIdx:
|
|
def test_skips_assistant_role_context_summary_marker(self, compressor):
|
|
"""A persisted assistant-role handoff is internal continuity state,
|
|
not the last reply the user saw. Tool-call-only assistant messages
|
|
after it must remain eligible for the fallback anchor."""
|
|
from agent.context_compressor import SUMMARY_PREFIX
|
|
|
|
messages = [
|
|
{"role": "assistant", "content": f"{SUMMARY_PREFIX}\nold handoff"},
|
|
{"role": "user", "content": "continue the task"},
|
|
{"role": "assistant", "content": None,
|
|
"tool_calls": [{"function": {"name": "t",
|
|
"arguments": "{}"}}]},
|
|
{"role": "tool", "content": "result", "tool_call_id": "c1"},
|
|
]
|
|
assert compressor._find_last_assistant_message_idx(
|
|
messages, head_end=0
|
|
) == 2
|
|
|
|
def test_multimodal_text_block_counts(self, compressor):
|
|
"""An assistant with multimodal list-content carrying a text
|
|
block (Anthropic / GPT-style ``[{type:text,text:...}]``)
|
|
counts as content-bearing."""
|
|
messages = [
|
|
{"role": "user", "content": "q"},
|
|
{"role": "assistant",
|
|
"content": [{"type": "text", "text": "hello"}]},
|
|
]
|
|
idx = compressor._find_last_assistant_message_idx(messages, head_end=0)
|
|
assert idx == 1
|
|
|
|
def test_respects_head_end_lower_bound(self, compressor):
|
|
"""An assistant message at or before ``head_end`` must be
|
|
ignored — it's already in the protected head region."""
|
|
messages = [
|
|
{"role": "system", "content": "sys"},
|
|
{"role": "assistant", "content": "in-head reply"}, # idx 1
|
|
{"role": "user", "content": "q"},
|
|
]
|
|
# head_end=2 means the compressible region starts at index 2;
|
|
# the assistant at index 1 is in the head and must be skipped.
|
|
idx = compressor._find_last_assistant_message_idx(messages, head_end=2)
|
|
assert idx == -1
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helper: _ensure_last_assistant_message_in_tail
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestEnsureLastAssistantMessageInTail:
|
|
def test_no_op_when_already_in_tail(self, compressor):
|
|
messages = [
|
|
{"role": "user", "content": "q"},
|
|
{"role": "assistant", "content": "reply"},
|
|
{"role": "user", "content": "q2"},
|
|
]
|
|
# cut_idx=1 means tail starts at index 1 — the reply is already in tail.
|
|
new_cut = compressor._ensure_last_assistant_message_in_tail(
|
|
messages, cut_idx=1, head_end=0
|
|
)
|
|
assert new_cut == 1
|
|
|
|
def test_walks_cut_idx_back_to_include_reply(self, compressor):
|
|
messages = [
|
|
{"role": "user", "content": "q1"},
|
|
{"role": "assistant", "content": "REPLY"}, # idx 1
|
|
{"role": "user", "content": "q2"},
|
|
{"role": "user", "content": "q3"},
|
|
]
|
|
# cut_idx=2 leaves the reply outside the tail; anchor must pull
|
|
# cut_idx back to 1 so messages[1:] contains the reply.
|
|
new_cut = compressor._ensure_last_assistant_message_in_tail(
|
|
messages, cut_idx=2, head_end=0
|
|
)
|
|
assert new_cut == 1
|
|
assert any(
|
|
isinstance(m.get("content"), str) and "REPLY" in m["content"]
|
|
for m in messages[new_cut:]
|
|
)
|
|
|
|
def test_re_aligns_through_preceding_tool_group(self, compressor):
|
|
"""When the anchored assistant is preceded by a
|
|
tool_call/result group, ``_align_boundary_backward`` must pull
|
|
``cut_idx`` even further back so the group isn't split — same
|
|
guarantee as ``_ensure_last_user_message_in_tail``."""
|
|
messages = [
|
|
{"role": "user", "content": "q1"},
|
|
{"role": "assistant", "content": None,
|
|
"tool_calls": [{"id": "c1",
|
|
"function": {"name": "t",
|
|
"arguments": "{}"}}]},
|
|
{"role": "tool", "content": "result", "tool_call_id": "c1"},
|
|
{"role": "assistant", "content": "REPLY"}, # idx 3
|
|
{"role": "user", "content": "q2"},
|
|
]
|
|
# cut_idx=4 leaves the reply outside the tail. Anchor pulls
|
|
# back to 3, then _align_boundary_backward sees the preceding
|
|
# tool group and pulls further back to 1 (before the assistant
|
|
# with tool_calls).
|
|
new_cut = compressor._ensure_last_assistant_message_in_tail(
|
|
messages, cut_idx=4, head_end=0
|
|
)
|
|
assert new_cut <= 3
|
|
# The tool_call assistant (1) and its tool_result (2) must NOT
|
|
# be split: either both in compressed region or both in tail.
|
|
if new_cut <= 1:
|
|
# Both in tail — tool group intact.
|
|
assert messages[new_cut].get("role") == "assistant"
|
|
else:
|
|
# Otherwise the anchor must land at the reply itself (3).
|
|
assert new_cut == 3
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Integration with _find_tail_cut_by_tokens
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestFindTailCutByTokensAnchorsAssistant:
|
|
def test_reporter_repro_long_tool_run_after_visible_reply(
|
|
self, compressor
|
|
):
|
|
"""The exact #29824 scenario: a tight token budget combined
|
|
with a long tail of tool-call/result messages after the
|
|
visible reply. Pre-fix, the token-budget walk hit its ceiling
|
|
on the tool output and parked ``cut_idx`` past the reply.
|
|
Post-fix, the assistant anchor pulls it back."""
|
|
c = compressor
|
|
c.tail_token_budget = 10 # force min-tail behaviour
|
|
messages = [
|
|
{"role": "system", "content": "sys"},
|
|
{"role": "user", "content": "msg1"}, # head_end=2
|
|
{"role": "user", "content": "q1"},
|
|
{"role": "assistant",
|
|
"content": "PREVIOUSLY VISIBLE REPLY"}, # idx 3
|
|
{"role": "user", "content": "q2"},
|
|
{"role": "assistant", "content": None,
|
|
"tool_calls": [{"id": "c1",
|
|
"function": {"name": "t",
|
|
"arguments": "{}"}}]},
|
|
{"role": "tool", "content": "x" * 200,
|
|
"tool_call_id": "c1"},
|
|
]
|
|
cut = c._find_tail_cut_by_tokens(messages, head_end=2)
|
|
tail_contents = [
|
|
m.get("content") for m in messages[cut:]
|
|
if isinstance(m.get("content"), str)
|
|
]
|
|
assert any(
|
|
"PREVIOUSLY VISIBLE REPLY" in (t or "") for t in tail_contents
|
|
), (
|
|
"REGRESSION (#29824): the visible reply was rolled into "
|
|
f"the compaction summary. Tail contents: {tail_contents!r}"
|
|
)
|
|
|
|
def test_user_and_assistant_anchors_compose(self, compressor):
|
|
"""Both anchors run in sequence; the tail must contain both
|
|
the latest user message AND the latest visible assistant
|
|
reply."""
|
|
c = compressor
|
|
c.tail_token_budget = 10
|
|
messages = [
|
|
{"role": "user", "content": "q1"},
|
|
{"role": "assistant", "content": "VISIBLE REPLY"},
|
|
{"role": "user", "content": "follow-up question"},
|
|
{"role": "user", "content": "and another"},
|
|
]
|
|
cut = c._find_tail_cut_by_tokens(messages, head_end=0)
|
|
tail_contents = [
|
|
m.get("content") for m in messages[cut:]
|
|
if isinstance(m.get("content"), str)
|
|
]
|
|
assert any("VISIBLE REPLY" in (t or "") for t in tail_contents)
|
|
assert any("and another" in (t or "") for t in tail_contents)
|
|
|
|
def test_oversized_tool_output_does_not_strand_reply(self, compressor):
|
|
"""The soft-ceiling logic in ``_find_tail_cut_by_tokens``
|
|
permits a single oversized tail message; the assistant anchor
|
|
must still recover the reply on the other side of it."""
|
|
c = compressor
|
|
c.tail_token_budget = 100 # soft ceiling 150
|
|
messages = [
|
|
{"role": "user", "content": "earlier"},
|
|
{"role": "assistant", "content": "VISIBLE REPLY"},
|
|
{"role": "user", "content": "read big file"},
|
|
{"role": "assistant", "content": None,
|
|
"tool_calls": [{"id": "c1",
|
|
"function": {"name": "read",
|
|
"arguments": "{}"}}]},
|
|
# ~500 chars ⇒ ~135 tokens, blows past soft ceiling of 150
|
|
{"role": "tool", "content": "y" * 500,
|
|
"tool_call_id": "c1"},
|
|
{"role": "user", "content": "ok"},
|
|
]
|
|
cut = c._find_tail_cut_by_tokens(messages, head_end=0)
|
|
tail_contents = [
|
|
m.get("content") for m in messages[cut:]
|
|
if isinstance(m.get("content"), str)
|
|
]
|
|
assert any("VISIBLE REPLY" in (t or "") for t in tail_contents)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# End-to-end: compress() preserves the reply
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestCompactionRollupReproduction:
|
|
"""End-to-end through ``compress()``: the visible reply text must
|
|
survive in the compressed transcript — either as its own
|
|
standalone assistant message OR concatenated onto the merged
|
|
summary-handoff tail message (the compressor's double-collision
|
|
fallback path; the WebUI re-splits these on the END marker so the
|
|
reply renders as a separate bubble — see ``splitCompactionContent``
|
|
in ``web/src/pages/SessionsPage.tsx``)."""
|
|
|
|
def test_compress_keeps_visible_reply_text(self, compressor):
|
|
from agent.context_compressor import (
|
|
SUMMARY_PREFIX,
|
|
COMPRESSED_SUMMARY_METADATA_KEY,
|
|
)
|
|
c = compressor
|
|
c.tail_token_budget = 10
|
|
# ``_generate_summary`` normally wraps the LLM body in
|
|
# ``SUMMARY_PREFIX`` via ``_with_summary_prefix``; mimic that so
|
|
# the merge-into-tail branch can identify the boundary.
|
|
_mocked = f"{SUMMARY_PREFIX}\nrolled-up middle summary"
|
|
messages = (
|
|
[{"role": "system", "content": "sys"},
|
|
{"role": "user", "content": "initial"}] # head (protect_first_n=2)
|
|
# Middle: long enough to be compressible.
|
|
+ [
|
|
{"role": "user", "content": f"middle q{i}"}
|
|
if i % 2 == 0
|
|
else {"role": "assistant", "content": f"middle reply {i}"}
|
|
for i in range(12)
|
|
]
|
|
+ [
|
|
{"role": "user", "content": "the visible question"},
|
|
{"role": "assistant",
|
|
"content": "THE VISIBLE REPLY THE USER JUST READ"},
|
|
{"role": "user", "content": "follow up"},
|
|
{"role": "assistant", "content": None,
|
|
"tool_calls": [{"id": "c1",
|
|
"function": {"name": "t",
|
|
"arguments": "{}"}}]},
|
|
{"role": "tool", "content": "z" * 500,
|
|
"tool_call_id": "c1"},
|
|
]
|
|
)
|
|
with patch.object(
|
|
c, "_generate_summary",
|
|
return_value=_mocked,
|
|
):
|
|
result = c.compress(messages, current_tokens=90_000)
|
|
# 1. A summary message exists (compression actually ran). Detect via
|
|
# the canonical metadata key rather than a content prefix: merge-into-
|
|
# tail summaries wrap prior content before the summary, so the prefix
|
|
# is no longer at the start of the message content (#56372).
|
|
assert any(
|
|
m.get(COMPRESSED_SUMMARY_METADATA_KEY)
|
|
for m in result
|
|
), "compress() did not insert a summary message"
|
|
# 2. The visible reply text must survive somewhere — either
|
|
# as its own message OR concatenated into the merged tail.
|
|
joined = "\n".join(
|
|
m.get("content") for m in result
|
|
if isinstance(m.get("content"), str)
|
|
)
|
|
assert "THE VISIBLE REPLY THE USER JUST READ" in joined, (
|
|
"REGRESSION (#29824): the visible reply was absorbed into "
|
|
"the compaction summary AND erased. Compressed transcript "
|
|
f"({len(result)} msgs): "
|
|
f"{[(m.get('role'), str(m.get('content'))[:50]) for m in result]}"
|
|
)
|
|
|
|
def test_standalone_summary_case_keeps_reply_as_own_message(
|
|
self, compressor
|
|
):
|
|
"""When the head and tail roles allow a standalone summary
|
|
message (no double-collision), the visible reply must remain
|
|
as its OWN assistant message — not merged with anything.
|
|
This is the common case; the merge-into-tail path is the
|
|
edge case for double-collision."""
|
|
from agent.context_compressor import (
|
|
SUMMARY_PREFIX,
|
|
COMPRESSED_SUMMARY_METADATA_KEY,
|
|
)
|
|
c = compressor
|
|
c.tail_token_budget = 10
|
|
_mocked = f"{SUMMARY_PREFIX}\nrolled-up middle summary"
|
|
# Head ends with ``assistant`` ⇒ summary_role flips to
|
|
# ``user`` ⇒ no collision with the assistant tail ⇒ standalone
|
|
# summary insert (no merge).
|
|
messages = (
|
|
[
|
|
{"role": "user", "content": "initial"},
|
|
{"role": "assistant", "content": "head reply"},
|
|
]
|
|
+ [
|
|
{"role": "user", "content": f"middle q{i}"}
|
|
if i % 2 == 0
|
|
else {"role": "assistant", "content": f"middle reply {i}"}
|
|
for i in range(12)
|
|
]
|
|
+ [
|
|
{"role": "user", "content": "the visible question"},
|
|
{"role": "assistant",
|
|
"content": "THE VISIBLE REPLY THE USER JUST READ"},
|
|
{"role": "user", "content": "follow up"},
|
|
]
|
|
)
|
|
with patch.object(
|
|
c, "_generate_summary",
|
|
return_value=_mocked,
|
|
):
|
|
result = c.compress(messages, current_tokens=90_000)
|
|
# Summary present (detect via the canonical metadata key — merge-into-
|
|
# tail summaries no longer start with SUMMARY_PREFIX after #56372):
|
|
summary_rows = [
|
|
m for m in result
|
|
if m.get(COMPRESSED_SUMMARY_METADATA_KEY)
|
|
]
|
|
assert len(summary_rows) == 1
|
|
# Visible reply as its OWN distinct assistant message
|
|
# (NOT merged into the summary row):
|
|
reply_rows = [
|
|
m for m in result
|
|
if m.get("role") == "assistant"
|
|
and isinstance(m.get("content"), str)
|
|
and "THE VISIBLE REPLY THE USER JUST READ" in m["content"]
|
|
and not m.get(COMPRESSED_SUMMARY_METADATA_KEY)
|
|
]
|
|
assert len(reply_rows) == 1, (
|
|
"REGRESSION (#29824): expected exactly one standalone "
|
|
f"assistant message carrying the visible reply, got "
|
|
f"{len(reply_rows)}"
|
|
)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Source guardrail
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestFindLastUserMessageIdxSkipsSummaryMarker:
|
|
"""A context-compaction handoff banner is inserted with ``role="user"``
|
|
when the head ends in an assistant/tool message (see the summary-role
|
|
selection in ``compress``). ``_find_last_user_message_idx`` must NOT treat
|
|
that banner as the latest user turn — otherwise, on a resumed or
|
|
multi-compaction session, ``_ensure_last_user_message_in_tail`` anchors the
|
|
tail to the summary and rolls the genuine last user message into the next
|
|
compaction, re-triggering the active-task loss the anchor exists to prevent.
|
|
(Salvaged from #36626 / issue #36624.)
|
|
"""
|
|
|
|
def test_skips_user_role_context_summary_marker(self, compressor):
|
|
from agent.context_compressor import SUMMARY_PREFIX
|
|
|
|
messages = [
|
|
{"role": "system", "content": "sys"},
|
|
{"role": "user", "content": "REAL current task"},
|
|
{"role": "assistant", "content": "working on it"},
|
|
# A handoff summary re-inserted as a user-role message after resume.
|
|
{"role": "user", "content": f"{SUMMARY_PREFIX}\n## Active Task\nold"},
|
|
{"role": "assistant", "content": "continuing from the real task"},
|
|
]
|
|
# Latest *real* user message is index 1, not the summary at index 3.
|
|
assert compressor._find_last_user_message_idx(messages, head_end=1) == 1
|
|
|
|
def test_returns_real_user_when_no_summary_present(self, compressor):
|
|
messages = [
|
|
{"role": "system", "content": "sys"},
|
|
{"role": "user", "content": "first"},
|
|
{"role": "assistant", "content": "reply"},
|
|
{"role": "user", "content": "second"},
|
|
]
|
|
assert compressor._find_last_user_message_idx(messages, head_end=1) == 3
|
|
|
|
def test_all_user_messages_are_summaries_returns_minus_one(self, compressor):
|
|
from agent.context_compressor import SUMMARY_PREFIX
|
|
|
|
messages = [
|
|
{"role": "system", "content": "sys"},
|
|
{"role": "assistant", "content": "reply"},
|
|
{"role": "user", "content": f"{SUMMARY_PREFIX}\nhandoff"},
|
|
]
|
|
assert compressor._find_last_user_message_idx(messages, head_end=1) == -1
|