# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
192 lines
8.4 KiB
Python
192 lines
8.4 KiB
Python
"""Tests for the context-halving bugfix.
|
|
|
|
Background
|
|
----------
|
|
When the API returns "max_tokens too large given prompt" (input is fine,
|
|
but input_tokens + requested max_tokens > context_window), the old code
|
|
incorrectly halved context_length via get_next_probe_tier().
|
|
|
|
The fix introduces:
|
|
* parse_available_output_tokens_from_error() — detects this specific
|
|
error class and returns the available output token budget.
|
|
* _ephemeral_max_output_tokens on AIAgent — a one-shot override that
|
|
caps the output for one retry without touching context_length.
|
|
* get_context_length_from_provider_error() — accepts only concrete
|
|
provider-reported lower context limits and refuses guessed probe-tier
|
|
step-downs when the provider gives no maximum.
|
|
|
|
Naming note
|
|
-----------
|
|
max_tokens = OUTPUT token cap (a single response).
|
|
context_length = TOTAL context window (input + output combined).
|
|
These are different and the old code conflated them; the fix keeps them
|
|
separate.
|
|
"""
|
|
|
|
from unittest.mock import MagicMock
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# parse_available_output_tokens_from_error — unit tests
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestParseAvailableOutputTokens:
|
|
"""Pure-function tests; no I/O required."""
|
|
|
|
def _parse(self, msg):
|
|
from agent.model_metadata import parse_available_output_tokens_from_error
|
|
return parse_available_output_tokens_from_error(msg)
|
|
|
|
# ── Should detect and extract ────────────────────────────────────────
|
|
|
|
def test_anthropic_canonical_format(self):
|
|
"""Canonical Anthropic error: max_tokens: X > context_window: Y - input_tokens: Z = available_tokens: W"""
|
|
msg = (
|
|
"max_tokens: 32768 > context_window: 200000 "
|
|
"- input_tokens: 190000 = available_tokens: 10000"
|
|
)
|
|
assert self._parse(msg) == 10000
|
|
|
|
def test_available_tokens_natural_language(self):
|
|
"""'available tokens: N' wording (no underscore)."""
|
|
msg = "max_tokens must be at most 10000 given your prompt (available tokens: 10000)"
|
|
assert self._parse(msg) == 10000
|
|
|
|
# ── Should NOT detect (returns None) ─────────────────────────────────
|
|
|
|
def test_prompt_too_long_is_not_output_cap_error(self):
|
|
"""'prompt is too long' errors must NOT be caught — they need context-overflow recovery."""
|
|
msg = "prompt is too long: 205000 tokens > 200000 maximum"
|
|
assert self._parse(msg) is None
|
|
|
|
def test_no_max_tokens_keyword(self):
|
|
"""Error not related to max_tokens at all."""
|
|
msg = "invalid_api_key: the API key is invalid"
|
|
assert self._parse(msg) is None
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Context-overflow recovery — only trust provider-reported limits
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestContextOverflowLimitSelection:
|
|
"""Context-overflow recovery must not invent a lower window size.
|
|
|
|
Some providers only say "input exceeds the context window" without telling
|
|
Hermes what the actual maximum is. In that case we may compress the
|
|
conversation, but must not silently probe-step from a user-configured 1M
|
|
window down to 256K/128K/64K/etc.
|
|
"""
|
|
|
|
def test_generic_overflow_without_provider_limit_keeps_context_length(self):
|
|
from agent.model_metadata import get_context_length_from_provider_error
|
|
from agent.model_metadata import parse_context_limit_from_error
|
|
|
|
old_ctx = 1_000_000
|
|
error_msg = (
|
|
"Your input exceeds the context window of this model. "
|
|
"Please adjust your input and try again."
|
|
)
|
|
|
|
assert parse_context_limit_from_error(error_msg) is None
|
|
assert get_context_length_from_provider_error(error_msg, old_ctx) is None
|
|
|
|
def test_explicit_provider_limit_still_selects_that_limit(self):
|
|
from agent.model_metadata import get_context_length_from_provider_error
|
|
|
|
error_msg = "prompt is too long: 300000 tokens > 272000 maximum"
|
|
|
|
assert get_context_length_from_provider_error(error_msg, 1_000_000) == 272_000
|
|
|
|
def test_reported_limit_not_lower_than_current_is_ignored(self):
|
|
from agent.model_metadata import get_context_length_from_provider_error
|
|
|
|
error_msg = "maximum context length is 1000000 tokens"
|
|
|
|
assert get_context_length_from_provider_error(error_msg, 272_000) is None
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# build_anthropic_kwargs — output cap clamping
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestBuildAnthropicKwargsClamping:
|
|
"""The context_length clamp only fires when output ceiling > window.
|
|
For standard Anthropic models (output ceiling < window) it must not fire.
|
|
"""
|
|
|
|
def _build(self, model, max_tokens=None, context_length=None):
|
|
from agent.anthropic_adapter import build_anthropic_kwargs
|
|
return build_anthropic_kwargs(
|
|
model=model,
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
tools=None,
|
|
max_tokens=max_tokens,
|
|
reasoning_config=None,
|
|
context_length=context_length,
|
|
)
|
|
|
|
def test_no_clamping_when_output_ceiling_fits_in_window(self):
|
|
"""Opus 4.6 native output (128K) < context window (200K) — no clamping."""
|
|
from agent.anthropic_adapter import _get_anthropic_max_output
|
|
kwargs = self._build("claude-opus-4-6", context_length=200_000)
|
|
assert kwargs["max_tokens"] == _get_anthropic_max_output("claude-opus-4-6") < 200_000
|
|
|
|
def test_explicit_max_tokens_clamped_when_exceeds_window(self):
|
|
"""Explicit max_tokens larger than a small window is clamped."""
|
|
kwargs = self._build("claude-opus-4-6", max_tokens=32_768, context_length=16_000)
|
|
assert kwargs["max_tokens"] == 15_999
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Ephemeral max_tokens mechanism — _build_api_kwargs
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestEphemeralMaxOutputTokens:
|
|
"""_build_api_kwargs consumes _ephemeral_max_output_tokens exactly once
|
|
and falls back to self.max_tokens on subsequent calls.
|
|
"""
|
|
|
|
def _make_agent(self):
|
|
"""Return a minimal AIAgent with api_mode='anthropic_messages' and
|
|
a stubbed context_compressor, bypassing full __init__ cost."""
|
|
from run_agent import AIAgent
|
|
agent = object.__new__(AIAgent)
|
|
# Minimal attributes used by _build_api_kwargs
|
|
agent.api_mode = "anthropic_messages"
|
|
agent.model = "claude-opus-4-6"
|
|
agent.tools = []
|
|
agent.max_tokens = None
|
|
agent.reasoning_config = None
|
|
agent._is_anthropic_oauth = False
|
|
agent._ephemeral_max_output_tokens = None
|
|
|
|
compressor = MagicMock()
|
|
compressor.context_length = 200_000
|
|
agent.context_compressor = compressor
|
|
|
|
# Stub out the internal message-preparation helper
|
|
agent._prepare_anthropic_messages_for_api = MagicMock(
|
|
return_value=[{"role": "user", "content": "hi"}]
|
|
)
|
|
agent._anthropic_preserve_dots = MagicMock(return_value=False)
|
|
agent.request_overrides = {}
|
|
return agent
|
|
|
|
def test_ephemeral_override_is_used_on_first_call(self):
|
|
"""When _ephemeral_max_output_tokens is set, it overrides self.max_tokens."""
|
|
agent = self._make_agent()
|
|
agent._ephemeral_max_output_tokens = 5_000
|
|
|
|
kwargs = agent._build_api_kwargs([{"role": "user", "content": "hi"}])
|
|
assert kwargs["max_tokens"] == 5_000
|
|
|
|
def test_ephemeral_override_is_consumed_after_one_call(self):
|
|
"""After one call the ephemeral override is cleared to None."""
|
|
agent = self._make_agent()
|
|
agent._ephemeral_max_output_tokens = 5_000
|
|
|
|
agent._build_api_kwargs([{"role": "user", "content": "hi"}])
|
|
kwargs = agent._build_api_kwargs([{"role": "user", "content": "hi"}])
|
|
assert kwargs["max_tokens"] != 5_000
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Integration: error handler does NOT halve context_length for output-cap errors
|
|
# ---------------------------------------------------------------------------
|