From e83816a4d1998245968949e88fa15f26d89800c0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 3 Sep 2026 09:44:26 -0700 Subject: [PATCH] review-fix(comments): restore lost #NNNN rationale comments across non-test source (mechanical sweep, condensed, code unchanged) For each issue anchor present in BASE 63279301bcb non-test .py and absent on HEAD, the BASE comment/docstring block was re-attached at the HEAD location of the code it explained (matched by the distinctive code line / enclosing def). Sentences already covered by an existing HEAD comment were deduped; the issue number always survives. Insert-only: no code lines changed. --- acp_adapter/commands.py | 2 + acp_adapter/content.py | 3 + acp_adapter/entry.py | 3 + acp_adapter/permissions.py | 1 + acp_adapter/server.py | 8 + acp_adapter/session.py | 13 + acp_adapter/tools.py | 5 + agent/account_usage.py | 5 + agent/activity_tracking.py | 7 + agent/agent_init.py | 61 +++ agent/agent_runtime_helpers.py | 240 ++++++++- agent/anthropic_adapter.py | 14 + agent/anthropic_endpoints.py | 7 +- agent/anthropic_message_convert.py | 19 +- agent/api_error_summary.py | 12 + agent/auxiliary_client.py | 217 +++++++- agent/backend_identity.py | 5 + agent/background_review.py | 58 ++- agent/bedrock_adapter.py | 15 +- agent/browser_provider.py | 7 + agent/chat_completion_helpers.py | 47 ++ agent/client_lifecycle.py | 29 +- agent/codex_responses_adapter.py | 55 +- agent/codex_runtime.py | 8 + agent/compaction_display.py | 12 + agent/compression_facade.py | 22 +- agent/context_compressor.py | 294 ++++++++++- agent/context_engine.py | 12 + agent/context_references.py | 2 + agent/conversation_compression.py | 259 +++++++++- agent/conversation_loop.py | 43 +- agent/copilot_acp_client.py | 3 + agent/credential_pool.py | 11 + agent/curator_backup.py | 1 + agent/deadline.py | 13 +- agent/error_classifier.py | 15 + agent/gemini_native_adapter.py | 27 +- agent/image_routing.py | 11 +- agent/insights.py | 8 +- agent/learning_mutations.py | 5 + agent/memory_manager.py | 14 + agent/message_sanitization.py | 53 ++ agent/micro_compaction.py | 9 + agent/moa_loop.py | 57 +++ agent/model_metadata.py | 40 +- agent/models_dev.py | 23 +- agent/native_compaction.py | 22 + agent/opencode_affinity.py | 12 + agent/outbound_webhooks.py | 8 +- agent/plan_prompt.py | 5 +- agent/plugin_llm.py | 6 +- agent/process_bootstrap.py | 10 + agent/prompt_builder.py | 87 +++- agent/prompt_cache_scope.py | 16 +- agent/prompt_caching.py | 6 + agent/proxy_sources/iron_proxy.py | 7 + agent/reasoning_effort.py | 12 +- agent/reasoning_params.py | 13 +- agent/redact.py | 98 +++- agent/relay_llm.py | 13 +- agent/relay_runtime.py | 7 + agent/relay_tools.py | 1 + agent/repetition_guard.py | 6 +- agent/replay_cleanup.py | 17 +- agent/secret_scope.py | 1 + agent/secret_sources/bitwarden.py | 2 + agent/secret_sources/registry.py | 10 +- agent/session_activity.py | 1 + agent/session_persistence.py | 17 +- agent/shell_hooks.py | 13 +- agent/skill_bundles.py | 10 +- agent/skill_commands.py | 53 +- agent/skill_utils.py | 25 +- agent/stream_delivery.py | 31 +- agent/transports/chat_completions.py | 33 +- agent/transports/codex.py | 66 +++ agent/transports/codex_app_server.py | 9 + agent/transports/codex_app_server_session.py | 5 + agent/tts_provider.py | 5 +- agent/turn_api_request.py | 6 + agent/turn_context.py | 48 +- agent/turn_context_compaction.py | 17 + agent/turn_facade.py | 5 + agent/turn_facade_lease.py | 3 + agent/turn_final_response.py | 10 + agent/turn_finalizer.py | 43 +- agent/turn_iteration_prep.py | 10 + agent/turn_liveness.py | 13 +- agent/turn_loop_errors.py | 11 + agent/turn_overflow.py | 21 + agent/turn_preflight.py | 38 ++ agent/turn_recovery.py | 20 + agent/turn_request_assembly.py | 4 + agent/turn_tool_round.py | 5 + agent/turn_truncation.py | 8 + agent/usage_pricing.py | 10 +- agent/verification_evidence.py | 4 + agent/vision_message_prep.py | 9 +- agent/web_search_provider.py | 6 +- agent/web_search_registry.py | 6 + batch_runner.py | 10 + cli.py | 210 +++++++- cron/jobs.py | 167 ++++++- cron/lifecycle_guard.py | 100 +++- cron/notepad.py | 1 + cron/scheduler.py | 308 +++++++++++- cron/scheduler_delivery.py | 69 ++- cron/scheduler_preflight.py | 30 +- cron/scheduler_prompt.py | 19 +- cron/scheduler_provider.py | 50 +- cron/scheduler_script.py | 27 +- cron/suggestions.py | 2 + gateway/authz_mixin.py | 33 +- gateway/channel_directory.py | 5 +- gateway/config.py | 43 ++ gateway/config_env.py | 8 + gateway/config_loader.py | 13 + gateway/control_socket.py | 12 +- gateway/delivery.py | 1 + gateway/delivery_ledger.py | 7 +- gateway/drain_control.py | 22 +- gateway/kanban_watchers.py | 1 + gateway/kanban_watchers_common.py | 6 + gateway/kanban_watchers_dispatcher.py | 3 + gateway/kanban_watchers_notifier.py | 16 + gateway/mirror.py | 6 + gateway/pairing.py | 21 + gateway/platform_registry.py | 9 + gateway/platforms/_shared.py | 4 + gateway/platforms/api_server.py | 107 +++- gateway/platforms/api_server_openai_routes.py | 11 +- .../platforms/api_server_run_idempotency.py | 4 + gateway/platforms/api_server_runs.py | 16 + gateway/platforms/base.py | 190 ++++++- gateway/platforms/bluebubbles.py | 9 +- gateway/platforms/qqbot/adapter.py | 9 + gateway/platforms/signal.py | 1 + gateway/platforms/webhook.py | 6 + gateway/platforms/weixin.py | 29 +- gateway/platforms/whatsapp_cloud.py | 3 + gateway/platforms/yuanbao.py | 15 + gateway/profile_routing.py | 4 + gateway/readiness.py | 1 + gateway/relay/__init__.py | 10 +- gateway/relay/adapter.py | 30 ++ gateway/relay/ws_transport.py | 4 + gateway/restart.py | 18 +- gateway/run.py | 470 ++++++++++++++++-- gateway/run_adapters.py | 114 ++++- gateway/run_agent_cache.py | 44 +- gateway/run_busy.py | 51 +- gateway/run_config_loaders.py | 21 +- gateway/run_goals.py | 5 + gateway/run_inbound.py | 23 + gateway/run_notifications.py | 27 +- gateway/run_shutdown.py | 107 +++- gateway/run_startup.py | 54 +- gateway/run_topics.py | 17 +- gateway/run_turn.py | 145 +++++- gateway/run_turn_runner.py | 41 +- gateway/run_voice.py | 11 +- gateway/run_watchers.py | 17 +- gateway/session.py | 31 +- gateway/session_context.py | 5 +- gateway/session_lifecycle.py | 28 +- gateway/session_persistence.py | 33 +- gateway/session_recovery.py | 11 +- gateway/session_stall.py | 10 +- gateway/session_state.py | 12 + gateway/session_transcript.py | 37 +- gateway/shutdown_flush.py | 10 +- gateway/slash_commands.py | 17 +- gateway/slash_commands_model.py | 27 + gateway/slash_commands_session.py | 21 +- gateway/slash_commands_status.py | 2 + gateway/status.py | 49 +- gateway/stream_consumer.py | 36 ++ gateway/stream_consumer_fallback.py | 9 + gateway/stream_consumer_transport.py | 21 +- gateway/turn_context.py | 5 + gateway/wake.py | 11 +- hermes_cli/_early_recovery.py | 16 + hermes_cli/_install_repair.py | 27 +- hermes_cli/_parser.py | 3 + hermes_cli/_scan_venv_blockers.py | 28 +- hermes_cli/_subprocess_compat.py | 55 ++ hermes_cli/active_sessions.py | 9 + hermes_cli/auth.py | 37 +- hermes_cli/auth_codex.py | 22 + hermes_cli/auth_device_flow.py | 16 +- hermes_cli/auth_oauth_grants.py | 2 + hermes_cli/auth_xai.py | 7 + hermes_cli/backup.py | 11 + hermes_cli/banner.py | 1 + hermes_cli/browser_connect.py | 3 + hermes_cli/claw.py | 2 + hermes_cli/cli_agent_setup_mixin.py | 18 +- hermes_cli/cli_chat_turn_mixin.py | 5 + hermes_cli/cli_commands_mixin.py | 58 ++- hermes_cli/cli_info_mixin.py | 8 + hermes_cli/cli_loops_mixin.py | 1 + hermes_cli/cli_modal_mixin.py | 42 +- hermes_cli/cli_model_switch_mixin.py | 26 + hermes_cli/cli_session_mixin.py | 41 +- hermes_cli/cli_status_bar_mixin.py | 15 +- hermes_cli/cli_stream_mixin.py | 1 + hermes_cli/cli_tui_mixin.py | 11 + hermes_cli/cli_voice_mixin.py | 8 +- hermes_cli/codex_models.py | 8 + hermes_cli/codex_runtime_plugin_migration.py | 13 + hermes_cli/commands.py | 5 +- hermes_cli/commands_completion.py | 4 + hermes_cli/commands_platforms.py | 11 + hermes_cli/config.py | 202 +++++++- hermes_cli/config_defaults.py | 77 +++ hermes_cli/config_providers.py | 11 +- hermes_cli/console_engine.py | 2 + hermes_cli/container_boot.py | 32 +- hermes_cli/credential_lifecycle.py | 11 + hermes_cli/cron.py | 27 + hermes_cli/dashboard_auth/__init__.py | 3 + hermes_cli/dashboard_auth/base.py | 9 +- hermes_cli/dashboard_auth/cookies.py | 16 +- hermes_cli/dashboard_auth/registry.py | 5 +- hermes_cli/dashboard_procs.py | 53 +- hermes_cli/debug.py | 1 + hermes_cli/default_soul.py | 11 +- hermes_cli/dep_ensure.py | 5 +- hermes_cli/doctor_config.py | 7 + hermes_cli/doctor_platform.py | 20 +- hermes_cli/doctor_state.py | 2 + hermes_cli/doctor_tools.py | 11 + hermes_cli/dump.py | 11 +- hermes_cli/env_loader.py | 37 +- hermes_cli/fallback_cmd.py | 2 + hermes_cli/gateway.py | 304 ++++++++++- hermes_cli/gateway_windows.py | 73 ++- hermes_cli/goals.py | 17 +- hermes_cli/inventory.py | 22 +- hermes_cli/kanban.py | 15 + hermes_cli/kanban_boards.py | 1 + hermes_cli/kanban_db.py | 53 +- hermes_cli/kanban_db_connect.py | 27 +- hermes_cli/kanban_db_dispatch.py | 53 +- hermes_cli/kanban_db_notify.py | 8 + hermes_cli/kanban_db_workspace.py | 11 +- hermes_cli/kanban_diagnostics.py | 7 +- hermes_cli/kanban_specify.py | 5 + hermes_cli/linux_desktop_entry.py | 40 +- hermes_cli/loops.py | 9 +- hermes_cli/macos_tcc_anchor.py | 4 + hermes_cli/main.py | 53 ++ hermes_cli/main_dashboard.py | 18 +- hermes_cli/main_desktop.py | 77 ++- hermes_cli/main_install_repair.py | 74 ++- hermes_cli/main_provider_setup.py | 5 +- hermes_cli/main_tui_launch.py | 46 +- hermes_cli/main_web_build.py | 37 +- hermes_cli/managed_uv.py | 38 +- hermes_cli/mcp_config.py | 11 + hermes_cli/mcp_security.py | 3 + hermes_cli/mcp_startup.py | 6 + hermes_cli/memory_setup.py | 9 + hermes_cli/moa_cmd.py | 5 + hermes_cli/moa_config.py | 35 +- hermes_cli/model_normalize.py | 8 + hermes_cli/model_setup_flows.py | 3 + hermes_cli/model_setup_flows_common.py | 1 + hermes_cli/model_setup_flows_custom.py | 3 + hermes_cli/model_switch.py | 51 +- hermes_cli/model_switch_providers.py | 8 + hermes_cli/models.py | 43 +- hermes_cli/models_validate.py | 6 + hermes_cli/nous_billing.py | 10 +- hermes_cli/nous_subscription.py | 6 + hermes_cli/npm_engine.py | 2 + hermes_cli/oneshot.py | 8 + hermes_cli/pairing.py | 1 + hermes_cli/plugin_packs.py | 2 + hermes_cli/plugins.py | 120 ++++- hermes_cli/plugins_cmd.py | 29 +- hermes_cli/plugins_dispatch.py | 19 +- hermes_cli/plugins_ledger.py | 14 +- hermes_cli/plugins_loader.py | 42 +- hermes_cli/plugins_manifest.py | 20 +- hermes_cli/plugins_state.py | 5 +- hermes_cli/process_identity.py | 16 +- hermes_cli/profile_cmd.py | 2 + hermes_cli/profile_describer.py | 1 + hermes_cli/profiles.py | 59 ++- hermes_cli/providers.py | 16 +- hermes_cli/pt_input_extras.py | 43 +- hermes_cli/pty_session.py | 2 + hermes_cli/runtime_provider.py | 30 +- hermes_cli/runtime_provider_backends.py | 3 + hermes_cli/runtime_provider_custom.py | 13 + hermes_cli/secrets_cli.py | 9 +- hermes_cli/service_manager.py | 36 +- hermes_cli/session_recovery.py | 7 + hermes_cli/sessions_cmd.py | 11 +- hermes_cli/setup.py | 3 + hermes_cli/setup_platforms.py | 3 + hermes_cli/setup_quick.py | 8 +- hermes_cli/skills_config.py | 5 +- hermes_cli/skills_hub.py | 10 +- hermes_cli/skin_cmd.py | 1 + hermes_cli/sqlite_safe_read.py | 3 + hermes_cli/sqlite_util.py | 6 +- hermes_cli/subcommands/login.py | 3 + hermes_cli/subcommands/secrets.py | 2 + hermes_cli/tools_config.py | 13 + hermes_cli/tools_config_cua.py | 32 +- hermes_cli/tools_config_post_setup.py | 12 + hermes_cli/update_abort_recovery.py | 8 +- hermes_cli/update_cmd.py | 77 ++- hermes_cli/update_cmd_config.py | 19 +- hermes_cli/update_cmd_deps.py | 102 +++- hermes_cli/update_cmd_fleet.py | 101 +++- hermes_cli/update_cmd_git.py | 39 +- hermes_cli/update_cmd_maint.py | 57 ++- hermes_cli/update_cmd_stash.py | 7 + hermes_cli/update_cmd_windows.py | 111 ++++- hermes_cli/update_cmd_zip.py | 38 +- hermes_cli/update_inventory.py | 26 +- hermes_cli/update_receipt.py | 16 + hermes_cli/voice.py | 35 +- hermes_cli/web_git.py | 10 +- hermes_cli/web_routers/actions.py | 14 +- hermes_cli/web_routers/analytics.py | 4 + hermes_cli/web_routers/chat_ws.py | 1 + hermes_cli/web_routers/config_env.py | 9 + hermes_cli/web_routers/dashboard_ui.py | 5 +- hermes_cli/web_routers/files.py | 9 +- hermes_cli/web_routers/messaging.py | 4 + hermes_cli/web_routers/models.py | 2 + hermes_cli/web_routers/ops.py | 9 + hermes_cli/web_routers/profiles.py | 2 + hermes_cli/web_routers/sessions.py | 11 + hermes_cli/web_routers/status.py | 9 + hermes_cli/web_server.py | 8 + hermes_cli/web_server_chat.py | 10 +- hermes_cli/web_server_config.py | 16 +- hermes_cli/web_server_dashboard.py | 39 ++ hermes_cli/web_server_gateway.py | 5 + hermes_cli/web_server_memory.py | 3 + hermes_cli/web_server_profiles.py | 15 +- hermes_cli/web_server_sessions.py | 3 + hermes_cli/worktree_ops.py | 5 + hermes_constants.py | 48 +- hermes_startup_watchdog.py | 1 + hermes_state.py | 98 +++- hermes_state_common.py | 50 +- hermes_state_compression.py | 30 +- hermes_state_dbfile.py | 5 +- hermes_state_errors.py | 12 +- hermes_state_fts.py | 16 +- hermes_state_gateway.py | 33 +- hermes_state_guard.py | 14 +- hermes_state_maintenance.py | 25 +- hermes_state_messages.py | 79 ++- hermes_state_portability.py | 16 +- hermes_state_readpool.py | 30 +- hermes_state_schema.py | 81 ++- hermes_state_search.py | 49 +- hermes_state_sessions.py | 113 ++++- hermes_state_telegram.py | 25 +- hermes_state_usage.py | 24 +- hermes_state_wal.py | 34 +- mcp_serve.py | 8 + model_tools.py | 25 + plugins/dashboard_auth/_shared.py | 4 + plugins/disk-cleanup/disk_cleanup.py | 5 + plugins/image_gen/krea/__init__.py | 1 + plugins/image_gen/openai-codex/__init__.py | 16 + plugins/kanban/dashboard/plugin_api.py | 13 +- plugins/memory/hindsight/__init__.py | 43 +- plugins/memory/hindsight/embedded.py | 22 +- plugins/memory/hindsight/settings.py | 2 + plugins/memory/holographic/store.py | 9 +- plugins/memory/honcho/client.py | 17 +- plugins/memory/honcho/client_cache.py | 6 +- plugins/memory/honcho/session.py | 6 +- plugins/memory/honcho/session_context.py | 12 +- plugins/memory/openviking/__init__.py | 36 +- plugins/model-providers/copilot/__init__.py | 1 + plugins/model-providers/custom/__init__.py | 4 + plugins/model-providers/meta-ai/__init__.py | 1 + .../model-providers/openrouter/__init__.py | 14 + plugins/model-providers/upstage/__init__.py | 3 + plugins/platforms/a2a/adapter.py | 14 +- plugins/platforms/a2a/tools.py | 8 +- plugins/platforms/buzz/adapter.py | 222 ++++++++- plugins/platforms/dingtalk/adapter.py | 24 +- plugins/platforms/discord/adapter.py | 134 ++++- plugins/platforms/email/adapter.py | 45 +- plugins/platforms/feishu/adapter.py | 63 ++- plugins/platforms/google_chat/adapter.py | 18 +- plugins/platforms/google_chat/cards.py | 1 + plugins/platforms/homeassistant/adapter.py | 7 + plugins/platforms/irc/adapter.py | 2 + plugins/platforms/line/adapter.py | 37 +- plugins/platforms/matrix/adapter.py | 75 ++- plugins/platforms/mattermost/adapter.py | 9 + plugins/platforms/photon/adapter.py | 23 +- plugins/platforms/photon/cli.py | 5 + plugins/platforms/raft/adapter.py | 8 +- plugins/platforms/simplex/adapter.py | 3 + plugins/platforms/slack/adapter.py | 270 +++++++++- plugins/platforms/sms/adapter.py | 8 + plugins/platforms/teams/adapter.py | 18 +- plugins/platforms/telegram/adapter.py | 346 ++++++++++++- .../platforms/telegram/telegram_network.py | 20 +- plugins/platforms/wecom/adapter.py | 1 + plugins/platforms/wecom/callback_adapter.py | 11 +- plugins/video_gen/fal/__init__.py | 1 + plugins/web/ddgs/provider.py | 12 +- run_agent.py | 42 +- scripts/release.py | 5 + .../google-workspace/scripts/gws_bridge.py | 7 + tools/ansi_strip.py | 22 +- tools/approval.py | 11 + tools/approval_context.py | 14 +- tools/approval_floors.py | 7 + tools/approval_gateway_wait.py | 3 + tools/approval_human_wait.py | 23 +- tools/approval_prompt.py | 8 + tools/approval_smart.py | 7 +- tools/arg_coercion.py | 2 + tools/async_delegation.py | 24 +- tools/bot_failure_reasons.py | 1 + tools/bot_mode_dm.py | 19 +- tools/bot_relay.py | 39 +- tools/browser_cdp_tool.py | 2 + tools/browser_tool.py | 14 + tools/browser_tool_cloud.py | 2 + tools/browser_tool_install.py | 15 +- tools/browser_tool_lifecycle.py | 30 +- tools/browser_tool_session.py | 9 + tools/browser_use_cli.py | 5 + tools/budget_config.py | 20 +- tools/checkpoint_manager.py | 8 +- tools/clarify_gateway.py | 8 +- tools/clarify_tool.py | 16 +- tools/code_execution_env.py | 9 +- tools/code_execution_tool.py | 12 + tools/code_kernel.py | 6 +- tools/computer_use/backend.py | 8 +- tools/computer_use/cua_backend.py | 21 +- tools/computer_use/cua_backend_capture.py | 16 +- tools/computer_use/cua_backend_driver.py | 17 +- tools/computer_use/cua_backend_input.py | 10 +- tools/computer_use/cua_backend_parse.py | 25 +- tools/computer_use/cua_backend_session.py | 26 +- tools/computer_use/doctor.py | 5 +- tools/computer_use/permissions.py | 6 +- tools/computer_use/tool.py | 22 +- tools/computer_use/vision_routing.py | 5 + tools/credential_files.py | 21 + tools/cronjob_job_args.py | 17 +- tools/cronjob_prompt_scan.py | 11 + tools/cronjob_tools.py | 44 +- tools/daemon_pool.py | 2 + tools/delegate_tool.py | 15 + tools/delegate_tool_child_run.py | 6 + tools/delegate_tool_config.py | 30 +- tools/delegate_tool_dispatch.py | 9 + tools/delegate_tool_progress.py | 1 + tools/delegate_tool_tasks.py | 6 +- tools/drive_preview_tool.py | 1 + tools/env_probe.py | 9 +- tools/environments/base.py | 24 +- tools/environments/base_output.py | 14 + tools/environments/base_session_env.py | 18 +- tools/environments/docker.py | 45 +- tools/environments/local.py | 30 +- tools/environments/local_env_policy.py | 50 +- tools/environments/local_pythonpath.py | 7 +- tools/environments/ssh.py | 1 + tools/file_operations.py | 39 +- tools/file_tools.py | 24 + tools/file_tools_paths.py | 7 +- tools/file_tools_read_tracking.py | 6 +- tools/file_tools_write_guards.py | 22 +- tools/image_generation_tool.py | 10 +- tools/interrupt.py | 5 +- tools/kanban_tools.py | 40 +- tools/lazy_deps.py | 26 +- tools/mcp_oauth.py | 91 +++- tools/mcp_oauth_manager.py | 21 +- tools/mcp_oauth_provider.py | 2 + tools/mcp_tool.py | 38 ++ tools/mcp_tool_agent.py | 1 + tools/mcp_tool_content.py | 15 +- tools/mcp_tool_discovery.py | 25 +- tools/mcp_tool_errors.py | 16 +- tools/mcp_tool_handlers.py | 55 +- tools/mcp_tool_health.py | 26 +- tools/mcp_tool_lifecycle.py | 13 + tools/mcp_tool_loop.py | 3 + tools/mcp_tool_registration.py | 22 +- tools/mcp_tool_schema.py | 18 +- tools/mcp_tool_server_run.py | 34 +- tools/mcp_tool_transport.py | 52 +- tools/memory_tool_store.py | 18 +- tools/osv_check.py | 4 + tools/process_registry.py | 119 ++++- tools/process_registry_notifications.py | 7 +- tools/project_tools.py | 4 + tools/read_extract.py | 7 +- tools/registry.py | 6 + tools/schema_sanitizer.py | 18 +- tools/send_message_senders.py | 17 +- tools/send_message_tool.py | 7 + tools/session_search_tool.py | 17 + tools/setup_mcp_tool.py | 6 + tools/skill_ledger.py | 5 +- tools/skill_manager_batch.py | 4 + tools/skill_manager_guards.py | 26 +- tools/skills_guard.py | 11 +- tools/skills_hub_github.py | 6 + tools/skills_hub_install.py | 13 + tools/skills_hub_models.py | 3 + tools/skills_sync.py | 12 +- tools/skills_sync_bundled_ops.py | 2 + tools/subagent_worktree.py | 3 + tools/terminal_scope.py | 8 +- tools/terminal_tool.py | 16 + tools/terminal_tool_lifecycle.py | 6 + tools/terminal_tool_result.py | 12 + tools/tirith_security.py | 9 + tools/todo_tool.py | 1 + tools/tool_backend_helpers.py | 22 +- tools/tool_search_validation.py | 6 +- tools/tour_tool.py | 1 + tools/transcription_cloud.py | 2 + tools/transcription_command.py | 10 + tools/transcription_common.py | 2 + tools/transcription_local.py | 9 +- tools/transcription_tools.py | 10 + tools/tts_streaming.py | 11 +- tools/tts_text_normalize.py | 7 +- tools/tts_tool.py | 20 +- tools/tts_tool_plugins.py | 9 +- tools/tts_tool_providers.py | 1 + tools/tts_tool_speaker.py | 1 + tools/video_generation_tool.py | 2 + tools/vision_tools.py | 22 + tools/voice_mode.py | 24 +- tools/voice_mode_transcript.py | 1 + tools/web_result_cache.py | 7 +- tools/web_tools.py | 40 +- tools/web_tools_truncate.py | 7 + tools/write_approval.py | 5 +- tools/x_search_tool.py | 9 +- tools/xai_http.py | 9 + toolsets.py | 13 +- tui_gateway/agent_callbacks.py | 2 + tui_gateway/change_watcher.py | 14 +- tui_gateway/compute_host_bridge.py | 6 +- tui_gateway/entry.py | 38 +- tui_gateway/git_probe.py | 7 +- tui_gateway/host_supervisor.py | 3 + tui_gateway/methods_bot_relay.py | 5 + tui_gateway/methods_complete.py | 2 + tui_gateway/methods_config.py | 1 + tui_gateway/methods_config_set.py | 5 + tui_gateway/methods_profiles.py | 41 +- tui_gateway/methods_projects.py | 9 +- tui_gateway/methods_prompt.py | 49 +- tui_gateway/methods_slash.py | 3 + tui_gateway/methods_tools.py | 6 + tui_gateway/methods_voice.py | 13 +- tui_gateway/model_switch.py | 4 + tui_gateway/prompt_turn.py | 24 + tui_gateway/server.py | 145 +++++- tui_gateway/session_auto_continue.py | 26 +- tui_gateway/session_compression.py | 37 +- tui_gateway/session_history.py | 31 +- tui_gateway/session_lifecycle.py | 22 +- tui_gateway/session_notifications.py | 12 +- tui_gateway/session_reaper.py | 12 +- tui_gateway/session_workdir.py | 13 +- tui_gateway/slash_worker.py | 8 +- tui_gateway/tool_progress.py | 7 + tui_gateway/ws.py | 2 + utils.py | 4 + 586 files changed, 13883 insertions(+), 829 deletions(-) diff --git a/acp_adapter/commands.py b/acp_adapter/commands.py index 1de88e0d94..d140b2ace6 100644 --- a/acp_adapter/commands.py +++ b/acp_adapter/commands.py @@ -236,6 +236,8 @@ class SlashCommandsMixin: original_count = len(state.history) # Include system prompt + tool schemas so the figure reflects real request pressure. + # See #6217. + # See #6217. _sys_prompt = getattr(agent, "_cached_system_prompt", "") or "" _tools = getattr(agent, "tools", None) or None approx_tokens = _estimate_tokens(state.history, agent, _sys_prompt, _tools) diff --git a/acp_adapter/content.py b/acp_adapter/content.py index 93842be335..d2137ffd38 100644 --- a/acp_adapter/content.py +++ b/acp_adapter/content.py @@ -103,6 +103,9 @@ def _decode_text_bytes(data: bytes, mime_type: str | None) -> str | None: return data.decode(encoding) except UnicodeDecodeError: continue + # Binary (ELF/Mach-O/PE), not a shell script: feeding its decoded bytes back into the guard tokenizes + # machine code into bogus NUL-bearing paths and crashes the scanner (#77703). Mirror + # lifecycle_guard._read_referenced_script and treat it as nothing to scan. return data.decode("utf-8", errors="replace") diff --git a/acp_adapter/entry.py b/acp_adapter/entry.py index 6b89d68e2c..13dcbcbcc4 100644 --- a/acp_adapter/entry.py +++ b/acp_adapter/entry.py @@ -185,6 +185,9 @@ def main(argv: list[str] | None = None) -> None: # MCP discovery from config.yaml runs in a background daemon thread so the ACP server is # responsive immediately (blocking here cost 2-5 s); per-session MCP servers registered via # asyncio.to_thread are unaffected. Metadata-only hosts can opt out of the global startup. + # Previously this blocked asyncio.run() for 2-5 s. (ACP also registers per-session MCP servers + # dynamically via asyncio.to_thread inside the event loop; that path is unaffected.) Moved from + # model_tools.py module scope to avoid freezing the gateway's loop on lazy import (#16856). if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1": try: from hermes_cli.mcp_startup import start_background_mcp_discovery diff --git a/acp_adapter/permissions.py b/acp_adapter/permissions.py index b1ede2d535..32bc642bb4 100644 --- a/acp_adapter/permissions.py +++ b/acp_adapter/permissions.py @@ -37,6 +37,7 @@ def _build_permission_options( # A gate that re-asks every time (allow_session=False, e.g. protected # agent-instruction writes) collapses to the same two options as a Smart # DENY override — offering a scope Hermes discards would re-prompt every write. + # See #81887. once_only = smart_denied or not allow_session options = [PermissionOption(option_id="allow_once", kind="allow_once", name="Allow once")] if not once_only: diff --git a/acp_adapter/server.py b/acp_adapter/server.py index bf4b237a46..44f8ce2da9 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -554,6 +554,14 @@ class HermesACPAgent(SlashCommandsMixin, acp.Agent): Best-effort: a corrupt message must not turn the load into an error.""" if replay_verb: try: + # Per ACP spec, `session/load` must stream the prior conversation back to the client via + # `session/update` notifications BEFORE responding, so the client receives the full + # transcript within the load request's lifetime. Awaiting the replay here matches Codex / + # Claude Code / OpenCode / Pi and the Zed client (which registers the session-update routing + # entry before awaiting the loadSession RPC specifically so in-call history replay updates + # can find the thread). Deferring this via `loop.call_soon` (as we did briefly in May 2026) + # broke every spec-compliant ACP client that measures notifications synchronously against + # the load response — see #12285 follow-up. await self._replay_session_history(state) except Exception: logger.warning( diff --git a/acp_adapter/session.py b/acp_adapter/session.py index 5ba60a984c..c0b0c157d2 100644 --- a/acp_adapter/session.py +++ b/acp_adapter/session.py @@ -44,6 +44,11 @@ def _normalize_cwd_for_compare(cwd: str | None) -> str: # ``/private/tmp``) that otherwise drop a workspace's own sessions; it is lexical # for missing paths (e.g. WSL-translated drives). try: + # Resolve symlink aliases so equivalent spellings of the same directory compare equal — macOS + # reports editor workspaces as ``/var/...`` while sessions get stored under ``/private/var/...`` + # (and ``/tmp`` vs ``/private/tmp``), which made ACP history filters silently drop a workspace's own + # sessions. WSL-translated Windows drives — keep the previous normpath behavior. Ported from + # PrimeIntellect-ai/prime-agent#628. return os.path.realpath(expanded) except OSError: return os.path.normpath(expanded) @@ -340,6 +345,14 @@ class SessionManager: # incrementally (append_message) and keeps pre-compaction turns as archived # active=0 rows; replace_messages() would DELETE those (and, after a compression # id rotation, clobber the ended parent transcript). Skip it in that case. + # Calling replace_messages() here would then be a redundant double-write that DELETEs exactly + # those archived rows (and, after a compression-driven id rotation where agent.session_id no + # longer equals state.session_id, clobbers the ended parent transcript) — silent data loss for + # any ACP conversation long enough to compress. Only fall back to the destructive atomic replace + # when the agent is NOT persisting itself to this DB (e.g. a test agent factory, or a fresh + # create/fork whose copied history the agent has not flushed yet). That path still rolls back on + # a mid-rewrite failure so the previously persisted conversation survives (salvaged from + # #13675). agent = state.agent if getattr(agent, "_session_db", None) is db and getattr(agent, "_session_db_created", False): return diff --git a/acp_adapter/tools.py b/acp_adapter/tools.py index 8dda3affa3..bf1541a6a2 100644 --- a/acp_adapter/tools.py +++ b/acp_adapter/tools.py @@ -306,6 +306,11 @@ def _format_read_file_result(tool_name: str, data: Args, a: Args) -> Optional[st @_structured() def _format_search_files_result(tool_name: str, data: Args, args: Args) -> Optional[str]: files, matches = data.get("files"), data.get("matches") + # Surface file/image attachments as compact text markers. The thread-context fetch is text-only, so + # without this the agent has no idea prior messages carried images/files at all (#69185, #32315): "@bot + # what do you think of the chart above?" reads as a question about nothing. Markers keep context bounded + # — the agent can ask for a re-share (or the caller may separately deliver the thread root's image, see + # _collect_thread_root_images). if isinstance(files, list): shown = min(len(files), 20) lines = ["File search results", f"Found {_plural(data.get('total_count', len(files)), 'file')}; showing {shown}.", ""] diff --git a/agent/account_usage.py b/agent/account_usage.py index b3a104beea..b6de6e65bb 100644 --- a/agent/account_usage.py +++ b/agent/account_usage.py @@ -304,6 +304,11 @@ def _resolve_codex_usage_credentials( # and hand back a DIFFERENT pool account's usage; such errors must propagate to the fail-open outer guard. # account_id is best-effort: a partial singleton store must not sink a usable credential. try: + # Tier 2: the native runtime resolver. It ALREADY falls back to the credential pool when the + # singleton is empty (see ``resolve_codex_runtime_credentials`` — issue #32992), so in a pool-only + # setup this returns a usable ``source="credential_pool"`` token. A refresh/network error must + # propagate — the outer ``fetch_account_usage`` guard fails open (shows nothing this turn) rather + # than reporting the wrong account. creds = resolve_codex_runtime_credentials(refresh_if_expiring=True) account_id: Optional[str] = None try: diff --git a/agent/activity_tracking.py b/agent/activity_tracking.py index 72acb63203..0f2a385208 100644 --- a/agent/activity_tracking.py +++ b/agent/activity_tracking.py @@ -33,6 +33,8 @@ class ActivityTrackingMixin: ``_touch_activity`` stamps under it and the liveness watchdog samples/commits under it, so a stall observation can never abort a turn that resumed in between. + + Created lazily so ``AIAgent.__new__``-based test doubles keep working. See #95663. """ return _activity_lock(self) @@ -48,6 +50,9 @@ class ActivityTrackingMixin: projection. ``provenance`` names special writers (compression); ``force_persist`` bypasses the SessionDB rate limit. Module-level lock helper, not ``self._liveness_activity_lock()``: doubles bind only ``_touch_activity`` (tests/run_agent/test_session_activity_persist.py). + + Bridge is rate-limited (60s) and best-effort — it never raises into the agent loop. See #31752. + See #72016, #72039. """ from agent.session_activity import ( bound_activity_description, normalize_activity_provenance, @@ -117,6 +122,8 @@ class ActivityTrackingMixin: Keeps ``_last_activity_ts`` so idle/watchdog clocks stay continuous across turns; clears description + provenance so idle agents / SessionDB listings stop advertising the last mid-turn stamp. + + See #15654, #72039. """ self._last_activity_desc = "" self._last_activity_provenance = ActivityProvenance.UNKNOWN diff --git a/agent/agent_init.py b/agent/agent_init.py index ce8e870e7e..d9a739004f 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -443,6 +443,11 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url): # Covers api.meta.ai → codex_responses (prompt caching: 0% on chat vs 93-99%). # URL-driven, not provider-name-driven: `providers.meta` may point anywhere. try: + # Note: provider="meta" without an api.meta.ai base_url (or with a non-api.meta.ai base_url) + # intentionally falls through to chat_completions here. The wire protocol for Meta is URL-driven + # BY DESIGN, not provider-name-driven, because user config `providers.meta` may point at any + # OpenAI-compatible endpoint, and forcing `codex_responses` on the provider name alone would + # break custom endpoints named "meta" that do not host the Responses API. See #63425. from hermes_cli.providers import host_mandated_api_mode as _host_mandated_api_mode _mandated = _host_mandated_api_mode(base_url or "") except Exception: @@ -453,6 +458,8 @@ def _resolve_api_mode(agent, api_mode, provider_name, base_url): def _finalize_routing(agent, api_mode, credential_pool): # Credential-pool validation runs AFTER provider auto-detection so a pool scoped to # "anthropic" isn't rejected for provider=None + anthropic.com URL. + # Regression from #63048 which placed this check before the URL-based auto-detection block above (fixed + # #63425). if credential_pool is not None: try: from agent.credential_pool import credential_pool_matches_provider @@ -481,6 +488,13 @@ def _finalize_routing(agent, api_mode, credential_pool): # exceptions live in _provider_model_requires_responses_api. _base_lower = str(agent.base_url or "").lower() if ( + # GPT-5.x models usually require the Responses API path, but some providers have exceptions (for + # example Copilot's gpt-5-mini still uses chat completions). ACP runtimes are excluded: an ACP + # client handles its own routing and does not implement the Responses API surface. Keyed on the + # `acp://` scheme, not one vendor, so every ACP client is covered. When api_mode was explicitly + # provided, respect it — the user knows what their endpoint supports (#10473). Exception: Azure + # OpenAI serves gpt-5.x on /chat/completions and does NOT support the Responses API — skip the + # upgrade for Azure (openai.azure.com), even though it looks OpenAI-compatible. api_mode is None and agent.api_mode == "chat_completions" and agent.provider != "copilot-acp" @@ -647,6 +661,11 @@ def _init_prompt_cache_config(agent): # unknown values keep "5m". A falsy/off value disables caching entirely (OAuth plans # billing cache writes, proxies adding their own cache_control); the disable survives # /model switches and fallback re-derivation. + # Anthropic supports "5m" (default) and "1h" cache TTL tiers. Read from config.yaml under + # prompt_caching.cache_ttl; unknown values keep "5m". 1h tier costs 2x on write vs 1.25x for 5m, but + # amortizes across long sessions with >5-minute pauses between turns (#14971). This is useful for OAuth + # subscription users where cache writes bill against "extra usage" or for third-party proxies that + # inject their own cache_control markers (#13477). agent._cache_ttl = "5m" with suppress(Exception): from hermes_cli.config import load_config_readonly as _load_pc_cfg @@ -721,6 +740,7 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout): return # ANTHROPIC_TOKEN fallback only for native Anthropic — other anthropic_messages providers # must use their own key or Anthropic credentials leak to third-party endpoints. + # Falling back would send Anthropic credentials to third-party endpoints (Fixes #1739, #minimax-401). _is_native_anthropic = agent.provider == "anthropic" effective_key = api_key or (resolve_anthropic_token() if _is_native_anthropic else None) or "" @@ -742,6 +762,10 @@ def _init_anthropic_client(agent, api_key, base_url, _provider_timeout): agent._anthropic_api_key = effective_key # OAuth only for native Anthropic: third-party anthropic_messages providers must never # trip OAuth paths — those inject Claude-Code identity headers → 401/403. + # Only mark the session as OAuth-authenticated when the token genuinely belongs to native Anthropic. + # Third-party providers (MiniMax, Kimi, GLM, LiteLLM proxies) that accept the Anthropic protocol must + # never trip OAuth code paths — doing so injects Claude-Code identity headers and system prompts that + # cause 401/403 on their endpoints. See #1739. from agent.anthropic_adapter import _is_oauth_token as _is_oat agent._is_anthropic_oauth = _is_oat(effective_key) if (_is_native_anthropic and isinstance(effective_key, str)) else False agent._anthropic_client = build_anthropic_client(effective_key, base_url, timeout=_provider_timeout) @@ -758,6 +782,12 @@ def _init_moa_client(agent, api_key): # build_moa_facade relays "moa.*" events through tool_progress_callback so every surface # shows each reference's answer before the aggregator acts. Display-only; shared with # fallback-restore so a restored facade keeps emitting. + # build_moa_facade wires the reference relay that routes reference-model outputs to the agent's + # tool_progress_callback so every surface that already consumes it (CLI spinner/scrollback, TUI, + # desktop, gateway) can show each reference's answer as a labelled block before the aggregator acts. The + # facade emits "moa.reference", "moa.progress", "moa.phase", and "moa.aggregating" events, forwarded + # through the same callback the tool lifecycle uses. Best-effort and cache-safe — display-only events, + # they never touch the message history. See #53802. agent.client = build_moa_facade(agent, agent.model) agent._client_kwargs = {} agent.api_key = api_key or "moa-virtual-provider" @@ -837,6 +867,9 @@ def _routed_client_kwargs(agent, fallback_model, _provider_timeout) -> Dict[str, # No credentials: try the fallback chain BEFORE failing (an exhausted single-entry pool # must not die with a misleading "No LLM provider configured"); only explicitly named # providers keep the missing-key diagnostic. + # An exhausted single-entry pool (typically ``openrouter`` under free-tier daily quotas) must still + # reach the chain instead of dying at init with a misleading "No LLM provider configured" error. See + # #17929. _explicit = (agent.provider or "").strip().lower() for _fb in _fallback_entries(fallback_model): try: @@ -1237,6 +1270,11 @@ def _init_memory(agent, _agent_cfg, skip_memory, platform): agent._iters_since_skill = 0 # skip_memory skips the external *provider*; enabled_toolsets=["memory"] still gets the # built-in store so the memory tool never sees store=None. + # Flush/background agents can still pass enabled_toolsets=["memory"] so the built-in file store exists + # and the memory tool does not fail with store=None (#65429). A toolset on disabled_toolsets is not a + # request: a caller that denylists memory while its default toolset still names it must not get + # MEMORY.md loaded by an enabled-only check. (Cron agents now run with skip_memory=False and take the + # normal path here.) _memory_toolset_requested = ( "memory" in (agent.enabled_toolsets or []) and "memory" not in (agent.disabled_toolsets or []) @@ -1768,6 +1806,9 @@ def _select_context_engine(_agent_cfg): # parent's. Uncopyable state (locks, DB conns) → built-in with an ACCURATE message. import copy try: + # Copy can fail for engines holding uncopyable state (locks, DB connections, clients); in + # that case fall back to the built-in compressor with an ACCURATE message rather than + # silently mislabelling it "not found". See #42449. _selected_engine = copy.deepcopy(_candidate) except Exception as _copy_err: _copy_failed = True @@ -1812,6 +1853,9 @@ def _build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_c # External engines own compaction policy — the host threshold (and its Codex # autoraise) never reaches the plugin, so drop the notice. agent._compression_threshold_autoraised = None + # External engines own compaction policy: the host compression threshold (including the Codex + # gpt-5.5 autoraise above) only configures the built-in ContextCompressor and never reaches the + # plugin, so the autoraise notice would announce a change that does not apply. (#44439) from agent.model_metadata import get_model_context_length _plugin_ctx_len = get_model_context_length( agent.model, base_url=agent.base_url, api_key=getattr(agent, "api_key", ""), @@ -1924,6 +1968,14 @@ def _inject_context_engine_tools(agent): # Context engine tool schemas (lcm_*), deduped against existing names (plugins may # register the same schemas; duplicates 400 provider-side) and gated on enabled_toolsets # so `platform_toolsets: telegram: []` can't leak them. + # Skip names that are already present — the _ra().get_tool_definitions() quiet_mode cache returned a + # shared list pre-#17335, so a stray mutation here would poison subsequent agent inits in the same + # Gateway process and trip provider-side 'duplicate tool name' errors. Even with the cache fix, dedup is + # the right defense against plugin paths that may register the same schemas via ctx.register_tool(). + # Mirrors the memory tools dedup above. Respect the platform's enabled_toolsets configuration (#5544): + # context engine tools follow the same gating pattern as memory provider tools — without the gate, + # `platform_toolsets: telegram: []` would still leak lcm_* tools into the tool surface and incur the + # same local-model latency penalty. agent._context_engine_tool_names: set = set() if ( agent.context_compressor @@ -1939,6 +1991,7 @@ def _inject_context_engine_tools(agent): if _schema is None: # A nameless tool makes strict providers 400 and disables the whole toolset. _ra().logger.warning( + # Skip it. See #47707. "Context engine returned a tool schema with no resolvable " "name; skipping to avoid poisoning the request (%r)", _raw_schema, @@ -2002,6 +2055,11 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length): ) # Recalibrate the compressor to the served window: every request runs at num_ctx, so a # trigger derived from the probed model window could sit above it and never fire. + # A config that sets only model.ollama_num_ctx (without model.context_length) previously left the + # compressor targeting the probed window while the server truncated/rejected at num_ctx — the compaction + # trigger could sit several times ABOVE the real served window and never fire. Clamp the compressor's + # window to the effective num_ctx so threshold math operates on the context the server actually serves. + # (Overlaps #60103's silent-clamp dead zone; this is the init-order half.) _cc_window = getattr(agent.context_compressor, "context_length", 0) or 0 if agent._ollama_num_ctx and agent._ollama_num_ctx > 0 and _cc_window and agent._ollama_num_ctx < _cc_window: _ra().logger.info( @@ -2020,6 +2078,9 @@ def _emit_compression_summary(agent, cs): _autoraise = agent._compression_threshold_autoraised or {} _autoraise_notice = None if ( + # A change in the raised threshold (or the autoraised model) updates the marker state and + # re-notifies once. The config display gate (compression.codex_gpt55_autoraise_notice) still + # suppresses the banner entirely without disabling the threshold autoraise. See #54432. bool(_autoraise) and cs.enabled and cs.autoraise_notice_enabled diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 5f3d0ede46..870e46f479 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -223,6 +223,14 @@ def sanitize_tool_call_arguments( ``cursor["prefix"]`` holds strong refs (not ``id()``: address reuse aliases) to the messages validated last call; the ``is``-identical prefix is skipped. Safe because only the surrogate sanitizers mutate live dicts; every other path replaces dicts, breaking identity. + + Safety argument for skipping: a message in the matched prefix was fully scanned before — every tool_call + argument was either already valid JSON or was rewritten to ``"{}"`` (valid). The only code paths that + mutate ``function["arguments"]`` on live history dicts between calls are the surrogate / non-ASCII + sanitizers, which substitute characters *inside* JSON string values and cannot invalidate JSON syntax. + Compression, repair, undo, and steer paths replace or reorder message dicts, which breaks the identity + match and forces a re-scan. Holding strong references (the objects themselves, not ``id()``s) makes + address reuse aliasing (#50372-style) impossible. """ log = logger or logging.getLogger(__name__) if not isinstance(messages, list): @@ -252,6 +260,8 @@ def sanitize_tool_call_arguments( continue # Canonical ``call_id || id`` precedence so scan and stub share the id the pipeline # uses; bare ``id`` misses Codex call_id results and orphans a stub. + # Keying on bare ``id`` here would fail to find a result built with ``call_id`` (Codex Responses + # format) and insert a duplicate stub that itself becomes an orphan (#58168). tool_call_id = _ra().AIAgent._get_tool_call_id_static(tool_call) or None function_name = function.get("name", "?") # Log the FULL (bounded) argument string: we are about to overwrite the only copy, which @@ -365,6 +375,15 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None: else: # Drop a stale ``tool_calls: []`` at the source: strict providers (DeepSeek v4, Kimi) 400 on # it and it persists into replayed history. + # Neither turn carries tool calls, but the surviving turn may still carry a stale ``tool_calls: []`` + # from the earlier message. An empty array is semantically "no tool calls", yet strict + # OpenAI-compatible providers (DeepSeek v4, Moonshot/Kimi) reject it with HTTP 400 ("Invalid + # 'messages[N].tool_calls': empty array..."). Drop the key HERE, at the source: + # ``sanitize_api_messages`` only fixes the per-call wire copy, so a ``[]`` left on the repaired turn + # survives in the live/persisted trajectory returned to callers (gateway/WebUI transcripts, session + # resume, subagents, cron) and is replayed on the next turn — which is how #58755 kept reproducing + # after the chokepoint fix (#77921). Popping is non-destructive: an empty array carries no + # information. prev.pop("tool_calls", None) # Concatenate plain-text content only; leave multimodal (list) content alone. prev_content = prev.get("content") @@ -374,6 +393,8 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None: joined = "\n".join(p for p in (prev_content.strip(), new_content.strip()) if p) prev["content"] = joined # A falsy new_content leaves ``joined`` == prev_content; that is not a rewrite. + # "") strips to nothing and ``joined`` collapses back to ``prev_content`` unchanged -- that must NOT + # count as a rewrite (wz-heng, #78063 review). content_rewritten = joined != prev_content elif not prev_content and new_content is not None: prev["content"] = new_content @@ -384,6 +405,18 @@ def _merge_assistant_into(prev: Dict, msg: Dict) -> None: prev["reasoning_content"] = msg["reasoning_content"] # A stale ``api_content`` sidecar overrides ``content`` at API-build time and would replay # pre-merge bytes; drop it only when content actually changed. + # ``prev`` may carry an ``api_content`` sidecar (the exact bytes previously sent to the API, e.g. a + # sanitize-divergence stamp — see ``_flush_messages_to_session_db``) from BEFORE this merge. The sidecar + # takes priority over ``content`` at API-build time (``conversation_loop``'s ``api_messages`` build + # substitutes it back in for role ``assistant``), so leaving it in place while ``prev["content"]`` + # changes would silently replay the pre-merge bytes and discard everything this merge just concatenated + # on — the same stale-field-survives-the-merge shape as the ``tool_calls`` gap above, just for a + # different field. Only drop it when the merge actually changed the resulting value (e.g. the later + # turn's content is ``None``, or either side is multimodal/list — both branches skip the reassignment + # and ``prev["content"]`` is untouched; a falsy ``new_content`` that strips to nothing also leaves + # ``joined`` equal to the original ``prev_content``): in those cases the sidecar is still the exact + # bytes previously sent for the UNCHANGED content, and dropping it would break the prompt-cache replay + # invariant for no reason (wz-heng, #78063 review). if content_rewritten: drop_stale_api_content(prev) @@ -416,6 +449,11 @@ def _drop_stray_tool_results(messages: List[Dict]) -> Tuple[List[Dict], int]: alias is not replayed to strict providers.""" repairs = 0 known_tool_ids: Dict[str, int] = {} # alias -> group id; reset by assistant/user turns + # Pass 1: drop stray tool messages that don't follow a known assistant tool call. A Responses call can + # have several equivalent spellings (call_id, id, response_item_id, or a composite ``call|item`` id), so + # consume the whole alias group when one spelling is matched. Alias expansion lives in + # ``agent.message_sanitization.tool_call_id_variants`` / ``tool_result_id_variants`` (single policy + # owner) — which also handles SDK tool_call objects, preserving the #91768 dict-or-object tolerance. matched_tool_groups: set = set() next_tool_group = 0 filtered: List[Dict] = [] @@ -644,6 +682,18 @@ def _is_entitlement_403(agent, status_code, error_context) -> bool: if status_code != 403: return False haystack = " ".join( + # Subscription/entitlement 403s look like auth failures on the wire but refresh cannot fix them — + # the OAuth token is already valid, the account simply lacks the entitlement. Without this guard, + # the refresh path keeps minting fresh tokens against the same unsubscribed account and the main + # agent loop spins re-issuing the same 403 until the user Ctrl+C's. Defense-in-depth for #26847: + # xAI's backend has been seen to 403 standard SuperGrok subscribers with bodies that don't match the + # existing entitlement keyword set in ``_is_entitlement_failure``. Any 403 against ``xai-oauth`` is + # treated as entitlement here so the refresh loop can't spin in those cases either. Exception + # (#29344): xAI's ``[WKE=unauthenticated:...]`` suffix and the ``OAuth2 access token could not be + # validated`` phrasing are xAI's authoritative "this is a stale token, not entitlement" signal. When + # either fires we must NOT apply the catch-all override — refresh is the recoverable path for these + # bodies, and blanket-classifying them as entitlement was the bug that left long-running TUI + # sessions stuck on stale tokens until the user exited and reopened. str(error_context.get(k) or "").lower() for k in ("message", "reason", "code", "error") if isinstance(error_context, dict) @@ -746,6 +796,11 @@ def recover_with_credential_pool( # The pool belongs to the PRIMARY provider: acting on fallback errors would corrupt its state # and reset base_url to the primary endpoint. Empty pool provider means unscoped; empty agent # provider is a mismatch (swap would leave provider="" model=""). + # Defensive guard: if a fallback provider is active and its provider name doesn't match the pool's + # provider, the pool belongs to the PRIMARY provider. Mutating it based on fallback errors would corrupt + # the primary's credential state (see #33088) and, via _swap_credential, overwrite the agent's base_url + # back to the primary's endpoint — every subsequent request then goes to the wrong host and 404s (see + # #33163). The pool should only act when the agent is still on the same provider that seeded the pool. current_provider = (getattr(agent, "provider", "") or "").strip().lower() pool_provider = (getattr(pool, "provider", "") or "").strip().lower() if pool_provider and not credential_pool_matches_provider( @@ -860,6 +915,10 @@ def _rebuild_primary_client(agent, rt: Dict[str, Any], *, reason: str) -> None: # reference_callback relay survives recovery. from agent.moa_loop import build_moa_facade agent.client = build_moa_facade(agent, agent.model) + # MoA is a virtual chat-completions provider. It never has real OpenAI client kwargs; restoring it + # after a fallback must recreate the facade, not call OpenAI() with an empty api_key. Use the shared + # factory so the restored facade keeps the reference_callback relay wired at init — a bare + # MoAClient() would silently stop emitting moa.reference/moa.aggregating display events (#53802). agent._anthropic_client = None elif agent.api_mode == "anthropic_messages": _build_anthropic_client_from_runtime(agent, rt) @@ -885,6 +944,11 @@ def try_recover_primary_transport( try: # Never hard-close the shared client here: stale streaming workers may still be unwinding on # the old pool; _retire_shared_openai_client defers FD release to GC. + # Retire the existing client to release stale connections. #70773: never hard-close the shared + # client here — this runs on the conversation-loop thread while workers from stale-killed streaming + # attempts may still be unwinding their SSL BIOs on the old pool. ``_retire_shared_openai_client`` + # shuts the sockets down (FD-safe from any thread) and defers the FD release to GC, which cannot + # complete until every borrowing thread has unwound. if getattr(agent, "client", None) is not None: with contextlib.suppress(Exception): agent._retire_shared_openai_client(agent.client, reason="primary_recovery") @@ -893,6 +957,9 @@ def try_recover_primary_transport( if agent.api_mode == "anthropic_messages": _build_anthropic_client_from_runtime(agent, rt) elif (agent.provider or "").strip().lower() == "moa": + # MoA is a virtual provider with empty client_kwargs — rebuilding via _create_openai_client + # would raise "api_key client option must be set". Recreate the facade through the shared + # factory so the reference_callback relay survives recovery (#53802). from agent.moa_loop import build_moa_facade agent.client = build_moa_facade(agent, agent.model) else: @@ -1021,6 +1088,8 @@ def _rebind_primary_credential_pool(agent, primary_provider, matches_primary, lo return if matches_primary(entry): # _swap_credential rebuilds the client and reapplies base-url-scoped headers. + # ``_swap_credential`` rebuilds the OpenAI/Anthropic client, reapplies base-url-scoped headers, and + # carries the accumulated base_url / OAuth-detection fixes (#33163). agent._swap_credential(entry) logger.info( "Restore re-selected pool entry %s (%s)", @@ -1043,6 +1112,11 @@ def restore_primary_runtime(agent) -> bool: # _fallback_index past the chain end and silently block future fallbacks. agent._fallback_index = 0 return False + # Reset the chain index even when no fallback was activated this turn. Without this, a turn where + # _try_activate_fallback() was called but returned False (chain exhausted or provider not configured) + # leaves _fallback_index >= len(_fallback_chain) while _fallback_activated stays False. The next turn + # skips this block entirely, stranding the index and silently blocking all future fallback attempts for + # the session. Fixes #20465. if getattr(agent, "_rate_limited_until", 0) > time.monotonic(): return False # primary still in rate-limit cooldown, stay on fallback rt = agent._primary_runtime @@ -1150,6 +1224,7 @@ def extract_reasoning(agent, assistant_message) -> Optional[str]: if not parts and isinstance(content, list): # DeepSeek V4 Pro returns typed content blocks ({"type": "thinking", ...}); dropping them # makes the next turn fail with HTTP 400 "thinking must be passed back". + # Refs #21944. for block in content: if isinstance(block, dict) and block.get("type") == "thinking": _add((block.get("thinking") or block.get("text") or "").strip()) @@ -1253,7 +1328,12 @@ def _raw_cache_ttl_from_config(default: Any) -> Any: def prompt_caching_disabled_from_config() -> bool: - """True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``).""" + """True when ``prompt_caching.cache_ttl`` is configured as off (same detection as ``agent_init``). + + Same disable detection as ``agent_init`` (via ``cache_ttl_means_disabled``) so stub-based policy paths + (MoA slot decoration, auxiliary fallback replan) honor the same config contract without holding a live + ``AIAgent`` (#76085 / #33555). + """ return cache_ttl_means_disabled(_raw_cache_ttl_from_config("5m")) @@ -1281,11 +1361,18 @@ def plan_cache_sections_for_destination( """Plan request-local cache sections for one resolved destination (MoA / auxiliary senders): stripped copies (non-caching route) or a ``build_prompt_cache_plan`` layout; never mutates inputs. ``cache_disabled``/``cache_ttl`` default to live config so the operator's disable and - tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint.""" + tier are honored; ``static_system_prefix`` gives the system prompt the main loop's early breakpoint. + + ``cache_disabled`` threads the operator's ``prompt_caching.cache_ttl`` disable into the blank policy + stub. When omitted, the live config is consulted so MoA/auxiliary paths cannot re-enable markers after + the user turned caching off (#76085). + """ from agent.prompt_caching import ( build_prompt_cache_plan, effective_cache_ttl, envelope_tool_part_cache_markers_supported, strip_anthropic_cache_control, strip_anthropic_tool_cache_control, ) + # The policy function reads agent.* only as fallbacks for kwargs we don't pass; blank_cache_policy_stub + # is the only sanctioned stub so _cache_disabled cannot be left off again (#76085). stub = blank_cache_policy_stub(cache_disabled) dest = dict(provider=provider, base_url=base_url, api_mode=api_mode, model=model) should_cache, native_layout = anthropic_prompt_cache_policy(stub, **dest) @@ -1386,6 +1473,11 @@ def anthropic_prompt_cache_policy( envelope (OpenRouter / OpenAI-wire proxies; Qwen/Alibaba too). The operator disable is read from ``_cache_disabled`` (not ``_cache_ttl``, unset during init) so it survives switches and restores. Branch ORDER is load-bearing (see inline notes). + + Qwen / Alibaba-family models on OpenCode, OpenCode Go, and direct Alibaba (DashScope) also honour + Anthropic-style ``cache_control`` markers on OpenAI-wire chat completions. Upstream pi-mono #3392 / pi + #3393 documented this for opencode-go Qwen. Without markers these providers serve zero cache hits, + re-billing the full prompt on every turn. """ if getattr(agent, "_cache_disabled", False): return (False, False) @@ -1403,6 +1495,10 @@ def anthropic_prompt_cache_policy( is_claude = "claude" in model_lower # Kimi/Moonshot via OpenRouter uses the same envelope cache_control as Claude; without this it # serves ~1% cache hits. Family matcher covers bare k1./k2. slugs. + # Without this branch moonshotai/kimi-k2.6 falls through to (False, False), serving ~1% cache hits on + # 64K-token prompts and re-billing the full prompt on every turn. Observed within-turn progression with + # cache enabled: 1% → 67% → 84% → 97% (#25970). Reuses the canonical family matcher (covers bare + # k1./k2./k25 release slugs the substring check missed). from agent.anthropic_adapter import _model_name_is_kimi_family is_kimi = _model_name_is_kimi_family(eff_model) or "moonshot" in model_lower is_openrouter = base_url_host_matches(eff_base_url, "openrouter.ai") @@ -1471,6 +1567,11 @@ def anthropic_prompt_cache_policy( # Qwen/Alibaba on OpenCode and DashScope accept envelope cache_control on the OpenAI wire # (pi-mono's "alibaba" cacheControlFormat). DeepSeek on OpenCode is excluded: its relay 400s on # block-array content. Family set/predicate shared with the effective_cache_ttl clamp. + # Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire transport that accepts + # Anthropic-style cache_control markers and rewards them with real cache hits. Without this branch + # qwen3.6-plus on opencode-go reports 0% cached tokens and burns through the subscription on every turn. + # OpenCode Zen's relay rejects the Anthropic-style content block format that cache markers produce + # (content becomes a block array instead of a plain string), causing HTTP 400 (#77217). from agent.prompt_caching import ALIBABA_FAMILY_PROVIDERS, is_qwen_model if provider_lower in ALIBABA_FAMILY_PROVIDERS and is_qwen_model(model_lower): return True, False @@ -1569,9 +1670,20 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo from agent.ssl_verify import resolve_httpx_verify # Treat client_kwargs as read-only: callers pass agent._client_kwargs, and in-place mutation # leaks into later requests (a torn-down httpx transport got reused). + # Callers pass agent._client_kwargs (or shallow copies of it) in; any in-place mutation leaks back into + # the stored dict and is reused on subsequent requests. #10933 hit this by injecting an httpx.Client + # transport that was torn down after the first request, so the next request wrapped a closed transport + # and raised "Cannot send a request, as the client has been closed" on every retry. The revert resolved + # that specific path; this copy locks the contract so future transport/keepalive work can't reintroduce + # the same class of bug. client_kwargs = dict(client_kwargs) # The MoA virtual provider has no OpenAI wire endpoint; the facade *is* the client. Rebuild the # facade, never a native client (TypeError; relay re-wire). + # Rebuilding a native OpenAI client while agent.provider == "moa" (client replacement, stream-retry pool + # cleanup, credential rotation, fallback+restore) drops the facade: the next primary call either raises + # a `_moa_prepared_request` TypeError (#78382) or, when _client_kwargs carry an unrelated relay + # base_url, leaks the request to a foreign gateway. Rebuild the facade instead (build_moa_facade also + # re-wires the reference relay, see #53802). if (getattr(agent, "provider", "") or "").strip().lower() == "moa": from agent.moa_loop import build_moa_facade return build_moa_facade(agent, getattr(agent, "model", None) or "default") @@ -1604,12 +1716,26 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo # behind a per-client view whose ``close()`` is a no-op for the pool, so a closed wrapper # never takes a sibling's (or the successor's) connections with it # (tests/agent/test_shared_http_transport.py). + # Without this, a peer that drops mid-stream leaves the socket in a state where epoll_wait never fires, + # ``httpx`` read timeout may not trigger, and the agent hangs until manually killed. Probes after 30s + # idle, retry every 10s, give up after 3 → dead peer detected within ~60s. Safety against #10933: the + # ``client_kwargs = dict(client_kwargs)`` above means this injection only lands in the local per-call + # copy, never back into ``agent._client_kwargs``. Each ``_create_openai_client`` invocation therefore + # gets its OWN fresh ``httpx.Client`` whose lifetime is tied to the OpenAI client it is passed to. When + # the OpenAI client is closed (rebuild, teardown, credential rotation), the paired ``httpx.Client`` + # closes with it, and the next call constructs a fresh one — no stale closed transport can be reused. if "http_client" not in client_kwargs: keepalive_http = agent._build_keepalive_http_client(client_kwargs.get("base_url", ""), verify=httpx_verify) if keepalive_http is not None: client_kwargs["http_client"] = keepalive_http # Retries belong to the outer conversation loop (honors Retry-After); SDK retries would # double-retry inside it. auxiliary_client keeps SDK retries as it isn't wrapped. + # Delegate all rate-limit / 5xx retry to hermes's outer conversation loop, which honors Retry-After and + # applies adaptive/jittered backoff. The OpenAI SDK default (max_retries=2) uses its own 1-2s backoff + # that ignores Retry-After and double-retries inside our loop — the same deadlock the Anthropic clients + # hit (#26293). This is the single chokepoint every primary OpenAI/aggregator client passes through + # (init, switch_model, recovery, restore, request-scoped); auxiliary_client builds its own clients and + # keeps SDK retries because it is NOT wrapped by the conversation loop. client_kwargs.setdefault("max_retries", 0) _ensure_copilot_headers(client_kwargs) # OpenCode Free is served anonymously: any unrecognized bearer is a 401, so an empty @@ -1768,6 +1894,9 @@ def _build_switched_client(agent, new_provider, api_key, base_url, api_mode, new ) # Read live config, not agent._custom_providers, so mid-session ssl_ca_cert / ssl_verify # edits are honored. + # Read custom_providers from live config (not the init-time snapshot on ``agent._custom_providers``) + # so ssl_ca_cert / ssl_verify edits are honored when switching mid-session, matching the + # context-length reload below (#15779). apply_custom_provider_tls_to_client_kwargs( agent._client_kwargs, str(effective_base or ""), get_compatible_custom_providers(load_config_readonly()), @@ -1906,6 +2035,7 @@ def _build_primary_runtime_snapshot(agent, api_mode) -> Dict[str, Any]: "reasoning_echo_flag": getattr(agent, "_reasoning_echo_flag", False), # Overrides must travel with the switched-to identity or a later recovery/restore resurrects # PRE-switch overrides from the stale init snapshot. + # See #75091. "request_overrides": dict(getattr(agent, "request_overrides", {}) or {}), "runtime_capabilities": dict(getattr(agent, "runtime_capabilities", {}) or {}), "compressor_model": getattr(cc, "model", agent.model), @@ -1973,6 +2103,12 @@ def switch_model( snapshot and re-raises (callers catch).""" old_model = agent.model old_provider = agent.provider + # ── Reload credential pool for the new provider (issue #52727) ── Without this, + # ``recover_with_credential_pool`` sees a ``pool.provider != agent.provider`` mismatch and + # short-circuits, leaving the new provider with no rotation/recovery on 401/429 and burning the original + # pool's entries. Only reload when the provider actually changed (or the pool was missing) — + # re-selecting the same provider must not churn the pool reference. A reload failure is logged + + # swallowed: the switch itself must still complete. old_norm = (old_provider or "").strip().lower() new_norm = (new_provider or "").strip().lower() api_mode, base_url, destination_capabilities = _resolve_switch_destination( @@ -2131,6 +2267,12 @@ def repair_tool_call(agent, tool_name: str) -> str | None: # VolcEngine api/plan leaks XML attribute fragments into tool_use.name (`terminal" # parameter="command" ...`); trim at the first quote/angle bracket. Do NOT split on whitespace: # "write file" must reach ``_norm`` -> ``write_file``. + # `terminal" parameter="command" string="true` `execute_code" parameter="code" string="true` + # `session_search" parameter="session_id" string="true` We trim at the first unambiguous XML/quote + # character so the rest of the repair pipeline (lowercase / snake_case / fuzzy match) can resolve the + # cleaned name to a real tool. Crucially we DO NOT split on whitespace: legitimate inputs like "write + # file" must keep flowing through ``_norm`` -> ``write_file`` (covered by test_space_to_underscore in + # tests/run_agent/test_repair_tool_call_name.py). See #33007. for _xml_sep in ('"', "'", "<", ">"): _idx = tool_name.find(_xml_sep) if _idx > 0: @@ -2171,12 +2313,19 @@ _INTERRUPTED_PLACEHOLDER = "[response interrupted]" # Escalate repeated heals once per session window, then stay quiet. Default threshold; tunable via # ``agent.sanitizer_heal_escalation_threshold`` (<= 0 disables). +# Repeated heals of the same poisoned transcript used to WARNING on every send (#96870). +# ``_EMPTY_HEAL_ESCALATE_AFTER`` is the built-in default; deployments tune it via +# ``agent.sanitizer_heal_escalation_threshold`` in config.yaml (<= 0 disables escalation entirely — WARNINGs +# still fire per window). _EMPTY_HEAL_ESCALATE_AFTER = 3 _EMPTY_HEAL_WINDOW_S = 600.0 _empty_heal_log_state: Dict[str, Dict[str, Any]] = {} _empty_heal_log_lock = threading.Lock() # Sessions already told ONCE (out-of-band, never in conversation context); kept apart from the # windowed log state so a new window never re-arms the notice. +# Session keys that already received the one-time user notice. Separate from the windowed log state so a new +# 10-minute window never re-notifies: the user is told ONCE per session, ever (#96870 — out-of-band, +# delivery channel only, never injected into conversation context). _empty_heal_user_notified: set = set() # One-shot pending notices keyed by session, drained via ``consume_pending_sanitizer_heal_notice`` # and delivered via the status/warning callback. @@ -2374,12 +2523,31 @@ def _drop_invalid_roles(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: def _drop_empty_tool_calls_arrays(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """Strict providers 400 on ``tool_calls: []``; normalize on shallow copies so history stays byte-stable.""" + # --- Drop empty / malformed tool_calls arrays on assistant messages --- An assistant message carrying + # ``tool_calls: []`` (an empty array) — or a non-list value under the key — is semantically identical to + # an assistant message with no tool calls, but strict OpenAI-compatible providers reject the empty array + # outright: DeepSeek v4 returns HTTP 400 "Invalid 'messages[N].tool_calls': empty array. Expected an + # array with minimum length 1, but got an empty array instead." (#58755, follow-up to #56980). Empty + # arrays reach here from session resume, host-fed histories, or the consecutive-assistant merge in + # ``repair_message_sequence`` (which preserves a pre-existing ``[]`` on the surviving turn). This is the + # final pre-API chokepoint, so normalize defensively — and, per the #56980 review, do it HERE on the + # per-call copy rather than in ``repair_message_sequence``, which would destructively rewrite the + # persisted trajectory. Shallow-copy the message before dropping the key so stored history (and prompt + # caching) stays byte-stable. normalized: List[Dict[str, Any]] = [] dropped = 0 for msg in messages: if ( isinstance(msg, dict) and msg.get("role") == "assistant" + # Defense-in-depth: a strict OpenAI-compatible provider (e.g. onerouter / Qwen, DeepSeek v4) + # rejects an assistant message carrying ``tool_calls: []`` (empty array) with HTTP 400 "Empty + # tool_calls is not supported in message." The pre-API sanitizer in agent_runtime_helpers drops + # these, but only on the conversation_loop path — other routes can reach the wire without it. + # For every request that serializes through this transport (conversation loop and any caller + # using it), this is the last boundary, so normalize here. Requests built by fully separate + # payload paths (e.g. some auxiliary clients) never pass through this layer and are out of scope + # for it. (#58755 follow-up) and "tool_calls" in msg and not (isinstance(msg["tool_calls"], list) and msg["tool_calls"]) ): @@ -2444,6 +2612,22 @@ def _pair_tool_calls_positionally(messages: List[Dict[str, Any]]) -> List[Dict[s """Positional tool_call <-> tool_result pairing: strict providers (DeepSeek v4, Kimi) require results IMMEDIATELY after their call. Drops positional orphans, stubs unanswered declared ids; matching is alias-aware.""" + # --- Positional tool_call <-> tool_result pairing --- Strict OpenAI-compatible providers (DeepSeek v4, + # Kimi) enforce the POSITIONAL invariant: an assistant message carrying tool_calls must be IMMEDIATELY + # followed by tool messages covering every tool_call_id. The previous implementation compared global id + # sets, which misses the failure mode where a result exists somewhere in the transcript but not in the + # run right after its call — an interrupted turn or a compression window can displace a result past a + # user turn. The id then survives in the global result set, so the call looks answered, no stub is + # injected, and the provider rejects the payload with HTTP 400 "An assistant message with 'tool_calls' + # must be followed by tool messages responding to each 'tool_call_id' (insufficient tool messages + # following tool_calls message)". Rewritten as a single rolling walk on the per-call copy (#94704): (a) + # tool results that do not immediately follow an assistant message declaring their id are dropped + # (positional orphans — includes results appearing BEFORE their call, which strict providers also + # reject); (b) declared ids not covered by the immediately-following tool run get a stub result injected + # at the end of that run, even when a mispositioned result exists elsewhere. Matching is variant-aware + # (``tool_call_id_variants`` / ``tool_result_id_variants``): a result keyed on ANY alias spelling + # (``id`` / ``call_id`` / ``response_item_id`` / composite bridge) answers the call, preserving the + # unified alias policy from #55626/#63000/#93251. paired: List[Dict[str, Any]] = [] declared_calls: Dict[str, tuple] = {} dropped = 0 @@ -2504,6 +2688,24 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any] (not ids ever seen) because llama.cpp reuses one constant id, and whole variant groups so alias-keyed results are not deleted.""" outstanding: Dict[str, int] = {} # every alias of an unanswered call -> its group id + # 3. Deduplicate tool_call_ids. Strict providers (DeepSeek) reject a payload where the same tool_call_id + # appears more than once with HTTP 400 "Duplicate value for 'tool_call_id'" (#58327). Duplicates can + # arise from retries, crash/resume glitches, or a compression window that re-emits a tool result. This + # is the final pre-API chokepoint, so dedup defensively here even though repair_message_sequence also + # consumes matched ids. (a) collapse duplicate tool_calls WITHIN an assistant message (b) drop tool + # results that answer no OUTSTANDING tool call (b) tracks outstanding calls rather than every id ever + # seen, because ``tool_call_id`` is NOT globally unique in practice: llama.cpp emits a single constant + # id for every tool call it ever returns (verified: three separate completions from one server all + # carry the same id). A seen-once-drop-forever rule reads the SECOND legitimate tool result of such a + # session as a duplicate and deletes it, so from the second tool call onward the model never sees any + # result — it announces its next action and the turn dies with the work unfinished. Outstanding-call + # semantics keep both protections intact: a re-emitted result still answers no pending call and is + # still dropped, while a genuine new call that reuses the id re-arms that id first. Variant-group + # tracking: answering or deduping one spelling consumes its siblings too. A Codex/Responses tool_call + # registers ``id`` (fc_...), ``call_id`` (call_...), ``response_item_id``, and composite spellings + # (#55626/#58168/#63000); tracking only the coalesced id here made a result keyed on any OTHER variant + # look like it answered no outstanding call, so this pass deleted the very result step 2's + # variant-aware matching had just preserved (issue #93251 — whole parallel batches vanished). outstanding_groups: Dict[int, frozenset] = {} next_group_id = 0 deduped: List[Dict[str, Any]] = [] @@ -2536,6 +2738,9 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any] continue if candidate_groups: # Consume EVERY variant of the matched call; ids are re-armed by the next call reusing them. + # Consume the whole alias group so a SECOND result replaying any sibling spelling falls into + # the drop branch below — strict providers reject duplicate tool_call_ids with HTTP 400 + # (#58327, #66974). Credit: #55436. group_id = min(candidate_groups) for variant in outstanding_groups.pop(group_id, frozenset()): if outstanding.get(variant) == group_id: @@ -2552,6 +2757,22 @@ def _dedupe_tool_call_ids(messages: List[Dict[str, Any]]) -> List[Dict[str, Any] def _realign_tool_result_names(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """Align each tool result's wire ``name`` with its call's function name (per-call copy only): Google 400s on a mismatch, routine when tool_search bridges via ``tool_call``.""" + # 4. Google matches functionResponse.name against functionCall.name and rejects a mismatch with HTTP 400 + # "Request contains an invalid argument" (INVALID_ARGUMENT); behind an OpenAI-compatible gateway that + # surfaces only as a generic "Provider returned error". When tool_search defers MCP/plugin tools the + # model calls the bridge tool ``tool_call``, while ``make_tool_result_message()`` labels the result + # with the unwrapped internal tool name (``mcp__github__create_issue``) that dispatch, hooks, logging, + # and guardrails need. #72089 fixed exactly this for the native Gemini adapter, which now prefers + # ``tool_name_by_call_id`` over the result name; requests that reach Gemini through the + # OpenAI-compatible path (OpenRouter, Vertex/LiteLLM proxies, any OpenAI-shaped gateway) skip that + # translation entirely and still send the internal name on the wire. Normalizing here rather than in + # the OpenAI-compat serializer keeps it provider-agnostic: Gemini reaches Hermes under many model + # strings and base URLs, so sniffing for "is this really Google?" is unreliable, and every other + # provider either ignores the field or agrees with the call name. Runs on the per-call copy, so the + # stored trajectory keeps the real tool name for the session DB and the UI — only the wire payload + # changes. A no-op for the native Gemini path, which already resolves the same name. A result whose + # assistant call frame is missing entirely never reaches here — pass 1 above drops it as an orphan — + # so the only results this pass sees are ones whose call name is knowable. call_names: Dict[str, str] = {} for msg in messages: if msg.get("role") == "assistant": @@ -2688,7 +2909,14 @@ def reapply_reasoning_echo_for_provider(agent, api_messages: list) -> int: """Re-pad or strip assistant turns' reasoning_content for the CURRENT provider after a fallback switch: ``api_messages`` is shaped for the primary; require-side providers (DeepSeek/Kimi/MiMo) 400 without the pad, strict ones (Mistral, Cerebras, Groq) 400/422 - with it. Idempotent; returns the number of assistant turns changed.""" + with it. Idempotent; returns the number of assistant turns changed. + + * Switching TO a strict provider that rejects the field (Mistral, Cerebras, Groq, SambaNova, …): + assistant turns built under a reasoning primary carry a ``reasoning_content`` pad (often a single space + ``" "``), and the strict provider rejects it with HTTP 400/422 ("Extra inputs are not permitted"). This + is the exact cross-provider fallback bug from #45655 — a DeepSeek primary pads history with ``" "``, the + request falls back to Mistral, and Mistral 422s on the stale pad. + """ from agent.message_sanitization import reapply_reasoning_echo return reapply_reasoning_echo(api_messages, agent._needs_thinking_reasoning_pad()) @@ -2701,7 +2929,11 @@ def _iter_httpx_pools_with_owner(http_client: Any): ``owner`` is ``None`` for a pool this client owns outright, or the ``_SharedTransport`` view id when the pool is process-shared with other clients (``process_bootstrap.build_keepalive_http_client``). Callers must then touch only the - in-flight requests stamped with that owner.""" + in-flight requests stamped with that owner. + + Walking the default transport alone makes ``force_close_tcp_sockets`` return 0 while a stream is still + mid-recv — the interrupt logs success and the provider keeps burning the slot (#72975). + """ seen_pools: set[int] = set() try: transports = [getattr(http_client, "_transport", None)] diff --git a/agent/anthropic_adapter.py b/agent/anthropic_adapter.py index 99b0956ca7..cd21338fa5 100644 --- a/agent/anthropic_adapter.py +++ b/agent/anthropic_adapter.py @@ -568,6 +568,17 @@ def build_anthropic_kwargs( kwargs["tool_choice"] = _TOOL_CHOICE_MAP.get(tool_choice) or { "type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice } + # Map reasoning_config to Anthropic's thinking parameter. Claude 4.6+ models use adaptive thinking + + # output_config.effort. Older models use manual thinking with budget_tokens. MiniMax Anthropic-compat + # endpoints support thinking (manual mode only, not adaptive). Haiku does NOT support extended thinking + # — skip entirely. Kimi / Moonshot models also use adaptive thinking: their Anthropic-compatible + # endpoints (api.moonshot.cn/anthropic, api.kimi.com/coding) accept ``thinking.type="adaptive"`` + + # ``output_config.effort``, and the replay-validation 400s that originally motivated dropping the + # parameter (#13848) no longer occur. (Kimi on chat_completions enables thinking via extra_body in the + # ChatCompletionsTransport — see #13503.) On 4.7+ the `thinking.display` field defaults to "omitted", + # which silently hides reasoning text that Hermes surfaces in its CLI. We request "summarized" so the + # reasoning blocks stay populated — matching 4.6 behavior and preserving the activity-feed UX during + # long tool runs. if reasoning_config and isinstance(reasoning_config, dict): kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens)) # Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits everywhere callers @@ -638,6 +649,9 @@ def _stream_final_message(stream_fn, api_kwargs, log_prefix, on_stream_event, on try: on_stream_event(event) except TimeoutError: + # The callback is the caller's deadline seam (#99692: the host waiting on this summary has + # already given up). Abandon the stream — the ``with`` closes it — instead of streaming an + # answer nobody will read. raise except Exception: logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True) diff --git a/agent/anthropic_endpoints.py b/agent/anthropic_endpoints.py index 7cb1cd10d9..e261c1d34d 100644 --- a/agent/anthropic_endpoints.py +++ b/agent/anthropic_endpoints.py @@ -76,7 +76,12 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool: """DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking`` blocks to round-trip while the generic third-party path strips them; its blocks are unsigned, so it gets the same strip-signed / keep-unsigned policy as Kimi. Pinned to the ``/anthropic`` - path so the OpenAI-compatible base URL is not misclassified.""" + path so the OpenAI-compatible base URL is not misclassified. + + Per DeepSeek's published compatibility matrix the blocks are unsigned (no Anthropic-proprietary + signature, no ``redacted_thinking`` support), so this endpoint is handled with the same strip-signed / + keep-unsigned policy used for Kimi's ``/coding`` endpoint. See hermes-agent#16748. + """ return base_url_host_matches(base_url or "", "api.deepseek.com") and "/anthropic" in _normalized_lower(base_url) diff --git a/agent/anthropic_message_convert.py b/agent/anthropic_message_convert.py index 124574584b..cbfeb5c7cb 100644 --- a/agent/anthropic_message_convert.py +++ b/agent/anthropic_message_convert.py @@ -109,6 +109,7 @@ def normalize_model_name(model: str, preserve_dots: bool = False) -> str: if model.lower().startswith("anthropic/"): model = model[len("anthropic/"):] if not preserve_dots and not _is_bedrock_model_id(model) and model.lower().startswith(("claude-", "anthropic/")): + # Only convert dots to hyphens for Anthropic/Claude models. See issue #17171. model = model.replace(".", "-") return model @@ -150,6 +151,8 @@ def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]: for t in tools or []: fn = t.get("function", {}) name = fn.get("name", "") + # Defensive dedup: Anthropic rejects requests with duplicate tool names. Upstream injection paths + # already dedup, but this guard converts a hard API failure into a warning. See: #18478 if name and name in seen_names: logger.warning("convert_tools_to_anthropic: duplicate tool name '%s' — dropping second occurrence", name) continue @@ -370,6 +373,12 @@ def _replay_ordered_blocks(m: Dict[str, Any], ordered_blocks: List[Any]) -> Opti def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: """Assistant message -> Anthropic content blocks (thinking, text, tool_use, Kimi/DeepSeek reasoning_content injection).""" + # apply_anthropic_cache_control marks an assistant turn with non-empty text by writing cache_control + # INTO ``content`` (see _apply_cache_marker's list branch), not at the top level. This branch rebuilds + # the message from ordered_blocks and never reads ``content``, so that marker would be dropped -- and + # because _can_carry_marker already counted this message as a carrier, the breakpoint is burned rather + # than relocated. #56195 covered the complementary shape (blank content -> top-level marker); this is + # the interleaved thinking + preamble-text + tool_use shape. content = m.get("content", "") ordered_blocks = m.get("anthropic_content_blocks") if isinstance(ordered_blocks, list) and ordered_blocks: @@ -395,6 +404,8 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: # (injected as a fallback upstream). Prepend, since thinking must precede text/tool_use. Skip # when reasoning_details already supplied (signed) thinking blocks: a duplicate unsigned one # would be downgraded to a spurious text block on the last assistant message. + # See hermes-agent#13848. Accept empty string "" — _copy_reasoning_content_for_api() injects "" as a + # tier-3 fallback for Kimi tool-call messages that had no reasoning. reasoning_content = m.get("reasoning_content") if isinstance(reasoning_content, str) and not _has_block_type(blocks, _THINKING_TYPES): blocks.insert(0, {"type": "thinking", "thinking": reasoning_content}) @@ -598,7 +609,13 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None: """Anthropic requires messages[0].role == user; prepend a placeholder turn otherwise. A second auto-compaction can leave a role=assistant summary first, which the API rejects (often masked as a misleading tool_use/tool_result 400). The filler must be non-whitespace text or it trades - that 400 for the blank-block one.""" + that 400 for the blank-block one. + + The inserted text block must be non-whitespace: Anthropic separately rejects any text content block + whose text is empty or whitespace-only ("text content blocks must contain non-whitespace text"), so a + single space here traded the "leading assistant turn" 400 for that one (#69512 class). Uses the same + placeholder as every other synthesized filler block in this module for consistency. + """ if result and result[0].get("role") != "user": result.insert(0, {"role": "user", "content": [_text_block(_EMPTY_TEXT_PLACEHOLDER)]}) diff --git a/agent/api_error_summary.py b/agent/api_error_summary.py index ff57e44d31..3e907dbaf3 100644 --- a/agent/api_error_summary.py +++ b/agent/api_error_summary.py @@ -57,6 +57,15 @@ class ApiErrorSummaryMixin: the pool. xAI returns the same permission-denied text for BOTH cases; a ``[WKE=unauthenticated:...]`` suffix (or "access token could not be validated") means stale token → return False so the refresh path runs. + + Disambiguator for xAI (#29344): the same ``code`` text ("The caller does not have permission to + execute the specified operation") is returned for BOTH an unsubscribed account AND a stale OAuth + access token. xAI ships an explicit signal in the ``error`` field that tells the two apart: a + ``[WKE=unauthenticated:...]`` suffix (and/or the ``OAuth2 access token could not be validated`` + phrasing) means the credentials failed validation — that's recoverable by refreshing the token, NOT + by surfacing an entitlement message. When either signal is present we return False eagerly so the + credential-pool refresh path runs, letting long-running TUI sessions recover from stale tokens + without an exit/reopen cycle. """ if status_code not in {401, 403, None}: return False @@ -162,6 +171,9 @@ class ApiErrorSummaryMixin: # SDK may leave body empty while httpx has the payload. Redact: the body is attacker-influenced # and may echo Authorization / x-api-key / request JSON. + # Redact before returning: the raw provider/proxy error body is attacker-influenced and may echo + # Authorization / x-api-key / request JSON, which would otherwise leak into final_response + logs + # (this path widens exposure vs the old empty-body "HTTP 400" string). See #36109. response = getattr(error, "response", None) if response is not None: try: diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 1a2afc8262..9796ac8d8c 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -177,6 +177,12 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any: _apply_required_codex_headers(kwargs, access_token=api_key, base_url=base_url) # Hermes owns aux retry/fallback policy; the SDK default (max_retries=2) would triple # wall time on a hung endpoint before Hermes sees one failure. + # Hermes owns auxiliary retry + provider/model fallback policy (the same-provider transient retry in + # call_llm plus the except-chain fallback). The OpenAI SDK's own default (max_retries=2 → up to 3 + # attempts) silently multiplies the effective wall time of every aux call by 3× on a slow/hung endpoint, + # so a 120s timeout can stall ~360s before Hermes sees a single failure (issue #54465). Disable + # SDK-internal retries by default and let Hermes control the budget; explicit callers can still override + # via kwargs. kwargs.setdefault("max_retries", 0) return OpenAI(api_key=api_key, base_url=base_url, **kwargs) @@ -184,6 +190,13 @@ def _create_openai_client(*, api_key: str, base_url: str, **kwargs: Any) -> Any: # Interrupt protection for atomic aux tasks: a compression summary killed by an ordinary # gateway interrupt degrades to a static marker, so a thread-local flag marks such calls # protected. Explicit host cancel (Ctrl+C, /stop) still overrides it, timeouts still fire. +# ── Interrupt protection for atomic auxiliary tasks ────────────────────── Some auxiliary tasks must NOT be +# aborted mid-flight by a gateway interrupt (e.g. an incoming user message while the agent is busy). Context +# compression is the prime case: if the summary LLM call is interrupted part-way, compression falls back to +# a static "summary unavailable" marker and the real handoff is lost (#23975). A thread-local flag lets such +# a task mark its in-flight LLM call as interrupt-protected; the Codex Responses stream's cancellation check +# honors it. TIMEOUTS still fire (a hung call must die), and all OTHER aux tasks (vision, web_extract, +# title_generation, …) remain freely interruptible. _aux_interrupt_protection = threading.local() @@ -279,6 +292,12 @@ _aux_provider_response = threading.local() # Absolute monotonic deadline of the waiting HOST. The stream's own ceiling # (_aux_stream_total_ceiling, >= the host's and started later) would otherwise leave an # orphaned stream still billing after every host-ceiling timeout. +# Absolute wall-clock deadline (time.monotonic) of the HOST waiting for this auxiliary call, when it has one +# (#99692). Liveness alone is not enough: a host also stops waiting at its own total ceiling, and the +# streamed consumer below bounds itself only by _aux_stream_total_ceiling() — a budget derived from the aux +# request timeout, which is >= the host ceiling for every configured value AND starts counting later. So the +# stream that outlives its abandoned host is not an edge case; it is the guaranteed outcome of every +# total-ceiling timeout. _aux_stream_deadline = threading.local() @@ -411,6 +430,12 @@ def aux_stream_deadline(deadline: Optional[float]): ``None`` is a passthrough; re-entrant-safe. Host->worker return leg of the progress hook: without it the isolated provider daemon streams to its own ceiling after the host stopped waiting, billing a summary the commit fence refuses. + + ``8207862212`` releases the compression OWNER when the fence is cancelled, but the isolated provider + daemon (:func:`_run_protected_sync_provider_call`) that holds the socket keeps streaming to its own + ``_aux_stream_total_ceiling`` budget — >= the host's ceiling by construction — billing an abandoned + summary the commit fence is already guaranteed to refuse, and stacking one fresh orphan per turn on a + session that compression never managed to shrink. See #99692. """ previous = getattr(_aux_stream_deadline, "value", None) _aux_stream_deadline.value = deadline if isinstance(deadline, (int, float)) else previous @@ -446,6 +471,9 @@ def _run_protected_sync_provider_call(callback: Callable[[dict[str, Any]], Any], dispatch_hook = getattr(_aux_dispatch, "hook", None) provider_response_hook = getattr(_aux_provider_response, "hook", None) host_deadline = _current_aux_stream_deadline() + # #99692: the stream is consumed on the daemon below, and thread-locals do not cross that boundary — an + # owner-thread-only deadline would leave the fix inert on exactly the path large-session compression + # takes (protected call + hard-cancel source installed). provider_context = contextvars.copy_context() done = threading.Event() outcome: dict[str, Any] = {} @@ -1104,6 +1132,18 @@ class _CodexStreamGuard: self.total_timeout = total_timeout self._start = time.monotonic() self.no_progress_timeout = _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS + # Progress-aware stream deadlines (supersedes the old single absolute kill at ``total_timeout``). + # Three regimes: 1. First token: the stream must produce its first substantive payload within + # ``no_progress_timeout`` (60s default) or we fail fast and let the caller's normal retry/fallback + # chain run — a dead (or keepalive-only zombie) Codex stream no longer holds the full 300s + # compression budget before falling back (masoria report, Aug 2026: 3 stacked 300s waits -> 20+ min + # stuck on "Summarizing"). 2. Streaming: every substantive event re-arms the deadline by + # ``no_progress_timeout`` — a live stream is never killed by an absolute total, so a long reasoning + # summary that is actually producing tokens completes instead of timing out at 300s and falling back + # (#54915's original complaint, fixed properly). Keepalive/lifecycle frames do NOT re-arm, mirroring + # the commit-fence progress gating (#96707). 3. Hard ceiling: an absolute backstop from + # ``_aux_stream_total_ceiling`` (max(600s, 4x configured timeout) — the same bound the streamed + # chat.completions path uses) so a pathological one-token-per-59s drip still terminates. if total_timeout is not None: self.no_progress_timeout = min(self.no_progress_timeout, float(total_timeout)) self.hard_deadline = self._start + _aux_stream_total_ceiling(total_timeout) @@ -1193,6 +1233,9 @@ class _CodexStreamGuard: # FD-safe — ``close()`` releases the raw TLS fd while the owner's OpenSSL BIO still # caches it, the kernel recycles it (e.g. into a SQLite handle), and the owner's TLS # flush corrupts that file. The owner does the real close in its ``finally``. + # This callback has two callers — ``_check_cancelled`` on the owning thread, and the daemon watchdog + # ``threading.Timer``, which is a stranger thread. The owning thread performs the real close in the + # ``finally`` below, which is where the FD release belongs. See #70773. self.timeout_release_pending.set() if threading.get_ident() == self._owner_tid: _close_quietly(self._client, "client close during timeout failed") @@ -1212,6 +1255,9 @@ class _CodexStreamGuard: # The aux client cache wraps this same client; drop the entry so the next aux call # doesn't reuse the dead transport and fail fast. try: + # After we close the httpx transport above, the cache must drop that entry — otherwise the next + # auxiliary call (compression retry, memory flush, etc.) reuses the dead client and fails fast + # with a connection error. See issue #23432. _evict_cached_client_instance(self._client) except Exception: logger.debug("Codex auxiliary: cache eviction on timeout failed", exc_info=True) @@ -1227,6 +1273,8 @@ class _CodexStreamGuard: # interrupt (degraded fallback marker); explicit host cancel has its own exception. if _aux_interrupt_cancel_requested(): raise AuxiliaryExplicitCancellation() + # Explicit host cancellation has its own frozen exception; timeouts above still fire and other + # aux tasks remain interruptible. See #23975. if is_interrupted() and not _aux_interrupt_protected(): raise InterruptedError("Codex auxiliary Responses stream interrupted") except InterruptedError: @@ -1260,6 +1308,8 @@ class _CodexStreamGuard: # TTFP telemetry records every frame, but forward progress (compression commit fence, # no-progress window) counts only substantive payloads — keepalives must not re-arm, # so a zombie stream dies at the same window as a dead connection. + # #93650: keep bulk wire-format payload out of the SDK's GIL-holding request transform on auxiliary + # calls too. if _codex_event_has_content(_event): self.record_progress() self.saw_content.set() @@ -1289,6 +1339,15 @@ class _CodexCompletionsAdapter: def _build_responses_kwargs(self, kwargs: Dict[str, Any]) -> Tuple[Dict[str, Any], str, Any]: """chat.completions kwargs → Responses API kwargs, ``(resp_kwargs, model, timeout)``; mirrors codex.py::build_kwargs.""" from utils import base_url_host_matches + # Separate system/instructions from replayable conversation messages, then route the rest through + # the SINGLE shared chat->Responses converter used by the main agent transport + # (agent/transports/codex.py). Maintaining a private conversion loop here let chat-style messages + # with role="tool" leak straight into Responses input[] — which the Responses API rejects with + # "Invalid value: 'tool'. Supported values are: 'assistant', 'system', 'developer', and 'user'." + # (issue #5709, hit hard by flush_memories() / compression replaying real session history that + # includes assistant tool_calls + role="tool" results). The shared converter encodes assistant tool + # calls as `function_call` items and tool results as `function_call_output` items with a valid + # call_id, so every Responses path normalizes tool history identically and cannot drift. from agent.codex_responses_adapter import _chat_messages_to_responses_input model = kwargs.get("model", self._model) host = str(getattr(self._client, "base_url", "") or "") @@ -1309,6 +1368,9 @@ class _CodexCompletionsAdapter: # Copilot binds replayed codex_message_items ids to a backend connection that doesn't # survive credential rotation (401 on replay) — same guard as build_kwargs. Aux calls # never send ``context_management`` (main-turn feature): no compaction checkpoint. + # Auxiliary calls (context compression, flush_memories, MoA aggregation) go through this adapter + # instead of agent/transports/codex.py's build_kwargs, so they need the same guard applied + # independently. See #32716. input_items = _chat_messages_to_responses_input( replay_messages, is_github_responses=is_copilot, native_compaction_eligible=False ) @@ -1374,6 +1436,9 @@ class _CodexCompletionsAdapter: # conversation (rotation-stable logical scope, else the physical session id). Skip the # key where the main transport does: xAI takes it in extra_body, GitHub opts out. try: + # Reuse the Responses transport's single authoritative hash algorithm and session-scope + # normalization so equivalent static prefixes route to the same cache bucket across modes, + # without concentrating unrelated sessions into one shared bucket (see #78941). from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key from agent.transports.codex import _default_prompt_cache_retention_for_request if not (is_xai or is_github) and "prompt_cache_key" not in resp_kwargs: @@ -1471,6 +1536,11 @@ class _AsyncAuxiliaryClientBase: self.api_key = sync_wrapper.api_key self.base_url = sync_wrapper.base_url if hasattr(sync_wrapper, "_real_client"): + # Mirror the sync wrapper's _real_client so cache eviction by leaf OpenAI client (e.g. + # _close_client_on_timeout in #23482) drops this async entry too. Without this, sync and async + # cache entries diverge on poisoning: the sync entry is evicted but the async entry keeps + # reusing the closed transport, failing every subsequent async aux call with 'Connection error' + # until the gateway restarts. self._real_client = sync_wrapper._real_client @@ -1587,6 +1657,9 @@ class _AnthropicCompletionsAdapter: # response_format: top-level gets the same translation as the extra_body form; when both # are present the extra_body form wins. Passthrough excludes ``reasoning``/``response_format`` # (already TRANSLATED to native fields — raw would 400 on strict gateways) and ``_`` Hermes plumbing. + # The adapter builds the Messages body from a fixed allow-list of kwargs, so before this an + # unrecognized top-level kwarg was dropped on the floor: the request succeeded but the schema + # contract silently became prompt compliance (#85626 review, point 2). top_level_response_format = kwargs.get("response_format") if top_level_response_format is not None: _translate_anthropic_response_format(anthropic_kwargs, top_level_response_format) @@ -2471,6 +2544,10 @@ def set_runtime_main( Context-local so concurrent gateway sessions don't clobber each other; legacy mirrors are updated for old readers. ``cache_scope`` is the rotation-stable logical cache scope, preferred over ``session_id`` for prompt_cache_key derivation. + + ``cache_scope`` is the rotation-stable logical cache scope (compression- lineage root — + agent/prompt_cache_scope.py) resolved once per turn by turn_context; auxiliary Responses calls prefer it + over ``session_id`` for prompt_cache_key derivation (#79017). """ runtime = { "provider": (provider or "").strip().lower(), @@ -2538,6 +2615,8 @@ def _resolve_custom_runtime() -> Tuple[Optional[str], Optional[str], Optional[st if base_url_host_matches(custom_base, "openrouter.ai"): return None, None, None # requested='custom' falls back to OpenRouter when unconfigured. # Local servers (Ollama, vLLM, ...) ignore auth but the SDK needs a non-empty key. + # Use a placeholder key — the OpenAI SDK requires a non-empty string but local servers ignore the + # Authorization header. Same fix as cli.py _ensure_runtime_credentials() (PR #2556). if not isinstance(custom_key, str) or not custom_key.strip(): custom_key = "no-key-required" if not isinstance(custom_mode, str) or not custom_mode.strip(): @@ -2931,6 +3010,7 @@ def _is_rate_limit_error(exc: Exception) -> bool: OpenAI's RateLimitError may omit .status_code — matched by class name. A generic 429 without billing keywords counts as a rate limit. """ + # (PR #8023 pattern) if type(exc).__name__ == "RateLimitError": return True if getattr(exc, "status_code", None) != 429: @@ -3116,6 +3196,8 @@ def _is_invalid_aux_response_error(exc: Exception) -> bool: # Tasks on a user-visible critical path (compression blocks resuming an oversized session; vision # stalls the serialised turn queue). A same-provider retry after a full-budget timeout costs another # whole ``timeout`` window, so they skip straight to fallback; fast blips still retry. +# Fast blips (a streaming-close or a 5xx) still retry, since those are cheap. See issue #54465 for the +# compression case. _TIMEOUT_NO_RETRY_TASKS = frozenset({"compression", "vision"}) @@ -3294,6 +3376,10 @@ def _prepare_same_provider_retry( ) # Preserve per-request attribution headers (e.g. Copilot ``x-initiator``) so the retry keeps capability gating. if extra_headers: + # Copilot's ``x-initiator: user``) across the rebuilt-client retry — dropping them here would let a + # recovery retry silently lose capability gating (#60293). + # Preserve per-request attribution headers across the rebuilt-client retry — see the sync variant + # above (#60293). retry_kwargs["extra_headers"] = dict(extra_headers) if _is_anthropic_compat_endpoint(resolved_provider, retry_base): retry_kwargs["messages"] = _convert_openai_images_to_anthropic(retry_kwargs["messages"]) @@ -3432,7 +3518,13 @@ def _coerce_positive_timeout(raw: Any) -> Optional[float]: def _fallback_entry_timeout(task: Optional[str], fb_label: str) -> Optional[float]: """Per-entry ``timeout`` for a configured fallback candidate, or None (keep the task-level - timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks.""" + timeout). Inheriting the primary's deadline used to kill healthy-but-slower fallbacks. + + A fallback candidate previously inherited the exact timeout the primary provider was called with. When + that deadline was tuned for the primary (or the primary simply consumed its whole budget before failing + over), the fallback aborted on the same clock even when independently healthy — a 163k-token compression + that needs ~90s on the fallback died at the primary's 30s deadline every turn (#62452). + """ entry = _fallback_chain_entry(task, fb_label) return _coerce_positive_timeout(entry.get("timeout") if entry else None) @@ -3598,7 +3690,12 @@ def _call_fallback_candidate_sync( ) -> Optional[Any]: """Call one fallback candidate with stale-credential recovery: on an auth error refresh its credentials and retry once with a rebuilt client; if that also auth-fails, quarantine the - provider and return None so the caller moves on. Non-auth errors raise.""" + provider and return None so the caller moves on. Non-auth errors raise. + + ``effective_timeout`` is the task-level deadline; a configured-chain candidate with its own ``timeout`` + entry gets that instead, so a fallback tuned differently from the primary is allowed its own budget + (#62452). + """ destination, fb_kwargs, rebuild = _plan_fallback_candidate( fb_client, fb_model, fb_label, task=task, effective_timeout=effective_timeout, apply_fast_lane=True, messages=messages, tools=tools, temperature=temperature, @@ -3751,6 +3848,17 @@ def _try_main_agent_model_fallback( # too-small aux models; runtime chains must too, or compression stops at a reachable-but-too-small # candidate. ``None`` (unknown) passes through. +# ── Context-window screening for runtime fallback chains (issue #52392) ── When the runtime auxiliary +# fallback chain selects a candidate that is reachable but has a context window smaller than the compression +# task requires, the call errors out instead of continuing to the next, viable candidate. The startup +# feasibility check in ``agent.conversation_compression.check_compression_model_feasibility`` already +# filters too-small auxiliary models at startup, but the runtime fallback chain +# (``_try_configured_fallback_chain`` and ``_try_main_fallback_chain``) does not apply the same filter, so +# compression can stop at the first alive door even if the room behind it is too small. The helpers below +# screen each candidate by its effective context window before it is returned. ``None`` results from +# ``get_model_context_length`` are passed through (we cannot prove a model is too small, so we do not block +# it). This preserves the existing fallback surface for unrecognised/custom models while closing the gap on +# the well-known ones. def _task_minimum_context_length(task: Optional[str]) -> Optional[int]: """Minimum context length for an auxiliary task; None = no floor (only ``compression`` has one).""" return MINIMUM_CONTEXT_LENGTH if task == "compression" else None @@ -3991,6 +4099,7 @@ def _try_main_provider_route( explicit_base_url = None elif runtime_base_url: # Config-less named custom provider (live runtime only): anonymous custom arm + runtime key. + # See #34777. resolved_provider = "custom" explicit_api_key = runtime_api_key or None elif runtime_api_key: @@ -4121,6 +4230,7 @@ def _to_async_client(sync_client, model: str, is_vision: bool = False): _apply_required_codex_headers(async_kwargs, access_token=sync_client.api_key, base_url=sync_base_url) async_kwargs = {**_openai_http_client_kwargs(sync_base_url, async_mode=True), **async_kwargs} # Hermes owns the auxiliary retry/timeout budget; disable SDK-internal retries. + # See #54465. async_kwargs.setdefault("max_retries", 0) return AsyncOpenAI(**async_kwargs), model @@ -4187,6 +4297,7 @@ def _build_bedrock_client(provider: str, model: Optional[str], *, raw_codex: boo return None, None # Region must match the main runtime's resolution (bedrock.region in config first, then # env/profile) so aux calls never leave the primary runtime's configured region. + # See #53880, #65076. region = resolve_bedrock_runtime_region() default_model = "anthropic.claude-haiku-4-5-20251001-v1:0" final_model = _normalize_resolved_model(model or default_model, provider) or default_model @@ -4406,6 +4517,8 @@ def _resolve_custom_branch(req: _ResolveRequest) -> _ResolveResult: elif main_runtime: # Reuse main_runtime's concrete base_url + api_key for a named custom provider; # re-resolving from bare "custom" loses the name and lands on the wrong provider. + # Re-resolution loses the provider name and falls back to OpenRouter or a wrong API-key provider — + # the main agent already solved this, we just need to reuse its answer. (#45472) _main_base = str(main_runtime.get("base_url") or "").strip().rstrip("/") _main_key = str(main_runtime.get("api_key") or "").strip() if _main_base and _main_key: @@ -4487,6 +4600,7 @@ def _resolve_named_custom_branch(req: _ResolveRequest) -> Optional[_ResolveResul provider, final_model, entry_api_mode or "chat_completions") # anthropic_messages: route via AnthropicAuxiliaryClient (mirrors _try_custom_endpoint); # the Anthropic SDK sees the original (un-rewritten) URL. + # Mirrors the anonymous-custom branch in _try_custom_endpoint(). See #15033. if entry_api_mode == "anthropic_messages": try: from agent.anthropic_adapter import build_anthropic_client @@ -4710,6 +4824,29 @@ def resolve_provider_client( # Excluded: ``auto`` (a stale main slug could pair with any picked provider) and Nous + vision (the # Portal's tier-aware vision recommendation must win over a text-only model). if not model and provider != "auto" and not (provider == "nous" and is_vision): + # ``auto`` is intentionally excluded: `_resolve_auto(main_runtime=...)` returns the model paired + # with the provider it actually selected. Pre-filling an auto call from `_read_main_model()` can + # leak a stale process-global runtime into a different provider (for example Claude model slug on + # Codex OAuth) and override that correctly resolved model. 1. ``model`` argument (caller knew what + # they wanted) 2. Provider's catalog default — cheap/fast model the provider registered via + # ``ProviderProfile.default_aux_model`` or the legacy ``_API_KEY_PROVIDER_AUX_MODELS_FALLBACK`` + # dict. 3. User's main model from ``model.model`` in config.yaml. This is the load-bearing step for + # OAuth providers: an xai-oauth user with grok-4.3 configured gets grok-4.3 for title generation + # instead of silently dropping to whatever Step-2 fallback (#31845). When the main provider is MoA, + # ``_read_main_model_for_aux()`` substitutes the preset's aggregator model — the preset NAME is + # never a valid wire model id, so unset aux models default to the preset's acting model instead. + # Each provider branch below sees a non-empty ``model`` whenever the user has *anything* configured + # — no provider-specific empty-model guards needed. When the user has NOTHING configured (fresh + # install, main_model also empty), the branches still hit their own missing-credentials returns and + # ``_resolve_auto`` falls through to the Step-2 chain as before. Do NOT pre-fill a blank ``auto`` + # request from the config/main default here. Claude model sent to Codex after the main lane fell + # back to gpt-5.5). Let _resolve_auto() return the actual current runtime model when the caller did + # not explicitly request one. (# compression-current-model) Nous + vision is the one carve-out: the + # branch below resolves its model from the Portal's tier-aware vision recommendation + # (``_try_nous(vision= True)``), and ``final_model = model or default`` means anything pre-filled + # here wins over that. The main chat model is routinely text-only (e.g. a ``:free`` chat SKU), so + # pre-filling it sends the image to a model that cannot accept one and the Portal 404s. Leave + # ``model`` unset and let the Portal slot through; only an explicit caller model may override it. model = _get_aux_model_for_provider(provider) or _read_main_model_for_aux() or model req = _ResolveRequest( provider, original_provider, model, async_mode, raw_codex, @@ -4974,6 +5111,8 @@ def auxiliary_max_tokens_param(value: int, *, model: Optional[str] = None) -> di # Client cache: (provider, async_mode, base_url, api_key, api_mode, runtime_key) -> (client, default_model, loop) # Loop identity is NOT part of the key: stale-loop entries are replaced in place on async hits, # bounding growth to one entry per provider config (avoids fd accumulation in gateways). +# This bounds cache growth to one entry per unique provider config rather than one per (config × +# event-loop), which previously caused unbounded fd accumulation in long-running gateway processes (#10200). _client_cache: Dict[tuple, tuple] = {} _client_cache_lock = threading.Lock() _CLIENT_CACHE_MAX_SIZE = 64 # safety belt — evict oldest when exceeded @@ -5056,6 +5195,12 @@ def _refresh_nous_auxiliary_client( to ``_get_cached_client`` when the stale client was acquired — so the fresh client overwrites the exact entry the stale one is served from. Keying on the resolved model or an empty task would leave the expired client immortal and every auxiliary call 401ing forever. + + See #56889. + For ``provider == "auto"`` the task participates in the cache key (task-specific fallback policy), so it + MUST be carried into the key here for the same reason as ``lookup_model``; otherwise an auto-provider + client refreshed on a 401 lands under the ``task=""`` key while the stale entry survives under the + task-scoped key (#58894). """ runtime = _resolve_nous_runtime_api(force_refresh=True, stale_access_token=api_key) if runtime is None: @@ -5212,6 +5357,10 @@ def _get_cached_client( Async clients bind to the loop they were created on, so every async hit validates the cached loop is the current, open loop; stale entries are replaced in place (bounded, no cross-loop reuse). + + This keeps cache size bounded to one entry per unique provider config, preventing the fd-exhaustion that + previously occurred in long-running gateways where recycled worker threads created unbounded entries + (#10200). """ current_loop = _current_event_loop() if async_mode else None runtime = _normalize_main_runtime(main_runtime) @@ -5265,6 +5414,14 @@ def _get_cached_client( _AUX_DIRECT_API_BASE_URLS: Dict[str, str] = {"openai": "https://api.openai.com/v1"} +# MoA virtual provider: an *explicit* `provider: moa` override (either the caller-passed `provider` arg or +# `auxiliary..provider` in config.yaml) reaches this function directly — it never goes through +# _resolve_auto(), which only unwraps the *implicit* "main provider is moa" case (#53827). Left as-is, "moa" +# is returned verbatim and resolve_provider_client() looks it up in PROVIDER_REGISTRY (which has no "moa" +# entry — it's not a real HTTP provider), falls to the unknown-provider dead end, and call_llm surfaces a +# nonsensical "MOA_API_KEY environment variable" error for a provider that was never meant to be reached +# over the wire. Auxiliary tasks don't need the reference fan-out — resolve to the preset's aggregator slot +# instead, exactly like the implicit path does (shared helper: _resolve_moa_aggregator). def _unwrap_moa_provider(prov: str, mdl: Optional[str]) -> Tuple[str, Optional[str]]: """Resolve an *explicit* ``provider: moa`` to its preset's aggregator slot (_resolve_auto() only unwraps the implicit case; "moa" isn't in PROVIDER_REGISTRY and would dead-end).""" @@ -5350,6 +5507,7 @@ def _resolve_task_provider_model( # An explicit provider without base_url adopts the task's configured endpoint (same or # unnamed provider) so the early return below carries it. Explicit "auto" is excluded — it # must keep flowing through auto-resolution. + # See #58515. if provider and provider != "auto" and not base_url and cfg_base_url and cfg_provider in (None, provider): base_url = cfg_base_url if not api_key: @@ -5375,6 +5533,11 @@ _DEFAULT_AUX_TIMEOUT = 30.0 # Reasoning compression models can exceed the default 120 s config timeout, falling back to the # deterministic marker. Bounded *floor* for config-derived compression timeouts only; never # overrides an explicit per-call timeout. +# Compression summarises large conversation histories; a reasoning auxiliary model (e.g. Codex / GPT-5.5) +# can legitimately take longer than the default ``auxiliary.compression.timeout`` (120 s), causing the +# stream to time out and the compressor to fall back to the deterministic context marker (#54915). A floor +# is harmless for fast compression models (they finish before the deadline) and is a minimum, so a higher +# config value is kept unchanged. _COMPRESSION_TIMEOUT_FLOOR_SECONDS = 300.0 @@ -5531,6 +5694,9 @@ def _get_task_extra_body(task: str) -> Dict[str, Any]: # Per-task concurrency limiting: many sessions can spawn unbounded background aux calls, each # retrying across the fallback chain during incidents. +# During provider incidents each call also retries / fans out across the fallback chain, multiplying request +# volume on already-degraded endpoints. A per-task semaphore caps in-flight calls so retry amplification +# stays bounded. See #23324. _aux_sync_semaphores: Dict[str, Tuple[int, threading.BoundedSemaphore]] = {} _aux_async_semaphores: Dict[Tuple[str, int], Tuple[int, Any]] = {} _aux_sem_lock = threading.Lock() @@ -5833,6 +5999,11 @@ def _validate_llm_response( Also the single aux-usage accounting chokepoint: every successful non-streaming response passes here exactly once; *provider*/*base_url* are optional hints. + + See #7264. + Recording is best-effort and never affects validation. *provider*/*base_url* are optional accounting + hints — fallback-path calls omit them and the row keeps the model (read from the response itself) with + an empty route. See #23270. """ if response is None: raise RuntimeError(f"Auxiliary {task or 'call'}: LLM returned None response") @@ -6007,6 +6178,7 @@ _AFFORDABLE_TOKENS_RE = re.compile(r"can only afford\s+([0-9][0-9,]*)", re.IGNOR # Below the floor the affordable budget can't fit a useful aux output — treat as exhaustion; # the margin keeps provider-side token-count rounding from 402-ing the retry. _AFFORDABLE_RETRY_FLOOR_TOKENS = 512 +# See #49785. _AFFORDABLE_RETRY_MARGIN_TOKENS = 64 @@ -6067,7 +6239,18 @@ def _create_with_progress_once( """create() that streams (and re-aggregates, ticking the hook per substantive chunk) when a progress hook is active or the provider is stream-only; plain ``create(**kwargs)`` otherwise or when the adapter streams internally. Streaming rejections fall back to a plain call — - except under ``force_stream``.""" + except under ``force_stream``. + + Behavior is byte-for-byte identical to a plain ``create(**kwargs)`` when neither trigger applies (every + existing caller/task) or when the client's wire adapter streams internally. With a hook + a + chunk-capable client, the request is sent with ``stream=True`` and aggregated, ticking the hook only for + substantive chunks. The configured ``timeout`` acts per stream read (idle) rather than as a total + budget, and outer liveness watchdogs see tokens moving. ``force_stream=True`` (stream-only providers + such as Tencent Copilot — credit @kudi88, PR #60686) takes the same streamed path even without a hook. + Providers that reject the streamed request fall back to the plain non-streaming call — except under + ``force_stream``, where a stream-only provider rejects the plain call by definition, so the original + error is surfaced to the normal recovery chains instead. + """ _notify_aux_dispatch() _notify_aux_progress() # Preserve the watchdog's historical dispatch tick. if (not _aux_progress_active() and not force_stream) or _client_streams_internally(client): @@ -6141,6 +6324,9 @@ class _ChatStreamAccumulator: self._total_ceiling = total_ceiling # Absolute instant the waiting host gives up; checked alongside (not instead of) the # ceiling, and unaffected by pre-construction dispatch/TTFT. + # Checked as well as (not instead of) the ceiling above: the ceiling still bounds callers with no + # host deadline, and the host deadline is absolute, so it is unaffected by however long dispatch and + # TTFT took before this accumulator was constructed. See #99692. self._host_deadline = host_deadline self.content_parts: List[str] = [] self.reasoning_parts: List[str] = [] @@ -6626,6 +6812,12 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute): response) bypass the explicit-provider gate — the provider cannot serve this request regardless of user intent. Auth errors only fall back in auto mode.""" task, tag, resolved_provider = route.task, route.tag, route.resolved_provider + # Respect explicit provider choice for transient errors (auth, request validation, etc.) but allow + # fallback when the provider clearly cannot serve the request due to capacity: payment/quota exhaustion + # and connection failures are capacity problems, not request constraints. See #26803: daily token quota + # (429 + "too many tokens per day") must fall back just like a 402 credit error. + # Rate limits are included: after retries are exhausted, a 429 means the provider is at capacity. See + # #52228. See #26803: daily token quota must fall back like a 402 credit error. is_auto = resolved_provider in {"auto", "", None} reason = next((label for predicate, label in _FALLBACK_REASONS if predicate(first_err)), None) is_capacity_error = any( @@ -6669,6 +6861,9 @@ def _ladder_provider_fallback(first_err: Exception, route: _LadderRoute): break # All fallback layers exhausted — one user-visible warning, then re-raise. logger.warning("Auxiliary %s%s: %s on %s and all fallbacks exhausted " + # All fallback layers exhausted — emit a single user-visible warning so the operator + # knows aux task is about to fail. (#26882) The error itself is re-raised below. + # (#26882) "(fallback_chain + main agent model). Raising original error.", task or "call", tag, reason, resolved_provider) return None @@ -6705,6 +6900,10 @@ def _aux_recovery_ladder( return resp # Connection/timeout errors poison the cached client (closed transport, half-read # stream); evict so the next aux call rebuilds a fresh one. + # Drop it from the cache regardless of whether we found a fallback above so the next auxiliary call + # rebuilds a fresh client instead of reusing the dead one. See issue #23432. + # Mirror the sync path: drop poisoned clients on connection/timeout so the next aux call rebuilds. See + # issue #23432. if _is_connection_error(first_err): try: _evict_cached_client_instance(client) @@ -6924,6 +7123,17 @@ def _call_llm_impl( return _relay_sync_stream(client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode) def _primary(**validate_kw: Any) -> Any: + # Retry on the same provider for a transient transport blip (connection reset / streaming-close / + # incomplete chunked read / 5xx / 408) before the except-chain below escalates to provider/model + # fallback. A dropped connection shouldn't abandon an otherwise-healthy provider — this especially + # matters for pinned auxiliary calls like MoA reference advisors, where "fallback to another + # provider" is not a meaningful recovery (the advisor is a specific model), so a transient blip that + # isn't retried simply loses that advisor for the turn (root of the run2 double-advisor "Connection + # error" collapse — a genuine upstream blip hitting both parallel advisors at once). Attempts are + # bounded and use exponential backoff. Count is configurable via auxiliary.transient_retries + # (default 2 retries → 3 total attempts); a second/third failure or any non-transient error falls + # through to ``first_err`` and the existing fallback handling unchanged. Unified home for the + # transient retry every auxiliary task shares. (PR #16587) return _validate_llm_response( _relay_sync_completion( client, kwargs, provider=request_provider, api_mode=req.resolved_api_mode, @@ -7087,6 +7297,7 @@ async def _async_call_llm_impl( client, kwargs, request_provider = req.client, req.kwargs, req.request_provider try: # Retry ONCE on the same provider for a transient blip before fallback (see call_llm()). + # (PR #16587) _force_stream_async = ( _provider_requires_stream(request_provider, req.base_info or req.resolved_base_url) and not isinstance(client, ( diff --git a/agent/backend_identity.py b/agent/backend_identity.py index 593b2d0b62..5ba5556068 100644 --- a/agent/backend_identity.py +++ b/agent/backend_identity.py @@ -78,6 +78,9 @@ def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool: (stranded failover). Same label = same configured credential; custom entries can each carry their own api_key, so a shared URL alone is only a weak signal when a label is missing.""" if a.provider and b.provider: + # Different labels = different credential config (first-class registry providers explicitly so — + # #70893; custom entries can each carry their own api_key, so sameness is unprovable and we must not + # skip). return a.provider == b.provider return bool(a.base_url and a.base_url == b.base_url) @@ -100,6 +103,8 @@ def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool: if not (a.provider and b.provider and a.provider == b.provider): return bool( a.base_url + # Same-host different-label shims: same URL + same model IS the same deployment even when the + # alias labels differ (#22548) — unless both labels are first-class registry providers (#70893). and a.base_url == b.base_url and a.model and a.model == b.model diff --git a/agent/background_review.py b/agent/background_review.py index 3959ca0d1a..a83382d032 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -119,11 +119,18 @@ def _interrupt_background_review(review_agent: Any) -> None: def cancel_background_review_for_live_turn(agent: Any) -> None: """Cancel the current review and await its request-phase acknowledgement. Foreground priority: past the bounded deadline, warn and let the live turn proceed — self-improvement work must - never block a user-facing turn.""" + never block a user-facing turn. + + Foreground priority is preserved: if the review does not acknowledge within the bounded deadline, a + warning is logged and the live turn proceeds anyway. See #84423. + """ with _optional_lock(agent, "_background_review_lock"): run = getattr(agent, "_background_review_run", None) legacy_agent = getattr(agent, "_background_review_agent", None) review_agent = legacy_agent if run is None else run.cancel() + # Attribute the review fork's usage to the PARENT session. Snapshot BEFORE unregister/close so counters + # survive teardown. Placed in this finally so a fork that consumed tokens and THEN raised is still + # attributed (issue #87250). Best-effort: the recorder never raises into the review thread. if review_agent is not None: _interrupt_background_review(review_agent) if run is None: @@ -594,7 +601,10 @@ def summarize_background_review_actions( skill-management tool results from the review agent's messages, skipping tool messages already present in ``prior_snapshot`` so inherited results are not re-surfaced as fresh work. ``notification_mode``: ``off`` -> no actions; ``on`` -> generic "Memory updated"/tool messages; - ``verbose`` -> content previews from the tool-call arguments.""" + ``verbose`` -> content previews from the tool-call arguments. + + See #14944. + """ mode = str(notification_mode or "on").lower() if mode == "off": return [] @@ -814,6 +824,13 @@ def build_cache_parity_fork( # Same model only: share the warm cached system prompt (~26% cost cut; a rebuilt prompt misses # the byte-exact prefix key) and pin session_start so any re-render (compression, plugin # hooks) stays byte-identical. + # Inherit the parent's cached system prompt verbatim so the review fork's outbound HTTP request hits the + # same Anthropic/OpenRouter prefix cache the parent warmed. Without this, the fork rebuilds the system + # prompt from scratch (fresh _hermes_now() timestamp, fresh session_id, narrower toolset → different + # skills_prompt) and the byte-exact prefix-cache key misses. See issue #25322 and PR #17276 for the full + # analysis + measured impact (~26% end-to-end cost reduction on Sonnet 4.5). When routed to a different + # model the parent's cached prompt is for the wrong model/cache key and would miss anyway, so let the + # routed fork build its own. if not _routed: review_agent._cached_system_prompt = agent._cached_system_prompt review_agent.session_start = agent.session_start @@ -824,6 +841,9 @@ def build_cache_parity_fork( return review_agent, _rt, _routed +# Install a non-interactive approval callback on this worker thread so any dangerous-command guard the +# review agent trips resolves to "deny" instead of falling back to input() -- which deadlocks against the +# parent's prompt_toolkit TUI (#15216). Same pattern as _subagent_auto_deny in tools/delegate_tool.py. def _bg_review_auto_deny(command, description, **kwargs): """Non-interactive approval: dangerous-command guards resolve to "deny" instead of input(), which would deadlock against the parent's TUI.""" @@ -876,6 +896,21 @@ def _review_tool_whitelist(review_agent: Any, task_cfg: Optional[Dict[str, Any]] whitelist |= {"read_file", "search_files"} # ``extra_tools`` admits named parent tools (e.g. a human-gated proposal tool). The whitelist # can only admit, never advertise: a listed tool must already exist in the inherited schema. + # Read-only file tools are whitelisted too (#61521, #39996): the model naturally reaches for + # read_file/search_files to inspect a skill before patching it. Denying them caused a per-review denial + # storm (~142 denials + ~204 read-before-write refusals over 2 days on one deployment) that starved the + # self-improvement loop — the model never loaded SKILL.md the way the read-before-write guard requires, + # so almost no patch landed. This is a DISPATCH-side change only: the advertised ``tools[]`` stays + # byte-identical to the parent's, so prompt-cache parity is untouched. read_file registers the read with + # the read-before-write guard (tools/file_tools.py), so a read_file → skill_manage(patch) sequence now + # succeeds. Write tools (write_file/patch/terminal) stay denied — autonomous maintenance must go through + # skill_manage's validation, and the deny message below names that substitute so one denial redirects + # the model instead of a storm. + # Profile-configured opt-in tools (#44672, salvage #82146 by @BrinShadewater): + # ``auxiliary.background_review.extra_tools`` admits named parent tools to the review whitelist — e.g. a + # human-gated proposal tool or a memory-provider write surface. Read from task_cfg (the + # auxiliary.background_review block already loaded for this spawn) so no extra config I/O happens per + # review. configured_extra_tools: set = set() try: extra_raw = _background_review_task_config(task_cfg).get("extra_tools", []) @@ -972,7 +1007,10 @@ def _run_review_in_thread( """Daemon-thread worker: build the fork, run the prompt, surface the action summary via ``agent._safe_print`` / ``background_review_callback``. ``review_run`` (from :func:`prepare_background_review_run`) cancelled before the first provider call aborts - without entering ``run_conversation()``.""" + without entering ``run_conversation()``. + + See #84423. + """ if review_run is not None and review_run.cancel_requested.is_set(): finish_background_review_run(agent, review_run) return @@ -993,11 +1031,25 @@ def _run_review_in_thread( try: # Silence stdout/stderr for THIS thread only: a process-global redirect would blank every # other thread's console for the whole review. + # A process-global ``contextlib.redirect_stdout(devnull)`` here would also blank + # ``sys.stdout``/``sys.stderr`` for every other thread — including a gateway event-loop thread + # driving a Telegram long-poll — for the full duration of the review (tens of seconds), swallowing + # their console output (#55769 / #55925). ``thread_scoped_silence`` routes only this thread's writes + # to devnull and leaves all other threads on the real streams. with thread_scoped_silence(): _run_review_fork(agent, messages_snapshot, prompt, task_cfg, review_run, st) # A buggy/legacy tool response shape must NOT take down the whole review (the outer # except would discard every action the fork DID complete), so coerce to an empty list. try: + # Scan the review agent's messages for successful tool actions and surface a compact summary to + # the user. Tool messages already present in messages_snapshot must be skipped, since the review + # agent inherits that history and would otherwise re-surface stale "created"/"updated" messages + # from the prior conversation as if they just happened (issue #14944). ``_change`` returned as a + # list instead of a dict, #59437) must NOT take down the whole review with an AttributeError, + # since the caller's outer except logs only "Background memory/skill review failed" and discards + # every successful action the fork DID complete before the crash. Coerce an exception into an + # empty actions list so the partial valid actions from earlier in the messages are returned + # instead. actions = summarize_background_review_actions( st.review_messages, messages_snapshot, notification_mode=getattr(agent, "memory_notifications", "on"), diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index 2771372ae3..26f0152516 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -24,6 +24,11 @@ logger = logging.getLogger(__name__) # boto3 is not in the [all] extras; lazy_deps installs it on demand. try: + # --------------------------------------------------------------------------- Ensure boto3/botocore are + # installed before any code in this module runs. Upstream removed boto3 from [all] extras (PRs #24220, + # #24515); lazy_deps handles on-demand installation so the Bedrock provider still works in the EKS + # deployment without baking boto3 into the base image. + # --------------------------------------------------------------------------- from tools.lazy_deps import ensure ensure("provider.bedrock", prompt=False) except Exception: @@ -260,7 +265,12 @@ def resolve_aws_auth_env_var(env: Optional[Dict[str, str]] = None) -> Optional[s def has_aws_credentials(env: Optional[Dict[str, str]] = None) -> bool: - """True if any AWS credential source (env vars or boto3 chain) is detected.""" + """True if any AWS credential source (env vars or boto3 chain) is detected. + + This two-tier approach mirrors the pattern from OpenClaw PR #62673: cloud environments (EC2, ECS, + Lambda) provide credentials via instance metadata, not environment variables. The env-var check is a + fast path for local development; the boto3 fallback covers all cloud deployments. + """ return resolve_aws_auth_env_var(env) is not None or _boto3_chain_has_credentials() @@ -431,6 +441,8 @@ def convert_tools_to_converse(tools: List[Dict]) -> List[Dict]: # Converse rejects empty OR whitespace-only text blocks, so the placeholder must be non-whitespace. +# A lone space is whitespace and is rejected too — the placeholder MUST itself be non-whitespace. Ref: issue +# #9486. _EMPTY_TEXT_PLACEHOLDER = "(empty)" _PLACEHOLDER_BLOCK = {"text": _EMPTY_TEXT_PLACEHOLDER} @@ -447,6 +459,7 @@ def _image_block_from_data_url(url: str) -> Dict: header, _, data = url.partition(",") media_type = (header[5:].split(";")[0] if header.startswith("data:") else "") or "image/jpeg" try: + # Ref: #33317. raw_bytes = base64.b64decode(data) except Exception: raw_bytes = data.encode("utf-8") diff --git a/agent/browser_provider.py b/agent/browser_provider.py index 5b975543d9..84889ca242 100644 --- a/agent/browser_provider.py +++ b/agent/browser_provider.py @@ -58,6 +58,13 @@ class BrowserProvider(ProviderBase): # Legacy ``CloudBrowserProvider`` names still used by ``tools.browser_tool`` and out-of-tree subclasses. + # ------------------------------------------------------------------ Backward-compat shims for the + # legacy CloudBrowserProvider API ------------------------------------------------------------------ The + # pre-PR-#25214 ABC exposed ``is_configured()`` and ``provider_name()``; ``tools.browser_tool`` has ~6 + # callers that still use those names. Rather than churn every callsite (and break out-of-tree downstream + # code that subclassed CloudBrowserProvider), we expose the old names as thin delegations to the new + # API. Subclasses MUST implement :meth:`is_available` and :attr:`name`; they may override + # ``is_configured`` / ``provider_name`` for compatibility with the legacy ABC but it is not required. def is_configured(self) -> bool: """Backward-compat alias for :meth:`is_available`.""" return self.is_available() diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index d95e5eac36..6a1350082c 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -30,6 +30,9 @@ from agent.errors import EmptyStreamError from agent.fast_mode import effective_request_overrides from agent.turn_context import substitute_api_content from agent.gemini_native_adapter import is_native_gemini_base_url +# Remote endpoints must never be fingerprinted: the probe waterfall is only valid for local/LM-Studio/Ollama +# boxes. Non-Ollama remotes (sglang, vLLM, OpenAI-compat) expose Ollama-compat endpoints that can +# misidentify and, without an api_key, return 401 on every leg (issue #89863). from agent.model_metadata import is_local_endpoint from agent.message_content import flatten_message_text from agent.message_metadata import append_message, stamp_message_timestamp @@ -1630,6 +1633,18 @@ def _assistant_tool_call_dict(agent, tool_call, index: int) -> dict: "function": {"name": tool_call.function.name, "arguments": tool_call.function.arguments}} # Preserve extra_content (Gemini thought_signature) or Gemini 3 thinking # models 400 on the next request. + # Tool-call arguments are intentionally NOT redacted here. This dict enters the in-memory conversation + # history that is replayed to the model on every subsequent turn AND persisted to state.db, which is + # itself replayed verbatim on session resume (get_messages_as_conversation). Masking a credential to + # `***` here poisons that replay: the model reads back its own `PGPASSWORD='***' psql ...` call and + # copies the placeholder into the next tool call, breaking every credential-dependent command on the + # second turn (#43083). The masking also provided no real protection — the same secret still leaks + # verbatim through tool OUTPUT (file contents, command output, diffs, the compaction block), none of + # which this pass ever touched. Keeping secrets out of the replayable store is a separate + # tokenization/vault concern, not something arg-redaction can deliver without breaking replay. + # Storage-time redaction remains governed by the `security.redact_secrets` toggle. (#19798 introduced + # this; #43083 removed it.) Preserve extra_content (e.g. Gemini thought_signature) so it is sent back on + # subsequent API calls. Without this, Gemini 3 thinking models reject the request with a 400 error. extra = getattr(tool_call, "extra_content", None) if extra is not None: tc_dict["extra_content"] = _dump_if_model(extra) @@ -1658,6 +1673,11 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic elif assistant_tool_calls and agent._needs_thinking_reasoning_pad(): # DeepSeek v4 / Kimi thinking modes 400 on a replayed tool-call message without # reasoning_content; pad with a single space (empty string is rejected too). + # Without it, replaying the persisted message causes HTTP 400 ("The reasoning_content in the + # thinking mode must be passed back to the API"). Include streamed reasoning text when captured; + # otherwise pad with a single space — DeepSeek V4 Pro tightened validation and rejects empty string + # ("The reasoning content in the thinking mode must be passed back to the API"). A space satisfies + # non-empty checks everywhere without leaking fabricated reasoning. Refs #15250, #17400, #17341. msg["reasoning_content"] = reasoning_text or " " elif reasoning_text: # Streaming-only providers accumulate reasoning via deltas and never set @@ -1665,6 +1685,16 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic # Promote ONLY when nothing set the field: SDK reasoning_content and the # tool-call pad win, and reasoning-less turns leave the field absent so # the replay-time leak guard and promotion tiers still apply. + # Additive fallback (refs #16844, #16884). Streaming-only providers (glm, MiniMax, gpt-5.x via aigw, + # Anthropic via openai-compat shims) accumulate reasoning through ``delta.reasoning_content`` chunks + # but never land it on the message object as a top-level attribute, so neither branch above fires + # and the chain-of-thought is stored only under the internal ``reasoning`` key. When the user later + # replays that history through a DeepSeek-v4 / Kimi thinking model, the missing + # ``reasoning_content`` causes HTTP 400 ("The reasoning_content in the thinking mode must be passed + # back to the API."). Promote the already-sanitized streamed ``reasoning_text`` to + # ``reasoning_content`` at write time, but ONLY when no prior branch already set it AND we actually + # captured reasoning text. This preserves every existing behavior: - SDK-exposed + # ``reasoning_content`` (OpenAI/Moonshot/DeepSeek SDK) still wins. msg["reasoning_content"] = reasoning_text if getattr(assistant_message, "reasoning_details", None): @@ -1874,6 +1904,8 @@ def _should_skip_fallback_candidate(agent, fb: dict, fb_key: tuple, fb_provider: return True # Identity semantics (axes, shim aliases, credential surfaces, multi-endpoint pools) # are owned by agent.backend_identity — do not re-implement comparisons here. + # Skip entries that resolve to the same backend that just failed — falling back to it loops the failure. + # See #22548, #62984, #70893. from agent.backend_identity import BackendIdentity, should_skip_candidate current_ident = BackendIdentity.build(provider=getattr(agent, "provider", ""), model=getattr(agent, "model", ""), base_url=str(getattr(agent, "base_url", "") or "")) @@ -1937,6 +1969,8 @@ def _reresolve_fallback_reasoning_config(agent) -> None: """Per-model override > global reasoning_effort (YAML False = disabled); a config load failure keeps the current reasoning_config rather than killing the swap.""" try: + # Re-resolve reasoning_config for the new fallback model (Closes #21256). Wrapped in try/except + # because a config load failure must not kill the swap. from hermes_cli.config import load_config from hermes_constants import resolve_reasoning_config agent.reasoning_config = resolve_reasoning_config(load_config() or {}, agent.model) @@ -2032,6 +2066,7 @@ def try_activate_fallback(agent, reason: "FailoverReason | None" = None) -> bool # Clear the per-config context_length override so the fallback model's own context # window is resolved instead of the previous model's stale value. + # See #22387. agent._config_context_length = None agent.model, agent.provider, agent.requested_provider = fb_model, fb_provider, fb_provider agent.base_url, agent.api_mode = fb_base_url, fb_api_mode @@ -2102,6 +2137,13 @@ def _iteration_summary_api_messages(agent, messages: list) -> list: api_msg.pop(key, None) # api_content holds the exact bytes the main loop sent; substituting (not popping) # keeps the summary's prefix identical instead of re-prefilling the largest context. + # Strict OpenAI-compatible gateways (Fireworks-backed OpenCode Go, Mistral, Moonshot/Kimi) reject + # any message key outside the Chat Completions schema. The main loop drops these via + # ChatCompletionsTransport.convert_messages(), but the summary path hand-builds messages and calls + # chat.completions.create() directly, bypassing the transport — so mirror that sanitization here: + # tool_name (SQLite FTS bookkeeping), the codex_* reasoning carriers, timestamp (preserved on + # gateway user replay entries for the stale-confirmation expiry check — #47868 rejection class), and + # every Hermes-internal underscore-prefixed scaffolding key. substitute_api_content(api_msg) if needs_sanitize: agent._sanitize_tool_calls_for_strict_api(api_msg, model=sanitize_model) @@ -2243,6 +2285,8 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str: if getattr(agent, "suppress_status_output", False): # Strict machine-readable mode (-Q, oneshot): keep diagnostics off stdout. quiet_mode is # NOT the gate — the interactive CLI runs quiet_mode=True by default and must see this. + # Strict machine-readable mode (hermes chat -Q, oneshot, background review): keep diagnostics out of + # stdout so wrappers receive only the final assistant content (#93220 class). logger.warning(warning) else: agent._safe_print(warning) @@ -2797,6 +2841,8 @@ class _StreamingCall: as in-stream chunks (choices=None + error_type/error_message), which would otherwise surface as a misleading EmptyStreamError plus retries.""" usage = chunk.usage if hasattr(chunk, "usage") and chunk.usage else None # final usage chunk + # Without this check the error is silently dropped and the stream ends empty → EmptyStreamError → + # misleading "empty stream" message and pointless retries on the same bad request. (#65631) _err_type = getattr(chunk, "error_type", None) _err_msg = getattr(chunk, "error_message", None) if _err_type or _err_msg: @@ -2806,6 +2852,7 @@ class _StreamingCall: raise ProviderStreamError(status_code=_status, body=body, raw_text=f"{_err_type}: {_err_msg}") # Nous Portal usage frames (choices=[] + lastOne=true, no [DONE]) are a # clean terminal, not a drop; relabelled upstreams send 1 / "true". + # See #90848. last_one = getattr(chunk, "lastOne", None) if last_one is None and isinstance(getattr(chunk, "model_extra", None), dict): last_one = chunk.model_extra.get("lastOne") diff --git a/agent/client_lifecycle.py b/agent/client_lifecycle.py index c54b9c8ec2..fca0d90975 100644 --- a/agent/client_lifecycle.py +++ b/agent/client_lifecycle.py @@ -117,6 +117,13 @@ class ClientLifecycleMixin: ``close()`` releases FDs from the calling thread while other threads may still hold the fd in an SSL BIO; a recycled fd then gets a TLS record written into an unrelated file (SQLite-header corruption). + + The shared primary client has no single owning thread — worker threads from stale-killed attempts + may still be unwinding their SSL BIOs, and the codex-direct / MoA paths stream on the shared client + itself. If we release an FD while another thread's SSL layer still caches the raw integer fd, the + kernel can recycle it into an unrelated ``open()`` (e.g. ``kanban.db``) and the unwinding TLS flush + then writes an application-data record into that file — the SQLite-header corruption documented in + #29507/#70773. """ if client is None: return @@ -134,6 +141,12 @@ class ClientLifecycleMixin: The worker may be blocked in an OpenSSL read; hard-closing from the timeout thread releases FDs under a live BIO (native corruption / SIGSEGV). Only ``shutdown()`` so the read sees EOF and the worker closes itself. + + See #94248. + A delegation deadline abandons this agent's daemon worker while it may still be blocked inside an + in-flight OpenSSL ``read`` (Codex Responses stream, httpx request). This helper only ``shutdown()``s + pooled sockets (safe from any thread), settling blocked reads with EOF/EPIPE so the worker can + unwind and run the real close from its own thread. See #70773, #94248. """ drained = 0 # Shared primary client (codex-direct / MoA stream on it directly). @@ -194,6 +207,11 @@ class ClientLifecycleMixin: return False self.client = new_client # Never hard-close the replaced shared client (another thread may still be unwinding on the old pool). + # #70773: never hard-close the replaced shared client from here — the caller may not be the thread + # whose request is still unwinding on the old pool (credential rotation and dead-connection cleanup + # run on the turn thread while stale-killed workers unwind; the codex-direct path streams on the + # shared client itself). Retire it instead: sockets are shut down (FD-safe), FD release deferred to + # GC. self._retire_shared_openai_client(old_client, reason=f"replace:{reason}") return True @@ -310,6 +328,10 @@ class ClientLifecycleMixin: try: shutdown_count = self._force_close_tcp_sockets(client) # Zero sockets shut down means the worker stays blocked — WARN, not success. + # tcp_force_closed=0 means the stranger-thread abort found no sockets to shut down — the worker + # stays blocked in recv and the provider keeps the slot (#72975). Surface that as WARNING so it + # cannot be mistaken for a successful abort in the logs. + # See #72975. _log = logger.warning if shutdown_count == 0 else logger.info _log( "%s client aborted (%s, shared=False, tcp_force_closed=%d, deferred_close=stranger_thread) %s%s", @@ -572,7 +594,12 @@ class ClientLifecycleMixin: def _try_refresh_env_client_credentials(self) -> bool: """Adopt ``~/.hermes/.env`` credential/base-url edits at the turn boundary (a Settings save updates ``.env`` - but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``.""" + but a live worker keeps init-time values). Adoption rule: ``_should_adopt_env_credentials``. + + Covers api-key registry providers and named custom providers with a ``key_env`` (#67935) — the + latter resolve to ``provider="custom"`` with no registry entry, so they are matched through the + runtime provider's config lookup instead. + """ if self.api_mode != "chat_completions" or getattr(self, "_fallback_activated", False): return False resolved = self._resolve_env_credentials() diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index bcd706f136..fcf58d0278 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -426,10 +426,38 @@ def _chat_messages_to_responses_input( ``native_compaction_eligible``: THIS request carries ``context_management``; gates both replaying ``compaction`` checkpoints and ``prune_pre_checkpoint_items``. Checkpoints persist across model swaps / compression flips / resume, so without the gate one checkpoint would erase pre-checkpoint history on a model that cannot decrypt it (lossless: - local history is never truncated).""" + local history is never truncated). + + Earlier (PR #26644, May 2026) we believed xAI's OAuth/SuperGrok ``/v1/responses`` surface rejected + replayed ``encrypted_content`` reasoning items minted by prior turns, and we stripped them. That + decision was wrong — xAI explicitly relies on Hermes threading encrypted reasoning back across turns for + cross-turn coherence (the whole point of their partnership integration). We now replay encrypted + reasoning on every Responses transport (xAI, native Codex, custom relays) and let xAI tell us explicitly + if a specific surface ever rejects a payload. + The Copilot backend (api.githubcopilot.com/responses) binds these ids to a specific backend "connection" + — credential-pool rotation, a gateway restart, or routine load-balancer churn between turns all + invalidate it — and rejects a stale id with HTTP 401 "input item ID does not belong to this connection" + even for short ids (see #32716). ``phase``/ ``status``/``content`` are still replayed; only ``id`` is + unsafe to reuse across a Copilot connection. + ``native_compaction_eligible`` mirrors, for THIS request, the decision made by + ``native_compaction.native_compaction_context_management`` — it is True only when that gate returned a + payload, i.e. when the request actually carries ``context_management``. It controls two things that must + never outlive the gate: replaying ``type: "compaction"`` checkpoint items, and restructuring the wire + around them (``prune_pre_checkpoint_items``). Checkpoints are persisted in the ``codex_reasoning_items`` + sidecar and survive a mid-session model swap, a ``compression.enabled: false`` flip, the rejection kill + switch and a resumed session; without this flag a single captured checkpoint would keep deleting every + pre-checkpoint item from every later request, on a model that cannot decrypt the blob (#85914). Default + False = pre-feature wire, which is also correct for every caller that never sends ``context_management`` + (auxiliary/compression client, ad-hoc ``convert_messages``). Dropping the checkpoint costs nothing: + Hermes' local history is never truncated by native compaction, so the full conversation is still on the + wire. + """ items: List[Dict[str, Any]] = [] # Parallel to ``items``: source chat message per item. Pruning reads a summary # carrier's provenance from the source; the converted item may be a lossy shape. + # Pruning needs this to read a canonical summary carrier's up-to-date, provenance-tagged content + # directly — the converted `item` can be a lossy shape (stale exact-replay, or a typed + # `function_call_output` wrapper) that no longer carries it (#90976). item_sources: List[Optional[Dict[str, Any]]] = [] seen_item_ids: set = set() def emit(new_items: List[Dict[str, Any]], msg: Dict[str, Any]) -> None: @@ -470,6 +498,17 @@ def _chat_messages_to_responses_input( # The server renders nothing placed before a compaction item, so pre-checkpoint history is # dead weight and plaintext asks / merged summaries silently vanish. Keep the newest checkpoint # first, retain pre-checkpoint USER and SUMMARY messages within a token budget, leave the tail. + # Native server-side compaction: when a replayed checkpoint is present, restructure the wire around it. + # Gated on the CURRENT request's native eligibility, not merely on the presence of a checkpoint: a + # persisted checkpoint outlives the gate, and pruning for a request that carries no + # ``context_management`` deletes history the server never compacted. ``item_sources`` (parallel to + # ``items``) carries the raw chat message each converted item came from. A canonical summary carrier's + # content can be lost or gone stale by the time it becomes a Responses item — a merge-into-tail + # tool-result carrier becomes a typed ``function_call_output`` (no ``content``/``role`` at all), and a + # merge-into-tail assistant carrier can be shadowed by a stale exact ``codex_message_items`` replay from + # before the merge rewrote its content. Pruning reads the source message's own up-to-date, + # provenance-tagged content directly instead of trying to recover it from whatever shape the conversion + # produced (#90976). if not native_compaction_eligible: return items from agent.native_compaction import prune_pre_checkpoint_items @@ -478,7 +517,12 @@ def _chat_messages_to_responses_input( class ResponsesRouteFlags(NamedTuple): """Which special Responses-API route an agent is talking to. Single owner of the - codex/xai/github predicates — every site must call :func:`classify_responses_route`.""" + codex/xai/github predicates — every site must call :func:`classify_responses_route`. + + Every site that needs these flags (request kwargs build, preflight estimation, silent- reject hints) + must call :func:`classify_responses_route` instead of re-implementing the string comparisons inline — + inline copies drift (backend-identity class: #22548/#70893/#59561/#72468). + """ is_codex_backend: bool is_xai_responses: bool is_github_responses: bool @@ -505,7 +549,12 @@ def estimate_native_responses_preflight_tokens( agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, ) -> Optional[int]: """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted - session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails.""" + session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails. + + Automatic preflight previously counted the full durable transcript. On a natively compacted Codex + session that overstates the wire by several times and fires local compression against history the main + request will never send (#96155). + """ if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list): return None route = classify_responses_route(agent)._asdict() diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index 4e078705b2..62268dd81e 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -307,6 +307,9 @@ def make_codex_app_server_event_bridge(agent) -> Callable[[dict], None]: def _fire_delta(params: dict, attr: str) -> None: text = params.get("delta") or params.get("text") or "" + # Single-writer guard (#65991): a superseded stream must not pollute the turn's accumulated text + # (which also feeds the interim-visible-text de-dup comparison), even when a caller reaches this + # directly (the tool-suppressed content path) rather than through _fire_stream_delta. if isinstance(text, str) and text: agent_cb(attr, f"{attr} raised", args=(text,)) @@ -380,6 +383,11 @@ def _ensure_codex_session(agent) -> None: auto_approve_requests = is_approval_bypass_active() except Exception: logger.debug("codex app-server: approval-bypass lookup failed; keeping fail-closed default", exc_info=True) + # Bridge codex JSON-RPC notifications (item/started, item/completed, item/agentMessage/delta, ...) into + # Hermes' gateway UI callbacks (tool_progress_callback, _fire_stream_delta, + # _emit_interim_assistant_message). Without this, Discord/Telegram users see no live tool-progress or + # interim commentary while codex_app_server is running — only the final answer (#33200). Supersedes the + # narrower item/started-only bridge from #38835. agent._codex_session = CodexAppServerSession( cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), approval_callback=approval_callback, request_routing=_ServerRequestRouting(auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests), diff --git a/agent/compaction_display.py b/agent/compaction_display.py index ba04e93a92..53d42ea0dd 100644 --- a/agent/compaction_display.py +++ b/agent/compaction_display.py @@ -11,6 +11,18 @@ _COMPACTION_INTERNAL_FIELDS = ( "tool_calls", "finish_reason", "reasoning", + # Provider replay/metadata fields that ride the wire on every request but are invisible to + # ``msg["content"]``/``msg["tool_calls"]`` accounting. Codex Responses sessions in particular carry + # ``codex_reasoning_items`` blobs of ``encrypted_content`` that can dominate the serialized session (a + # measured 214-turn session held ~115K tokens / 27% of its payload there — #55572). + # ``reasoning_details`` is handled separately (see ``_reasoning_details_text_chars``): its signed/base64 + # envelope is excluded from the budget, mirroring the preflight estimator's exclusion in + # ``model_metadata._estimate_message_tokens_without_images`` (#73298). + # An assistant turn may carry only reasoning/thinking content with no visible text (extended-thinking + # turns, thinking-only recovery responses). Such a turn is persisted with its reasoning fields and is + # recallable from the transcript, but dropping it here as "empty" makes it vanish from the + # resumed/reloaded session view while the desktop's reasoning disclosure has nothing to render. Keep it + # when it carries reasoning so the "Thinking…" block still shows. (#44022) "reasoning_content", "reasoning_details", "codex_reasoning_items", diff --git a/agent/compression_facade.py b/agent/compression_facade.py index fe99989d79..6d55c2ab6d 100644 --- a/agent/compression_facade.py +++ b/agent/compression_facade.py @@ -114,6 +114,14 @@ def _run_under_progress_timeout( from agent.conversation_compression import CompressionCommitFence, run_compress_context_with_progress_timeout def _snapshot_worker(fence=None): + # #76354 review F3: the pooled worker must NEVER share the caller's live transcript. Plugin/legacy + # context engines are allowed to mutate their input list in place; after a host timeout the worker + # stays alive, so a shared list would let a late engine rewrite the live conversation (roles, + # ordering, persisted content) behind the caller's back. Deep-snapshot here, on the worker thread, + # so the caller's list object is never touched by pooled code. Results are published to + # caller-visible state only via the returned value of an ADMITTED commit (the host discards results + # on timeout/cancel); durable SessionDB mutation is already gated behind the commit fence inside + # compress_context. snapshot = copy.deepcopy(messages) result_msgs, result_prompt = run(fence, target_messages=snapshot) return (messages if result_msgs is snapshot else result_msgs), result_prompt @@ -185,9 +193,18 @@ class CompressionFacadeMixin: ) -> tuple: """Forwarder — see ``agent.conversation_compression.compress_context``. ``force=True`` (manual /compress) bypasses the summary-failure cooldown; ``bypass_cooldown=True`` - (provider-proven overflow recovery) runs one real attempt while the cooldown stays armed.""" + (provider-proven overflow recovery) runs one real attempt while the cooldown stays armed. + + ``force=True`` is passed by the manual ``/compress`` slash command so users can bypass the + summary-failure cooldown after an auto-compress abort. Auto-compress callers use the default + ``force=False``. See #100661. + """ # Per-attempt timeout signal for turn-start preflight and in-loop consumers: a stalled # compression must not be mistaken for a structural no-op. Thread-local + per-agent lock. + # A stalled compression must not be mistaken for a structural no-op and followed by the oversized + # provider request it was meant to prevent. The typed helper upgrades the simple attribute to + # thread-local state guarded by a per-agent lock so overlapping automatic/manual entrypoints cannot + # clobber each other's outcome (#98741). from agent.conversation_compression import ( CompressionCommitFence, compress_context, reset_context_compression_timeout_outcome, resolve_context_compression_timeouts, @@ -206,6 +223,9 @@ class CompressionFacadeMixin: root = self._conversation_root_id() if root: token = set_conversation_context(root) + # Initialized alongside `token`: the turn-lease timeout/interrupt early returns leave the try block + # before set_affinity_scope() runs, and the finally reads this name unconditionally + # (UnboundLocalError otherwise — the 4 red cross-process lease tests on PR #97158). affinity_token = None if get_affinity_scope() is None: declared = declared_conversation_scope_safe(self) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index d33a834d6e..d53770436b 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -46,6 +46,19 @@ def _safe_int(value: Any) -> int | None: # summary sees it while the detached stalled worker does not. A stall raises nothing, so the aux client's # exception-path fallback never fires; the host pins a fallback route for exactly ONE retry (the sole aux # call per compaction). The main-model retry must NOT re-issue the pin. +# ── Pinned summary route ───────────────────────────────────────────────── The summary call normally +# resolves its provider/model from ``auxiliary.compression``. One caller needs to override that for a single +# attempt: after the host's progress-aware timeout aborts a stalled summary (#78981), +# ``agent.conversation_compression`` re-runs compression with the route pinned to a configured +# ``fallback_chain`` entry. Nothing raised out of the stalled call, so the auxiliary client's own fallback +# handling — which only runs from its exception path — never saw that failure. A ContextVar, not an +# attribute on the compressor: the aborted worker is detached and still alive on the pool, and the +# compressor object is shared with it. Context is copied per worker (``propagate_context_to_thread``), so +# the pin reaches the retry's whole synchronous call chain and cannot leak into the stalled attempt or any +# unrelated auxiliary call. Coverage is the single ``_generate_summary`` LLM call only. That is one call per +# compression run (its only non-recursive call site is the compress path; the two recursive calls are the +# deliberate main-model retry that must NOT re-issue the pin). The summary call is the ONLY auxiliary LLM +# call a lean compaction attempt makes (#96603) — there are no sibling digest calls. _SUMMARY_ROUTE_PIN: contextvars.ContextVar[Optional[Dict[str, Any]]] = ( contextvars.ContextVar("hermes_summary_route_pin", default=None) ) @@ -93,7 +106,10 @@ _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = ( def _is_hygiene_preagent_only_cooldown(error: object) -> bool: """Return True for a cooldown that belongs only to pre-agent hygiene. Hygiene watchdog timeouts / turn-hold deferrals are not evidence of an auxiliary-model failure and - must never block the in-agent compressor.""" + must never block the in-agent compressor. + + See #74136, #86972. + """ text = str(error or "").strip().casefold() return any(marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS) @@ -114,6 +130,10 @@ def _response_finish_reason(response: Any) -> str: # Marker for a length-stopped (PARTIAL) summary; the except-branch classifier keys # on this exact substring, so keep raise sites and classifier in sync. +# RuntimeError marker raised when the summarizer's generation stopped on the output-token cap +# (``finish_reason == "length"``). A length stop means the summary text is PARTIAL — persisting it as a +# compaction checkpoint would silently truncate the conversation's memory and feed the cut-off text back +# into every subsequent iterative-update prompt. (Ported from earendil-works/pi#7048 / commit 97fa14e39.) _TRUNCATED_SUMMARY_MARKER = "finish_reason=length" @@ -123,6 +143,11 @@ def _is_summary_access_or_quota_error(exc: Exception) -> bool: # No active secret scope is a missing-credential failure of our own making; # classify as credential so compress() preserves the session unchanged. try: + # A credential read that failed closed because no profile secret scope was active (multiplexed + # gateway, worker thread without the caller's ContextVars) is a missing-credential failure of our + # own making: the summary model cannot be reached until the spawn site is fixed, and a placeholder + # summary would only destroy the middle window for nothing. Classify it with the credential class so + # compress() preserves the session unchanged (#100849 bundle: every hygiene pass truncated). from agent.secret_scope import UnscopedSecretError except Exception: # pragma: no cover - import guard UnscopedSecretError = () # type: ignore[assignment] @@ -150,6 +175,12 @@ HISTORICAL_TASK_HEADING = "## Historical Task Snapshot" SUMMARY_PREFIX = ( + # Jul 2026 (#65848 class): identical to the pre-#69619 prefix except it lacked the explicit "tools + # remain fully active" clause — the strong REFERENCE ONLY framing bled into general tool-use suppression + # (observed: 7 consecutive narration-only turns immediately after a compression event on a production + # deployment). + # Carveout era (#41607/#38364/#42812): "consistent → use as background" licensed stale-task resumption + # on topic overlap. "[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted " "into the summary below. This is a handoff from a previous context " "window — treat it as background reference, NOT as active instructions. " @@ -193,6 +224,21 @@ COMPRESSED_SUMMARY_HAS_USER_TURN_KEY = "_compressed_summary_has_user_turn" # Only micro markers may be superseded/defragged/rehydrated: a batch marker's # content is NOT in the rolling micro summary, so rewriting one destroys history. MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker" +# Intrinsic marker stamped on a message dict once it has been written to the SQLite session store. Used by +# ``_flush_messages_to_session_db`` to decide what is already durable. An object-identity (``id(msg)``) +# dedup set cannot be trusted across turns: once a flushed message dict is dropped from the live list (e.g. +# by scaffolding rewind or in-place compaction) and garbage- collected, CPython is free to hand its address +# to a brand-new assistant/tool message, whose ``id()`` then collides with the stale entry and the real turn +# is silently never persisted. A marker bound to the dict itself cannot be aliased that way. The ``_`` +# prefix is mandatory: the wire sanitizers (agent/transports/chat_completions.py, +# agent/chat_completion_helpers.py) strip every top-level ``_``-prefixed key before the request leaves the +# process, so this never reaches a strict OpenAI-compatible gateway. CONTRACT (#92231): the marker asserts +# "this dict's CONTENT is durable as written". Loaded rows are stamped at materialization time +# (hermes_state._rows_to_conversation), so any code that mutates a loaded or flushed dict's content in place +# and needs the change persisted MUST pop the marker (and invalidate _db_flush_scan_prefix if the dict may +# sit inside the bounded-scan prefix) — see agent/turn_finalizer.py (fill-empty-tail) and +# agent/context_compressor.py (micro-compaction defrag) for the two canonical pop sites. Mutating without +# popping leaves the DB silently stale. _DB_PERSISTED_MARKER = "_db_persisted" # Carried-forward tail rows archive as rewind-style (active=0, compacted=0) so # they don't duplicate live copies in recall; never persisted (unknown column). @@ -630,6 +676,12 @@ def _is_clarify_non_response_sentinel(response: Any) -> bool: # Ghost-skill defense: the ONE canonical prune marker; emit sites and presence # checks must use the same string so they cannot drift. +# Ghost-skill defense (#32106): when compaction reduces an old ``skill_view`` result to a 1-line metadata +# summary, the model still believes the skill is loaded even though its instructions are gone. The marker +# below is the ONE canonical prune signal — ``_skill_pruned_marker()`` builds it and every presence check +# matches against the same string, so the emit side and the check side can never drift apart (the original +# PR #44166 emitted ``[SKILL_PRUNED:`` but presence-checked ``[SKILL_PRUNED]``, making re-injection fire +# even when the marker had survived). SKILL_PRUNED_MARKER_PREFIX = "[SKILL_PRUNED:" # Small skill_view results stay verbatim; shared by emit site and summarizer scan. _SKILL_VIEW_PRUNE_MIN_CHARS = 5000 @@ -783,6 +835,13 @@ def _build_recovery_footer(session_id: str, region_len: int) -> str: # Detailed session log comes from the SAME single summary request (one aux LLM # call per attempt); coverage via input sampling, exact needles via anchor index. +# One flat 2-3K-token summary cannot carry a 400K+ region's specifics — the eval showed recall collapsing to +# ~33% when the big tail (which accidentally archived restated facts) shrank. The detailed, +# identifier-preserving session log is produced by the SAME single summary request as the narrative summary +# (one auxiliary LLM call per compaction attempt, total — #96603: the earlier per-chunk digest loop made up +# to 28 extra aux calls and pushed compactions to 7-11 minutes on slow aux routes). Coverage over oversized +# regions comes from even input sampling (see ``_sample_summary_input``), and exact-needle defense comes +# from the LLM-free anchor index below. _LEAN_SESSION_LOG_HEADING = "## Detailed Session Log (oldest first)" # Extra output-token guidance for the session-log section (single response). _LEAN_SESSION_LOG_BUDGET_TOKENS = 4_000 @@ -852,6 +911,8 @@ def _build_anchor_index(turns: List[Dict[str, Any]]) -> str: # Message-count window (distinct from the token-based tail boundary) in which a # just-loaded skill_view body must survive the Phase-1 prune. +# A skill_view call within this many trailing messages counts as "just loaded": its full instruction body +# must survive the Phase-1 prune even when the token-budget boundary would otherwise demote it (#32106). _SKILL_PRUNE_RECENT_WINDOW = 10 @@ -912,12 +973,15 @@ _MAX_TAIL_MESSAGE_FLOOR = 8 # Skip the LLM call when the compressible middle is below this fraction of the # threshold (and a prior ineffectiveness strike exists); dropping alone suffices. +# See #60451. _FEASIBILITY_SKIP_MIDDLE_FRACTION = 0.10 # Under pressure, demote large tool outputs even inside the protected region but # keep this many trailing messages verbatim. _PRESSURE_KEEP_RECENT_MESSAGES = 3 # Newest image-bearing tool results kept verbatim; older image payloads retire # even inside protect_last_n (matches the Anthropic adapter's keep-window). +# Native vision_analyze / computer_use screenshots that sit inside the protected tail cannot be demoted by +# pass 2, so they ride every later request until anti-thrash disables compression (#92699). _MAX_KEEP_TOOL_IMAGES = 3 # Below this window the threshold is floored (raise-only): at 50% the incompressible @@ -929,6 +993,8 @@ _SMALL_CTX_THRESHOLD_PERCENT = 0.75 _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") # MEDIA directives must not reach the summarizer or they get re-emitted as active. +# MEDIA delivery directives must not reach the summarizer — if one leaks into the summary, the downstream +# model may re-emit it as an active directive on the next turn, triggering bogus attachment sends (#14665). _MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") _HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)") @@ -1079,6 +1145,9 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) - tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN # Charge only thinking TEXT, never the signed/base64 envelope; skip when the # same text already rides in reasoning/reasoning_content. + # When the same thinking text already rides in ``reasoning``/``reasoning_content`` (measured + # byte-identical on Anthropic-wire sessions), skip it here entirely so the prose is not charged twice on + # top of the envelope exclusion. See #73298. if not (msg.get("reasoning") or msg.get("reasoning_content")): tokens += _reasoning_details_text_chars(msg.get("reasoning_details")) // _CHARS_PER_TOKEN return tokens @@ -1256,6 +1325,11 @@ def _strip_historical_media(messages: List[Dict[str, Any]]) -> List[Dict[str, An # non-empty for the zero-user-turn guard). Rule 2: superseded tool-result image, even in the tail. return ( (0 < anchor and index < anchor) + # When the ONLY image-bearing user message is the very first one (``anchor == 0``) and newer + # tool-result images exist, the model has moved on — but the opening base64 blob used to survive + # every compaction forever, which is half the wedge in #89938 (the reported session opened with + # a ~200KB poster). When nothing newer exists the opening image IS the newest image and is kept, + # consistent with keep-newest everywhere else. or (anchor == 0 and index == 0 and tool_anchor > 0) or (message.get("role") == "tool" and index != tool_anchor) ) @@ -1437,6 +1511,8 @@ def _json_dict(text: Any) -> dict: """Parse ``text`` as a JSON object; ``{}`` for empty, invalid, or non-object input.""" try: parsed = json.loads(text) if text else {} + # Just-loaded / actively-referenced skills survive verbatim (#32106). Pass-4 pressure demotion overrides + # this. except (json.JSONDecodeError, TypeError): return {} return parsed if isinstance(parsed, dict) else {} @@ -1484,6 +1560,10 @@ def _memory_provider_section(memory_context: str) -> str: def _today_for_prompt() -> str: """Date-only (user tz) for temporal anchoring; "" when the clock fails. Cache-safe: the summary is outside the prefix.""" try: + # Date-only granularity matches system_prompt.py:337 (PR #20451) and the user's configured timezone + # via hermes_time.now(). The compaction summary is a mid-conversation message that is NOT part of + # the cached prefix, so a date here never affects prompt-cache stability. Resolved defensively — a + # clock failure must never block compaction. from hermes_time import now as _hermes_now return _hermes_now().strftime("%Y-%m-%d") except Exception: # pragma: no cover - clock resolution is best-effort @@ -1724,7 +1804,17 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): """Clear all per-session compaction state at a real session boundary. Session end (CLI exit, gateway expiry, id rotation) — NOT /new or /reset. Every per-session flag/counter can contaminate the next live session (suppressed compression, stale cooldowns, - misleading warnings), so the whole surface is reset here.""" + misleading warnings), so the whole surface is reset here. + + Session end (CLI exit, gateway expiry, session-id rotation) goes through this method rather than + ``on_session_reset()`` (/new, /reset). The original fix (#38788) only cleared ``_previous_summary``, + but the same cross-session contamination risk applies to every per-session variable that + ``on_session_reset()`` clears: stale ``_ineffective_compression_count`` can suppress compression in + a subsequent live session; ``_summary_failure_cooldown_until`` can block summary generation; + ``_last_compress_aborted`` can make callers think compression is still aborted; + ``_last_aux_model_failure_*`` can surface stale error warnings; ``_last_summary_dropped_count`` / + ``_last_summary_fallback_used`` can produce misleading user warnings. + """ self._reset_session_compaction_state() def _reset_real_usage_pairing(self) -> None: @@ -1880,7 +1970,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._durable_write("set_compression_ineffective_count", "compression ineffective count", self._ineffective_compression_count) def _load_anti_thrash_recovery_deadline(self) -> None: - """Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed.""" + """Restore the durable recovery deadline (wall-clock epoch); missing storage leaves it disarmed. + + See #100185. + """ self._load_durable("_anti_thrash_recovery_deadline", "get_compression_recovery_deadline", "compression recovery deadline", float, 0.0) def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None: @@ -1924,6 +2017,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._verify_compaction_cleared_threshold = True if feasibility_skip: # A pre-LLM feasibility skip is not a summary-quality verdict: it must neither extend nor reset the streak. + # A deliberate pre-LLM feasibility skip (#60451) is not a summary-quality verdict: it must + # neither extend a fallback streak (two skips would otherwise latch the >= 2 breaker and disable + # compression entirely — including the cheap deterministic dropping the skip exists to reach) + # nor reset one (a skip proves nothing about the summary model's health). if not self.quiet_mode: logger.info( "Compaction completed via pre-LLM feasibility skip; fallback_compression_streak unchanged (%d)", @@ -1980,6 +2077,9 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): return None # Hygiene-only cooldowns share the column but are not a 429/aux fault; the in-agent compressor may run. # A hygiene write may have overwritten an aux-model row; drop the in-memory cooldown too. + # Hygiene watchdog timeouts and turn-hold deferrals persist the same column so the pre-agent pass + # can skip (#74136), but they are not evidence of a 429/aux-model fault. The in-conversation + # compressor has its own budget and must still be allowed to run (#86972). if _is_hygiene_preagent_only_cooldown(state.get("error")): self._summary_failure_cooldown_until, self._last_summary_error = 0.0, None return None @@ -1994,6 +2094,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _record_compression_failure_cooldown(self, cooldown_seconds: float, error: Optional[str]) -> None: # Never shorten a longer live deadline; record the latest error text only. self._summary_failure_cooldown_until = max(self._summary_failure_cooldown_until, time.monotonic() + float(cooldown_seconds)) + # A later stall or timeout records the latest error text but keeps the later of the two clocks. See + # #96775. self._last_summary_error = error cooldown_until = time.time() + max(0.0, self._summary_failure_cooldown_until - time.monotonic()) if not getattr(self, "_session_db", None) or not getattr(self, "_session_id", ""): @@ -2020,6 +2122,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): def _compression_cancelled(self) -> bool: """Read the host-owned cooperative cancellation signal, if installed.""" + # #76354 review F4: fence check BEFORE cooldown-clear. A late worker whose host already timed out + # (and recorded a timeout cooldown) must not undo that cooldown when its summary eventually + # succeeds. The hook is installed by compress_context for the duration of the fenced call; when it + # reports cancellation, keep the host's cooldown. cancelled_check = getattr(self, "_compression_cancelled_check", None) if not callable(cancelled_check): return False @@ -2042,6 +2148,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._base_threshold_percent = resolve_model_threshold(model, self.model_thresholds, _config_pct) self.threshold_percent = self._effective_threshold_percent(context_length, self._base_threshold_percent) # max_tokens=None means "unspecified": keep the existing output reservation. + # A switch that genuinely changes the output budget passes the new value explicitly. (#43547) if max_tokens is not None: self.max_tokens = self._coerce_max_tokens(max_tokens) self.threshold_tokens = self._compute_threshold_tokens(context_length, self.threshold_percent, self.max_tokens) @@ -2072,6 +2179,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): _MIN_CTX_TRIGGER_RATIO = 0.85 # Anti-thrash recovery: after this long blocked, allow ONE probe (counters drop to 1 strike). + # Anti-thrash recovery window (#14694): once the ineffective/fallback breaker trips, automatic + # compaction stays blocked for this long, then ONE probe attempt is allowed (counters drop to 1 strike, + # so another ineffective pass re-trips immediately). Long enough that a genuinely incompressible session + # isn't compacting in a loop; short enough that a session which has since grown real compressible + # material recovers well before it rides into the provider's hard context limit. _ANTI_THRASH_RECOVERY_SECONDS = 300.0 # Structural no-op (nothing eligible) is not an ineffective attempt: defer retries instead of striking. @@ -2109,7 +2221,21 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): ) -> int: """Compute the compaction trigger in tokens from the effective input budget. Base is ``(context_length - max_tokens) * threshold_percent`` floored at MINIMUM_CONTEXT_LENGTH; - when the floor binds it is capped at 85% of the budget so small windows can still fire.""" + when the floor binds it is capped at 85% of the budget so small windows can still fire. + + The base value is ``effective_input_budget * threshold_percent``, floored at + ``MINIMUM_CONTEXT_LENGTH`` so large-context models don't compress prematurely at 50%. BUT that floor + degenerates at small windows: for a model whose ``context_length`` is at/below the minimum (e.g. a + 64K local model), ``max(0.5*64000, 64000) == 64000`` makes the threshold equal the ENTIRE window — + auto-compression can never fire because the provider rejects the request before usage reaches 100% + (#14690). + The provider reserves ``max_tokens`` of output space out of the same window, so the usable INPUT + budget is ``context_length - max_tokens``. With a large ``max_tokens`` (e.g. 65536 on a custom + provider) the input budget is materially smaller than the raw window, and a threshold based on the + full window lets the session hit a provider 400 before compaction fires (#43547). The percentage and + the degenerate-window check below both operate on the effective input budget. ``max_tokens=None`` + (provider default) conservatively assumes no reservation (full window). + """ effective_window = context_length - (max_tokens or 0) if effective_window <= 0: effective_window = context_length @@ -2166,6 +2292,11 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # Usable input = context_length - max_tokens; only a positive int counts as a reservation. self.max_tokens = self._coerce_max_tokens(max_tokens) # True: summary failure aborts (messages unchanged); False: insert deterministic handoff and drop middle. + # Output-token reservation: the provider carves max_tokens out of the context window, so the usable + # input budget is context_length - max_tokens. None = provider default => assume no reservation. + # (#43547) Coerce defensively: only a positive int is a real reservation; any other value (None, + # non-numeric, <=0) means "no reservation" so the threshold arithmetic never sees a non-int (e.g. a + # test MagicMock). self.abort_on_summary_failure = abort_on_summary_failure # Micro-compaction is OFF by default: each pass breaks the prompt-cache prefix every turn. @@ -2173,18 +2304,30 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._reset_micro_compact_cursor_state() self._micro_compact_defrag_threshold_tokens = 2000 # Set when _defrag_rolling_summary pops _DB_PERSISTED_MARKER in place; finalize_turn resets the flush cursor. + # Set by _defrag_rolling_summary when it pops _DB_PERSISTED_MARKER from a live dict in place; + # consumed by finalize_turn to invalidate the agent's bounded flush-scan cursor (sibling of the + # #75170 site). self._flush_scan_cursor_invalidated: bool = False self._micro_compact_passes = self._micro_compact_tokens_saved_total = self._micro_compact_turns_since_pass = 0 # Cadence dial: how often the cache-breaking pass is paid. 1 = every turn. self._micro_compact_every_n_turns: int = 1 # Deferred: get_model_context_length() may issue a sync HTTP probe that must not block construction. # Floor and cap are applied on first resolution (see _resolve_context_length / threshold_tokens). + # The small-context threshold floor and the absolute threshold cap both need the resolved window, so + # they are applied on first resolution (see _resolve_context_length / the threshold_tokens property) + # instead of here. update_model() re-derives the floor for a new window from + # _config_threshold_percent (the raw config value snapshotted above), so switching small -> large + # correctly drops back to the configured value. See #32221. self._config_context_length = config_context_length self._configured_threshold_percent = self.threshold_percent self._resolved_context_length: int | None = None self._threshold_tokens = self._tail_token_budget = self._max_summary_tokens = None self.compression_count = 0 # The init log reports resolved budgets; emit it on first resolution to keep construction non-blocking. + # The "initialized" log reports resolved token budgets, which would force the deferred + # get_model_context_length() probe to run inside __init__ and re-introduce the exact synchronous + # blocking this change removes (#32221). Emit it on first context-length resolution instead so + # construction stays non-blocking on every path (not just quiet). self._log_init_summary = not quiet_mode self._context_probed = False # True after a step-down from context error self.last_prompt_tokens = self.last_completion_tokens = 0 @@ -2227,6 +2370,16 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._pending_request_rough_tokens = 0 # Anti-thrash verdict lives HERE: effectiveness is "prompt under threshold" per the provider's real count, # not "messages shrank"; should_compress() runs twice per turn with mixed measures and would reset it. + # Anti-thrashing verdict, judged HERE because this is the only place that sees the provider's + # real prompt count for the just-compacted conversation. Effectiveness is "did the prompt get + # under the threshold?", not "did the message list shrink?": compaction can only shrink + # messages, while the system prompt and tool schemas are an incompressible floor (with 50+ + # tools, 20-30K tokens — see #14695). When that floor alone meets the threshold, every pass + # shrinks messages by a healthy margin yet leaves the prompt over the line, so the next turn + # compacts again, forever. It must NOT live in should_compress(): that runs twice per turn with + # two different measures (a rough preflight estimate and the real post-response count, #36718), + # and the rough one can dip below the threshold and reset the strike every turn, re-opening the + # loop. Keying on real usage compares like with like and fires exactly once per compaction. if self._verify_compaction_cleared_threshold: if self.last_prompt_tokens >= self.threshold_tokens: self._record_ineffective_compression_verdict(self._ineffective_compression_count + 1) @@ -2351,6 +2504,10 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): # probe by dropping counters to 1 strike (persisted). Deadline is armed lazily and persisted on the row. if self._tripped(): # Wall clock: the deadline is persisted so a rebuilt compressor resumes the SAME window. + # Wall clock, not monotonic: the deadline is persisted on the session row (#100185) so a fresh + # compressor bound to the same session — the gateway rebuilds the AIAgent on every cache + # eviction — resumes the SAME window instead of restarting it. Without that, a blocked messaging + # session never earned its probe and stayed blocked forever. _now = time.time() if self._anti_thrash_recovery_deadline <= 0.0 or ( # Clock jumped backwards: never wait longer than one window from now. @@ -2359,6 +2516,20 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._set_anti_thrash_recovery_deadline(_now + self._ANTI_THRASH_RECOVERY_SECONDS) elif _now >= self._anti_thrash_recovery_deadline: self._set_anti_thrash_recovery_deadline(0.0) + # Anti-thrashing: back off if recent compressions were ineffective. The back-off must not be + # permanent (#14694): the tripped state was judged against the transcript as it existed THEN + # (e.g. a middle region too small to matter), but the conversation keeps growing and can + # accumulate plenty of compressible material later. Without a recovery path the session + # never auto-compacts again and rides into the provider's hard context limit. Recovery is a + # probation probe: after _ANTI_THRASH_RECOVERY_SECONDS of continuous block, allow ONE + # attempt by dropping the tripped counter(s) to 1 strike (persisted, so sibling agents on + # the same session row unblock too). If the probe is ineffective again the very next verdict + # re-trips the guard, so the worst case in the truly-incompressible state is one compaction + # attempt per recovery window — bounded, not thrash. The clock is armed lazily on the first + # BLOCKED evaluation and persisted on the session row (#100185): a fresh process/compressor + # that loads a durable tripped counter (#69872) with no stored deadline starts a full window + # blocked, preserving the restart-must-not-disarm contract (#54923) — but one that loads an + # already-armed deadline resumes that window instead of restarting it. if self._ineffective_compression_count >= 2: self._record_ineffective_compression_verdict(1) if self._fallback_compression_streak >= 2: @@ -2548,6 +2719,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): prune_boundary = self._prune_boundary(result, protect_tail_count, protect_tail_tokens) pruned = self._dedupe_tool_results(result) # Just-loaded / tail-referenced skills keep full skill_view bodies through the ordinary passes. + # Without this, a skill loaded moments before a compaction can be demoted to metadata while the + # model still believes its instructions are in context. See #32106. protected_skills = _collect_protected_skill_names(result, prune_boundary) # Pass 2: summarize old tool results. Pass 3: shrink large tool_call arguments INSIDE the parsed JSON so # the result stays valid; otherwise providers 400 on every turn until the call leaves the window. @@ -2559,6 +2732,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): self._truncate_tool_call_args_at(result, i) # Pass 3.5: retire image payloads inside the protected tail; re-sent embeds otherwise make # compression look ineffective and trip anti-thrash. Newest frames stay live. + # Newest frames stay live for follow-up QA; older ones become placeholders. See #92699. pruned += _retire_stale_tool_result_images(result) if protect_tail_tokens is not None and protect_tail_tokens > 0 and result: pruned += self._pressure_demote_tail( @@ -2645,7 +2819,18 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): object as ``(messages, 0)``. The rearm gate is measured on message bodies only, so it is bypassed (never the reclaim gate) when a provider-billed ``current_tokens`` reading already puts the request over ``threshold_tokens`` (#101889); every no-op taken while over threshold - is logged once per distinct reason.""" + is logged once per distinct reason. + + ``_prune_old_tool_results`` runs all deterministic passes: (1) dedup byte-identical tool results — + keeps the newest full copy and back-references older exact duplicates ANYWHERE in the list + (including the protected tail), so no unique content is ever lost; (2) summarize non-tail tool + results larger than ``min_prune_chars``; (3) truncate oversized tool_call arguments on non-tail + assistant messages; (3.5) retire image payloads on all but the newest ``_MAX_KEEP_TOOL_IMAGES`` + image-bearing tool results — tail-agnostic and lossy by design (#92699). Only pass (2)'s floor is + raised by ``proactive_prune_min_result_chars``; passes (1) and (3) keep their own fixed floors. The + recent-tail protection applies to passes (2) and (3); pass (1) is tail-agnostic by design because + dedup is lossless. + """ if self.proactive_prune_tokens <= 0 or ( current_tokens is not None and current_tokens < self.proactive_prune_tokens ): @@ -2689,6 +2874,7 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): logger.warning("Proactive tool-result prune DB commit failed; keeping the original transcript: %s", exc) return messages, 0 # Shared post-commit stamp site with the in-place commit and micro-compaction sync. + # See #98450. stamp_db_persisted_markers(pruned_msgs) self._proactive_prune_rearm_tokens = next_rearm_tokens # Reclamation just ran: let a future lockout warn again. @@ -2864,6 +3050,10 @@ None recoverable from deterministic fallback. ## Critical Context Summary generation was unavailable, so this is a best-effort deterministic fallback for {len(turns_to_summarize)} compacted message(s).{reason_text}""" # Per-turn truncation cuts [SKILL_PRUNED] markers; re-derive from raw turns and re-inject. + # Ghost-skill defense (#32106): the fallback's per-turn truncation (``_FALLBACK_TURN_MAX_CHARS``) + # routinely cuts [SKILL_PRUNED: ...] markers out of the compacted turns. Re-derive the ghosted + # skills from the raw turn contents and re-inject deterministically, exactly like the LLM-summary + # path. _pruned_names = _collect_ghosted_skill_names(turns_to_summarize) del _pruned_names[_MAX_PRUNED_SKILL_MARKERS:] summary = self._with_summary_prefix(_redact_compaction_text(body.strip())) @@ -3001,6 +3191,10 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb call_kwargs["model"] = self.summary_model # Pinned route (stall fallback) overrides task routing so the retry leaves the stalled backend. call_kwargs.update(_pinned_summary_call_kwargs()) + # Compression is atomic: protect the in-flight summary call from a mid-turn gateway interrupt. + # Without this, an incoming user message aborts the summary and compression falls back to a degraded + # static marker, losing the real handoff (#23975). Re-entrant: a main-model retry (_generate_summary + # recursion) re-enters harmlessly. _aux_call_start = time.monotonic() _latency_info: Dict[str, int] = {"prompt_build_ms": max(0, int((_aux_call_start - prompt_started_at) * 1000))} call_kwargs["latency_info"] = _latency_info @@ -3026,8 +3220,25 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb # Reasoning-field fallback (DeepSeek/Qwen/Kimi put the summary in reasoning_content); capped. content = extract_content_or_reasoning(response, max_reasoning_chars=8000) where = f"(provider={self.provider or 'auto'} model={self.summary_model or self.model})" + # Some OpenAI-compatible proxies (e.g. cmkey.cn, one-api channels) return a well-formed HTTP 200 + # with an empty or whitespace-only ``content`` instead of an error or empty ``choices``. That + # payload passes ``_validate_llm_response`` (a ``message`` exists), so it reaches here and would + # otherwise be stored as a prefix-only summary with no body — silently wiping the compacted turns + # and making the model forget the in-progress task (#11978, #11914). Treat empty content as a + # failure so it routes through the same main-model fallback + cooldown machinery as a transport + # error, rather than replacing real context with an empty summary. if not content.strip(): raise RuntimeError(f"Context compression LLM returned empty content {where}") + # A finish_reason of "length" means the summarizer hit its output token cap mid-generation: the text + # present is PARTIAL. Persisting a partial summary as the compaction checkpoint silently truncates + # the conversation's memory — the cut-off text replaces the real middle turns AND is fed back into + # every subsequent iterative update prompt, compounding the loss across compactions. Treat it as a + # failure so it routes through the same main-model fallback + abort machinery as other degraded + # responses instead of becoming a checkpoint. (Ported from earendil-works/pi#7048.) + # A length stop means the merged rolling summary is partial — persisting it would silently drop the + # tail of the merge and feed the cut-off text into every later micro-compact pass. Leave the + # exchange unabsorbed instead; a later pass retries it. (Same class as _generate_summary's guard; + # pi#7048.) if _response_finish_reason(response) == "length": raise RuntimeError( f"Context compression summary was truncated ({_TRUNCATED_SUMMARY_MARKER}): generation hit the output " @@ -3046,6 +3257,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb # bypass_cooldown: provider-proven overflow gets ONE real attempt while armed. if prompt_started_at < self._summary_failure_cooldown_until and not bypass_cooldown: logger.debug( + # See #100661. "Skipping context summary during cooldown (%.0fs remaining)", self._summary_failure_cooldown_until - prompt_started_at, ) @@ -3076,6 +3288,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb # The summarizer may echo secrets verbatim; redact the output too. summary = _redact_compaction_text(content.strip()) # Restore any [SKILL_PRUNED] marker the summarizer paraphrased away. + # See #32106. summary = _reinject_pruned_skill_markers(summary, _pruned_skill_names) summary = self._ground_historical_task_snapshot(summary, turns_to_summarize) summary = self._augment_summary_lean(summary, turns_to_summarize) @@ -3229,6 +3442,12 @@ Write only the summary body. Do not include any preamble or prefix.""" """Classify a summary-call failure; retry once on the main model (returning its result) or arm a cooldown (None).""" # Only a genuine no-provider RuntimeError gets the long cooldown; empty/invalid-response # RuntimeErrors are transient and must get the main-model retry below first. + # ``call_llm`` raises ``RuntimeError`` for two very different cases: 1. 2. An empty/invalid response + # from a configured provider (``_validate_llm_response`` empty-``choices``/``None``, or our + # empty-``content`` guard above) — a transient/proxy fault that should fall back to the main model + # first, exactly like the transport errors handled below. Only (1) belongs in the long no-provider + # cooldown; (2) and every other exception flow into the generic fallback logic so they get a + # main-model retry before any cooldown. (#11978, #11914) if isinstance(e, RuntimeError) and "no llm provider configured" in str(e).lower(): self._record_compression_failure_cooldown(_SUMMARY_FAILURE_COOLDOWN_SECONDS, "no auxiliary LLM provider configured") self._last_summary_error = "no auxiliary LLM provider configured" @@ -3269,6 +3488,11 @@ Write only the summary body. Do not include any preamble or prefix.""" # Terminal network/empty-content failure after any fallback: flag so compress() ABORTS # and preserves the session; independent of abort_on_summary_failure. if kind.streaming_closed: + # A terminal connection/network failure or empty-content response from a degraded provider (we + # reach this branch only after any main-model fallback has already been tried or is + # unavailable). Flag it so compress() ABORTS and preserves the session unchanged instead of + # destroying the middle window for a placeholder marker — retrying once the provider recovers is + # strictly better than dropping context (#29559, #25585, #94448). self._last_summary_network_failure = True elif kind.truncated: self._last_summary_truncated_failure = True @@ -3494,6 +3718,8 @@ Write only the summary body. Do not include any preamble or prefix.""" """Find handoff summaries inside a compression window.""" n = len(messages) # Clamp: callers may pass end = len(messages)+1. + # Defensive: clamp bounds so a caller passing an out-of-range end (e.g. tail-cut returning + # len(messages)+1 when head_end >= n) cannot trigger IndexError. (#75588) start = max(0, min(start, n)) end = max(start, min(end, n)) return [ @@ -3578,7 +3804,12 @@ Write only the summary body. Do not include any preamble or prefix.""" @staticmethod def _tool_call_id_variants(tc) -> set: - """Return every id variant a result might reference *tc* by (forwards to message_sanitization).""" + """Return every id variant a result might reference *tc* by (forwards to message_sanitization). + + Thin forwarder — the policy owner is ``agent.message_sanitization.tool_call_id_variants``, which + also expands ``response_item_id`` and composite ``call|item`` bridge spellings (#63000), so the + compressor's pairing tolerance matches the pre-call sanitizer's exactly and the two can never drift. + """ from agent.message_sanitization import tool_call_id_variants return set(tool_call_id_variants(tc)) @@ -3648,7 +3879,11 @@ Write only the summary body. Do not include any preamble or prefix.""" return self.protect_first_n def _protect_head_size(self, messages: List[Dict[str, Any]]) -> int: - """Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows.""" + """Head messages to protect: the system prompt (if present) plus the decaying ``protect_first_n`` extra rows. + + The ``protect_first_n`` portion DECAYS after the first compression (see _effective_protect_first_n) + so early user turns don't fossilize across repeated compactions (#11996). + """ head = 1 if messages and messages[0].get("role") == "system" else 0 return head + self._effective_protect_first_n(messages) @@ -3762,6 +3997,16 @@ Write only the summary body. Do not include any preamble or prefix.""" from agent.conversation_compression import _is_real_user_message last_user_idx = -1 + # Find the newest user message that carries at least one image part. We anchor on image-bearing user + # messages (not all user messages) so a plain text follow-up after a big-image turn still strips the + # old image — matching the problem kilocode#9434 set out to solve. + # Newest tool message carrying an image. Tool-result images (``vision_analyze``, + # screenshot-returning tools) accumulate on their own timeline and the user anchor never protects + # the stale ones: a session whose only image-bearing user message is the FIRST one leaves ``anchor + # <= 0`` and strips nothing at all, so twenty tool results keep multi-MB of base64 in every request + # body until the provider answers 413 -- and the 413 handler's recovery compaction lands right back + # here and frees nothing, which is the wedge in #89938. Keep the newest tool image, since that is + # the one the model is reasoning about, and drop every older one wherever it sits. for i in range(len(messages) - 1, -1, -1): msg = messages[i] # _is_real_user_message also rejects metadata-flagged scaffolding @@ -3898,7 +4143,16 @@ Write only the summary body. Do not include any preamble or prefix.""" def _ensure_last_n_user_messages_in_tail( self, messages: List[Dict[str, Any]], cut_idx: int, head_end: int, n: int, ) -> int: - """Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method.""" + """Keep the last N actionable user messages in the tail; n <= 1 delegates to the single-message method. + + Only REAL actionable user turns count toward N — the collector uses the same + ``_is_actionable_user_turn`` / ``_is_synthetic_compression_user_turn`` pair as + ``_find_last_user_message_idx``, so blank platform echoes, compaction handoffs, continuation + markers, and todo-snapshot rows never consume a slot (#69291 bug class). + A user message is already a clean boundary — there is no tool_call/result group that spans across + it, so ``_align_boundary_backward`` is intentionally NOT called. Calling it can pull the cut past + the user message into the preceding assistant(tool_calls)→tool group and split it (#22566). + """ if n <= 1: return self._ensure_last_user_message_in_tail(messages, cut_idx, head_end) @@ -3957,6 +4211,8 @@ Write only the summary body. Do not include any preamble or prefix.""" cut_idx = self._align_boundary_backward(messages, cut_idx) # Latest user message must stay in the tail (active task). Latest assistant reply must stay too; # anchors only walk backward, so chaining is monotonic. + # Ensure the most recent user message is always in the tail so the active task is never lost to + # compression (fixes #10896). cut_idx = self._ensure_last_user_message_in_tail(messages, cut_idx, head_end) cut_idx = self._ensure_last_assistant_message_in_tail(messages, cut_idx, head_end) @@ -4275,6 +4531,10 @@ Write only the summary body. Do not include any preamble or prefix.""" self.compression_count += 1 # Replace historical image payloads with placeholders; multi-MB base64 blobs otherwise # exceed body limits. + # Replace image parts in all compressed messages before the newest image-bearing user turn with a + # short text placeholder. Without this, tail messages keep their original multi-MB base-64 image + # payloads forever, which can push every subsequent API request past the provider's body-size limit + # and wedge the session. Port of Kilo-Org/kilocode#9434. compressed = _strip_historical_media(compressed) # Like-for-like savings: current_tokens includes system prompt/tool schemas, new_estimate is @@ -4300,6 +4560,10 @@ Write only the summary body. Do not include any preamble or prefix.""" # Compaction frees the biggest allocation: hand pages back to the OS (glibc/config-gated, # rate-limited, #70782). debug, not warning: compression must never fail because of a trim. try: + # A successful compaction just freed the largest allocation a long session ever drops (the + # compressed-away message dicts), which makes this the natural point to hand allocator pages + # back to the OS. #76905's trim lifecycle covers the gateway/TUI housekeeping loops but not the + # CLI compression path, so RSS keeps the pre-compaction high-water mark until exit. (#70782) from hermes_cli.mem_trim import trim_memory trim_memory(reason="post-compression") except Exception as exc: @@ -4317,7 +4581,19 @@ Write only the summary body. Do not include any preamble or prefix.""" ) -> List[Dict[str, Any]]: """Summarize the middle turns: prune tool results and blank echoes (survives an abort), protect head and a token-budget tail, summarize, clean orphaned tool pairs. ``force`` clears the failure cooldown and bypasses - the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown.""" + the feasibility skip; ``bypass_cooldown`` runs the summary LLM without clearing the cooldown. + + Args: focus_topic: Optional focus string for guided compression. When provided, the summariser will + prioritise preserving information related to this topic and be more aggressive about compressing + everything else. Inspired by Claude Code's ``/compact``. force: If True, clear any active + summary-failure cooldown before running so a manual ``/compress`` can retry immediately after an + auto-compression abort, and bypass the pre-LLM feasibility skip so an explicit user request always + exercises the full summary path. Auto-compress callers pass False. memory_context: Optional + provider-supplied context to preserve in the summary prompt. Whitespace-only values are ignored. + bypass_cooldown: If True, run the summary LLM even while the summary-failure cooldown is armed, + WITHOUT clearing it (#100661). Set by provider-proven overflow recovery, which is already bounded by + the caller's attempt budget. + """ telemetry = self._begin_compress_attempt(current_tokens, force) n_messages = len(messages) # Only need head + 3 tail messages minimum (token budget decides the real tail size) diff --git a/agent/context_engine.py b/agent/context_engine.py index ef9c6f4c93..54c10d5335 100644 --- a/agent/context_engine.py +++ b/agent/context_engine.py @@ -62,6 +62,8 @@ class ContextEngine(ABC): # Compaction parameters (read by run_agent.py for preflight). protect_first_n counts # non-system head messages kept verbatim IN ADDITION to the always-protected system # prompt (3 keeps the historical head shape). + # These control the preflight compression check. Subclasses may override via __init__ or property; + # defaults are sensible for most engines. See #13754. threshold_percent: float = 0.75 protect_first_n: int = 3 protect_last_n: int = 6 @@ -183,6 +185,16 @@ class ContextEngine(ABC): def on_session_reset(self) -> None: """/new or /reset: reset per-session state (default: counters and token tracking).""" + # Reset cross-call calibration state captured under the PREVIOUS model. These fields encode "the + # provider proved this prompt fit" / "preflight can be deferred" decisions that are only valid for + # the model that produced them. Carrying them across a switch to a smaller-context model would let + # should_defer_preflight_to_real_usage() suppress a preflight compression the new model actually + # needs — the exact oversized-send-after-switch failure in #23767. The new model's first response + # repopulates them via update_from_response(). Setting last_prompt_tokens to 0 (NOT -1) is + # deliberate: 0 is the documented "no real usage yet -> use the rough estimate" state, so the post- + # response should_compress path falls back to estimate_request_tokens_rough rather than skipping + # compression. -1 is a different sentinel (#36718, "compression just ran, await real usage") and + # must not be set here. self.last_prompt_tokens = 0 self.last_completion_tokens = 0 self.last_total_tokens = 0 diff --git a/agent/context_references.py b/agent/context_references.py index 5b12083b9c..b20da47720 100644 --- a/agent/context_references.py +++ b/agent/context_references.py @@ -20,6 +20,8 @@ from hermes_cli.sizefmt import format_bytes # ── Plugin context-reference provider API ──────────────────────────────────── +# --------------------------------------------------------------------------- Plugin context-reference +# provider API (Issue #26193) --------------------------------------------------------------------------- BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"}) _context_reference_providers: dict[str, "ContextReferenceProvider"] = {} diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index c807ae85f0..85f38efdf3 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -53,6 +53,8 @@ _TERMINAL_COMPRESSION_PROVENANCES = frozenset( # Split failures are usually transient lease/DB conditions, so use the FIRST # timeout-ladder rung (60s), not the 600s summary-provider cooldown. +# Cooldown armed when a compression SPLIT fails (session_split_failed / rotation rollback, #97948 symptom +# B). _SPLIT_FAILURE_COOLDOWN_SECONDS = 60 # Marker tui_gateway/server.py::_status_update matches to tag kind="compacting" for drivers' "Summarizing…" UI. Keep @@ -108,6 +110,11 @@ COMPRESSION_RETRY_CONTEXT_REDUCED_STATUS_TEMPLATE = ( # FAILURE-class notice: compression blocked, so the session grows until the provider limit kills it. Must stay visible # on gateways: never add it to ROUTINE_COMPRESSION_STATUS_SAMPLES or _TELEGRAM_NOISY_STATUS_RE. +# FAILURE-CLASS notice — a deliberate carve-out from routine-compression silence (#16775 class): the context +# is over the compression threshold but compression is blocked (summary-LLM cooldown / anti-thrash breaker), +# so the session will keep growing until the hard provider token limit kills it. Do NOT add it to +# ROUTINE_COMPRESSION_STATUS_SAMPLES or the gateway noise regex (_TELEGRAM_NOISY_STATUS_RE); it is pinned +# un-swallowed in tests/gateway/test_telegram_noise_filter.py::VISIBLE_COMPRESSION_MESSAGES. CONTEXT_OVERFLOW_BLOCKED_WARNING_TEMPLATE = ( "⚠ Context is over the compression threshold (~{tokens:,} tokens >= {threshold:,}) " "but compression is currently blocked ({reason}). The model may stop responding. Run /new to start a fresh " @@ -188,6 +195,19 @@ def _snapshot_compressor_attempt_state(compressor: Any) -> dict[str, Any]: # Attempt ownership: stall-fallback detaches a timed-out worker and reuses the compressor, so its late unwind could # restore a stale snapshot or clear the fallback's cancel check. Generation guards ATTRIBUTE writes; fence, COMMITs. +# --------------------------------------------------------------------------- Attempt ownership (#96634 +# follow-up). The stall-fallback path deliberately DETACHES a timed-out primary worker (fence cancel wins; +# the future stays on the shared pool) and immediately starts a fallback attempt against the SAME +# ContextCompressor. Two races follow from that overlap: 1. The late primary's unwind still calls +# _restore_compressor_attempt_state with the PRIMARY's pre-attempt snapshot. Landing after the fallback's +# commit, it rolls _previous_summary / cooldown / provenance / telemetry back to pre-primary values — +# silently discarding fallback-owned state. 2. _compression_cancelled_check is one shared attribute: the +# late primary's ``finally`` clears the callback the fallback just installed, so the fallback's F4 +# cancellation consult reads None. Both are fixed with a monotonic per-compressor attempt generation, +# claimed under one module lock. Restores and callback set/clear are keyed to the claiming generation and +# no-op when a newer attempt owns the compressor. The commit fence still owns COMMIT admission; the +# generation owns compressor-ATTRIBUTE writes — two different boundaries. +# --------------------------------------------------------------------------- _COMPRESSOR_ATTEMPT_LOCK = threading.Lock() @@ -268,7 +288,11 @@ def _restore_compressor_attempt_state( ) -> None: """Restore the per-attempt snapshot after a pre-commit hard cancel. A restore stamped with a stale ``attempt_generation`` no-ops so a timed-out primary's late unwind cannot - roll back state owned by the fallback attempt.""" + roll back state owned by the fallback attempt. + + ``attempt_generation`` (when provided) is the claim the calling attempt took via + :func:`_claim_compressor_attempt`. See #96634. + """ if attempt_generation is not None and not _compressor_attempt_is_current(compressor, attempt_generation): logger.warning( "Skipping stale compressor attempt-state restore: attempt " @@ -348,10 +372,25 @@ class CompressionCommitFence: self._cancelled = False self._commit_started = False # Readable WITHOUT the lock (begin_commit holds it until finish_commit): hosts see a hung commit. + # Lock-free commit-phase marker (#76354 review F1). ``begin_commit`` RETAINS ``self._lock`` until + # ``finish_commit``, so any host-side observation that needs the lock (``try_cancel_before_commit``) + # blocks/space-outs for the whole commit. This Event is set inside ``begin_commit`` while the lock + # is held but is READABLE WITHOUT the lock, so a host can observe "a commit was admitted and may be + # in flight" even while the commit itself is hung — which is exactly when the overrun warning must + # be able to fire. self._commit_phase = threading.Event() # Set on ANY host unwind without the fence lock so FUTURE commits are blocked; bool store is atomic. + # Lock-free admission revocation (#76354 review F2). Set by :meth:`revoke_commit_admission` on ANY + # host unwind (KeyboardInterrupt, cancellation, unexpected exception) without touching the fence + # lock, so a host that cannot afford to block behind an in-flight commit can still guarantee no + # FUTURE commit is admitted. self._admission_revoked = False # Holder-scoped release published by the worker once it owns the durable lock (no ABA on a NEW holder). + # Holder-qualified durable-lock release hook (#76354 review F4; transplanted from PR #71569 by + # @ciabata-git). The worker publishes an idempotent, holder-scoped release callable once it owns the + # durable compression lock; a timed-out host invokes it to free the lease without racing a NEW + # holder (DB release is holder-qualified, so a stale release can never delete a replacement's row — + # no ABA). self._lock_release_guard = threading.Lock() self._cancelled_lock_release: Optional[Callable[[], None]] = None self._cancelled_lock_release_requested = False @@ -389,7 +428,14 @@ class CompressionCommitFence: @property def deadline_monotonic(self) -> float | None: - """Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting.""" + """Armed deadline (absolute monotonic); the worker's stream consumer stops when the host stops waiting. + + :meth:`set_total_ceiling_seconds` documents this deadline as "shared by the host and worker", but + until #99692 only the host could read it — ``deadline_exceeded`` answers "is it past?" for a caller + that is already polling, which is useless to a worker blocked inside a provider stream. Publishing + the instant itself lets the worker's stream consumer stop at exactly the moment the host stops + waiting (see ``auxiliary_client.aux_stream_deadline``). + """ return self._deadline def seconds_since_progress(self) -> float: @@ -454,7 +500,14 @@ class CompressionCommitFence: self._retain_cancelled_lock_until_worker_done = True def mark_commit_watermark_fenced(self) -> None: - """Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission.""" + """Record a watermark-bounded commit (later rows survive as tail); a detached worker may keep admission. + + Called by the compression worker right after it captures ``get_active_message_watermark()`` under + the durable compression lock (#75316/#87484). A watermark-fenced commit archives ONLY rows at or + below the watermark; rows appended later — e.g. the user turn the host released at the turn-hold + boundary (#97963) — are cloned as live concurrent tail. That is exactly the property a host needs + before letting a detached worker keep its commit admission. + """ self._commit_watermark_fenced = True @property @@ -481,6 +534,11 @@ class CompressionCommitFence: # ── Holder-qualified durable-lease cancellation: release is DELETE WHERE # holder = ?, so a stale release can never free a NEW holder's lease (no ABA). + # ── Holder-qualified durable-lease cancellation (#76354 F4) ────────── Transplanted from PR #71569 + # (@ciabata-git): the worker publishes an idempotent, holder-scoped release hook once it owns the + # durable compression lock, and the host invokes it after winning cancellation. ABA safety comes from + # SessionDB.release_compression_lock being holder-qualified (DELETE ... WHERE holder = ?), so a stale + # release can never free a NEW holder's lease. def begin_lock_setup(self) -> bool: """Hold the fence across lock acquisition + release-hook publication so a timeout cannot win between.""" self._lock.acquire() @@ -524,6 +582,9 @@ DEFAULT_CONTEXT_TIMEOUT_SECONDS = 120.0 DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS = 600.0 # Unlike explicit_interrupt, a /stop after the stall window arms the durable backoff (no automatic re-entry). +# Distinct from ``explicit_interrupt``: a /stop that arrived after the summary stream had already crossed +# the no-progress stall window (#96775). Ordinary early /stop stays cooldown-neutral; this class arms the +# durable backoff so the next automatic turn does not re-enter the same stalled strategy. STALL_INTERRUPTED_FAILURE_CLASS = "stall_interrupted" # Daemon pool so a fence-cancelled hung worker cannot block interpreter exit; never shut down per call. @@ -535,6 +596,8 @@ _COMMIT_OVERRUN_WAIT_SLICE_SECONDS = 30.0 # A worker exiting within the grace proves no provider call is in flight, so its lease may be released even # on the total-ceiling path; one that doesn't exit is orphaned behind the poison fence and keeps its lease. +# Bounded grace given to a fence-cancelled compression worker to actually exit before the host moves on +# (#97488). _CANCELLED_WORKER_TEARDOWN_GRACE_SECONDS = 5.0 @@ -561,6 +624,16 @@ def _join_cancelled_worker(future: Any, grace_seconds: float) -> bool: # The executor queue is unbounded and a queued job would run stale, so admission is capped at the worker # count (fail fast, continue uncompressed). Slots free via done-callback; a never-returning worker loses one. +# Bounded admission for the shared compress-timeout pool (#76354 review F6). The stdlib executor queue is +# unbounded: with all four workers wedged in hung summaries, a fifth compression would queue silently, wait +# out its whole timeout without ever starting, and remain eligible to run as a stale job whenever a worker +# recovered. Admission is therefore capped at the worker count — when every worker slot is occupied (running +# OR admitted-not-started) submission FAILS FAST and the caller continues without compression. Recovery +# contract when all workers are wedged: new compressions fail fast (no queue growth, conversation continues +# uncompressed, a warning is logged each attempt); wedged workers are fence-cancelled so they cannot publish +# anything when they eventually return, and each recovery frees its admission slot via the future +# done-callback, restoring normal service. If a worker NEVER returns, its slot is lost for the process +# lifetime — bounded, observable degradation instead of an unbounded stale-job queue. _COMPRESS_EXECUTOR_MAX_WORKERS = 4 _compress_admission_lock = threading.Lock() _compress_admitted_count = 0 @@ -632,7 +705,12 @@ def compression_attempt_stalled( ) -> bool: """Return whether a pre-commit cancel landed after the stall window. An early ``/stop`` stays cooldown-neutral; an interrupt after the inactivity budget counts as a stall so - the next automatic turn does not blindly retry.""" + the next automatic turn does not blindly retry. + + When the fence (or, without a fence, the attempt clock) has already sat idle for the configured + compression inactivity budget, the interrupt is a stalled attempt — the same condition the host timeout + uses — and the next automatic turn must not blindly retry that strategy (#96775). + """ idle = idle_timeout_seconds if idle is None: idle, _ceiling = resolve_context_compression_timeouts() @@ -674,6 +752,8 @@ def _record_stall_interrupted_backoff( if not compression_attempt_stalled(commit_fence=commit_fence, started_at=started_at): return False compressor = getattr(agent, "context_compressor", None) + # Same timeout cooldown ladder as summary-LLM timeouts (#62452): avoid re-burning the full idle budget + # every turn. record = getattr(compressor, "record_timeout_failure", None) if not callable(record): return False @@ -741,7 +821,17 @@ def _retry_compression_on_fallback_chain( """Re-run an aborted compression once with the summary route pinned. Returns ``(messages, system_prompt)`` on real compression, else ``None`` and the caller degrades as before. The entry's ``timeout`` sets the idle window. Re-runs the whole worker, so pre-compression - callbacks must be idempotent.""" + callbacks must be idempotent. + + The retry is bounded the same way the primary was: silence for one idle window ends it, while a fallback + that is streaming keeps its ceiling. The entry's own ``timeout`` (when declared) sets that idle window, + so a fallback tuned for a slower-but-healthy backend is not held to a deadline the stalled primary + defined (#62452 semantics, applied to the stall path). + Known limitation (accepted, #96634 review): the retry re-runs the COMPLETE worker, which repeats + memory/plugin pre-compression callbacks. Built-in callbacks are idempotent (re-reads and overwrites of + attempt-scoped state); third-party plugin callbacks are advised to be. Splitting the worker to resume + mid-pipeline would couple this path to every host's callback ordering — deliberately out of scope. + """ # An explicit stop is not a stalled route. The retry worker would abort on # the same event anyway, but starting one at all makes /stop look ignored. hard_cancel = getattr(telemetry_agent, "_hard_interrupt_requested", None) @@ -930,6 +1020,9 @@ def run_compress_context_with_progress_timeout( executor = _get_compress_timeout_executor() # Refuse rather than queue when the pool is full: a queued job would wait out # its budget unstarted and run stale later. Skip compression this cycle. + # A queued job would silently wait out its whole budget without starting and stay eligible to run as a + # stale cancelled job when a worker recovers. Fail fast: continue without compression this cycle. See + # #76354. if not _try_admit_compression_job(): logger.warning( "Context compression pool saturated (%d workers busy) — refusing new compression this cycle and continuing without " @@ -980,6 +1073,14 @@ def run_compress_context_with_progress_timeout( # cancel() is a no-op for a running worker (fence handles that path). future.cancel() total_exhausted = time.monotonic() - wait_started >= ceiling or fence.deadline_exceeded + # #97488 teardown (total-ceiling path only): give the cancelled worker a bounded grace to actually + # exit before this host moves on. The worker checks the poison fence between provider phases, so a + # cooperative worker exits quickly; an uninterruptible provider call is orphaned behind the fence + # after the grace elapses (its late result is discarded and cannot touch session state). The + # idle-stall path intentionally skips the join: its worker is by definition silent/hung, the + # stall-fallback retry below needs a prompt host return (pinned by the #76354 S3 latency contract), + # and the fence poison + attempt-generation supersession already protect state against its late + # unwind. if total_exhausted: # A total-ceiling candidate may be unwinding a healthy provider call; keep its # lease until it exits so no other attempt overlaps the unchanged source. @@ -999,6 +1100,9 @@ def run_compress_context_with_progress_timeout( handled_exit = True _release_cancelled_worker(future, fence, total_exhausted=total_exhausted, ceiling=ceiling) waited = time.monotonic() - wait_started + # #76354 S3 analogue for this wait: charge the idle budget from the LAST PROGRESS event, not from + # the start of this wait slice. Waiting a full ``idle`` after progress that landed early in the + # previous slice would allow silence to approach 2x the budget. since_progress = fence.seconds_since_progress() # Lease is free, so run the fallback BEFORE on_timeout: that callback records # the summary-failure cooldown, which would no-op the retry's summary call. @@ -1222,7 +1326,15 @@ def compression_blocked_transiently(agent: Any) -> bool: """Type-pinned read of the transient-block signal. Set when an automatic pass no-ops on a TRANSIENT guard (summary-failure cooldown or structural backoff). Consumers must defer, not count it toward ``compression_exhausted``, or an overflow auto-reset wipes a - session that was merely cooling down. The permanent ``ineffective`` breaker never sets it.""" + session that was merely cooling down. The permanent ``ineffective`` breaker never sets it. + + See #97488. + Consumers (the overflow-recovery loops in ``conversation_loop``) must treat such a no-op as a temporary + defer, NOT as evidence the session is incompressible: counting it toward ``compression_exhausted`` lets + a real upstream ``context_length_exceeded`` auto-reset (wipe) a session whose compression was merely + cooling down (#97488). The permanent ``ineffective`` breaker intentionally does NOT set this signal — a + genuinely incompressible session must still be able to exhaust. + """ _sig = getattr(agent, "_compression_blocked_transient", None) return isinstance(_sig, str) and bool(_sig) @@ -1265,7 +1377,14 @@ def _adopt_live_compression_child( ) -> Optional[List[Dict[str, Any]]]: """Move a stale compression contender onto the live continuation tip. Resolve and load first, then mutate the agent, so ambiguous lineage or an unreadable handoff fails closed. - Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live.""" + Uses the transitive ``get_compression_tip`` walk; a tip is adopted only while its row is still live. + + Resolution uses the canonical transitive walk ``get_compression_tip`` so a lineage with >=2 compression + hops (root -> mid -> tip) recovers to the live tip — the depth-1 ``find_live_compression_child`` lookup + this used to call finds no live *direct* child in that shape and skipped recovery (#82001). The tip walk + returns the input id when no continuation exists, and a resolved tip is adopted only while its row is + still live — both cases fail closed exactly as before. + """ resolver = getattr(type(session_db), "get_compression_tip", None) row_getter = getattr(type(session_db), "get_session", None) loader = getattr(type(session_db), "get_messages_as_conversation", None) @@ -1591,6 +1710,12 @@ def _lower_threshold_to_aux_context( safe_pct = int((aux_context / main_ctx) * 100) if main_ctx else 50 # Mirror the compressor's threshold math (percent floor, output reservation, 64K floor): a suggestion it # would override is silently ignored and this warning reappears every session. External engines: keep it plain. + # The "lower the threshold" suggestion must survive the built-in trigger recomputation (#67422): + # _effective_threshold_percent() raises sub-75% values back up for main windows under 512K, and + # _compute_threshold_tokens() further applies the output-token reservation, the 64K floor, and the + # degenerate-window guard. Recommending a value those would override is silently ignored and this + # warning would reappear every session — so mirror the compressor's own math and only offer the option + # when the recomputed trigger actually fits the auxiliary model's context. from agent.context_compressor import ContextCompressor as _CC recomputed_threshold = None if main_ctx and isinstance(compressor, _CC): @@ -1979,6 +2104,12 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list) """Preserve human intent, not merely a synthetic user-role placeholder.""" if any(_is_real_user_message(message) for message in compressed) or _compressed_has_busy_steer(compressed): return "already_present" + # Post-commit contract (#98450, mirrors _sync_micro_compact_to_db): archive_and_compact just durably + # wrote every dict in `compressed` as the new active set, but compress() returned marker-swept COPIES + # (_strip_persistence_markers, #57491). These exact dict instances become the live message list the + # caller keeps, so without the stamp the next _persist_session → _flush_messages_to_session_db_unlocked + # walk treats the whole compacted transcript as unpersisted and re-INSERTs it — the live set doubles on + # every compaction (~58K → ~512K tokens in production). from agent.context_compressor import ( _INFLIGHT_REPLAY_MERGED_KEY, COMPRESSION_CONTINUATION_USER_CONTENT, _fresh_compaction_message_copy, ) @@ -1987,6 +2118,9 @@ def _ensure_compressed_has_user_turn(original_messages: list, compressed: list) return "already_present" # One reversed scan over BOTH kinds: scanning steer then user would let an older # consumed steer outrank a newer real user request and replay it. + # One reversed positional scan: the anchor is whichever intent-bearing row is LAST in the original + # transcript — a real ``role=user`` turn or a steer marker riding inside a ``role=tool`` result. See + # #100053. for message in reversed(original_messages): if _is_real_user_message(message): return _insert_real_user_anchor(compressed, _fresh_compaction_message_copy(message)) @@ -2421,6 +2555,13 @@ def _adopt_grown_durable_parent(agent: Any, lease: _CompressionLease, messages: return None # In-memory carries this turn's un-persisted user tail; flush it via the normal # rotation-boundary path before adopting, else skip adoption (would drop input). + # The in-memory transcript carries the CURRENT turn's un-persisted user tail (anchored by + # _persist_user_message_idx) that the durable snapshot read above does not contain yet. Flush that tail + # through the normal rotation-boundary path (conversation_history = the already-durable prefix, #68196 + # boundary) BEFORE adopting, then re-read the durable parent so the adopted snapshot includes the live + # input. If the flush fails (or the anchor is unknown), skip adoption entirely: replacing the in-memory + # transcript with a snapshot that lacks the user's input would silently drop it from the summarized and + # rotated history (#adopt-live-tail). _preflush_idx = getattr(agent, "_persist_user_message_idx", None) # No un-persisted tail means the transcript is fully durable: adopting the longer parent cannot drop input. _preflush_ok = True @@ -2523,6 +2664,9 @@ def _run_summary_dispatch( # A LATE successful summary must not undo the host's timeout cooldown: the # compressor checks cancellation before clearing; removed in finally (no leak). if commit_fence is not None: + # Install a cancellation check the compressor consults BEFORE clearing the failure cooldown; removed + # in the finally below so it cannot leak into later attempts (e.g. a manual /compress force-clear). + # See #76354. _install_compression_cancelled_check( agent.context_compressor, lambda: commit_fence.is_cancelled, attempt_generation ) @@ -2594,11 +2738,20 @@ def _fold_todo_snapshot(agent: Any, compressed: list) -> None: if todo_snapshot: # If this boundary pruned skill bodies, the policy behind the todos is gone: # add a reload notice after TODO_INJECTION_HEADER so both strip together. + # Retention parity (#84718): the snapshot below re-injects the imperative verbatim. If this same + # boundary pruned skill bodies to [SKILL_PRUNED: ...] markers, the policy that governed those tasks + # is gone — couple a reload instruction to the snapshot so the imperative never crosses the boundary + # alone. _reload_notice = _pruned_skill_reload_notice(compressed) if _reload_notice: todo_snapshot = f"{todo_snapshot}\n\n{_reload_notice}" # Fold the snapshot into a trailing REAL user msg (no synthetic user/user pair); # strip old snapshots first. Scaffolding tails must not absorb it (provenance). + # Any snapshot merged at an earlier boundary is stripped first so repeated compactions refresh + # rather than accumulate todo state (#26981). Scaffolding tails (continuation marker, summary + # handoff, a bare stale snapshot row) must never absorb the snapshot: merging would upgrade them to + # "real user" evidence and break zero-user provenance (#69292), so those keep the flagged standalone + # append and the real-user preservation pass continues to see todo scaffolding, not human intent. from agent.context_compressor import _append_text_to_content merged = False _tail = compressed[-1] if compressed and isinstance(compressed[-1], dict) else None @@ -2630,6 +2783,12 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str: # Refresh tool schemas at the commit boundary: forever-sessions never restart, # so config reaches agent.tools here. Keep list identity if byte-equal (cache). try: + # Refresh dynamic tool schemas at the same admitted-commit boundary that rebuilds the system prompt + # (maintainer-directed, #95681 arc): forever-sessions (Bot Mode chats, gateway channels) never + # restart, so compaction is the ONLY point where a config change — image model swap, delegation + # depth, code_execution mode — can reach agent.tools. The prompt cache is already broken here, so + # the refresh is free; when nothing changed the snapshot is byte-equal and we keep the existing list + # object (identity matters to provider-side tool-block caching on some backends). _refresh_agent_tool_definitions(agent) except Exception: # noqa: BLE001 logger.warning( @@ -2638,6 +2797,13 @@ def _rebuild_system_prompt_at_boundary(agent: Any, system_message: str) -> str: # ALWAYS rebuild the prompt here: keeping old bytes meant prompt-builder changes # never reached long sessions. Equal bytes keep KV; preserve object identity. + # ALWAYS rebuild the prompt at the admitted-commit boundary (maintainer-directed, #95681 arc). The + # previous "keep-prompt" containment branch put the OLD bytes back whenever the reloaded memory blocks + # were already embedded — which meant prompt-builder changes (guidance diets, new blocks, renames) NEVER + # reached a long-lived session. The cache argument for keeping bytes was hollow: when nothing changed, + # the rebuild is byte-identical and local KV prefixes survive on equality; when something changed, the + # cache was stale by definition and propagation is the point. Preserve OBJECT identity on byte-equality + # for backends that key on it. rebuilt_system_prompt = agent._build_system_prompt(system_message) if cached_system_prompt is not None and rebuilt_system_prompt == cached_system_prompt: new_system_prompt = agent._cached_system_prompt = cached_system_prompt @@ -2662,6 +2828,13 @@ def _salvage_or_refuse_grown_transcript( Compares like-for-like rough estimates; on growth tries one mechanical salvage pass, else treats the attempt as a refused no-op. Returns ``(compressed, None)`` to proceed or ``(None, prompt)`` when refused (caller releases the lease).""" + # Anti-growth guard at the COMMIT SITE: never persist a compression that makes the transcript larger + # (observed: 379K -> 687K when the generated summary plus retained reasoning exceeded what it replaced). + # Compare like-for-like (both rough estimates of the same message shape) so an "actual vs estimate" + # measurement mismatch cannot produce a false verdict. The gateway has a rotation-path-only guard + # (#83339), but in-place compaction commits inside this method via archive_and_compact — before the + # gateway can inspect the result — so the guard must live here to protect both paths. On growth, treat + # the attempt as a no-op: the original transcript stays untouched and durable. _rough_in = estimate_messages_tokens_rough(messages) _rough_out = estimate_messages_tokens_rough(compressed) if _rough_out > _rough_in: @@ -2699,6 +2872,9 @@ def _salvage_or_refuse_grown_transcript( # Count the refusal as an ineffective-compaction strike so the anti-thrash # breaker latches; otherwise auto-compress retries the same summary every turn. with _swallow('could not record rejected-compaction strike', exc_info=True): + # Without this, the unchanged transcript stays over the compression threshold and automatic + # compression retries the identical summary request on every turn (#88568). Manual /compress + # keeps bypassing the latch (force=True skips the guards). agent.context_compressor.record_rejected_compaction() _restore_prune_rearm_tokens(agent.context_compressor, attempt_snapshot) return None, _existing_sp @@ -2726,6 +2902,9 @@ def _carry_session_state_to_child(agent: Any, old_session_id: str, old_title: An transfer clears the ancestor's row, then restored so an inherited auto-title stays upgradeable. """ with _swallow('Could not migrate goal on compression: %s'): + # Carry a persistent /goal onto the continuation session. Compression mints a fresh child id; + # load_goal does a flat per-session lookup with no parent walk, so without this an active goal + # silently dies at the boundary (#33618). from hermes_cli.goals import migrate_goal_to_session migrate_goal_to_session(old_session_id, agent.session_id, reason="compression") with _swallow('Could not migrate heartbeat on compression: %s'): @@ -2980,6 +3159,10 @@ def _candidate_rejected( # Compare semantic state, not identity: engines may return an equal copy or # mutate the live list. ``==`` first (subclass __eq__), then marker-insensitive. + # Neither case may rotate or rewrite the session. The raw ``==`` leg runs FIRST so a list subclass + # returned by an engine keeps its ``__eq__`` semantics (tests seam on this); the marker-insensitive leg + # (#92231) then covers the cold-resume shape where the stamped snapshot differs from the marker-swept + # compress() output only by ``_db_persisted``. if compressed == messages_before_compression or ( _strip_marker_for_comparison(compressed) == _strip_marker_for_comparison(messages_before_compression) ): @@ -3095,6 +3278,19 @@ def _commit_compaction( agent._last_flushed_db_idx = 0 else: # Bind old_session_id first: it is the rollback key in the handler below. + # ── Rotation (legacy): end this session, fork a continuation ─ Flush any un-persisted + # current-turn messages to the OLD session before ending it, so they survive in the + # preserved parent transcript (#47202). (In-place skips this — see above.) Pass the + # already-durable prefix as conversation_history so the flush skips it by identity (#68196). + # Preflight compression runs BEFORE the normal turn flush has stamped the cold-resumed + # history dicts with _DB_PERSISTED_MARKER, so without a boundary + # _flush_messages_to_session_db treats every restored row as new and re-appends the whole + # transcript to the parent. turn_context anchors _persist_user_message_idx at the + # current-turn user message before preflight runs, so messages[:idx] is exactly the + # persisted prefix; only the current turn's new messages get written. Bound to + # old_session_id, hoisted above the flush: the ``except`` handler below keys its in-memory + # rollback off this name, so anything that fails from here on rolls the transcript back + # instead of leaving the failed attempt's compacted snapshot in place. old_session_id = agent.session_id _publish_rotated_compaction( agent, messages, compressed, new_system_prompt=new_system_prompt, lease=lease, @@ -3117,6 +3313,23 @@ def _commit_compaction( ): if rotation_rollback: old_session_id = None + # In-place sibling of the rotation rollback above (#99477). archive_and_compact() is atomic, + # so a raise before it returned means EVERY pre-compaction row is still ``active = 1`` in + # state.db — nothing was archived and the compacted set was never inserted. But + # ``compressed`` is the marker-swept output of compress() (_strip_persistence_markers, + # #57491) and the post-commit ``stamp_db_persisted_markers`` never ran, so handing it back + # makes the next append-only flush treat the whole compacted transcript as new and INSERT it + # ON TOP of the rows it was supposed to replace. The active set then holds the summary AND + # the turns it summarized; the next resume reloads both, the token count goes UP, preflight + # fires again, and each failed attempt appends another copy of the protected head + tail + # (#99477: ~15 real turns stored as 3,814 rows, the first user message repeated 893 times). + # Gate on ``split_status`` rather than ``compacted_in_place``: it is assigned on the + # statement immediately after the atomic commit returns, so a committed compaction can never + # be rolled back into a live/durable mismatch of the opposite sign. The deepcopy carries + # each row's _DB_PERSISTED_MARKER from the pre-compression snapshot, so the restored + # transcript is correctly skipped by the flush, and replacing every dict breaks + # _db_flush_scan_prefix identity (same reasoning as the rotation branch — no explicit clear + # needed). messages[:] = copy.deepcopy(messages_before_compression) compressed = messages made_progress = False @@ -3134,6 +3347,7 @@ def _commit_compaction( # Arm the failure cooldown so the next turn can't rerun the doomed compression; # try/except so a stub compressor can't mask the original error in this handler. with _swallow('could not record split-failure cooldown', exc_info=True): + # See #97948. agent.context_compressor._record_compression_failure_cooldown( _SPLIT_FAILURE_COOLDOWN_SECONDS, f"session_split_failed: {e}" ) @@ -3270,6 +3484,13 @@ def _begin_compression_attempt(agent: Any, *, force: bool, defer_notification: b agent._last_compression_attempt_recorded = True agent._last_compression_attempt_in_place = None agent._compression_skipped_due_to_lock = None + # Clear the lock-skip signal at the VERY TOP, before the codex route and the breaker gates below can + # early-return (per-attempt state rule, #58630/#69853). A stale ``True``/holder value from a prior + # lock-skip must never make a later breaker/codex no-op look like lock contention to the automatic-path + # consumers (compression_deferred, #49874) — the second clear before lock acquisition below stays for + # the same reason it was added in #69870 and is simply idempotent now. + # Transient-block signal (#97488): cleared with the same per-attempt rule; set by the breaker gates + # below when a TRANSIENT guard (cooldown / structural backoff) no-ops this pass. agent._compression_blocked_transient = None started_at = time.monotonic() attempt_id = uuid.uuid4().hex @@ -3327,7 +3548,24 @@ def compress_context( """Compress conversation context and split the session in SQLite. ``force`` (manual /compress) clears the summary-failure cooldown; ``bypass_cooldown`` (provider-proven overflow) skips it once, breakers still apply. ``commit_fence`` stops a timed-out worker mutating session - state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split.""" + state. Returns ``(messages, system_prompt)``; on abort input is unchanged, NOT split. + + Args: agent: The owning :class:`AIAgent`. messages: Current message history (will be summarised). + system_message: Current system prompt; used when compression needs a rebuilt cached prompt. + approx_tokens: Pre-compression token estimate, logged for ops. task_id: Tool task scope (used for + clearing file-read dedup state). focus_topic: Optional focus string for guided compression — the + summariser will prioritise preserving information related to this topic. Inspired by Claude Code's + ``/compact ``. force: If True, bypass any active summary-failure cooldown. Set by the manual + ``/compress`` slash command so users can retry immediately after an auto-compress abort. Auto-compress + callers use the default ``False``. bypass_cooldown: If True, the automatic breaker gates ignore ONLY the + summary-failure cooldown for this attempt (#100661). Set by the provider-proven overflow recovery path: + the provider already rejected the request, so deferring until the cooldown lapses wedges the session. + Unlike ``force`` it does not clear the cooldown, and the ineffective/structural breakers still apply; a + failed attempt records its cooldown normally. defer_context_engine_notification: Delay the existing + context-engine hook until a manual host commits its outer history transaction. commit_fence: Optional + cooperative fence for executor callers that may time out. It prevents a late worker from mutating + session state after its caller has moved on. + """ attempt = _begin_compression_attempt(agent, force=force, defer_notification=defer_context_engine_notification) # Codex owns the real thread; route compaction to its own compact (config @@ -3395,6 +3633,11 @@ def compress_context( # Interrupts/redirects must not tear a summary in half. Use the explicit stop # Event (message fields race) + fence timeout so pool slots free promptly. + # Explicit stop surfaces set a separate Event atomically; never infer cause from the racy message + # fields. A host timeout also cancels the attempt's commit fence. Feed BOTH into the protected + # auxiliary-call seam so the compression owner unwinds promptly while an isolated provider stream + # finishes or closes in its daemon worker. Otherwise four timed-out streams retain all four shared + # compression-pool slots until the auxiliary stream's longer absolute ceiling expires. See #23975. _hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None) phase = _run_summary_phase( agent, messages, lease=lease, in_place=in_place, checkpoint_required=checkpoint_required, diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 898fa3a52d..3e0b0c7a94 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -54,6 +54,14 @@ from hermes_logging import set_session_context # patch them here, so they must stay bound in this namespace. from agent.conversation_compression import conversation_history_after_compression # noqa: F401 from agent.model_metadata import ( # noqa: F401 + # ----------------------------------------------------------------- Session hygiene: auto-compress + # pathologically large transcripts Long-lived gateway sessions can accumulate enough history that every + # new message rehydrates an oversized transcript, causing repeated truncation/context failures. Detect + # this early and compress proactively — before the agent even starts. (#628) Token source priority: 1. + # Actual API-reported prompt_tokens from the last turn (stored in session_entry.last_prompt_tokens) 2. + # Rough char-based estimate (str(msg)//4). Overestimates by 30-50% on code/JSON-heavy sessions, but that + # just means hygiene fires a bit early — safe and harmless. + # ----------------------------------------------------------------- estimate_messages_tokens_rough, estimate_request_tokens_rough, save_context_length, @@ -91,7 +99,13 @@ def _midturn_request_pressure_tokens( """Token figure the mid-turn pre-API compression guard compares: the pruned native-Responses estimate when native compaction eligibility is proven (the generic estimate overstates the wire on compacted sessions, #96995), else messages+tools. - The system prompt is counted exactly once.""" + The system prompt is counted exactly once. + + When the upcoming request is eligible for native Responses compaction the transport will + checkpoint-prune the payload before sending, so the generic durable-history estimate overstates the wire + by orders of magnitude on a compacted session and fires a 600s local compression the main request never + needed (#96995). + """ try: from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens native = estimate_native_responses_preflight_tokens( @@ -182,11 +196,18 @@ def _should_skip_model_call_for_reference_handoff( # Fallback final_response for the sole-handoff skip (#80622); finalize_turn appends it as a # fresh assistant row, so it must not replay the last assistant text. +# Deliberately NOT a replay of the last assistant text: finalize_turn's non-assistant-tail chokepoint +# (#43849) appends final_response as a fresh assistant row, so recovering the previous turn's prose here +# would duplicate it in the durable transcript AND re-deliver it to the user as if it were this turn's +# answer. A short status is honest and idempotent. _HANDOFF_SKIP_FINAL_RESPONSE = ( "Context was compacted. The previous response is complete — awaiting your next message." ) # Terminal final_response when compression timed out while the request was still oversized (#98722). +# Terminal final_response for a turn ended because context compression hit its host progress-aware timeout +# while the request was still oversized (#98722, salvaged from #98741). Sending the unchanged request would +# only bounce off the provider's overflow error and re-enter compression in the same turn. _COMPRESSION_TIMEOUT_FINAL_RESPONSE = ( "Context compression timed out without reducing this conversation. No messages were " "dropped. Start a fresh session with /new, or check auxiliary.compression before retrying /compress." @@ -228,6 +249,12 @@ def _is_interpreter_shutdown_error(exc: Exception) -> bool: """True for a fatal interpreter-shutdown RuntimeError. The RuntimeError type gate stays here: a ValueError carrying similar text must not match (#93269).""" if isinstance(exc, RuntimeError): + # ── Interpreter finalization: abandon immediately ── The process is exiting (TUI quit, SIGTERM, + # one-shot done) while this turn — typically the post-turn review fork's daemon thread — is + # mid-flight. Retries, credential rotation, and fallbacks are all futile ("cannot schedule new + # futures..."), and the buffered ⚠️/❌ retry trace spams the shell after the TUI already exited. End + # the turn with a single log line: no print, no traceback, no debug dump, no retry. Same class as + # cron delivery (#55924/#58720) and concurrent tool submission — shared predicate. from tools.interpreter_shutdown import interpreter_shutting_down return interpreter_shutting_down(exc) return False @@ -296,6 +323,12 @@ def _apply_active_turn_redirect(agent: Any, messages: List[Dict[str, Any]], text # Transcript shows the user's own words; the provider replays the scaffolded form. append_message(messages, {"role": "user", "content": text, "api_content": correction}) + # Stateful scrubber for spans split across stream deltas (#5719). sanitize_context() + # alone can't survive chunk boundaries because the block regex needs both tags in one string. + # Stateful scrubber for reasoning/thinking tags in streamed deltas (#17924). Replaces the per-delta + # _strip_think_blocks regex that destroyed downstream state (e.g. MiniMax-M2.7 streaming '' as + # delta1 and 'Let me check' as delta2 — the regex erased delta1, so downstream state machines never + # learned a block was open and leaked delta2 as content). agent._current_streamed_assistant_text = "" agent._stream_needs_break = True @@ -991,6 +1024,14 @@ def _provider_overflow_exhausted_result( "remains over threshold at ~%s tokens.", agent.log_prefix, max_compression_attempts, f"{request_pressure_tokens:,}", ) + # Host progress-aware timeout (#98722, salvaged from #98741): the provider proved the request does not + # fit, but this recovery pass spent the full wait budget without a committed summary. Re-sending the + # unchanged request would bounce off the same overflow error and re-enter compression in the same turn. + # End the turn with the typed recovery contract instead — transcript intact, no further doomed provider + # sends. + # Prior <3 retries (or an earlier successful tool batch) leave a tool-result tail. Closing it here + # matches interrupt aborts (#48879 / #52592) so the next user turn is not tool→user for strict + # providers. agent._persist_session(messages, conversation_history) return _partial_turn_result( "Context length exceeded: compression could not reduce the rebuilt request below the safe threshold.", diff --git a/agent/copilot_acp_client.py b/agent/copilot_acp_client.py index c32fc6478c..d5876e15ac 100644 --- a/agent/copilot_acp_client.py +++ b/agent/copilot_acp_client.py @@ -119,6 +119,7 @@ def _build_subprocess_env() -> dict[str, str]: # Copilot ACP drives a model and needs LLM provider credentials; the central helper still # strips Tier-1 secrets (bot tokens, GitHub auth, infra). + # See #29157. env = hermes_subprocess_env(inherit_credentials=True) env["HOME"] = _resolve_home_dir() apply_subprocess_home_env(env) @@ -303,6 +304,8 @@ class CopilotACPClient: try: from hermes_cli._subprocess_compat import windows_hide_flags # hide the Windows console flash (#56747); pipes intact for the ACP wire + # Hide the console the CLI child would otherwise flash on Windows (#56747). Hide-only — stdio + # pipes stay intact for the ACP wire. proc = subprocess.Popen( [self._acp_command] + self._acp_args, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding='utf-8', errors='replace', bufsize=1, cwd=self._acp_cwd, env=_build_subprocess_env(), diff --git a/agent/credential_pool.py b/agent/credential_pool.py index a906cbb33e..4286903c0e 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -146,6 +146,14 @@ FAILURE_REASON_BILLING_UNVERIFIED = "billing_unverified" # every model call; on Windows several processes share one rotating log behind # a cross-process lock, and per-selection logging stormed that lock, pegged a # core, and stalled the event loop (Desktop backend readiness timeouts). +# Credential selection runs on a hot path (every model call, plus auxiliary tasks like +# compression/moa/titles), so when a pool is empty or fully exhausted the un-throttled log fires on *every* +# selection. On Windows several Hermes processes share one rotating log guarded by concurrent-log-handler's +# cross-process lock; that per-selection volume storms the lock (``RuntimeError: Cannot acquire lock after +# 20 attempts``), pegs a core, and stalls the asyncio event loop long enough to fail the Desktop backend +# readiness handshake ("Timed out connecting to Hermes backend after 15000ms"). Logging the condition at +# most once per window preserves the signal while removing the storm — same class of fix as the warn-once +# dedup in #58265. NO_AVAILABLE_ENTRIES_LOG_THROTTLE_SECONDS = 60.0 # Pool key prefix for custom OpenAI-compatible endpoints: all share @@ -731,6 +739,8 @@ def _write_through_provider_state_to_global_root( a failed write-through degrades to root-stale and must never break the profile's own successful save. Mirrors ``hermes_cli.auth._write_through_xai_oauth_to_global_root``. + + See #48415. """ try: global_path = _guarded_global_root(auth_mod._global_auth_file_path()) @@ -2717,6 +2727,7 @@ def _seed_custom_pool(pool_key: str, entries: List[PooledCredential]) -> Tuple[b # The pool may be keyed under the durable ``providers.`` # slug or legacy ``custom:``; accept any candidate, or # seeding is skipped when the pool holds the other identity. + # Check if this model's base_url matches our custom provider. See #100413. matched_keys = { str(key).strip().lower() for key in custom_provider_pool_key_candidates(model_base_url) } diff --git a/agent/curator_backup.py b/agent/curator_backup.py index cd194e0bd8..3078f6d03b 100644 --- a/agent/curator_backup.py +++ b/agent/curator_backup.py @@ -33,6 +33,7 @@ DEFAULT_KEEP = 5 # is the backup dir itself; .git is repository metadata — rolling it back breaks git tracking, and snapshots that include it grow # with the full history (once backups are committed back, each snapshot contains the prior ones: 38MB of skills inflated to 24GB # in weeks). The tar filter in ``snapshot_skills`` applies the same set to nested paths, so a nested ``.git`` is skipped too. +# See #91449. _EXCLUDE_TOP_LEVEL = {".curator_backups", ".hub", ".git"} # Snapshot id: UTC ISO with colons replaced by dashes (Windows-safe filename); optional ``-NN`` suffix for same-second snapshots. diff --git a/agent/deadline.py b/agent/deadline.py index 96355b8934..19d9d5fed9 100644 --- a/agent/deadline.py +++ b/agent/deadline.py @@ -41,6 +41,8 @@ _LOOP_BLOCKED_DUMP_GRACE_S = 5.0 # ``Event.wait`` is a C-level block: KeyboardInterrupt / SetAsyncExc only land when the # thread returns to Python, so the sync wait is sliced to observe /stop or SIGINT promptly. +# Slice the wait so a /stop or SIGINT during a bounded sync call is observed within this window rather than +# at the full deadline (#94285, tools/test_local_interrupt_cleanup). _BOUNDED_SYNC_WAIT_SLICE_S = 0.2 @@ -165,6 +167,12 @@ def resolve_timeout(key: str, *, default: Optional[float], env_var: Optional[str # second timer dumps all thread stacks when the loop provably failed to process the expiry. +# --------------------------------------------------------------------------- Bounded execution — async +# flavor. Generalizes plugins/platforms/telegram/adapter.py:_await_with_thread_deadline (the #63309 fix): +# the deadline is driven by a daemon threading.Timer so a blocked event loop cannot disable it, and a second +# timer dumps all thread stacks when the loop provably failed to process the expiry — the one piece of +# information loop-blocked hangs otherwise never surface. +# --------------------------------------------------------------------------- def _consume_abandoned(task: "asyncio.Future[Any]") -> None: """Observe an abandoned task's outcome so it never logs 'never retrieved'.""" try: @@ -282,7 +290,10 @@ def run_bounded_sync( """Run ``fn`` in a daemon worker thread under a wall-clock deadline; exceptions re-raise in the caller. On expiry the worker is **abandoned** (every timeout leaks one daemon thread, so do NOT use per-item in hot loops) and ``on_timeout`` runs best-effort in the caller's thread. - The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive.""" + The worker runs under ``contextvars.copy_context()`` so secret scope / session id survive. + + See #94285. + """ timeout_s = clamp_timeout(timeout) start = time.monotonic() if timeout_s is None: diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 093a79195a..7bf3b8716f 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -140,6 +140,9 @@ _USAGE_LIMIT_TRANSIENT_SIGNALS = ( # Anthropic's "request_too_large" type without one). _PAYLOAD_TOO_LARGE_PATTERNS = ( "request entity too large", "payload too large", "error code: 413", "request_too_large", + # Normally arrives with an HTTP 413 status (handled by the status path), but aggregators/proxies can + # re-wrap it into a plain message with no status attribute — route it to the same compression recovery. + # (port of anomalyco/opencode#37848) "request exceeds the maximum size", ) @@ -181,6 +184,8 @@ _CONTEXT_OVERFLOW_PATTERNS = ( "超过最大长度", "上下文长度", "tokens in request more than max tokens allowed", "input is too long", "max input token", "input token", "exceeds the maximum number of input tokens", + # Together/Fireworks-style: "Input length 131393 exceeds the maximum allowed input length of 131040 + # tokens." No other pattern in this list matches that wording. (port of anomalyco/opencode#37848) "maximum allowed input length", ) @@ -548,6 +553,16 @@ def _by_transport(c: _Ctx) -> Optional[Verdict]: if any(p in msg for p in _SERVER_DISCONNECT_PATTERNS) and not c.status_code: # Reasoning models: far more likely the gateway idle-killed a long # thinking stream — never compress on a phantom overflow (#52310). + # Reasoning-model override: a transport disconnect on a reasoning model is much more likely the + # upstream proxy idle-killing a long thinking stream than a true context overflow — even on large + # sessions. The default disconnect+large-session routing below would otherwise send the user into + # the compression branch (should_compress=True) and silently delete conversation history on a + # phantom context-length error. Reasoning models have multi-minute thinking phases that routinely + # exceed the cloud gateway's idle window (NVIDIA NIM ~120s — first-party repro at + # NVIDIA/NemoClaw#4846; OpenAI worker / Anthropic stream-idle similar). The per-reasoning-model + # stale-timeout floor in agent/reasoning_timeouts.py raises the stale-detector threshold to tolerate + # long thinking, so a true transport-layer failure here is recoverable via the retry path — not via + # context compression. Reclassify as timeout. (Part 1 of Fixes #52310.) from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor if get_reasoning_stale_timeout_floor(c.model) is not None: return _V_TIMEOUT diff --git a/agent/gemini_native_adapter.py b/agent/gemini_native_adapter.py index 0d2af89f4a..9dafa4d6bf 100644 --- a/agent/gemini_native_adapter.py +++ b/agent/gemini_native_adapter.py @@ -84,7 +84,13 @@ def bare_gemini_model_id(model: str) -> str: def gemini_requires_tool_call_ids(model: str) -> bool: """Gemini 3+ needs explicit functionCall/functionResponse ids so replayed parallel tool calls - pair with their responses; 2.x rejects the field.""" + pair with their responses; 2.x rejects the field. + + Gemini 3+ models require explicit tool call IDs in replayed history — without them, multi-tool turns can + be rejected or mismatched. Older Gemini models (2.x) reject unexpected ``id`` fields, so this is gated + on the major version. Mirrors earendil-works/pi#7494 (their fix for the same class of bug in the + google-shared converter). + """ match = re.match(r"gemini-(\d+)", bare_gemini_model_id(model).lower()) return match is not None and int(match.group(1)) >= 3 @@ -242,6 +248,10 @@ def _translate_tool_result_to_gemini( parsed = json.loads(content) if content.strip().startswith(("{", "[")) else None except json.JSONDecodeError: parsed = None + # Gemini 3 resolves JSON-Schema ``$ref`` pointers inside a functionResponse.response payload and rejects + # unknown references with HTTP 400 INVALID_ARGUMENT ("referenced name '#/$defs/...' does not match a + # display_name"; see vercel/ai#14369). A tool result that is itself a JSON Schema (e.g. tool_describe + # output for an MCP tool) must therefore be forwarded as opaque text, not as a structured response. structured = isinstance(parsed, dict) and not _looks_like_json_schema(parsed) function_response: Dict[str, Any] = {"name": name, "response": parsed if structured else {"output": content}} if include_ids and tool_call_id: @@ -264,6 +274,17 @@ def _merge_alternating(contents: List[Dict[str, Any]]) -> List[Dict[str, Any]]: functionResponse + functionResponse still merge); 3) the split pair stays API-valid via an interposed placeholder model turn.""" merged: List[Dict[str, Any]] = [] + # Compatibility contract for native Gemini generateContent: 1) Same-role adjacent contents still merge + # in general (strict user/model alternation for ordinary text turns and parallel tool-result grouping; + # consecutive same-role contents are rejected with HTTP 400 "Please ensure that multiturn requests + # alternate between user and model"). 2) Exception: do NOT fuse a human user text turn into a preceding + # user content that only carries functionResponse parts (or vice versa). Gemini 3 accepts that fold with + # HTTP 200 but then reads the trailing text as a continuation of the tool result — it returns an empty + # candidate or "finishes the user's sentence" instead of answering (same defect gemini-cli fixed in + # google-gemini/gemini-cli#28700). 3) Because rule 1's HTTP 400 makes two consecutive user contents + # unsafe to emit (#55125 — the reason this merge exists), the split pair is kept API-valid by + # interposing a placeholder model turn between the functionResponse content and the human text content, + # mirroring gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER repair. for content in contents: prev = merged[-1] if merged else None same_role = prev is not None and prev["role"] == content["role"] @@ -677,6 +698,10 @@ class AsyncGeminiNativeClient: self.api_key, self.base_url = sync_client.api_key, sync_client.base_url self.chat = SimpleNamespace(completions=SimpleNamespace(create=self._create_chat_completion)) + # Expose the underlying sync client as _real_client so the auxiliary cache's eviction-by-leaf-client + # helper (#23482) can find and drop this async entry when the sync GeminiNativeClient is poisoned. + # GeminiNativeClient is itself the leaf (no OpenAI client beneath it), so we point at the sync_client + # directly. async def _create_chat_completion(self, **kwargs: Any) -> Any: result = await asyncio.to_thread(self._sync.chat.completions.create, **kwargs) return self._async_stream(result) if kwargs.get("stream") else result diff --git a/agent/image_routing.py b/agent/image_routing.py index 03de0d79d3..19004139f6 100644 --- a/agent/image_routing.py +++ b/agent/image_routing.py @@ -212,7 +212,11 @@ def _resolve_inference_base_url(cfg: Optional[Dict[str, Any]], provider: str) -> def _resolve_inference_api_key(cfg: Optional[Dict[str, Any]], provider: str) -> str: """Best-effort API key, resolved like :func:`_resolve_inference_base_url` so it matches the base URL actually probed; otherwise the local server-type probe hits - a keyed remote endpoint without Authorization and sprays 401s on every image turn.""" + a keyed remote endpoint without Authorization and sprays 401s on every image turn. + + Mirrors :func:`_resolve_inference_base_url`'s resolution order (runtime value, then ``model.api_key``, + then the providers blocks) so the key matches the base URL actually being probed. See #89863. + """ return _resolve_inference_value(cfg, provider, "api_key", runtime_ok=lambda _: True) @@ -270,6 +274,11 @@ def _probe_models_dev(provider: str, model: str, cfg: Optional[Dict[str, Any]]) The fetch is cached (4h TTL) and backoff-limited.""" from agent.models_dev import get_model_capabilities + # allow_network=True on purpose: vision-capability lookup runs when an image actually needs routing (not + # per turn), and the #31179 text-only-main guard depends on catalog data — a cold cache returning + # "unknown" would fall back to attempting the call and reintroduce the bug. This preserves the + # historical network-on-cold-cache behavior for this one path; the fetch is cached (4h TTL) and + # backoff-limited after failures. caps = get_model_capabilities(provider, model, allow_network=True) return None if caps is None else bool(caps.supports_vision) diff --git a/agent/insights.py b/agent/insights.py index 5a77136ace..aa3739711b 100644 --- a/agent/insights.py +++ b/agent/insights.py @@ -16,7 +16,11 @@ _SKILL_TOOLS = {"skill_view", "skill_manage"} def _fmt_est_cost(est_cost: float) -> str: - """Shared label helper so sub-cent totals render at 4dp, not "~$0.00".""" + """Shared label helper so sub-cent totals render at 4dp, not "~$0.00". + + Routes through ``format_cost_label`` so sub-cent aggregates render at 4dp instead of collapsing to + "~$0.00" (#79220 bug class — the same dishonesty this module's cost buckets exist to fix, #77223). + """ return format_cost_label(Decimal(str(est_cost))) @@ -450,6 +454,8 @@ class InsightsEngine: @staticmethod def _cost_lines(o: Dict, templates: tuple) -> List[str]: """One formatted line per non-zero cost bucket (estimated, included, unknown).""" + # Cost breakdown — surface the three buckets so subscription-included and unknown-cost sessions are + # visible instead of silently collapsing to $0. See #77223. est_cost = o.get("estimated_cost", 0.0) values = (_fmt_est_cost(est_cost) if est_cost > 0 else "", o.get("included_cost_sessions", 0), o.get("unknown_cost_sessions", 0)) return [tpl.format(v) for tpl, v in zip(templates, values) if v] diff --git a/agent/learning_mutations.py b/agent/learning_mutations.py index a64f73fde0..2ee948af1b 100644 --- a/agent/learning_mutations.py +++ b/agent/learning_mutations.py @@ -111,6 +111,11 @@ def delete_node(node_id: str) -> dict[str, Any]: def _delete_skill(name: str) -> dict[str, Any]: from tools import skill_usage + # Pin must be respected by autonomous maintenance. The curator already skips pinned skills from every + # auto-transition; the background review fork is the same kind of autonomous, no-user-present actor, so + # it must not write to a pinned skill either (issue #25839). This is stricter than the foreground + # ``_pinned_guard`` (which only blocks deletion) precisely because there is no user in the loop to + # consent to an edit here. if skill_usage.get_record(name).get("pinned"): return {"ok": False, "message": f"'{name}' is pinned — unpin it first (hermes curator unpin {name})"} ok, message = skill_usage.archive_skill(name) diff --git a/agent/memory_manager.py b/agent/memory_manager.py index b50b22a103..ce74dc11b4 100644 --- a/agent/memory_manager.py +++ b/agent/memory_manager.py @@ -127,6 +127,7 @@ def inject_memory_provider_tools(agent: Any) -> int: if not memory_provider_tools_exposed(agent): # Say so once: a silent 0 leaves the provider looking "half on" with no clue which # config key (platform_toolsets / disabled_toolsets) gated it. + # See #81014. _providers = [p for p in getattr(memory_manager, "providers", None) or [] if getattr(p, "name", "") != "builtin"] if _providers: @@ -344,6 +345,9 @@ class MemoryManager: # Core tool names are reserved: built-ins always win at agent init, so a shadowing # provider tool would linger in ``_tool_to_provider`` and hijack dispatch. + # ``clarify``, ``delegate_task``). Reject it here, at the door, so it never enters the routing table + # at all — matching the built-ins-always-win invariant used by the TTS/browser/search provider + # registries. See #40466. from toolsets import _HERMES_CORE_TOOLS for raw_schema in provider.get_tool_schemas(): @@ -592,6 +596,16 @@ class MemoryManager: ``on_session_end`` (LLM-bound, seconds) must run strictly BEFORE ``on_session_switch`` rebinds provider state; an ad-hoc thread raced the inline switch and misattributed transcripts. + + Running extraction inline blocked the /new command for the whole LLM round-trip (#16454); running it + on an ad-hoc thread raced the inline switch — providers key off internal state, so a late + ``on_session_end`` ran against post-switch bindings (transcript misattributed to the new session id, + double-ingest of the old turn buffer, new-session buffers cleared). + Submitting BOTH hooks as one task on the manager's single background worker gives both properties at + a single chokepoint: the caller returns immediately, and the worker's FIFO order serializes + end→switch against every other provider write (per-turn ``sync_all``, prefetches), which already + share the same worker. If the executor is unavailable, ``_submit_background`` degrades to inline + execution — the pre-#16454 synchronous behavior, slow but correct. """ if not self._providers: return diff --git a/agent/message_sanitization.py b/agent/message_sanitization.py index 54f8f91e24..4d9bb17ed2 100644 --- a/agent/message_sanitization.py +++ b/agent/message_sanitization.py @@ -253,6 +253,7 @@ _IMAGE_REJECTION_PHRASES = ( "does not support images", "does not support image input", "does not support multimodal", "does not support vision", "model does not support image", # DashScope-style gateways reject non-text blocks with this generic body. + # Some OpenAI-compatible endpoints (e.g. (issue #57948) "unexpected item type in content", # ChatGPT-account Codex backend rejects data:image URLs in input_image; keyed on the # field-path apostrophe so other URL errors don't false-trip. Second: its wording for @@ -262,8 +263,17 @@ _IMAGE_REJECTION_PHRASES = ( "unknown variant `image_url`, expected `text`", "unknown variant image_url, expected text", # OpenRouter HTTP 404 when no upstream endpoint accepts image input (passes the 4xx # gate; without this the gateway queue wedges behind the stuck turn). + # Without this phrase the agent never strips the images, the retry loop re-sends the same rejected + # request until exhaustion, and the gateway leaves every subsequent message queued behind the stuck turn + # — the P1 in issue #21160. "no endpoints found that support image input", # Kimi/Moonshot et al. reject truncated/corrupt image bytes baked into history. + # Kimi / Moonshot / other OpenAI-compatible Chinese providers reject truncated or corrupt image bytes + # with HTTP 400 "Invalid request: prepare image failed ... failed to decode image: invalid or + # unsupported image format". Like the Codex case above, the bad bytes are baked into immutable + # conversation history and re-sent on every retry, wedging the session. Strip the images so the turn + # recovers instead of exhausting retries. (issue #76884; complements the proactive full-decode + # validation in tools/vision_tools._normalize_to_supported_image) "failed to decode image", ) @@ -305,6 +315,17 @@ def _tc_set(tc: Any, key: str, value: Any) -> None: tc.__setitem__(key, value) if isinstance(tc, dict) else setattr(tc, key, value) +# --------------------------------------------------------------------------- call_id policy — single owner +# (audit F4, incident chain I4) --------------------------------------------------------------------------- +# Three forked policy sites converged here: * agent/codex_responses_adapter.py `_deterministic_call_id` — +# hash synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2). * +# run_agent.AIAgent._get_tool_call_id_static — `call_id or id` coalescing for dicts and SDK objects. * +# run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with deterministic `_d` suffixes +# (#58327 loss class). NOT consolidated (different scheme on purpose): +# agent/transports/codex_event_projector._deterministic_call_id maps codex app-server ITEM ids +# (`codex__`), not chat tool-call content; merging the two would change ids and invalidate +# prompt caches. HARD INVARIANT: everything here must stay deterministic (never uuid4) and byte-identical +# for existing inputs — these ids feed prompt-cache prefixes. def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str: """Deterministic call_id fallback when the API omits one (random ids would break caching).""" seed = f"{fn_name}:{arguments}:{index}" @@ -392,6 +413,19 @@ def uniquify_tool_call_ids(tool_calls: list) -> list: # empty-string pads → " ". Strict side (400/422 "Extra inputs are not permitted"): everyone # else — Mistral, Cerebras, Groq, SambaNova, … Strip the key entirely, even a one-space pad. +# --------------------------------------------------------------------------- reasoning_content policy — +# single owner (audit F4) --------------------------------------------------------------------------- The +# strip-vs-repad decision was previously forked across the wire files in separate incident commits +# (2b3a4f0af8 strip for strict providers, b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi +# pad). The POLICY — which provider direction gets which treatment — lives here as one rule table + apply +# functions; adapters keep only SYNTAX mapping (e.g. anthropic_adapter turning reasoning_content into a +# thinking block). Direction table: require-side (echo-back enforced; replays 400 without the field): kimi +# — provider kimi-coding/kimi-coding-cn, or host api.kimi.com / moonshot.ai / moonshot.cn. Host-driven on +# purpose: aggregators re-exporting kimi models reject the echo. deepseek — provider "deepseek", model +# contains "deepseek", or host api.deepseek.com (#15250; V4 rejects empty-string pads, hence the " " +# single-space pad, #17341). mimo — provider "xiaomi", model contains "mimo", or host *.xiaomimimo.com. +# strict side (field rejected with 400/422 "Extra inputs are not permitted"): everyone else — Mistral, +# Cerebras, Groq, SambaNova, … (#45655). Strip the key entirely, even a single-space pad. _REASONING_ECHO_RULES: tuple = ( # (family, exact providers (raw), exact providers (lowered), model substrings (lowered), hosts) ("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (), ("api.kimi.com", "moonshot.ai", "moonshot.cn")), @@ -449,6 +483,15 @@ def apply_reasoning_content_policy(source_msg: dict, api_msg: dict, needs_thinki api_msg.pop("reasoning_content", None) return existing, reasoning = source_msg.get("reasoning_content"), source_msg.get("reasoning") + # 1. Explicit reasoning_content already set. When the active provider enforces the thinking-mode + # echo-back (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their own space-placeholder + # written at creation time and any valid reasoning from the same provider. Sessions persisted BEFORE + # #17341 have empty-string placeholders pinned at creation time; DeepSeek V4 Pro rejects those with + # HTTP 400, so upgrade "" → " " on replay. When the active provider does NOT enforce echo-back, strip + # the field entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq, SambaNova, …) + # reject ANY reasoning_content key in input messages with HTTP 400/422 ("Extra inputs are not + # permitted"), even an empty string or a single-space pad. Stripping here covers the rebuild path; + # ``reapply_reasoning_echo`` covers the already-built api_messages path. Refs #45655. if isinstance(existing, str): # Explicit value: preserve verbatim, upgrading legacy "" to " " (DeepSeek V4 400s on ""). api_msg["reasoning_content"] = existing or " " @@ -475,6 +518,16 @@ def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int: for api_msg in api_messages: if api_msg.get("role") != "assistant": continue + # 3. Healthy session: promote 'reasoning' field to 'reasoning_content' for providers that use the + # internal 'reasoning' key. This must happen before the unconditional empty-string fallback so + # genuine reasoning content is not overwritten (#15812 regression in PR #15478). Only promote for + # providers that enforce echo-back — strict providers reject the field (refs #45655). + # 4. DeepSeek / Kimi thinking mode: all assistant messages need reasoning_content. Inject a single + # space to satisfy the provider's requirement when no explicit reasoning content is present. + # Covers both tool-call turns (already-poisoned history with no reasoning at all) and plain text + # turns. Space (not "") because DeepSeek V4 Pro tightened validation and rejects empty string with + # HTTP 400 ("The reasoning content in the thinking mode must be passed back to the API"). Refs + # #17341. if needs_thinking_pad: if not api_msg.get("reasoning_content"): apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad) diff --git a/agent/micro_compaction.py b/agent/micro_compaction.py index 6e4e5a01c6..275b77b6a4 100644 --- a/agent/micro_compaction.py +++ b/agent/micro_compaction.py @@ -180,6 +180,11 @@ class MicroCompactionMixin: # in-place pop on a live dict would be identity-skipped by the bounded flush scan; # flag the finalizer. entry.pop(_cc()._DB_PERSISTED_MARKER, None) + # Sibling of the finalize_turn pop site (#75170): this pop also strips the marker from a LIVE + # dict in place, so the bounded flush-scan cursor would identity-skip the rewritten marker and + # the defragged summary would never reach state.db. The compressor holds no agent reference, so + # raise a flag the finalizer consumes to invalidate agent._db_flush_scan_prefix. (The pop sites + # at module scope — fresh copies in strip-marker helpers — break identity and need no flag.) self._flush_scan_cursor_invalidated = True logger.info( "Micro-compaction defrag: rolling summary re-summarized (%d -> %d chars)", @@ -313,6 +318,9 @@ class MicroCompactionMixin: self._micro_compact_tokens_saved_total -= delta or 0 self._micro_compact_passes += 1 # Cached reads only: the lazy properties can fire a synchronous /models probe. + # The ``threshold_tokens`` / ``context_length`` properties resolve lazily and can fire a + # synchronous /models probe on first access (#32221) — telemetry must never be the thing that + # blocks a turn. Unresolved simply reports null. threshold = self._threshold_tokens has_occupancy = threshold and tokens_after is not None and threshold > 0 occupancy = round(tokens_after / threshold * 100, 1) if has_occupancy else None @@ -343,6 +351,7 @@ class MicroCompactionMixin: # Every row except the marker is a carried-forward original: archive rewind-style. session_db.archive_and_compact(session_id, compacted_messages, tail_count=max(0, len(compacted_messages) - 1)) # Shared post-commit stamp site with batch commit and proactive prune. + # See #98450. _cc().stamp_db_persisted_markers(compacted_messages) except Exception: logger.info( diff --git a/agent/moa_loop.py b/agent/moa_loop.py index e5ac42fb3d..d0b0968698 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -30,6 +30,19 @@ logger = logging.getLogger(__name__) # Privacy filter (moa.privacy_filter: '' | display | full): PII classes agent.redact # leaves alone. The phone pattern requires explicit delimiters so line numbers, # dates, times, SHAs, IPs and versions never match. +# Advisor (reference) outputs can echo PII from the conversation — emails, phone numbers, credentials pasted +# by the user — into surfaces the user may not expect: the labelled reference blocks rendered in the UI, +# saved MoA trace files, and (in `full` mode) the guidance block injected into the aggregator prompt (issue +# #59959). Secret/credential shapes (API-key prefixes, JWTs, private keys, DB connection strings, E.164 +# phone numbers) are handled by the repo's central redactor, ``agent.redact .redact_sensitive_text`` — the +# MoA filter never re-implements those. The two patterns below cover the PII classes the central redactor +# deliberately leaves alone for log/tool output (emails and formatted phone numbers). Pattern safety: +# advisory text is frequently code-review-shaped — line numbers, timestamps, git SHAs, IDs, IP addresses. A +# bare 10-digit match would mangle all of those, so the phone pattern requires clearly delimited formatting: +# a parenthesized area code and/or explicit `-`/`.` separators between groups ((555) 123-4567, 555-123-4567, +# 555.123.4567, +1 555-123-4567). Undelimited digit runs (5551234567), dates (2026-07-12), times (12:34:56), +# hex IDs, and dotted quads never match. International numbers in E.164 form (+14155551234) are already +# masked by the central redactor. _MOA_EMAIL_RE = re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b") _MOA_PHONE_RE = re.compile( r"(? Any: # Cold-start caches: preset and per-(provider, model) runtime are immutable for a turn. +# A MoA preset switch used to re-resolve the full config + preset + every slot's provider runtime on EACH +# create() call (once per tool-loop iteration), serially before the parallel fan-out could start — adding +# 5-30s of "frozen" latency on complex presets (#66793). _preset_cache_lock = threading.Lock() _preset_cache: dict[tuple, Any] = {} @@ -286,6 +302,9 @@ def _maybe_apply_moa_cache_control( legacy system-and-3 fallback is used. ``cache_disabled`` is stamped onto the stub so ``cache_ttl: off`` is honored; ``cache_ttl`` is clamped per destination. Returns the messages unchanged on any error. + + ``cache_disabled`` (or the live config when omitted) is stamped onto the policy stub so + ``prompt_caching.cache_ttl: off`` is not bypassed by the blank-agent pattern (#76085). """ try: from agent.agent_runtime_helpers import anthropic_prompt_cache_policy, blank_cache_policy_stub @@ -353,6 +372,10 @@ def _run_reference( # Trim to THIS model's window (advisors may be smaller than the aggregator); the # advisory view is append-only across iterations, so cache_control lets # iteration N+1 replay N's cached prefix. + # Reference models may have a smaller window than the aggregator (e.g. kimi-k2.7-code @ 262K + # advising a glm-5.2 @ 1M conversation); without this trim the provider returns a hard HTTP 400 + # which the except below silently converts to a [failed: …] note (issue #60345). Estimated AFTER the + # advisory system prompt is prepended so its tokens count against the budget too. trimmed = _trim_messages_for_reference( messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache, ) @@ -421,6 +444,11 @@ def _trim_messages_for_reference( body and the trailing user turn plus one preceding turn (even if still over budget). ``context_length_cache`` memoizes the window per (provider, model); unresolvable windows leave messages unchanged. + + Reference models may have a smaller context window than the aggregator or the main conversation. Without + this trim, a reference whose window is exceeded gets a hard HTTP 400 from the provider, which + ``_run_reference``'s try/except silently converts to a ``[failed: …]`` note — the MoA turn silently + degrades to fewer references (issue #60345). """ if not messages or not slot.get("model"): return messages @@ -480,6 +508,13 @@ def _settle_interrupted( for future, idx in futures.items(): if results[idx] is not None: continue + # #38922: a slow confirmation does NOT necessarily mean the send failed — but we must distinguish + # two cases via future.cancel()'s return value: cancel() == False -> the coroutine was already + # running on the gateway loop when the timeout fired; the request is in flight on the wire and + # cannot be un-sent. Re-sending via standalone would be a guaranteed DUPLICATE, so treat it as + # delivered (assume-delivered). cancel() == True -> the scheduled callback never started executing + # (loop wedged/backlogged for the full 60s), so nothing was sent. We MUST fall through to the + # standalone path or the message is silently dropped (worse than a duplicate). cancelled = future.cancel() if not cancelled and future.done(): results[idx] = future.result() @@ -753,6 +788,12 @@ def aggregate_moa_context( Failures become model-specific notes instead of aborting the loop. ``reference_max_tokens`` caps ONLY the fan-out (capping the aggregator truncated long syntheses). ``agent`` makes the fan-out interruptible. + + ``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is + never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when + it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright. + A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing + ``reference_max_tokens`` to both calls here would silently reintroduce that regression. """ reference_models = [slot for slot in reference_models if slot.get("enabled", True)] reference_outputs = _run_references_parallel( @@ -1015,6 +1056,9 @@ class MoAChatCompletions: planning_messages = peel_reference_guidance(agg_messages, str(guidance)) if guidance else agg_messages # Tri-state cache_disabled: facades built via __new__ have no _agent; forcing # False would suppress the planner's config fallback. + # plan_cache_sections_for_destination never mutates its inputs and always returns request-local + # copies, so the prepared state stays canonical. Tri-state: only pass a bool when a live agent + # snapshot exists. See #76085. _agent = getattr(self, "_agent", None) cache_disabled, cache_ttl = _agent_cache_opts(_agent) # Agent TTL + stable system prefix so MoA does not regress 1h → 5m. @@ -1090,6 +1134,19 @@ class MoAChatCompletions: view changes. "every_n:": iteration 1 of a turn, then every Nth; in-between iterations return the pinned last on-cadence key (HIT: no calls, no re-emit). """ + # "user_turn" (default — cheapest cadence, #67199): advisors run ONCE per user turn; subsequent tool + # iterations reuse that turn's advice and the aggregator acts alone (the original MoA shape: + # synthesize at the start, then let the acting model work). Implemented by hashing only the prefix + # up to the LAST USER message so mid-turn growth doesn't change the signature — iteration 2+ becomes + # a cache HIT. "per_iteration": advisors re-run whenever the advisory view changes — i.e. every tool + # iteration, since the view grows with each tool result; advice tracks live task state at the cost + # of multiplying advisor latency/spend by tool-loop depth. "every_n:" (N >= 2): the middle ground + # (issue #63393 — advisor fan-out multiplies latency/cost by the tool-iteration count). Advisors run + # on iteration 1 of a user turn and then every Nth tool iteration; the iterations in between REUSE + # the cached guidance from the last on-cadence run (same mechanism as user_turn's cache HIT — the + # aggregator still gets advice every iteration, it's just not refreshed against the very latest tool + # results). The iteration counter is scoped per user turn and resets on a new user message, so every + # turn starts with fresh advice. fanout_mode = str(preset.get("fanout") or "user_turn").strip().lower() every_n = 0 if fanout_mode.startswith("every_n:"): diff --git a/agent/model_metadata.py b/agent/model_metadata.py index e094b27e1f..1bbf221ddf 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -112,6 +112,11 @@ _ENDPOINT_MODEL_CACHE_TTL = 300 # server swap on the same port is re-detected; None gets the short TTL so a # transient failure recovers in minutes without re-running the waterfall each turn. _ENDPOINT_PROBE_TTL_SECONDS = 3600.0 +# A failed probe verdict (server_type is None — no known endpoint answered) is cached for a much shorter +# window: the in-memory entry exists only to keep one image-bearing turn from re-running the 5-request +# waterfall on every subsequent turn (#89863 — a keyed remote endpoint answered 401 to each leg and the None +# verdict was never cached, so every turn re-probed). Short TTL keeps a transient failure (server starting +# up, key being fixed) recoverable within minutes instead of pinning "undetected" for an hour. _ENDPOINT_PROBE_FAILURE_TTL_SECONDS = 300.0 _endpoint_probe_path_cache: Dict[str, tuple] = {} # Routable-but-dead endpoints (corp LAN off-VPN) blackhole TCP: once ANY probe paid @@ -674,6 +679,9 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]: return result +# Cache the negative verdict in memory only (never on disk — a failure is often transient: server starting, +# key being fixed) so the very next turn does not re-run the whole waterfall against an endpoint that just +# answered nothing (#89863). def _iter_nested_dicts(value: Any): if isinstance(value, dict): yield value @@ -778,6 +786,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any try: _ensure_requests() # (connect, read) tuple: a flat timeout lets urllib3 block per retry stage through proxies that 403 CONNECT. + # See #46620. response = requests.get(OPENROUTER_MODELS_URL, timeout=(5, 10), verify=_resolve_requests_verify()) response.raise_for_status() cache = {} @@ -1186,6 +1195,8 @@ def _ollama_show(server_url: str, api_key: str, bare_model: str, timeout: float def _is_ollama_server(base_url: str, api_key: str) -> bool: try: + # Forward the API key: a remote API-keyed endpoint answers the probe waterfall with 401s without it, + # and an unauthorized probe can never produce a positive verdict (#89863). return detect_local_server_type(base_url, api_key=api_key) == "ollama" except Exception: return False @@ -1447,7 +1458,11 @@ _CODEX_900K_SNAPSHOT_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$") def _bare_codex_slug(model: Optional[str]) -> str: - """Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``).""" + """Lowercased slug without ``vendor/`` (display/auxiliary callers pass ``openai/gpt-5.6-sol-900k``). + + Display/auxiliary callers pass ids like ``openai/gpt-5.6-sol-900k``; the main-agent path normalizes the + namespace away earlier, but this resolver must accept both shapes (#92797 review). + """ return (model or "").strip().lower().rsplit("/", 1)[-1] @@ -1567,6 +1582,10 @@ def _resolve_codex_oauth_context_length_with_source(model: str, access_token: st return bumped, source return ctx, source # The Codex catalog only knows the base slug (no -900k, no vendor/). + # ``-900k`` variants are Hermes picker aliases — the Codex catalog only knows the base slug, so resolve + # against the stripped id. Also drop any ``vendor/`` namespace (``openai/gpt-5.6-sol-900k``): the + # main-agent path normalizes it away before reaching here, but display/auxiliary callers pass it through + # (#92797 review). lookup_bare = _bare_codex_slug(strip_codex_context_variant_suffix(model_bare)) if access_token: live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token) @@ -1650,6 +1669,10 @@ def _validate_cached_context_length(model: str, base_url: str, cached: int, is_b _invalidate_cached_context_length(model, base_url) return bedrock_ctx return cached + # For local endpoints, run the probe that respects configured Modelfile context values first. + # _query_local_context_length prefers num_ctx from Modelfile, while _query_ollama_api_show returns the + # GGUF training max first which can be larger and would create a false-safe window for compression + # (#63122). Non-local endpoints preserve the existing GGUF-first behavior. if is_local_endpoint(base_url): return _reconcile_local_cached_context_length(model, base_url, cached, api_key=api_key) return cached @@ -1740,12 +1763,17 @@ def _config_override_context_length(model: str, base_url: str, provider: str, cu """Steps 0b-0c: config-only overrides (never touch the network). 0b: EXPLICIT model_overrides only — fill-gap _default entries apply inside lookup_models_dev_context once the catalog has missed, so a _default can never preempt custom_providers or live probes. 0c: custom_providers.""" + # This is the supported self-unblock path for models with wrong context in models.dev (#84482) and for + # custom/local models (#8731). if provider and model: with contextlib.suppress(Exception): # fall through to other resolution paths from agent.models_dev import _override_context_window mo_ctx = _override_context_window(provider, model) if mo_ctx is not None and mo_ctx > 0: return mo_ctx + # 0c. custom_providers per-model override — check before any probe. This closes the gap where /model + # switch and display paths used to fall back to 128K despite the user having a per-model context_length + # set. See #15779. if custom_providers and base_url and model: with contextlib.suppress(Exception): # fall through to probing from hermes_cli.config import get_custom_provider_context_length @@ -1981,6 +2009,16 @@ def _strip_stale_thinking_for_estimate(messages: List[Dict[str, Any]]) -> List[D # pinned (strong ref in the entry, so the id can't be reused and immutability makes id-equality # value-equality); numbers/bools/None by value; dicts/lists structurally in key order (``str(shadow)`` # depends on it); any other type aborts the memo. api_messages shallow-copies dicts but shares the strings. +# ``estimate_messages_tokens_rough`` is called on the full history every loop iteration (conversation_loop +# preflight), repeatedly during compaction telemetry, and inside an O(n^2) shrink loop in moa_loop. The +# per-message helpers are pure functions of the message's value, so a memo keyed on a fingerprint that +# uniquely determines the value is exactly equivalent. Fingerprint design (soundness argument): While the +# entry lives, that id cannot be reused by another object, so id-equality implies object-equality — strings +# are immutable, so value-equality too (no #50372-style aliasing). Equal fingerprints therefore imply +# deep-equal messages built from identical immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical +# estimate. Because the api_messages build shallow-copies history dicts each iteration, the copies share the +# same content strings — so unchanged history messages hit the memo even though the outer dicts are fresh +# objects every turn. _MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {} _MSG_TOKENS_CACHE_MAX = 4096 diff --git a/agent/models_dev.py b/agent/models_dev.py index 1ae381ce38..a03c8c6207 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -498,7 +498,10 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool """Context window in tokens for provider+model, or None if not found. An EXPLICIT ``model_overrides`` entry wins over the catalog; ``_default`` fills the gap only when the catalog has no answer (the self-unblock path for wrong/missing context in models.dev). Catalog entries with context=0 are - skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn.""" + skipped in favour of later candidates. ``allow_network`` defaults to False — runs every turn. + + See #84482. + """ override_ctx = _override_context_window(provider, model) if override_ctx is not None: return override_ctx @@ -513,6 +516,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool # catalog. ``._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to # models the catalog does not know and never displace catalog data. Provider keys accept the Hermes # or models.dev id; model ids match exactly, then case-insensitively (mirroring catalog lookup). +# Resolution semantics: 1. 2. See #84482, #8731. _OVERRIDE_WARNED_KEYS: set = set() # Safe defaults for models absent from the catalog (tools on, vision/reasoning off, 200K context); # shared by get_model_capabilities and get_model_info so the two unknown-model paths agree. @@ -589,6 +593,7 @@ def _override_context_window(provider: str, model: str) -> Optional[int]: return _override_int(ov, "context_window") if ov is not None else None +# Catalog miss — a _default override may fill the gap (#84482). def _default_override_context(provider: str) -> Optional[int]: """Fill-gap context from a ``_default`` override, for catalog misses.""" default = _default_model_override(provider) @@ -655,7 +660,13 @@ def _entry_supports_vision(entry: Dict[str, Any]) -> bool: def get_model_capabilities(provider: str, model: str, *, allow_network: bool = False) -> Optional[ModelCapabilities]: """Capability metadata from the models.dev cache, or None if unresolvable. EXPLICIT ``model_overrides`` patch catalog fields; ``_default`` fills the gap only for models the catalog does not know. Unspecified - fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path).""" + fields fall through to the catalog, or to safe defaults. ``allow_network`` defaults to False (hot path). + + EXPLICIT ``model_overrides`` entries (per-provider+model) win over catalog values for the fields they + set. ``_default`` entries fill the gap only for models the catalog does not know — the supported + self-unblock path for custom/local models (#8731) and for models with wrong metadata in models.dev + (#84482). + """ models = _get_provider_models(provider, allow_network=allow_network) entry = _find_model_entry(models, model) if models is not None else None raw = _apply_overrides(provider, model, entry) @@ -751,7 +762,13 @@ def get_provider_info(provider_id: str, *, allow_network: bool = True) -> Option def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = False) -> Optional[ModelInfo]: """Full model metadata by Hermes or models.dev provider ID (exact match, then case-insensitive), or None if not found. EXPLICIT ``model_overrides`` patch known catalog models; ``_default`` fills the gap - only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths.""" + only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths. + + ``model_overrides`` entries use the SAME canonical schema as every other consumer (``context_window``, + ``max_output_tokens``, ``supports_*``, ``model_family``) — they are translated into the catalog shape at + this boundary, and sub-dicts (``limit``, ``modalities``) are merged rather than clobbered. See #84482, + #8731. + """ mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id) models = _registry_models(mdev_id, allow_network=allow_network) mid, entry = next(_iter_model_entries(models, model_id, suffix_fallback=False), (model_id, None)) if models is not None else (model_id, None) diff --git a/agent/native_compaction.py b/agent/native_compaction.py index cfe2b184eb..0202770c4d 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -206,6 +206,18 @@ def prune_pre_checkpoint_items( that is itself a canonical summary carrier is read from the SOURCE and retained as a synthesized ``role="assistant"`` message. - ``enable_summary_retention`` is a test override, not a config surface. + + The server drops every input item that precedes a replayed ``compaction`` item (live-verified Aug 2026), + so sending pre-checkpoint history is dead weight AND silently erases the user's plaintext asks — + including any local-compression summary the agent already produced, which previously vanished here + because it carries ``role="assistant"``, not ``"user"`` (#90975). + A summary is never byte/character-sliced: Hermes summaries carry structural framing (handoff prefix, end + marker, merge-into-tail delimiters) that a blind slice can corrupt, so one that doesn't fit whole is + dropped instead. A summary already retained once (identical text) is never duplicated, so repeated + checkpoints stay idempotent. - ``enable_summary_retention`` is a function-level override (used by tests + and callers that need the pre-#90975 behavior back); it is not wired to a user-facing config surface. + Without ``item_sources`` (default), retention only sees what survived conversion, matching pre-#90976 + behavior (#90976). """ if not isinstance(items, list) or not items: return items @@ -241,6 +253,9 @@ def prune_pre_checkpoint_items( continue # Source-based detection sees past a lossy conversion; it only fires # when the source itself is a provenance-tagged summary carrier. + # Canonical source-based summary detection: reads the ORIGINAL chat message's own content, so it + # sees past a lossy conversion (a typed `function_call_output` wrapper, or a stale exact-replay + # message) that erased the summary from `item` itself (#90976). if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source): text = flatten_message_text(source.get("content")) _src_role = source.get("role") @@ -291,6 +306,13 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: Drives one-shot recovery (strip, disable for the session, retry), so matching is narrow: a transient 5xx that merely ECHOES the request must not downgrade native compaction. Requires ``status_code`` 400 (or unknown) AND the field name with rejection language. + + See #82777. + * ``status_code`` is 400 (or unknown/None — some transports surface only a message string; field-name + matching alone is then the best available signal, preserving pre-#82777 behavior for them), and * the + error text names ``context_management`` / ``compact_threshold`` alongside rejection language ("unknown", + "unsupported", "invalid", "unexpected", "not permitted"...). A bare field-name echo without rejection + language does not match. """ text = str(error or "").lower() if "context_management" not in text and "compact_threshold" not in text: diff --git a/agent/opencode_affinity.py b/agent/opencode_affinity.py index 4f0cc5522e..be355eb121 100644 --- a/agent/opencode_affinity.py +++ b/agent/opencode_affinity.py @@ -56,6 +56,18 @@ def opencode_session_headers( from agent.transports.codex import _cache_scope_from_session_id key = _cache_scope_from_session_id( + # Top-level session_id → OpenRouter's sticky routing key. Per their prompt-caching docs it is + # used directly as the routing key instead of hashing the opening messages, and it activates + # stickiness on the first successful request rather than only after a cache hit. Resolve it from + # the declared routing scope first (set only by a host that names its own conversation, #96811), + # then the ambient conversation contextvar, with the explicit argument as fallback. The gap this + # closes is the auxiliary call sites — compression, title generation, vision, web_extract, + # session_search, MoA slots — which funnel through ``agent.auxiliary_client``. That module has + # no session handle and passes no ``session_id``, so those calls sent NO sticky key at all and + # each routed independently of the conversation it belonged to (#70820). Mirrors the Nous Portal + # profile, which resolves the same way (f2f4df064d). The ambient value is the session-lineage + # ROOT, so it also stays stable for installs that opt out of the default ``compression.in_place: + # true`` and across delegate-subagent trees. get_affinity_scope() or get_conversation_context() or session_id ) except Exception: diff --git a/agent/outbound_webhooks.py b/agent/outbound_webhooks.py index 61c8aa0e1e..14cb117e8e 100644 --- a/agent/outbound_webhooks.py +++ b/agent/outbound_webhooks.py @@ -133,7 +133,12 @@ def flush(timeout: float = 5.0) -> bool: def re_register_config_hooks() -> None: """Re-register outbound webhooks after a plugin force-reload cleared ``_hooks``. Only the current home's idempotence keys are cleared so a force-reload in one profile cannot - invalidate another profile's still-live registration.""" + invalidate another profile's still-live registration. + + Mirrors ``agent.shell_hooks.re_register_config_hooks``: config-owned outbound-webhook callbacks live in + the same ``_hooks`` dict that ``PluginManager.discover_and_load(force=True)`` clears via ``unload()``, + so without this the force-reloaded profile's outbound webhooks go silently inert (#92682 review). + """ from hermes_cli.config import load_config _forget_home_registrations(_registered, _registered_lock) register_from_config(load_config()) @@ -235,6 +240,7 @@ def _serialize_payload(event: str, kwargs: Dict[str, Any], delivery_id: str) -> (also the ``X-Hermes-Delivery`` header) and ``timestamp`` live inside the HMAC-signed body, so they double as replay protection.""" # Profile resolved at fire time so a multiplexed gateway's receivers can tell which profile emitted. + # See #92674. from hermes_cli.profiles import get_active_profile_name payload = { "hook_event_name": event, "profile": get_active_profile_name(), **_payload_fields(kwargs), diff --git a/agent/plan_prompt.py b/agent/plan_prompt.py index d8030feea5..c9fdfa9996 100644 --- a/agent/plan_prompt.py +++ b/agent/plan_prompt.py @@ -58,7 +58,10 @@ Interaction style: def build_plan_prompt(task: str = "") -> str: - """Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context.""" + """Build the plan-mode prompt; empty *task* asks the agent to infer it from conversation context. + + See #36821. + """ task = (task or "").strip() task_block = f"Task to plan:\n{task}\n" if task else ( "No explicit task was given with /plan — infer the task from the " diff --git a/agent/plugin_llm.py b/agent/plugin_llm.py index d084d465e2..0fb46fd42a 100644 --- a/agent/plugin_llm.py +++ b/agent/plugin_llm.py @@ -211,7 +211,11 @@ def _check_task(policy: _TrustPolicy, *, plugin_id: str, requested_task: Optiona registered itself → allowed; a built-in key → only with ``allow_task_override``; anything else raises + logs. Never silently downgraded to ``auto``: that would mask the misconfiguration and could route to a main model the user steered - elsewhere on purpose.""" + elsewhere on purpose. + + A foreign/unknown key raises :class:`PluginLlmTrustError` and logs a warning naming the offending plugin + and key. See #64174, #64182. + """ task = (requested_task or "").strip() if not task or task.lower() == "auto": return None diff --git a/agent/process_bootstrap.py b/agent/process_bootstrap.py index 4d640dbe30..4416ec063f 100644 --- a/agent/process_bootstrap.py +++ b/agent/process_bootstrap.py @@ -209,6 +209,11 @@ def enable_happy_eyeballs_on_client(client) -> None: For callers that build clients inline (Codex OAuth/device-login). Proxy-backed pools are skipped (TCP connect goes to the proxy host); async clients need nothing (anyio already races per RFC 8305). Best-effort. + + Proxy-backed transports (``httpcore.HTTPProxy`` / SOCKS pools) are left untouched: with a proxy in play + the TCP connect goes to the proxy host, which is out of scope for the direct-transport racing added in + #94388. Async clients are also left untouched — httpcore's async backend already performs RFC 8305 + racing natively via ``anyio.connect_tcp(happy_eyeballs_delay=0.25)``. """ try: import httpcore @@ -312,6 +317,8 @@ def _shared_transport_cls(): mounted object absorbs that close while the shared pool keeps serving other clients. ``handle_request`` stamps the owning view into ``request.extensions`` so socket-abort sweeps target only this client's in-flight connections on the shared pool. + + See #10933. """ __slots__ = ("_inner", "_closed") @@ -391,6 +398,9 @@ def build_keepalive_http_client(base_url: str = "", *, async_mode: bool = False, ``HTTPTransport`` through a ``_SharedTransport`` view, so N delegated children share one connection pool + SSL context. Async clients are never shared: an httpcore async pool is bound to the event loop that first used it. Proxy-backed clients keep httpx's own transport. + + See #12952, #54049. + See #10933. """ try: import httpx diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 6fc87fee80..3b418e89af 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -150,6 +150,10 @@ HERMES_AGENT_HELP_GUIDANCE_NO_SKILLS = ( ) +# Memory guidance (#95681, consolidated): ONE block from ONE builder. The opening frame adapts to which +# stores config enables; everything else is written exactly once. Leads with the positive posture (save +# proactively, replace when full) — the routing rules come after, as refinements, not as the headline. WHAT +# belongs in memory is the memory tool schema's job and is never re-taught here. def build_memory_guidance(memory_enabled: bool = True, profile_enabled: bool = True) -> str: """ONE memory-guidance block whose opening frame adapts to the enabled store(s); "" when both are off. @@ -199,6 +203,17 @@ SESSION_SEARCH_GUIDANCE = ( # ("After completing a complex task (5+ tool calls)... save the approach as a skill...") on subscription OAuth # credentials, surfacing as a billing-shaped HTTP 400. If you rewrite it, re-verify with a subscription OAuth # token — sk-ant-api keys do not hit the filter. The safety-rule heading is referenced by tests and compaction summaries. +# Anthropic's server-side content filter rejects the previous phrasing ("After completing a complex task (5+ +# tool calls), fixing a tricky error, or discovering a non-trivial workflow, save the approach as a skill +# with skill_manage so you can reuse it next time.") on subscription OAuth credentials, and surfaces that +# rejection as a billing-shaped HTTP 400 ("You're out of extra usage"), which sends users to buy quota they +# do not need. Bisected against the live API: that sentence alone reproduces the 400 and removing it alone +# clears it; size and the system[0] identity gate were both ruled out. The reword is empirically validated, +# not understood — if you rewrite this sentence, re-verify against a subscription OAuth token, not an +# sk-ant-api… key, which does not hit the filter. Dieted (#95681, maintainer-directed): the record-it / +# patch-it coaching that used to open this block duplicated the ## Skills section (which teaches both "offer +# to save as a skill" and "fix it with skill_manage(action='patch')") and skill_manage's own schema. Only +# the compaction-pruning contract lives here — nothing else teaches it. SKILLS_GUIDANCE = ( "When you work out a non-trivial workflow, record it with skill_manage for future reuse.\n\n" "## Skill Safety Rule\n" @@ -308,6 +323,14 @@ TOOL_USE_ENFORCEMENT_MODELS = ("gpt", "codex", "gemini", "gemma", "grok", "glm", # traces showed the same failure modes; Muse Spark stops after a chat-only turn on defaults). Gemini/Gemma get # GOOGLE_MODEL_OPERATIONAL_GUIDANCE instead; Claude does not exhibit these modes. Any model can opt in via # config.yaml (`true` or a substring list). +# Model name substrings whose sessions receive OPENAI_MODEL_EXECUTION_GUIDANCE (execution discipline: tool +# persistence, mandatory tool use for arithmetic, external-write read-back, count reconciliation, literal +# preservation, verification-gated completion) when agent.execution_guidance is "auto". gpt/codex/grok are +# the historical set; deepseek/kimi/qwen/glm/minimax/ mimo/mistral were added after Composio agentic-eval +# traces showed the same failure modes on those families (financial math in prose, no read-back after +# external writes, identifier "repair", completeness claims despite count mismatches). GLM's +# tool-calls-as-plain-text stall (#53847) and MiMo (#41874) are covered here too. Gemini/Gemma are excluded +# — they get the more specific GOOGLE_MODEL_OPERATIONAL_GUIDANCE block instead. EXECUTION_GUIDANCE_MODELS = ( "gpt", "codex", "grok", "deepseek", "kimi", "qwen", "glm", "minimax", "mimo", "mistral", "muse", @@ -329,6 +352,20 @@ TASK_COMPLETION_GUIDANCE = ( # Universal parallel-tool-call guidance (ALL models): the runtime already executes independent calls # concurrently. Supersedes the former Google-only bullet so no model receives the steer twice. +# Why this matters for cost: every assistant turn resends the entire accumulated conversation (and, on +# cache-friendly providers, re-reads the cached prefix and pays for the newly-appended turn). A model that +# issues one tool call per turn multiplies the number of round-trips — and therefore the resent context — +# for any task that needs several independent reads, searches, or safe lookups. Batching independent calls +# into a single assistant response collapses N turns into one, cutting both latency and the resent-context +# cost that compounds over a long conversation. The hermes-agent runtime already executes a batch of tool +# calls concurrently when they are independent (read-only tools always; path-scoped file ops when their +# targets don't overlap — see run_agent._execute_tool_calls / tool_dispatch_helpers). The missing piece was +# telling the *model* to emit those calls together in the first place. Until now the only batching steer in +# the prompt lived in GOOGLE_MODEL_OPERATIONAL_GUIDANCE — Gemini/Gemma got it, every other model got +# nothing. Short on purpose — shipped in the cached system prompt to every user, every session. Token cost +# is paid once at install and amortised across all sessions via prefix caching. Keep it tight. Ported from +# cline/cline#11514 ("encourage parallel tool calls"), adapted from Cline's TypeScript tool-surface guidance +# to hermes-agent's Python prompt-assembly architecture. PARALLEL_TOOL_CALL_GUIDANCE = ( "# Parallel tool calls\n" "When you need several pieces of information that don't depend on each other, request them together in a " @@ -342,6 +379,15 @@ PARALLEL_TOOL_CALL_GUIDANCE = ( # Execution-discipline guidance for models that abandon partial results, skip prerequisite lookups, answer # from memory, or declare "done" unverified. Body is family-agnostic (OPENAI_ prefix reflects origin). # Injection gate: system_prompt.py via config.yaml ``agent.execution_guidance`` (auto/true/false/list). +# OpenAI GPT/Codex-specific execution guidance. Addresses known failure modes where GPT models abandon work +# on partial results, skip prerequisite lookups, hallucinate instead of using tools, and declare "done" +# without verification. Inspired by patterns from OpenAI's GPT-5.4 prompting guide & OpenClaw PR #38953. +# Also applied to xAI Grok — same failure modes in practice (claims completion without tool calls, suggests +# workarounds instead of using existing tools, replies with plans/suggestions instead of executing). As of +# the Composio agentic-eval follow-up, the block is no longer fenced to gpt/codex/grok: eval traces showed +# DeepSeek/Kimi doing financial math in prose, skipping read-back verification after external writes, +# "repairing" malformed identifiers, and claiming completeness despite count mismatches — exactly the +# failure modes this block targets. OPENAI_MODEL_EXECUTION_GUIDANCE = ( "# Execution discipline\n" "\n" @@ -462,6 +508,13 @@ def format_steer_marker(steer_text: str) -> str: STEER_CHANNEL_NOTE = ( # Only what the marker cannot say about itself: it is the ONLY trusted shape and carries full user authority. + # Dieted (#95681, maintainer-directed). History: #40240 added this note when the marker was bare and + # models refused steers as prompt injection (screenshot-verified). The marker has since become + # self-describing — it declares its own provenance ("a direct message from the user...") and its own + # replay rule ("not a new delivery when replayed from conversation history") at delivery time — so the + # prompt-side briefing keeps only what the marker cannot say about itself: it is the ONLY trusted shape + # (anti-lookalike), and it carries full user authority. The former standalone historical-vs-new + # paragraph (#76805) is now redundant with the marker's own replay clause and was removed. "## Mid-turn user steering\n" "Mid-turn, the user can steer you: Hermes appends their message to the end of a tool result, wrapped exactly as:\n" f"{STEER_MARKER_OPEN}\n\n{STEER_MARKER_CLOSE}\n" @@ -681,6 +734,12 @@ PLATFORM_HINTS = { # Telegram rich-messages extension — injected only with # ``platforms.telegram.extra.rich_messages: true`` (gateway.* or top-level). +# NOTE: a "webui" hint lived here until 2026-08-29. It was a ghost (verified in the all-platform hint audit, +# PR #97873): no code path constructs platform="webui" — the dashboard chat resolves to 'desktop' or 'tui' +# (tui_gateway/server.py:_resolve_session_platform), and the browser chat tab is an xterm.js PTY hosting the +# TUI, not an HTML chat renderer. Its content (tables/LaTeX/Mermaid, MEDIA: rich previews incl. Excalidraw) +# described a renderer that does not exist anywhere in web/. If a real WebUI chat surface ships, write a +# hint from its actual renderer — do not resurrect this text. TELEGRAM_RICH_MESSAGES_HINT = ( "Telegram now supports rich Markdown, so lean into it: whenever it makes the answer clearer or easier to scan, " "actively reach for real Markdown tables (pipe `| col | col |` syntax), bullet and numbered lists, task lists (`- " @@ -737,7 +796,13 @@ def _plugin_backend_is_remote(backend: str) -> bool: def _windows_marketing_version() -> str: - """"10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000).""" + """"10"/"11" (``platform.release()`` says 10 for both; 11 is build >= 22000). + + ``platform.release()`` reports the kernel version, which is ``10`` for BOTH Windows 10 and Windows 11 — + the prompt then claims "Windows (10)" on Windows 11 hosts and misleads the model about the OS (#51755). + Windows 11 is distinguished by build number: >= 22000 is 11. Falls back to ``platform.release()`` on any + lookup failure. + """ try: return "11" if sys.getwindowsversion().build >= 22000 else "10" # type: ignore[attr-defined] except Exception: @@ -974,6 +1039,9 @@ def drain_truncation_warnings() -> list: # Skills index (two-layer cache: in-process LRU, then disk snapshot). # One entry per profile × platform (key carries skills_dir); a multiplexing gateway needs more than a handful. +# Sized for multi-profile processes: since #86313 the cache key carries a per-profile skills_dir (one entry +# per profile × platform), so the old cap of 8 could thrash on a gateway multiplexing default + several bots +# (each miss = full os.walk manifest rebuild). ~32 costs low single-digit MB worst case. _SKILLS_PROMPT_CACHE_MAX = 32 _SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict() _SKILLS_PROMPT_CACHE_LOCK = threading.Lock() @@ -1361,6 +1429,11 @@ def load_soul_md(context_length: Optional[int] = None, home_override: "Path | No Callers must pass ``skip_soul=True`` to ``build_context_files_prompt`` so it isn't injected twice. ``home_override`` pins the profile home (a thread that lost the HERMES_HOME ContextVar reads the wrong one). + + ``home_override`` scopes the read to an explicit profile home (the agent knows its own home from its + session_db path). Without it, resolution is ambient — which on a thread that lost the HERMES_HOME + ContextVar falls back to the launch home and reads the wrong profile's SOUL.md (#50233, same class as + the skills-index leak fixed in #86313). """ try: from hermes_cli.config import ensure_hermes_home @@ -1423,6 +1496,14 @@ def _load_agents_md(cwd_path: Path, context_length: Optional[int] = None) -> str Per directory the first of ``AGENTS.override.md`` / ``AGENTS.md`` / ``agents.md`` wins (a gitignored personal override shadows the committed file); identical content seen again down the chain is skipped. + + Each directory on the chain (see ``_agents_md_directory_chain``) contributes its ``AGENTS.override.md`` + / ``AGENTS.md`` / ``agents.md`` (first name wins per directory) as its own provenance-labelled section. + ``AGENTS.override.md`` wins over ``AGENTS.md`` so a developer can keep a personal, typically-gitignored + override next to the committed project instructions without editing the tracked file (same convention as + earendil-works/pi#7681). Identical content encountered again further down the chain (copied or symlinked + files) is deduplicated. With a single match — the common case, and always the case outside a git repo — + output is identical to the historical single-file behavior. """ cwd_resolved = cwd_path.resolve() sections: list[str] = [] @@ -1483,6 +1564,10 @@ def build_context_files_prompt( cwd_path = Path(cwd if cwd is not None else os.getcwd()).resolve() # A FALLBACK-picked cwd inside the Hermes install tree must not gain system-prompt authority (the desktop # default would load this repo's contributor AGENTS.md). An explicit cwd is honored verbatim. + # An explicitly configured cwd is honored verbatim — the Hermes tree is a legitimate workspace when the + # user deliberately points a session at it — and CLI-style surfaces pass + # allow_install_tree_fallback=True because their launch dir IS the user's shell cwd (developing Hermes + # in-tree). See #64590. from agent.runtime_cwd import _is_install_tree if cwd is None and not allow_install_tree_fallback and _is_install_tree(cwd_path): logger.warning( diff --git a/agent/prompt_cache_scope.py b/agent/prompt_cache_scope.py index 18437a6b53..c8da04f88a 100644 --- a/agent/prompt_cache_scope.py +++ b/agent/prompt_cache_scope.py @@ -73,6 +73,8 @@ def _conversation_generation(session_key: str, source: str, session_db: Any) -> The declared key survives ``/new``, so hashing it alone would reuse one scope across conversations. The counter advances with each reset boundary, independent of prunable rows and wall-clock; compression does not advance it. + + The declared key names a chat and deliberately survives `/new` and policy resets. See #79017, #86733. """ reader = getattr(session_db, "latest_conversation_boundary", None) if not callable(reader): @@ -99,6 +101,11 @@ def declared_conversation_scope(agent: Any) -> Optional[str]: if sid and db is not None: try: # One read for both halves of the row identity (fork verdict + source). + # One read for both halves of the row's identity: the fork verdict and the source the peer + # queries match on live on the same ``sessions`` row, and asking for them separately read it + # twice per resolution (@teknium1 on #98811). A SessionDB without the combined view keeps the + # original call, so nothing that predates it — including the doubles that certify the + # fail-closed contract below — changes behaviour. identity = getattr(db, "declared_scope_identity", None) if callable(identity): is_fork, row_source = identity(sid) @@ -158,7 +165,14 @@ def declared_conversation_scope_safe(agent: Any) -> Optional[str]: def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]: """Never-raising variant of :func:`resolve_prompt_cache_scope` (None = use the physical id). - At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding.""" + At turn_context an exception inside ``set_runtime_main(...)`` would skip the whole binding. + + Returns None on any failure (or when there is no scope). Consumers treat None/empty as "fall back to the + physical session_id", so a resolution failure degrades to pre-#79017 behavior instead of blocking the + caller — important at turn_context's call site, where an exception raised inside the + ``set_runtime_main(...)`` argument list would otherwise skip the whole runtime binding, not just the + cache scope. + """ try: return resolve_prompt_cache_scope(agent) or None except Exception: diff --git a/agent/prompt_caching.py b/agent/prompt_caching.py index 63551c7167..6e511b1b3d 100644 --- a/agent/prompt_caching.py +++ b/agent/prompt_caching.py @@ -280,6 +280,12 @@ def apply_anthropic_cache_control( marker and the remaining two go to the latest cacheable non-system messages; otherwise the legacy system-and-3 layout applies. Idempotent: pre-existing markers are stripped from a per-message copy first. Returns a shallow list copy with deep copies of modified messages. + + Idempotent: pre-existing ``cache_control`` markers are stripped from a per-message copy before new ones + are placed, so calling this twice (or handing it messages a prior call already marked) can never + accumulate past 4 markers. Only messages that already carry a marker pay the copy cost — a shallow + top-level copy suffices because :func:`strip_anthropic_cache_control` is copy-on-write on content parts + — and the rest of the copy-on-write contract is unchanged (#90971). """ if not api_messages: return api_messages diff --git a/agent/proxy_sources/iron_proxy.py b/agent/proxy_sources/iron_proxy.py index f0b4c9f342..f7c76df9d4 100644 --- a/agent/proxy_sources/iron_proxy.py +++ b/agent/proxy_sources/iron_proxy.py @@ -71,6 +71,13 @@ _BEARER_PROVIDERS: Dict[str, Tuple[str, ...]] = { # two require-rules on one host would reject each other's requests; the sandbox gets the token # under every name. Authorization is also matched for Anthropic/Azure (SDKs may send Bearer); # Gemini's ``?key=`` style is covered by match_query. +# Providers whose API authenticates with a NON-Authorization header. iron-proxy v0.39's +# ``secrets.replace.match_headers`` targets arbitrary header names (case-insensitive; confirmed by the +# iron-proxy author on PR #30179 and verified in the pinned v0.39.0 source — ``swapHeaders`` + +# ``parseHeaderMatchers``), so these are first-class swapped providers, not "uncovered". ``aliases`` are +# interchangeable env-var names for the SAME upstream credential (Hermes' auth.py keys Google on both +# GEMINI_API_KEY and GOOGLE_API_KEY). The sandbox receives the minted token under the canonical name AND +# every alias so SDKs reading either work. _HEADER_AUTH_PROVIDERS: Dict[str, Dict[str, Tuple[str, ...]]] = { "ANTHROPIC_API_KEY": {"hosts": ("api.anthropic.com",), "match_headers": ("x-api-key", "Authorization"), "aliases": ()}, "AZURE_OPENAI_API_KEY": {"hosts": ("*.openai.azure.com", "*.cognitiveservices.azure.com", "*.services.ai.azure.com"), diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index bad9ca9387..ea196a68a3 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -16,6 +16,7 @@ import re from typing import Optional, Sequence #: Matches ``k3`` as a delimited token (``k3``, ``k3-256k``, ``kimi-k3-cot``), never K2-era names (``kimi-k2.6``). +# From #76427 by @ruizanthony. _KIMI_K3_SLUG_RE = re.compile(r"(?:^|[^a-z0-9])k3(?:[^a-z0-9]|$)") # Canonical low→high ordering for nearest-level clamping. Includes "none" so an explicit @@ -32,6 +33,7 @@ CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh" CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh") #: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high. +# : Backward-compat alias (pre-#68365-verification name). XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh") XAI_LEGACY_EFFORTS: tuple[str, ...] = ("low", "medium", "high") @@ -59,6 +61,9 @@ SOLAR_EFFORTS: tuple[str, ...] = ("low", "medium", "high") #: widens it to a graded scale (live-verified, monotonic). ``xhigh`` requests the top tier. GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} +# : GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified : live on +# api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every : level accepted with monotonic +# reasoning-token scaling (low=4, medium=11, : high=98, max=125 on the probe prompt). GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} @@ -80,7 +85,12 @@ def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]: def kimi_supported_efforts(model: Optional[str]) -> tuple[str, ...]: - """Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3).""" + """Supported effort set for a Moonshot/Kimi slug (bare ``k3``, ``k3-256k``, ``kimi-k3*`` → K3). + + K3 is served as the bare slug ``k3``, plan variants like ``k3-256k``, and the ``kimi-k3*`` aliases; its + documented set is low/high/max. Everything earlier speaks low/medium/high. Boundary-matched so K2-era + names (``kimi-k2.6``) never match (detection regex from #76427 by @ruizanthony). + """ m = (model or "").strip().lower().split("/")[-1] return KIMI_K3_EFFORTS if _KIMI_K3_SLUG_RE.search(m) else KIMI_K2_EFFORTS diff --git a/agent/reasoning_params.py b/agent/reasoning_params.py index 3186f072cb..973fb64ab1 100644 --- a/agent/reasoning_params.py +++ b/agent/reasoning_params.py @@ -130,7 +130,12 @@ class ReasoningParamsMixin: def _needs_thinking_reasoning_pad(self) -> bool: """True when the provider enforces ``reasoning_content`` echo-back on tool-call replays (DeepSeek, Kimi, MiMo thinking all 400 without it). Cached per (provider, model, base_url), invalidated by - ``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn.""" + ``switch_model()`` / ``_try_activate_fallback()`` — called ~16× per turn. + + DeepSeek v4 thinking and Kimi / Moonshot thinking both reject replays of assistant tool-call + messages that omit ``reasoning_content`` (refs 15250, #17400). Xiaomi MiMo thinking mode has the + same requirement. + """ key = (self.provider, self.model, getattr(self, "_base_url_lower", self.base_url)) cached = getattr(self, "_thinking_pad_cache", None) if cached is not None and cached[0] == key: @@ -162,7 +167,11 @@ class ReasoningParamsMixin: return matches_reasoning_echo_family("kimi", self.provider, None, self.base_url) def _needs_deepseek_tool_reasoning(self) -> bool: - """True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400).""" + """True when the current provider is DeepSeek thinking mode (omitting the echo is an HTTP 400). + + DeepSeek V4 thinking mode requires ``reasoning_content`` on every assistant tool-call turn; omitting + it causes HTTP 400 when the message is replayed in a subsequent API request (#15250). + """ return matches_reasoning_echo_family("deepseek", (self.provider or "").lower(), self.model, self.base_url) def _needs_mimo_tool_reasoning(self) -> bool: diff --git a/agent/redact.py b/agent/redact.py index d2a1d22ebe..34a8df803f 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -19,6 +19,8 @@ logger = logging.getLogger(__name__) # Sensitive query-string param names (case-insensitive): opaque tokens / OAuth # codes / pre-signed signatures with no vendor prefix. +# Ported from nearai/ironclaw#2529 — catches tokens whose values don't match any known vendor prefix regex +# (e.g. opaque tokens, short OAuth codes). _SENSITIVE_QUERY_PARAMS = frozenset({ "access_token", "refresh_token", "id_token", "token", "api_key", "apikey", "client_secret", "password", "auth", "jwt", "session", "secret", "key", @@ -28,6 +30,11 @@ _SENSITIVE_QUERY_PARAMS = frozenset({ # Snapshot at import time so runtime env mutations (e.g. an LLM-generated # `export HERMES_REDACT_SECRETS=false`) cannot disable redaction mid-session. # ON by default; `security.redact_secrets: false` bridges to this env var. +# ON by default — secure default per issue #17691. Users who need raw credential values in tool output (e.g. +# working on the redactor itself) can opt out via `security.redact_secrets: false` in config.yaml (bridged +# to this env var in hermes_cli/main.py, gateway/run.py, and cli.py) or `HERMES_REDACT_SECRETS=false` in +# ~/.hermes/.env. An opt-out warning is logged at gateway and CLI startup so operators see the downgrade — +# see `_log_redaction_status()` in gateway/run.py and cli.py. _REDACT_ENABLED = os.getenv("HERMES_REDACT_SECRETS", "true").lower() in {"1", "true", "yes", "on"} # Known API key prefixes -- match the prefix + contiguous token chars. @@ -76,6 +83,7 @@ _PREFIX_PATTERNS = [ r"fw_[A-Za-z0-9]{30,}", # Fireworks AI API key r"fpk_[A-Za-z0-9]{30,}", # Fireworks AI project key # GitLab token families (each keeps a full literal prefix for the pre-screen). + # Ported from openclaw/openclaw#112954; follow-up invited in #4541. r"glpat-[A-Za-z0-9_\-]{10,}", # GitLab personal access token r"gloas-[A-Za-z0-9_\-]{10,}", # GitLab OAuth application secret r"gldt-[A-Za-z0-9_\-]{10,}", # GitLab deploy token @@ -97,12 +105,17 @@ _PREFIX_PATTERNS = [ # tolerate spaces around "=" and allow the keyword embedded anywhere # (``MYTOKEN=…``) — an all-caps key is almost never prose. Bare ``KEY``/``PASS``/ # ``PW`` suffixes are included; _key_has_secret_keyword rejects ``KEYBOARD=``. +# The regex is IGNORECASE so lowercase env names (``openai_key=…``) are caught here too. The secret name +# must sit at a word boundary (``_``-delimited or whole-word) so generic prose words (``password=``, +# ``token=``, ``KEYBOARD=``, ``PASSAGE=``) do not match — those are handled by the config/form/URL paths, +# and a bare ``password=…`` in a form body must not be swallowed greedily by ``\S+``. See #77484. _SECRET_ENV_NAMES = r"(?:API_?KEY|KEY|TOKEN|SECRET|PASSWORD|PASSWD|PASS|PW|CREDENTIAL|AUTH)" _ENV_ASSIGN_RE = re.compile(rf"([A-Z0-9_]{{0,50}}{_SECRET_ENV_NAMES}[A-Z0-9_]{{0,50}})\s*=\s*(['\"]?)(\S+)\2") # Lowercase env names: only underscore-boundary forms (``openai_key=``) — NOT # bare ``password=``/``token=``, which appear in prose, URLs, and form bodies. # The lookbehind anchors each attempt to the start of an identifier run; without # it re.sub retries the greedy prefix at every byte of a long opaque payload. +# See #77484. _ENV_ASSIGN_LOWER_RE = re.compile( rf"(? neither can match. @@ -158,6 +179,16 @@ _YAML_ASSIGN_RE = re.compile( # only at a word boundary: key edge, next to a non-letter, or a camelCase # transition (``clientSecret``, ``APIToken``); trailing plural ``s`` is part of # it. Concatenations match via explicit alternatives (``authtoken``, ``apikey``). +# The side effect: ordinary prose/document words that merely CONTAIN a keyword also matched — ``Secretary: +# J.Smith`` (secret), ``tokenizer: cl100k_base`` (token), ``author=Smith`` (auth) — mangling legitimate +# content on the surfaces that run these passes (browser snapshots, log lines, kanban summaries, CLI-echoed +# command output). Ported from nearai/ironclaw#6129, where the same substring false positive ("Secretary of +# the Treasury" matching the ``secret`` marker) scrubbed legitimate tool results from the replayed +# transcript and sent the model into a re-fetch loop. Common concatenated compounds keep matching via +# explicit alternatives (``authtoken`` ngrok, ``authkey`` tailscale, ``secretkey`` minio, ``apikey``). +# Embedded occurrences inside a larger word (``secretary``, ``tokenizer``, ``authored``, ``credentialing``) +# no longer match. ALL-CAPS keys keep the legacy embedded matching (``MYTOKEN=…``) — an all-caps key is +# almost never prose, the same rationale as _ENV_ASSIGN_RE. _KEY_KEYWORD_RE = re.compile( r"(?:api|auth|access|refresh|session|secret)[ _.\\-]?(?:key|token)" r"|token|secret|passwd|password|pass|pw|credential|auth|key", @@ -223,6 +254,12 @@ def _should_redact_assignment(key: str, value: str, *, check_keyword: bool) -> b """Shared gate for the ENV / JSON / YAML assignment passes: skip programmatic env lookups used as values, optionally require a word-bounded keyword in the key, then redact when the key is unambiguously credential-bearing or the value looks opaque.""" + # Programmatic env lookups reference variable *names*, not secret values — masking them corrupts code + # snippets in prose/log contexts (issue #2852): ``KEY=os.getenv('X')``. + # Same programmatic-env-lookup exception as _redact_env above (issue #2852): "apiKey": "os.getenv('X')" + # is a code snippet, not a leaked secret value. + # Same programmatic-env-lookup exception as _redact_env above (issue #2852): api_key: os.getenv('X') is + # a code snippet, not a leaked secret value. if _ENV_LOOKUP_VALUE_RE.match(value): return False if check_keyword and not _key_has_secret_keyword(key): @@ -254,6 +291,11 @@ _PRIVATE_KEY_RE = re.compile(r"-----BEGIN[A-Z ]*PRIVATE KEY-----[\s\S]*?-----END # Database connection strings: protocol://user:PASSWORD@host. The userinfo and # password groups forbid whitespace so a match can never span a line break (a # greedy ``[^@]+`` once ran to a decorator's ``@`` on the next code line). +# Database connection strings: protocol://user:PASSWORD@host Catches postgres, mysql, mongodb, redis, amqp +# URLs and redacts the password. A real DSN password never contains whitespace; without this bound the +# greedy [^@]+ would scan past the end of a code line to the next stray "@" (e.g. a Python decorator), +# swallowing intervening lines and corrupting tool OUTPUT for any source containing a postgresql:// f-string +# template. See issue #33801. _DB_CONNSTR_RE = re.compile( r"((?:postgres(?:ql)?|mysql|mongodb(?:\+srv)?|redis|amqp)://[^:\s]+:)([^@\s]+)(@)", re.IGNORECASE, @@ -265,6 +307,11 @@ _DB_CONNSTR_RE = re.compile( # bare userinfo. ``user:pass@`` passes through (class forbids ``:``); DB schemes # belong to _DB_CONNSTR_RE. 8+ char floor skips short usernames; the class # forbids ``/`` so an ``@`` in a path/query (``?q=user@example.com``) never counts. +# This is the ``git remote set-url origin https://PASSWORD@github.com/...`` shape from issue #6396 — a +# single opaque credential in the userinfo position with NO ``user:pass`` colon. The colon form +# ``user:pass@`` is deliberately left to pass through (commit "pass web URLs through unchanged", #34029) and +# is NOT matched here — the token class forbids ``:``. DB schemes are handled by _DB_CONNSTR_RE above and +# excluded here. Guards against false positives: _URL_BARE_TOKEN_RE = re.compile( r"((?:https?|wss?|git|ssh|ftp|ftps|sftp)://)" # scheme r"([^\s:@/]{8,})" # bare token (no colon/slash/@), 8+ chars @@ -320,6 +367,10 @@ def _mask_control_split_tokens(text: str, mask_fn) -> str: Match on a control-stripped copy, then mask the corresponding span in the ORIGINAL — only when that span holds solely token-body and control chars, so a match can never cross into another line's unrelated text. + + A credential like ``sk-abc\\x1bdef456…`` or ``ghp_abc\\n123def…`` has its token body interrupted, so the + contiguous _PREFIX_RE cannot match it and the secret leaks verbatim (issue #77484). ``EXA_API_KEY=*** is + rejected). """ stripped = _CONTROL_CHARS_RE.sub("", text) if stripped == text: @@ -428,7 +479,12 @@ def _redact_form_body(text: str) -> str: def _mask_token_nonreusable(token: str) -> str: """Redact a prefix-matched credential to a NON-REUSABLE sentinel: no head/tail chars (an agent once wrote a truncated-looking mask back into a config file), - only the vendor prefix label so the credential KIND stays visible.""" + only the vendor prefix label so the credential KIND stays visible. + + * cannot be mistaken for a usable-but-truncated key, so an agent that reads it from a config file and + writes it back does NOT corrupt the stored credential into a dead 13-char string (issue #35519); and * + still does not leak the secret material (no head/tail chars). + """ label = next((sub for sub in _PREFIX_SUBSTRINGS if token.startswith(sub)), "") if token else "" return f"«redacted:{label}…»" if label else "«redacted-secret»" @@ -451,9 +507,20 @@ def _redact_assignments(text: str) -> str: _redact_env = _assignment_sub(lambda g: f"{g[0]}={g[1]}{_mask_token(g[2])}{g[1]}", check_keyword=True) text = _ENV_ASSIGN_RE.sub(_redact_env, text) if "://" not in text: # lowercase names would match URL params + # Skip URLs — the query string may contain ``token=``/``key=`` params that are intentionally + # passed through (see note near the bottom of this function; _redact_strict_url_credentials + # handles the opt-in case). The uppercase regex above is all-caps-only, so it never matches URL + # params; the lowercase one would (issue #77484). text = _ENV_ASSIGN_LOWER_RE.sub(_redact_env, text) # The keyword pre-gate is exact and matters: _CFG_DOTTED_RE backtracks # quadratically on long unbroken [A-Za-z0-9_.\-] runs. + # Lowercase/dotted config keys (issue #16413). Skip URLs entirely — web-URL query params are + # intentionally passed through (see note near the bottom of this function); _DB_CONNSTR_RE still + # guards connection-string passwords. Extra gate: every _CFG_*_RE match requires a secret keyword in + # the key, so a text without any secret keyword cannot match — skipping is exact. This matters + # because _CFG_DOTTED_RE backtracks quadratically on long unbroken [A-Za-z0-9_.\-] runs (e.g. + # base64/hex blobs in compaction payloads); the linear keyword scan prevents that pathological path + # on secret-free text. if "://" not in text and _CFG_SECRET_WORD_RE.search(text): text = _CFG_DOTTED_RE.sub(_redact_env, text) text = _CFG_ANCHORED_RE.sub(_redact_env, text) @@ -507,6 +574,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F Every regex sits behind a cheap substring gate that its pattern requires, so the gates are never false-negative. + + Set file_read=True for file *content* returned to the agent (read_file / search_files / cat). The old + mask looked like a real-but-truncated key, so an agent reading it from config.yaml and writing it back + silently corrupted the stored credential into a dead 13-char value → 401 (issue #35519). The sentinel is + syntactically invalid as a token, so it can't be mistaken for a usable key or written back as one. """ if text is None: return None @@ -518,6 +590,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F # Control/zero-width chars can split a token body so _PREFIX_RE alone misses it. if _has_known_prefix_substring(text): _prefix_sub = _mask_token_nonreusable if file_read else _mask_token + # Control/zero-width chars (\\n, \\r, ESC, U+200B, …) split a token body so _PREFIX_RE cannot match + # across them — a secret smuggled as ``sk-abc\\x1bdef…`` leaks verbatim (issue #77484). Mask such + # runs by first matching on a control-stripped copy, then re-masking the corresponding span in the + # original (the stripped copy and the original are aligned 1:1 for non-control chars). text = _mask_control_split_tokens(text, _prefix_sub) text = _PREFIX_RE.sub(lambda m: _prefix_sub(m.group(1)), text) @@ -534,6 +610,11 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F if "BEGIN" in text and "-----" in text: text = _PRIVATE_KEY_RE.sub("[REDACTED PRIVATE KEY]", text) + # Database connection string passwords. With code_file=True, a password group that is a pure ``{...}`` + # brace expression is an f-string template reference (e.g. f"postgresql://{user}:{pass}@{host}"), not a + # literal credential — preserve it. Literal passwords are still redacted. The regex forbids whitespace + # in the password group, so a single-line template's group(2) is exactly the brace expression. See issue + # #33801. if "://" in text: text = _redact_url_credentials(text, code_file) @@ -541,6 +622,15 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F text = _JWT_RE.sub(lambda m: _mask_token(m.group(0)), text) if redact_url_credentials: # opt-in; known credential shapes in URLs are caught above + # NOTE: Web-URL redaction (query params + userinfo + HTTP access-log request targets) is + # intentionally OFF. Many legitimate workflows pass opaque tokens through query strings — magic-link + # checkouts, OAuth callbacks the agent is meant to follow, pre-signed share URLs — and + # blanket-redacting param values by name breaks those skills mid-flow. DB connection-string + # passwords are still caught by _DB_CONNSTR_RE. The ONE userinfo case still redacted is the + # colon-less bare-token form ``scheme://TOKEN@host`` (#6396, handled by _URL_BARE_TOKEN_RE in the + # ``://`` block above): a bare credential in userinfo is never a round-trip workflow token (those + # live in the query string), so masking it can't break a skill. The ``user:pass@`` form is left to + # pass through per #34029. text = _redact_strict_url_credentials(text) if "&" in text and "=" in text: @@ -555,6 +645,10 @@ def redact_sensitive_text(text: str, *, force: bool = False, code_file: bool = F # Commands whose stdout is an env-var dump: terminal redaction runs the # ENV-assignment pass (code_file=False) for these so opaque tokens with no vendor # prefix are masked; everything else uses code_file=True (``MAX_TOKENS=100``). +# Commands whose stdout is an environment-variable dump (KEY=value lines), NOT source code. +# ``MY_SERVICE_TOKEN=abc123randomstring``) are still masked. For all other commands, code_file=True is used +# to avoid mangling legitimate source/config dumps (``MAX_TOKENS=100``, ``"apiKey": "x"`` fixtures, +# ``postgresql://{user}`` f-string templates). See issue #43025. _ENV_DUMP_COMMANDS = frozenset({"env", "printenv", "set", "export", "declare"}) # Commands that read file contents to stdout. A ``.env`` target is a credential @@ -702,6 +796,8 @@ def _has_known_prefix_substring(text: str) -> bool: # ADDITIVE-ONLY: a plugin can extend what gets masked but cannot weaken a # built-in, so it can only over-redact. Keyed by registration source so plugin # unload has a clean seam to drop ONE plugin's patterns. +# There is deliberately no public removal API — additive-only stands; unload is a host-owned lifecycle +# concern. See #64229. _PLUGIN_PREFIX_PATTERNS: dict = {} _registry_lock = threading.Lock() diff --git a/agent/relay_llm.py b/agent/relay_llm.py index d75f5ae8ff..d61b2d84f8 100644 --- a/agent/relay_llm.py +++ b/agent/relay_llm.py @@ -98,6 +98,10 @@ class _ManagedAttempt: Relay can invoke callbacks while another still owns the captured Context (hence the copy); nested relay calls run unmanaged — see relay_runtime.managed_callback_guard.""" def guarded() -> Any: + # See #77244. + # See #77244. + # Hermes-side callbacks run while the native pipeline drives this stream; nested relay calls + # they make must bypass managed execution (#77244). with relay_runtime.managed_callback_guard(): return callback(*args) @@ -223,7 +227,14 @@ def stream_current( With ``completed_response_predicate`` set, a factory that ignores ``stream=True`` and returns a complete response is unwrapped and returned directly (pre-Relay behavior). Detecting that primes the lazy pipeline: a genuine first chunk is buffered, but provider latency and pre-first-yield - errors may surface before this returns.""" + errors may surface before this returns. + + AnthropicAuxiliaryClient and other shims that ignore ``stream=True``), unwrap and return the completed + response directly. This mirrors the pre-Relay behavior where ``call_llm(stream=True)`` returned the raw + response and the consumer's own ``hasattr(stream, "choices")`` check handled it (#11732, #55933) — + without the unwrap the response stays trapped as ``final_response`` on the inner ManagedLlmStream and + the outer consumer sees an empty stream. + """ session_id = _current_session_id() # Inside a managed callback (on the Relay session's loop) a nested ManagedLlmStream would be # iterated synchronously on that loop, which asyncio forbids; the outer stream tracks this attempt. diff --git a/agent/relay_runtime.py b/agent/relay_runtime.py index abc0b68bae..3102f61fed 100644 --- a/agent/relay_runtime.py +++ b/agent/relay_runtime.py @@ -1139,6 +1139,13 @@ def resolve_execution_context(session_id: str) -> tuple[RelayRuntime | None, Rel # Nested managed execution is impossible (see _MANAGED_CALLBACK_DEPTH); the outer scope # still records the tool-level event. if _MANAGED_CALLBACK_DEPTH.get() > 0 or not relay_instrumentation_enabled(): + # A managed Relay callback is already executing on this logical call path (e.g. the native + # ``tools.execute`` pipeline is mid-dispatch of a Hermes tool). Nested managed execution here is + # structurally impossible: the native pipeline binds its Futures to the OUTER call's event loop, + # which is blocked inside the synchronous tool callback until the tool returns. A nested managed LLM + # call (the vision_analyze auxiliary path) therefore awaits a foreign-loop Future that can never + # complete — "attached to a different loop" at best, deadlock at worst, and "Event loop is closed" + # during shutdown when the orphaned Future is completed late (#77244). return None, None, None turn = active_turn(session_id) host = turn.lease.live_runtime() if turn is not None else None diff --git a/agent/relay_tools.py b/agent/relay_tools.py index bba27a7db8..ce00788b99 100644 --- a/agent/relay_tools.py +++ b/agent/relay_tools.py @@ -30,6 +30,7 @@ def execute( # Everything the tool transitively calls (incl. auxiliary LLM calls on worker # threads) must bypass managed Relay: the pipeline's Futures bind to THIS loop, # which is blocked until the tool returns. + # See #77244. with relay_runtime.managed_callback_guard(): return callback(final_args) diff --git a/agent/repetition_guard.py b/agent/repetition_guard.py index e2aa371a27..6b536de629 100644 --- a/agent/repetition_guard.py +++ b/agent/repetition_guard.py @@ -25,7 +25,11 @@ _DOMINANCE_RATIO = 0.5 def is_repetition_dominated(text: str) -> bool: """True when a single 60+ char substring recurs often enough to cover at least half - of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input.""" + of ``text`` — the signature of a repetition loop. Fail-open for non-string/short input. + + That shape is the signature of a model repetition loop (issue #86581), and continuing such a fragment is + pointless — the continuation nudge would just stitch more repeated text into the final response. + """ if not isinstance(text, str): return False n = len(text) diff --git a/agent/replay_cleanup.py b/agent/replay_cleanup.py index da1c78ff07..7af23b143e 100644 --- a/agent/replay_cleanup.py +++ b/agent/replay_cleanup.py @@ -95,7 +95,12 @@ def strip_interrupted_tool_tails(agent_history: List[Dict[str, Any]]) -> List[Di def strip_dangling_tool_call_tail(agent_history: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """Strip a trailing ``assistant(tool_calls)`` with NO answers — a call that killed the gateway itself (``docker restart``) left zero ``tool`` rows, invisible to ``strip_interrupted_tool_tails``. A partially - answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results.""" + answered block still resumes. Read-only tails are dropped; side-effecting ones get UNKNOWN-effect results. + + On resume the model sees an unanswered tool call at the tail and naturally re-issues it — which restarts + the gateway again, producing the infinite reboot loop in #49201. ``strip_interrupted_tool_tails`` does + not catch this because there is no tool result to inspect for an interrupt marker. + """ if not agent_history: return agent_history last = agent_history[-1] @@ -124,6 +129,8 @@ def sanitize_replay_history(agent_history: List[Dict[str, Any]]) -> List[Dict[st # --- Stale dangerous-confirmation text expiry --- # Short on purpose: a dangerous confirmation must not survive any restart or resume gap. +# ────────────────────────────────────────────────────────────────────── Stale dangerous-confirmation text +# expiry (#59607) ────────────────────────────────────────────────────────────────────── _DANGEROUS_CONFIRMATION_EXPIRY_SECONDS = 60.0 # Phrases that unlock destructive host actions; case-insensitive substring match so trailing punctuation / @@ -155,7 +162,13 @@ def strip_stale_dangerous_confirmations( ) -> List[Dict[str, Any]]: """Redact IN PLACE dangerous-confirmation text older than ``expiry_seconds`` in user messages: a confirmation surviving a restart reads as a fresh re-confirmation minutes later. Untimestamped messages (legacy - transcripts, test scaffolding) are left untouched.""" + transcripts, test scaffolding) are left untouched. + + See #59607. + On the next inbound message — possibly a casual "are you there?" from the user minutes later — the LLM + sees the stale confirmation and may interpret the new turn as a fresh re-confirmation, re-executing the + destructive action. This is the failure mode reported in #59607. + """ if not agent_history: return agent_history cleaned: List[Dict[str, Any]] = [] diff --git a/agent/secret_scope.py b/agent/secret_scope.py index 9831aadb2c..d7b85a6e09 100644 --- a/agent/secret_scope.py +++ b/agent/secret_scope.py @@ -76,6 +76,7 @@ _GLOBAL_ENV_EXACT = frozenset({ # API-server LISTENER settings — deployment config (compose/systemd env), # which the scoped runner reload must keep seeing or containers silently # lose the api_server platform. API_SERVER_KEY is a credential: NOT here. + # See #64674, #69379. "API_SERVER_ENABLED", "API_SERVER_HOST", "API_SERVER_PORT", "API_SERVER_CORS_ORIGINS", # Relay-connector ROUTING stamps injected by managed deploys. Every reader diff --git a/agent/secret_sources/bitwarden.py b/agent/secret_sources/bitwarden.py index 287e6170e1..a219a30d13 100644 --- a/agent/secret_sources/bitwarden.py +++ b/agent/secret_sources/bitwarden.py @@ -242,6 +242,8 @@ def _b64e(raw: bytes) -> str: def _derive_encrypted_cache_key(access_token: str, salt: bytes) -> bytes: """HKDF the local cache key from the bootstrap BWS token. cryptography is imported lazily: eagerly mapping ``_rust.pyd`` on Windows blocks the updater replacing it.""" + # Keep the native cryptography extension lazy. Most CLI commands import this module while building + # argparse, even though only encrypted-cache reads/writes need it. See #73381. from cryptography.hazmat.primitives import hashes from cryptography.hazmat.primitives.kdf.hkdf import HKDF diff --git a/agent/secret_sources/registry.py b/agent/secret_sources/registry.py index 936a278c99..9294e771b5 100644 --- a/agent/secret_sources/registry.py +++ b/agent/secret_sources/registry.py @@ -174,7 +174,12 @@ def list_sources(*, scope: Optional[str] = None) -> List[SecretSource]: def list_plugin_sources() -> List[SecretSource]: """Sources registered outside the bundled set: global ``"plugin"`` origins - plus every scoped registration (bundled sources register with scope=None).""" + plus every scoped registration (bundled sources register with scope=None). + + Includes both legacy global plugin registrations (``_SOURCE_ORIGINS == "plugin"``) and the current + scope's profile-keyed registrations — every scoped entry is plugin-registered by definition, since + bundled sources register with ``scope=None`` (#64229 profile isolation). + """ _ensure_builtin_sources() with _REGISTRY_LOCK: merged = {n: s for n, s in _SOURCES.items() if _SOURCE_ORIGINS.get(n) == "plugin"} @@ -374,6 +379,9 @@ def apply_all(secrets_cfg: dict, home_path: Path, Profile aliasing: under a named profile an applied ``FOO_`` (credential-shaped suffixes only) also hydrates canonical ``FOO``, under the same guards; disabled with ``secrets.profile_alias: false``. + + 1. 2. 3. 4. See #58073. + See #51447. """ env = environ if environ is not None else os.environ report = ApplyReport() diff --git a/agent/session_activity.py b/agent/session_activity.py index 639b21271b..ddd04c3ab7 100644 --- a/agent/session_activity.py +++ b/agent/session_activity.py @@ -22,6 +22,7 @@ class ActivityProvenance(str, Enum): UNKNOWN = "unknown" # Compression writers: heartbeat, host timeout, cooldown, turn hold. + # See #72424. AGENT_COMPRESSION = "agent.compression" AGENT_COMPRESSION_TIMEOUT = "agent.compression_timeout" AGENT_COMPRESSION_COOLDOWN = "agent.compression_cooldown" diff --git a/agent/session_persistence.py b/agent/session_persistence.py index 0454e83df0..3a392e4a55 100644 --- a/agent/session_persistence.py +++ b/agent/session_persistence.py @@ -309,7 +309,12 @@ class SessionPersistenceMixin: def _persist_session(self, messages: List[Dict], conversation_history: List[Dict] = None): """Save to JSON log and SQLite on any exit path. Trailing empty-response scaffolding is dropped from - the live list; the persist override is applied to the DB row only.""" + the live list; the persist override is applied to the DB row only. + + The persist user-message *override* is NOT applied here — it is resolved inside + ``_flush_messages_to_session_db`` and written only to the DB row, never mutating the live message + list used by the API call (#48677 is thus closed for every persist caller, not just this one). + """ from agent.agent_runtime_helpers import note_turn_persisted with _persist_lock(self): self._drop_trailing_empty_response_scaffolding(messages) @@ -356,7 +361,15 @@ class SessionPersistenceMixin: """Persist un-flushed messages to SQLite. Dedup is the intrinsic ``_DB_PERSISTED_MARKER`` on each written dict — not positional slices (drift after sequence repair) nor an ``id(msg)`` set (address reuse). The persist override touches the written row only. A compression-closed session adopts its live tip and - retries exactly once.""" + retries exactly once. + + Deduplicates via an intrinsic ``_DB_PERSISTED_MARKER`` stamped on each written message dict, so + repeated calls (from multiple exit paths) only write truly new messages — preventing the + duplicate-write bug (#860) without relying on positional slices that can drift after + message-sequence repair, and without a retained ``id(msg)`` set that CPython could alias onto a + freed-then-reused address (#50372). The ``_flushed_db_message_ids`` attribute is now only a one-shot + seed (translated to markers, then cleared each flush), not a persisted set. + """ # Persistence-isolated agents (background review fork) share the parent's session_id for cache warmth; # a write here would land the curator's turn in the user's real history. if getattr(self, "_persist_disabled", False) or not self._session_db: diff --git a/agent/shell_hooks.py b/agent/shell_hooks.py index 1cf992436b..e552740cf6 100644 --- a/agent/shell_hooks.py +++ b/agent/shell_hooks.py @@ -181,7 +181,18 @@ def iter_configured_hooks(cfg: Optional[Dict[str, Any]]) -> List[ShellHookSpec]: def re_register_config_hooks() -> None: """Re-register after a plugin force-reload cleared the manager's hooks; only this home's keys - are cleared (profile A's reload never drops B), never re-prompts.""" + are cleared (profile A's reload never drops B), never re-prompts. + + ``PluginManager.discover_and_load(force=True)`` unloads via the ownership ledger and clears the + manager's ``_hooks`` dict, which silently drops shell hooks that were registered from ``config.yaml`` at + startup (they are config-owned, not plugin-owned, so the ledger cannot restore them). Clear the + idempotence set and re-run ``register_from_config()`` so hooks are wired again (#60036 / PR #60267; + tracking #64178 — salvaged from PR #64188). + Only the idempotence keys for the *current* Hermes home are cleared — ``discover_and_load(force=True)`` + only unloads the manager scoped to that one home, so clearing every home's keys would make a + force-reload in profile A drop profile B's still-live registration from the ledger and duplicate it on + B's next registration call (#92682 review). + """ _forget_home_registrations(_registered, _registered_lock) from hermes_cli.config import load_config register_from_config(load_config()) diff --git a/agent/skill_bundles.py b/agent/skill_bundles.py index 401468d6b6..8fcd581834 100644 --- a/agent/skill_bundles.py +++ b/agent/skill_bundles.py @@ -129,7 +129,15 @@ def build_bundle_invocation_message( loaded_skill_names, missing_skill_names)`` or ``None`` if the bundle wasn't found. Uninstalled members are skipped with a note; disabled ones too, since ``_load_skill_payload`` bypasses the scan-time filter (``platform`` scopes - that check — gateway passes it, None resolves from env).""" + that check — gateway passes it, None resolves from env). + + Disabled skills are also skipped: bundles load members via ``_load_skill_payload`` directly, bypassing + the scan-time disabled filter in ``get_skill_commands()``, so the disabled list must be re-applied here. + ``platform`` scopes the check to a specific platform's ``skills.platform_disabled`` config (gateway + dispatch passes it explicitly because the gateway handles multiple platforms in one process); when + *None*, the platform resolves from session env vars and the global disabled list still applies. Mirrors + the stacked-skill gate in gateway dispatch (#58888). + """ info = get_skill_bundles().get(cmd_key) if not info: return None diff --git a/agent/skill_commands.py b/agent/skill_commands.py index 893f2fd451..6a09f774c0 100644 --- a/agent/skill_commands.py +++ b/agent/skill_commands.py @@ -61,7 +61,13 @@ def append_user_instruction(parts: list, instruction: str) -> str: """Append the instruction line to ``parts``; return the stable prefix, which ends exactly at the instruction marker so (registered with ``agent.prompt_cache_boundary``) the cache planner can break on the scaffold. - Single construction site guarantees the prefix is a byte-prefix of the message.""" + Single construction site guarantees the prefix is a byte-prefix of the message. + + Shared by every builder that ends a static skill scaffold with the caller-supplied volatile instruction + (single-skill invocations, cron job prompts). Keeping construction in one place guarantees the + registered prefix stays a byte-prefix of the built message — the invariant the request-time split + depends on. See #81867. + """ stable_prefix = "\n".join(parts) + "\n" + _SINGLE_SKILL_INSTRUCTION parts.append(f"{_SINGLE_SKILL_INSTRUCTION}{instruction}") return stable_prefix @@ -116,7 +122,11 @@ def _cut_after(message: str, marker: str, stop_marker: str, find) -> Optional[st def _resolve_skill_commands_platform() -> Optional[str]: """Current platform scope for disabled-skill filtering, or None (CLI, RL, scripts). A change invalidates the scan cache so each platform sees its - own ``skills.platform_disabled`` view.""" + own ``skills.platform_disabled`` view. + + Used to detect when the active platform has shifted so :func:`get_skill_commands` can drop a stale cache + that was populated for a different platform's ``skills.platform_disabled`` view (#14536). + """ try: from gateway.session_context import get_session_env resolved_platform = os.getenv("HERMES_PLATFORM") or get_session_env("HERMES_SESSION_PLATFORM") @@ -127,7 +137,14 @@ def _resolve_skill_commands_platform() -> Optional[str]: def _resolve_skill_commands_home() -> str: """Effective Hermes home the scan is scoped to (profiles carry their own - ``skills.external_dirs``, so a profile switch must invalidate the cache).""" + ``skills.external_dirs``, so a profile switch must invalidate the cache). + + A gateway session can switch between profiles that each carry their own ``skills.external_dirs`` (via + ``set_hermes_home_override``), but the module-level scan only tracked + ``_resolve_skill_commands_platform()``. Switching profiles without a platform change left the previous + profile's skill list cached, so ``get_skill_commands()`` reported a cache miss for skills that only + exist under the new profile (#88023). + """ from hermes_constants import get_hermes_home return str(get_hermes_home()) @@ -248,6 +265,10 @@ def _build_skill_message( parts.append("") # Everything before the volatile instruction is a stable scaffold; the # registered boundary lets the cache planner break there (see append_user_instruction). + # Everything before the caller-supplied instruction is a stable scaffold; declare the exact boundary + # so the Anthropic cache planner can put a breakpoint on it instead of caching the whole message as + # one atomic block (#81867). The static instruction prose stays on the stable side; the volatile + # instruction (webhook payload, ticket IDs, timestamps) and any runtime note ride in the tail. stable_prefix = append_user_instruction(parts, user_instruction) if runtime_note: parts += ["", f"[Runtime note: {runtime_note}]"] @@ -263,6 +284,9 @@ def _render_skill_block( """Bump Curator usage tracking (never fatal) and build the message block for one loaded skill.""" loaded_skill, skill_dir, skill_name = loaded try: + # Track active usage for Curator lifecycle management (#17782) + # Track active usage for Curator lifecycle management (#17782) + # Track active usage for Curator lifecycle management (#17782) from tools.skill_usage import bump_use bump_use(skill_name, task_id=task_id) except Exception: @@ -344,6 +368,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: global _skill_commands, _skill_commands_platform, _skill_commands_home platform = _resolve_skill_commands_platform() home = _resolve_skill_commands_home() + # Build into a local map and publish once, at the end. Writing straight into the global made a scan's + # partial results visible to everything else in the process: a second, overlapping scan deduped against + # its own (empty) ``seen_names`` but collided against the first scan's already- published slugs, logging + # one bogus "already claimed" warning per skill — each naming the same skill as its own incumbent + # (#74574). commands: Dict[str, Dict[str, Any]] = {} try: from tools.skills_tool import _skills_dir, _get_disabled_skill_names @@ -356,6 +385,7 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: # Precedence: project (through the quarantine chokepoint) > local > external. # Resolve the local dir at call time: import-time SKILLS_DIR is frozen to # the launch home, but a multiplexed profile scope may have changed it. + # See #67277. skills_dir = _skills_dir() iters = [iter_project_skill_files(d) for d in get_project_skills_dirs()] local = [skills_dir] if skills_dir.exists() else [] @@ -372,6 +402,11 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]: # could accept the new map under a stale platform tag and serve another # platform's disabled-skill view. with _publish_lock: + # Bare assignments are not atomic together: a reader landing between them sees the NEW map still + # carrying the OLD platform tag, and if that stale tag happens to match its own platform it accepts + # the map without rescanning — serving another platform's disabled-skill view, exactly the leak + # #14536 closed. Only the publish/lookup pair is locked; the scan above (file I/O, deferred imports) + # stays outside it. _skill_commands = commands _skill_commands_platform = platform _skill_commands_home = home @@ -382,7 +417,10 @@ def get_skill_commands() -> Dict[str, Dict[str, Any]]: """Return the current skill commands mapping (scan first if empty). Rescans when the platform scope (one gateway serving Telegram and Discord) or the active profile's home (Desktop profile switch) changes, so each sees its - own ``platform_disabled`` / ``external_dirs`` view.""" + own ``platform_disabled`` / ``external_dirs`` view. + + See #14536, #88023. + """ current_platform = _resolve_skill_commands_platform() current_home = _resolve_skill_commands_home() with _publish_lock: @@ -539,7 +577,12 @@ def build_preloaded_skills_prompt(skill_identifiers: list[str], task_id: str | N """Load skills for session-wide CLI/TUI preloading; returns (prompt_text, loaded_skill_names, missing_identifiers). Disabled skills count as missing: this path bypasses the scan-time filter, and ``hermes -s `` must not - force-load an operator-disabled skill.""" + force-load an operator-disabled skill. + + Disabled skills are treated the same as missing ones: this loads via a raw identifier straight into + ``_load_skill_payload``, bypassing ``get_skill_commands()``'s scan-time disabled filter — mirrors the + bundle-invocation gate (#59156). + """ loaded_names, missing, _disabled, prompt_parts = _load_skill_blocks( [(raw or "").strip() for raw in skill_identifiers], lambda identifier: _load_skill_payload(identifier, task_id=task_id), diff --git a/agent/skill_utils.py b/agent/skill_utils.py index 5f85589409..7c33210e7f 100644 --- a/agent/skill_utils.py +++ b/agent/skill_utils.py @@ -289,7 +289,10 @@ def parse_config_string_list(value) -> List[str]: """Normalize a config value that may hold a JSON-array string into a list. ``hermes config set`` stores lists as quoted JSON/Python-literal strings; treating one as a single name would silently filter nothing. A scalar - string still means one name.""" + string still means one name. + + See #13026, #86661. + """ if isinstance(value, str): if value.strip().startswith("["): try: @@ -412,7 +415,14 @@ _PROJECT_ROOT_MAX_DEPTH = 64 # walk-up bound for pathological cwds def find_project_root(start: Optional[Path] = None) -> Optional[Path]: """Nearest ancestor containing ``.git`` (dir or worktree file), or None. Without *start*, the surface's ``TERMINAL_CWD`` wins over process cwd so - cron/API surfaces inherit an interactive trust decision by project identity.""" + cron/API surfaces inherit an interactive trust decision by project identity. + + When *start* is not given, the surface's working directory wins over the process cwd: ``TERMINAL_CWD`` + is the same per-surface workdir the terminal tool and cron jobs use (a cron job sets it from its per-job + ``workdir`` without chdir'ing the scheduler process). This is what lets non-interactive surfaces inherit + a prior interactive trust decision by project identity — and a surface with no workdir in a trusted repo + simply resolves no project and loads nothing (#48975). + """ try: if start is None: from agent.runtime_cwd import scope_terminal_cwd @@ -501,6 +511,16 @@ def get_untrusted_project_skills_root() -> Optional[Tuple[Path, int]]: # cached under HERMES_HOME, never inside the repo); "dangerous" excludes the # skill from index, list, view and slash commands ("caution" loads, as on the hub). +# ── Project skill quarantine (scan-time injection defense) ──────────────── Trust (`hermes skills trust`) +# is a REPO-level decision made once; the repo's skill content keeps changing underneath it with every pull. +# The hub install path runs skills_guard on install, but project skills are read straight from a checkout — +# without this gate a `git pull` could inject a malicious skill into an already-trusted repo with no scan +# anywhere (#48974). Every project SKILL.md's parent dir is scanned with the same skills_guard scanner the +# hub uses (content-hash cached, so the cost is one scan per skill per content change). A "dangerous" +# verdict quarantines the skill: it is excluded from the index, skills_list, skill_view, and slash commands. +# "caution" loads (matches hub behavior for prose-level keyword hits) — the quarantine is for +# high-confidence findings only. The scan cache lives under HERMES_HOME, never inside the repo (we don't +# write artifacts into the user's checkout). _PROJECT_SCAN_SOURCE = "project-local" _PROJECT_QUARANTINE_CACHE: Dict[str, bool] = {} # skill_dir -> quarantined @@ -552,6 +572,7 @@ def normalize_skill_lookup_name(identifier: str) -> str: # (which follows the live profile-scoped HERMES_HOME), so normalization # must agree with that exact root. Import deferred (cycle). try: + # See #67277. from tools import skills_tool as _skills_tool primary_root = _skills_tool._skills_dir() except Exception: diff --git a/agent/stream_delivery.py b/agent/stream_delivery.py index 399e99a8ae..34b040df29 100644 --- a/agent/stream_delivery.py +++ b/agent/stream_delivery.py @@ -58,6 +58,15 @@ class StreamDeliveryMixin: self._deliver_to_stream_callbacks(tail) self._record_streamed_assistant_text(tail) + # Flush any benign partial-tag tail held by the think scrubber first (#17924): an innocent '<' at + # the end of the stream that turned out not to be a tag prefix should reach the UI. Then flush the + # context scrubber. Order matters — the think scrubber's output feeds into the context scrubber's + # state. + # Suppress reasoning/thinking blocks via the stateful scrubber (#17924). Earlier versions ran + # _strip_think_blocks per-delta here, which destroyed downstream state machines when a tag was split + # across deltas (e.g. MiniMax-M2.7 sends '' and its content as separate deltas — regex case 2 + # erased the first delta, so the CLI/gateway state machine never saw the open tag and leaked the + # reasoning content as regular response text). if think_scrubber is not None: think_tail = think_scrubber.flush() deliver(ctx_scrubber.feed(think_tail) if think_tail and ctx_scrubber is not None else think_tail) @@ -194,7 +203,10 @@ class StreamDeliveryMixin: self._deliver_interim(visible, already_streamed=already_streamed, record=undelivered_parts or [visible]) def _ensure_stream_writer_state(self) -> None: - """Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``).""" + """Lazily create the single-writer guard fields (``AIAgent.__new__``-built instances skip ``agent_init``). + + See #65991. + """ if getattr(self, "_stream_writer_lock", None) is None: self._stream_writer_lock = threading.Lock() if getattr(self, "_stream_writer_tls", None) is None: @@ -209,6 +221,8 @@ class StreamDeliveryMixin: Every attempt (each provider path, each retry) claims right before consuming; claiming bumps the shared token, so an earlier attempt still alive on another thread is superseded and its late chunks fenced out. Stored per-thread: a thread that never claimed can never be fenced. + + See #65991. """ self._ensure_stream_writer_state() with self._stream_writer_lock: @@ -217,11 +231,18 @@ class StreamDeliveryMixin: return token def _stream_writer_is_current(self, token: int) -> bool: - """True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded.""" + """True when ``token`` is still the active writer, so a stream loop can bail the instant it is superseded. + + active writer — i.e. no newer stream attempt has claimed the sink since (#65991). + """ return token == getattr(self, "_stream_writer_token", token) def _stream_writer_superseded(self) -> bool: - """True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer).""" + """True when this thread claimed the sink but a newer attempt has since claimed it (never for a non-claimer). + + stream attempt has since claimed it — i.e. this thread is a stale writer whose chunks must be + dropped (#65991). + """ token = getattr(getattr(self, "_stream_writer_tls", None), "token", None) return token is not None and token != getattr(self, "_stream_writer_token", token) @@ -261,6 +282,7 @@ class StreamDeliveryMixin: """Fire all registered stream delta callbacks (display + TTS).""" # A superseded stream must not interleave its tokens alongside the retry that replaced it. if self._stream_writer_superseded(): + # See #65991. self._note_dropped_stream_writer("_fire_stream_delta") return # One paragraph break before the first text delta after a tool iteration, without @@ -274,6 +296,7 @@ class StreamDeliveryMixin: # tag was split across deltas; memory-context spans split across chunks must not leak to # the UI. Legacy callers lack the scrubber attributes and get the whole-string fallbacks. think_scrubber = getattr(self, "_stream_think_scrubber", None) + # See #5719. scrubber = getattr(self, "_stream_context_scrubber", None) text = think_scrubber.feed(text) if think_scrubber is not None else self._strip_think_blocks(text) text = scrubber.feed(text) if scrubber is not None else sanitize_context(text) @@ -291,6 +314,8 @@ class StreamDeliveryMixin: def _fire_reasoning_delta(self, text: str) -> None: """Fire reasoning callback if registered; superseded writers are fenced like content deltas.""" if self._stream_writer_superseded(): + # Single-writer guard (#65991): fence out a superseded stream's reasoning deltas the same way as + # content deltas. self._note_dropped_stream_writer("_fire_reasoning_delta") return self._call_quietly(self.reasoning_callback, text) diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 5577a6bf89..3000dda5ee 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -20,6 +20,14 @@ from agent.transports.types import NormalizedResponse, ToolCall, Usage # xAI reserves ``tool_search`` for its server-side tool (HTTP 400 on client # declarations); aliased on the wire, mapped back in normalize_response. +# xAI's chat-completions API reserves the function name ``tool_search`` for its own server-side tool and +# rejects any request declaring a client function with that name (HTTP 400 "The function name tool_search is +# reserved for the tool_search tool", #95003). The Tool Search bridge (tools/tool_search.py) assembles its +# client-side discovery tool under the same literal name for every provider, so Grok providers are unusable +# whenever the bridge is active. Mirror the web_search treatment in transports/codex.py +# (_rename_client_web_search_for_xai): alias the wire declaration and map the alias back in +# normalize_response. The alias value matches _CODEX_TOOL_SEARCH_ALIAS from the Codex-side fix for the same +# reserved-name class (#83122) so the two transports stay consistent. _XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search" # Persistence-only / cross-transport message keys that strict OpenAI-compatible @@ -70,7 +78,19 @@ def _add_prompt_cache_key( survives compression rotation. A caller-supplied key is authoritative but is bounded to OpenAI's 64-char cap in place. Shares the Responses transport's hash so equivalent prefixes hit one bucket across modes. + + ``cache_scope_id``, when provided, is the rotation-stable logical scope (compression-lineage root — + agent/prompt_cache_scope.py) and takes precedence over the physical ``session_id`` so the key survives + context-compression session rotation (#79017). """ + # Stable prompt-cache routing for the Codex/Responses aux path, mirroring the main transport + # (agent/transports/codex.py::build_kwargs, which sets prompt_cache_key = + # _content_cache_key(instructions, tools)). Without this, MoA acting-aggregator and other auxiliary + # Responses calls stay cache-cold while the main Responses transport is warm (issue #53735). The key is + # content-addressed from the static prefix (instructions + tool schemas) so it stays warm across + # turns/fires. Guard the top-level field the same way the main transport does: xAI Responses takes the + # key in extra_body (not top-level) and GitHub/Copilot Responses opts out of cache-key routing entirely + # — for those hosts, skip it here. from agent.transports.codex import ( _bound_prompt_cache_key_field, _cache_scope_from_session_id, _content_cache_key ) @@ -90,7 +110,14 @@ def _add_prompt_cache_key( def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None: - """Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary.""" + """Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary. + + Hermes' internal effort set extends the wire vocabulary with ``ultra`` (the /reasoning command documents + none..xhigh|max|ultra). OpenAI- compatible wires — OpenRouter chief among them — accept exactly + max|xhigh|high|medium|low|minimal|none and reject the extension with HTTP 400 (#89503). Clamp against + the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with + narrower sets clamp again downstream. + """ if not isinstance(reasoning_config, dict): return reasoning_config effort = str(reasoning_config.get("effort") or "").strip().lower() @@ -104,6 +131,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> return None normalized_model = (model or "").strip().lower().removeprefix("google/") # Gemini-only; Gemma/PaLM on the same provider 400 on the field even as ``{"includeThoughts": False}``. + # ``thinking_config`` is a Gemini-only request parameter. The same ``gemini`` provider also serves Gemma + # (and historically PaLM/Bard); those reject the field with HTTP 400 "Unknown name 'thinking_config': + # Cannot find field" — including the polite ``{"includeThoughts": False}`` form. Omit the field entirely + # on non-Gemini models. (#17426) if not normalized_model.startswith("gemini"): return None effort = str(reasoning_config.get("effort", "medium") or "medium").strip().lower() diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 2af1c21780..77578a5a7b 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -12,6 +12,8 @@ from typing import Any, Callable, Optional from agent.reasoning_effort import ( ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, + # Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort): + # per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365). codex_supported_efforts, ) from agent.transports.base import ProviderTransport @@ -21,6 +23,7 @@ logger = logging.getLogger(__name__) # Cron fires use ``cron__``; the per-fire timestamp is # stripped so repeat fires of one job share a cache scope. +# See #51395, #52295. _CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$") @@ -64,6 +67,10 @@ _XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search" # OpenCode /v1/responses rejects client tools using these names (HTTP 400 # "custom function name 'X' is reserved"); xAI reserves ``tool_search`` for # Grok's native Tool Search. Aliased as hermes_. +# OpenCode's /v1/responses endpoints (Zen and Go, including custom providers pointing at opencode.ai) +# reserve certain function names server-side and reject client tools that use them with HTTP 400 ("custom +# function name 'X' is reserved"). Same treatment as the xAI web_search collision: rename on the wire +# (hermes_), map back in normalize_response so Hermes dispatch is unaffected. See #85589. _OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files") _XAI_RESERVED_TOOL_NAMES = ("tool_search",) _RESERVED_TOOL_ALIAS_PREFIX = "hermes_" @@ -126,6 +133,11 @@ def _xai_prefers_native_web_search() -> bool: """True when xAI Responses should use Grok's native ``web_search`` built-in. Web-search registry first, then the legacy ``_get_search_backend`` probe; fails closed to native (True). + + Delegates to the web-search registry's provider resolution (which reads ``web.search_backend`` / + ``web.backend`` from config) and checks whether the resolved provider is xAI. On any resolution failure, + returns True (fail-closed to native — preserves the #48108 incomplete-hang fix rather than risk + reintroducing it). """ try: from agent.web_search_registry import get_active_search_provider @@ -160,9 +172,25 @@ def _alias_wire_tools(response_tools: Any, params: dict[str, Any], is_xai_respon {**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools ] wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search" + # OpenCode Responses backends reserve web_search / search_files as function names (HTTP 400 "custom + # function name 'X' is reserved", #85589). Alias them on the wire; normalize_response maps them back. if response_tools and _is_opencode_responses_backend(params): response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES) wire_aliases.update(_oc_aliases) + # xAI server-side web search vs Hermes web providers. grok models on xAI's /v1/responses surface have a + # *native*, server-executed web search. A client-side function literally named ``web_search`` collides + # with that engine: declared as a plain ``function`` rather than ``{"type": "web_search"}``, the search + # dispatches but never reconciles → incomplete turn + 3 retries. Verified live against + # grok-composer-2.5-fast (2026-06); see #48108. Two modes, chosen by the user's web-search backend + # config: 1. **Native** (active/configured backend is ``xai``, or resolution fails): drop the client + # ``web_search`` function and declare xAI's built-in instead. 1:1 swap only when client ``web_search`` + # was already present — never an additive grant. 2. **Client** (Firecrawl / Tavily / Exa / … configured + # or resolved): keep Hermes dispatch so ``web.backend`` / ``web.search_backend`` is honored, but rename + # the wire tool to ``hermes_web_search`` so Grok cannot hijack the name. The alias is mapped back to + # ``web_search`` in ``normalize_response``. Request-local alias provenance: every wire alias THIS + # request emits is recorded here and stashed on the transport, so the reverse rewrite in + # ``normalize_response`` applies only to aliases that were actually sent (never to a real tool that + # merely shares an alias-shaped name). if is_xai_responses and response_tools: response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES) wire_aliases.update(_xai_aliases) @@ -183,6 +211,10 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]: elif reasoning_config.get("effort"): reasoning_effort = reasoning_config["effort"] + # Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker + # supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that + # repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a + # model's real ceiling (#87279). if params.get("is_xai_responses", False): from agent.model_metadata import is_grok_46_family @@ -215,6 +247,11 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op hostname = base_url_hostname(str(base_url or "")).lower() # Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without). + # Meta Model API (api.meta.ai) only achieves prompt-cache hits on the Responses API with + # prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured). Exact-hostname + # match per #32243. + # Meta Model API: prompt caching only on Responses API (0% on chat/completions vs 93-99% on /responses + # with retention). See #32243. if hostname == "api.meta.ai": return "24h" parts = hostname.split(".") @@ -229,6 +266,12 @@ def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], """``pck_`` of (scope_id, instructions, name-sorted tools), or None if nothing static. Routing hint only; ``scope_id`` keeps unrelated sessions off one bucket. + + ``scope_id`` (pass ``_cache_scope_from_session_id(session_id)``) keeps unrelated sessions — independent + conversations, main vs. child/subagent, sibling children — from concentrating onto the same bucket + merely because their static prefix matches (see #78941), while still letting recurring cron fires of one + job share a stable key across their timestamped session_ids (the original #51395/#52295 fix this built + on). Sorting tools by name keeps the hash insertion-order independent. """ if not instructions and not tools: return None @@ -421,6 +464,20 @@ class ResponsesApiTransport(ProviderTransport): cache key / xAI conv header), max_tokens, timeout, request_overrides, provider, base_url, is_github_responses, is_codex_backend, is_xai_responses, github_reasoning_extra, context_management, replay_encrypted_reasoning. + + params: instructions: str — system prompt (extracted from messages[0] if not given) + reasoning_config: dict | None — {effort, enabled} session_id: str | None — transcript/session id; + drives the Codex ``session_id`` header, and is the cache-scope fallback when no ``cache_scope_id`` + is given cache_scope_id: str | None — rotation-stable logical scope id (compression-lineage root; + see agent/prompt_cache_scope.py). Preferred over session_id when deriving the prompt_cache_key + content hash and the xAI x-grok-conv-id header; the Codex x-client-request-id header mirrors the + resulting body key. Keeps the cache warm across context-compression session rotation (#79017) + max_tokens: int | None — max_output_tokens timeout: float | None — per-request timeout forwarded to + the SDK request_overrides: dict | None — extra kwargs merged in provider: str | None — provider name + for backend-specific logic base_url: str | None — endpoint URL base_url_hostname: str | None — + hostname for backend detection is_github_responses: bool — Copilot/GitHub models backend + is_codex_backend: bool — chatgpt.com/backend-api/codex is_xai_responses: bool — xAI/Grok backend + github_reasoning_extra: dict | None — Copilot reasoning params """ from run_agent import DEFAULT_AGENT_IDENTITY @@ -490,6 +547,8 @@ class ResponsesApiTransport(ProviderTransport): _bound_prompt_cache_key_field(kwargs) # Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing. + # Grok 4.6 accepts Priority Processing, but continue stripping stale or unsupported tier values on + # every other xAI path. See #28490 and #84799. if is_xai_responses: from agent.model_metadata import is_grok_46_family @@ -521,6 +580,13 @@ class ResponsesApiTransport(ProviderTransport): _merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope}) # xAI reads prompt_cache_key from the body; extra_body survives SDK builds whose # Responses.stream() dropped the typed kwarg. An explicit request_overrides value wins. + # Scoped like the body cache key below — otherwise cron's per-fire timestamp in session_id + # (cron__) pins every fire of the same job to a different xAI backend server (#78941). + # xAI Responses cache-routing — body-level field per + # https://docs.x.ai/developers/advanced-api-usage/prompt-caching/maximizing-cache-hits. A + # caller's request_overrides={"prompt_cache_key": ...} lands on the top-level kwarg set above — + # read it back here so an explicit override actually governs the field xAI reads, instead of + # being silently outrun by the auto-derived cache_key (#78941). existing_extra_body = kwargs.get("extra_body") kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {} kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key)) diff --git a/agent/transports/codex_app_server.py b/agent/transports/codex_app_server.py index 03912a7ebc..1d2d348057 100644 --- a/agent/transports/codex_app_server.py +++ b/agent/transports/codex_app_server.py @@ -48,6 +48,14 @@ class CodexAppServerClient: ) -> None: self._codex_bin = codex_bin # codex needs LLM provider creds but must not receive Tier-1 Hermes secrets (gateway/GitHub/infra tokens). + # codex app-server is a model-driving CLI executor: it runs a model-chosen agentic loop that + # executes shell commands, so it legitimately needs LLM provider credentials + # (inherit_credentials=True) to authenticate against the model endpoint. But the previous + # `os.environ.copy()` also handed it every Tier-1 Hermes secret — gateway bot tokens, GitHub auth, + # Modal/Daytona infra tokens, the dashboard session token, AUXILIARY_* side-LLM keys, + # GATEWAY_RELAY_* auth — none of which a coding subprocess has any use for. Route through the + # centralized helper so Tier-1 + dynamic-internal secrets are always stripped while provider creds + # still flow, matching copilot_acp_client (#29157 sibling spawn-site gap). spawn_env = hermes_subprocess_env(inherit_credentials=True) if env: spawn_env.update(env) @@ -71,6 +79,7 @@ class CodexAppServerClient: # Hide the console the codex child would otherwise flash on Windows (#56747). # Hide-only — stdio pipes stay intact for the app-server wire. + # See #56747. from hermes_cli._subprocess_compat import windows_hide_flags self._proc = subprocess.Popen( diff --git a/agent/transports/codex_app_server_session.py b/agent/transports/codex_app_server_session.py index 1290bf5b31..5ff11c73e4 100644 --- a/agent/transports/codex_app_server_session.py +++ b/agent/transports/codex_app_server_session.py @@ -328,6 +328,11 @@ class CodexAppServerSession: """Send a user message and block until turn/completed, bridging approvals and projecting items. post_tool_quiet_timeout: silence this long after a tool completes fast-fails and retires. + + post_tool_quiet_timeout: if codex emits a tool completion and then goes quiet for this many seconds + without emitting another item or `turn/completed`, fast-fail and mark the session for retirement. + Mirrors openclaw beta.8's post-tool completion watchdog (#81697) so a wedged codex doesn't burn the + full turn deadline. """ result = TurnResult() if self._start_for(result): diff --git a/agent/tts_provider.py b/agent/tts_provider.py index 99cdc8545a..9b49fcaea8 100644 --- a/agent/tts_provider.py +++ b/agent/tts_provider.py @@ -86,7 +86,10 @@ class TTSProvider(CatalogProviderBase): """Whether output suits voice-bubble delivery (mirrors ``tts.providers..voice_compatible``): True → the gateway converts to Opus via ffmpeg if needed; False → regular audio attachment. Default - False (opt in).""" + False (opt in). + + See #17843. + """ return False diff --git a/agent/turn_api_request.py b/agent/turn_api_request.py index fbbb321a85..a77aa6c6dc 100644 --- a/agent/turn_api_request.py +++ b/agent/turn_api_request.py @@ -122,6 +122,12 @@ def build_api_request( api_kwargs = agent._build_api_kwargs(api_messages, tools_for_api=tools_for_api) # Surrogate chokepoint: tool descriptions, extra_body and kwargs strings can carry # invalid code points (HTTP 400). One walk makes the payload json.dumps()-safe. + # Outbound-request surrogate chokepoint (#50959): the messages were scrubbed above, but the rest of the + # request body — tool/function descriptions (session_search's ±-heavy text is the recorded repro), + # extra_body, system strings routed via kwargs — can still carry invalid code points that providers + # reject with a non-retryable HTTP 400 ("invalid unicode code point"). One in-place walk here guarantees + # the entire payload json.dumps()-safe regardless of which leaf produced the string. Fast no-op when the + # payload is clean. _sanitize_structure_surrogates(api_kwargs) if agent._force_ascii_payload: _sanitize_structure_non_ascii(api_kwargs) diff --git a/agent/turn_context.py b/agent/turn_context.py index 6d266ddb9a..9163a0e3a1 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -172,6 +172,7 @@ def _maybe_title_session_at_turn_start(agent: Any, messages: List[Any]) -> None: main_runtime = { k: getattr(agent, k, None) for k in ("model", "provider", "base_url", "api_key", "api_mode") } + # See #19027. maybe_auto_title( session_db, session_id, @@ -197,7 +198,17 @@ def reanchor_current_turn_user_idx(messages: List[Any], user_message: Any) -> in Prefers the LAST user message whose content exactly matches this turn's text, else the last user-originated turn; compaction handoffs are never the fallback. - Returns -1 when there is no user-originated message.""" + Returns -1 when there is no user-originated message. + + Compression replaces list entries with fresh copies (and may append a todo-snapshot user message or a + restored user turn AFTER the surviving copy of the current turn's message), so a pre-compression index + is meaningless. Prefer the LAST user message whose content exactly matches this turn's text — the + surviving copy in the common case — so the injection stamp and the #48677 persist override can't land on + a todo-snapshot or historical row. Fall back to the last *user-originated* turn when no exact match + survives (merge-summary-into-tail rewrites the content but the trackers still need a live anchor). + Compaction handoffs must never become the fallback anchor (#80622) — they are reference-only + scaffolding, not the active ask. + """ from agent.context_compressor import user_originated_turn_view fallback = -1 @@ -225,11 +236,24 @@ def compression_made_progress( orig_len: int, new_len: int, orig_tokens: int, new_tokens: int ) -> bool: """``True`` if a compression pass materially reduced the request: fewer rows, or a - >5% token cut with the same rows (same floor as the overflow-handler retry).""" + >5% token cut with the same rows (same floor as the overflow-handler retry). + + Compression can succeed by summarising message contents — reducing the estimated request token count — + without reducing the message row count. Treating row count as the sole progress signal false-positives + on size-only wins and surfaces a misleading "Cannot compress further" failure even when post-compression + tokens are well below the model context window. See issue #39548 for an observed case: 220 → 220 + messages, ~288k → ~183k tokens on a 1M-context model still triggered auto-reset. + The token reduction must be *material* (>5%) to count as progress — the same floor the overflow-handler + retry path uses (conversation_loop.py, 39550) — so a sub-5% wobble doesn't keep the multi-pass loop + spinning. See #39550. + """ return new_len < orig_len or (orig_tokens > 0 and new_tokens < orig_tokens * 0.95) # Back-compat alias: gateway callers and tests patch ``_compression_made_progress``. +# Back-compat alias: this predicate was module-private until the gateway's session-hygiene recovery gate +# needed the same semantics (#79624). Keeping the old name bound means existing callers and any test that +# patches ``_compression_made_progress`` continue to work unchanged. _compression_made_progress = compression_made_progress @@ -296,7 +320,12 @@ def _should_idle_compact( produced (``ContextCompressor.last_compression_rough_tokens``, same rough shape as ``tokens``) — raises the floor to ``last + floor_tokens`` so the transcript must gain a floor's worth of NEW content first. ``0`` (nothing compacted yet / counter reset) keeps - the original semantics exactly.""" + the original semantics exactly. + + A session that compacted to well above that target therefore stays above it forever, so every later idle + resume re-runs a full summarisation over a transcript that has not grown — minutes of silently blocked + prompt on a slow route, reclaiming nothing (#97239). + """ if not enabled or idle_after_seconds <= 0 or idle_gap_seconds < idle_after_seconds or cooldown_active: return False effective_floor = floor_tokens @@ -350,6 +379,12 @@ def _publish_runtime_main(agent: Any) -> None: from agent.prompt_cache_scope import resolve_prompt_cache_scope_safe # Rotation-stable prompt-cache scope (lineage root), memoized per segment; a new # session uses the physical id until build_api_kwargs re-resolves. + # Memoized per segment on the agent, so this is a DB walk at most once per segment — except a + # brand-new session whose row lands later in turn setup (_ensure_db_session); that first turn falls + # back to the physical id here and the first build_api_kwargs re-resolves. Stays valid through a + # mid-turn compression rotation because the lineage root is by definition rotation-invariant + # (#79017). Resolved with the never-raising variant OUTSIDE the argument list, so a resolution + # failure can only lose the scope — never the whole runtime binding. _cache_scope = resolve_prompt_cache_scope_safe(agent) or "" set_runtime_main( _str_attr(agent, "provider"), _str_attr(agent, "model"), @@ -569,6 +604,8 @@ def _collect_pre_llm_call_context( sender_id=getattr(agent, "_user_id", None) or "", ) try: + # Spill oversized per-hook context to disk so a runaway plugin can't inflate every subsequent + # turn's prompt. Ported from openai/codex PR #21069 ("Spill large hook outputs from context"). from tools.hook_output_spill import ( get_spill_config as _spill_cfg, spill_if_oversized as _spill_if_oversized ) @@ -738,6 +775,11 @@ def build_turn_context( # Tag log records on this thread with the session ID for ``hermes logs``; bind the # skill write-origin ContextVar; restore the primary runtime after a fallback turn. + # NOTE: the DB session row is created later, AFTER the system prompt is restored/built (see + # _ensure_db_session() below the system-prompt block). Creating it here — before _cached_system_prompt + # is populated — inserts a row with system_prompt=NULL on a fresh API/gateway agent that carries + # client-managed history, which then trips the "stored system prompt is null; rebuilding from scratch" + # warning and a needless first-turn prefix cache miss. (Issue #45499.) set_session_context(agent.session_id) set_current_write_origin(getattr(agent, "_memory_write_origin", "assistant_tool")) agent._restore_primary_runtime() diff --git a/agent/turn_context_compaction.py b/agent/turn_context_compaction.py index 14481f11c1..415a54fefd 100644 --- a/agent/turn_context_compaction.py +++ b/agent/turn_context_compaction.py @@ -42,6 +42,14 @@ class CompactionOutcome: def _clear_overflow_warn(agent: Any) -> None: """Re-arm the context-overflow warning dedup (test doubles may lack the method).""" + # Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow + # warning dedup so a future blocked-over-threshold turn can warn again. Mirrors the turn-context + # preflight reset (silent-overflow fix #62625). getattr guard: test doubles built via object.__new__ + # lack the method (gateway test-double pitfall) — treat absence as no-op. + # Compression is actually running (block cleared / was never blocked) — reset the blocked-overflow + # warning dedup so a future blocked-over-threshold turn can warn again (silent-overflow fix #62625). + # getattr guard: test doubles built via object.__new__ lack the method (gateway test-double pitfall) — + # treat absence as no-op. _clear_warn = getattr(agent, "_clear_context_overflow_warn", None) if callable(_clear_warn): _clear_warn() @@ -94,6 +102,10 @@ def _apply_grown_window(agent: Any, compressor: Any, grown: int) -> None: def _refund_api_call(agent: Any, api_call_count: int) -> int: """A pass that never reached the provider refunds the call count and budget.""" + # Host progress-aware timeout (#98722, salvaged from #98741): this preflight iteration never reached the + # provider. Refund its provisional call/budget exactly like a successful pre-API compaction, then stop + # before the unchanged oversized request reaches the provider — its overflow error would only invoke + # compression again on the same transcript with the wait budget already spent. api_call_count -= 1 agent._api_call_count = api_call_count agent.iteration_budget.refund() @@ -200,6 +212,7 @@ def _codex_native_auto_compaction(agent: Any) -> bool: """Codex app-server threads are compacted by the codex agent itself; Hermes only initiates compaction in "hermes" mode.""" return ( + # See #36801. getattr(agent, "api_mode", None) == "codex_app_server" and str( getattr(agent, "codex_app_server_auto_compaction", "native") or "native" @@ -367,6 +380,10 @@ def _run_preflight_passes( # Lock-skip: another path holds the lock, so this is a DEFER, not proof of # incompressibility — don't arm the blocker; stop passes this turn. logger.info( + # That is a temporary DEFER, not proof the transcript cannot compress — do NOT arm the + # insufficient-progress blocker (the loop's error handlers must keep their provider-proven + # retry budget) and stop preflight passes for this turn; the lock winner is shrinking the + # same session concurrently. See #69870. "Preflight compression deferred: compression lock " "held by another path (session %s)", agent.session_id or "none", diff --git a/agent/turn_facade.py b/agent/turn_facade.py index 7b7d3e3f4e..6252e3ba2e 100644 --- a/agent/turn_facade.py +++ b/agent/turn_facade.py @@ -31,6 +31,8 @@ class TurnFacadeMixin: """Forwarder — see ``agent.conversation_loop.run_conversation``.""" # A review shares this session_id for cache parity: fence review startup or interrupt # an admitted request and await its exit before opening live-turn instrumentation. + # Foreground priority is retained if the review does not acknowledge within the bounded deadline + # (#84423). from agent.background_review import cancel_background_review_for_live_turn cancel_background_review_for_live_turn(self) @@ -106,6 +108,9 @@ class TurnFacadeMixin: # affinity scope falls back to it; accounting handles route aux usage to the session. token = set_conversation_context(self._conversation_root_id()) affinity_token = set_affinity_scope(declared_conversation_scope_safe(self)) + # Publish the session accounting handles the same way so auxiliary calls record their token + # usage into session_model_usage (task dimension) — the fix for aux spend being invisible in + # analytics (issue #23270). acct_token = set_accounting_context( getattr(self, "_session_db", None), getattr(self, "session_id", None) ) diff --git a/agent/turn_facade_lease.py b/agent/turn_facade_lease.py index 4987b1dd76..5fb4f5c052 100644 --- a/agent/turn_facade_lease.py +++ b/agent/turn_facade_lease.py @@ -220,6 +220,9 @@ def _durable_session_exists(db, session_id: str) -> bool: # A locked / non-WAL read is not proof the row is absent; treating probe failure as "fresh" # ran fail-open at the exact contention point. Acquire, or fail closed. logger.warning( + # Acquire (or fail closed if acquire itself cannot) rather than start load/run/flush + # unsynchronized. get_session returns None — it does not raise — when the row is missing. See + # #84234. "Could not check durable session before turn lease; " "will acquire rather than run without serialization", exc_info=True, diff --git a/agent/turn_final_response.py b/agent/turn_final_response.py index fca3dbe7f6..4907d6b09b 100644 --- a/agent/turn_final_response.py +++ b/agent/turn_final_response.py @@ -103,6 +103,16 @@ def finish_text_response( agent._emit_pending_fallback_notice() agent._clear_status_buffer() + # Defensive: repair malformed role-alternation before API call. Catches cases where the history got + # wedged into a ``tool → user`` or ``user → user`` tail (e.g. after empty- response scaffolding was + # stripped and a new user message landed after an orphan tool result). Most providers return empty + # content on malformed sequences, which would otherwise retrigger the empty-retry loop indefinitely. + # repair_message_sequence_with_cursor also recomputes the SessionDB flush cursor (_last_flushed_db_idx) + # when repair compacts the list, so the turn-end flush doesn't skip the assistant/tool chain (#44837). + # One-time repeated-heal escalation notice (#96870): if the sanitizer above just crossed the per-session + # heal threshold, deliver the queued notice through the status/warning callback — the normal out-of-band + # delivery channel (gateway status message / CLI print). NEVER appended to messages/api_messages: + # conversation context and the cached prompt prefix stay byte-identical. from agent.agent_runtime_helpers import ( intent_ack_continuation_mode, trailing_continue_intent ) diff --git a/agent/turn_finalizer.py b/agent/turn_finalizer.py index 9fd951f17b..7fcea4d05d 100644 --- a/agent/turn_finalizer.py +++ b/agent/turn_finalizer.py @@ -46,7 +46,12 @@ def _record_kanban_budget_exhausted( Routed via ``_record_task_failure`` (not ``kanban_block``) so it counts toward the consecutive-failure circuit breaker. Idempotent via the ``_end_run`` CAS - (``WHERE ended_at IS NULL``), so safe from multiple exit paths.""" + (``WHERE ended_at IS NULL``), so safe from multiple exit paths. + + This is a bounded fallback (#87096): the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``) + guarantees idempotence — if another path already closed the run this is a no-op — so it is safe to call + from multiple exit paths. + """ try: from hermes_cli import kanban_db as _kb _conn = _kb.connect() @@ -149,6 +154,17 @@ def _resolve_budget_fallback( # A kanban worker must record a terminal outcome whether or not a fallback path # was eligible, so the dispatcher learns the worker could not complete. _kanban_task = os.environ.get("HERMES_KANBAN_TASK") if budget_exhausted else None + # If running as a kanban worker, signal the dispatcher that the worker could not complete (rather than + # treating it as a protocol violation). This applies whether the user-facing fallback came from the + # summary call or an explicitly pending continuation; both exhausted the task budget and must advance + # the failure circuit. We route through ``_record_task_failure(outcome="timed_out")`` rather than + # ``kanban_block`` so this counts toward the dispatcher's consecutive-failure circuit breaker (#29747 + # gap 2). + # Bounded fallback (#87096): budget was exhausted but none of the normal fallback paths were eligible + # (interrupted / failed / anomalous exit_reason). If running as a kanban worker we must still record a + # terminal outcome so the task does not remain in an ambiguous lifecycle state. The worker's run is + # closed via ``_record_task_failure`` (compare-and-swap receipt path) which is a no-op if another path + # closed it — the CAS invariant in ``_end_run`` (``WHERE ended_at IS NULL``) guarantees idempotence. if _kanban_task: _record_kanban_budget_exhausted(_kanban_task, api_call_count, agent.max_iterations, logger) return final_response, _turn_exit_reason, preserved_verification_fallback @@ -202,6 +218,16 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed) # row; enforce "delivered final_response ⇒ assistant row" here. Compare content, # not role, so a matching verification candidate isn't dup'd. if final_response and not interrupted: + # Some recovery/fallback paths return a real final_response without adding a closing assistant + # message to the transcript (e.g. the partial-stream and prior-turn-content recovery ``break`` sites + # in ``conversation_loop``). If persisted as-is, the durable session can end at a tool/user message + # even though the caller — and the gateway platform — already saw a completed assistant response. + # The next turn then replays a user-only backlog and the model re-answers every "unanswered" + # message. Close the durable turn at the source, at the single chokepoint every recovery ``break`` + # flows through, so the invariant "delivered final_response ⇒ assistant row in transcript" holds + # regardless of which path produced it. (#43849 / #44100) Compare content (not just role) so a + # verification candidate that matches the final response is not duplicated at budget exhaustion. + # (#65919 §7) _tail = messages[-1] if messages else None if not isinstance(_tail, dict) or _tail.get("role") != "assistant": append_message(messages, {"role": "assistant", "content": final_response}) @@ -219,6 +245,8 @@ def _close_transcript_tail(agent, messages, final_response, interrupted, failed) # Request is complete, so replace API-local voice/model/skill guidance with the # clean user input before the durable snapshot (earlier flushes still needed them). + # Earlier turn-start flushes use the DB-only override because their messages are still needed for the + # API request; this finalizer runs after that request is complete (#48677 / #63766). _apply_override = getattr(agent, "_apply_persist_user_message_override", None) if callable(_apply_override): _apply_override(messages) @@ -302,6 +330,13 @@ def _append_file_mutation_footer(agent, final_response, logger): """Append the verifier advisory when ``write_file`` / ``patch`` calls failed and were never superseded by a successful write to the same path (surfaces over-claiming).""" try: + # File-mutation verifier footer. This catches the specific case — reported by Ben Eng + # (#15524-adjacent) — where a model issues a batch of parallel patches, half of them fail with + # "Could not find old_string", and the model summarises the turn claiming every file was edited. The + # user then has to manually run ``git status`` to catch the lie. With this footer the truth is + # surfaced on every turn, so over-claiming is structurally impossible past the model. Gate: only + # applied when a real text response exists for this turn and the user didn't interrupt. + # Empty/interrupted turns already have other surface text that shouldn't be augmented. _failed = getattr(agent, "_turn_failed_file_mutations", None) or {} if _failed and agent._file_mutation_verifier_enabled(): footer = agent._format_file_mutation_failure_footer(_failed) @@ -473,6 +508,12 @@ def finalize_turn( # Surrogate chokepoint: RAW SDK text with a lone UTF-16 surrogate crashes downstream # consumers (stdout, Telegram ``utf16_len``, JSON); scrub once where it leaves the loop. + # Class-level surrogate chokepoint (#80366, #55143, #55309, #19819): ``final_response`` is often the RAW + # SDK content (``assistant_message.content``), not the sanitized copy stored in history by + # ``build_assistant_message``. Any lone UTF-16 surrogate (U+D800–U+DFFF) in it crashes downstream + # consumers — oneshot stdout writes, Telegram's ``utf16_len`` length check, Signal formatting, JSON + # envelope encodes — on every provider (Ollama, NVIDIA NIM, …). Scrub once here, where model text leaves + # the conversation loop, so every delivery surface receives valid Unicode. if isinstance(final_response, str): final_response = _sanitize_surrogates(final_response) diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index 78051669db..1b4917758c 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -273,6 +273,11 @@ def begin_iteration( # Grace call: budget exhausted but the model gets one more call. Consume the # flag so the loop exits after this iteration regardless of outcome. if agent._budget_grace_call: + # Iteration budget: the LLM is only notified when it actually exhausts the iteration budget + # (api_call_count >= max_iterations). At that point we inject ONE message, allow one final API call, + # and if the model doesn't produce a text response, force a user-message asking it to summarise. No + # intermediate pressure warnings — they caused models to "give up" prematurely on complex tasks + # (#7915). agent._budget_grace_call = False elif not agent.iteration_budget.consume(): _turn_exit_reason = "budget_exhausted" @@ -340,6 +345,11 @@ def apply_retry_restarts( retry_count += 1 _retry.restart_with_compressed_messages = False if _should_skip_model_call_for_reference_handoff( + # Compression rebuilt the list (tail messages are fresh compaction copies), so the + # pre-compression index of this turn's user message is stale. Re-anchor both index trackers: the + # api_content stamp below, the loop's injection site, and the flush's persist-override row + # (#48677) must all target the surviving dict, not a stale position. Exact-content match first + # so a todo-snapshot user message appended after the tail can't steal the anchor. messages, user_message ): logger.info( diff --git a/agent/turn_liveness.py b/agent/turn_liveness.py index a1eace5761..538ca2a583 100644 --- a/agent/turn_liveness.py +++ b/agent/turn_liveness.py @@ -127,6 +127,11 @@ class TurnLivenessWatchdog: return None # Observational only: the commit below can still veto the abort if progress # resumed; the definitive settlement is _surface_committed_abort. + # Pre-commit surface is OBSERVATIONAL only: it reports the stall and that a recovery attempt is + # beginning. It must not claim the abort or the lease withdrawal has committed — the next operation + # can still veto the outcome. The definitive aborted/lease-stopped settlement is published by + # _surface_committed_abort only after _commit_abort succeeds and the turn is deactivated (#95663 + # review). self._surface_stall(snapshot) message = f"Turn made no progress for {int(snapshot.idle_seconds)}s; aborting to release the session." if not self._commit_abort(snapshot, message): @@ -178,7 +183,13 @@ class TurnLivenessWatchdog: ) def _surface_committed_abort(self, snapshot: ActivitySnapshot) -> None: - """Publish the definitive settlement once the abort has authority.""" + """Publish the definitive settlement once the abort has authority. + + Runs only once ``_commit_abort`` succeeded (the interrupt was published) and the turn lease was + deactivated: the turn IS force-aborted and lease renewal IS stopped, so stating that is now true. + Separated from the pre-commit surface so a declined abort never reports a committed outcome (#95663 + review). + """ logger.error( "Turn liveness watchdog aborted turn for session %s: " "no progress for %.1fs; turn interrupted and lease renewal " diff --git a/agent/turn_loop_errors.py b/agent/turn_loop_errors.py index 32c7b20d41..8ebcb669c6 100644 --- a/agent/turn_loop_errors.py +++ b/agent/turn_loop_errors.py @@ -56,6 +56,17 @@ def handle_outer_loop_error( _outer_error_count += 1 # Interpreter shutdown makes every executor op raise: break. + # Phase-aware error classification. The huge outer try/except spans both the actual API request and all + # local post-processing of the returned assistant message. Deterministic local bugs (e.g. passing a + # multimodal content list into a regex helper after a vision turn or context compaction) should not be + # retried: they will fail identically on every iteration and only burn the iteration budget. We classify + # an error as local by inspecting the traceback: if the exception propagated through any of the known + # local post-processing helpers and never entered the interruptible API-call helpers, it is almost + # certainly a local processing bug. (#66267) Interpreter shutdown: if the process is tearing down, every + # executor-backed operation (API call, tool dispatch, memory sync) raises ``RuntimeError: cannot + # schedule new futures after interpreter shutdown``. Retrying is pointless — the executor is gone for + # good — and each retry just spams another traceback. Break immediately so the turn exits cleanly. + # (#93217) if sys.is_finalizing() or _is_interpreter_shutdown_error(e): error_msg = f"Interpreter is shutting down — cannot continue (API call #{api_call_count}): {e}" try: diff --git a/agent/turn_overflow.py b/agent/turn_overflow.py index dca76b10a4..79a17683af 100644 --- a/agent/turn_overflow.py +++ b/agent/turn_overflow.py @@ -109,6 +109,9 @@ class _Recovery(OverflowVerdict): "failed": True, } if compression_exhausted: + # Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The + # bloated transcript remains intact while future input can move to a clean session instead of + # replaying the summarize-timeout loop. result["compression_exhausted"] = True result.update(extra) return self.done("return", result) @@ -190,6 +193,14 @@ class _Recovery(OverflowVerdict): if deferred is not None: return deferred, False, original_tokens messages = self.messages + # Re-measure after compression. Same-message-count compression (tool-result pruning, in-place + # summarization) can materially reduce request size without reducing the message array (#39550), and + # — the image-dominated case — compaction's historical-media aging (#97160) can free megabytes of + # base64 that the token estimate never counted. Bytes are the yardstick for a 413; tokens are kept + # only for status display. + # Re-estimate tokens after compression. Same-message-count compression (tool-result pruning, + # in-place summarization) can materially reduce request size without reducing the message array. + # (#39550) new_tokens = estimate_messages_tokens_rough(messages) shrank_tokens = new_tokens > 0 and new_tokens < original_tokens * 0.95 if len(messages) < original_len: @@ -223,6 +234,13 @@ def _recover_payload_too_large(st: _Recovery, _retry: TurnRetryState) -> Overflo messages = st.messages original_len = len(messages) + # A 413 is a BYTE-size error, so this branch scores progress in BYTES of the serialized messages payload + # — exact and free — never the token estimate. The estimator prices every image at a flat per-image + # token cost (see estimate_messages_tokens_rough) so screenshots don't trigger premature compaction; + # that deliberate byte-blindness means compaction can free megabytes of base64 (real case: two vision + # results = 96.6% of the request body but ~3.7% of the estimate) while the token delta stays under any + # threshold. Token-scored progress here burned all attempts on "no progress" and wedged the session + # permanently. (#88960 / #47339) original_bytes = serialized_messages_bytes(messages) deferred = st.compress(st.request_tokens()) if deferred is not None: @@ -439,6 +457,9 @@ def recover_from_overflow( # failover or generic retries. The classifier also covers 400/disconnect + # large-session heuristics. st.is_context_length_error = ( + # Check for context-length errors BEFORE generic 4xx handler. The classifier detects context + # overflow from: explicit error messages, generic 400 + large session heuristic (#1630), and server + # disconnect + large session pattern (#2153). classified.reason == FailoverReason.context_overflow or wrapped_output_cap_budget is not None ) diff --git a/agent/turn_preflight.py b/agent/turn_preflight.py index a7801786a2..c131002fd2 100644 --- a/agent/turn_preflight.py +++ b/agent/turn_preflight.py @@ -262,9 +262,17 @@ def compress_after_tool_results( ) _compressor = agent.context_compressor + # Use real token counts from the API response to decide compression. prompt_tokens + completion_tokens + # is the actual context size the provider reported plus the assistant turn — a tight lower bound for the + # next prompt. Tool results appended above aren't counted yet, but the threshold (default 50%) leaves + # ample headroom; if tool results push past it, the next API call will report the real total and trigger + # compression then. If last_prompt_tokens is 0 (stale after API disconnect or provider returned no usage + # data), fall back to rough estimate to avoid missing compression. Without this, a session can grow + # unbounded after disconnects because should_compress(0) never fires. (#2153) if _compressor.last_prompt_tokens > 0: # Only prompt_tokens: thinking models inflate completion_tokens with # reasoning that uses no context → premature compression. + # Only use prompt_tokens — completion/reasoning tokens don't consume context window space. (#12026) _real_tokens = _compressor.last_prompt_tokens elif _compressor.last_prompt_tokens == -1: # Compression just ran, no API prompt count yet: don't treat a rough @@ -274,6 +282,12 @@ def compress_after_tool_results( # Include tool schemas (20-30K tokens the messages-only estimate misses) and # stay route-aware: on a compacted native-Codex session the generic # durable-history figure would false-trigger. + # Include tool schemas — with 50+ tools enabled these add 20-30K tokens the messages-only estimate + # misses, which can skip compression past the configured threshold (#14695). Route-aware + # (#96995/#97602 class): on a compacted native-Codex session the generic durable-history figure + # overstates the wire and would false-trigger compression here exactly like the pre-API guard — this + # fallback runs precisely when no provider usage is available (post-disconnect / gateway restart), + # the unanchored case from #97602's repro. _real_tokens = _midturn_request_pressure_tokens( agent, messages, active_system_prompt or "", estimate_request_tokens_rough(messages, tools=agent.tools or None), @@ -298,6 +312,30 @@ def compress_after_tool_results( if messages is _post_tool_input and compression_skipped_due_to_lock(agent): # Lock-skip no-op is a temporary defer, not evidence about compressibility: # refund so a lock-loser loop doesn't burn the budget toward exhausted. + # #69870 lock-skip / #97488 transient-block: this pass no-oped for a TEMPORARY reason (another + # path holds the compression lock, or a timed cooldown/backoff guard is active). That is a + # temporary DEFER, not evidence about compressibility — refund the attempt (it must not burn the + # shared overflow-recovery budget toward compression_exhausted → gateway auto-reset, + # #9893/#35809) and leave the insufficient-progress blocker unarmed. Proceed with the current + # request: if it truly does not fit, the provider's 413/overflow handler returns the soft + # compression_deferred result with that stronger signal. + # #69870 lock-skip: the provider proved the request does not fit, but this compression pass + # no-oped only because another path holds the session's compression lock. Temporary defer, not + # exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the + # session (#9893/#35809). + # #97488 transient-block: compression no-oped because a timed guard (host-timeout cooldown / + # structural backoff) is active — a temporary defer, not evidence of incompressibility. Never + # classify it as compression_exhausted (gateway auto-reset). + # bypass_cooldown=True, # #100661 provider-proven overflow + # #97488: timed transient guard — defer, never exhaustion (gateway auto-reset). + # #69870 lock-skip: the provider proved the request does not fit, but this compression pass + # no-oped only because another path holds the session's compression lock. Temporary defer, not + # exhaustion — refund the attempt and end the turn softly so the gateway does NOT auto-reset the + # session (#9893/#35809). + # #97488 transient-block: a timed guard (host-timeout cooldown / structural backoff) no-oped + # this pass — defer softly, never compression_exhausted (which would auto-reset the session). + # #69870 lock-skip: this pass no-oped because another path holds the session's compression lock + # — a temporary defer, not evidence about compressibility. compression_attempts -= 1 else: conversation_history = conversation_history_after_compression( diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index ceb549067e..a25f37a994 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -145,6 +145,9 @@ def _recover_unicode_encode_error( # Non-ASCII in the API key makes httpx fail encoding the Authorization header — the # usual persistent cause after message/tool sanitization. Entra ID bearer providers # are callables minting ASCII JWTs; skip them (``_strip_non_ascii`` would crash). + # Sanitize the API key — non-ASCII characters in credentials (e.g. ʋ instead of v from a bad copy-paste) + # cause httpx to fail when encoding the Authorization header as ASCII. This is the most common cause of + # persistent UnicodeEncodeError that survives message/tool sanitization (#6843). _credential_sanitized = False _raw_key = getattr(agent, "api_key", None) or "" if _raw_key and isinstance(_raw_key, str): @@ -543,6 +546,8 @@ def recover_after_classification( # Anthropic OAuth subscription rejected the 1M-context beta: disable it for this # session, rebuild the client, retry once. Reactive so capable subscriptions keep 1M. if ( + # See PR #17680 for the original report (we chose reactive recovery over the proposed unconditional + # omit so capable subscriptions don't silently lose the capability). classified.reason == FailoverReason.oauth_long_context_beta_forbidden and agent.api_mode == "anthropic_messages" and agent._is_anthropic_oauth @@ -698,6 +703,8 @@ def nonretryable_client_error_result( logger.error("%sNon-retryable client error: %s", agent.log_prefix, api_error) # Skip persistence on likely context-overflow (400 + large session): persisting the # failed message grows the session and repeats the failure. + # Persisting the failed user message would make the session even larger, causing the same failure on the + # next attempt. (#1630) if status_code == 400 and (approx_tokens > 50000 or len(api_messages) > 80): _vlines(agent, "⚠️ Skipping session persistence for large failed session to prevent growth loop.") else: @@ -981,6 +988,9 @@ def compute_error_backoff( _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") if _ra_raw: try: + # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap + # caused us to retry before the actual reset window and re-trip the limit. 600s covers all + # realistic provider reset windows while still rejecting pathological values. (#26293) _retry_after = min(float(_ra_raw), 600) except (TypeError, ValueError): pass @@ -1307,6 +1317,9 @@ def route_classified_error( # Overhead-aware request size so recovery arms on the true request # (msgs + tools + system), not the tool-blind message count. messages, active_system_prompt = agent._compress_context( + # Route the overhead-aware _real_tokens (computed above) into compression, not the bare + # last_prompt_tokens — which is 0 in the no-usage fallback, hiding the true request size + # from the engine's overflow guard (upstream PR #77169 review). messages, system_message, approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None), task_id=effective_task_id, @@ -1331,6 +1344,12 @@ def route_classified_error( is_rate_limited = classified.reason in _RATE_LIMIT_REASONS # Some relays wrap upstream output-cap 400s as 429 (rate_limit). Only the max_tokens # clamp fixes it. Parsed once; gates the eager-fallback exemption and overflow entry. + # Relay-wrapped output-cap errors: some gateways wrap an upstream "[400]: max_tokens (...) exceeds + # model's maximum output tokens (...)" as HTTP 429, which classifies as rate_limit. The failure is a + # deterministic request-shape problem — falling back to another provider (or burning generic retries) + # can't fix it, but the output-cap clamp below can, in one retry (#72281). Parse once here; the result + # gates both the eager-fallback exemption and the widened is_context_length_error entry, and is reused + # as available_out inside the handler. _wrapped_output_cap_budget = ( parse_available_output_tokens_from_error(error_msg) if classified.reason == FailoverReason.rate_limit else None @@ -1348,6 +1367,7 @@ def route_classified_error( if _should_fallback and agent._fallback_index < len(agent._fallback_chain): # No eager fallback while credential pool rotation may recover. Exception: an # upstream-aggregator 429 — the pool can't help, always fall back. + # Fixes #11314. _is_upstream = classified.reason == FailoverReason.upstream_rate_limit pool_may_recover = ( False if _is_upstream else _ra()._pool_may_recover_from_rate_limit(agent._credential_pool) diff --git a/agent/turn_request_assembly.py b/agent/turn_request_assembly.py index 09bf5a5728..09cdd4beb3 100644 --- a/agent/turn_request_assembly.py +++ b/agent/turn_request_assembly.py @@ -234,6 +234,10 @@ def assemble_api_request( approx_tokens = estimate_messages_tokens_rough(api_messages, charge_stale_thinking=False) # Route-aware: native Responses compaction prunes the wire payload, so the raw # history figure overstates it and fires needless local compression. + # Route-aware pressure: when the upcoming request is eligible for native Responses compaction the + # transport will checkpoint-prune the payload before sending — the generic durable-history figure + # overstates the wire by orders of magnitude on a compacted session and fires a 600s local compression + # the main request never needed (#96995, mirroring the turn-prologue preflight #96644/#96155). request_pressure_tokens = _midturn_request_pressure_tokens( agent, api_messages, effective_system or "", approx_tokens ) diff --git a/agent/turn_tool_round.py b/agent/turn_tool_round.py index 16f7c04e87..daa009ac80 100644 --- a/agent/turn_tool_round.py +++ b/agent/turn_tool_round.py @@ -205,6 +205,11 @@ def run_tool_round( agent._session_messages = messages # Touch activity so slow post-tool work plus a slow follow-up API call can't exceed # the gateway inactivity timeout (HERMES_AGENT_TIMEOUT). + # Touch activity before continuing so the gateway's inactivity monitor never sees a stale timestamp + # between tool completion and the start of the next API call. Without this, a tool-call result (which + # takes ~0s to process) followed by slow post-tool processing (compression, persist) and a slow + # follow-up API call can exceed the gateway inactivity timeout (HERMES_AGENT_TIMEOUT, default 1800s) and + # the gateway kills the session before the next activity touch fires (#69559, #69131). agent._touch_activity(f"tool results posted, continuing iteration #{api_call_count}") return _verdict("continue") diff --git a/agent/turn_truncation.py b/agent/turn_truncation.py index 8dd7ba548c..6c81d581d3 100644 --- a/agent/turn_truncation.py +++ b/agent/turn_truncation.py @@ -436,6 +436,14 @@ def continue_codex_incomplete( agent._vprint(f"{agent.log_prefix}↻ Codex response incomplete; continuing turn ({n}/3)") # Spinner/heartbeat notice: these retries can take minutes and otherwise look # like infinite thinking. + # #70773: same FD-recycle corruption vector as #67142. The shared OpenAI client's connection pool + # must NOT be closed from this watchdog/poll thread — worker threads from previous stale-killed + # attempts may still be unwinding their SSL BIOs. The request-local client is already closed above + # via _close_request_client_once. The shared client will be replaced lazily by + # _ensure_primary_openai_client on the next request. + # Surface the continuation on the live spinner/status line (CLI/TUI/Desktop) and gateway heartbeat: + # each of these retries can spend minutes waiting on the provider, and without a distinct notice the + # user only sees a generic thinking spinner ("infinite thinking", #64434). agent._emit_wait_notice( f"↻ model returned reasoning with no final answer — asking it to continue ({n}/3)" ) diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index a0cbb0137e..9d60c1a070 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -17,6 +17,8 @@ _ONE_MILLION = Decimal("1000000") _NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1" # Below $0.01, render at 4 dp so cheap-model costs never display as $0.00. +# Sub-cent cost threshold: below $0.01, render at 4 decimal places so the display is non-zero (e.g. $0.0046 +# instead of $0.00). See #79220. _SUBCENT_THRESHOLD = Decimal("0.01") # Attached to every CostResult with status="included" so consumers can @@ -27,13 +29,19 @@ _INCLUDED_NOTE = "subscription-included; no provider invoice for usage" def format_cost_label(amount: Decimal) -> str: """Cost display label: zero → "$0.00"; sub-cent → "~$0.0046" (4 dp, or "~$<0.0001" when it rounds to 0.0000 so the label never reads as zero); - else "~$1.23". Shared by per-response labels and insights cost buckets.""" + else "~$1.23". Shared by per-response labels and insights cost buckets. + + This fixes #79220 where sub-cent per-turn costs on cheap models (DeepSeek, etc.) rendered as "$0.00" + despite amount_usd carrying full Decimal precision. + """ if amount == _ZERO: return "$0.00" if amount < _SUBCENT_THRESHOLD: label = f"~${amount:.4f}" # Compare the rendered label: a naive `< 0.00005` threshold misses # the exact boundary under ROUND_HALF_EVEN. + # A positive amount that rounds to 0.0000 at 4 dp would render "~$0.0000" — a zero-looking label, + # the exact #79220 dishonesty. return label if label != "~$0.0000" else "~$<0.0001" return f"~${amount:.2f}" diff --git a/agent/verification_evidence.py b/agent/verification_evidence.py index d641a21b27..3ef2db03fd 100644 --- a/agent/verification_evidence.py +++ b/agent/verification_evidence.py @@ -135,6 +135,10 @@ def _transaction() -> Iterator[sqlite3.Connection]: ``sqlite3.Connection`` as a context manager only commits/rolls back; without the close, each call leaks a connection (and WAL/SHM fds) until GC runs. + + Using ``with _connect()`` alone therefore leaks a connection — and its WAL/SHM file descriptors — on + every call, deferring the close to the garbage collector, which over a long-running process can exhaust + ``RLIMIT_NOFILE`` (the cron-ledger sibling of this bug was #69567 / PR #69594). """ conn = _connect() try: diff --git a/agent/vision_message_prep.py b/agent/vision_message_prep.py index a1aee7800f..9a0e4c5172 100644 --- a/agent/vision_message_prep.py +++ b/agent/vision_message_prep.py @@ -294,7 +294,14 @@ class VisionMessagePrepMixin: def _anthropic_preserve_dots(self) -> bool: """True for anthropic-compatible endpoints that keep dots in model names (DashScope, MiniMax, Xiaomi - MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form).""" + MiMo, OpenCode Go/Zen, ZAI/Zhipu; Bedrock's dotted inference-profile IDs 400 on the hyphenated form). + + Alibaba/DashScope keeps dots (e.g. qwen3.5-plus). OpenCode Go/Zen keeps dots for non-Claude models + (e.g. minimax-m2.5-free). ``global.anthropic.claude-opus-4-7``, + ``us.anthropic.claude-sonnet-4-5-20250929-v1:0``) and rejects the hyphenated form with ``HTTP 400 + The provided model identifier is invalid``. Regression for #11976; mirrors the opencode-go fix for + #5211 + """ if (getattr(self, "provider", "") or "").lower() in { "alibaba", "minimax", "minimax-cn", "opencode-go", "opencode-zen", "zai", "bedrock", "xiaomi", "vertex", }: diff --git a/agent/web_search_provider.py b/agent/web_search_provider.py index 656d927734..48b24c7f1b 100644 --- a/agent/web_search_provider.py +++ b/agent/web_search_provider.py @@ -24,7 +24,11 @@ from agent.provider_base import ProviderBase def get_provider_env(name: str) -> str: """Config-aware env lookup (``os.environ`` first, then ``~/.hermes/.env``) so credentials set through the config layer are visible in gateway sessions / - delegate children / subprocess runs. Stripped value, or ``""`` when unset.""" + delegate children / subprocess runs. Stripped value, or ``""`` when unset. + + Falls back to a bare ``os.getenv`` when the config module is unavailable (stripped installs, early + import contexts). See #40190. + """ try: from hermes_cli.config import get_env_value diff --git a/agent/web_search_registry.py b/agent/web_search_registry.py index 813f06bd95..a62c5f26da 100644 --- a/agent/web_search_registry.py +++ b/agent/web_search_registry.py @@ -160,6 +160,12 @@ def _disabled_web_plugin_for(configured: Optional[str] = None, *, capability: Op backend because a disabled provider fails the availability gate and silently drops to the default. Bundled web plugins live under ``web/`` with the provider name differing only by hyphen/underscore, so both are normalized. + + When a user sets ``web.extract_backend: firecrawl`` (or the search equivalent) but also lists + ``web-firecrawl`` in ``plugins.disabled``, the provider never registers and the dispatcher would + otherwise emit a misleading "No web extract provider configured. Set web.extract_backend to ..." error — + even though the backend IS configured correctly. This helper detects that case so the dispatcher can + point the user at the actual cause (issue #40190 follow-up: pi314's disabled-plugin symptom). """ def _norm(s: str) -> str: return s.strip().lower().replace("-", "_") diff --git a/batch_runner.py b/batch_runner.py index 53caf4a1c6..9722bd27c4 100644 --- a/batch_runner.py +++ b/batch_runner.py @@ -269,6 +269,10 @@ def _process_single_prompt( "tool_stats": tool_stats, "reasoning_stats": reasoning_stats, "completed": result["completed"], + # Sibling of the non-empty-response return below (#64686): the classifier's failure_reason must + # survive the empty-response normalization path too, or downstream consumers (TUI billing + # surface, transient-failure persistence) lose the structured reason exactly when the run + # produced no text. "partial": result.get("partial", False), "api_calls": result["api_calls"], "toolsets_used": selected_toolsets, @@ -814,6 +818,12 @@ class BatchRunner: "total_prompts": len(self.dataset), "total_batches": len(self.batches), "batch_size": self.batch_size, + # Snapshot the CLI-level credential/runtime fields BEFORE mutating them so a failed in-place + # agent swap can roll the whole CLI back to the old working model. Otherwise the broken + # credentials staged below leak into the next turn's resolution even though the agent itself + # rolled back (#50163). + # Snapshot CLI-level fields before mutation so a failed in-place swap rolls the whole CLI back + # to the old working model (#50163). "model": self.model, "completed_at": datetime.now().isoformat(), "duration_seconds": round(time.time() - start_time, 2), diff --git a/cli.py b/cli.py index 8026de3d0e..a19f88212a 100644 --- a/cli.py +++ b/cli.py @@ -188,6 +188,10 @@ def _strip_reasoning_tags(text: str) -> str: """Strip reasoning blocks (closed, unterminated, orphan-close) and leaked tool-call XML from display text. Keep in sync with ``run_agent._strip_think_blocks`` and the stream consumer's think-tag sets. + + Also strips tool-call XML blocks some open models leak into visible content (````, + ````, Gemma-style ``…``). Ported from + openclaw/openclaw#67318. """ cleaned = text for tag in _REASONING_TAGS: @@ -613,6 +617,11 @@ _cli_wake_owner = None _single_query_finalize_attempted_session_ids: set[str | None] = set() # /handoff sessions belong to the gateway: finalizing them here would stamp end_reason on # a row the gateway just reopened, making the handoff leg vanish from history. +# Session IDs that were handed off to the gateway via /handoff. The CLI process exits after a successful +# handoff, but the gateway now owns the session lifecycle — _run_cleanup must NOT call finalize_session on +# these, because doing so sets end_reason on a row the gateway just reopened and is actively writing to +# (#88234). The race made the handoff leg vanish from session history and broke session_search recall for +# the handed-off session. _handed_off_session_ids: set[str | None] = set() _active_agent_ref = None # active AIAgent, for memory-provider shutdown at exit _deferred_agent_startup_done = False @@ -621,6 +630,9 @@ _deferred_agent_startup_done = False _tui_input_modes_active = False +# Set True once the TUI's prompt_toolkit app starts (which enables focus reporting + mouse tracking). Gates +# the on-exit terminal reset so non-TUI one-shot CLI runs — which also register _run_cleanup via atexit — +# don't emit escape codes for modes they never enabled (#36823). def _mark_tui_input_modes_active() -> None: """Record that the TUI app started, so _run_cleanup resets input modes.""" global _tui_input_modes_active @@ -688,6 +700,10 @@ def _arm_exit_watchdog(timeout_s: float | None = None, *, from_signal: bool = Fa Backstop for a cleanup step wedged on network I/O and for interpreter teardown blocked joining non-daemon threads (ThreadPoolExecutor's atexit join). The daemon timer survives ``Py_FinalizeEx``'s joins. ``HERMES_EXIT_WATCHDOG_S=0`` disables. + + 1. 2. Interpreter teardown blocked joining non-daemon threads — stdlib ``ThreadPoolExecutor`` workers + are joined unconditionally by ``concurrent.futures``' atexit hook even after ``shutdown(wait=False)``, + so one tool thread wedged on a socket held the process open forever (#27563 class). """ if timeout_s is None: timeout_s = _exit_watchdog_timeout() @@ -728,6 +744,13 @@ def _arm_exit_watchdog_on_shutdown_signal() -> None: (main thread in a syscall, prompt_toolkit teardown never returning). Leash is 2x the cleanup timeout so a progressing cleanup is never cut short. Never arm at startup: the timer exits unconditionally. + + SIGTERM/SIGHUP establish unambiguous shutdown intent, but the graceful path from signal → + ``agent.interrupt()`` → ``app.exit()`` / ``KeyboardInterrupt`` → ``finally`` → ``_run_cleanup`` has + several wedge points BEFORE ``_run_cleanup`` arms the normal watchdog: a main thread parked in a syscall + that never observes the unwind, a prompt_toolkit teardown that never returns, or an agent worker + blocking the ``finally``. When that happens the process has NO backstop and a "dead" CLI lingers + (observed: ``hermes --tui`` alive ~47 min at 4% CPU after terminal close — the #65998 class). """ global _signal_watchdog_armed if _signal_watchdog_armed: @@ -754,6 +777,9 @@ def _shutdown_agent_memory_provider(agent) -> None: # no-arg fallback for stubs / partially-initialised agents. _session_msgs = getattr(agent, '_session_messages', None) _sid = getattr(agent, "session_id", None) or "" + # ``_session_messages`` is set on ``AIAgent.__init__`` and refreshed every turn via + # ``_persist_session``. Fall back to no-arg on test stubs / partially-initialised agents where the + # attribute is missing. See #15165. if isinstance(_session_msgs, list): logger.info("CLI cleanup calling memory shutdown for session %s with %d message(s)", _sid, len(_session_msgs)) agent.shutdown_memory_provider(_session_msgs) @@ -805,6 +831,7 @@ def _run_cleanup(*, notify_session_finalize: bool = True): _arm_exit_watchdog() # Reset terminal input modes FIRST: teardown below can take seconds and a later # step raising must not skip the reset. No-op unless the TUI ran. + # See #36823. _reset_terminal_input_modes_on_exit() for step, swallow in _CLEANUP_STEPS: @@ -824,6 +851,8 @@ def _run_cleanup(*, notify_session_finalize: bool = True): def _should_emit_cleanup_session_finalize(session_id: str | None) -> bool: # A handed-off session is owned by the gateway process — never finalize it here. + # The CLI must not finalize it on exit — that sets end_reason on a row the gateway reopened and is + # actively writing to, causing the handoff leg to vanish from session history (#88234). if session_id is not None and session_id in _handed_off_session_ids: return False if not _single_query_finalize_attempted_session_ids: @@ -898,6 +927,16 @@ def _flush_one_shot_session_store(cli) -> None: One-shot runs get a single turn, so nothing retries a transiently-failed transcript flush, closes the session row, or drains token deltas the kanban ``os._exit(0)`` path skips. Handed-off sessions are left alone. + + - a turn whose in-loop ``_flush_messages_to_session_db`` failed under write-lock contention (e.g. a busy + multiplex gateway sharing state.db) was silently lost — the reply reached stdout and agent.log but the + resumed session's stored history never changed (#88583); - the resumed/created titled session row was + left dangling open (``ended_at``/``end_reason`` NULL) on every one-shot exit; - queued async + token-accounting deltas relied on interpreter-exit hooks, which the kanban SIGTERM path's + ``os._exit(0)`` skips entirely. + Idempotent and best-effort: ``_persist_session`` dedupes via the per-message ``_DB_PERSISTED_MARKER`` + stamps (already-written turns are not re-written) and ``end_session`` no-ops on an already-ended row. + See #88234. """ agent, session_id = _oneshot_agent_and_session(cli) if agent is None or not session_id or session_id in _handed_off_session_ids: @@ -930,6 +969,8 @@ def _wait_for_oneshot_background_completions(cli) -> None: Waits on the whole registry: a one-shot process hosts one agent, and task_id filtering would skip processes registered before the session id settled. + + See #90879. """ from tools.process_registry import process_registry @@ -971,6 +1012,11 @@ def _reset_terminal_input_modes_on_exit() -> None: Ctrl+C / SIGTERM / crashes bypass prompt_toolkit's unwind, leaving focus events and mouse reports as visible text in the next shell. Writes to stdout when it is the terminal, else /dev/tty (the TUI may have run with stdout redirected). + + Called from ``_run_cleanup`` (atexit-registered + invoked on the normal / EOF / interrupt exit paths) + this covers normal quit, Ctrl+C and SIGTERM/SIGHUP. ``kill -9`` is uncatchable, and the kanban worker's + ``os._exit(0)`` path bypasses ``atexit``; neither runs this — but both are non-TTY / non-TUI, so there + is nothing to reset there. See #36823. """ global _tui_input_modes_active if not _tui_input_modes_active: @@ -1025,6 +1071,8 @@ from hermes_cli.worktree_ops import ( # noqa: F401 (mixins/tests/worktree_gc i _worktree_merge_cache_path, ) +# ============================================================================= Git Worktree Isolation +# (#652) ============================================================================= _active_worktree: Optional[Dict[str, str]] = None @@ -1182,6 +1230,9 @@ def _query_osc11_background() -> str | None: answer DA1, so its reply proves our OSC 11 was processed — otherwise a late reply leaks into prompt_toolkit's stdin as typed text. Skipped over SSH (round-trip too slow; a late BEL reads as Ctrl+G). A 50 ms drain after TCSAFLUSH catches stragglers. + + After the main read + TCSAFLUSH, a short drain window (50 ms) catches late-arriving bytes that slipped + past the flush — a race observed on VPS and container terminals under load (#40250). """ if not sys.stdin.isatty() or not sys.stdout.isatty(): return None @@ -1626,6 +1677,9 @@ def _cprint(text: str): loop = None try: # get_running_loop(): get_event_loop() warns from threads with no current loop. + # Use get_running_loop() instead of get_event_loop() to avoid the DeprecationWarning / + # RuntimeWarning emitted by Python 3.10+ when get_event_loop() is called from a thread that has no + # current event loop set (e.g. the process_loop background thread). Fixes #19285. current_loop = _asyncio.get_running_loop() except Exception: current_loop = None @@ -1880,6 +1934,10 @@ def _hermes_call_output_screen_diff( Inflates ``previous_screen.height`` when the new screen is taller so pt skips the cursor move that stamps chrome into scrollback; on a corrupt previous paint buffer (tmux re-attach) retries once as a first paint instead of crashing the loop. + + 1. 2. On AttributeError/TypeError from a corrupt previous paint buffer (classic after tmux attach with + same width), retry once with ``previous_screen=None`` so pt first-paints cleanly instead of crashing the + event loop with ``'cell' object has no attribute 'char'``. See #26137. """ try: if previous_screen is not None and hasattr(previous_screen, "height") and previous_screen.height < screen.height: @@ -1962,6 +2020,11 @@ def _apply_bracketed_paste_timeout_patch() -> None: # CPR replies (``ESC[;R``) can race past the input parser under resize storms # and land as literal text; the ``^[[...R`` form appears when a filter stripped the ESC. +# Cursor Position Report (CPR / DSR) response, format ``ESC[;R``. prompt_toolkit's _on_resize() + +# renderer send ``ESC[6n`` queries to the terminal; under resize storms or tab switches the terminal's reply +# can race past the input parser and end up in the input buffer as literal text (see issue #14692). Also +# matches the visible-form ``^[[;R`` that appears when the ESC byte was stripped by a prior +# filter. _DSR_CPR_ESC_RE = re.compile(r"\x1b\[\d+;\d+R") _DSR_CPR_VISIBLE_RE = re.compile(r"\^\[\[\d+;\d+R") _SGR_MOUSE_ESC_RE = re.compile(r"\x1b\[<\d+;\d+;\d+[Mm]") @@ -1990,6 +2053,9 @@ def _is_ghostty_terminal(env: Optional[Mapping[str, str]] = None) -> bool: Ghostty gets ONLY modifyOtherKeys: its Kitty disambiguate mode strips Alt from Backspace (upstream bug), breaking backward-kill-word. + + Ghostty implements modifyOtherKeys correctly (it then emits ``\\x1b[27;3;127~``, which the alias table + also maps). See #87630. """ env = os.environ if env is None else env return (env.get("TERM_PROGRAM") or "").strip() == "ghostty" or (env.get("TERM") or "").strip().lower() == "xterm-ghostty" @@ -2017,6 +2083,17 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] = stock prompt_toolkit barely maps (Ctrl+C once arrived as ``ESC[99;5u``), so ``install_modify_other_keys_aliases()`` must have run first. Ghostty gets only modifyOtherKeys. The exit reset pops both modes. + + Under either protocol the terminal re-encodes modified keys as escape sequences — Kitty disambiguate + mode as ``ESC[;u`` (plus the Esc key as ``ESC[27u``), modifyOtherKeys=2 as + ``ESC[27;;~``. Stock prompt_toolkit 3.x maps almost none of these, which is why the CSI + >1u push was temporarily removed in 87074 (Ctrl+C arrived as ``ESC[99;5u`` and died, #56684). + ``install_modify_other_keys_aliases()`` (called at CLI startup from ``hermes_cli.pt_input_extras``) now + populates ``ANSI_SEQUENCES`` with the full Ctrl/Alt/Shift/multi-modifier and functional-key tables under + BOTH formats, so every existing key binding continues to fire — including Ctrl+C, which is handled by + prompt_toolkit's ``c-c`` binding (raw mode clears ISIG, so the kernel INTR path was never in play for + the CLI). + See #87630. """ if not _terminal_supports_extended_enter_keys(env): return False @@ -2057,7 +2134,10 @@ def _apply_backslash_line_continuation(text: str) -> str: def _preserve_ctrl_enter_newline() -> bool: - """Environments delivering Ctrl+Enter as bare LF (Windows Terminal, WSL, SSH, Ghostty): c-j must stay newline.""" + """Environments delivering Ctrl+Enter as bare LF (Windows Terminal, WSL, SSH, Ghostty): c-j must stay newline. + + See issue #22379. + """ env = os.environ if ( sys.platform == "win32" @@ -2079,7 +2159,12 @@ def _preserve_ctrl_enter_newline() -> bool: def _bind_prompt_submit_keys(kb, handler, *, multiline_shortcuts_enabled: Optional[bool] = None) -> None: - """Enter always submits; c-j submits only with multiline shortcuts off AND where Ctrl+Enter isn't c-j.""" + """Enter always submits; c-j submits only with multiline shortcuts off AND where Ctrl+Enter isn't c-j. + + Even when the setting is disabled, environments where Ctrl+Enter is known to arrive as c-j (Windows, + WSL, SSH, Windows Terminal, Ghostty) keep c-j reserved for newline; otherwise Ctrl+Enter submits instead + of composing. See _preserve_ctrl_enter_newline() and issue #22379. + """ if multiline_shortcuts_enabled is None: multiline_shortcuts_enabled = _cli_multiline_shortcuts_enabled() kb.add("enter")(handler) @@ -2094,12 +2179,23 @@ def _disable_prompt_toolkit_cpr_warning(app) -> None: def _terminal_may_leak_cpr() -> bool: - """Suppress prompt_toolkit CPR queries (delayed replies leak into input); Windows keeps pt's default.""" + """Suppress prompt_toolkit CPR queries (delayed replies leak into input); Windows keeps pt's default. + + Delayed CPR replies (``ESC[;R`` / visible ``^[[;R``) leak into the status line and + can freeze input when the reply is slow (#13870 on SSH/slow PTYs). The same race hits local POSIX TTYs + under heavy subagent / status-line load — see ``tests/cli/test_cpr_local_leak.py``. + """ return os.environ.get("PROMPT_TOOLKIT_NO_CPR", "") == "1" or sys.platform != "win32" def _build_cpr_disabled_output(stdout): - """Vt100_Output with ``enable_cpr=False`` (``from_pty()`` doesn't expose it), or None on failure.""" + """Vt100_Output with ``enable_cpr=False`` (``from_pty()`` doesn't expose it), or None on failure. + + prompt_toolkit's renderer sends ``ESC[6n`` (Device Status Report) to learn the cursor row before + painting in non-fullscreen mode; the terminal replies ``ESC[;R``. When that reply is delayed + it races into the display as raw ``^[[39;1R`` and can stall the renderer's pending-CPR future (#13870; + also local POSIX under heavy subagent load). + """ try: import io as _io from prompt_toolkit.output.vt100 import Vt100_Output, _get_size @@ -2370,7 +2466,15 @@ def save_config_value(key_path: str, value: any) -> bool: def _normalize_moa_model(model: Optional[str]) -> tuple[Optional[str], Optional[str]]: - """``moa:`` -> ``("moa", preset)`` (same routing as ``/moa``); anything else -> ``(None, model)``.""" + """``moa:`` -> ``("moa", preset)`` (same routing as ``/moa``); anything else -> ``(None, model)``. + + Returns ``("moa", "")`` when *model* selects the MoA virtual provider, otherwise ``(None, + model)`` unchanged. This gives non-interactive ``hermes chat -Q -m moa:`` the same routing the + interactive ``/moa`` command and the model picker already use: ``resolve_runtime_provider`` handles + ``requested_provider == "moa"`` and ``agent_init`` builds the MoAClient off ``provider == "moa"``. + Without this the raw ``moa:`` string is sent to the real provider and rejected with a 401/400 + "model not supported" (#56828). + """ if isinstance(model, str) and model.strip().lower().startswith("moa:"): preset = model.strip().split(":", 1)[1].strip() if preset: @@ -2381,7 +2485,12 @@ _split_model_config_default = _lazy_shim("hermes_cli.config", "split_model_confi class _VoiceInputMessage: - """Sentinel for voice-transcribed input so the concise voice prefix never applies to typed text.""" + """Sentinel for voice-transcribed input so the concise voice prefix never applies to typed text. + + Distinguishes STT output from manually typed text while voice mode is active, so the + concise-voice-response prefix is applied only to messages that actually came from the microphone + (#65827). + """ __slots__ = ("text",) @@ -2603,6 +2712,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix _startup_api_key_override = _startup_route.api_key # ``moa:`` selects the MoA virtual provider before provider resolution so the # real provider never sees the unknown model; the prefix wins over --provider. + # A ``moa:`` model string selects the MoA virtual provider in one shot (parity with + # interactive ``/moa`` and the model picker). See #56828. _moa_provider_override, self.model = _normalize_moa_model(self.model) _env_mt = os.environ.get("HERMES_MAX_TOKENS") _mt = _model_config.get("max_tokens") @@ -2617,6 +2728,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix self._model_is_default = not model and not _config_model # --api-key wins; otherwise a URL-bearing startup alias carries its own credential. + # See #28660. self._explicit_api_key = api_key or _startup_api_key_override or None self._explicit_base_url = base_url @@ -2627,6 +2739,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix ) # `--provider ` without `-m` uses that entry's default_model, else the global # default goes to the custom endpoint and the compressor gets the wrong context length. + # Explicit `-m` still wins. See #86978. if not model and provider: try: from hermes_cli.runtime_provider import _get_named_custom_provider @@ -2705,6 +2818,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix self.prefill_messages = _load_prefill_messages(_resolve_prefill_messages_file(CLI_CONFIG)) # Per-model override > global reasoning_effort. + # Reasoning config (OpenRouter reasoning effort level) Per-model override > global reasoning_effort + # — resolved through the shared chokepoint in hermes_constants (Closes #21256). from hermes_constants import resolve_reasoning_config self.reasoning_config = resolve_reasoning_config(CLI_CONFIG, self.model) # --reasoning wins for this run only (never persisted); unparseable -> warn and ignore. @@ -2771,6 +2886,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix except Exception as e: # Without a store the transcript is NOT persisted while the chat looks healthy, # so surface it prominently rather than only logging. + # #41386: a failed session store means the transcript is NOT persisted to state.db — the live + # chat looks healthy but resume later shows a truncated/empty session. A buried log line is not + # enough; surface it prominently so the user knows persistence is off for this run and can fix + # the store before relying on resume. self._session_db_unavailable = True logger.warning("Failed to initialize SessionDB — session will NOT be indexed for search: %s", e) try: @@ -2798,6 +2917,9 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix self._terminal_io_broken = False # stdout EIO: freeze UI paints instead of spinning self._delete_session_on_exit = False # /exit --delete # /update: relaunch() runs from run() after prompt_toolkit restored terminal modes. + # /exit --delete: when True, the current session's SQLite history and on-disk transcripts are + # deleted during shutdown. Set by process_command() when the user runs /exit --delete or /quit + # --delete. Ported from google-gemini/gemini-cli#19332. self._pending_relaunch: list[str] | None = None self._last_ctrl_c_time = 0 # Blocking-prompt overlays (clarify / sudo / approval / slash-confirm / model picker). @@ -2889,6 +3011,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix surface=surface, config=self.config, # Writer identity: a re-claim by this process replaces its own entry. + # See #94595. metadata={"live_session_id": str(self.session_id)}, ) except Exception as exc: @@ -3143,7 +3266,12 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # A bare `/resume` prompt is one-shot: any other command disarms it so a later # number isn't swallowed as a stale selection. + # See #34584. if canonical not in {"resume", "sessions"}: + # Armed when a bare `/resume` prints the recent-sessions list so the very next bare numeric + # input (e.g. `3`) resolves to that session. Holds the exact list used for index resolution; + # one-shot (cleared on the next submitted input, whether it's the selection or anything else). + # See #34584. self._pending_resume_sessions = None entry = self._slash_handler(canonical) @@ -3202,6 +3330,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix timeout=30, env=build_subprocess_env(), creationflags=windows_hide_flags(), # no console flash on Windows (#56747) ) + # See #56747. output = result.stdout.strip() or result.stderr.strip() if output: from agent.redact import redact_sensitive_text @@ -3333,6 +3462,9 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix Busy-time input lands in ``_interrupt_queue`` and is only drained by the explicit interrupt path; a turn that finishes naturally would otherwise strand it and the CLI appears to hang. Never raises. + + Called once at the end of every turn from ``process_loop``'s ``finally`` block. Catches and swallows + ``Exception`` because the drain must never break the main loop. (#20271) """ try: while not self._interrupt_queue.empty(): @@ -3351,6 +3483,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # Inline tokens that bypass the destructive-slash confirmation modal (scripting, or # when the modal can't be marshaled onto the app loop). + # A general escape hatch for non-interactive use (scripting/automation) and for the degraded path where + # the modal can't be marshaled onto the app loop — lets users self-serve without flipping + # approvals.destructive_slash_confirm in config. (Native Windows now drives the modal normally — see + # #33961.) _DESTRUCTIVE_SKIP_TOKENS = frozenset({"now", "--yes", "-y"}) def _tui_process_loop(self): @@ -3385,6 +3521,8 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix def _tui_unwrap_input(self, user_input): """Unwrap ``_VoiceInputMessage`` / ``_SeededQueryMessage`` -> ``(text_or_tuple, is_voice_input, is_seeded_query)``.""" + # Voice-transcribed messages arrive wrapped in a sentinel so only genuine STT output gets the voice + # prefix (#65827). is_voice_input = isinstance(user_input, _VoiceInputMessage) if is_voice_input: user_input = user_input.text @@ -3487,6 +3625,10 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix if self._last_turn_interrupted: self._recover_terminal_after_interrupt() + # Re-queue any messages that arrived in _interrupt_queue while the agent was running and were never + # claimed by the explicit interrupt path. See _drain_interrupt_queue_to_pending_input for the full + # rationale. Regression of #17666 / #18760 — the drain block from the original PR #17939 was + # deferred as "worth its own review" and never re-landed (#20271). self._drain_interrupt_queue_to_pending_input() # /goal continuation (queued user input still preempts), then /loop tick completion. @@ -3529,6 +3671,12 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix with suppress(Exception): logger.debug("Received signal %s, triggering graceful shutdown", signum) # Arm the backstop IMMEDIATELY: if the unwind wedges, _run_cleanup never arms its own. + # Shutdown intent is now unambiguous — arm the exit backstop IMMEDIATELY, before the graceful unwind + # below. If any step of that unwind wedges (main thread parked in a syscall, prompt_toolkit teardown + # never returning), _run_cleanup never runs and would never arm its own watchdog — leaving a "dead" + # CLI alive for minutes (#65998 class). + # Arm the exit backstop now that shutdown intent is unambiguous — covers wedges in the unwind below + # that would otherwise leave the process alive with no watchdog (#65998 class). _arm_exit_watchdog_on_shutdown_signal() if self._agent_running: _interrupt_agent_for_signal(self.agent, signum) @@ -3616,6 +3764,9 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # Redaction is ON by default; be loud when the operator turned it off. with suppress(Exception): + # The redactor snapshots its state at import time so any toggle now won't affect the running + # process — we just want the operator to see that they're running without the safety net. See + # #17691. _redact_raw = os.getenv("HERMES_REDACT_SECRETS", "true") if _redact_raw.lower() not in {"1", "true", "yes", "on"}: self._console_print( @@ -3690,6 +3841,12 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # 0 (default) avoids fighting terminal auto-scroll in non-fullscreen mode. refresh_interval=float(CLI_CONFIG.get("display", {}).get("cli_refresh_interval", 0)), # Erase the bottom chrome on exit instead of freezing a copy into scrollback. + # Without this, prompt_toolkit's render_as_done teardown repaints the chrome one last time and + # leaves it stranded above the exit summary — so a dead status bar + empty prompt sit between + # the conversation transcript and the "Resume this session" block, and stack with the next + # session's UI on resume (#38252). The actual conversation transcript is printed through + # patch_stdout into normal scrollback and is unaffected; only the managed chrome is erased. + # Applies to every exit path (/exit, /quit, EOF, Ctrl+C). erase_when_done=True, **extra_kw, ) @@ -3760,6 +3917,16 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # paint, pushing chrome into scrollback where a column-shrink reflows it into # duplicates. Wrapping _output_screen_diff keeps its reserve-space branch from firing. try: + # Background: prompt_toolkit's renderer (renderer.py L232-242) explicitly moves the cursor to + # the bottom of the canvas after painting "to make sure the terminal scrolls up, even when the + # lower lines of the canvas just contain whitespace". In non-fullscreen mode this scrolls chrome + # content (status bar, input rules) into terminal scrollback on every render. When the terminal + # column-shrinks, the emulator reflows the previously rendered full-width rows into multiple + # narrower rows that get pushed up — leaving ghost duplicates AND polluting scrollback. Same + # issue as pt #29 (open since 2014), #1675, #1933. Surgical fix: wrap _output_screen_diff so + # that when its internal `if current_height > previous_screen.height` branch fires (the one that + # does the bottom-cursor-move), we make it fall through by inflating previous_screen.height + # first. import prompt_toolkit.renderer as _pt_renderer from prompt_toolkit.renderer import _output_screen_diff as _orig_osd @@ -3791,12 +3958,22 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix try: with patch_stdout(): try: + # run_in_terminal() may return either: • a coroutine / Future (prompt_toolkit ≥ 3.0) — + # must be scheduled via ensure_future so the coroutine is actually awaited; calling it + # bare would leave it unawaited and silently drop the output (fixes #23185 Bug A). • + # None (some mocks / older PT builds) — just call the inner function directly since PT + # already executed it synchronously. Do NOT fall back to a bare _pt_print when + # ensure_future raises, because run_in_terminal already invoked the lambda in that case + # (the mock path), which would double-print the line. import asyncio as _aio _aio.get_running_loop().set_exception_handler(self._tui_suppress_closed_loop_errors) except Exception: pass # no running loop -- nothing to patch # Record that the app enables focus reporting + mouse tracking so _run_cleanup # resets them; extended key modes are popped by the same reset. + # When multiline shortcuts are on, also ask supported terminals (e.g. iTerm2) to report + # modified keys distinctly (kitty protocol + modifyOtherKeys); the cleanup reset pops both + # modes. See #36823. _mark_tui_input_modes_active() if self._tui_multiline_shortcuts: _enable_extended_enter_keys(app.output) @@ -4095,12 +4272,26 @@ def _install_single_query_signal_handlers(cli): # Kanban: a non-daemon worker blocked in _wait_for_process survives KeyboardInterrupt # and the dispatcher sees 'running' forever, so os._exit(0) (SIGALRM deadman guards # a blocking flush). That skips atexit + the token-drain hook, hence the explicit flush. + # Kanban worker exit path (#28181): SIGTERM hits a dispatcher-spawned worker that's likely in a + # non-daemon thread waiting on a child subprocess in _wait_for_process. Raising KeyboardInterrupt + # only unwinds the main thread; the worker thread keeps running, the process gets reparented to + # init, and the dispatcher's _pid_alive check returns True forever — task stuck in 'running' + # indefinitely. Skip the controlled-unwind dance and call os._exit(0) so the kernel reclaims the PID + # immediately and detect_crashed_workers can reclaim the stale claim on the next tick. Flush logging + # + stdout/stderr first so the final debug trace isn't lost; SIGALRM deadman guards the flush + # against any rare blocking-I/O case (the reporter measured flush in <1ms; the alarm is a failsafe, + # not the common path). if os.environ.get("HERMES_KANBAN_TASK"): with suppress(Exception): if hasattr(_signal, "SIGALRM"): _signal.signal(_signal.SIGALRM, lambda *_: os._exit(0)) _signal.alarm(5) with suppress(Exception): + # Durable flush FIRST: memory-provider shutdown inside _run_cleanup can issue aux-LLM calls, + # and nothing after it may fail in a way that loses the turn (#88583). + # os._exit(0) skips atexit AND SessionDB's token-drain hook, so flush + finalize the session + # store here or the worker's turn (and its usage deltas) never become durable (#88583 / + # #50881 class). Best-effort under the SIGALRM deadman above. _flush_one_shot_session_store(cli) _flush_logging_and_stdio() os._exit(0) @@ -4269,6 +4460,13 @@ def _run_single_query_mode(cli, query, image, quiet, oneshot): return cli.run() cli._single_query_mode = True # agent waits the full MCP cold-start before its only tool snapshot # No user can answer approval prompts: the approval gate takes the deterministic path. + # One-shot mode: no between-turns MCP late-binding refresh, so the agent must wait the full MCP + # cold-start bound before its first (and only) tool snapshot. See #51316. + # Mark single-query for the approval gate. cli.py sets HERMES_INTERACTIVE earlier for interactive sudo + # prompts, but a -q run has NO user waiting to answer approval prompts. The gate reads this marker (via + # gateway.session_context.get_session_env, which falls back to os.environ when the session-context layer + # isn't engaged) and takes the deterministic approvals.single_query_mode path instead of waiting the + # full timeout. See #86878. os.environ["HERMES_SINGLE_QUERY_SESSION"] = "1" if not cli._claim_active_session("cli", stderr=bool(quiet)): sys.exit(1) diff --git a/cron/jobs.py b/cron/jobs.py index b78ffaf5e7..34a484f990 100644 --- a/cron/jobs.py +++ b/cron/jobs.py @@ -59,6 +59,11 @@ def _ensure_croniter() -> bool: # Cron is per-profile by design: anchor at get_hermes_home() (active profile home), NOT # get_default_hermes_root() — the shared root would funnel every profile's jobs into one jobs.json # and run them under the ticker's HERMES_HOME, leaking config/credentials/skills across profiles. +# Each profile owns its own cron store under its own HERMES_HOME, and a profile-scoped gateway runs that +# profile's jobs under that same HERMES_HOME — so a job authored in profile `coder` lives in +# `~/.hermes/profiles/coder/cron/jobs.json` and executes with `coder`'s `.env`, `config.yaml`, and skills. +# Do NOT change this to the default root: that re-breaks per-profile isolation. See also the dynamic +# `_get_hermes_home()` / `_get_lock_paths()` resolution in cron/scheduler.py. See #4707. HERMES_DIR = get_hermes_home().resolve() # Default-profile fallback and compatibility surface for callers/tests. Cross-profile callers must # scope paths with use_cron_store() instead of mutating these process-wide. @@ -66,6 +71,9 @@ CRON_DIR = HERMES_DIR / "cron" JOBS_FILE = CRON_DIR / "jobs.json" # Heartbeat: touched every ticker loop so `hermes cron status` can tell the ticker THREAD is alive, # not just the gateway PROCESS; success = last tick that completed WITHOUT raising. +# The gateway process and the (separate) ``hermes cron status`` process share it so status can tell whether +# the ticker THREAD is alive, not just whether the gateway PROCESS exists — a ticker that dies silently +# inside a live gateway would otherwise report healthy (#32612, #32895). TICKER_HEARTBEAT_FILE = CRON_DIR / "ticker_heartbeat" TICKER_SUCCESS_FILE = CRON_DIR / "ticker_last_success" # Single source of truth for the ticker interval (scheduler_provider.py) and the staleness @@ -169,7 +177,13 @@ def _oneshot_run_claim_ttl_seconds() -> float: def _job_running_in_this_process(job_id: str) -> bool: """True when the scheduler in THIS process is still running ``job_id``: the run_claim TTL alone cannot distinguish "claiming tick died" from "alive but slow". Lazy import: scheduler imports - us.""" + us. + + Direct liveness signal for stale-entry recovery (#62002): the run_claim TTL alone cannot distinguish + "the claiming tick died" from "the run is alive but slow" — a run stalled on network I/O (or a laptop + that slept mid-run) legitimately outlives the TTL. The in-process ticker and the run share this process, + so the scheduler's running set settles the common single-gateway case without any claim-age guesswork. + """ try: from cron.scheduler import get_running_job_ids return job_id in get_running_job_ids() @@ -242,6 +256,7 @@ def _jobs_lock(): # jobs.json stamp as of this section's load_jobs(): lets _save_jobs_unlocked skip the # shrink-merge parse when the file provably hasn't changed. Reset on entry/exit so stale # stamps from unlocked loads or prior sections can never suppress a needed merge. + # See #80703. _jobs_lock_state.load_stamp = None lock_fd = None try: @@ -495,7 +510,12 @@ def _is_recoverable_error_job(job: Dict[str, Any]) -> bool: """True for a recurring job stuck in ``state=error`` (set ONLY when ``compute_next_run()`` fails for a cron/interval job: croniter missing, malformed schedule). Such a job still has future occurrences once the issue resolves, so treating it as terminal would block due-scan self-heal, - pre-advance, dispatch claim and ``resume_job`` — wedging it forever.""" + pre-advance, dispatch claim and ``resume_job`` — wedging it forever. + + Unlike ``state=completed`` (a one-shot that genuinely has no more occurrences, ever), an error-state + recurring job still has a schedule with future occurrences once the underlying issue resolves — it is + stuck pending a ``next_run_at`` recompute, not truly done. See #16265. + """ return ( job.get("state") == "error" and (job.get("schedule") or {}).get("kind") in {"cron", "interval"} @@ -571,7 +591,14 @@ def ensure_dirs(): def normalize_repeat_value(repeat: Any) -> Optional[int]: """Coerce a repeat value (int or user-facing string) into ``Optional[int]``: ``'forever'``-family -> None, ``'once'``-family -> 1, numeric -> int, 0/negative -> None, - else ValueError.""" + else ValueError. + + The tool schema exposes ``repeat`` as an integer, but agents and users legitimately pass the user-facing + strings ``'forever'``/``'once'`` or numeric strings (``'3'``). Uncoerced strings previously died with + ``'<=' not supported between instances of 'str' and 'int'`` at create + (#66824/#64520/#7142/#71987/#95706) and were stored raw by update paths, breaking ``mark_job_run`` + later. + """ if repeat is None: return None if isinstance(repeat, str): @@ -716,6 +743,7 @@ def parse_schedule(schedule: str) -> Dict[str, Any]: is_every = schedule_lower.startswith("every ") rest = schedule[6:].strip() if is_every else schedule_lower cron_expr = _natural_every_to_cron(rest) + # Reuse the same helper — the phrase shape is identical without the "every " prefix. See #51975. if cron_expr is not None: example = "every monday 9am" if is_every else "weekdays at 9am" return _cron_schedule( @@ -737,6 +765,11 @@ def parse_schedule(schedule: str) -> Dict[str, Any]: dt = datetime.fromisoformat(schedule.replace('Z', '+00:00')) # Naive timestamps become aware in the CONFIGURED Hermes timezone (not server-local): # the due-check compares against hermes_time.now(). + # Make naive timestamps timezone-aware at parse time so the stored value doesn't depend on the + # system timezone matching at check time. UTC) while now() runs in Asia/Kolkata, the stored + # instant would land hours off from the user's wall-clock intent — far enough that one-shots + # never become due and recurring jobs fire at the wrong time. Using the configured zone makes + # "20:07" mean 20:07 on the same clock the scheduler checks against (#51021). if dt.tzinfo is None: dt = dt.replace(tzinfo=_hermes_now().tzinfo) return { @@ -834,6 +867,7 @@ def _compute_grace_seconds(schedule: dict) -> int: # A recurring dispatch within this many seconds of schedule renders "on time": a busy once-a-minute # ticker can slip a couple of minutes — normal cadence, not gateway downtime. +# See #99879. _LATE_DISPATCH_TOLERANCE_SECONDS = 300 @@ -861,7 +895,13 @@ def _job_is_stale_error_recurring( """True when a recurring job (caller-checked) is wedged in a stale persisted error state: ``last_status == "error"``, not running in this process (never re-arm a live run underneath itself), and ``last_run_at`` older than ``cadence + grace`` (a job merely erroring-and-retrying - on schedule stays fresh and is not flagged).""" + on schedule stays fresh and is not flagged). + + Condition (all must hold): * it has NOT successfully re-fired within its natural cadence — its + ``last_run_at`` is older than ``cadence + grace``, so this is not a normal transient-error retry that + will fire on its own soon, it is a job that has been sitting errored for a full period with no recovery; + See #62002. + """ if job.get("last_status") != "error": return False if _job_running_in_this_process(str(job.get("id") or "")): @@ -960,7 +1000,14 @@ def _cron_next_run_matches_expr(schedule: Dict[str, Any], next_run_dt: datetime) """Whether ``next_run_dt`` is an occurrence of the schedule's current expr (detects a hand-edited ``schedule.expr`` whose stored ``next_run_at`` came from the old one). Best-effort: anything uncheckable (non-cron, no expr, no croniter, malformed) reports a - match.""" + match. + + A direct ``jobs.json`` edit can change ``schedule.expr`` while leaving the stored ``next_run_at`` + computed under the *old* expression (#93049). The stored instant is stale exactly when it is not an + occurrence of the current expression. Validation is best-effort: anything that cannot be checked + (non-cron kind, missing expr, croniter unavailable, malformed input) reports a match so the fire path + keeps its existing semantics. + """ if schedule.get("kind") != "cron": return True expr = schedule.get("expr") @@ -991,7 +1038,14 @@ def _classify_stale_cron_next_run( normalized into the profile tz); treating it as an edit would skip a due, never-fired occurrence. Discriminator: normalization moved the wall clock AND the stored instant's OWN wall clock is a legal occurrence (when offsets agree a genuine expr edit can never be misread - as a migration).""" + as a migration). + + * ``expr_edit`` — a direct ``jobs.json`` edit changed ``schedule.expr`` while leaving ``next_run_at`` + computed under the old one (#93049). Upgrading from a UTC-scheduling build to one that honours the + profile timezone leaves legacy rows like ``2026-09-02T04:00:00+00:00`` for ``0 4 * * *``; normalizing to + Europe/Brussels turns that into ``06:00+02``, which the expression excludes. Treating it as a stale edit + re-anchored to tomorrow and silently skipped a due occurrence that had never fired. + """ if _cron_next_run_matches_expr(schedule, next_run_dt): return STALE_CRON_MATCH wall_clock_shifted = raw_next_run_dt.replace(tzinfo=None) != next_run_dt.replace(tzinfo=None) @@ -1076,7 +1130,16 @@ def _write_marker(name: str, text: str, tmp_prefix: str) -> None: def record_ticker_heartbeat(success: bool = False) -> None: """Record ticker liveness (+ last-success marker when ``success``) so `cron status` can tell - "alive but failing" from "firing"; scoped per profile store.""" + "alive but failing" from "firing"; scoped per profile store. + + The ticker calls this once per loop iteration. ``success=True`` additionally bumps the *last successful + tick* marker. We track two distinct signals so `hermes cron status` can tell a thread that is merely + *alive and looping* (heartbeat fresh, success stale) from one that is actually *firing jobs* (both + fresh) — a ticker stuck failing every tick would otherwise keep the plain heartbeat fresh and falsely + report healthy (#32612, #32895). + Resolution uses ``_current_cron_store()`` so the heartbeat is correctly scoped to the active profile's + store — critical under multiplex_profiles where each profile needs its own liveness signal (#69377). + """ _write_marker("ticker_heartbeat", str(time.time()), ".hb_") if success: _write_marker("ticker_last_success", str(time.time()), ".hb_") @@ -1093,12 +1156,22 @@ def _epoch_file_age(name: str) -> Optional[float]: def get_ticker_heartbeat_age() -> Optional[float]: """Seconds since the ticker loop last iterated; None = missing/unreadable ("cannot determine", - not "dead").""" + not "dead"). + + Resolution uses ``_current_cron_store()`` so the heartbeat is correctly scoped to the active profile — + critical under multiplex_profiles where ``hermes cron status`` must report per-profile liveness + (#69377). + """ return _epoch_file_age("ticker_heartbeat") def get_ticker_success_age() -> Optional[float]: - """Seconds since the ticker last completed a tick WITHOUT raising, or None.""" + """Seconds since the ticker last completed a tick WITHOUT raising, or None. + + Resolution uses ``_current_cron_store()`` so the heartbeat is correctly scoped to the active profile — + critical under multiplex_profiles where ``hermes cron status`` must report per-profile liveness + (#69377). + """ return _epoch_file_age("ticker_last_success") @@ -1234,7 +1307,11 @@ def _record_load_stamp(stamp: Optional[Tuple[int, int, int]]) -> None: """Remember jobs.json's stamp for the enclosing _jobs_lock() section (no-op outside one) so the save path can skip the shrink-merge when disk provably hasn't changed. Capture it BEFORE reading: a mid-read sibling then mismatches (fail-safe); stamping after would certify an - unseen write.""" + unseen write. + + Stamping after the read would let that sibling's write be certified as "seen" without being in the + loaded payload, wrongly suppressing the recovery. See #80703. + """ if getattr(_jobs_lock_state, "depth", 0): _jobs_lock_state.load_stamp = stamp @@ -1530,6 +1607,10 @@ def _compute_provider_model_snapshots( from hermes_cli.runtime_provider import resolve_runtime_provider runtime_kwargs = {"requested": None} + # Delegate all rate-limit / 5xx retry to hermes's outer conversation loop, which honors + # Retry-After. The SDK default (max_retries=2) uses its own 1-2s backoff that ignores + # Retry-After and double-retries inside our loop — burning request slots against a bucket that + # won't refill for minutes. (#26293) if normalized_base_url: runtime_kwargs["explicit_base_url"] = normalized_base_url snap = resolve_runtime_provider(**runtime_kwargs) @@ -1625,6 +1706,9 @@ def create_job( source run FIRST each tick; unchanged output suppresses the agent run (mutually exclusive, incompatible with ``no_agent``). reasoning_effort: per-job pin; capability NOT validated.""" parsed_schedule = parse_schedule(schedule) + # Normalize repeat: treat 0 or negative values as None (infinite). String forms + # ('forever'/'once'/numeric) coerce via normalize_repeat_value — the shared chokepoint with update paths + # (#66824/#64520/#7142/#71987/#95706). repeat = normalize_repeat_value(repeat) if parsed_schedule["kind"] == "once" and repeat is None: repeat = 1 @@ -2010,7 +2094,13 @@ def remove_job(job_id: str) -> bool: def _set_alert_flag(job_id: str, field: str, value: bool) -> bool: """Set/clear a persisted alert-dedup marker (alert exactly once until the condition heals; survives restarts) and return the PRIOR value. Fields: ``preflight_alerted``, - ``drift_alerted``.""" + ``drift_alerted``. + + The marker records that the operator was already alerted about this job's condition, so the scheduler + alerts exactly once and stays silent on subsequent ticks until the condition heals (same alert-once + shape as the dead-pin auto-pause in #73506). Fields: ``preflight_alerted`` (blocked config, T1-26) and + ``drift_alerted`` (#44585 drift-guard skip). + """ def apply(jobs, _i, job): prior = bool(job.get(field)) if value: @@ -2087,7 +2177,14 @@ def _record_run_outcome( def _advance_after_run(job: Dict[str, Any], now: str) -> None: """Bump ``repeat.completed`` and recompute ``next_run_at``; retire the record as a terminal completion when the repeat limit is reached or a one-shot has no further run.""" + # If no next run, decide whether this is terminal completion (one-shot) or a transient failure + # (recurring schedule couldn't compute — e.g. 'croniter' missing from the runtime env). Recurring jobs + # must NEVER be silently disabled: that turns a missing runtime dep into "job completed" and the user's + # schedule quietly goes off. See issue #16265. kind = job.get("schedule", {}).get("kind") + # One-shot dispatch-limit guard (issue #38758): a finite one-shot claimed via claim_dispatch() but whose + # tick died before mark_job_run could remove it will have completed >= times while still looking due + # (last_run_at was never written, so the recovery helper re-armed it). Remove it instead of re-firing. repeat = job.get("repeat") if repeat: times = repeat.get("times") @@ -2177,7 +2274,14 @@ def _write_oneshot_diagnostic(job: Dict[str, Any], text: str, what: str) -> bool def _write_wedged_oneshot_diagnostic(job: Dict[str, Any]) -> None: """Trace for a wedged one-shot removal: dispatch was claimed but mark_job_run never ran - (interrupted mid-run); removing it silently would leave no output, error, or record.""" + (interrupted mid-run); removing it silently would leave no output, error, or record. + + A finite one-shot whose dispatch was claimed (``repeat.completed`` >= ``repeat.times``) but which never + reached ``mark_job_run`` (``last_run_at`` is null) was interrupted mid-run — scheduler restart, gateway + kill, or a non-Exception escape (#73973). The recovery guards remove such jobs so they stop appearing + due, but a silent removal leaves the user with no output, no error, and no job record. Write a small + diagnostic file into the job's output directory so the removal is observable and debuggable. + """ if job.get("last_run_at") is not None: return # a prior run was recorded — normal completion race, not a wedge repeat = job.get("repeat") or {} @@ -2227,7 +2331,13 @@ def claim_dispatch(job_id: str) -> bool: """Atomically claim a finite one-shot dispatch BEFORE execution: ``repeat.completed`` is bumped and persisted under the jobs lock so a tick dying mid-execution cannot lose the dispatch (*at-most-times* instead of *at-least-once*). True if the caller may run the job; False when - the limit is already reached. Only ``kind == "once"`` with ``repeat.times > 0`` is claimed.""" + the limit is already reached. Only ``kind == "once"`` with ``repeat.times > 0`` is claimed. + + Increments ``repeat.completed`` under the cross-process jobs lock and persists the claim immediately, so + that if the tick dies mid-execution (gateway kill, OOM, segfault, hard-timeout) the dispatch is not + lost. This converts finite one-shot jobs from *at-least-once* to *at-most-times* semantics — a job that + self-destructs fires at most ``repeat.times`` times instead of infinitely (issue #38758). + """ def apply(jobs, i, job): repeat = job.get("repeat") or {} times = repeat.get("times") @@ -2250,6 +2360,7 @@ def claim_dispatch(job_id: str) -> bool: # A prior tick claimed the dispatch then died — a genuinely wedged claim. Remove it so # it stops appearing due, leaving an operator-visible diagnostic. jobs.pop(i) + # See #73973. save_jobs(jobs, removed_ids={job_id}) _write_wedged_oneshot_diagnostic(job) logger.info( @@ -2283,7 +2394,13 @@ def _refresh_claim(jobs: List[Dict[str, Any]], claim: Any, expected_owner: str) def heartbeat_run_claim(job_id: str, *, expected_owner: str) -> bool: """Refresh a one-shot's ``run_claim`` timestamp while its run is alive, so an expired claim really means the claiming process died. Compare-and-refresh on ``expected_owner`` stops a stale - runner from extending a claim another process has since taken over.""" + runner from extending a claim another process has since taken over. + + Called periodically from the scheduler's run monitor (#62002) so a legitimately long run keeps its claim + fresh: an expired claim then really does mean "the claiming process died", and neither another process's + tick nor this process's own next tick will re-dispatch or stale-remove the job while the run is in + flight. mark_job_run() clears the claim on completion. + """ def apply(jobs, _i, job): if job.get("schedule", {}).get("kind") != "once": return False @@ -2294,7 +2411,11 @@ def heartbeat_run_claim(job_id: str, *, expected_owner: str) -> bool: def clear_run_claim(job_id: str) -> bool: """Clear a one-shot's ``run_claim`` when dispatch itself fails: such a job never reaches - mark_job_run, so the stale claim would block re-dispatch until the TTL expires.""" + mark_job_run, so the stale claim would block re-dispatch until the TTL expires. + + Calling this on every early-exit path restores the "the job stays due and will fire on the next healthy + tick" invariant that the scheduler comment promises (#86522). + """ def apply(jobs, _i, job): if job.get("schedule", {}).get("kind") != "once" or job.get("run_claim") is None: return False # recurring, or already cleared @@ -2463,7 +2584,11 @@ def get_due_jobs() -> List[Dict[str, Any]]: """Return all jobs due now. A recurring job more than one period stale (gateway down, or a run overran the interval) has its backlog collapsed — next_run_at fast-forwards so nothing burst-fires — but still fires ONCE now (via mark_job_run, consuming one ``repeat.times`` run), - avoiding the perpetual-defer loop for runs longer than interval + grace.""" + avoiding the perpetual-defer loop for runs longer than interval + grace. + + This prevents the perpetual-defer loop (#33315) where a job whose runtime exceeds ``interval + grace`` + would be skipped forever. + """ with _jobs_lock(): return _get_due_jobs_locked() @@ -2726,6 +2851,11 @@ def _oneshot_dispatch_limit_reached(job: Dict[str, Any], scan: _DueScan) -> bool if times is None or times <= 0 or completed < times: return False name = job.get("name", job.get("id", "?")) + # A live run must never have its job record deleted underneath it (#62002): a run that outlives the + # run_claim TTL (stream stall, laptop asleep mid-run) satisfies the same completed >= times + + # expired-claim condition as a dead tick, but mark_job_run() still needs the record to land last_run_at + # / last_status / last_delivery_error. If this process is still running the job, it is slow, not stale — + # keep the entry and skip. if _job_running_in_this_process(job.get("id", "")): logger.info( "Job '%s': dispatch limit reached (%d/%d) but its run is still in flight in this " @@ -2804,6 +2934,7 @@ def _evaluate_due_job(job: Dict[str, Any], scan: _DueScan, run_claim_ttl: float) # late. if not manual_run and recurring: lateness = max(0.0, (now - d.next_run_dt).total_seconds()) + # See #99879. dispatch_stamp = { "scheduled_at": next_run, "dispatched_at": now.isoformat(), @@ -2858,6 +2989,9 @@ def _get_due_jobs_locked() -> List[Dict[str, Any]]: # Per-run output files (`cron/output//.md`) are capped so a frequent job can't fill # the disk. +# Unlike the quick-snapshot store (`hermes_cli.backup`, capped at 20) it had no retention, so a +# frequently-scheduled job on a long-running deploy accumulated one file per run forever and could fill the +# disk (#52383). Keep the most recent N files per job; a non-positive value disables pruning (opt-out). _CRON_OUTPUT_DEFAULT_KEEP = 50 @@ -2897,6 +3031,7 @@ def save_job_output(job_id: str, output: str): output_file = job_output_dir / f"{_hermes_now().strftime('%Y-%m-%d_%H-%M-%S')}.md" atomic_write_text(output_file, output, tmp_prefix=".output_") _secure_file(output_file) + # Bound per-job output growth so long-running deploys don't fill the disk (#52383). _prune_job_output(job_output_dir, _cron_output_keep()) return output_file diff --git a/cron/lifecycle_guard.py b/cron/lifecycle_guard.py index d68d1b8bb6..b21a89cead 100644 --- a/cron/lifecycle_guard.py +++ b/cron/lifecycle_guard.py @@ -33,12 +33,24 @@ _GATEWAY_LIFECYCLE_PATTERN = re.compile( # `hermes` from being a path component or word tail (`/docs/hermes gateway restart-notes.md`) # while every real command position (text start, whitespace, `;`/`&`/`|`, `$(`, backtick, # U+FFFD) still matches. + # See #77173. r"(?:(? -- ` (or `launchctl bootstrap gui/ `) creates + # a NEW keepalive job wrapping an arbitrary helper, which is how a blocked direct restart/kill gets + # laundered into a persistent restart loop instead (#62891) — same foot-gun, indirect shape. + # Neutral-label submissions that dodge this text anchor are caught separately by + # `contains_launchctl_submit_command` (execution-aware, label-independent). `bootout`/`remove`/`disable` + # sit alongside `unload`: Apple deprecated load/unload in favour of bootstrap/bootout, so `bootout` is + # the modern spelling of an already-listed verb, `remove` is its legacy sibling, and `disable` is what + # makes an unload durable across boots. Omitting them left the bypassable approval layer + # (tools/approval.py, skipped on force=True) as the only cover, while this hard block — documented as + # "force=True cannot help here" — let them through (#80260). r"|(?:launchctl\s+(?:kickstart|unload|load|stop|restart|submit|bootstrap|bootout|remove|disable)\b[^\n]*\bhermes[.\-]?gateway)" # Branch C: systemctl ops on a hermes-gateway unit. r"|(?:systemctl\s+(?:-\S+\s+)*(?:restart|stop|start)\b[^\n]*\bhermes[.\-]?gateway)" @@ -51,16 +63,27 @@ _GATEWAY_LIFECYCLE_PATTERN = re.compile( # Every branch uses `[^\n]*` between verb and label so matches cannot span unrelated lines. A POSIX # backslash-newline continuation is therefore collapsed to a space before matching (as the shell # does) rather than loosening `[^\n]*`. +# Every branch above uses `[^\n]*` between its verb and the gateway identifier so the match can't span +# unrelated lines of a longer cron prompt/script, but that also means a real multi-line shell invocation +# split across continuation lines (e.g. `launchctl submit \` / ` -l ai.hermes.gateway-... \` / ` -- ...`, +# the exact reported shape in #62891) would otherwise slip past. Collapse continuations to a single space +# before matching, mirroring what the shell itself does, rather than loosening `[^\n]*` and risking false +# positives across genuinely separate lines. _SHELL_LINE_CONTINUATION = re.compile(r"\\\r?\n[ \t]*") # Python argv-list punctuation (`subprocess.run(["launchctl", "bootout", ...])`) separates exec'd # words with brackets/commas. Stripped only for the token-join re-scan, never from raw text. +# See #68289. _ARGV_LIST_PUNCTUATION = re.compile(r"[\[\],]+") # Branch A2: `hermes -p gateway restart|stop` (also `--profile ` / # `--profile=`). The selector breaks Branch A's adjacency. A sibling-profile restart is a # legitimate fleet operation, so the profile name is captured and blocked only when it equals the # profile running the guard. `start` stays excluded as in Branch A. +# Unlike Branch A this form is NOT unconditionally self-targeting: issued from inside gateway `zeus`, +# `hermes -p venus gateway restart` operates on a sibling profile's gateway and is a legitimate fleet +# operation. The pattern captures the named profile so `contains_gateway_lifecycle_command` can block only +# the self-targeting shape (named profile == the profile running the guard). See #78028. _PROFILE_FLAG_LIFECYCLE_PATTERN = re.compile( r"(?i)" r"hermes\s+" @@ -77,6 +100,8 @@ _PROFILE_FLAG_LIFECYCLE_PATTERN = re.compile( # EARLIER `;`-segment (`label=${item%%:*}; launchctl bootout "gui/$uid/$label"`) leaves only # `$label` next to the verb. These verbs act on an EXISTING job, so the hermes-gateway label anchor # stays correct, but the check is "verb anywhere AND label anywhere". +# No profile identity available: cannot prove self-targeting, so do not block — sibling restarts must stay +# allowed (#78028). _LAUNCHCTL_LIFECYCLE_VERBS_RE = re.compile( r"(?i)\blaunchctl\s+(?:kickstart|unload|load|stop|restart|bootout|kill|disable|remove)\b" ) @@ -118,6 +143,10 @@ _ReadRemoteScriptFn = Callable[[str], Optional[str]] # Wrappers that hand execution to their argument tail: the real command sits further right, so a # first-token-only guard would let `sudo bash ~/restart.sh` / `sudo launchctl submit ...` walk past. +# A guard that reads only the first token sees `sudo`/`env`/`nohup` and never inspects what they run, so +# `sudo bash ~/restart.sh` walked past the same walk that stops `bash ~/restart.sh`, and `sudo launchctl +# submit ...` past the label-independent submit block (#62891). `_PIPE_TO_INTERPRETER` above already reads +# `sudo ` this way for the pipe case; this generalises that reading to the command position. _TRANSPARENT_COMMAND_PREFIXES = frozenset({ "sudo", "doas", "env", "nohup", "setsid", "nice", "ionice", "stdbuf", "timeout", "exec", "command", "builtin", "eatmydata", @@ -216,12 +245,25 @@ def contains_gateway_lifecycle_command(text: str) -> bool: escapes resolved (closes splice bypasses like ``kick"start"`` / ``kick\\start``); order-independent launchctl pass. Single choke point for every recursion level of ``_contains_unsafe_gateway_action``. + + That second pass exists because a real shell resolves quote-splicing (``kick"start"``) and + backslash-escaping (``kick\\start``) into one literal word — ``kickstart`` — before the command ever + runs. The raw text still has the quote or backslash sitting between the verb's two halves, so the first + pass alone lets a spliced verb reach ``launchctl``/``systemctl`` untouched while still executing as the + blocked lifecycle command (#80269, reported against #80260's bootout parity fix). Tokenizing closes that + gap while keeping the same gateway-label anchoring (``_GATEWAY_LIFECYCLE_PATTERN`` still requires a + ``hermes``/``gateway`` token) — this function is the single choke point + ``_contains_unsafe_gateway_action`` calls at every recursion level, so referenced-script and ``sh -c`` + payload scanning inherit the fix automatically. """ if not text: return False # Provably inert heredoc bodies (quoted delimiter, data-sink consumer like `cat > f <<'EOF'`) # are documentation, not commands. The stripper fails open on ANY ambiguity (unquoted delimiter, # shell consumer, unterminated body), so executable heredocs are still scanned. + # Heredoc bodies that are provably inert data (quoted delimiter, data-sink consumer like `cat > file + # <<'EOF'`) are masked before scanning (#88336): a runbook line "a human can run: hermes gateway + # restart" inside such a body is documentation, not a command this shell will execute. from tools.shell_heredoc import strip_inert_heredoc_bodies text = strip_inert_heredoc_bodies(text) @@ -229,6 +271,10 @@ def contains_gateway_lifecycle_command(text: str) -> bool: if _GATEWAY_LIFECYCLE_PATTERN.search(normalized): return True # Profile-flag form: blocked only when the named profile IS the one running the guard. + # Profile-flag form (#78028): `hermes -p gateway restart|stop` bypasses Branch A because the + # selector sits between `hermes` and `gateway`. It is only the same foot-gun when the named profile IS + # the profile running the guard — sibling-profile restarts are legitimate fleet operations and stay + # allowed. profile_match = _PROFILE_FLAG_LIFECYCLE_PATTERN.search(normalized) if profile_match: named = profile_match.group(1) or profile_match.group(2) @@ -238,6 +284,9 @@ def contains_gateway_lifecycle_command(text: str) -> bool: return True # Token-aware pass. Tokens are also re-joined with Python argv-list punctuation stripped, since # `subprocess.run(["launchctl", "bootout", ...])` separates argv words with commas/brackets. + # Token-aware second pass (#80269): re-run the pattern on shell-tokenized segments where quotes/escapes + # are resolved, closing splice bypasses like `kick"start"`. Runs after the profile-flag check so both + # passes apply independently. for segment in _iter_command_segments(normalized): joined = " ".join(segment) if joined and _GATEWAY_LIFECYCLE_PATTERN.search(joined): @@ -246,6 +295,10 @@ def contains_gateway_lifecycle_command(text: str) -> bool: if stripped != joined and _GATEWAY_LIFECYCLE_PATTERN.search(stripped): return True # The label may be built in an earlier `;`-segment, so no pass above sees verb + label together. + # Order-independent launchctl pass (#77083): a shell loop can build the gateway label from a variable + # defined in an earlier `;`-separated segment (`label=${item%%:*}; launchctl bootout + # "gui/$uid/$label"`), so neither the same-span regex nor same-segment tokenization sees verb and label + # together. Check "verb anywhere AND label anywhere" instead. return _contains_launchctl_gateway_lifecycle(normalized) @@ -256,6 +309,7 @@ def contains_gateway_lifecycle_command(text: str) -> bool: # lifecycle command) and is logged at WARNING so an operator can tell it from a real block. Sizes # sit well above any legitimate wrapper graph; remote reads are a backend roundtrip each, so they # get a far tighter cap. +# See #78398. _MAX_LIFECYCLE_SCAN_BYTES = _MAX_REFERENCED_SCRIPT_BYTES # 1 MiB across the walk _MAX_LIFECYCLE_SCAN_LINES = 16384 _MAX_LIFECYCLE_SCAN_LINE_BYTES = 64 * 1024 @@ -312,7 +366,10 @@ class _LifecycleScanBudget: def _capped_read_limit(max_bytes: Optional[int]) -> int: """Per-read byte cap: never above the per-file cap, never negative. One definition so local and - remote reads cannot diverge.""" + remote reads cannot diverge. + + See #76762, #77703. + """ if max_bytes is None: return _MAX_REFERENCED_SCRIPT_BYTES return min(_MAX_REFERENCED_SCRIPT_BYTES, max(0, int(max_bytes))) @@ -462,6 +519,8 @@ def contains_launchctl_submit_command(command: str) -> bool: Label-independent by design: a NEW job's label is attacker-chosen, so a neutral name defeats any label-anchored regex. Both verbs register a persistent launchd job — never safe in the gateway. + + See #62891. """ for segment in _iter_command_segments(command): index = _executed_command_index(segment) @@ -561,7 +620,14 @@ def _expand_candidate_path(candidate: str) -> Optional[Path]: """Sanitize a tokenized path candidate at the ingestion boundary. Tokens from shlex-splitting arbitrary (possibly binary-decoded) text can carry NUL or junk that each downstream ``Path`` op rejects differently (ValueError, RuntimeError when HOME is unset under launchd, OSError); reject - once here. ``None`` = not a real path, nothing to scan.""" + once here. ``None`` = not a real path, nothing to scan. + + Every OS-facing ``Path`` operation downstream (``expanduser``, ``os.open``, ``resolve``) raises a + *different* exception for the same junk (``ValueError: embedded null byte``, ``RuntimeError: Could not + determine home directory`` when HOME is unset under launchd, OSError for over-long paths). Rejecting + here — once, before any OS call — is the whole-class fix; catching per-syscall was the whack-a-mole that + produced #76762, #77703, #77780, and #78256. + """ if not candidate or "\x00" in candidate: return None try: @@ -720,6 +786,8 @@ def _read_referenced_script( FileProvider path is never opened — not even to check hydration — because an evicted placeholder's ``open()`` can hang preflight. Lexical check: direct paths; resolved: symlinks. ``max_bytes`` lowers the per-file cap to what the calling walk can still afford. + + See #88052. """ byte_limit = _capped_read_limit(max_bytes) if _on_cloud_path(path): @@ -732,13 +800,21 @@ def _read_referenced_script( # "nothing to scan" — never crash the guard. return None, False try: + # ValueError: an embedded NUL byte in *path* itself — a binary's decoded bytes tokenized into a + # bogus script path by the recursion (#77703). metadata = os.fstat(descriptor) + # Directories are not scripts. Docker Desktop writes ``fpath=(~/.docker/completions …)`` into + # ``~/.zshrc``; the walk then treats that dir as a referenced script and used to fail-closed, + # blocking ``source ~/.zshrc`` (#86753). Devices/sockets stay fail-closed. if not stat.S_ISREG(metadata.st_mode): # Directories are not scripts (`fpath=(~/.docker/completions …)` in ~/.zshrc must not # block `source ~/.zshrc`). Devices/sockets stay fail-closed. return None, not stat.S_ISDIR(metadata.st_mode) # Sniff a small prefix first: compiled binaries are never shell scripts, so skip them # WITHOUT reading the rest or feeding decoded garbage into the recursion. + # Deliberately NOT keyed on the mere presence of a NUL byte (#77927): bash executes a text script + # straight past an embedded NUL, so NUL-bearing text must fall through to the magic-number check + + # NUL-strip below. data = os.read(descriptor, _BINARY_SNIFF_BYTES) if _has_binary_magic(data): return None, False @@ -773,7 +849,13 @@ def _sanitize_remote_script_text( """Apply the local-read contract to text from an untrusted ``read_remote_script`` callback: NUL means binary (nothing to scan, checked first); oversized fails closed. Size compares re-encoded *bytes* (matching the ``head -c`` wire bound): a >1 MiB multibyte file truncated at the byte cap - decodes to fewer chars, and a char count would scan instead of failing.""" + decodes to fewer chars, and a char count would scan instead of failing. + + The recursion boundary must not trust its callbacks: any backend (SSH, Modal, Daytona, or a future one) + can hand back raw binary bytes decoded as text, or arbitrarily large output. Enforced here rather than + inside each callback so the guarantee holds for every callback, not just the ones we hardened. See + #76762, #77703. + """ if not text or "\x00" in text: return None, False byte_limit = _capped_read_limit(max_bytes) @@ -863,6 +945,11 @@ def contains_gateway_lifecycle_command_or_referenced_script( Total by construction: never raises. Direct scans are pure string ops; the referenced-script walk (filesystem, remote backends, shlex on decoded bytes) is best-effort defense-in-depth — an unexpected failure is logged and treated as "walk found nothing". + + This is the contract #76762 established ("a guarded path must never crash the guard") enforced at the + boundary instead of per-syscall: a guard crash propagates out of ``tools/terminal_tool.py`` and breaks + every terminal command until the gateway restarts (#77780, #78256), which is strictly worse than either + verdict. """ try: return _contains_unsafe_gateway_action( @@ -894,6 +981,11 @@ def check_gateway_lifecycle(prompt: Optional[str], script: Optional[str] = None) # Attribute the refusal correctly: not a lifecycle command, but a cloud path never opened. if resolved_script is not None and _on_cloud_path(resolved_script): raise GatewayLifecycleBlocked( + # Attribute the refusal correctly: the script is not known to contain a lifecycle command — + # it lives on a cloud-synced FileProvider path (iCloud Drive / ~/Library/CloudStorage) that + # the guard refuses to open because an evicted placeholder can hang preflight indefinitely + # (#88052). Fail closed with the real reason instead of implying a dangerous lifecycle + # command. "Blocked: the cron script lives on a cloud-synced path " "(iCloud Drive / ~/Library/CloudStorage). Opening an " "evicted FileProvider placeholder can hang the guard's " @@ -911,6 +1003,8 @@ def check_gateway_lifecycle(prompt: Optional[str], script: Optional[str] = None) # false-positive generator on Python sources (pathlib "/" resolves to the filesystem root). # The regex still scans the full text; non-regular/oversized files fail closed (sentinel). # The data-exemption masker tokenizes with shlex, so it is charged against the walk budget. + # The direct command regex below still scans the full text, so a literal `hermes gateway restart` + # embedded in a .py script is still blocked. See #77131, #78398. if not _LifecycleScanBudget().charge_text(combined): unsafe = _budget_exhausted("text", 0) else: diff --git a/cron/notepad.py b/cron/notepad.py index 6f0768fd2b..a84c37b363 100644 --- a/cron/notepad.py +++ b/cron/notepad.py @@ -22,6 +22,7 @@ from hermes_time import now as _hermes_now # Optional test override. Production resolves the path at transaction time so multiplexed profile # ticks (set_hermes_home_override) cannot leak one profile's notepad rows into the import-time home # — and remove_job's clear_notepad cannot wipe the wrong profile's DB. +# Same pattern as cron/executions.py. See #86519. NOTEPAD_FILE: Optional[Path] = None MAX_VALUE_BYTES = 16 * 1024 MAX_KEY_CHARS = 128 diff --git a/cron/scheduler.py b/cron/scheduler.py index 0829788f97..781d755966 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -53,6 +53,10 @@ logger = logging.getLogger(__name__) def _close_late_session_db_result(future: "concurrent.futures.Future") -> None: """Done-callback: close a SessionDB whose constructor finished after run_job's init timeout (worker abandoned via ``shutdown(wait=False)``), else its .db/WAL/SHM handles leak to EMFILE. + + If the constructor later completes inside that abandoned worker, the Future's result — an open SessionDB + holding .db / WAL / SHM file handles — would be orphaned and never closed, leaking descriptors until + EMFILE (#72782). This callback retrieves and closes that eventual late result. """ with contextlib.suppress(Exception): db = future.result() @@ -64,7 +68,16 @@ def _close_late_session_db_result(future: "concurrent.futures.Future") -> None: def _set_cron_session_title(session_db, session_id, base_title): """Persist a non-blank, unique title for a finished cron session; returns it (None if unset). Runs BEFORE end_session()/close() so no write races the close. Duplicate title (unique-index - ValueError) -> get_next_title_in_lineage(); if unavailable, raise rather than end up untitled.""" + ValueError) -> get_next_title_in_lineage(); if unavailable, raise rather than end up untitled. + + Centralizes the title write so the cron finally block can guarantee a non-blank, unique title is + persisted before end_session()/close() tear the connection down (issues #50535, #50536, #50537): + - #50535: never leaves the session blank. base_title already carries a cron-id fallback for nameless + jobs; this also guards a failed write. Recover by appending a #N suffix via get_next_title_in_lineage() + when supported, instead of swallowing the error and ending up untitled. - #50536: this runs + synchronously in the cron finally block ahead of the session close, so no in-flight title write can race + the close. + """ if not session_db or not session_id: return None title = (base_title or "").strip() @@ -233,6 +246,7 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str: # Script runner contract ("Script timed out after {n}s: {path}") — also for agent jobs with a # context script. Must precede generic timeout matching so it never claims a provider fallback. + # See #78503, #82460. if lower.startswith("script timed out"): return ( f"⚠️ Cron '{job_name}' failed: script timed out. " @@ -241,6 +255,9 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str: # Whole-token 429: substrings in job ids/ports/hashes tripped false rate-limit alerts. if provider_reachable and ( + # Provider/API failures are the common noisy path. Keep these short. Match 429 as a whole token + # (#83188 @cation98): bare substring matching let identifiers containing those digits (job ids, + # ports, hashes) trip a false "provider rate limit" alert. re.search(r"\b429\b", text) or "rate limit" in lower or "usage limit" in lower ): reason = "rate limit" @@ -256,6 +273,17 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str: # Scheduler inactivity watchdog shape ("idle for {n}s (limit {m}s)"). Must precede the generic # provider-timeout branch: the job's own tool going quiet involves no provider/fallback chain. + # The scheduler's own inactivity watchdog (see the TimeoutError raised above at "Cron job '{job_name}' + # idle for {secs}s (limit {limit}s) — last activity: {desc}") produces a message that contains the + # substring "timed out"/"timeout" nowhere, but DOES contain "idle for ... (limit ...)" — however + # older/other call sites can still phrase an inactivity abort using "timed out" wording, so match on the + # "idle for Ns (limit" shape specifically (case-insensitive) BEFORE the generic provider- timeout branch + # below. Without this, an inactivity timeout — the job's OWN tool call/turn going quiet, no provider or + # fallback chain ever involved — gets rewritten into a misleading "provider timeout / fallback chain + # exhausted" message, sending the operator to debug the wrong system entirely (field-reported: a stuck + # `terminal` tool call tripped the 600s inactivity limit and was reported as a provider/fallback + # failure). Mirrors the same reordering fix upstream issue #59549 applied for script timeouts vs + # provider timeouts — check the more specific, deterministic signature first. if re.search(r"idle for \d+s\s*\(limit \d+s\)", lower): return ( f"⚠️ Cron '{job_name}' failed: the job itself stalled — no tool/API " @@ -294,6 +322,11 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str: # error, which carries the failing symbol. Fail-safe: skew is None on non-git/no-fingerprint # (message unchanged); no_agent jobs excluded via the same mode gate (a fresh subprocess # resolves imports against disk, so its ImportError is the script's own problem). + # Import-class failures (#95294 part 3): a long-lived gateway whose checkout was updated underneath it + # (interrupted `hermes update`, manual git pull) serves MIXED modules — old entries frozen in + # sys.modules, new files loaded by lazy imports — and every agent cron job then dies with `cannot import + # name X` / ModuleNotFoundError. The error itself reads like a code bug, so operators debug the wrong + # thing (2 days on the reporting incident, 15 missed jobs). if provider_reachable and re.search( r"cannot import name|modulenotfounderror|importerror", lower ): @@ -348,14 +381,23 @@ def _mark_incident_alerted(incident_id: Optional[str]) -> None: class CronPromptInjectionBlocked(Exception): """Raised by _build_job_prompt when the assembled prompt (incl. runtime-loaded skill content, unseen by create-time scanning) trips the injection scanner; run_job turns it into a clean - "job blocked" delivery.""" + "job blocked" delivery. + + Assembled-prompt scanning (including loaded skill content) plugs the gap from #3968: create-time + scanning only covers the user-supplied prompt field; skill content loaded at runtime was never scanned, + so a malicious skill could carry an injection payload that reached the non-interactive (auto-approve) + cron agent. + """ def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]: """Toolsets a cron-spawned agent must never receive: ``messaging``/``clarify`` always (interactive); ``cronjob`` by default (loop prevention, not a security boundary — ``cron.allow_agent_scheduling: true`` lifts only that); ``agent.disabled_toolsets`` layered on - top so per-job ``enabled_toolsets`` cannot widen past config.yaml's denylist.""" + top so per-job ``enabled_toolsets`` cannot widen past config.yaml's denylist. + + See #25752. + """ cron_cfg = (cfg or {}).get("cron") or {} if cron_cfg.get("allow_agent_scheduling"): disabled = ["messaging", "clarify"] @@ -393,7 +435,15 @@ def _merge_mcp_into_per_job_toolsets(per_job: list[str], cfg: dict) -> list[str] def _resolve_cron_enabled_toolsets(job: dict, cfg: dict) -> list[str] | None: """Toolset list for a cron job. Precedence: per-job ``enabled_toolsets`` (+ MCP merge) > ``cron`` platform config (``_get_platform_tools``, which strips _DEFAULT_OFF_TOOLSETS so fresh - installs run without ``moa``) > ``None`` on any failure (full default set).""" + installs run without ``moa``) > ``None`` on any failure (full default set). + + 1. Per-job ``enabled_toolsets`` (set via ``cronjob`` tool on create/update). Keeps the agent's + job-scoped toolset override intact — #6130. Enabled MCP servers are layered on per + ``_merge_mcp_into_per_job_toolsets`` so a native-toolset allowlist does not silently strip MCP tools. 2. + Mirrors gateway behavior (``_get_platform_tools(cfg, platform_key)``) so users can gate cron toolsets + globally without recreating every job. 3. ``None`` on any lookup failure — AIAgent loads the full + default set (legacy behavior before this change, preserved as the safety net). + """ per_job = job.get("enabled_toolsets") if per_job: return _merge_mcp_into_per_job_toolsets(list(per_job), cfg or {}) @@ -445,7 +495,13 @@ SILENT_MARKER = "[SILENT]" def _is_cron_silence_response(text: str) -> bool: """True when a cron final response should suppress delivery: ``[SILENT]`` (or SILENT / NO_REPLY / NO REPLY) as the whole response OR its own first/last line — NOT mid-sentence. - Shares the webhook-lane matcher in :mod:`gateway.response_filters` so the two cannot drift.""" + Shares the webhook-lane matcher in :mod:`gateway.response_filters` so the two cannot drift. + + Recognizes the bracketed ``[SILENT]`` sentinel (whole-response, first line, or last line) plus the + bracketless ``SILENT`` / ``NO_REPLY`` / ``NO REPLY`` variants the model emits when it drops the brackets + (#51438, #46917). Whitespace-trimmed and case-insensitive. A token buried mid-sentence is treated as + real content and delivered. + """ from gateway.response_filters import is_autonomous_silence_response return is_autonomous_silence_response(text) @@ -484,6 +540,11 @@ _INFLIGHT_MIN_ALLOWANCE_MINUTES = 30.0 # ``last_status`` so a still-running agent thread can't overwrite "interrupted" with a false "ok". # Token keying scopes the flag to one execution (recurring jobs reuse IDs); legacy paths without a # fire owner fall back to the bare job ID. +# ``run_one_job``'s own completion path checks its OWN token before writing ``last_status`` so a cron agent +# thread that keeps running in-process after its tool was killed out from under it — and produces a +# plausible-looking final response from truncated output — can never overwrite the interrupted status with a +# false "ok" (#60432). Token keying keeps an interruption scoped to that exact execution: a later run of the +# same job ID (recurring jobs reuse the ID every fire) must not inherit the stale flag. _interrupted_job_ids: set = set() @@ -512,7 +573,12 @@ class _CombinedCancelEvent: def get_running_job_ids() -> "frozenset[str]": """Thread-safe snapshot of executing job IDs (dispatch until ``_process_job`` returns). Read by - the gateway shutdown drain, otherwise blind to cron work (runs outside ``_running_agents``).""" + the gateway shutdown drain, otherwise blind to cron work (runs outside ``_running_agents``). + + _drain_active_agents``) reads this to treat in-flight cron work as active the same way it already treats + in-flight chat sessions via ``_running_agents`` — cron jobs run through their own thread pool here, + entirely outside that dict, so without this the drain is structurally blind to them (#60432). + """ with _running_lock: return frozenset(_running_job_ids | _running_fire_owners.keys()) @@ -520,7 +586,15 @@ def get_running_job_ids() -> "frozenset[str]": def try_register_running_job(job_id: str) -> bool: """Atomically add ``job_id`` to the in-flight set; False (caller must skip) if already mid-run. Single dedupe owner for ticker + manual runs (the fire claim's 300s TTL is outlived by real - jobs). Callers MUST pair success with ``release_running_job`` in a ``finally``.""" + jobs). Callers MUST pair success with ``release_running_job`` in a ``finally``. + + This is the single dedupe owner shared by the ticker's ``_submit_with_guard`` and manual runs + (``tools/cronjob_tools``): the fire claim alone cannot prevent a double-fire because its TTL (300s) is + routinely outlived by real jobs, after which a manual ``cronjob(action='run')`` would claim successfully + and run the same job concurrently (idea from #53395 by @izumi0uu). + Registration also makes the run visible to ``get_running_job_ids`` (the gateway shutdown drain, #60432) + and ``mark_running_jobs_interrupted``. + """ with _running_lock: if job_id in _running_job_ids: return False @@ -817,6 +891,7 @@ def mark_running_jobs_interrupted( job_id) # Still report it: shutdown uses the returned IDs for the interrupted-cron notice. The # in-memory flag WAS recorded above; only the persisted last_status write is skipped. + # See #82232. marked.append(job_id) continue try: @@ -858,7 +933,13 @@ def _inactivity_watchdog_loop( future_done: Callable[[], bool], ) -> bool: """Poll idle time until limit (-> True), stop, or the future completes (-> False). Uses - ``threading.Event.wait``, not asyncio, so a blocked event loop cannot disable the watchdog.""" + ``threading.Event.wait``, not asyncio, so a blocked event loop cannot disable the watchdog. + + Driven by ``threading.Event.wait`` (a kernel timeout), not asyncio, so a blocked event-loop / + ``run_job`` thread cannot disable this watchdog the way ``asyncio.sleep`` / ``wait_for`` would (family A + of #94285 — the 4118s-idle-on-a-600s-limit cron hang). Returns True when *limit_s* of inactivity was + observed. + """ while not stop.wait(poll_s): if future_done(): return False @@ -934,7 +1015,15 @@ def _write_usage_audit(record: dict) -> None: def _interpreter_shutting_down(exc: Optional[BaseException] = None) -> bool: """True when the interpreter is finalizing (tick fired during gateway teardown): concurrent. futures/asyncio refuse new work, so delivery attempts only pollute errors.log — callers skip - with a warning. ``exc`` lets an already-raised scheduling error count as a shutdown signal.""" + with a warning. ``exc`` lets an already-raised scheduling error count as a shutdown signal. + + A cron tick can fire while the gateway is tearing down — SIGTERM from ``hermes update`` / ``hermes + gateway stop`` / systemd restart, or an OOM-kill. Once finalization starts, ``concurrent.futures`` + refuses new work with ``RuntimeError: cannot schedule new futures after interpreter shutdown`` and + asyncio's default executor is gone, so *any* attempt to schedule delivery (live-adapter, + ``asyncio.run``, or a fresh pool) is doomed and only pollutes ``errors.log`` with a traceback. See + #55924, #58720. + """ from tools.interpreter_shutdown import interpreter_shutting_down return interpreter_shutting_down(exc) @@ -946,7 +1035,12 @@ _hermes_home: Path | None = None def _get_hermes_home() -> Path: """Hermes home at call time (honouring the test override). Cron is per-profile: never freeze - this at import or anchor it at the shared default root — either breaks profile isolation.""" + this at import or anchor it at the shared default root — either breaks profile isolation. + + Cron is per-profile by design (#4707): the in-process ticker runs inside a profile-scoped gateway, so + resolving the active HERMES_HOME at call time means a profile's jobs are stored AND executed under that + profile's home (its .env, config.yaml, scripts, skills). + """ return _hermes_home or get_hermes_home() @@ -957,6 +1051,9 @@ def _get_lock_paths() -> tuple[Path, Path]: return lock_dir, lock_dir / ".tick.lock" +# Errnos that mean "another ticker (or manual tick) holds the tick lock", as opposed to a real failure +# opening/locking the file. Everything else — most importantly EMFILE/ENFILE (fd exhaustion, #87644) and +# EACCES on open() — must be surfaced, never swallowed as lock contention. def _is_lock_contention_errno(err: OSError) -> bool: """True when *err* from the lock syscall means another ticker holds the lock (POSIX flock: EWOULDBLOCK/EAGAIN, EACCES on some NFS; msvcrt.locking: EACCES/EDEADLK). Everything else — @@ -978,7 +1075,10 @@ def _is_fd_exhaustion_text(text: str) -> bool: def _is_fd_exhaustion(exc: BaseException) -> bool: """True when *exc* indicates fd exhaustion: EMFILE/ENFILE errno, or the "Too many open files" - wording for wrapped exceptions (load_jobs wraps the OSError in a RuntimeError).""" + wording for wrapped exceptions (load_jobs wraps the OSError in a RuntimeError). + + See #87644. + """ if isinstance(exc, OSError) and exc.errno in (errno.EMFILE, errno.ENFILE): return True return _is_fd_exhaustion_text(str(exc)) @@ -986,7 +1086,11 @@ def _is_fd_exhaustion(exc: BaseException) -> bool: def _reclaim_fds_best_effort() -> None: """Best-effort fd reclamation: gc.collect() closes file objects stuck in reference cycles; - apply_nofile_soft_limit() raises the RLIMIT_NOFILE soft limit for headroom. Never raises.""" + apply_nofile_soft_limit() raises the RLIMIT_NOFILE soft limit for headroom. Never raises. + + The cron FD-leak family (#60859, #79742, #80792) leaks descriptors from abandoned workers/sessions. Two + safe, idempotent levers: + """ with contextlib.suppress(Exception): import gc @@ -1288,6 +1392,7 @@ def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConf logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e) # Fail fast: an empty model otherwise reaches the provider as an opaque 400. + # See #23979. if not (isinstance(model, str) and model.strip()): raise RuntimeError( f"Cron job '{job_name}' has no model configured " @@ -1337,6 +1442,13 @@ def _preflight_or_block(job: dict, job_id: str, job_name: str, cfg: dict) -> Opt record blocked_config and alert once (`preflight_alerted` bit). Must run after the wake gate so silent ticks stay silent. Opt-out: `cron.preflight: false`. Returns failure tuple or None. """ + # --------------------------------------------------------------- Pre-dispatch configuration validation + # (T1-26). A job whose configuration cannot possibly produce a successful run — missing provider API key + # (no fallback chain), unready attached skill, unconfigured delivery platform — is refused HERE, before + # AIAgent is constructed and before the resolution below can feed a doomed runtime into it, so a + # misconfigured job never burns an LLM call. run_one_job keys off the BLOCKED_CONFIG_MARKER in the + # returned error to record last_status='blocked_config' and alert exactly once (dedup persisted via the + # job's `preflight_alerted` bit — the #73506 alert-once shape). _pf_reason = None try: if _cron_preflight_enabled(cfg): @@ -1480,6 +1592,9 @@ def _check_model_drift( return _changes = "; ".join(_drift) # A finite one-shot is consumed by this attempt, so "edit the job" is a dead end for it. + # Lifecycle-aware remediation (#72056, @sashmatash): a finite one-shot is consumed by this attempted + # dispatch — telling an operator to edit a spent job is a dead end. Recurring and repeatable jobs get + # the pin command instead. _repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {} _finite_oneshot = ( isinstance(job.get("schedule"), dict) @@ -1503,6 +1618,9 @@ def _check_model_drift( job_id, _changes, _remediation) # Alert-once via drift_alerted bit (silent marker suppresses delivery); a successful run # clears it and re-arms the alert. + # Alert-once (#73506 shape): persist the drift_alerted bit so only the FIRST drifted tick delivers; + # run_one_job suppresses delivery on the silent marker. mark_job_run clears the bit when a run succeeds + # (drift healed), re-arming the alert. _drift_already_alerted = False with contextlib.suppress(Exception): from cron.jobs import mark_drift_alerted @@ -1540,6 +1658,11 @@ def _init_cron_mcp_tools(job_id: str) -> None: """Register MCP servers for the agent's tool registry. Idempotent across ticks; non-fatal so a broken MCP server never kills a working job.""" try: + # Initialize MCP servers so configured mcp_servers are available to the agent's tool registry before + # AIAgent is constructed. Without this, cron jobs never saw any MCP tools — only the gateway / CLI + # paths called discover_mcp_tools() at startup. Idempotent: subsequent ticks short-circuit on + # already-connected servers inside register_mcp_servers(). Non-fatal on failure: a broken MCP server + # shouldn't kill an otherwise-working cron job. See #4219. from tools.mcp_tool import discover_mcp_tools _mcp_tools = discover_mcp_tools() if _mcp_tools: @@ -1552,6 +1675,14 @@ def _open_cron_session_db(job: dict): """Open the SQLite session store under its own timeout (HERMES_CRON_TIMEOUT only watches run_conversation). A wedged sqlite3.connect returns None (no session store) instead of wedging the worker thread.""" + # Initialize the SQLite session store so cron job messages are persisted and discoverable via + # session_search (same pattern as gateway/run.py) — only now, after every early-return path (wake-gate, + # prompt validation, drift skip) has passed, so a gated run never opens state.db just to abandon the + # handle (#96290). Bounded with its own timeout (separate from HERMES_CRON_TIMEOUT, which only watches + # the agent's run_conversation below): SessionDB.__init__ opens/migrates state.db synchronously and has + # no timeout of its own against a wedged sqlite3.connect (e.g. a stale flock left by a crashed sibling + # process). An unbounded hang here would wedge the job's worker thread, so the init is bounded and a + # timeout proceeds without a session store instead of blocking the run forever. _session_db_timeout = _get_session_db_timeout() try: from hermes_state import get_shared_session_db @@ -1567,6 +1698,10 @@ def _open_cron_session_db(job: dict): return _session_db_future.result(timeout=_session_db_timeout) except concurrent.futures.TimeoutError: # The abandoned worker may still finish; close its late result or its SQLite FDs leak. + # The worker is abandoned (shutdown below doesn't wait for it). If SessionDB() later completes + # inside it, the future's result would be orphaned and its SQLite FDs (.db, WAL, SHM) leak until + # process exit. Register a done-callback that retrieves and closes any eventual late result + # (#72782). _session_db_future.add_done_callback(_close_late_session_db_result) raise finally: @@ -1613,6 +1748,11 @@ def _run_agent_with_watchdog( _POLL_INTERVAL = 5.0 # Heartbeat the one-shot run_claim while alive: without it a long run looks like a dead owner # and gets re-dispatched / stale-removed out from under the live run. + # Keep the one-shot run_claim fresh while the run is alive (#62002): the claim TTL is a dead-owner + # detector, but without a heartbeat a run that legitimately outlives it (stream stall, laptop asleep + # mid-run) is indistinguishable from a dead tick — another process re-dispatches it and get_due_jobs + # stale-removes the job record out from under the live run. Refreshing the claim from this monitor keeps + # "expired claim" meaning "owner died". _job_schedule = job.get("schedule") _is_oneshot = isinstance(_job_schedule, dict) and _job_schedule.get("kind") == "once" _run_claim = job.get("run_claim") @@ -1670,6 +1810,9 @@ def _run_agent_with_watchdog( try: if _cron_inactivity_limit is not None: # Separate daemon thread so a hung get_activity_summary can't stop the limit firing. + # Daemon thread: kernel ``Event.wait`` timeout, independent of the ``run_job`` thread. A blocked + # loop / hung ``get_activity_summary`` on this thread can no longer keep the 600s inactivity + # limit from firing (#94285). _watch_thread.start() if _cron_inactivity_limit is None and not _is_oneshot and cancel_event is None: result = _cron_future.result() @@ -1706,6 +1849,11 @@ def _final_response_from_result(result: dict, job_id: str, job_name: str, AIAgen """Deliverable final response from a ``run_conversation`` result. Raises RuntimeError on `failed=True`/`completed=False`: the error text may sit in `final_response` and would otherwise be delivered as the reply with the job marked ok.""" + # If the agent itself reported failure (e.g. all retries exhausted on API errors, model abort, mid-run + # interrupt), do not silently mark the job as successful. run_agent populates + # `failed=True`/`completed=False` on these paths and may put the error into `final_response`, which + # would otherwise be delivered as if it were the agent's reply and the job's `last_status` set to "ok". + # Raise so the except handler below builds the proper failure tuple. (issue #17855) turn_exit_reason = str(result.get("turn_exit_reason") or "") final_response_text = (result.get("final_response") or "").strip() max_iteration_summary = ( @@ -1774,12 +1922,21 @@ def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_s except (Exception, KeyboardInterrupt) as e: with contextlib.suppress((Exception, KeyboardInterrupt)): _agent_session_id = getattr(agent, "session_id", None) + # CLI (single-process) path: the approval contextvar is only bound during gateway/TUI turns and + # HERMES_SESSION_KEY is not in the CLI environment, so the key resolves empty here. Since #64240 + # the CLI drains completions through a positive-ownership filter keyed on the durable + # AIAgent.session_id — an empty session_key would fail closed and the CLI could never claim its + # own completions, while a restored foreign event with an empty key could leak into any + # unfiltered consumer (#64484). Stamp the parent's durable session id instead; compression + # rotations are handled on the drain side via resolve_resume_session_id lineage resolution. if _agent_session_id: _final_cron_session_id = _agent_session_id logger.debug("Job '%s': failed to resolve cron compression tip: %s", job_id, e) # Title must persist BEFORE end_session()/close(). Run-time suffix keeps it unique against the # sessions.title index; the fallbacks below guarantee a non-blank title. try: + # Title the cron session from the job (name -> id) and PERSIST it BEFORE end_session()/close() tear + # the connection down, so the close can never run over an in-flight title write (#50536). _title_base = " ".join(job_name.split())[:60].strip() or f"cron {job_id}" _cron_title = f"{_title_base} · {_hermes_now().strftime('%b %d %H:%M')}" if not _set_cron_session_title(_session_db, _final_cron_session_id, _cron_title): @@ -1787,6 +1944,7 @@ def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_s except (Exception, KeyboardInterrupt) as e: logger.debug("Job '%s': failed to set cron session title: %s", job_id, e) # Never leave the session untitled. + # Try the next free title in the lineage, then a bare id-stamped title. See #50535. for _fallback in ( getattr(_session_db, "get_next_title_in_lineage", lambda b: b)(f"cron {job_id}"), f"cron {job_id} {_final_cron_session_id[-6:]}"): @@ -1798,6 +1956,16 @@ def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_s # Book cron_complete only when the last row is a real assistant reply ([SILENT] counts). Only a # POSITIVELY recognized bad status downgrades (keep tuple in sync with # session_lifecycle_statuses); unknown values / probe failures fail OPEN. + # Verified completion booking (#93820): the run may only be recorded as cron_complete when the session's + # LAST message row is a real assistant reply — a plain answer or the [SILENT] sentinel (both are + # assistant-text rows, so both classify as 'complete'). A turn that died after a tool call, + # mid-API-wait, or without any assistant text leaves the last row as a tool result / pending call / user + # prompt and must not surface as a healthy run. session_lifecycle_statuses is the existing cost-bounded + # classifier for exactly this shape. Only a POSITIVELY recognized pathological status (see the status + # vocabulary in hermes_state's session_lifecycle_statuses docstring — keep the tuple below in sync when + # it grows) downgrades the booking: an unknown value (newer classifier shape, test doubles) keeps the + # historical reason, and so does a failed probe — the booking itself is FAIL-OPEN on probe errors, + # because classification is best-effort metadata and must not mislabel a healthy run. _end_reason = "cron_complete" try: _statuses = _session_db.session_lifecycle_statuses([_final_cron_session_id]) @@ -1868,6 +2036,11 @@ def _prepare_job_prompt( # Wake-gate: run the pre-check script BEFORE building the prompt; its result is passed into # _build_job_prompt so the script runs only once. + # NOTE: the SQLite session store used to be initialized here, BEFORE the wake-gate and prompt-validation + # early returns below. Every gated run (``wakeAgent: false``, blocked prompt) opened state.db and + # returned without reaching the finally that closes it, relying on GC to release the handle. Init now + # happens inside the main try, right before the agent is constructed — after every early-return path + # (#96290). prerun_script = None script_path = job.get("script") if script_path: @@ -1939,6 +2112,15 @@ class _CronRunScope: chat_name="", # Cron can't receive completions after its turn; async delegation output could # otherwise route to an unrelated chat via the ambient session key => inline delegation. + # We clear the HERMES_SESSION_* routing keys just below, so an async delegation's completion + # event carries session_key="" — _enrich_async_delegation_routing cannot resolve it and + # _inject_watch_notification drops it ("no routing metadata"). And by the time a child finishes, + # run_job has already shipped the job's final response via _deliver_result; there is no turn + # left to re-enter. (Worse, get_current_session_key() can fall back to the ambient os.environ + # HERMES_SESSION_KEY, which risks routing a cron subagent's output into an unrelated user chat.) + # Declaring the channel stateless routes delegate_task to its existing inline/synchronous path, + # so results return within the job's own turn. See declare_stateless_channel(). Upstream: + # #53027, #63142. async_delivery=False, cwd=self.workdir or "", ) @@ -2113,7 +2295,18 @@ def run_job( """Execute a single cron job. Returns (success, full_output_doc, final_response, error). ``defer_agent_teardown``: if a list, the live agent is appended instead of torn down; the caller MUST call ``_teardown_cron_agent(agent)`` AFTER delivery (a torn-down async client can't - deliver). ``extra_prompt``: per-fire context, never persisted.""" + deliver). ``extra_prompt``: per-fire context, never persisted. + + ``defer_agent_teardown``: when a caller passes a list, ``run_job`` skips the agent's async-resource + teardown (``agent.close()`` + ``cleanup_stale_async_clients()``) in its ``finally`` block and instead + appends the live agent to that list. The caller is then responsible for calling + ``_teardown_cron_agent(agent)`` AFTER it has delivered the result. This closes the ordering window in + #58720 where delivery ran against a torn-down async client (defense-in-depth alongside the + interpreter-shutdown guard). When ``None`` (the default) teardown happens inline as before, so every + existing caller is unchanged. + ``extra_prompt``: optional per-run context from ``cronjob(action='run', prompt=...)`` (#57331). Appended + to the stored prompt for this fire only — never persisted to the job definition. + """ job_id = job["id"] job_name = str(job.get("name") or job.get("prompt") or job_id or "cron job") @@ -2180,6 +2373,11 @@ def run_job( _finalize_cron_session(_session_db, agent, job_id, job_name, _cron_session_id) # Tear down the ephemeral agent or the gateway leaks fds per tick (EMFILE). With deferred # teardown, hand the live agent back: delivery needs a live async client. + # Release subprocesses, terminal sandboxes, browser daemons, and the main OpenAI/httpx client held + # by this ephemeral cron agent. Without this, a gateway that ticks cron every N minutes leaks fds + # per job until it hits EMFILE (#10200 / "too many open files"). When the caller opted to defer + # teardown (passed a list), hand the live agent back instead of closing it here — delivery must run + # against a live async client, and the caller tears down afterwards (#58720). if defer_agent_teardown is not None: if agent is not None: defer_agent_teardown.append(agent) @@ -2191,7 +2389,12 @@ def _teardown_cron_agent( agent, job_id: str, *, timeout_seconds: Optional[float] = None ) -> None: """Release an ephemeral cron agent's async resources within a hard bound (this runs outside the - inactivity watchdog). Shared by ``run_job``'s finally and deferred post-delivery teardown.""" + inactivity watchdog). Shared by ``run_job``'s finally and deferred post-delivery teardown. + + Split out of ``run_job``'s ``finally`` so a caller that defers teardown (to deliver first — #58720) can + invoke the identical cleanup AFTER delivery. The timeout matters because this executes after + ``run_conversation`` has returned, outside the agent inactivity watchdog. + """ def _cleanup_agent() -> None: try: if agent is not None: @@ -2522,6 +2725,9 @@ def _save_compose_deliver( # Not a substring check: bare "SILENT"/"NO_REPLY" or a report quoting "[SILENT]" must # not be swallowed; bracketed-prefix / trailing-line tolerance is kept. if d.should_deliver and d.success and _is_cron_silence_response(deliver_content): + # Cron silence suppression — see _is_cron_silence_response. Replaces the old `SILENT_MARKER in + # ...upper()` substring check, which both leaked bracketless near-markers ("SILENT" / "NO_REPLY") + # and wrongly swallowed a real report that merely quoted "[SILENT]" mid-sentence (#51438, #46917). logger.info("Job '%s': agent returned %s — skipping delivery", job["id"], SILENT_MARKER) d.should_deliver = False @@ -2561,6 +2767,11 @@ def _finish_interrupted_run(job: dict, execution_id: str, delivery_error: Option fire or auto-delete the job); an unsent notice is recorded via update_job instead.""" if delivery_error: try: + # The gateway shutdown already wrote last_status for this run, so mark_job_run is skipped below + # — but it could not know that the notice we just tried to send never left the process (the + # adapters were torn down first, #82232). Record the delivery failure on its own via update_job: + # mark_job_run also advances next_run_at and the repeat counter, and running that a second time + # for one run would skip a fire or auto-delete the job early. from cron.jobs import update_job update_job(job["id"], {"last_delivery_error": delivery_error}) except Exception as _rec_err: @@ -2662,6 +2873,8 @@ def _run_one_job_body( try: # Commit a finite one-shot's dispatch BEFORE its side effect so a tick dying mid-run cannot # re-fire it forever on restart. No-op for recurring/infinite jobs (at-most-times). + # This lives here in the shared body so BOTH the built-in ticker and the external provider (Chronos + # fire_due) get at-most-times semantics. See #38758. if not claim_dispatch(job["id"]): logger.info( "Job '%s': one-shot dispatch limit reached — skipping", @@ -2685,15 +2898,39 @@ def _run_one_job_body( # Same for terminal policy (gateway/run.py _profile_runtime_scope): else the ticker reads # process-global TERMINAL_* env a concurrent profile pinned. Resolution failure installs a # refusal scope — terminal execution raises instead of using the launch process's policy. + # Same isolation for terminal settings (third profile seam; see gateway/run.py + # _profile_runtime_scope): installs the firing profile's COMPLETE terminal policy for this fire — + # run, delivery, and bookkeeping — resetting in this function's finally alongside the secret scope. + # See #68559. + # Bind the profile's COMPLETE terminal policy for the agent build (fail-closed: malformed policy → + # refusal scope) so _make_agent's terminal probing / cwd hints resolve the routed profile, never the + # launch process (#98581 class). + # Same authoritative terminal policy the gateway binds per turn (#68559): a docker-configured + # dashboard profile must never resolve the launch process's pinned env. + # Fourth profile seam: bind the session profile's COMPLETE terminal policy for this turn + # (dashboard/TUI analogue of the gateway's per-turn scope). #98581's unified-desktop reproduction + # ran a docker-configured profile on the host because terminal_tool read the launch process's pinned + # env. from tools.terminal_scope import ( install_profile_terminal_scope) _terminal_scope_token = install_profile_terminal_scope(_get_hermes_home()) # Defer agent teardown until AFTER delivery; closing first races the live send against a # torn-down async client. run_job hands the agent back via this list. + # Defer the cron agent's async-resource teardown until AFTER delivery. run_job normally closes the + # agent (and reaps stale async clients) in its finally block; doing that before _deliver_result runs + # means the live send races a torn-down async client (#58720). Passing a holder list makes run_job + # hand the agent back instead, and we tear it down below once delivery is done. Defense-in-depth + # alongside the interpreter-shutdown guard in _deliver_result. _deferred_agents: list = [] def _teardown_deferred() -> None: + # run_job's finally still hands back the agent when it raises; tear it down here so a failed run + # never leaks its async resources (#10200), then re-raise into the outer handler. BaseException + # (not just Exception) so a KeyboardInterrupt/SystemExit mid-run still triggers teardown before + # propagating. + # Tear down the deferred agent(s) now that save + delivery have run (or raised). Must happen on + # every path so cron agents never leak their subprocesses/clients (#10200). for _deferred_agent in _deferred_agents: _teardown_cron_agent(_deferred_agent, job["id"]) @@ -2750,6 +2987,13 @@ def _run_one_job_body( # Without mark_job_run(False) a finite one-shot is wedged: claim_dispatch consumed # repeat.completed but last_run_at is never written. Record first, then re-raise # non-Exception. Owner fencing still applies. + # BaseException, not Exception (#73973): the inner run_job handler re-raises CancelledError / + # KeyboardInterrupt / SystemExit after agent teardown, and none of those are Exception subclasses. + # If they escape without mark_job_run(False), a finite one-shot is left wedged — claim_dispatch() + # already consumed repeat.completed, but last_run_at is never written, so the job sits in state + # "scheduled" until the run-claim TTL expires and the dispatch-limit guard removes it with no output + # and no error. Owner fencing still applies: a stale worker must not record over a replacement claim + # owner. _err_text = str(e) or type(e).__name__ logger.error( "Error processing job %s: %s", @@ -3161,6 +3405,8 @@ def create_job_with_scheduler_registration(**kwargs) -> dict: # Dead-owner reap is throttled (opens the executions ledger). Tests may reset # _last_dead_owner_reap_at to None to force a reap next tick. +# Dead-owner claim reclaim throttle (#86721): recover_interrupted_executions opens the executions ledger, so +# the per-tick reap is rate-limited rather than run on every idle 60s cycle. _DEAD_OWNER_REAP_INTERVAL_SECONDS = 300.0 _last_dead_owner_reap_at: Optional[float] = None @@ -3241,6 +3487,11 @@ def _acquire_tick_lock(lock_file): tick.""" lock_fd = None try: + # Cross-platform file locking: fcntl on Unix, msvcrt on Windows. Only genuine lock contention + # (another ticker holds the lock) skips the tick silently. A real OSError — most importantly + # EMFILE/ENFILE from fd exhaustion — must NOT be swallowed as "another instance holds the lock": + # that previously made the scheduler appear healthy (tick returned 0, heartbeat recorded success) + # while no job ever ran again (#87644). lock_fd = open(lock_file, "w", encoding="utf-8") if fcntl: fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB) @@ -3278,6 +3529,12 @@ def _release_tick_lock(lock_fd) -> None: def _maybe_reap_dead_owners() -> None: """Dead-owner reclaim: a run that died mid-flight would leave its row 'claimed' forever. Only rows whose owner process is proved gone are touched (_owner_is_live). Throttled.""" + # Dead-owner claim reclaim (#86721): execution rows carry their owner pid + process start time, but + # recovery previously ran only at scheduler STARTUP. A one-shot `hermes cron run` that claimed a job and + # died mid-run (its runner thread lived in the exiting CLI process) left the row 'claimed' forever while + # the long-lived gateway ticker kept running — blocking every future run of that job. Reap provably-dead + # owners periodically so stale claims auto-clear without a gateway restart. Throttled so idle 60s ticks + # don't pay a ledger connection every cycle (#33612). global _last_dead_owner_reap_at _reap_now = time.monotonic() if ( @@ -3367,7 +3624,16 @@ def _submit_with_guard(job: dict, pool: concurrent.futures.ThreadPoolExecutor, p def _clear_run_claim_best_effort() -> None: """Best-effort claim cleanup on dispatch-failure paths. Only one-shots carry a run_claim; clear_run_claim takes _jobs_lock + full load/save and can raise on degraded paths - (shutdown, EMFILE) — a claim expiring at TTL beats crashing the tick.""" + (shutdown, EMFILE) — a claim expiring at TTL beats crashing the tick. + + Only one-shot jobs carry a ``run_claim`` (stamped by get_due_jobs, #59229), so recurring jobs skip + the call entirely — clear_run_claim acquires _jobs_lock (blocking cross-process flock) and does a + full load_jobs read, and the dispatch-failure paths fire exactly when the process can least afford N + pointless lock/read round-trips (interpreter shutdown, EMFILE). clear_run_claim itself does + load_jobs/save_jobs file I/O; on those degraded paths it can raise, and these early-exits exist + precisely to skip cleanly — a stale claim expiring at the TTL is a better outcome than crashing the + tick (#86522). + """ _schedule = job.get("schedule") if not (isinstance(_schedule, dict) and _schedule.get("kind") == "once"): return @@ -3383,6 +3649,13 @@ def _submit_with_guard(job: dict, pool: concurrent.futures.ThreadPoolExecutor, p logger.warning("Job '%s' not dispatched — interpreter is shutting down", job_label) # During interpreter shutdown pool.submit raises; skip — the job fires on the next tick. + # If the interpreter is finalizing (gateway SIGTERM / restart / OOM), scheduling any new delivery is + # futile — asyncio.run and a fresh ThreadPoolExecutor both raise "cannot schedule new futures after + # interpreter shutdown". Skip gracefully with a warning rather than emitting an ERROR traceback on every + # restart-race (#58720, #55924). + # A tick can race gateway teardown: once the interpreter is finalizing, ``pool.submit`` raises "cannot + # schedule new futures after interpreter shutdown" and crashes the tick. Skip cleanly — the job stays + # due and will fire on the next healthy tick (#58720, #55924). if _interpreter_shutting_down(): _not_dispatched_shutdown() _clear_run_claim_best_effort() @@ -3494,6 +3767,11 @@ def tick( if not due_jobs: # Idle tick: skip config load + pool setup, but still reap crashed jobs' MCP orphans. if verbose: + # Idle tick: skip config load + pool partitioning entirely (#33612 — the gateway ticker + # calls tick(verbose=False) every 60s, so idle ticks previously fell through to + # load_config()). Still run the post-tick MCP orphan sweep: main intentionally sweeps on + # idle ticks so orphaned stdio children from crashed jobs are reaped even when nothing is + # due. logger.info("%s - No jobs due", _hermes_now().strftime('%H:%M:%S')) _sweep_mcp_orphans() return 0 diff --git a/cron/scheduler_delivery.py b/cron/scheduler_delivery.py index 3b5da860a2..07db410ca7 100644 --- a/cron/scheduler_delivery.py +++ b/cron/scheduler_delivery.py @@ -75,7 +75,13 @@ def _resolve_cron_surface_mode(pconfig, logical_platform_name: str) -> str: def _resolve_origin(job: dict) -> Optional[dict]: """Extract origin info from a job. Non-dict origins (provenance strings, hand-edited - jobs.json) are treated as missing — otherwise every fire crashed on ``origin.get``.""" + jobs.json) are treated as missing — otherwise every fire crashed on ``origin.get``. + + Without this guard, a job tagged with e.g. ``"combined-digest-replaces-x-and-y"`` crashed every fire + attempt with ``'str' object has no attribute 'get'`` — ``mark_job_run`` recorded the failure, but the + next tick re-loaded the same poisoned origin and crashed identically until the field was patched + manually (#18722). + """ origin = job.get("origin") if isinstance(origin, dict) and origin.get("platform") and origin.get("chat_id"): return origin @@ -175,6 +181,11 @@ def _maybe_mirror_cron_delivery( from gateway.mirror import mirror_to_session # USER role + labelled prefix, NOT assistant: an assistant-role mirror lands # assistant→assistant and breaks strict alternation; consecutive user turns merge safely. + # The brief is not the agent speaking; an assistant-role mirror lands as assistant→assistant after + # the agent's last turn and breaks strict alternation (issue #2221, the exact failure #2313 + # removed). A user-role turn collapses safely via repair_message_sequence's consecutive-user merge + # on every provider, and the prefix preserves the "this came from cron" context that the dropped + # SQLite mirror metadata would otherwise lose on replay. ok = mirror_to_session( platform_name, str(chat_id), _cron_mirror_message(job, text), source_label="cron", thread_id=thread_id, user_id=user_id, role="user") @@ -401,7 +412,15 @@ def _home_env_lookup(env_var: str, suffix: str = "", *, strip: bool = False) -> def _env_home_target_chat_id(platform_name: str) -> str: - """Home chat id from the env mirror only (no config).""" + """Home chat id from the env mirror only (no config). + + Reads through ``get_secret`` (not raw ``os.getenv``) so a profile-scoped secret scope wins in a + multiplex gateway. ``DISCORD_HOME_CHANNEL`` lives in each profile's ``.env``; in a multiplex process the + winning cron tick runs with the job-owning profile's scope installed (run_one_job sets it), so reading + via ``get_secret`` resolves the OWNING profile's chat id rather than the host process's ``os.environ`` + (#83182, chat-id leg — the token leg was fixed earlier; chat id / thread id resolve through the same + leak). + """ env_var = _resolve_home_env_var(platform_name) return _home_env_lookup(env_var) if env_var else "" @@ -419,7 +438,13 @@ def _get_home_target_chat_id(platform_name: str) -> str: def _get_home_target_thread_id(platform_name: str) -> Optional[str]: """Optional thread/topic id for a platform home target. Telegram: ``TELEGRAM_CRON_THREAD_ID`` overrides ``TELEGRAM_HOME_CHANNEL_THREAD_ID`` — in topic mode a root-DM delivery lands in the - system-only lobby where the user cannot reply.""" + system-only lobby where the user cannot reply. + + When topic mode is enabled, deliveries that land in the root DM (thread_id unset) end up in the + system-only lobby where the user cannot reply — the gateway returns the lobby reminder and drops + ``reply_to_message_id`` (#24409). Pointing cron at a dedicated topic via this env var lets replies work + as expected without changing the lobby invariant. + """ if platform_name.lower() == "telegram": cron_thread = _home_env_lookup("TELEGRAM_CRON_THREAD_ID", strip=True) if cron_thread: @@ -885,7 +910,15 @@ def _confirm_adapter_delivery( ``success`` attr/key is NOT success (would log "delivered" while nothing was sent). ``delivered is False`` REJECTS even with truthy ``success`` (the silence-narration filter returns ``{"success": True, "delivered": False}``). No ``message_id``/``raw_response`` is still - accepted (some adapters return a bare success) but logged at WARNING as UNVERIFIED.""" + accepted (some adapters return a bare success) but logged at WARNING as UNVERIFIED. + + A live adapter that returns ``None`` (e.g. a swallowed exception, a busy platform, or a code path that + returns early without producing a ``SendResult``) must NOT be treated as success — doing so causes the + scheduler to log ``"delivered to via live adapter"`` while the gateway never actually sees the + message (#47056). + * No ``message_id`` and no ``raw_response`` means we have no positive evidence of a send. Telegram + ``SendResult`` objects carry ``message_id``; the dict-filter shape does not. See #77763. + """ if send_result is None: return False if isinstance(send_result, dict): @@ -914,7 +947,13 @@ def _is_channel_dm_topic(runtime_adapter: Any, chat_id: Any, loop: Any, job_id: """Is an ambiguous ``telegram::`` target a channel Direct-Messages topic (``direct_messages_topic_id``) rather than a private-chat forum topic (``message_thread_id``)? Shape cannot decide; signal is ``get_chat_info`` type == ``channel``. - Fails SAFE to False (thread routing) without a probe or on any probe error/timeout.""" + Fails SAFE to False (thread routing) without a probe or on any probe error/timeout. + + Callers gate this on the ambiguous shape first (``telegram::``) — + that shape is identical for both cases, so shape alone cannot decide (this was the #52060 regression). + Probe the live adapter's ``get_chat_info`` once and only return True when the chat is a channel. + See #22773. + """ # Resolve on the CLASS, not the instance: a MagicMock instance auto-creates a truthy # ``get_chat_info``, so an instance-level probe would misclassify test doubles. get_chat_info = getattr(type(runtime_adapter), "get_chat_info", None) @@ -1027,6 +1066,7 @@ def _resolve_target_transport( if isinstance(adapters, _sched.SharedRouteAdapters): # Credentialless satellite: the primary adapter serves THIS target only when an exact # primary route maps it to this profile; a miss fails closed below. + # See #101113. shared = adapters.get(platform, target) target_adapters = {platform: shared} if shared is not None else {} transport = resolve_delivery_transport(platform, config, target_adapters) @@ -1090,6 +1130,7 @@ def _live_route_metadata(t: _TargetDelivery) -> tuple[Optional[str], dict, dict] if is_ambiguous_telegram_topic and _is_channel_dm_topic( t.runtime_adapter, t.chat_id, t.loop, job["id"]): # Channel DM topic: direct_messages_topic_id, no bare thread_id; media mirrors text. + # See #22773. route_thread_id = None route_metadata = { "direct_messages_topic_id": str(thread_id), "job_id": job["id"], @@ -1098,6 +1139,10 @@ def _live_route_metadata(t: _TargetDelivery) -> tuple[Optional[str], dict, dict] media_metadata = {"direct_messages_topic_id": str(thread_id), "notify": t.notify_delivery} else: # Forum-style topic or non-topic target: message_thread_id. + # Put thread_id in *route_metadata* (not just the DeliveryTarget) deliberately — the + # DeliveryRouter's private-chat topic detection (gateway/delivery.py) demands a reply anchor when + # thread_id is absent from metadata; cron deliveries have no inbound reply anchor, so the metadata + # key bypasses that check and lets the adapter route via a plain message_thread_id. See #52060. route_thread_id = str(thread_id) if thread_id is not None else None route_metadata = {"job_id": job["id"], "notify": t.notify_delivery} if route_thread_id: @@ -1289,6 +1334,12 @@ def _deliver_via_live_adapter( # Media rides the same DM-topic-aware routing as text. Skipped after a confirmation # timeout (loop contended, text already assumed delivered) — record the drop instead. + # Send extracted media files as native attachments via the live adapter, using the same + # DM-topic-aware routing as the text send (#22773 — media previously used a bare thread_id and + # landed in the General lane for private DM topics). Skip on an in-flight confirmation timeout: the + # gateway loop is contended, so each media send would also block its 30s budget, and the text + # payload is already assumed delivered (#38922). Record the skipped attachments so the drop is + # visible rather than silently lost. if adapter_ok and not timed_out and media_files: _live_send_media(t, media_metadata, media_files, delivery_errors) elif timed_out and media_files: @@ -1301,6 +1352,9 @@ def _deliver_via_live_adapter( if adapter_ok: # Log WHERE it went: a ghost delivery in the wrong lane is otherwise indistinguishable. logger.info( + # Log WHERE it went, not just that it went: a ghost delivery that landed in the wrong lane + # (General topic instead of the routed thread) is indistinguishable from a real one without + # the routing identity (#77763). "Job '%s': delivered to %s:%s via live adapter thread=%s message_id=%s", job["id"], t.platform_name, t.chat_id, route_thread_id if route_thread_id is not None else "-", @@ -1520,6 +1574,11 @@ def _unresolved_delivery_outcome(job: dict, for_failure: bool) -> Optional[str]: return None if deliver_value == "origin": logger.info( + # deliver=origin with no resolvable origin and no configured home channels: treat as local + # rather than reporting an error. CLI-created jobs never capture a {platform, chat_id} origin, + # so failing here would make every CLI `deliver=origin` (or auto-detect) job emit a spurious "no + # delivery target resolved" error on every run (#43014). The output is still persisted in + # last_output for `cron list`/resume. "Job '%s': deliver=origin but no origin or home channels — " "skipping delivery (output saved in last_output)", job.get("name", job.get("id", "?"))) diff --git a/cron/scheduler_preflight.py b/cron/scheduler_preflight.py index 36abd32451..ad48c351f5 100644 --- a/cron/scheduler_preflight.py +++ b/cron/scheduler_preflight.py @@ -21,6 +21,9 @@ logger = logging.getLogger("cron.scheduler") BLOCKED_CONFIG_MARKER = "[blocked_config]" BLOCKED_CONFIG_SILENT_MARKER = "[blocked_config:silent]" # Drift-guard skip: same contract (drift_alerted bit on the job record). +# Same alert-once contract as blocked_config: run_one_job keys off it to record last_status and the +# ``:silent`` variant means "already alerted on a previous tick — do not deliver again" (the drift_alerted +# bit on the job record, #73506 shape). DRIFT_SKIP_MARKER = "[drift_skip]" DRIFT_SKIP_SILENT_MARKER = "[drift_skip:silent]" @@ -123,7 +126,13 @@ def _primary_profile_routes_for_current_home() -> list: holding its own token is a ``duplicate_credential`` fatal). Reads the primary config.yaml directly (top-level or nested ``gateway.``) instead of ``load_gateway_config()`` so no primary platform config leaks into this process. Shared by preflight rescue and delivery-time - resolution so they cannot drift.""" + resolution so they cannot drift. + + Under ``gateway.multiplex_profiles`` a satellite profile's cron jobs are ticked by the primary gateway's + in-process ticker (#69377) and delivered through the primary gateway's live adapters — the satellite + home never holds the platform credentials itself (giving it a token of its own is a + ``duplicate_credential`` fatal). + """ try: from hermes_constants import get_default_hermes_root, get_hermes_home primary_home = get_default_hermes_root() @@ -157,7 +166,10 @@ def _primary_profile_routes_for_current_home() -> list: def _delivery_platform_routed_from_primary_gateway(platform_name: str) -> bool: - """True when the primary gateway routes this platform to the profile being served.""" + """True when the primary gateway routes this platform to the profile being served. + + scheduler is currently serving (preflight rescue, #97476). + """ platform_key = platform_name.lower() return any( str(route.platform).lower() == platform_key @@ -169,7 +181,10 @@ class SharedRouteAdapters: """Read-only adapter map for a credentialless satellite profile. ``get(platform, target)`` resolves the PRIMARY adapter iff the inbound route matcher (``ProfileRoute.matches``) accepts the target; anything else (unmatched target, disabled route, other profile, or target-less - ``get(platform)``) is a miss — fail closed, never the default bot.""" + ``get(platform)``) is a miss — fail closed, never the default bot. + + See #101113. + """ def __init__(self, primary_adapters, routes) -> None: self._primary = dict(primary_adapters or {}) @@ -245,6 +260,9 @@ def _preflight_check_delivery(job: dict) -> Optional[str]: # Multiplex: a satellite served by the primary's adapters reads unconnected — no block. if ( platform_name.lower() not in connected + # Multiplex escape hatch: a satellite profile whose deliveries are routed by the primary + # gateway's profile_routes is served by the primary's adapters, so its own unconnected reading + # is a false block (#97476). and not _delivery_platform_routed_from_primary_gateway(platform_name) ): return ( @@ -295,7 +313,11 @@ def _preflight_check_skills(job: dict) -> Optional[str]: def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]: """Pre-dispatch validation: return a reason (missing key, unconfigured delivery, unready skill) so the caller refuses BEFORE building agent machinery or burning an LLM call. Every check fails - open — preflight blocks only on an affirmative misconfiguration verdict.""" + open — preflight blocks only on an affirmative misconfiguration verdict. + + Same fail-before-spend spirit as the #44585 drift guard and the fail-loud-on-hidden-tools direction in + #27948; alert dedup follows the alert-once pattern from the dead-pin auto-pause (#73506). + """ for name, check in ( ("provider_key", lambda: _preflight_check_provider_key(job, cfg)), ("skills", lambda: _preflight_check_skills(job)), diff --git a/cron/scheduler_prompt.py b/cron/scheduler_prompt.py index d0045c38bd..0a04263351 100644 --- a/cron/scheduler_prompt.py +++ b/cron/scheduler_prompt.py @@ -18,7 +18,11 @@ logger = logging.getLogger("cron.scheduler") def _parse_wake_gate(script_output: str) -> bool: """Wake gate: False only if the last non-empty stdout line is JSON ``{"wakeAgent": false}`` - (agent skipped entirely — no LLM run, no delivery); anything else wakes normally.""" + (agent skipped entirely — no LLM run, no delivery); anything else wakes normally. + + Any other output (non-JSON, missing flag, gate absent, or ``wakeAgent: true``) means wake the agent + normally. See #1232. + """ stripped_lines = [line for line in (script_output or "").splitlines() if line.strip()] if not stripped_lines: return True @@ -189,7 +193,12 @@ def _build_job_prompt( """Build the effective prompt for a cron job, optionally loading skills first. ``prerun_script``: cached ``(success, stdout)`` from a script the caller already ran (wake-gate check) — skips re-execution. ``extra_prompt``: per-run ``## Run Context`` for this fire only, - never persisted to the job.""" + never persisted to the job. + + When provided, the script is not re-executed and the cached result is used for prompt injection. When + omitted, the script (if any) runs inline as before. extra_prompt: Optional per-run context (from + ``cronjob(action='run')``, 57331 — salvaged from #57342 by @liuhao1024). + """ user_prompt = str(job.get("prompt") or "") if extra_prompt: user_prompt = f"{user_prompt}\n\n## Run Context\n{extra_prompt}" @@ -238,6 +247,9 @@ def _build_job_prompt( parts.append("") # Skill blocks are stable per job config; the appended instruction is volatile per-run. # Declare that boundary for the Anthropic cache planner. + # The skill blocks (and any skipped-skill notice) above are stable per job config; the appended + # instruction carries the volatile per-run data (cron hint + prompt + script output + run context). + # See #81867. stable_prefix = append_user_instruction(parts, prompt) assembled = _scan_assembled_cron_prompt("\n".join(parts), job, has_skills=True) if ( @@ -261,6 +273,9 @@ def _scan_assembled_cron_prompt( STRICT ``_scan_cron_prompt``; skills or injected data → LOOSER ``_scan_cron_skill_assembled`` (command-shape patterns dropped, invisible unicode sanitized not blocked, so a false positive cannot permanently kill a job); injected data without skills also scans ``user_prompt`` STRICT. + + Since cron runs non-interactively (auto-approves tool calls), a malicious skill carrying an injection + payload bypassed every gate. See #3968. """ from tools.cronjob_tools import _scan_cron_prompt, _scan_cron_skill_assembled if has_skills or has_injected_data: diff --git a/cron/scheduler_provider.py b/cron/scheduler_provider.py index 805cc22a87..8f129c736c 100644 --- a/cron/scheduler_provider.py +++ b/cron/scheduler_provider.py @@ -20,8 +20,15 @@ _EMFILE_BACKOFF_MAX_SECONDS = 15 * 60 DEFAULT_MISFIRE_GRACE_MINUTES = 10 +# Cap for the exponential tick backoff applied while consecutive ticks fail with fd exhaustion +# (EMFILE/ENFILE, #87644). Base is the tick interval (60s by default); each consecutive EMFILE failure +# doubles the wait, capped here so a still-alive-but-exhausted gateway never sleeps longer than this between +# recovery attempts. def _backoff_wait_seconds(interval: float, consecutive_failures: int) -> float: - """Plain ``interval`` while healthy; doubles per fd-exhaustion failure, capped.""" + """Plain ``interval`` while healthy; doubles per fd-exhaustion failure, capped. + + Exponential tick backoff shared by both ticker loops (#87644). + """ if consecutive_failures <= 0: return interval return min(interval * (2 **(consecutive_failures - 1)), _EMFILE_BACKOFF_MAX_SECONDS) @@ -29,7 +36,12 @@ def _backoff_wait_seconds(interval: float, consecutive_failures: int) -> float: def _note_tick_failure(exc: BaseException, consecutive_failures: int) -> int: """On fd exhaustion: reclaim fds and bump the backoff counter; any other failure resets it — - backoff is reserved for the EMFILE storm.""" + backoff is reserved for the EMFILE storm. + + Shared by both ticker loops (#87644): on fd exhaustion, attempt reclamation (gc.collect + raise the soft + nofile limit) so the NEXT tick can succeed, and bump the counter so ``_backoff_wait_seconds`` backs off + exponentially while the process has no chance of making progress. + """ from cron.scheduler import _is_fd_exhaustion, _reclaim_fds_best_effort if _is_fd_exhaustion(exc): @@ -46,7 +58,14 @@ def _profile_entry(entry) -> tuple: def _existing_profile_homes(profile_homes: list) -> list: """Drop homes no longer on disk: ticking/heartbeating a deleted home would recreate its - ``cron/`` workspace and silently resurrect the profile.""" + ``cron/`` workspace and silently resurrect the profile. + + Ticking or heartbeating a deleted home recreates its ``cron/`` workspace (``record_ticker_heartbeat`` -> + ``ensure_dirs`` -> ``mkdir(parents=True)``) on every 60s cycle, so the "deleted" profile silently comes + back on disk and in ``hermes profile list`` (#47368). Filtering on directory existence leaves a deleted + profile's home untouched, which is the correct invariant: a home that does not exist cannot hold jobs to + fire. + """ return [entry for entry in profile_homes if Path(_profile_entry(entry)[1]).is_dir()] @@ -56,6 +75,10 @@ def _profile_cron_scope(home): from cron.jobs import use_cron_store from hermes_constants import set_hermes_home_override, reset_hermes_home_override + # Record per-profile heartbeat after each tick cycle. Distinguish a COMPLETED cycle (``_tick_error`` + # unset) — where each profile's beat reflects its own outcome, so a yielding profile does not darken + # healthy siblings — from an aborted one (exception), where no profile completed and all beats are + # unsuccessful (#32612). home_token = set_hermes_home_override(str(home)) try: with use_cron_store(home): @@ -247,6 +270,9 @@ def fire_overdue_jobs( continue job_id = str(job.get("id") or "") # One-shots past ONESHOT_GRACE_SECONDS "will never fire"; don't resurrect them hours late. + # One-shot jobs share the module-wide policy: more than ONESHOT_GRACE_SECONDS past their run time + # means "will never fire" (create/update/resume/recovery and, since #89571, the due-scan all enforce + # it). The misfire backstop must not resurrect them hours late after downtime — that's #93526. schedule = job.get("schedule") or {} if str(schedule.get("kind") or "") == "once" and overdue_seconds > ONESHOT_GRACE_SECONDS: logger.warning( @@ -349,6 +375,11 @@ class InProcessCronScheduler(CronScheduler): logger.info("In-process cron scheduler started (interval=%ds)", interval) # Multiplex: tick EACH profile's store every cycle, heartbeats/recovery scoped per profile. + # ── Multiplex profiles ──────────────────────────────────────────── When profile_homes is set + # (multiplex_profiles on), tick EACH profile's cron store on every tick cycle so secondary-profile + # jobs actually fire instead of languishing in a store no ticker owns (#69377). Without this, only + # the process-global HERMES_HOME (the default profile) is ticked. Heartbeats and recovery are also + # scoped per profile so `hermes cron status` reflects liveness for every profile independently. if profile_homes: self._start_multiplex( stop_event, profile_homes=profile_homes, adapters=adapters, loop=loop, @@ -380,6 +411,11 @@ class InProcessCronScheduler(CronScheduler): except BaseException as e: # BaseException, not Exception: a SystemExit must not silently kill the ticker; # KeyboardInterrupt is caught on purpose — shutdown is driven by stop_event. + # Catch BaseException (not just Exception) so a SystemExit from a misbehaving provider SDK / + # agent retry path does not kill the ticker thread silently (#32612). KeyboardInterrupt is + # intentionally caught here too — gateway shutdown is driven by stop_event (set by the main + # thread's signal handler), not by an exception in this daemon thread, so swallowing it and + # re-checking stop_event keeps shutdown clean. if isinstance(e, CronTickYielded): # Expected while a fresh gateway owns the lock; still recorded for status. logger.info("Cron tick yielded: %s", e) @@ -389,6 +425,10 @@ class InProcessCronScheduler(CronScheduler): record_ticker_error(f"{type(e).__name__}: {e}") consecutive_failures = _note_tick_failure(e, consecutive_failures) # Liveness every iteration; success marker only on a clean tick. + # EMFILE: reclaim fds + back off exponentially so the exhausted process stops hammering the + # store while it has no chance of making progress (#87644). + # Record liveness every iteration; bump the success marker only on a clean tick, so status can + # tell "alive but failing every tick" from "actually firing jobs" (#32612, #32895). record_ticker_heartbeat(success=ok) if ok: clear_ticker_error() @@ -427,6 +467,8 @@ class InProcessCronScheduler(CronScheduler): return tick_adapters # Recovery + heartbeat per profile; one broken store must not abort startup for the others. + # A profile may have been deleted since this snapshot was taken; never recreate a deleted home's + # cron workspace via the heartbeat below (#47368). for entry in _existing_profile_homes(profile_homes): _, home = _profile_entry(entry) try: @@ -449,6 +491,7 @@ class InProcessCronScheduler(CronScheduler): _tick_error = None _profile_errors: dict[str, str] = {} # Worst failure this cycle (fd exhaustion wins); backoff applied once per cycle. + # See #87644. _cycle_exc: BaseException | None = None cycle_homes = [_profile_entry(e) for e in _existing_profile_homes(profile_homes)] if profile_gate is not None: @@ -484,6 +527,7 @@ class InProcessCronScheduler(CronScheduler): except BaseException as e: logger.error("Cron tick error: %s", e, exc_info=True) _tick_error = f"{type(e).__name__}: {e}" + # EMFILE: reclaim fds + exponential backoff (#87644). consecutive_failures = _note_tick_failure(e, consecutive_failures) # Completed cycle: each profile's own outcome; aborted cycle: all beats unsuccessful. for _, home in cycle_homes: diff --git a/cron/scheduler_script.py b/cron/scheduler_script.py index 14e7252254..a803c3c258 100644 --- a/cron/scheduler_script.py +++ b/cron/scheduler_script.py @@ -260,6 +260,12 @@ def _resolve_script_path(script_path: str) -> tuple[Optional[Path], Optional[str # Reject NUL eagerly: on Windows Path ops raise ValueError *after* expanduser so the try below # would not catch it. str() first so the guard itself cannot raise on a non-str script_path. + # Same ingestion contract as cron.lifecycle_guard._expand_candidate_path: a NUL-bearing value can never + # name a real script, and on Windows the Path operations raise ValueError *after* expanduser (expanduser + # never expands "~user" there, so the try below never fires) — reject eagerly so both platforms fail + # cleanly instead of crashing the scheduler. str() first so the guard itself can never raise TypeError + # on a non-str script_path (e.g. a Path passed by a future caller) — the guard must be crash-proof even + # though every current call site passes a plain str (#86832 review). if "\x00" in str(script_path): return None, f"Blocked: script path contains a NUL byte: {script_path!r}" try: @@ -311,7 +317,13 @@ def _run_job_script( """Execute a cron job's script and return ``(success, output)``; on failure *output* is the error message for the LLM to report. Env goes through ``build_subprocess_env`` (SECURITY.md §2.3). ``workdir`` sets the subprocess cwd only; the Python process cwd is NEVER mutated (an - ``os.chdir()`` would leak into concurrent gateway sessions).""" + ``os.chdir()`` would leak into concurrent gateway sessions). + + Args: script_path: Path to the script. Relative paths are resolved against HERMES_HOME/scripts/. + Absolute and ~-prefixed paths are also validated to ensure they stay within the scripts dir. workdir: + Optional absolute path to use as the script's cwd. When set, the subprocess runs in this directory + instead of the scripts-dir parent. See #69396. + """ path, err = _resolve_script_path(script_path) if path is None: return False, err @@ -327,11 +339,18 @@ def _run_job_script( popen_kwargs = { "creationflags": _sched.windows_hide_flags() | getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0), + # Lossy UTF-8 decode — locale-mismatched bytes from the STT command must not raise in the + # reader threads on non-UTF-8 Windows (#45099). + # Lossy UTF-8 decode — locale-mismatched bytes from the TTS command must not raise in the + # reader threads on non-UTF-8 Windows (#45099). "encoding": "utf-8", "errors": "replace"} env = build_subprocess_env() env.update(env_overlay) # Subprocess cwd only (default: scripts-dir parent). NEVER os.chdir() the process. + # Use the job's workdir as the subprocess cwd when configured, otherwise default to the scripts-dir + # parent (back-compat). NEVER mutate the Python process cwd — that would leak into concurrent + # gateway sessions (#69396). proc = subprocess.Popen( argv, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, cwd=workdir or str(path.parent), env=env, **popen_kwargs) @@ -347,6 +366,12 @@ def _run_job_script( if remaining <= 0: _sched._terminate_cron_script_tree(proc) _drain_script_pipes(proc) + # Phase 4a (#85125): a script timeout must leave ZERO living descendants. killpg only + # reaches the script's own process group — a grandchild that called setsid (backgrounded + # shell jobs, watchdogs) escapes it and keeps running after the job reports failure (#71148 + # / #59549). agent.deadline.kill_process_tree snapshots the descendant set via psutil BEFORE + # signalling, so own-session grandchildren are reached too — the unified deadline layer's + # tree-kill (#85147, d6a5cb9725). return False, f"Script timed out after {script_timeout}s: {path}" try: stdout_raw, stderr_raw = proc.communicate(timeout=min(0.1, remaining)) diff --git a/cron/suggestions.py b/cron/suggestions.py index 9ca896127e..15e1602832 100644 --- a/cron/suggestions.py +++ b/cron/suggestions.py @@ -28,6 +28,8 @@ logger = logging.getLogger(__name__) # Per-profile by design (anchored on get_hermes_home(), see cron/jobs.py). Optional test override; # production resolves the path at CALL time so multiplexed profile ticks (set_hermes_home_override) # cannot leak one profile's suggestions into the import-time home. +# Per-profile by design (issue #4707): suggestions live alongside the active profile's cron store. Anchor on +# get_hermes_home() (profile home), not the shared default root. Same pattern as cron/executions.py. SUGGESTIONS_FILE: Optional[Path] = None # Protects load->modify->save cycles (the background review fork and the main agent can both write). diff --git a/gateway/authz_mixin.py b/gateway/authz_mixin.py index 6f5efab2b3..cd53d3aa39 100644 --- a/gateway/authz_mixin.py +++ b/gateway/authz_mixin.py @@ -30,6 +30,10 @@ _ALLOW_ALL_ENV = {p: v.replace("_ALLOWED_USERS", "_ALLOW_ALL_USERS") for p, v in _GROUP_USER_ENV = {Platform.TELEGRAM: "TELEGRAM_GROUP_ALLOWED_USERS"} _GROUP_CHAT_ENV = {Platform.TELEGRAM: "TELEGRAM_GROUP_ALLOWED_CHATS", Platform.QQBOT: "QQ_GROUP_ALLOWED_USERS"} _ALLOW_BOTS_ENV = { + # Bots admitted by {PLATFORM}_ALLOW_BOTS bypass the human allowlist (#4466). Checked before the + # no-user-id guard below: some platforms deliver bot/automation traffic with no user_id at all -- e.g. + # Slack Workflow Builder posts arrive as subtype=bot_message with user=None -- so deferring past the + # guard would reject them outright (the same reason the chat-scoped allowlist above runs early). Platform.DISCORD: "DISCORD_ALLOW_BOTS", Platform.FEISHU: "FEISHU_ALLOW_BOTS", Platform.TELEGRAM: "TELEGRAM_ALLOW_BOTS", @@ -43,6 +47,10 @@ def _platform_gate_env(name: str, default: str = "") -> str: With a profile secret scope installed AND multiplexing active, a scoped miss returns ``default`` instead of falling through to ``os.environ``, which may hold ANOTHER profile's first-writer bridged value (allowlist leak). Single-profile deployments behave exactly like ``os.getenv``. + + Under multiplex the process env may hold ANOTHER profile's first-writer-bridged value (the YAML→env + bridges in the Discord/Telegram adapters' ``_apply_yaml_config`` are first-writer-wins), so falling + through would leak profile A's allowlist into profile B (issue #72348). """ if not name: return default @@ -93,6 +101,9 @@ def _adapter_config_extra(adapter) -> dict: # Nostr npub -> hex (Buzz): ``BUZZ_ALLOWED_USERS`` accepts hex or ``npub1…`` but inbound pubkeys # are always hex. Pure stdlib; mirrors plugins/platforms/buzz/adapter.py. +# Without decoding, the central allowlist comparison string-matches the raw npub against the hex pubkey and +# an operator who listed only their npub sees every message rejected ("Unauthorized user: ", +# #78428). _BECH32_CHARSET = "qpzry9x8gf2tvdw0s3jn54khce6mua7l" _BECH32_GENERATOR = (0x3B6A57B2, 0x26508E6D, 0x1EA119FA, 0x3D4233DD, 0x2A1462B3) @@ -149,7 +160,12 @@ def _npub_to_hex(npub: str) -> Optional[str]: def _normalize_nostr_allow_entries(entries: set) -> set: - """Add the hex form of every valid ``npub1…`` entry; invalid entries are kept as-is.""" + """Add the hex form of every valid ``npub1…`` entry; invalid entries are kept as-is. + + Hex entries pass through unchanged; each valid ``npub1…`` entry is decoded and its 64-char hex form + added, so either form authorizes the same identity (#78428). Invalid entries are kept as-is (they simply + never match an inbound hex pubkey). + """ return set(entries) | {h for e in entries if e.lower().startswith("npub1") and (h := _npub_to_hex(e))} @@ -173,6 +189,10 @@ def _principal_matches_allowlist(source, user_id: str, allowed_ids: set) -> bool check_ids.add(source.user_name) # Buzz: allowlist may hold npub or hex; inbound pubkeys are hex. if platform_value == "buzz": + # Buzz (Nostr-based): BUZZ_ALLOWED_USERS accepts npub or hex, but inbound event pubkeys are always + # 64-char hex. Decode npub entries to hex so an operator who listed only their npub authorizes the + # same identity as the hex form (#78428). Hex entries pass through unchanged, so existing hex-only + # allowlists keep working. allowed_ids = _normalize_nostr_allow_entries(allowed_ids) hex_user = _npub_to_hex(user_id) if user_id.startswith("npub") else None if hex_user: @@ -366,6 +386,9 @@ class GatewayAuthorizationMixin: # bridge), so read the live adapter's. entry = _registry_entry(source.platform) if entry and entry.allowed_users_env: + # Buzz) carry the same operator-configured allowlist in + # ``PlatformConfig.extra.allowed_users``. An absent/empty entry changes nothing here — the + # default-deny below still applies. See #82871, #98738. adapter_allow = extra.get("allowed_users") if not adapter_allow: return False @@ -526,6 +549,14 @@ class GatewayAuthorizationMixin: Order: explicit per-platform config; Email → "ignore" (inboxes hold arbitrary mail); explicit non-default global; adapter dm_policy (pairing → "pair", allowlist/disabled → "ignore"); any configured allowlist → "ignore" (spamming unknown contacts with codes is noisy and leaks); else "pair". + + 1. 2. Email defaults to ``"ignore"`` unless explicitly opted into pairing. 3. Explicit global + ``unauthorized_dm_behavior`` in config — wins for chat-shaped platforms when no per-platform + override is set. 4. When an adapter-level DM policy opts into pairing or silent drop, honor it. 5. + When an allowlist (``PLATFORM_ALLOWED_USERS``, ``PLATFORM_GROUP_ALLOWED_USERS`` / + ``PLATFORM_GROUP_ALLOWED_CHATS``, or ``GATEWAY_ALLOWED_USERS``) is configured, default to + ``"ignore"`` — the allowlist signals that the owner has deliberately restricted access; spamming + unknown contacts with pairing codes is both noisy and a potential info-leak. (#9337) 6. """ config = getattr(self, "config", None) if ( diff --git a/gateway/channel_directory.py b/gateway/channel_directory.py index 3ad56083a4..7688762786 100644 --- a/gateway/channel_directory.py +++ b/gateway/channel_directory.py @@ -312,7 +312,10 @@ async def _build_slack(adapter) -> List[Dict[str, Any]]: def _build_from_sessions(platform_name: str) -> List[Dict[str, str]]: - """Known channels/contacts from session origins: state.db first, sessions.json fallback (pre-migration).""" + """Known channels/contacts from session origins: state.db first, sessions.json fallback (pre-migration). + + state.db is the primary source (#9006): gateway session rows persist origin_json. + """ return _build_from_sessions_db(platform_name) or _build_from_sessions_json(platform_name) diff --git a/gateway/config.py b/gateway/config.py index 4b810ec934..0a7f36a605 100644 --- a/gateway/config.py +++ b/gateway/config.py @@ -330,6 +330,8 @@ class SessionResetPolicy: notify: bool = True # Notify the user when auto-reset occurs notify_exclude_platforms: tuple = ("api_server", "webhook") # A background process this old no longer blocks reset (not killed, only ignored by the guard). + # A forgotten preview server should not keep a session alive forever (#29177). Raise this if you run + # legitimate multi-day jobs whose liveness should pin the conversation open. bg_process_max_age_hours: int = 24 def to_dict(self) -> Dict[str, Any]: @@ -369,6 +371,8 @@ class ChannelOverride: # Platforms whose primary credential is ``PlatformConfig.token`` → its env var (empty-token # warnings; multiplex primary-startup gate in ``gateway.run``). Platforms absent here # authenticate another way and must never be skipped for a missing token. +# Platforms absent from this map authenticate some other way (session files, port-bound webhooks, +# api_key-only) and must never be skipped for a missing token. See #64674. PLATFORM_TOKEN_ENV_NAMES: dict["Platform", str] = { Platform.TELEGRAM: "TELEGRAM_BOT_TOKEN", Platform.DISCORD: "DISCORD_BOT_TOKEN", @@ -458,6 +462,11 @@ class StreamingConfig: buffer_threshold: int = DEFAULT_STREAMING_BUFFER_THRESHOLD cursor: str = DEFAULT_STREAMING_CURSOR # >0: final edit becomes a fresh message once the preview was visible this long (Telegram only; 0 = off). + # Ported from openclaw/openclaw#72038. When >0, the final edit for a long-running streamed response is + # delivered as a fresh message if the original preview has been visible for at least this many seconds, + # so the platform's visible timestamp reflects completion time instead of the preview creation time. + # Currently applied to Telegram only (other platforms ignore the setting). Default 0 disables the + # fresh-message replacement path; set >0 to opt in. fresh_final_after_seconds: float = 0.0 def to_dict(self) -> Dict[str, Any]: @@ -538,6 +547,9 @@ class GatewayConfig: quick_commands: Dict[str, Any] = field(default_factory=dict) # slash commands that bypass the agent loop sessions_dir: Path = field(default_factory=lambda: get_hermes_home() / "sessions") # Legacy sessions.json mirror of the routing index (primary: state.db) for external tooling / downgrades. + # The primary copy lives in state.db (gateway_routing table, #9006). Default True for backward + # compatibility with external tooling and downgrade safety; set gateway.write_sessions_json: false in + # config.yaml to stop producing the file. write_sessions_json: bool = True always_log_local: bool = True # Always save cron outputs to local files # Drop outbound "silence narration" (*(silent)*, 🔇, a bare ".") that ping-pongs in bot-to-bot @@ -561,6 +573,11 @@ class GatewayConfig: # stalls (adapter reconnect doing sync socket I/O) so a short block does not cause restart churn. # max_strikes ~= 90-120s sustained block; the heartbeat-fsync false positive is fixed at the root # (off-loop write + two-witness probe), so raising it would only delay recovery. + # On by default; set gateway.loop_watchdog: false in config.yaml to disable. Telegram/Discord reconnect + # doing synchronous socket I/O during a network blip — so a short block does not force exit code 75 and + # trigger a restart churn that stalls cron dispatch (recurring fleet incidents on 2026-08-17, kanban + # t_0f76430f/t_70483f23). A genuine wedge (event loop frozen for the full tolerance window) still + # escalates to a supervised restart. See #69089. loop_watchdog: bool = True loop_watchdog_probe_interval_s: float = DEFAULT_LOOP_WATCHDOG_INTERVAL_S loop_watchdog_probe_timeout_s: float = DEFAULT_LOOP_WATCHDOG_TIMEOUT_S @@ -605,6 +622,21 @@ class GatewayConfig: try: from gateway.platform_registry import platform_registry with contextlib.suppress(Exception): + # Iterate built-in platforms plus any registered plugin platforms so plugin authors get the + # same shared-key bridging (#24836). + # Registry-driven enable for plugin platforms. Built-ins have explicit blocks above. A + # plugin platform is enabled when its credentials are configured (``is_connected``) and its + # dependencies are either present (passive ``check_fn``) or installable on demand + # (``ensure_deps_fn``, run later by ``create_adapter()`` — never here). Plugins that need to + # seed ``PlatformConfig.extra`` from env vars (e.g. Google Chat's project_id / + # subscription_name) can supply ``env_enablement_fn`` on their PlatformEntry — called here + # BEFORE adapter construction. Enablement gate (#31116): when a plugin registers + # ``is_connected`` (the "has the user actually configured credentials for this?" check), we + # MUST consult it before flipping ``enabled = True``. Otherwise ``check_fn`` alone — a + # passive "is the SDK importable?" probe — silently enables platforms the user never opted + # into, and the gateway then tries to connect to Discord / Teams / Google Chat with no token + # and emits noisy retry-forever errors. ``_platform_status`` was already fixed for the same + # bug class in commit 7849a3d73; this is the runtime counterpart. from hermes_cli.plugins import discover_plugins discover_plugins() entry = platform_registry.get(platform.value) @@ -767,6 +799,14 @@ def load_gateway_config() -> GatewayConfig: config_loader.load_yaml_layer(_home, gw_data) except Exception as e: logger.warning( + # DingTalk settings → env vars: migrated to the dingtalk plugin's apply_yaml_config_fn hook + # (plugins/platforms/dingtalk/adapter.py). #41112 / #3823. + # Mattermost config bridge moved into plugins/platforms/mattermost/ + # adapter.py::_apply_yaml_config — see #25443 (apply_yaml_config_fn). + # Matrix settings → env vars: migrated to the matrix plugin's apply_yaml_config_fn hook + # (plugins/platforms/matrix/adapter.py). #41112 / #3823. + # Feishu settings → env vars: migrated to the feishu plugin's apply_yaml_config_fn hook + # (plugins/platforms/feishu/adapter.py). #41112 / #3823. "Failed to process config.yaml — falling back to .env / gateway.json values. " "Check %s for syntax errors. Error: %s", _home / "config.yaml", e, @@ -791,6 +831,9 @@ def _validate_gateway_config(config: "GatewayConfig") -> None: policy.idle_minutes = 1440 try: + # Reject known-weak placeholder tokens. Ported from openclaw/openclaw#64586: users who copy + # .env.example without changing placeholder values get a clear startup error instead of a confusing + # "auth failed" from the platform API. from hermes_cli.auth import has_usable_secret except ImportError: has_usable_secret = None diff --git a/gateway/config_env.py b/gateway/config_env.py index 5cf58c7077..a25c4b8fea 100644 --- a/gateway/config_env.py +++ b/gateway/config_env.py @@ -397,6 +397,14 @@ def _enable_plugin_platform(config: GatewayConfig, entry) -> None: return # Dependencies LAST — only for platforms already enabled or past the credential gate. try: + # ``check_fn`` is a PASSIVE probe (never installs); a platform whose deps are missing but which + # registered ``ensure_deps_fn`` still gets enabled here — the registry's ``create_adapter()`` runs + # the active installer at gateway start, when the user actually wants the platform up. Historically + # the ACTIVE installer was wired as ``check_fn`` and this sweep pip-installed + # Discord/Telegram/Slack/Feishu/Dingtalk SDKs on every ``load_gateway_config()`` call — including + # the desktop/dashboard readiness probe (``GET /api/status``) — blocking startup until every install + # finished and boot-looping the desktop app at 94%. The check_fn/ensure_deps_fn split (#79812) makes + # that impossible by construction. deps_ok = bool(entry.check_fn()) except Exception as e: logger.debug("check_fn for %s raised: %s", entry.name, e) diff --git a/gateway/config_loader.py b/gateway/config_loader.py index 2c88960570..52167d67bd 100644 --- a/gateway/config_loader.py +++ b/gateway/config_loader.py @@ -290,6 +290,10 @@ def apply_plugin_yaml_hooks(yaml_cfg: dict, gateway_platforms: Any, platforms_da if registry is None: return for entry in registry.all_entries(): + # Plugin-owned YAML→env config bridges (#24836). See ``PlatformEntry.apply_yaml_config_fn`` for the + # hook contract. Order: shared-key loop (above) → this dispatch → legacy hardcoded blocks (below; + # no-op when a hook already set their env var) → ``_apply_env_overrides()`` after + # ``GatewayConfig.from_dict``. if entry.apply_yaml_config_fn is None: continue platform_cfg, _ = platform_section(yaml_cfg, entry.name, gateway_platforms) @@ -315,9 +319,18 @@ def bridge_core_env_settings(yaml_cfg: dict, platforms_data: dict) -> None: if tl_require_mention is not None and "require_mention" not in (yaml_cfg.get("telegram") or {}): tg_plat = platforms_data.setdefault(Platform.TELEGRAM.value, {}) tg_plat.setdefault("extra", {}).setdefault("require_mention", tl_require_mention) + # Also bridge to the TELEGRAM_REQUIRE_MENTION env var that the adapter reads at runtime. This used + # to live in the telegram_cfg block in core; it stays in core because it keys off the TOP-LEVEL + # require_mention (not a telegram: block), so the telegram plugin's apply_yaml_config_fn hook — + # which only runs when a telegram config block exists — can't cover the no-telegram-block case + # (#3979). if not os.getenv("TELEGRAM_REQUIRE_MENTION"): os.environ["TELEGRAM_REQUIRE_MENTION"] = str(tl_require_mention).lower() + # Telegram settings → env vars / extra: migrated to the telegram plugin's apply_yaml_config_fn hook + # (plugins/platforms/telegram/adapter.py). #41112 / #3823. + # WhatsApp settings → env vars: migrated to the whatsapp plugin's apply_yaml_config_fn hook + # (plugins/platforms/whatsapp/adapter.py). #41112 / #3823. signal_cfg = yaml_cfg.get("signal", {}) if isinstance(signal_cfg, dict) and "require_mention" in signal_cfg and not os.getenv("SIGNAL_REQUIRE_MENTION"): os.environ["SIGNAL_REQUIRE_MENTION"] = str(signal_cfg["require_mention"]).lower() diff --git a/gateway/control_socket.py b/gateway/control_socket.py index 6a767a7a01..bac58ce6b2 100644 --- a/gateway/control_socket.py +++ b/gateway/control_socket.py @@ -77,7 +77,12 @@ def resolve_client_socket_path(home: Path) -> Optional[Path]: def _detect_supervisor() -> str: - """Supervisor kind for THIS process from its own launch env (not inferred outside-in).""" + """Supervisor kind for THIS process from its own launch env (not inferred outside-in). + + Unlike the outside-in `_detect_supervisor_for_pid` scan, this answers from the process's own launch + context — which is exactly the provenance the 92091 design wants declared rather than inferred. See + #92091. + """ env = os.environ if env.get("INVOCATION_ID"): return "systemd" @@ -330,5 +335,8 @@ def identify_gateway(home: Path, *, timeout: float = _DEFAULT_CLIENT_TIMEOUT) -> def pause_gateway_for_update(home: Path, *, timeout: float = _DEFAULT_CLIENT_TIMEOUT) -> Optional[dict[str, Any]]: """Ask the gateway serving ``home`` to drain and exit for an update. Returns the ACK ``{"pausing", "already_stopping", "pid", "drain_timeout"}`` or None when no gateway answers (old gateway without - the verb, no/dead socket) — the caller then uses the legacy signal/tree-kill pause path.""" + the verb, no/dead socket) — the caller then uses the legacy signal/tree-kill pause path. + + Step 2 of the socket migration (#92091). + """ return query_gateway_control(home, "pause-for-update", timeout=timeout) diff --git a/gateway/delivery.py b/gateway/delivery.py index b8478da44f..105175801b 100644 --- a/gateway/delivery.py +++ b/gateway/delivery.py @@ -256,6 +256,7 @@ class DeliveryRouter: # risk). Cron output is an ARTIFACT, not model chatter: a legitimately terse job ("...", a single 🔇) # has no mirror loop, and dropping it while returning success is how a cron gets logged as delivered # with nothing on the wire. Cron sends carry job_id in metadata; everything else is filtered. + # See #77763. is_cron_artifact = "job_id" in (metadata or {}) if self._filter_silence_narration_enabled() and not is_cron_artifact and _is_silence_narration(content): logger.warning("Dropped silence-narration outbound to %s (chat=%s): %r", diff --git a/gateway/delivery_ledger.py b/gateway/delivery_ledger.py index fac8e2b219..95c35ef937 100644 --- a/gateway/delivery_ledger.py +++ b/gateway/delivery_ledger.py @@ -95,7 +95,12 @@ def _initialize_schema(conn: sqlite3.Connection) -> None: def _transaction() -> Iterator[sqlite3.Connection]: """Open a connection, commit/rollback on exit, and ALWAYS close it: ``sqlite3.Connection`` as a context manager only commits/rolls back, so ``with _connect()`` alone leaks a connection (and its - WAL/SHM fds) per call — ``record_obligation`` runs on every final response; exhausts RLIMIT_NOFILE.""" + WAL/SHM fds) per call — ``record_obligation`` runs on every final response; exhausts RLIMIT_NOFILE. + + On a long-running gateway that exhausts ``RLIMIT_NOFILE`` (the cron-ledger sibling of this bug was + #69567 / PR #69594). ``record_obligation`` runs on every outbound final response, so this ledger is the + highest-frequency leaker. + """ conn = _connect() with closing(conn), conn: yield conn diff --git a/gateway/drain_control.py b/gateway/drain_control.py index 232d45ae4a..a849f076e4 100644 --- a/gateway/drain_control.py +++ b/gateway/drain_control.py @@ -29,6 +29,8 @@ _log = logging.getLogger(__name__) _DRAIN_REQUEST_FILENAME = ".drain_request.json" # Drain-gated lifecycle actions complete in minutes; an hour bounds the wedge a leaked # marker can cause. Long drains refresh the marker instead of raising this. +# Max-age fallback for a same-epoch orphaned marker (#85433). Long-running drains refresh the marker via +# write_drain_request() (idempotent re-write bumps ``requested_at``) rather than raising this bound. DRAIN_REQUEST_MAX_AGE_SECONDS = 3600.0 # Dedup for the expired-marker warning (the watcher re-reads every second); keyed by # ``requested_at`` so a keep-alive re-write that later expires logs again. @@ -95,6 +97,8 @@ def _marker_is_expired(body: dict[str, Any]) -> bool: Missing/unparseable and future-dated (clock skew) timestamps are honoured. Logged once per marker, not per poll — the operator's breadcrumb for a leak. + + See #85433. """ global _expiry_logged_for raw = body.get("requested_at") @@ -126,7 +130,17 @@ def _active_drain_body(home: Optional[Path]) -> Optional[dict[str, Any]]: def drain_requested(*, home: Optional[Path] = None) -> bool: - """True iff an active (present, same-epoch, unexpired) begin-drain marker exists.""" + """True iff an active (present, same-epoch, unexpired) begin-drain marker exists. + + A marker whose ``epoch`` does not match the current instantiation epoch is treated as absent: it + survived a container/VM restart (HERMES_HOME is a durable Fly volume on Hermes Cloud) and the lifecycle + action that triggered the drain has already completed — honouring it would wedge the freshly-restarted + gateway in ``draining`` (NS-570). A marker whose ``requested_at`` is older than + :data:`DRAIN_REQUEST_MAX_AGE_SECONDS` is likewise treated as absent: it is a same-epoch orphan whose + drain-gated action completed without a restart and was never cancelled (#85433). Both staleness checks + are lenient (see :func:`_marker_epoch_is_stale` / :func:`_marker_is_expired`): a legacy/corrupt marker + with no epoch and no timestamp, or an environment without ``/proc``, still reads as drain-active. + """ return _active_drain_body(home) is not None @@ -135,6 +149,12 @@ def drain_notification_suppressed(*, home: Optional[Path] = None) -> bool: Same activeness rule as :func:`drain_requested`, so an orphan can never silence a fresh gateway; a marker without the field reads False (fail toward louder). + + "Active" means exactly what :func:`drain_requested` means — a marker present AND stamped with the + current instantiation epoch AND not past its max-age. A stale (other-epoch) marker that survived a + machine restart on the durable HERMES_HOME volume, or an expired same-epoch orphan (#85433), is ignored + here just as it is for drain state (NS-570): we must never let an orphaned marker's flag silence a + *fresh* gateway's legitimate shutdown broadcast. """ body = _active_drain_body(home) return bool(body and body.get("suppress_notification")) diff --git a/gateway/kanban_watchers.py b/gateway/kanban_watchers.py index f38e6ee422..a95837ce37 100644 --- a/gateway/kanban_watchers.py +++ b/gateway/kanban_watchers.py @@ -279,6 +279,7 @@ class GatewayKanbanWatchersMixin: # Re-read the auto-decompose toggle live so disabling it # takes effect on the next tick, not on restart. _ad_enabled, _ad_per_tick = _resolve_auto_decompose_settings(_load_config) + # See #49638. if _ad_enabled: await _to_thread_process_service(dispatcher.auto_decompose_tick, _ad_per_tick) results = await _to_thread_process_service(dispatcher.tick_once) diff --git a/gateway/kanban_watchers_common.py b/gateway/kanban_watchers_common.py index 167e24d49b..c690c9f8be 100644 --- a/gateway/kanban_watchers_common.py +++ b/gateway/kanban_watchers_common.py @@ -68,6 +68,12 @@ def _resolve_auto_decompose_settings(load_config: Callable[[], Any]) -> "tuple[b Fails safe: a config read error returns ``(False, 3)`` rather than re-enabling a feature the user turned off. ``per_tick`` is clamped to ``>= 1``. + + Read fresh from config on every dispatcher tick (#49638) so that flipping ``kanban.auto_decompose: + false`` to STOP runaway fan-out takes effect on the next tick instead of requiring a gateway restart. + Auto-decompose is a safety toggle — a user who sees it create and launch tasks they didn't intend + reaches for this flag to halt it, and a stale boot-captured value silently ignoring that change is the + bug reported in #49638. """ try: cfg = load_config() diff --git a/gateway/kanban_watchers_dispatcher.py b/gateway/kanban_watchers_dispatcher.py index 28688b6068..29b57422bc 100644 --- a/gateway/kanban_watchers_dispatcher.py +++ b/gateway/kanban_watchers_dispatcher.py @@ -83,6 +83,9 @@ def _resolve_dispatcher_settings(kanban_cfg: dict, kb: Any) -> _DispatcherSettin # Fallback profile for tasks created without an assignee (e.g. via the # dashboard). Empty (the schema default) keeps skipping them. + # When set, the dispatcher applies it to unassigned ready tasks instead of skipping them indefinitely + # (#27145). Empty string (the schema default) means "no fallback, keep skipping" — backward-compatible + # with existing installs. default_assignee = (kanban_cfg.get("default_assignee") or "").strip() or None if default_assignee: logger.info("kanban dispatcher: default_assignee=%r (unassigned ready tasks " diff --git a/gateway/kanban_watchers_notifier.py b/gateway/kanban_watchers_notifier.py index 128b7ec037..8cf7499b92 100644 --- a/gateway/kanban_watchers_notifier.py +++ b/gateway/kanban_watchers_notifier.py @@ -25,6 +25,18 @@ _WAKE_KINDS = ("completed", "gave_up", "crashed", "timed_out", "blocked", "revie # Consecutive send failures (adapter raised OR reported SendResult(success=False)) # before a sub is dropped as a dead chat. 12 ≈ 60s at the 5s cadence: a transient # API outage must not permanently unsubscribe a live review-gate channel. +# Subscriptions are removed only when the task reaches the irreversible archived status. ``done`` is +# reversible in review/controller flows, so removing its subscription would silence a later reopen. We used +# to also unsub on any terminal event kind (gave_up / crashed / timed_out / blocked), but that silently +# dropped the user out of the loop whenever the dispatcher respawned the task: a worker that crashes, gets +# reclaimed, runs again, and crashes a second time would only notify on the first crash because the +# subscription was deleted after the first event. Same shape as the reblock-after-unblock cycle that PR +# #22941 fixed for `blocked`. Keeping the subscription alive until the task is archived lets the cursor +# (advanced atomically by claim_unseen_events_for_sub) handle dedup, and any retry-loop event reaches the +# user. Per-subscription send-failure counter. Adapter.send raising means the chat is dead (deleted, bot +# kicked, etc.) — after N consecutive send failures the sub is dropped so we don't spin against a dead chat +# every 5 seconds forever. A genuinely dead chat still drops, just ~60s later — a fine trade for an +# unattended gate where a false drop means silent work pileup. MAX_SEND_FAILURES = 12 _LOCAL_PATH_RE = re.compile(r"(? Path: return PAIRING_DIR if PAIRING_DIR is not None else get_hermes_dir("platforms/pairing", "pairing") @@ -51,6 +63,7 @@ def _default_pairing_dir() -> Path: # an already-configured allowlist (revoke removes them) so the operator's list stays # the visible source of truth. Platforms absent here (or with no allowlist # configured) keep the pairing store as the sole grant record (authz union). +# See #23778. _PLATFORM_ALLOWLIST_ENV = { "telegram": "TELEGRAM_ALLOWED_USERS", "discord": "DISCORD_ALLOWED_USERS", "whatsapp": "WHATSAPP_ALLOWED_USERS", "whatsapp_cloud": "WHATSAPP_CLOUD_ALLOWED_USERS", @@ -119,6 +132,8 @@ def _read_allowlist_env(env_var: str) -> str: miss must return empty rather than borrow it; unscoped callers keep the legacy ``os.getenv`` read. Writes (``save_env_value``/``remove_env_value``) target the active profile's ``.env`` / installed scope, not ``os.environ``. + + See #88441. """ with contextlib.suppress(Exception): from agent.secret_scope import UnscopedSecretError, get_secret @@ -241,6 +256,10 @@ def _load_json_file(path: Path) -> dict: return data if isinstance(data, dict) else {} except PermissionError as e: try: + # Surface this loudly: a 0600 file owned by a different user (classic Docker symptom: `docker + # exec` runs as root and writes the file, then the gateway process — running as `hermes` after + # gosu drop — can't read it) would otherwise be swallowed by the generic OSError branch below, + # silently leaving the user marked unauthorized. See issue #10270. st = path.stat() owner_info = f"owner_uid={st.st_uid} mode={oct(st.st_mode)[-4:]}" except OSError: @@ -455,6 +474,8 @@ class PairingStore: Returns ``{user_id, user_name}``, or ``None`` if the code is invalid/expired OR the platform is locked out (disambiguate with ``_is_locked_out``). Constant-time salted-hash compare; legacy plaintext entries are ignored and pruned at TTL. + + See #10195. """ with self._lock: self._cleanup_expired(platform) diff --git a/gateway/platform_registry.py b/gateway/platform_registry.py index 3feb57dde5..4e6fc5eb15 100644 --- a/gateway/platform_registry.py +++ b/gateway/platform_registry.py @@ -52,6 +52,15 @@ class PlatformEntry: # ACTIVE installer, run by ``create_adapter()`` only when ``check_fn`` is False (platform # enabled+configured, about to connect); None = a False check_fn is a hard block. Split # from check_fn because one field either installed from every status display or never. + # ACTIVE dependency installer: make the platform's dependencies available, installing them (pip / + # lazy_deps) if needed. Returns True once deps are importable, False if they could not be installed. + # None = no auto-install; a False ``check_fn`` is then a hard block (correct for platforms with no + # optional deps). Why two fields (#79812): when the ACTIVE installer was registered as ``check_fn``, + # every status display pip-installed SDKs as a side effect (desktop boot-loop at 94%, see + # gateway/config.py enablement comments); when the PASSIVE probe was registered instead, + # ``create_adapter()`` returned None before ``connect()`` could lazy-install, so the deps never + # installed at all (Teams deadlock). Splitting the two roles makes both call sites correct by + # construction. ensure_deps_fn: Optional[Callable[[], bool]] = None # Connected/configured for this PlatformConfig (``get_connected_platforms``, setup UI); # None falls back to ``validate_config`` or ``check_fn``. diff --git a/gateway/platforms/_shared.py b/gateway/platforms/_shared.py index 6c92935c64..3edb326d32 100644 --- a/gateway/platforms/_shared.py +++ b/gateway/platforms/_shared.py @@ -9,6 +9,7 @@ from __future__ import annotations import os from typing import Any +# Profile-scoped secret reader for multiplexing support (PR #50094) from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError from agent.secret_scope import get_secret as _scoped_get_secret @@ -30,6 +31,9 @@ def get_scoped_secret(name: str, default: Any = None) -> Any: def profile_scoped() -> bool: + # --------------------------------------------------------------------------- YAML → env config bridge + # (apply_yaml_config_fn, #25443) + # --------------------------------------------------------------------------- """True when running inside a multiplexed secondary profile's scope. Secondary-profile adapters are constructed/connected inside diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 930f644d17..e4ebd27146 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -559,7 +559,12 @@ def _reap_disconnected_agent_processes( agent: Any, *, source: str = "api_server_sse_disconnect") -> None: """Reap background processes an abandoned API-server turn created (these turns bypass ``TurnRunner``). Daemon-thread fire-and-forget; epoch-gated so a stale reaper never kills - a newer run's process on a shared task_id.""" + a newer run's process on a shared task_id. + + Mirrors the gateway-turn cleanup in ``gateway/run.py`` (#76115) for this API-server surface, which runs + its own agent lifecycle via ``_run_agent`` and never passes through ``TurnRunner`` — so it needs its own + trigger for the same baseline-diff reap. + """ process_task_id = getattr(agent, "_gateway_turn_process_task_id", "") process_baseline = getattr(agent, "_gateway_turn_process_baseline", None) if not process_task_id or process_baseline is None: @@ -1101,6 +1106,9 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): # Stateless request/response (``send()`` is a stub): async-delivery tools must not promise # delivery here, and a resumed turn completes the work rather than asking. supports_async_delivery: bool = False + # Same statelessness applies to the startup auto-resume prompt: no client is waiting to answer "session + # restored — what next?", so a resumed turn should complete the interrupted work rather than acknowledge + # (#57056). interactive_resume: bool = False # Admission-gated OpenAI-compatible entry points (bodies live in the mixin). @@ -1124,6 +1132,11 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): self._model_routes: Dict[str, Dict[str, Any]] = self._parse_model_routes(extra.get("model_routes")) # Opt-in bare ``model`` passthrough on OpenAI-compatible surfaces (generic clients # hardcode "gpt-4o" etc., hence off by default). + # Off by default: generic OpenAI clients routinely hardcode model names ("gpt-4o", ...), and + # existing deployments rely on those falling back to the gateway default rather than switching the + # executing model. Requests that send an explicit ``provider`` — and the Hermes-native session-chat + # and /v1/runs endpoints — are always honored regardless of this flag. (Idea credit: PR #22825 by + # @mssteuer.) self._direct_model_requests: bool = _coerce_request_bool( extra.get("direct_model_requests"), default=False) self._app: Optional["web.Application"] = None @@ -1141,6 +1154,9 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): self._session_db_lock: Optional[asyncio.Lock] = None # single-flight for lazy init self._max_concurrent_runs: int = self._resolve_max_concurrent_runs() # 0 disables # In-flight _run_agent() turns (/v1/runs tracks its own via _active_run_tasks). + # Concurrency cap shared across all agent-serving endpoints (/v1/chat/completions, /v1/responses, + # /v1/runs). Read from config.yaml gateway.api_server.max_concurrent_runs; 0 disables the cap. + # Bounds CPU / memory / upstream-LLM-quota exhaustion from a request flood (#7483). self._inflight_agent_runs: int = 0 # Every agent inside _run_agent() for shutdown interrupt, keyed by id() (the strong ref # keeps the id() from recycling); distinct from the run_id-keyed _active_run_agents. @@ -1447,7 +1463,11 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): def _profile_scope(profile: Optional[str]): """Enter the multiplex profile runtime scope, or a no-op when unset. No prefix AND multiplexing active enters the DEFAULT profile's scope (an unscoped run would raise - UnscopedSecretError on its first credential read); single-profile gateways no-op.""" + UnscopedSecretError on its first credential read); single-profile gateways no-op. + + Single-profile gateways keep the no-op — ``get_secret`` falls through to ``os.environ`` there, + unchanged. See #61276. + """ if not profile: with suppress(Exception): from agent.secret_scope import is_multiplex_active @@ -1563,7 +1583,13 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): def _bind_declared_conversation( self, session_id: Optional[str], gateway_session_key: Optional[str]) -> None: """Record the declared conversation key on the session row (AIAgent writes it unkeyed); - ``include_compression_ancestors`` covers a mid-turn rotation. UPDATE: no-op w/o a row.""" + ``include_compression_ancestors`` covers a mid-turn rotation. UPDATE: no-op w/o a row. + + ``include_compression_ancestors`` carries the key up a mid-turn compression rotation so the pre- and + post-rotation rows of one conversation share it, while that same walk deliberately stops at + ``/branch``, delegate and tool children (#79161). The statement is an UPDATE, so it is a harmless + no-op on a turn that failed before the row was created. + """ key = (gateway_session_key or "").strip() sid = str(session_id or "").strip() db = self._ensure_session_db() if key and sid else None @@ -2460,7 +2486,15 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): def _artifact_store_for(self, profile: str) -> ArtifactStore: """Profile-scoped artifact store, lazily created under the profile's data dir and cached - BY RESOLVED PROFILE (on a multiplex listener profile A must never pin B to A's root).""" + BY RESOLVED PROFILE (on a multiplex listener profile A must never pin B to A's root). + + The store root lives under the profile's data directory + (``/plugin-data/.../artifacts``-style controlled root), so artifacts never escape the + profile boundary. Stores are cached BY RESOLVED PROFILE — on a multiplex listener, profile A + touching the artifact route first must never pin profile B to A's physical root (same frozen-handle + class as the per-profile session-storage fix in #88734). The root itself is created on first use; + TTL cleanup runs on every store/load/prune. + """ profile_key = str(profile or "default") store = self._browser_control_artifacts.get(profile_key) if store is not None: @@ -2732,6 +2766,10 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): # A canonical Bot Chat auto-archived by the orphan reaper would make `hermes peer dm` # mint transient sessions: resurrect and re-list; deliberate archives stay put. try: + # Recoverable-archive resurrection (#92687): a canonical Bot Chat archived by the ws-orphan + # reaper / older agent cleanup is invisible to list_sessions_rich (include_archived=False), + # which would fail `hermes peer dm` resolution and mint transient sessions — same accident + # the tui_gateway lookups heal. from tools.bot_mode_probe import BOT_CHAT_TITLE stale = db.get_session_by_title(title_filter) if title_filter == BOT_CHAT_TITLE else None if stale and stale.get("archived") and db.unarchive_recoverable_session(stale["id"]): @@ -3015,6 +3053,14 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): """Sanitized runtime metadata for a finished session-chat turn.""" runtime = self._result_runtime(result, usage) return self._sanitize_runtime_metadata( + # Same shared ladder /api/status uses. Before this was unified, the two endpoints disagreed on + # the same page load — the sidebar strip read "running" (it probed GATEWAY_HEALTH_URL and scoped + # to the requested profile) while the Channels page rendered "The gateway is not running" (it + # did neither). Cross-container, profile-scoped, and launch-service-managed deployments each hit + # that split. profile_home is passed when the request was scoped to a named profile: + # gateway/status readers resolve process-level paths and do NOT follow the HERMES_HOME + # contextvar override (#56986 / #69143), so the profile's directory has to be handed over + # explicitly or messaging silently reports another profile's gateway (#71211). runtime=runtime, requested_runtime=runtime_request.get("requested"), route_source=runtime_request.get("route_source") or "global", @@ -3073,6 +3119,7 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): queue, _event_payload = events.queue, events.payload # Claim ownership inside the request's profile scope before any run-keyed state # exists, so /v1/runs/{id}* control is confined to the starting profile. + # See #93689. self._run_owners[run_id] = self._run_idempotency_scope(request) self._set_run_status( run_id, "queued", session_id=session_id, model=ctx["body"].get("model", self._model_name)) @@ -3492,7 +3539,10 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): """Bind session contextvars for an API-server agent run — the SINGLE chokepoint for every agent-entry path. Hardwires ``platform="api_server"`` + ``async_delivery=False`` (HTTP can never wake the agent after the turn) so no route reintroduces the silent no-op bug. - Returns reset tokens for ``clear_session_vars`` in a ``finally`` (request-scoped).""" + Returns reset tokens for ``clear_session_vars`` in a ``finally`` (request-scoped). + + See #10760. + """ from gateway.session_context import set_session_vars return set_session_vars( platform="api_server", chat_id=chat_id, session_key=session_key, session_id=session_id, @@ -3545,6 +3595,7 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): "output_tokens": getattr(agent, "session_completion_tokens", 0) or 0, "total_tokens": getattr(agent, "session_total_tokens", 0) or 0} # Effective session id lets callers track compression-triggered rotations. + # (#16938) _eff_sid = getattr(agent, "session_id", session_id) if isinstance(_eff_sid, str) and _eff_sid: result["session_id"] = _eff_sid @@ -3606,7 +3657,17 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): effective_task_id = session_id or str(uuid.uuid4()) # Process baseline for disconnect reaping (this surface bypasses TurnRunner) # + shutdown-interrupt registration, once for every caller. + # Baseline for selective background-process reaping on SSE client disconnect — mirrors + # gateway/run.py's gateway-turn cleanup (#76115); this API-server surface runs its own + # agent lifecycle and doesn't go through TurnRunner, so it needs its own baseline. + # /v1/runs runs its own agent lifecycle (no TurnRunner, no _run_agent) — record turn + # process ownership so stop/cancel can reap only the background processes this run + # created (#76115). _publish_turn_process_ownership(agent, effective_task_id) + # Registering here, once, covers every _run_agent() caller — the same reason the + # _ProviderAuthResolutionError handler below lives here rather than in each route. Only + # two callers pass ``agent_ref``, and only /v1/runs has a run_id, so neither is a usable + # hook for the rest. See #63529. self._shutdown_interruptible_agents[id(agent)] = agent result = agent.run_conversation( user_message=user_message, conversation_history=conversation_history, @@ -3633,6 +3694,11 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): self._shutdown_interruptible_agents.pop(id(agent), None) # Bind the declared key to the row the turn actually ended on # (agent.session_id carries a mid-turn rotation). Opt-in per route. + # Record the declared conversation on the row the turn actually ended on — + # ``agent.session_id`` already carries a mid-turn compression rotation (#16938), so + # the next reply resolves the live transcript rather than its retired parent. + # Opt-in: only the routes that resolve their session id from the declared key + # (/v1/responses, /v1/runs) record one, so no other caller's rows change shape. if bind_declared_conversation: self._bind_declared_conversation( getattr(agent, "session_id", None) or session_id, gateway_session_key) @@ -3752,6 +3818,14 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): # Config error, not transient: a bare ``return False`` would make the reconnect watcher # re-instantiate the adapter (+ sqlite connection) until EMFILE. self._set_fatal_error( + # A rejected API_SERVER_KEY is a configuration error, not a transient blip — the key will + # not become valid on its own. A bare ``return False`` makes the reconnect watcher in + # gateway.run treat it as retryable and loop forever at the backoff cap, re-instantiating + # the adapter (and its ResponseStore sqlite connection) every retry (#38803: ~501 leaked + # connections / 1002 fds over 2.5 days until EMFILE took the whole gateway down). + # Non-retryable drops it from the reconnect queue — same treatment as the port-conflict + # guard (api_server_port_in_use). The guard already logged the specific rejection reason + # just above. "api_server_key_invalid", "API_SERVER_KEY was rejected by the startup guard (missing, " "placeholder/too short, or strength unverifiable — see the " @@ -3800,6 +3874,13 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): await self._runner.setup() # Bind directly (a pre-probe raced the bind, misreporting TIME_WAIT as "in use"). # SO_REUSEADDR off on macOS (BSD can split traffic between two listeners). + # Bind directly instead of probing 127.0.0.1 first — the old single-family pre-probe raced the + # real bind and reported a TIME_WAIT socket as "in use" (#10297), failing gateway restarts for + # up to ~60s. SO_REUSEADDR is platform-dependent (same rationale as the webhook adapter, + # #65482): - macOS (BSD semantics): two sockets with SO_REUSEADDR can silently split traffic + # while both report success — disable. - Linux: SO_REUSEADDR only permits rebinding past + # TIME_WAIT (a second live listener needs SO_REUSEPORT, never set), so keep the default + # (enabled) for instant restart rebinds. self._site = web.TCPSite( self._runner, self._host, self._port, reuse_address=False if sys.platform == "darwin" else None) try: @@ -3811,6 +3892,13 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): if getattr(exc, "errno", None) == errno.EADDRINUSE: # Config error: non-retryable, or the reconnect watcher leaks fds forever. self._set_fatal_error( + # A port conflict is a configuration error, not a transient blip — another process + # holds the port for its lifetime. A bare ``return False`` makes the reconnect + # watcher in gateway.run treat it as retryable and loop forever at the backoff cap + # (observed: 1568+ retries over 5 days across multi-profile setups all defaulting to + # the same port, #52132), filling errors.log and leaking the adapter's ResponseStore + # fds each retry. Non-retryable drops it from the reconnect queue; the operator + # recovers with ``/platform resume api_server`` after changing the port. "api_server_port_in_use", f"Port {self._port} already in use. Set " f"platforms.api_server.port in config.yaml to a " @@ -3832,7 +3920,14 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): async def disconnect(self) -> None: """Stop the aiohttp server and release every owned resource, including the ResponseStore - connection (the reconnect loop builds a fresh adapter per retry; leaked fds hit EMFILE).""" + connection (the reconnect loop builds a fresh adapter per retry; leaked fds hit EMFILE). + + Without this, every adapter instance leaks 2 file descriptors (the database file and its WAL + sidecar) — the reconnect loop in ``gateway.run`` constructs a fresh adapter on every retry, so 2 + fds/retry × 300s backoff cap ≈ 12 fds/hour, which exhausts the default 2560 fd limit after ~12h of + failed reconnects and turns the whole gateway into a zombie (OSError: [Errno 24] Too many open + files, #37011). + """ self._mark_disconnected() if self._response_store is not None: try: diff --git a/gateway/platforms/api_server_openai_routes.py b/gateway/platforms/api_server_openai_routes.py index 9dbaa819d6..bed7181586 100644 --- a/gateway/platforms/api_server_openai_routes.py +++ b/gateway/platforms/api_server_openai_routes.py @@ -67,6 +67,8 @@ def _result_flags(result: Any) -> tuple: def _finish_reason(completed, is_partial, is_failed, err_msg, agent_error=None) -> str: """OpenAI ``finish_reason``: "length" for truncation, "error" for failure, else "stop".""" + # OpenAI uses "length" for truncation, "stop" for normal completion, and downstream SDKs accept "error" + # / custom codes. See issue #22496. if is_partial and err_msg and "truncat" in err_msg.lower(): return "length" if agent_error is not None or is_failed or (not completed and err_msg): @@ -411,6 +413,7 @@ class OpenAICompatRoutesMixin: _content_has_visible_payload, _derive_chat_session_id, _error_response, _invalid_request, _multimodal_validation_error, _normalize_chat_content, _normalize_multimodal_content, _openai_error, _redact_api_error_text, _resolve_media_to_data_urls) + # Bound total in-flight agent runs (configurable; #7483). limited = self._concurrency_limited_response() if limited is not None: return limited @@ -731,6 +734,7 @@ class OpenAICompatRoutesMixin: _content_has_visible_payload, _error_response, _invalid_request, _multimodal_validation_error, _normalize_multimodal_content, _redact_api_error_text, _resolve_media_to_data_urls, _responses_usage_payload) + # Bound total in-flight agent runs (configurable; #7483). limited = self._concurrency_limited_response() if limited is not None: return limited @@ -958,7 +962,12 @@ class OpenAICompatRoutesMixin: ) -> List[Dict[str, Any]]: """This turn's assistant/tool messages in client-safe shape: clients accumulating ``assistant.delta`` into one buffer cannot reconstruct assistant segments that preceded - tool calls, so ``run.completed`` carries the authoritative per-turn transcript.""" + tool calls, so ``run.completed`` carries the authoritative per-turn transcript. + + Emitting the authoritative per-turn transcript on ``run.completed`` lets any SSE consumer reconcile + its live view against ground truth without a separate ``GET /messages`` round-trip. Purely additive: + clients that ignore the field are unaffected. Refs #34703. + """ agent_messages = result.get("messages") if isinstance(result, dict) else None if not isinstance(agent_messages, list) or not agent_messages: return [] diff --git a/gateway/platforms/api_server_run_idempotency.py b/gateway/platforms/api_server_run_idempotency.py index 64d584d7b1..232714131a 100644 --- a/gateway/platforms/api_server_run_idempotency.py +++ b/gateway/platforms/api_server_run_idempotency.py @@ -71,6 +71,10 @@ class RunIdempotencyStore: try: self._conn = sqlite3.connect(db_path, check_same_thread=False, timeout=30) except Exception as exc: + # Docker may create the container object before `docker run` fails to start it (e.g. exit code + # 125 when the daemon isn't ready, or a timeout mid-pull). That orphan is left in "Created" + # state — which the exited-only orphan reaper (reap_orphan_containers, status=exited) never + # catches, so it leaks permanently. Remove it by its known name before re-raising. See #7439. logger.warning( "Run idempotency storage is unavailable; falling back to " "process memory, so replay will not survive a restart: %s", exc) diff --git a/gateway/platforms/api_server_runs.py b/gateway/platforms/api_server_runs.py index b7dcfbb3b7..2231b4f22a 100644 --- a/gateway/platforms/api_server_runs.py +++ b/gateway/platforms/api_server_runs.py @@ -465,6 +465,12 @@ def _run_agent_sync(self, run: _RunLaunch, agent, approval_notify, *, _api_serve from gateway.session_context import clear_session_vars from gateway.hosted_room_execution_policy import ( RoomExecutionPolicy, bind_room_execution_policy, reset_room_execution_policy) + # No eager slash-worker pre-warm: slash.exec spawns one on demand (its error path already relies on that + # respawn to recover from a dead worker). Each worker child runs its own MCP discovery (#61891), so + # pre-warming one per session forks the full stdio MCP fleet — ~20 OS processes per retained session on + # a config with a few stdio servers — even for sessions that never run a worker-routed command. Sessions + # held by a live transport are never reaped, so with the desktop app open for days those fleets + # accumulate until the OS refuses new process spawns. from tools.approval import ( register_gateway_notify, reset_current_session_key, set_current_session_key, unregister_gateway_notify) @@ -517,6 +523,9 @@ def _make_approval_notify(self, run: _RunLaunch, *, _api_server) -> Callable[[Di def _approval_notify(approval_data: Dict[str, Any]) -> None: event = dict(approval_data or {}) # Clients must never receive the raw flagged command: redact before it hits the stream. + # Redact credentials from the command before it enters the SSE/API event stream — same egress bug as + # #48456, second transport: API/desktop clients would otherwise receive the raw command Tirith + # flagged. Reuse the gateway seam. if "command" in event: from gateway.run import _redact_approval_command event["command"] = _redact_approval_command(event.get("command")) @@ -616,6 +625,9 @@ def _request_owns_run(self, request: "web.Request", run_id: str) -> bool: return owner == scope # No in-memory owner: only a durable record under the caller's scope admits it. # Under multiplex_profiles every profile holds a valid key, so ownerless = allow-all. + # Run state that exists without an owner stamp is an unanswered authorization question, not a run anyone + # may control — under gateway.multiplex_profiles every served profile holds a valid key, so admitting it + # would make the boundary allow-all (#93689). return self._run_idempotency_store.owns_run(scope, run_id) @@ -656,6 +668,10 @@ async def _handle_run_events(self, request: "web.Request", *, _api_server) -> "w if not self._request_owns_run(request, run_id): return _run_not_found(_api_server._openai_error, run_id) # Allow subscribing slightly before the run is registered (race window). + # Confirm the force-kill actually reaped the process before we clear its PID file / scoped locks. + # SIGKILL can fail to take (e.g. an uninterruptible-sleep or zombie-reaping parent), and if we blindly + # clear the metadata and start a fresh instance we end up with two live gateways fighting over the same + # token — the duplicate-gateway failure in #19471. for _ in range(20): if run_id in self._run_streams: break diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index aa1acac24c..eb0fbb6395 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -184,7 +184,13 @@ def build_auto_tts_output_path(platform) -> str: """Unique temp output path for gateway auto-TTS: ``.ogg`` for ``OPUS_VOICE_PLATFORMS`` (the tool's ``_repair_ogg_container`` then guarantees real Opus bytes), else ``.mp3``. Platform-awareness lives HERE because ``_clear_session_env`` wipes the TTS tool's - ``HERMES_SESSION_PLATFORM`` contextvar before the post-handler auto-TTS block runs.""" + ``HERMES_SESSION_PLATFORM`` contextvar before the post-handler auto-TTS block runs. + + Platforms whose native voice bubbles require Ogg/Opus (``tools.tts_tool.OPUS_VOICE_PLATFORMS`` — the + single source of truth) get an explicit ``.ogg`` path; the tool's central container repair + (``_repair_ogg_container``) then guarantees real Ogg/Opus bytes for every provider, including MP3-only + backends like Edge TTS. Everything else keeps the MP3 default. See #36685, #57049. + """ from tools.tts_tool import OPUS_VOICE_PLATFORMS ext = "ogg" if _platform_name(platform) in OPUS_VOICE_PLATFORMS else "mp3" audio_path = os.path.join( @@ -195,7 +201,10 @@ def build_auto_tts_output_path(platform) -> str: def utf16_len(s: str) -> int: """UTF-16 code units in *s* — Telegram's 4 096 limit counts those, so astral chars - (emoji, CJK Ext B) cost **two** units although Python's ``len()`` counts one.""" + (emoji, CJK Ext B) cost **two** units although Python's ``len()`` counts one. + + Ported from nearai/ironclaw#2304 which discovered the same discrepancy in Rust's ``chars().count()``. + """ return len(s.encode("utf-16-le")) // 2 @@ -430,6 +439,8 @@ if TYPE_CHECKING: from agent.display import ToolPreview @dataclass +# --------------------------------------------------------------------------- Streaming TTS format +# descriptor and handle (#60671) --------------------------------------------------------------------------- class AudioFormat: """Declared PCM format for a streaming-TTS session: every ``write_streaming_tts`` chunk must be raw little-endian PCM at this rate / channels / sample width.""" @@ -510,6 +521,15 @@ IMAGE_CACHE_DIR = get_hermes_dir("cache/images", "image_cache") # Inbound media cap (``gateway.max_inbound_media_bytes``): payloads are buffered fully in memory, # so an uncapped upload (Discord Nitro: 500 MB) could OOM-kill the gateway. +# Inbound image / audio / video payloads are buffered fully into process memory before being written to the +# cache directory. With no cap, a single large upload (Discord Nitro allows 500 MB) — or a remote URL in an +# inbound message payload pointing at an arbitrarily large file — can spike RAM and OOM-kill the gateway. +# The ``cache_*_from_bytes`` helpers (the shared funnel every platform reaches eventually) and the +# ``cache_*_from_url`` downloaders enforce this cap, so the protection holds regardless of which platform +# adapter or code path produced the bytes. Configurable via ``gateway.max_inbound_media_bytes`` in +# config.yaml. ``0`` disables the cap. Default 128 MiB — generous enough for ordinary photos/voice +# notes/short clips while still bounding a hostile upload. +# --------------------------------------------------------------------------- See #13145. DEFAULT_INBOUND_MEDIA_MAX_BYTES = 128 * 1024 * 1024 @@ -759,7 +779,13 @@ _ROOT_CREDENTIAL_PATHS = ( def _profile_cache_roots() -> List[Path]: """Per-profile cache roots ``/profiles//cache/{images,...}`` (the static safe roots cover only the active HERMES_HOME). Enumerated at check time so profiles created after - startup count and are allowlisted BEFORE the ``/root`` denylist (HERMES_HOME symlinked).""" + startup count and are allowlisted BEFORE the ``/root`` denylist (HERMES_HOME symlinked). + + ``HERMES_HOME=/opt/data``) while the model emits a profile-scoped path silently fails delivery. + Enumerated dynamically at check time so profiles created after startup are covered, and so the resolved + profile path is allowlisted *before* the ``/root`` system denylist is consulted (which otherwise wins + when HERMES_HOME is symlinked under a denied prefix and $HOME is not that prefix). See issue #31733. + """ try: profile_dirs = [p for p in (_HERMES_ROOT / "profiles").iterdir() if p.is_dir()] except OSError: @@ -897,7 +923,12 @@ def _docker_sandbox_dir_candidates(session_key: str = "") -> List[str]: """Candidate host sandbox dir names for the delivering session, best first. Mirrors ``_resolve_container_task_id`` (tools/terminal_tool.py): containers are PROFILE-scoped (``default``, else ``profile:``); legacy ``session:`` sandboxes stay as a fallback. - The key is passed explicitly because delivery runs after the turn's contextvars were cleared.""" + The key is passed explicitly because delivery runs after the turn's contextvars were cleared. + + Takes the key explicitly because the delivery pipeline runs after ``_handle_message_with_agent`` cleared + the turn's session contextvars (#93950) — an ambient lookup here would silently collapse onto + ``default`` and miss the session's real sandbox. + """ try: from tools.environments.path_utils import sanitize_task_id_for_path except Exception: @@ -974,7 +1005,10 @@ def _cache_dir_container_mounts() -> List[Tuple[Path, Path]]: def _warn_unresolved_docker_media(candidate: Path, session_key: str, reason: str) -> None: """Name WHY a container-absolute MEDIA path failed translation (otherwise the only signal is - the generic "Skipping unsafe MEDIA directive path" line). Docker-only; host rejections quiet.""" + the generic "Skipping unsafe MEDIA directive path" line). Docker-only; host rejections quiet. + + See #93950. + """ if not _docker_env_active(): return logger.warning("Docker MEDIA path %s did not resolve to a host sandbox file (%s%s); " @@ -1112,6 +1146,19 @@ SUPPORTED_IMAGE_DOCUMENT_TYPES = { # Media-delivery ext allowlist — SINGLE SOURCE OF TRUTH for both extractors and the cleanup # regexes: a tag is stripped only when deliverable, unknown-ext paths survive in the body. +# Both extractors that turn response text into native attachments derive their extension set from this +# tuple: * ``extract_media()`` — explicit ``MEDIA:`` tags * ``extract_local_files()`` — bare +# absolute/home paths the agent mentions Historically these two carried independently-maintained extension +# lists. ``extract_media`` had a narrow list (no .md/.json/.yaml/.xml/.html/...) while +# ``extract_local_files`` had a broad one. Combined with the unconditional ``MEDIA:\\s*\\S+`` cleanup at the +# dispatch sites, that mismatch created a silent black hole: a ``MEDIA:/report.md`` tag failed the narrow +# extract_media match, got stripped from the body by the loose cleanup regex, and was then invisible to +# extract_local_files — the file was never delivered (issue #34517). Keeping one list eliminates the drift; +# building the cleanup regexes from the same set means a tag is only stripped when its extension is one we +# can actually deliver, so an unknown-extension path survives in the body instead of vanishing. Covers +# images (inline), video (inline where supported), audio (voice/audio), documents/spreadsheets/presentations +# (send_document), archives, and rendered web output. The dispatch partition (image vs video vs document) +# lives in ``gateway/run.py``. --------------------------------------------------------------------------- MEDIA_DELIVERY_EXTS: Tuple[str, ...] = ( ".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp", ".tiff", ".svg", # images (embed inline) ".mp4", ".mov", ".avi", ".mkv", ".webm", ".3gp", # video (embed inline where supported) @@ -1133,6 +1180,22 @@ _MEDIA_EXT_ALTERNATION = "|".join(sorted((e.lstrip(".") for e in MEDIA_DELIVERY_ # ``MEDIA:`` as a boundary so glued tags (``MEDIA:/a.pngMEDIA:/b.png``) never merge; sentence-final # ``.`` is a boundary only before whitespace/EOL (``data.csv.`` -> ``data.csv``, ``archive.tar.gz`` # extends past ``.tar``); CJK full-width punctuation terminates paths (``早报.pdf(782.6 KB)``). +# A ``MEDIA:`` tag with an unknown extension is left in the text so it can still be picked up by the +# bare-path detector (extract_local_files) downstream rather than silently deleted. Path anchors: ``~/`` +# (Unix home-relative), ``/`` (Unix absolute), ``X:\\`` or ``X:/`` (Windows drive-letter absolute — #34632). +# Emphasis tolerance: models routinely wrap the tag in Markdown emphasis (``**MEDIA:/x.pdf**``, +# ``*MEDIA:/x.pdf*``, ``_MEDIA:/x.pdf_``) when they present a file to the user. The old single-quote anchor +# (``[`"']?``) and the closing lookahead (which lacked ``*``/``_``) failed to match such tags, so the file +# was silently never delivered and the literal ``MEDIA:`` text leaked into the chat. Allow a short run of +# emphasis/quote markers on both sides so the tag is recognised regardless of cosmetic Markdown. Code-block +# / inline-code / blockquote contexts are still neutralised earlier by ``_mask_protected_spans`` (#35695), +# so example tags remain non-deliverable. The trailing lookahead also accepts ``MEDIA:`` as a boundary, so +# the next tag stops the current match cleanly (#68773). The whitespace guard keeps multi-part extensions +# intact — for ``archive.tar.gz`` the ``.`` after ``tar`` is followed by ``g``, so the match must extend to +# ``.gz`` instead of stopping early at ``.tar``. CJK full-width punctuation accepted as MEDIA path +# terminators, mirroring the ASCII set in the looka below. Chinese-language agent output naturally writes +# ``MEDIA:D:\path\早报.pdf(782.6 KB)`` or ``MEDIA:...pdf:内容`` — without these, the lookahead fails and the +# attachment is silently dropped (#88038). _MEDIA_CJK_TERMINATORS = "()〈〉《》:,。;!?、\u201c\u201d\u2018\u2019【】" MEDIA_TAG_CLEANUP_RE = re.compile( @@ -1146,6 +1209,23 @@ MEDIA_TAG_CLEANUP_RE = re.compile( # ``validate_media_delivery_path`` accepts them, so injected paths that don't validate stay visible. # The bare path class is whitespace-bounded (a tag glued to the next ``MEDIA:`` or prose must not # absorb it); spaced paths are recovered by ``_match_extensionless_path`` with on-disk validation. +# Paths NOT covered by MEDIA_TAG_CLEANUP_RE's extension alternation — both extension-less files (Caddyfile, +# Dockerfile, Makefile) and files with an unknown extension (.py, .log, .weirdext, ...) — are validated and +# delivered via MEDIA_EXTENSIONLESS_TAG_RE. Every ``MEDIA:`` path is therefore deliverable regardless of +# file type (#36060): known extensions extract unconditionally via the anchored pattern above, everything +# else extracts only after ``validate_media_delivery_path`` accepts it (exists on disk, not under the +# credential/system denylist, strict-mode rules honored), so prompt-injection paths that do not validate are +# left visible instead of silently dropped. The path class uses a tempered-greedy token (``[^\s\n`"']+?`` +# followed by a ``(?=...)`` lookahead) instead of the prior ``[^\s\n`"']+`` so a tag glued to the next +# ``MEDIA:`` keyword (``MEDIA:/a.pngMEDIA:/b.png``) or to arbitrary following text (``MEDIA:/a.pngSome +# text``) cannot silently absorb the next path — that earlier behavior merged the two paths into one invalid +# string and dropped the file (#68773). The bare form stays non-greedy and whitespace-bounded — spaced paths +# are NOT absorbed at the regex level, because greedy space-tolerance would reintroduce the #68773 bug class +# (gluing the next MEDIA: tag or trailing prose into one invalid path). Instead, unknown-extension paths +# containing spaces (``MEDIA:/data/map data.kmz``, ``C:\...\My Documents\x.log``) are recovered by +# ``_match_extensionless_path`` (#24032): when the bare match fails validation, the candidate is +# progressively extended forward across single spaces — bounded, stopping at newline / the next ``MEDIA:`` +# keyword — and the first extension that validates on disk wins. MEDIA_EXTENSIONLESS_TAG_RE = re.compile( r'''[`"'*_]{0,3}MEDIA:\s*''' r'''(?P`[^`\n]+`|"[^"\n]+"|'[^'\n]+'|''' @@ -1158,7 +1238,14 @@ MEDIA_EXTENSIONLESS_TAG_RE = re.compile( def _match_extensionless_path(scan_text: str, match: "re.Match") -> Optional[Tuple[str, int]]: """Extensionless MEDIA tag match -> validated on-disk ``(safe_path, end_offset)`` or None: the captured path first, then extended across single spaces (max 8 tokens, never past a newline - or the next ``MEDIA:``).""" + or the next ``MEDIA:``). + + When that fails validation, the candidate is progressively extended forward across single spaces + (validation-gated, bounded at 8 tokens, never past a newline or a subsequent ``MEDIA:`` keyword) so + unknown-extension paths containing spaces deliver (#24032). Returns ``(safe_path, end_offset)`` where + ``end_offset`` is the index in ``scan_text`` just past the matched path, or ``None`` when nothing + validates. + """ path = _normalize_media_tag_path(match.group("path")) if not path: return None @@ -1192,7 +1279,13 @@ def _normalize_media_tag_path(raw: str) -> str: def _path_lacks_deliverable_extension(path: str) -> bool: """True when ``path`` has no extension or one outside MEDIA_DELIVERY_EXTS — such paths - take the validated delivery pass so nonexistent / denylisted ones stay visible.""" + take the validated delivery pass so nonexistent / denylisted ones stay visible. + + ``path`` — either the basename has no extension at all (Caddyfile, Makefile, …) or the extension is not + in MEDIA_DELIVERY_EXTS (.py, .log, .weirdext, …). Such paths route through the validated delivery pass + (``validate_media_delivery_path``) instead of the unconditional one, so every file type is deliverable + (#36060) while nonexistent / denylisted paths stay visible in the text. + """ return Path(path).suffix.lower() not in MEDIA_DELIVERY_EXTS @@ -1264,7 +1357,10 @@ def _strip_media_tag_directives(text: str) -> str: """Remove MEDIA: tags and [[audio_as_voice]] / [[as_document]] markers so they never render as text (backstop after ``extract_media``). Protected spans are mask-located only — tags inside them are neither stripped nor mangled, matching ``extract_media`` so display and - delivery agree. Empty/None text is returned as-is.""" + delivery agree. Empty/None text is returned as-is. + + See #16434. + """ if not text or not _has_media_directives(text): return text cleaned = text.replace("[[audio_as_voice]]", "").replace("[[as_document]]", "") @@ -1748,6 +1844,12 @@ class BasePlatformAdapter(ABC): supports_inchannel_continuable: bool = False # A human can answer "session restored — what next?"; webhook-style platforms set False so # auto-resume finishes the work instead of asking nobody. + # The startup auto-resume turn (``_schedule_resume_pending_sessions`` → the ``_is_resume_pending`` + # branch in ``_handle_message_with_agent``) reads this to pick its guidance: interactive platforms + # (Telegram, Slack, Discord DMs, …) get "report the restore and ask what the user wants next"; + # non-interactive event platforms (webhook) get "finish the interrupted work" because nobody is there to + # answer, and an acknowledgement would silently abandon the task (#57056). Read generically via + # ``getattr(adapter, "interactive_resume", True)`` — no per-platform branching at the call site. interactive_resume: bool = True # Back-reference to the running ``GatewayRunner`` (set by gateway/run.py); ``build_source`` # resolves the inbound profile via ``runner._profile_name_for_source``. @@ -1769,6 +1871,7 @@ class BasePlatformAdapter(ABC): # Strong refs to shielded fatal-error handler tasks that outlive their carrier task # (asyncio keeps only weak refs); without them the loop can GC the detached handler # mid-flight — the "handler killed mid-flight" class (#81335). + # See #81335. self._detached_fatal_tasks: set = set() # Lock takeover armed only for the initial connect of ``gateway run --replace``. self._platform_lock_takeover_allowed = self._platform_lock_takeover_attempted = False @@ -1799,6 +1902,9 @@ class BasePlatformAdapter(ABC): self._auto_tts_default: bool = False self._auto_tts_enabled_chats, self._auto_tts_disabled_chats = set(), set() # Turn keys where streaming TTS already delivered audio; whole-file auto-TTS skips them. + # When the gateway streaming-TTS consumer successfully delivers audio, it adds the turn key here so + # the base adapter's whole-file auto-TTS path skips the duplicate. Cleared after the turn completes. + # See #60671. self._streaming_tts_completed_turns: set[str] = set() # Chats whose typing indicator is paused (approval waits); _keep_typing skips them. self._typing_paused: set = set() @@ -1919,7 +2025,11 @@ class BasePlatformAdapter(ABC): def _should_auto_tts_for_chat(self, chat_id: str) -> bool: """Whether auto-TTS fires for ``chat_id``: explicit ``/voice on|tts`` wins, - then explicit ``/voice off``, then the global ``voice.auto_tts`` default.""" + then explicit ``/voice off``, then the global ``voice.auto_tts`` default. + + Decision layers (Issue #16007): 1. Explicit ``/voice on`` or ``/voice tts`` → always fire (even if + ``voice.auto_tts`` is False). 2. 3. + """ return chat_id in self._auto_tts_enabled_chats or ( chat_id not in self._auto_tts_disabled_chats and bool(self._auto_tts_default)) @@ -1999,6 +2109,7 @@ class BasePlatformAdapter(ABC): # cancels; unshielded, the handler died mid-flight — adapter popped from the # gateway map but never queued for background reconnect, leaving a zombie # gateway with no platforms and no pending retries (#81335). + # See #81335. task = asyncio.ensure_future(result) # Strong ref: asyncio only keeps weak refs to tasks ("save a reference ... to # avoid a task disappearing mid-execution"); matches @@ -2351,7 +2462,12 @@ class BasePlatformAdapter(ABC): async def delete_message(self, chat_id: str, message_id: str) -> bool: """Delete a sent message; True on success (platforms without a deletion API return False and callers leave it). Used by the stream consumer's fresh-final cleanup to remove stale - previews.""" + previews. + + Used by the stream consumer's fresh-final cleanup path (see openclaw/openclaw#72038) to remove + long-lived preview messages after sending the completed reply as a fresh message so the platform's + visible timestamp reflects completion time. + """ return False def _get_ephemeral_system_ttl_default(self) -> int: @@ -2603,6 +2719,11 @@ class BasePlatformAdapter(ABC): # ── Streaming TTS contract: voice adapters accept PCM chunks while the LLM generates. # Defaults report "unsupported" (whole-file fallback). + # ------------------------------------------------------------------ Streaming TTS adapter contract + # (#60671) ------------------------------------------------------------------ Voice-capable adapters + # (LiveKit, Discord voice, …) override these to accept PCM audio chunks while the LLM is still + # generating. The default implementations report "unsupported" so existing adapters are + # source-compatible and keep the whole-file auto-TTS fallback. def supports_streaming_tts(self, chat_id: str, audio_format: AudioFormat) -> bool: """Return True when this adapter can accept streaming PCM for *chat_id*.""" return False @@ -2654,7 +2775,12 @@ class BasePlatformAdapter(ABC): self, chat_id: str, media_path: str, *, is_voice: bool = False, metadata: Optional[Dict[str, Any]] = None) -> None: """User-visible notice when a MEDIA attachment upload failed: the tag was - already stripped from the text, so silence would be a silent drop.""" + already stripped from the text, so silence would be a silent drop. + + The non-streaming dispatch loop strips ``MEDIA:`` tags before sending attachments. When the + subsequent upload returns ``success=False`` (for example Discord accepted the message but attached + nothing), the user must see a failure notice instead of a silent drop (#66797). + """ ext = Path(media_path).suffix.lower() if is_voice or should_send_media_as_audio(self.platform, ext, is_voice=is_voice): text = _media_failure_text("audio") @@ -2707,6 +2833,7 @@ class BasePlatformAdapter(ABC): continue # This is a MEDIA path quote, not inline code # A whole tag in inline code (`MEDIA:/path.csv`) is a real directive (models format # paths as code): deliver IF it validates; non-existent examples stay masked. + # See #35695. inner = m.group(0)[1:-1].strip() if inner.upper().startswith("MEDIA:"): candidate = _normalize_media_tag_path(inner[6:]) @@ -2722,7 +2849,12 @@ class BasePlatformAdapter(ABC): like ``{"result": "MEDIA:/x/stale.png"}``) so they are never re-delivered. Only value-context strings (``:,{[`` before the ``"``) and bare paths (``/``, ``~/``, ``X:\\``) count; ``MEDIA:"..."`` quoted tags and line-start/prose tags are untouched. Offsets - preserved.""" + preserved. + + Here the ``MEDIA:`` is part of stored text, not an outbound directive, but the bare-path branch of + ``MEDIA_TAG_CLEANUP_RE`` would still match it and re-deliver a stale file. (Regression report + #34375.) + """ if '"' not in content or "MEDIA:" not in content: return content # Value-context string: quote preceded by : , { or [; escape-aware body to the closing @@ -2744,6 +2876,11 @@ class BasePlatformAdapter(ABC): # Scan a masked copy so example/stored MEDIA paths (code, quotes, JSON values) are never # delivered; dedupe on the expanded path so a file referenced twice uploads once. scan_content = _mask_media_scan_text(content) + # - code blocks / inline code / blockquotes hold prose examples (#35695) - serialized JSON string + # values hold stored tool-result text (#34375) Both maskers are offset-preserving (chars -> + # spaces) so match offsets stay valid; chaining them masks the union of both protected regions. + # Dedupe on the expanded path (first occurrence wins) so the same file referenced twice in one + # response — e.g. a MEDIA tag inline AND in a summary footer — is uploaded once, not twice (#29131). seen_paths: set = set() def _add(path: str) -> None: @@ -2785,6 +2922,9 @@ class BasePlatformAdapter(ABC): mutilated. Dispatch by type lives in ``gateway/run.py``.""" ext_part = '|'.join(e.lstrip('.') for e in MEDIA_DELIVERY_EXTS) # Lookbehind rejects URL/relative matches (https://…/img.png, ./foo.png). + # (? None: """Release a finished owner task's guard, dropping its ``_session_tasks`` entry ONLY if the guard was released: after a concurrent guard swap the done-task entry lets - ``_session_task_is_stale`` heal the orphan.""" + ``_session_task_is_stale`` heal the orphan. + + Release-then-conditional-delete is the #48300 fix: when a concurrent path (reset/new command, drain + handoff) swapped ``_active_sessions[key]`` to a different guard, ``_release_session_guard`` skips on + the guard mismatch and the lock stays installed. If we deleted ``_session_tasks`` unconditionally + (the old order), ``_session_task_is_stale`` would later see no owner task and report "not stale", so + the orphaned guard would never be healed — a permanent session deadlock. Keeping the done-task entry + when the guard survives lets the on-entry self-heal detect the stale lock and clear it on the next + inbound message. + """ self._release_session_guard(session_key, guard=interrupt_event) if session_key not in self._active_sessions: self._session_tasks.pop(session_key, None) diff --git a/gateway/platforms/bluebubbles.py b/gateway/platforms/bluebubbles.py index 8a0bead2ea..8c0eb2788c 100644 --- a/gateway/platforms/bluebubbles.py +++ b/gateway/platforms/bluebubbles.py @@ -199,6 +199,7 @@ class BlueBubblesAdapter(BasePlatformAdapter): return False from aiohttp import web # Tighter keepalive so idle CLOSE_WAIT drains promptly. + # See #18451. from gateway.platforms._http_client_limits import platform_httpx_limits self.client = httpx.AsyncClient(timeout=30.0, limits=platform_httpx_limits()) try: @@ -215,6 +216,9 @@ class BlueBubblesAdapter(BasePlatformAdapter): return False # client_max_size makes aiohttp enforce the cap on every read path, incl. chunked requests # with no Content-Length. + # Explicit body cap: BlueBubbles webhook events are small JSON (or form-encoded) payloads. + # client_max_size makes aiohttp enforce the cap on every read path — including chunked requests that + # carry no Content-Length (same pattern as webhook.py / raft, #58536/#58902). app = web.Application(client_max_size=_WEBHOOK_MAX_BODY_BYTES) app.router.add_get("/health", lambda _: web.Response(text="ok")) app.router.add_post(self.webhook_path, self._handle_webhook) @@ -317,7 +321,10 @@ class BlueBubblesAdapter(BasePlatformAdapter): """Resolve an email/phone to a chat GUID (raw ``a;-;b`` GUIDs pass through). Matches strictly on ``chatIdentifier`` / ``identifier``; participant membership is intentionally NOT a fallback — the same contact appears in a 1:1 DM and any number of groups, so a participant match could - leak a DM reply into a group thread. ``None`` lets the caller create a fresh DM.""" + leak a DM reply into a group thread. ``None`` lets the caller create a fresh DM. + + See #24157. + """ target = (target or "").strip() if not target or ";" in target: return target or None diff --git a/gateway/platforms/qqbot/adapter.py b/gateway/platforms/qqbot/adapter.py index 5148e25980..5596b83eef 100644 --- a/gateway/platforms/qqbot/adapter.py +++ b/gateway/platforms/qqbot/adapter.py @@ -79,6 +79,14 @@ def check_qq_requirements() -> bool: _VOICE_EXTENSIONS = (".silk", ".amr", ".mp3", ".wav", ".ogg", ".m4a", ".aac", ".speex", ".flac") _STT_PROVIDER_BASE_URLS = { "zai": "https://open.bigmodel.cn/api/coding/paas/v4", + # Aliases that target direct REST APIs not modeled as first-class providers in PROVIDER_REGISTRY. Used + # for ``auxiliary..provider`` so users can write the obvious name and have it resolve to a working + # ``custom`` endpoint without needing to know our internal provider IDs. Why these specifically: + # PROVIDER_REGISTRY has ``openai-codex`` (OAuth) and ``custom`` (manual base_url + OPENAI_API_KEY) but + # no plain ``openai`` for direct API-key access. Users predictably type ``provider: openai`` and expect + # it to use OPENAI_API_KEY against api.openai.com. Previously this silently fell back to the user's main + # provider, sending OpenAI model names to e.g. DeepSeek and producing cryptic ``unknown variant + # 'image_url'`` errors (issue #31179). "openai": "https://api.openai.com/v1", "glm": "https://open.bigmodel.cn/api/coding/paas/v4"} _AUDIO_URL_EXTENSIONS = {".silk", ".amr", ".mp3", ".wav", ".ogg", ".m4a", ".aac", ".flac"} @@ -195,6 +203,7 @@ class QQAdapter(BasePlatformAdapter): try: # Tighter keepalive pool so idle CLOSE_WAIT sockets drain faster behind proxies. + # See #18451. from gateway.platforms._http_client_limits import platform_httpx_limits from tools.url_safety import create_ssrf_safe_async_client self._http_client = create_ssrf_safe_async_client( diff --git a/gateway/platforms/signal.py b/gateway/platforms/signal.py index 3883cd00eb..cfb06ffa52 100644 --- a/gateway/platforms/signal.py +++ b/gateway/platforms/signal.py @@ -232,6 +232,7 @@ class SignalAdapter(BasePlatformAdapter): except Exception as e: logger.warning("Signal: Could not acquire phone lock (non-fatal): %s", e) # Tighter keepalive so idle CLOSE_WAIT drains promptly. + # See #18451. from gateway.platforms._http_client_limits import platform_httpx_limits self.client = httpx.AsyncClient(timeout=30.0, limits=platform_httpx_limits()) try: diff --git a/gateway/platforms/webhook.py b/gateway/platforms/webhook.py index c84022b468..9c928a474c 100644 --- a/gateway/platforms/webhook.py +++ b/gateway/platforms/webhook.py @@ -149,6 +149,8 @@ class WebhookAdapter(BasePlatformAdapter): """Generic webhook receiver that triggers agent runs from HTTP POSTs.""" # Event-triggered, no human present: startup auto-resume must FINISH the interrupted work, not ask "what next?". + # The startup auto-resume turn must instruct the model to FINISH the interrupted work instead of + # emitting an interactive acknowledgement that abandons the task (#57056). interactive_resume: bool = False def __init__(self, config: PlatformConfig): @@ -531,6 +533,7 @@ class WebhookAdapter(BasePlatformAdapter): return web.json_response({"status": "ignored", "reason": "filter", "route": route_name}) # Script, prompt render and skill lookup read the profile's home (skills/, config); the runner # only enters the routed profile's scope later around handle_message, so enter it here. + # See #67277. with self._profile_scope(profile): script = route_config.get("script") if script: @@ -720,6 +723,9 @@ class WebhookAdapter(BasePlatformAdapter): try: # Off-loop: `gh` does network I/O up to its 30s timeout; inline it froze every adapter and # timer on the gateway event loop. + # Running it inline froze every adapter and timer on the gateway event loop for the duration + # (Pattern A, #91912 class). asyncio.to_thread keeps the loop serving while the subprocess runs; + # the worker thread is bounded by the subprocess timeout below. result = await asyncio.to_thread( subprocess.run, ["gh", "pr", "comment", str(pr_int), "--repo", repo, "--body", content], capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=30) diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index 3a70663617..1316af4c2d 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -41,7 +41,10 @@ from agent.secret_scope import UnscopedSecretError, get_secret def _wx_secret(name: str, default: Optional[str] = None) -> Optional[str]: """Scope-aware WEIXIN_* read. Secondary profiles run scoped: a miss returns ``default`` (never borrow ``os.environ``). The DEFAULT profile runs *unscoped* under multiplexing, - where ``get_secret`` raises; there ``os.environ`` is its own value, so fall back.""" + where ``get_secret`` raises; there ``os.environ`` is its own value, so fall back. + + Same pattern as the Slack ``SLACK_APP_TOKEN`` read (#59739) and WhatsApp's ``_get_wsecret``. + """ try: return get_secret(name, default) except UnscopedSecretError: @@ -90,7 +93,13 @@ def _is_session_expired(resp: Dict[str, Any], ret: Any, errcode: Any) -> bool: def _make_ssl_connector() -> Optional["aiohttp.TCPConnector"]: """TCPConnector with certifi's CA bundle (``ilinkai.weixin.qq.com`` fails some system stores, e.g. Homebrew OpenSSL); None without certifi so aiohttp's default (honors ``SSL_CERT_FILE`` under trust_env) applies. - ``keepalive_timeout=2`` + ``enable_cleanup_closed`` drain idle CLOSE_WAIT sockets behind proxies like Warp.""" + ``keepalive_timeout=2`` + ``enable_cleanup_closed`` drain idle CLOSE_WAIT sockets behind proxies like Warp. + + Uses a tight ``keepalive_timeout=2`` (default aiohttp: 30s) so idle connections drain promptly behind + proxies like Cloudflare Warp that leave peer-initiated FIN in ``CLOSE_WAIT`` (same class as #18451). + ``enable_cleanup_closed=True`` helps the connector clean up sockets that the remote side has already + closed. + """ try: import ssl import certifi @@ -531,7 +540,13 @@ def _extract_text(item_list: List[Dict[str, Any]]) -> str: # Tencent's ``voice_item.text`` is their STT output and is wrong for non-Chinese audio. # When raw audio exists return "" so gateway/run.py's central STT transcribes the download; # otherwise use Weixin's transcript but mark its voice origin. + # #27300: Tencent Cloud's `voice_item.text` is their STT output, which is wrong for any + # non-Chinese audio (the original report was a Russian voice message that came back as English + # gibberish). Return empty so the central STT pipeline in ``gateway/run.py`` produces the body + # from the downloaded audio instead. voice_item = item.get("voice_item") or {} + # Use it, but preserve the voice origin so the agent can distinguish this from text the user + # typed (#65022). voice_text = str(voice_item.get("text") or "") if not (voice_item.get("media") or {}) and voice_text: return f"[Voice transcription provided by Weixin]\n{voice_text}" @@ -841,6 +856,11 @@ class WeixinAdapter(BasePlatformAdapter): # Full failure streak: recycle the session. Failed connects through a local proxy (e.g. # Clash) strand sockets the keepalive reaper never sees; on macOS the 256-fd soft limit # then yields EMFILE and a crash. Closing the session tears down every socket. + # Clash on 127.0.0.1:7890) can strand sockets that never return to the connector's + # keepalive pool, so the tight keepalive_timeout never reaps them. On macOS the default + # 256-fd soft limit turns that drip into `[Errno 24] Too many open files` and a gateway + # crash (#79889). Closing the session tears down its connector and every socket it + # holds; a fresh session starts the next attempt from zero fds. await self._recycle_poll_session() async def _recycle_poll_session(self) -> None: @@ -958,6 +978,11 @@ class WeixinAdapter(BasePlatformAdapter): aes_key_b64 = media.get("aes_key") if item_key == "image_item" and payload.get("aeskey"): # image_item may carry a raw hex ``aeskey`` beside the media block aes_key_b64 = base64.b64encode(bytes.fromhex(str(payload.get("aeskey")))).decode("ascii") or aes_key_b64 + # #27300: previously short-circuited when ``voice_item.text`` was set on the assumption that + # Tencent Cloud's STT was good enough. For non-Chinese audio that text is garbage (e.g. a + # Russian message comes back as English phonemes) — we must always download the raw audio so + # ``gateway/run.py``'s central STT pipeline can re-transcribe with the user's configured + # mlx-whisper / whisper.cpp / faster-whisper backend. data = await _download_and_decrypt_media( self._poll_session, cdn_base_url=self._cdn_base_url, encrypted_query_param=media.get("encrypt_query_param"), aes_key_b64=aes_key_b64, full_url=media.get("full_url"), timeout_seconds=timeout_seconds) diff --git a/gateway/platforms/whatsapp_cloud.py b/gateway/platforms/whatsapp_cloud.py index 333fec7727..f8ca5fea4d 100644 --- a/gateway/platforms/whatsapp_cloud.py +++ b/gateway/platforms/whatsapp_cloud.py @@ -259,9 +259,12 @@ class WhatsAppCloudAdapter(WhatsAppBehaviorMixin, BasePlatformAdapter): self._set_fatal_error(code, message, retryable=False) return False # Tighter keepalive so idle CLOSE_WAIT drains promptly. + # Outbound HTTP client. See #18451. from gateway.platforms._http_client_limits import platform_httpx_limits self._http_client = httpx.AsyncClient(timeout=30.0, limits=platform_httpx_limits()) # client_max_size backstops the bounded reader in _handle_webhook. + # Inbound webhook server. client_max_size backstops the bounded reader in _handle_webhook — aiohttp + # enforces the cap on request.read()/post() paths too (#58536/#58902/#59180 pattern). app = web.Application(client_max_size=WEBHOOK_MAX_BODY_BYTES) app.router.add_get(self._health_path, self._handle_health) app.router.add_get(self._webhook_path, self._handle_verify) diff --git a/gateway/platforms/yuanbao.py b/gateway/platforms/yuanbao.py index 3e42248dcb..90f15ee7b6 100644 --- a/gateway/platforms/yuanbao.py +++ b/gateway/platforms/yuanbao.py @@ -84,6 +84,12 @@ MAX_RECONNECT_ATTEMPTS = 100 DEFAULT_SEND_TIMEOUT = 30.0 # WS biz request timeout # Caps the WS close handshake: websockets' own 5s close_timeout waits for a close echo an idle # server never sends, stalling shutdown; a responsive server finishes well under 1s. +# Upper bound on the WS close handshake during teardown (#40383). The websockets connection's own +# close_timeout (5s) blocks until the server echoes the close frame; an idle/unresponsive server never +# replies, stalling gateway shutdown by the full timeout. Bounding the close await here keeps teardown fast +# — a responsive server completes the handshake in well under a second, so this only caps the pathological +# hang. Also bounds the reconnect / connect-failure cleanup paths that reuse _cleanup_ws(), where a graceful +# close is unnecessary anyway (the socket is being discarded to redial). WS_CLOSE_TIMEOUT_S = 1.0 NO_RECONNECT_CLOSE_CODES = {4012, 4013, 4014, 4018, 4019, 4021} # permanent errors — never reconnect HEARTBEAT_TIMEOUT_THRESHOLD = 2 # consecutive missed pongs before reconnect @@ -645,6 +651,11 @@ class RecallGuardMiddleware(InboundMiddleware): logger.warning("[%s] Recall: failed to resolve session: %s", adapter.name, exc) return try: + # Load transcript from canonical store (state.db). Since PR #29278 added a + # ``platform_message_id`` column to the messages table and ``append_to_transcript`` wires the + # incoming dict's ``message_id`` into it, ``load_transcript`` returns rows with ``message_id`` + # set for any message that was observed with one — Branch A1 (exact id match) is the canonical + # path again. transcript = store.load_transcript(sid) except TranscriptReadError as exc: # Not an empty transcript — the rows are unreadable, so recall has @@ -1686,6 +1697,10 @@ class DispatchMiddleware(InboundMiddleware): async def _dispatch_inbound_event() -> None: if any(mt.startswith(("application/", "text/")) for mt in ctx.media_types): + # Classification: DOCUMENT wins over PHOTO/VIDEO/AUDIO for mixed attachments — run.py's + # image handling keys off the per-path image/* mime types regardless of message_type, but + # document-context injection gates strictly on MessageType.DOCUMENT (same precedence as + # Email/Signal, PR #44695). msg_type = MessageType.DOCUMENT else: # yuanbao-local subtypes (CHAT_RECORD) are deep-parsed into text → TEXT downstream msg_type = ctx.msg_type if isinstance(ctx.msg_type, MessageType) else MessageType.TEXT diff --git a/gateway/profile_routing.py b/gateway/profile_routing.py index 7a5f90c863..ae19d6a808 100644 --- a/gateway/profile_routing.py +++ b/gateway/profile_routing.py @@ -93,6 +93,10 @@ def _coerce_route_id(value: Any) -> Optional[str]: PyYAML loads unquoted numeric IDs as ``int`` while ``SessionSource`` fields are ``str``. Only ``int`` (not ``bool``) is coerced; floats stringify to something (``"123.0"``) that can never match, so they get a load-time warning instead. + + ``bool`` is an ``int`` subclass but never a valid id; floats and other types stringify to something + (``"123.0"``) that can never equal an inbound id — recreating the silent no-match this exists to fix — + so they are passed through with a load-time warning instead of being silently "fixed" (#86470). """ if value is None or isinstance(value, str): return value diff --git a/gateway/readiness.py b/gateway/readiness.py index 266f093732..6c0ca67182 100644 --- a/gateway/readiness.py +++ b/gateway/readiness.py @@ -30,6 +30,7 @@ def _probe_state_db(home: Path) -> dict[str, Any]: # writers. ``closing`` is required — sqlite3's context manager only commits/rolls # back, never closes, so a bare ``with connect()`` leaks a connection per poll. with closing(sqlite3.connect(f"file:{path.as_posix()}?mode=ro", uri=True, timeout=1.0)) as conn: + # A readiness probe must never compete with normal state writers. See #69567, #69678. conn.execute("PRAGMA query_only = ON") conn.execute("SELECT name FROM sqlite_master LIMIT 1").fetchone() return _check("ok") diff --git a/gateway/relay/__init__.py b/gateway/relay/__init__.py index d1e032d49a..b49a240787 100644 --- a/gateway/relay/__init__.py +++ b/gateway/relay/__init__.py @@ -169,7 +169,15 @@ def relay_display_name() -> Optional[str]: """The human-facing agent display name forwarded at provision — the connector's multi-agent reply-attribution prefix (``**:** ``). ``GATEWAY_RELAY_DISPLAY_NAME`` env, then the skin's branded agent name (a skin - rename propagates on the next boot). Absent -> connector's linked-owner fallback.""" + rename propagates on the next boot). Absent -> connector's linked-owner fallback. + + The PRIMARY source for the connector's multi-agent reply-attribution prefix (gateway-gateway #171): in a + multi-agent scope the shared bot prepends ``**:** `` to this instance's replies. + Gateway-asserted but safely scoped exactly like ``relay_instance_id()`` / ``relay_wake_url()`` — the + tenant stays token-verified, so a dishonest gateway can only label its OWN instance. Absent -> the + connector stores null and attribution falls back to the instance's linked-owner identity, else skips the + prefix. + """ value = os.environ.get("GATEWAY_RELAY_DISPLAY_NAME", "").strip() if not value: try: diff --git a/gateway/relay/adapter.py b/gateway/relay/adapter.py index a82c3b8ccd..021d723764 100644 --- a/gateway/relay/adapter.py +++ b/gateway/relay/adapter.py @@ -586,6 +586,8 @@ class RelayAdapter(BasePlatformAdapter): adapter's keyword contract, not a card_id. ``fallback_text``/``title`` are accepted for parity but not forwarded (the connector's plan-mode stream renders task chunks; field limits are enforced connector-side). + + See #85476. """ merged_meta = dict(metadata or {}) if reply_to and "thread_ts" not in merged_meta: @@ -636,6 +638,16 @@ class RelayAdapter(BasePlatformAdapter): # durable buffer and replay on re-handshake; routine WS drops are handled by # the transport's own reconnect supervisor. if self._transport is None: + # ``is_reconnect`` is part of the BasePlatformAdapter.connect contract: the gateway's reconnect + # watcher (gateway/run.py) re-establishes a platform after a fatal adapter error by building a + # fresh adapter and calling ``connect(is_reconnect=True)``. Relay MUST accept the kwarg or that + # recovery path raises TypeError and the relay platform can never come back through the watcher. + # The flag exists so adapters with a server-side update queue (e.g. Telegram's Bot API) preserve + # that queue across an outage instead of dropping it (#46621). Routine WS drops are handled + # entirely by the transport's own reconnect supervisor (WebSocketRelayTransport, + # reconnect=True); a watcher-driven reconnect builds a fresh transport from scratch (the + # fatal-error handler disconnect()s the old adapter first, cancelling its supervisor), so there + # is nothing at the adapter layer to preserve. raise RuntimeError("RelayAdapter has no transport configured") self._transport.set_inbound_handler(self._on_inbound) # Interrupts and passthrough-plane forwards (Discord interactions, Twilio, …) @@ -1016,6 +1028,10 @@ class RelayAdapter(BasePlatformAdapter): # read off the wire (engages /sethome's via_relay guard). delivered_via_upstream_relay=True, # Profile routing (multiplex mode), mirroring _event_from_wire. + # The HERMES profile this interaction is routed to (multiplex mode) — mirrors _event_from_wire's + # profile stamping for plain relayed messages (#60586). Without this, a Team-Gateway's Discord + # slash-command/button/modal always fell back to the legacy agent:main namespace even when the + # connector resolved a specific profile for it. profile=getattr(forward, "profile", None), ) event = MessageEvent(text=text, message_type=message_type, source=source) @@ -1327,6 +1343,11 @@ class RelayAdapter(BasePlatformAdapter): triggering ts IS the thread anchor and the final reply's ONLY threading signal (dropping it unconditionally exiled finals to the DM root while progress stayed threaded). Removes an anchor, never adds one. + + It does NOT: * regress real-thread streaming — a real thread carries a distinct ``thread_id`` in + metadata, so the guard leaves ``reply_to`` alone; * regress channel autoThread — a channel/group + top-level reply carries ``thread_id`` (the message's own ts) in metadata when threading is on, so it + is left alone; and a non-DM chat is never matched here. See #18859. """ if reply_to is None: return None @@ -1472,6 +1493,15 @@ class RelayAdapter(BasePlatformAdapter): self, chat_id: str, metadata: Optional[Dict[str, Any]], content: Optional[str], lane: str ) -> None: """One ``typing`` frame (``content`` None = omit; "" = Slack clear). Cosmetic: never raises.""" + # Thread anchor for the status surface. Slack's status line ("is thinking…" in the thread's replies + # footer — works with plain chat:write, confirmed on native no-assistant bots) is THREAD-only: the + # connector's typing case no-ops without a thread_ts. But the typing lane's metadata (base.py + # _thread_metadata_for_source) has no anchor for a top-level DM — source.thread_id is None — so + # every heartbeat was silently dropped. In thread-per-message mode the turn's thread root IS the + # triggering message ts (run.py's synthetic root); synthesize it here from the per-chat inbound + # cache, exactly like native send_typing resolves thread_ts from metadata.message_id. Flat mode + # (reply_in_thread=false) keeps the no-anchor no-op: there is no thread and must not be one + # (#18859). md = self._with_status_thread_anchor(chat_id, metadata) frame: Dict[str, Any] = {"op": "typing", "chat_id": chat_id, "metadata": self._with_scope(chat_id, md)} if content is not None: diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index f3dfeede82..3dfbedae22 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -258,6 +258,10 @@ class PassthroughForward: # None keeps the legacy ``agent:main`` namespace. Without it a relayed Discord # slash-command/button/modal fell back to agent:main even when the equivalent # plain message routed to the right profile. + # Mirrors the ``profile`` field _event_from_wire already carries on the ``inbound`` frame's + # SessionSource (#60586) — the connector stamps it when NAS resolves the target profile for a + # Team-Gateway interaction; absent for a single-profile gateway, where it stays None and session keys + # keep the legacy ``agent:main`` namespace. profile: Optional[str] = None diff --git a/gateway/restart.py b/gateway/restart.py index 9b36120ac3..db149b9665 100644 --- a/gateway/restart.py +++ b/gateway/restart.py @@ -10,6 +10,7 @@ from hermes_cli.config import DEFAULT_CONFIG GATEWAY_SERVICE_RESTART_EXIT_CODE = 75 # EX_CONFIG (sysexits.h): fatal configuration error (token collision, no platforms); # the s6 finish script maps it to exit 125 so the supervisor stops restarting. +# See #51228. GATEWAY_FATAL_CONFIG_EXIT_CODE = 78 # Set by ``hermes gateway run --external-supervisor``. Unlike systemd's INVOCATION_ID @@ -35,6 +36,7 @@ DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT = float(DEFAULT_CONFIG["agent"]["cron_drain_t CRON_DRAIN_CLEANUP_RESERVE_S = 10.0 # systemd TimeoutStopSec headroom after the stop-path drain budget, and the floor when that # budget is still the default immediate (0s) chat drain. Keep in lockstep with generate_systemd_unit(). +# See #94759. SYSTEMD_STOP_HEADROOM_S = 30.0 SYSTEMD_TIMEOUT_STOP_SEC_FLOOR = 60.0 @@ -47,6 +49,11 @@ def is_global_startup_conflict(error_code: str | None) -> bool: Adapters emit ``{scope}_lock`` with ``retryable=True`` so a *mid-run* reconnect can recover; at startup a live foreign holder is a configuration conflict (two gateways cannot poll one token), not a transient blip. Matches by error CODE only, never text. + + ``BasePlatformAdapter._acquire_platform_lock`` emits ``{scope}_lock`` with ``retryable=True`` on + purpose: a *mid-run* reconnect must be able to recover once the live holder exits or a stale record is + cleared (#54167). This matches by error CODE only (the ``{scope}_lock`` / ``lock_conflict`` families + every adapter emits for scoped-lock and identity conflicts), never by message text. """ code = (error_code or "").strip().lower() return bool(code) and (code == "lock_conflict" or code.endswith("_lock")) @@ -97,7 +104,11 @@ def parse_restart_after_turn_timeout(raw: object) -> float: def parse_cron_drain_timeout(raw: object) -> float: - """Parse the cron-only drain floor (``0`` = opt out; cron interrupted on the chat budget).""" + """Parse the cron-only drain floor (``0`` = opt out; cron interrupted on the chat budget). + + ``0`` is a deliberate opt-out — cron work is then interrupted on the same budget as chat work, the + pre-#82161 behaviour — and must not fall through to the default, unlike empty/missing input. + """ return _parse_timeout_keeping_zero(raw, DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT) @@ -131,7 +142,10 @@ def resolve_systemd_timeout_stop_sec( ) -> int: """Seconds systemd ``TimeoutStopSec`` must cover: the stop path may first wait ``cron_drain_timeout`` + ``cleanup_reserve_s`` for cron work, so sizing from the chat drain - alone lets systemd SIGKILL an in-budget drain. A zero cron timeout is an opt-out.""" + alone lets systemd SIGKILL an in-budget drain. A zero cron timeout is an opt-out. + + ``restart_drain_timeout`` is only the chat-turn interrupt budget (default 0). See #94759. + """ drain = _seconds(drain_timeout) cron = _seconds(cron_drain_timeout) cron_budget = (cron + _seconds(cleanup_reserve_s)) if cron > 0.0 else 0.0 diff --git a/gateway/run.py b/gateway/run.py index 0ef86b40dc..0522699357 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -48,6 +48,11 @@ _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT = 30.0 # Telegram connect proves a real getUpdates round trip; must cover polling-start deadlines + readiness. _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT = 180.0 # The initial Telegram connect gates `running` for EVERY platform, so it must not spend the full 180s. +# Cold-start cap for Telegram (#85993): the initial connect awaited before the gateway reaches `running` +# must not spend the full 180s budget — an unreachable Telegram would hold EVERY platform's serving state +# hostage for the whole window. The initial attempt gets one bounded try; on timeout the platform is queued +# for the reconnect watcher, which retries with the full 180s budget (is_reconnect=True preserves the +# offline update queue, #46621). _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT = 45.0 _ADAPTER_DISCONNECT_TIMEOUT_SECS_DEFAULT = 5.0 # End reasons meaning the USER deliberately closed this thread. Shared by _classify_completion_target and @@ -68,6 +73,7 @@ _TELEGRAM_NOISY_STATUS_RE = re.compile( r"|no\s+auxiliary\s+llm\s+provider\s+configured" r"|auto-lowered\s+compression\s+threshold" # the auto-lower notice was reworded to "Auto-lowered this session's threshold..." — cover both. + # See #69332. r"|auto-lowered\s+(?:this\s+)?session'?s?\s+threshold" r"|configured\s+auxiliary\s+compression\s+provider\s+.+\s+unavailable" r"|skipping\s+concurrent\s+compression" @@ -108,6 +114,15 @@ def _hygiene_cooldown_for_failure(gateway, session_key: str, base_cooldown_secon """Bump the hygiene failure streak and return the escalated cooldown (x1/x3/x9 over base, clamped). Hygiene's per-run ``AIAgent`` is fresh, so the streak lives in SQLite keyed by rotation-stable session_key. + + It exists because the in-agent equivalent is unreachable from here: + ``ContextCompressor.record_timeout_failure`` escalates on an absolute 60 -> 300 -> 900s ladder driven by + the in-memory ``_consecutive_timeout_failures`` counter, which ``bind_session_state`` zeroes. Session + hygiene constructs a FRESH ``AIAgent`` per run and re-binds state every time, so from the gateway that + streak is structurally always 0 and only the flat ``hygiene_failure_cooldown_seconds`` could ever be + recorded — a session whose summary model always times out retried on that same fixed interval forever + (#79624). The streak is mirrored to SQLite by rotation-stable ``session_key`` so it outlives both the + per-run agent and gateway restarts; ``PersistentState`` keeps the hot in-process view. """ streak, state = 1, None try: @@ -157,7 +172,15 @@ def hygiene_compaction_recovered( """True when a hygiene run actually recovered the session (extracted to be unit testable). Requires no abort, a real rewrite (the no-op path reuses pre-compression counts) and material shrink per - :func:`compression_made_progress` (a bare ``<`` misses row-count wins and counts estimate noise).""" + :func:`compression_made_progress` (a bare ``<`` misses row-count wins and counts estimate noise). + + * the compressor did not abort (no summary produced at all); * the transcript was actually rewritten — + either rotated into a new session or compacted in place. The degenerate "did not rotate or compact in + place" path (#21301) reuses the pre-compression counts, so relying on the numbers alone would read a + no-op as success; * the request materially shrank, per the canonical :func:`compression_made_progress` + (#39548) — a row-count drop counts even when the summary keeps the token estimate flat, and a sub-5% + token wobble does not count at all. + """ if aborted or not (rotated or in_place): return False return compression_made_progress(msg_count, new_count, approx_tokens, new_tokens) @@ -201,7 +224,13 @@ async def run_codex_hygiene_compaction( The real context is the server-side thread; the local transcript is a never-replayed mirror, so rewriting it shrinks nothing and evicting the live agent starts the next turn on an EMPTY thread. So: compact the LIVE agent via ``thread/compact/start``, keep it cached, never build a detached compressor. ``native``/``off`` - skip without local fallback. Returns ``compacted``, ``skipped:`` or ``failed:``.""" + skip without local fallback. Returns ``compacted``, ``skipped:`` or ``failed:``. + + See #73503. + * Evicting the cached live agent afterwards destroys the only real context: the next turn spawns an + EMPTY thread and the model starts blank while Hermes still mirrors a full history (abrupt amnesia — the + user-facing damage documented on #73503). + """ mode = str(auto_mode or "native").lower() if mode not in {"native", "hermes", "off"}: mode = "native" @@ -255,7 +284,10 @@ def hygiene_wait_should_extend( ) -> bool: """Whether the hygiene host should keep waiting for a slow summary. - A cancelled commit fence cannot commit: extending only queues inbound messages behind a doomed attempt.""" + A cancelled commit fence cannot commit: extending only queues inbound messages behind a doomed attempt. + + Stop extending immediately so the turn can continue. See #96953. + """ return not fence_cancelled and idle < timeout and waited < ceiling @@ -264,6 +296,10 @@ def _record_hygiene_cooldown( """Persist a session-hygiene compression-failure cooldown to the state DB (survives restarts). ``error`` must be forwarded: the recorder writes compression_failure_error UNCONDITIONALLY (NULL clobber). + + Uses the same ``compression_failure_cooldown_until`` column and ``record_compression_failure_cooldown`` + method that the in-conversation compression path (``agent/context_compressor.py``) already uses, so the + cooldown survives gateway restarts (#74136). """ recorder = getattr(_gateway_session_db_inner(gateway), "record_compression_failure_cooldown", None) if recorder is None: @@ -283,6 +319,11 @@ def _status_template_to_regex(template: str) -> str: # ROUTINE compression progress statuses, derived from the SAME template constants the emit sites format. +# Used ONLY by the opt-in ``compression.progress_notices`` gate below (#52995) to decide which of the noisy +# statuses matched by _TELEGRAM_NOISY_STATUS_RE are compression progress (deliverable when the user opted +# in) versus unrelated aux/retry chatter (always suppressed on chat surfaces). Failure notices and manual +# /compress feedback never match _TELEGRAM_NOISY_STATUS_RE in the first place, so they are unaffected by +# this gate. _COMPRESSION_PROGRESS_STATUS_RE = re.compile( "|".join( _status_template_to_regex(_template) @@ -298,7 +339,10 @@ _COMPRESSION_PROGRESS_STATUS_RE = re.compile( def _gateway_compression_progress_notices_enabled() -> bool: """True when ``compression.progress_notices`` is on (default False: chat is silent by design). - Read live (mtime-cached) so a config edit applies at the next status; fail-closed on read error.""" + Read live (mtime-cached) so a config edit applies at the next status; fail-closed on read error. + + Reads ``compression.progress_notices`` from the gateway's raw YAML config (#52995). + """ try: config = _load_gateway_config() compression_cfg = config.get("compression") if isinstance(config, dict) else None @@ -451,7 +495,13 @@ _TRANSIENT_NETWORK_ERROR_CLASS_NAMES = frozenset({ def _is_transient_network_error(exc: BaseException) -> bool: """True for transient network errors safe to log + swallow (the next poll recovers; never crash). - Walks the cause chain so wrapped errors (PTB ``NetworkError`` over ``httpx.ConnectError``) match.""" + Walks the cause chain so wrapped errors (PTB ``NetworkError`` over ``httpx.ConnectError``) match. + + The crash class targeted by #31066 / #31110: an unhandled Telegram ``TimedOut`` (or peer + ``NetworkError`` / ``httpx`` connection error) propagating to the event loop and killing the entire + gateway process. These are by definition transient — the next poll cycle or user action recovers — so + they must never crash the process. + """ seen: set[int] = set() cur: Optional[BaseException] = exc depth = 0 @@ -471,7 +521,11 @@ def _gateway_loop_exception_handler( loop: "asyncio.AbstractEventLoop", context: Dict[str, Any]) -> None: """Loop-level safety net for transient network errors (installed once by ``start_gateway``). - Logs WARNING with traceback; non-transient errors go to the default handler so real bugs surface.""" + Logs WARNING with traceback; non-transient errors go to the default handler so real bugs surface. + + Catches the ``telegram.error.TimedOut`` crash class (issues #31066 / #31110) and any peer transient + network error before it can kill the gateway process. + """ exc = context.get("exception") if exc is not None and _is_transient_network_error(exc): task = context.get("future") or context.get("task") @@ -492,7 +546,13 @@ def _redact_gateway_user_facing_secrets(text: str) -> str: """Secret redaction before text can leave the gateway. Shared ``redact_sensitive_text`` with ``force=True`` (holds even when ``security.redact_secrets`` is off); - ``_GATEWAY_SECRET_PATTERNS`` is a second pass so redaction degrades gracefully if that import fails.""" + ``_GATEWAY_SECRET_PATTERNS`` is a second pass so redaction degrades gracefully if that import fails. + + Delegates to the authoritative ``agent.redact.redact_sensitive_text`` — the same Tirith-grade redactor + already applied to logs, tool output, and approval-command prompts — so the outbound chat path masks the + full credential set the startup banner promises ("chat responses are scrubbed before delivery"), not a + divergent subset. See #23810. + """ redacted = str(text or "") try: from agent.redact import redact_sensitive_text @@ -508,7 +568,14 @@ def _redact_gateway_user_facing_secrets(text: str) -> str: def _redact_approval_command(cmd: "str | None") -> str: """Redact credentials from a command before it goes into an approval prompt. - Else a Tirith-flagged credential echoes verbatim to chat; ``force=True`` holds even with redaction off.""" + Else a Tirith-flagged credential echoes verbatim to chat; ``force=True`` holds even with redaction off. + + Tirith's *findings* are already redacted, but the gateway approval prompt is built from the raw command + string, so a credential-shaped value Tirith flagged would otherwise be echoed verbatim to the chat + platform (#48456). Uses ``redact_sensitive_text(force=True)`` — the same Tirith-grade redactor — so the + prompt honors redaction even when ``security.redact_secrets`` is off. Module-level so the wiring is + unit-testable (the call site is a deeply nested gateway closure that cannot be driven directly). + """ from agent.redact import redact_sensitive_text return redact_sensitive_text(str(cmd or ""), force=True) @@ -584,11 +651,18 @@ def _sanitize_gateway_final_response(platform: Any, text: str) -> str: return text # Lone UTF-16 surrogates make Telegram/Signal ``.encode()`` raise; last defense for legacy/plugin paths. + # Lone UTF-16 surrogates (U+D800–U+DFFF) in model output crash chat surfaces downstream: Telegram's + # ``utf16_len`` length check and Signal formatting both ``.encode()`` the reply and raise + # UnicodeEncodeError before any send (#55143, #55309). The stored-history copy is already sanitized by + # ``build_assistant_message`` and ``finalize_turn`` scrubs the returned ``final_response``, but this + # boundary is the last line of defense for every legacy/plugin delivery path that hands us raw text. + # Raw-text/programmatic surfaces above keep passthrough — their JSON consumers escape surrogates safely. from agent.message_sanitization import _sanitize_surrogates text = _sanitize_surrogates(str(text)) # Cancellation metadata, not prose; ACP/TUI already suppress this sentinel, chat surfaces should too. + # See #7921. if str(text).strip().startswith(INTERRUPT_WAITING_FOR_MODEL_PREFIX): return "" @@ -628,7 +702,10 @@ def render_notice_line(notice) -> str: async def _send_or_update_status_coro(adapter, chat_id, status_key, content, metadata): """Route a status through adapter.send_or_update_status when supported (edits the previous - bubble for the same status_key instead of appending); otherwise fall back to plain send.""" + bubble for the same status_key instead of appending); otherwise fall back to plain send. + + See #30045. + """ sender = getattr(adapter, "send_or_update_status", None) if callable(sender): return await sender(chat_id, status_key, content, metadata=metadata) @@ -693,6 +770,8 @@ def _resolve_progress_thread_id( ``reply_in_thread=False`` (Slack): no synthetic-thread fallback, else the final flat reply inherits a thread. A source.thread_id equal to the event's message id is the adapter's synthetic session key: no thread. + + See #18859. """ platform_key = str(getattr(platform, "value", platform) or "").lower() if not reply_in_thread: @@ -887,7 +966,12 @@ def build_resume_recovery_note( reason: Optional[str], message: str = "", *, interactive: bool = True) -> str: """Build the resume-pending recovery system note for an interrupted turn (empty ``message`` = auto-resume). - Interactive platforms report the restore and ask what next; non-interactive ones finish the work.""" + Interactive platforms report the restore and ask what next; non-interactive ones finish the work. + + On non-interactive event platforms (webhook, API server — adapters with ``interactive_resume = False``) + nobody can answer; the resumed turn must instead complete the interrupted work, or the task is silently + abandoned behind a "restored" acknowledgement that goes nowhere (#57056). + """ reason_phrase = ( "a gateway restart" if reason == "restart_timeout" else "a gateway shutdown" if reason == "shutdown_timeout" else "a gateway interruption") @@ -925,7 +1009,15 @@ def _prepare_resume_pending_message( reason: Optional[str], message: Optional[str], *, interactive: bool = True) -> tuple[str, str]: """Return the recovery message and the user text to persist. - Empty original: persist the note (a "" user row trips the pre-call sanitizer). Real text: persist clean.""" + Empty original: persist the note (a "" user row trips the pre-call sanitizer). Real text: persist clean. + + Resume turns replace the startup event's text with a recovery note before entering the agent. When the + original message is empty (the synthesized auto-resume turn), persist the note too — persisting the + empty string left a blank user row in state.db that the pre-call sanitizer re-healed on every later call + forever (#86580). When the user sent REAL text while the resume was pending, keep persisting their clean + words: the transcript stays scaffold-free (the model still receives the wrapped note), and a non-empty + row never trips the sanitizer. + """ recovery_message = build_resume_recovery_note(reason, message or "", interactive=interactive) persist_message = message if isinstance(message, str) and message.strip() else recovery_message return recovery_message, persist_message @@ -933,6 +1025,19 @@ def _prepare_resume_pending_message( # Assistant fields that must survive replay for CLI parity (reasoning continuity, prefix-cache hits, provider # echo): unreconstructable thinking text (DeepSeek/Kimi), opaque signatures, Codex blobs (caching degrades). +# ``reasoning`` and ``reasoning_details`` were the original three preserved by PR #2974 (schema v6). +# ``reasoning_content``, ``codex_reasoning_items``, ``codex_message_items``, and ``finish_reason`` were +# added to the DB later but the gateway's replay whitelist was never expanded to match — so any pure-text +# assistant turn (no ``tool_calls``) silently dropped them on replay, regressing the CLI-vs-gateway +# behavioural parity. Why each field matters on replay: ``_copy_reasoning_content_for_api`` promotes +# ``reasoning`` → ``reasoning_content`` at send time, but only when the strings happen to match. Carrying +# the original ``reasoning_content`` verbatim avoids reconstruction loss for providers that return them as +# distinct fields (DeepSeek/Kimi/Moonshot thinking modes). * ``reasoning_details``: opaque structured array +# (signature, encrypted_content) used by OpenRouter/Anthropic to maintain reasoning continuity across turns. +# * ``codex_reasoning_items``: encrypted reasoning blobs for the OpenAI Codex Responses API. * +# ``codex_message_items``: exact assistant message items with ``phase``. OpenAI docs: "preserve and resend +# phase on all assistant messages — dropping it can degrade performance." Required for prefix cache hits. * +# ``finish_reason``: informational; cheap to keep so transcripts replay identically across CLI and gateway. _ASSISTANT_REPLAY_FIELDS: tuple[str, ...] = ( "reasoning", "reasoning_content", "reasoning_details", "codex_reasoning_items", "codex_message_items", "finish_reason") @@ -944,7 +1049,15 @@ def _build_replay_entry( """Build a replay entry for a non-tool-calling message, preserving ``_ASSISTANT_REPLAY_FIELDS``. ``preserve_timestamp``: only user rows need it (stale-dangerous-confirmation stripper). Falsy fields are - dropped EXCEPT ``reasoning_content``: DeepSeek/Kimi treat "" as a sentinel; dropping it can 400.""" + dropped EXCEPT ``reasoning_content``: DeepSeek/Kimi treat "" as a sentinel; dropping it can 400. + + Empty values: most fields are dropped when falsy (matching the original PR #2974 behaviour) since an + empty list/string for those carries no information. The exception is ``reasoning_content``: + DeepSeek/Kimi thinking-mode replay treats an empty string as a meaningful sentinel that + ``_copy_reasoning_content_for_api`` upgrades to a single space. Dropping it here would make the gateway + send no ``reasoning_content`` at all on the next turn, which can cause HTTP 400 from strict thinking + providers. + """ entry: Dict[str, Any] = {"role": role, "content": content} # api_content sidecar keeps the request prefix byte-stable — ONLY if this pipeline did not rewrite content. _sidecar = msg.get("api_content") @@ -999,6 +1112,7 @@ def _slack_ignored_channels_from_gateway_config(config: Any) -> set[str]: raw = getattr(platform_cfg, "extra", {}).get("ignored_channels") if raw is None: # Top-level ``slack.ignored_channels`` arrives via the plugin's YAML→env bridge, not PlatformConfig.extra. + # See #46925. raw = os.getenv("SLACK_IGNORED_CHANNELS") or None return _csv_or_list_to_set(raw) @@ -1077,6 +1191,10 @@ def _build_gateway_agent_history( agent_history = strip_interrupted_tool_tails(agent_history) # Strip a dangling assistant(tool_calls) tail (SIGKILL-mid-tool-call); else the model re-issues it forever. + # Strip a dangling assistant(tool_calls) tail with no tool answers — the signature of a SIGKILL + # mid-tool-call (e.g. the tool itself ran `docker restart`/`kill` and took the gateway down before the + # result was persisted). Without this the model re-issues the unanswered call on resume and loops the + # restart forever (#49201). agent_history = strip_dangling_tool_call_tail(agent_history) # Strip expired dangerous-confirmation phrases; replayed, a follow-up could read as a fresh confirmation. @@ -1091,7 +1209,14 @@ def _select_cached_agent_history( """Prefer the cached live transcript only when it is longer AND has a real, non-ephemeral unpersisted row. Guards FTS write-corruption amnesia (stale reload while the cached agent holds unpersisted rows). Length - alone is not enough: a longer all-durable list can be an expected replay-filtering delta.""" + alone is not enough: a longer all-durable list can be an expected replay-filtering delta. + + Guards the FTS write-corruption case (#50502): when message writes fail silently through corrupt FTS + triggers, the next turn reloads a stale/empty ``conversation_history`` from disk even though the same + cached ``AIAgent`` still holds unpersisted real rows in ``_session_messages``. Replacing those rows with + the shorter persisted copy causes immediate same-session amnesia. Length alone does not trigger + retention. + """ if isinstance(live_history, list) and len(live_history) > len(persisted_history): from run_agent import _is_ephemeral_scaffolding @@ -1197,7 +1322,16 @@ def _collect_auto_append_media_tags( Producer allowlist: docs/logs/search results contain example MEDIA: strings that must never become attachments. If mid-run compression shrank the list below the history length the slice is - untrustworthy, so scan every message (dedup via history_media_paths).""" + untrustworthy, so scan every message (dedup via history_media_paths). + + 1. Producer-tool allowlist: only tools that intentionally emit deliverable artifacts (TTS) are eligible. + (Fixes the original report behind #16721.) 2. Current-turn isolation: only messages produced this turn + are scanned, so a tool result from an earlier turn (still present in the full message list) cannot leak + onto a later text-only reply (#34608). + When that happens the slice boundary is no longer trustworthy, so fall back to scanning every message + and rely on ``history_media_paths`` for dedup, preserving the compression-safe behaviour of #160. The + producer-tool allowlist still applies on the fallback path. + """ history_media_paths = history_media_paths or set() new_messages = (messages[history_offset:] if history_offset and len(messages) >= history_offset else messages) @@ -1242,7 +1376,11 @@ def _collect_auto_append_media_tags( def _collect_history_media_paths(agent_history: List[Dict[str, Any]]) -> set: - """Dedup set of media paths already delivered (JSON-payload and assistant-message shapes alike).""" + """Dedup set of media paths already delivered (JSON-payload and assistant-message shapes alike). + + Missing the JSON-payload shape caused #46627; missing the assistant-message shape caused repeated + delivery when the model echoed a previous MEDIA tag. + """ paths: set = set() tool_name_by_call_id = _tool_name_by_call_id(agent_history) @@ -1432,7 +1570,12 @@ def _multiplex_profile_homes(config: object) -> list[tuple[str, "Path"]]: def _enable_multiplex_log_routing(config: object) -> bool: """Route agent.log/errors.log/gateway.log records to their owning profile (inert single-profile). ``setup_logging(mode="gateway")`` binds file handlers to the launch home, so under multiplexing - every secondary profile's records would land in the default profile's logs.""" + every secondary profile's records would land in the default profile's logs. + + Swap the static handlers for the profile routers from #99440 — the same primitive the Desktop cron + ticker uses — once the served-profile set is known. Inert for single-profile gateways + (``enable_profile_log_routing`` is a no-op below two homes). + """ if not getattr(config, "multiplex_profiles", False): return False try: @@ -1493,6 +1636,7 @@ def _terminal_scope_cwd(default: str = "") -> str: def _load_profile_secret_scope(profile_home: "Path") -> dict: """Hydrate and load one profile's secrets under its home override.""" from hermes_constants import set_hermes_home_override, reset_hermes_home_override + # Caller already hydrated external sources off-loop (#99519). from agent.secret_scope import build_profile_secret_scope from hermes_cli.env_loader import hydrate_profile_secret_sources @@ -1525,6 +1669,8 @@ def _profile_runtime_scope( secrets = build_profile_secret_scope(Path(profile_home)) secret_token = set_secret_scope(secrets) # Install the routed profile's COMPLETE terminal policy, never ambient TERMINAL_* a prior turn set. + # Without it terminal_tool reads the process-global TERMINAL_* vars a previous profile's turn may have + # pinned (first-writer-wins backend leak; #68559). from tools.terminal_scope import install_and_reset_profile_terminal_scope with install_and_reset_profile_terminal_scope(Path(profile_home)): @@ -1547,7 +1693,10 @@ def load_gateway_config_for_runner() -> "GatewayConfig": """Load gateway config for the process-level GatewayRunner. Multiplexed: reload under the default profile's ``_profile_runtime_scope`` so platform tokens in its ``.env`` resolve via the secret scope; unscoped ``_getenv`` falls to ``os.environ``, which often lacks a token living only under - ``profiles//.env``. Off -> identical to ``load_gateway_config()``.""" + ``profiles//.env``. Off -> identical to ``load_gateway_config()``. + + See #64674. + """ cfg = load_gateway_config() if not getattr(cfg, "multiplex_profiles", False): return cfg @@ -1566,7 +1715,12 @@ def load_gateway_config_for_runner() -> "GatewayConfig": async def _discover_gateway_mcp_tools(config: object) -> None: """Run startup MCP discovery for every profile this gateway serves: ``discover_mcp_tools`` reads ``mcp_servers`` from ``get_hermes_home()``'s config, so an unscoped call only connects the launch - profile's servers (single-profile gateways keep the unscoped call).""" + profile's servers (single-profile gateways keep the unscoped call). + + Under multiplex, run it once per served profile inside that profile's ``_profile_runtime_scope`` and + carry the scope into the executor thread with ``copy_context()`` (the same shape as + ``_run_in_executor_with_context``). See #95518. + """ from tools.mcp_tool import discover_mcp_tools loop = asyncio.get_running_loop() if not getattr(config, "multiplex_profiles", False): @@ -1592,6 +1746,13 @@ def _platform_has_bot_credential(platform: "Platform", platform_config: "Platfor return True # Matrix also authenticates by password; a token-only check would evict a reconnectable config from # the retry queue. Read ONLY extra (build_config() copies env there): env fallback = every config OK. + # Those credentials land in ``extra`` rather than ``.token``, so a token-only check reads a perfectly + # reconnectable password-auth config as credential-less and evicts it from the retry queue on the first + # transient failure — after which it stays down until the gateway is restarted by hand. Mirror the + # adapter's own gate: homeserver + user_id + password. Read ONLY from extra, never os.getenv: + # build_config() already copies all three env vars onto extra, and importing this module loads + # ~/.hermes/.env, so an env fallback would report "has credential" for every Matrix config on the box — + # including the empty-primary multiplex case (#64674) this check exists to evict. if platform is not Platform.MATRIX: return False extra = getattr(platform_config, "extra", None) or {} @@ -1736,6 +1897,10 @@ def _bridge_config_to_env(_cfg: dict) -> None: _auxiliary_cfg = _cfg.get("auxiliary", {}) if _auxiliary_cfg and isinstance(_auxiliary_cfg, dict): _bridge_auxiliary_config_to_env(_auxiliary_cfg) + # config.yaml is the documented, authoritative source for these settings — it unconditionally wins over + # .env values. Previously the guards below read `if X not in os.environ` and let stale .env entries + # (e.g. HERMES_MAX_ITERATIONS=60 written by an old `hermes setup` run) silently shadow the user's + # current config. See PR #18413 / the 60-vs-500 max_turns incident. _agent_cfg = _cfg.get("agent", {}) _bridge_max_turns_to_env(_agent_cfg) _bridge_section_to_env(_agent_cfg, _AGENT_ENV_BRIDGE) @@ -1846,6 +2011,13 @@ from gateway.config import ( ChannelOverride, Platform, GatewayConfig, PlatformConfig, _getenv, load_gateway_config) from gateway.session import ( AsyncSessionStore, SessionStore, SessionSource, SessionContext, build_session_key) +# Telegram topic routing (#22773, regression fixed #52060): a +# ``telegram::`` cron target is ambiguous — a forum-style topic in a +# private chat and a genuine Bot API channel Direct-Messages topic share the same shape and need OPPOSITE +# routing. Disambiguate at delivery time via ``_is_channel_dm_topic`` (see its docstring for the full +# rationale); ``thread_id`` goes in ``route_metadata`` so the anchorless cron send bypasses the +# DeliveryRouter's private-chat reply-anchor requirement. Compute the routed metadata ONCE so both the text +# send (via DeliveryRouter) and the media send agree. from gateway.delivery import ( DeliveryRouter, resolve_delivery_transport, # noqa: F401 (re-exported: run_* mixins + tests resolve gateway.run.) @@ -1950,6 +2122,11 @@ _AGENT_PENDING_SENTINEL = object() # _running_agents*/_active_session_leases/_busy_ack_ts/_turn_lease_tokens (_release_running_agent_state + # dispatch finally); _session_run_generation (monotonic; clearing breaks stale-run detection); # _agent_cache (_evict_cached_agent); approval/slash-confirm (_clear_session_boundary_security_state). +# The state itself now lives in ``SessionState.conversation`` (see gateway/session_state.py) and boundaries +# clear it structurally via ``ConversationState.clear()`` — adding a field to ConversationState means every +# boundary picks it up automatically. History: boundaries used to each carry a hand-copied pop-list that +# drifted whenever a new dict was added (#48031, #58403, #10702, #35809). - _agent_cache: has its own +# eviction path (_evict_cached_agent) with resource cleanup; boundaries call it explicitly. _CONVERSATION_SCOPED_STATE: tuple = ( "_session_model_overrides", "_pending_one_turn_model_restores", @@ -1959,6 +2136,7 @@ _CONVERSATION_SCOPED_STATE: tuple = ( "_last_resolved_model", "_queued_events", # Stall-watchdog "already notified" latch; cleared on /new so a fresh conversation can warn again. + # See #72016. "_session_stall_notified", # Sidecar notes staged but never consumed (turn aborted before run_sync) must not leak into a # future conversation's first user message — session keys are source-derived and REUSED. @@ -1978,6 +2156,8 @@ def _resolve_runtime_agent_kwargs() -> dict: runtime = resolve_runtime_provider() except AuthError as auth_exc: # Rate-limit cap vs real auth failure: both use the fallback chain; the log must not mislabel. + # Distinguish a transient rate-limit/quota cap (credentials are fine, re-auth cannot help) from a + # genuine auth failure (expired/revoked token). See #32790. if is_rate_limited_auth_error(auth_exc): logger.warning("Primary provider rate-limited (429): %s — trying fallback", auth_exc) else: @@ -2156,6 +2336,9 @@ def _try_resolve_fallback_provider() -> dict | None: explicit_api_key=resolve_entry_api_key(entry)) # Log the config `provider`, not the runtime category (Ollama would log "openrouter"). logger.info( + # Log the literal `provider` key from config, not the resolved runtime category — an + # Ollama fallback resolves through the OpenAI-compatible path and would otherwise be + # logged as "openrouter", contradicting the operator's config (#32790). "Fallback provider resolved: %s model=%s", entry.get("provider") or runtime.get("provider"), entry.get("model")) return {**_runtime_agent_kwargs(runtime), "model": entry.get("model")} @@ -2441,7 +2624,11 @@ def _is_control_interrupt_message(message: Optional[str]) -> bool: def _strip_response_attachments_for_direct_send(response: str, adapter) -> str: """Return the visible text portion of a response before direct send(). Only explicit ``MEDIA:`` attachments are stripped; bare paths/URLs stay visible. No broad regex after - ``extract_media()``: it deliberately preserves protected code spans and unvalidated tags.""" + ``extract_media()``: it deliberately preserves protected code spans and unvalidated tags. + + Queued follow-up resends only replay explicit ``MEDIA:`` attachments in this path. Keep bare local paths + and ordinary image URLs visible because the post-stream uploader intentionally ignores them (#20834). + """ _, cleaned = adapter.extract_media(response) return cleaned.replace("[[audio_as_voice]]", "").replace("[[as_document]]", "").strip() @@ -2577,6 +2764,9 @@ def _load_gateway_config(config_path: "Path | None" = None) -> dict: # Canonicalize model-id aliases (model.name/model.model → model.default) and migrate stale root # provider/base_url: the gateway bypasses load_config(), else ``model: {name: }`` is empty. try: + # The gateway bypasses load_config() (it reads raw YAML for speed), so the normalization that + # load_config() applies must be replayed here or the gateway would resolve an empty model for + # ``model: {name: }`` configs while the CLI resolves it correctly. See issue #34500. Fail-open. from hermes_cli.config import _normalize_root_model_keys raw = _normalize_root_model_keys(raw) except Exception: @@ -2777,7 +2967,15 @@ def _normalize_empty_agent_response( agent_result: dict, response: str, *, history_len: int = 0) -> str: """Normalize empty/None agent responses into user-facing messages. Covers ``failed``, work done (api_calls > 0) with no text, and never-ran (api_calls == 0, the - post-/stop silent-drop from a stale generation token) with a retry hint.""" + post-/stop silent-drop from a stale generation token) with a retry hint. + + Consolidates the existing ``failed`` handler and adds a catch-all for the case where the agent did work + (api_calls > 0) but returned no text. Fix for #18765. + Also surfaces a retry hint when the agent never ran at all (api_calls == 0) for a non-interrupted, + non-failed turn -- this is the silent-drop pattern observed after ``/stop`` where the next user message + hits a stale generation token and returns an empty result, leaving the platform with nothing to send. + (#31884) + """ if response: return response if agent_result.get("failed"): @@ -2810,12 +3008,26 @@ def _normalize_empty_agent_response( if agent_result.get("interrupted"): # Interrupted with api_calls > 0 = deliberately stopped/steered; silence is intentional (queued # messages arrive via the recursive drain). ZERO api_calls = never processed (stale /stop flag). + # An interrupted run that did work (api_calls > 0) is the drain of a run the user deliberately + # stopped or steered — its silence is intentional, and any queued/interrupting message is delivered + # by the recursive drain inside _run_agent before this result is seen. An interrupted run with ZERO + # api_calls never processed the user's message at all: it was killed at the top of the tool loop by + # an interrupt flag left over from a recent /stop (#44212). Pure silence there swallows a real user + # message, so surface it. + # api_calls == 0, not failed, not interrupted: the agent never ran for this turn. This is the + # post-/stop generation-race pattern where the gateway would otherwise silently drop the turn + # (response=0 chars) and the user sees no reply at all. Surface a short retry hint so the message + # isn't lost in silence. (#31884) if api_calls == 0: return ( "⚠️ Your message was interrupted before processing started " "(likely by a recent /stop). Please send it again.") return response if api_calls > 0: + # Hidden-reasoning-only retry exhaustion: the loop's sentinel text ("Codex response remained + # incomplete after 3 continuation attempts") doubles as final_response, so it would be delivered + # verbatim into the channel — where peer agents can ingest it as a completed assistant turn + # (#51628). Blank it here so the normal empty-response handling (and the suppression below) applies. if _is_gateway_hidden_reasoning_incomplete_turn(agent_result): return "" if agent_result.get("partial"): @@ -2878,7 +3090,18 @@ def _preserve_queued_followup_history_offset( async def _dispose_unused_adapter(adapter: "BasePlatformAdapter | None") -> None: """Best-effort dispose for an adapter that never made it onto ``self.adapters`` (may be ``None``). Nothing else calls ``disconnect()`` on it, so ``__init__`` resources (e.g. SQLite fds) would leak - until GC (not prompt for asyncio-bound objects) and exhaust the fd ulimit over a long retry loop.""" + until GC (not prompt for asyncio-bound objects) and exhaust the fd ulimit over a long retry loop. + + The reconnect watcher in ``GatewayRunner._platform_reconnect_watcher`` constructs a fresh adapter on + every retry attempt. When the connect call fails — for any of the three reasons (non-retryable error, + retryable error, exception during connect) — the adapter is dropped without ever being installed, so + nothing else will call its ``disconnect()``. ``APIServerAdapter`` opens a SQLite ``ResponseStore`` that + holds 2 fds — the db file and its WAL sidecar) stay open until garbage collection sweeps the unreachable + object, which Python's cyclic GC does not do promptly for asyncio-bound objects with native handles. The + cumulative leak is 2 fds × every retry at the 300s backoff cap ≈ 12 fds/hour, and the default 2560-fd + ulimit is exhausted in ~12h of continuous failure, after which every open() call on the gateway raises + ``OSError: [Errno 24] Too many open files`` and the gateway becomes a zombie (#37011). + """ if adapter is None: return try: @@ -3100,6 +3323,8 @@ class GatewayRunner( return [(key, state.turn.agent) for key, state in self._sessions_map().items() if state.turn.agent is not None] # Loop-liveness / watchdog handles; class-level defaults so partially constructed test runners work. + # Class-level defaults so partial construction in tests doesn't blow up on access; the real values are + # set in __init__ / start() / stop(). See #66892, #69089. _loop_heartbeat_task: Optional["asyncio.Task"] = None _loop_floor_timer_handle: Optional[Any] = None _loop_liveness_watchdog: Optional[Any] = None @@ -3112,6 +3337,7 @@ class GatewayRunner( global _gateway_runner_ref # With multiplex_profiles on, load under the default profile secret scope so bot tokens in its # .env resolve as secondary profiles' do; explicit config= injection (tests) is left untouched. + # See #64674. self.config = config if config is not None else load_gateway_config_for_runner() # Multiplexer flag flips agent.secret_scope.get_secret() to fail-closed on unscoped credential # reads, so a missed migration crashes loudly instead of leaking a cross-profile value. @@ -3123,6 +3349,7 @@ class GatewayRunner( self.adapters: Dict[Platform, BasePlatformAdapter] = {} # Non-None means SessionDB init failed — the gateway broadcasts a one-time warning to the home # channel(s) after connecting so the user learns persistence is broken before /resume fails. + # See #88235. self._session_db_init_error: Optional[str] = None # Non-default profiles' adapters by profile then Platform; self.adapters stays the default's map. self._profile_adapters: Dict[str, Dict[Platform, BasePlatformAdapter]] = {} @@ -3205,6 +3432,27 @@ class GatewayRunner( # to one session_id (switch_session's many-to-one mapping), which routing-key guards cannot see. self._turn_leases = SessionTurnLeaseRegistry() # Stall-notified keys clear when pending clears / activity resumes / conversation boundary. + # Tokens for held turn leases, keyed by (routing key, run generation) so release is granted per-turn + # and a stale unwind can never free a newer turn's lease (#28686 ownership lesson). Held turn-lease + # tokens live on SessionState.turn.lease_token / .lease_generation (the old dict was keyed (routing + # key, generation) so a stale unwind could never free a newer turn's lease — the generation field + # preserves that ownership check, #28686). Runner-level queued interrupt text lives on + # SessionState.persistent.pending_command_text (NOTE: distinct from the adapter-level + # _pending_messages Dict[str, MessageEvent] in gateway/platforms/base.py, which shares the legacy + # name). Last successfully-resolved (non-empty) model, keyed by session. Used as a fallback when a + # fresh config read transiently returns an empty model (e.g. an mtime-keyed config-cache miss during + # a post-interrupt recovery turn). Without this, the agent is built with model="" and every API call + # fails HTTP 400 "No models provided" — the session goes silent until the user manually re-sends. + # See #35314. The ``"*"`` session entry holds a process-wide last-known-good for sessions seen for + # the first time. Lives on SessionState.conversation.last_resolved_model. Overflow buffer for + # explicit /queue commands. The adapter-level _pending_messages dict is a single slot per session + # (designed for "next-turn" follow-ups where repeated sends collapse into one event). /queue has + # different semantics: each invocation must produce its own full agent turn, in FIFO order, with no + # merging. When the slot is occupied, additional /queue items land here and are promoted + # one-at-a-time after each run's drain. Cleared on /new and /reset. /model and other mid-session + # operations preserve the queue. Lives on SessionState.conversation.queued_events; native image + # paths, busy-ack debounce timestamps and the monotonic run-generation counter (#28686, NEVER reset) + # live on SessionState too. See gateway.session_stall. self._session_stall_notified: Dict[str, bool] = {} # Startup restore gate: while restart-interrupted sessions auto-resume, real inbound messages # queue instead of competing with the synthetic resume turns; drained after all resume tasks end. @@ -3226,6 +3474,7 @@ class GatewayRunner( self._completion_delivery_retention = 2048 # Agent-triggered terminal completions from one conversation often land in the same scheduler # tick; hold them briefly so the agent gets one synthetic turn instead of one per process. + # See #70300. self._completion_notification_batches: dict[tuple[str, ...], list[tuple[str, dict, asyncio.Future]]] = {} self._completion_notification_batch_tasks: dict[tuple[str, ...], asyncio.Task] = {} self._completion_notification_batch_flush_tasks: set[asyncio.Task] = set() @@ -3265,6 +3514,9 @@ class GatewayRunner( # on unattended gateways — surface it so operators knowingly enable one. try: from hermes_cli.config import load_config as _load_full_config + # Startup heads-up (#30882): a gateway in manual approval mode with no automated risk assessor + # (tirith disabled AND no auxiliary.approval model) can only gate dangerous commands / + # execute_code scripts via live in-chat approval. _appr_cfg = _load_full_config() _appr_mode = str( cfg_get(_appr_cfg, "approvals", "mode", default="manual") or "manual" @@ -3286,6 +3538,10 @@ class GatewayRunner( """Open the session DB for the active scope and run opportunistic state.db / checkpoint maintenance.""" # Session DB is a property caching one AsyncSessionDB per path (a handle bound here would pin the # root home under multiplex); priming here keeps startup diagnostics at init. + # Initialize session database for session_search tool support. Same frozen-handle class of bug as + # SessionStore._db (#88532): a handle bound here is pinned to the process's root home, but /resume, + # /title, /history and session search all run inside _profile_runtime_scope on a multiplexed gateway + # and must see that profile's own state.db. self._session_db_pinned: Any = _SESSION_DB_UNPINNED self._session_db_handles: Dict[Path, Any] = {} self._session_db_handles_lock = threading.Lock() @@ -3301,6 +3557,10 @@ class GatewayRunner( # Opportunistic state.db maintenance (prune + optional VACUUM), at most once per min_interval_hours. # A few blocking seconds per day is fine for a long-lived gateway; failures log, never raise. + # Surface the failure to the user via their home channel(s) once the gateway connects. Without this, + # state.db corruption or NFS/SMB lock failures silently degrade the entire gateway — messages may + # flow but nothing is persisted, and the user has no indication until they try /resume and find + # nothing (#88235). if self._session_db is not None: try: from hermes_cli.config import load_config as _load_full_config @@ -3354,6 +3614,7 @@ class GatewayRunner( self._background_tasks: set = set() # Event-loop liveness heartbeat: rewritten every 30s while the loop dispatches; supervisors use # the file mtime / updated_at to tell "process alive" from "loop frozen". + # See #66892. self._gateway_started_at: float = time.time() self._loop_heartbeat_task: Optional[asyncio.Task] = None self._loop_floor_timer_handle = self._loop_liveness_watchdog = None @@ -3368,7 +3629,20 @@ class GatewayRunner( def _open_session_db_for_active_scope(self, raise_on_error: bool = False) -> Any: """AsyncSessionDB for the active profile scope, resolved per access (not in ``__init__``) since ``SessionDB()`` reads the context-local HERMES_HOME; one handle cached per path. Construction - failure enters bounded backoff; ``raise_on_error=True`` (priming) propagates it.""" + failure enters bounded backoff; ``raise_on_error=True`` (priming) propagates it. + + Same per-path cache as ``SessionStore._open_session_db_for_active_scope`` (#88532): ``SessionDB()`` + resolves ``_default_db_path()`` at call time through the context-local HERMES_HOME override + installed by ``_profile_runtime_scope``, so resolving per access — instead of once in ``__init__`` — + is what lets /resume, /title, /history and session search on a multiplexed gateway read the *serving + profile's* store rather than the root one. + One ``AsyncSessionDB`` is cached per resolved path, so the wrapper identity is stable per profile + (callers compare and stash it) and two profiles never share a handle. A construction failure enters + bounded backoff; one caller retries after the deadline while concurrent callers continue to see the + unavailable fallback. ``raise_on_error=True`` (construction-time priming) propagates the failure + after recording that recoverable state so ``__init__`` can record ``_session_db_init_error`` for the + #88235 broadcast. + """ from hermes_state import AsyncSessionDB, _default_db_path, get_shared_session_db from gateway.session_db_recovery import RecoverableHandleCache path = Path(_default_db_path()) @@ -3382,6 +3656,12 @@ class GatewayRunner( def _open(): # Borrow the SessionStore's handle (same path) so state.db doesn't get two writers/pools. # The store owns/sweeps it at shutdown; this cache holds only the async wrapper (close_all). + # Both caches resolve the SAME ``_default_db_path()``, so the process was holding two writer + # connections and two read pools against one state.db — the fd budget doubled for nothing, and + # doubled again per profile on a multiplexed gateway (#98573). A borrowed wrapper cannot go + # stale in practice: the store's cache only drops handles in close_all_db_handles() (shutdown), + # and while the store's own open is failing there is nothing to borrow, so nothing is cached + # here either. store = getattr(self, "session_store", None) borrowed = getattr(store, "_db", None) if store is not None else None if borrowed is not None: @@ -3420,13 +3700,18 @@ class GatewayRunner( """Close every per-profile AsyncSessionDB this runner opened. Drained under the lock, closed outside it; a pinned handle is the pinner's to close. Wrappers - BORROWED from ``session_store`` are skipped: the store's sweep (runs first) closes them.""" + BORROWED from ``session_store`` are skipped: the store's sweep (runs first) closes them. + + See #98573. + """ def _close(db) -> None: if getattr(db, "__dict__", {}).get("_hermes_borrowed_handle"): return inner = getattr(db, "_db", db) if inner is None or not hasattr(inner, "close"): return + # Shared instances no-op on close() (the registry owns the lifecycle). Release the refcount + # instead (#90837). from hermes_state import release_or_close try: release_or_close(inner) @@ -3529,7 +3814,16 @@ class GatewayRunner( def _normalize_source_for_session_key(self, source: SessionSource) -> SessionSource: """Apply Telegram DM topic recovery to a source for session-key purposes. Always derive override storage keys from the result: ``_handle_message_with_agent`` rewrites ``thread_id`` before - deriving the session key, so keys from the raw ``event.source`` are never read next turn.""" + deriving the session key, so keys from the raw ``event.source`` are never read next turn. + + ``_handle_message_with_agent`` rewrites ``source.thread_id`` via + ``_recover_telegram_topic_thread_id`` *before* deriving the session key for a normal message turn (a + lobby/stripped reply gets pinned to the user's last-active topic). Session-scoped command handlers + like ``/model`` and ``/reasoning`` derive their override key from the raw inbound ``event.source``, + which skips that recovery — so the override is stored under a different key than the next message + turn reads, and the override is silently dropped on Telegram forum topics and after compression + session splits (#30479). + """ try: recovered = self._recover_telegram_topic_thread_id(source) except Exception: @@ -3721,6 +4015,14 @@ class GatewayRunner( if getattr(source, "platform", None) == Platform.SLACK: # Per-turn egress identity: Slack chat.startStream needs recipient_user_id/team_id; the relay # adapter's _with_scope fallback reads per-chat caches a CONCURRENT turn overwrites. + # Slack's chat.startStream requires recipient_user_id (+ recipient_team_id) when streaming to a + # channel, and the relay connector fills those from metadata.user_id / metadata.scope_id. The + # relay adapter's _with_scope fallback resolves BOTH from per-chat caches keyed only by chat_id + # — mutable state that a CONCURRENT turn overwrites: two users with overlapping turns in one + # channel would open U1's stream with U2 as the recipient. Stamp the authentic per-turn values + # from THIS turn's source here, where they are still turn-scoped; _with_scope only fills keys + # that are absent, so the cache degrades to what it should be — a restart/synthetic-send + # fallback. See #210. team_id = getattr(source, "scope_id", None) user_id = getattr(source, "user_id", None) if team_id or user_id: @@ -3732,6 +4034,7 @@ class GatewayRunner( metadata.setdefault("user_id", str(user_id)) # Routed profile for shared state.db namespaces: under profile_routes the transport adapter's # stamp is not the profile that wrote the binding (Telegram prune path needs it). + # See #76423. profile = str(getattr(source, "profile", None) or "").strip() if profile and metadata is not None: metadata = dict(metadata) @@ -3874,6 +4177,7 @@ class GatewayRunner( # (section, key) config values baked into the agent at construction: a change MUST invalidate the # cached agent or a mid-gateway edit is silently ignored. Add new baked-in settings here. + # _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816) _CACHE_BUSTING_CONFIG_KEYS: tuple = ( ("model", "context_length"), ("model", "max_tokens"), ("compression", "enabled"), ("compression", "progress_notices"), ("compression", "threshold"), @@ -3902,7 +4206,13 @@ class GatewayRunner( """Reset per-turn state on a cached agent before a new turn starts. The activity ts/desc/provenance triple resets together and only at depth 0 — else a session idle 29 min trips the watchdog before the first call; interrupt-recursive turns keep it so stuck-turn - idle time accumulates to the 30-min timeout.""" + idle time accumulates to the 30-min timeout. + + ``_last_activity_ts``, ``_last_activity_desc``, and ``_last_activity_provenance`` are only reset for + fresh external turns (depth 0); they are a semantic triple - description and provenance describe the + activity *at* ts, so updating one without the others would make get_activity_summary() misleading. + See #15654, #9051. + """ if interrupt_depth == 0: from agent.session_activity import ActivityProvenance agent._last_activity_ts = time.time() @@ -3910,6 +4220,7 @@ class GatewayRunner( agent._last_activity_provenance = ActivityProvenance.UNKNOWN # Reset the SessionDB flush cursor so the new turn's messages are fully persisted — a stale # value from the previous turn makes `_flush_messages_to_session_db` skip new rows. + # See #44327. if hasattr(agent, "_last_flushed_db_idx"): agent._last_flushed_db_idx = 0 agent._api_call_count = 0 @@ -4028,7 +4339,14 @@ def _run_planned_stop_watcher( poll_interval: float = 0.5) -> None: """Poll for the planned-stop marker and trigger graceful shutdown (Windows lacks ``add_signal_handler``, so ``hermes gateway stop`` would never drain). Runs everywhere; on POSIX - the signal handler consumes the marker first and ``_running``/``_draining`` guard re-triggers.""" + the signal handler consumes the marker first and ``_running``/``_draining`` guard re-triggers. + + On Windows, ``asyncio.add_signal_handler`` raises NotImplementedError for SIGTERM/SIGINT, so the + standard signal-driven shutdown path never runs when ``hermes gateway stop`` signals the gateway. The + consequence is that the drain loop is skipped — in-flight agent sessions are killed mid-turn and + ``resume_pending`` is never set, so the next gateway boot has no idea those sessions need to be + auto-resumed (issue #33778, v0.13.0 session-resume feature broken on native Windows). + """ from gateway.status import ( _get_planned_stop_marker_path, planned_stop_marker_targets_self) marker_path = _get_planned_stop_marker_path() @@ -4040,6 +4358,16 @@ def _run_planned_stop_watcher( and getattr(runner, "_running", False)): # A marker may target a PREVIOUS instance that exited before stop() cleaned up; # firing on it means an "UNKNOWN" exit and a watchdog crash-loop; probe unlinks stale. + # A marker existing is NOT sufficient — it may have been written for a PREVIOUS gateway + # instance (different PID) and left behind because that process exited before the CLI's + # stop() could clean it up. Firing the handler on a stale/foreign marker drives the gateway + # into shutdown, then consume_planned_stop_marker_for_self() correctly reports a PID + # mismatch — but by then we're already stopping, so it's logged as an unexpected "UNKNOWN" + # exit and the watchdog crash-loops the gateway (issue #34597, a regression from PR #33798 + # which added this watcher without the PID check). Only fire when the marker actually + # targets us. The probe is non-destructive on a match (the handler does the authoritative + # consume on the loop thread) and self-heals by unlinking stale/malformed markers so they + # cannot wedge a freshly booted gateway. if not planned_stop_marker_targets_self(): stop_event.wait(poll_interval) continue @@ -4148,6 +4476,9 @@ def _housekeeping_auto_archive() -> None: def _housekeeping_deferred_fts_retry() -> None: """A SessionDB opened while another process held the rebuild lock fails closed onto the LIKE fallback and the gateway stays up for days. Non-blocking, rate-limited inside SessionDB; no-op when not stale.""" + # Retry here, on the existing tick, against the shared instances this process already holds: + # non-blocking admission, no new thread, rate-limited inside SessionDB. No-op when nothing is stale (one + # attribute read per instance). See #100108. from hermes_state_registry import live_shared_session_dbs for _sdb in live_shared_session_dbs(): _retry = getattr(_sdb, "retry_deferred_fts_recovery", None) @@ -4253,7 +4584,10 @@ async def _await_thread_exit( thread: Optional[threading.Thread], timeout: float, poll: float = 0.1) -> bool: """Wait for a daemon thread to exit WITHOUT blocking the event loop; True if it exited in time. A synchronous ``join()`` freezes the loop — fatal for the cron ticker, whose in-flight delivery is a - coroutine on *this* loop: it could never run, so the join timed out and the message dropped.""" + coroutine on *this* loop: it could never run, so the join timed out and the message dropped. + + See #58818. + """ if thread is None: return True deadline = asyncio.get_running_loop().time() + max(0.0, timeout) @@ -4266,7 +4600,10 @@ async def _shutdown_mcp_servers_nonblocking(timeout: float = 5.0) -> bool: """Close MCP servers off-loop with a bounded wait; True when done within ``timeout``. ``shutdown_mcp_servers()`` can block ~15s; on the loop thread short-grace supervisors (s6 3s) SIGKILL us before ``mark_exited()`` runs, so every later boot reports a phantom unclean death. - On timeout shutdown proceeds and the daemon thread is left to finish or die.""" + On timeout shutdown proceeds and the daemon thread is left to finish or die. + + See #82874. + """ def _do() -> None: try: from tools.mcp_tool import shutdown_mcp_servers @@ -4302,12 +4639,30 @@ def _gateway_stderr_formatter() -> logging.Formatter: return RedactingFormatter("%(asctime)s %(levelname)s %(name)s: %(message)s") +# ownership guard inserted below (PR #93084) def _replace_target_belongs_to_other_profile(existing_pid: int) -> bool: """Return True when ``--replace`` must refuse to signal ``existing_pid``. A poisoned/stale PID record can point at another profile's LIVE gateway (cross-profile SIGTERM restart loop). Ownership is decided by the persisted identity record ALONE, bound to the live target by exact PID + start-time; live argv can never PROVE ownership (no HERMES_HOME), it is only a consistency check. Missing, legacy, conflicting or unprovable identity → refuse (fail closed).""" + # On Windows there is no systemd/launchd service query at all (_get_service_pids() returns an empty + # set), so a gateway supervised by a Scheduled Task / Startup VBS looks like an unsupervised orphan to + # the process scan (#86098). The same holds on every platform for a healthy gateway launched standalone + # (no service registration) whose PID the runtime record can see (#83683). Exempt the recorded healthy + # gateway PID and its parent chain: a recorded, liveness-verified gateway is by definition not an orphan + # "the pidfile/runtime record can't see", and the Scheduled-Task bootstrap's argv (``gateway run``) + # matches the gateway scan — killing that bootstrap takes the detached gateway it spawned down with it. + # Exclusion evidence comes from the RAW registration record, not the liveness-validated probe. + # ``get_running_pid`` (any flags) returns None whenever a record fails validation — start-time mismatch + # after PID-reuse checks, argv drift, lock hiccups — which is exactly when a healthy standalone gateway + # (no service supervisor — e.g. `hermes gateway run` on Windows) is at risk: its PID never joins the + # exclusion set and the sweep hard-kills it. On Windows SIGTERM is TerminateProcess, so the gateway's + # planned-stop watcher never gets a chance to drain. Reading the raw pidfile + lock records (no + # validation, no unlink side effects) is strictly safer for a KILL exclusion list: a stale recorded PID + # at worst spares one process this sweep, while a validation false-negative would kill a live gateway. + # The validated probe is still consulted for the runtime-status fallback PID it can surface when no + # pidfile exists. try: from gateway.status import ( _get_pid_path, _get_process_hermes_home, _get_process_start_time, _pid_from_record, @@ -4628,9 +4983,18 @@ async def _start_gateway_start_control_socket(runner): import atexit _control_server = None try: + # Started immediately after the PID-file claim: winning that O_EXCL race is the moment this process + # becomes the authoritative gateway for its HERMES_HOME, so from here on "does a socket answer?" is + # a truthful liveness/identity query for updater and fleet consumers. Strictly non-fatal: a bind + # failure only means consumers fall back to the process-scan/state-file layer, exactly as before + # this feature. See #92091. from gateway.control_socket import GatewayControlServer # pause-for-update: the updater asks us to drain + exit (freeing venv handles) vs. a tree-kill # (same path as SIGUSR1). Handler runs on the socket executor thread, so marshal onto the loop. + # pause-for-update (#92091 step 2): the updater asks this gateway to drain in-flight turns and exit + # cleanly — releasing every venv file handle — instead of being tree-killed mid-turn. Same drain + # path as SIGUSR1/service restarts (request_restart(via_service=True)); the updater (or the service + # manager) relaunches after the code swap. _main_loop = asyncio.get_running_loop() def _pause_for_update_handler() -> dict: @@ -4765,6 +5129,12 @@ async def _start_gateway_shutdown_tail( return False # Never join(): an in-flight cron delivery is a coroutine on THIS loop; a sync join would drop it. + # Stop cron scheduler + housekeeping cleanly. These MUST be awaited cooperatively, not join()ed. A cron + # delivery in flight when the gateway restarts is a coroutine scheduled onto THIS event loop + # (safe_schedule_threadsafe); the ticker thread is blocked on its future.result(). A synchronous + # cron_thread.join() would block the loop, so that delivery could never run — it timed out and the + # message was silently dropped (#58818). Awaiting keeps the loop alive so the in-flight delivery + # finishes before we tear down. cron_stop.set() _stop_cron_provider(cron_provider) if not await _await_thread_exit(cron_thread, timeout=_CRON_SHUTDOWN_DRAIN_TIMEOUT): @@ -4822,6 +5192,7 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = runner = GatewayRunner(config) # Multiplex: swap the launch-home file handlers for per-profile routers so each profile's records # land in its own logs/. Must run after the runner resolved (possibly None) config and setup_logging. + # See #82936. _enable_multiplex_log_routing(runner.config) # ``--replace`` is explicit startup authority, not a durable reconnect policy: GatewayRunner scopes # it to cold adapter connects and clears it before the background reconnect watcher starts. @@ -4839,6 +5210,12 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = loop = asyncio.get_running_loop() # Swallow transient network errors from background tasks; one unhandled httpx error would kill us. + # Issues #31066 / #31110: an unhandled ``telegram.error.TimedOut`` (or peer NetworkError / httpx + # connection error) in any awaited coroutine would propagate to the loop and kill the gateway process, + # taking down every profile attached to the same runner. systemd then restarts the service after ~5s but + # the active conversation turn is lost. The fix is intentionally narrow: only well-known transient + # network errors are swallowed (and logged with full traceback so the originating call site is still + # discoverable). Anything else is forwarded to the default handler so real bugs still surface. loop.set_exception_handler(_gateway_loop_exception_handler) if threading.current_thread() is threading.main_thread(): @@ -4854,6 +5231,14 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = # Windows has no add_signal_handler, so `hermes gateway stop`'s SIGTERM would never drain; poll the # planned-stop marker (written BEFORE the kill) instead. Runs everywhere so masked-SIGTERM drains. + # Windows fallback: asyncio.add_signal_handler raises NotImplementedError on Windows, so `hermes gateway + # stop`'s SIGTERM (which Python maps to TerminateProcess on Windows) never invokes + # shutdown_signal_handler. That means the drain loop never runs, mark_resume_pending never fires, and + # sessions are silently lost across restarts (issue #33778). The fix is a marker-polling thread: `hermes + # gateway stop` writes the planned-stop marker BEFORE killing, and this thread notices it and drives the + # same shutdown path the signal handler would have. Runs on every platform (cheap, defensive) so + # non-signal-bearing environments (Windows native, sandboxed CI runners that mask SIGTERM) still get a + # clean drain. _planned_stop_watcher_stop = threading.Event() _planned_stop_watcher_thread = threading.Thread( target=_run_planned_stop_watcher, @@ -4884,6 +5269,10 @@ async def start_gateway(config: Optional[GatewayConfig] = None, replace: bool = # discover_mcp_tools() blocks up to 120s; on the loop thread it would freeze platform heartbeats. try: + # MCP tool discovery — run in an executor so the asyncio event loop stays responsive even when a + # configured MCP server is slow or unreachable. discover_mcp_tools() uses a blocking 120s wait + # internally; calling it from the loop thread would freeze platform heartbeats (Discord shard, + # Telegram polling) until it returned. See #16856. await _discover_gateway_mcp_tools(runner.config) except Exception as e: logger.debug("MCP tool discovery failed: %s", e) @@ -4997,6 +5386,16 @@ def main(): # start_gateway() completes teardown before returning/raising SystemExit; force-exit after so a # wedged non-daemon worker can't block Py_FinalizeEx's join. SystemExit caught so EVERY path exits. try: + # start_gateway() performs the full graceful teardown (adapters disconnected, sessions saved + + # flushed, SQLite closed, cron/MCP stopped, PID file + runtime lock released) before it returns OR + # raises SystemExit with an explicit code. Force-exit afterwards so a wedged non-daemon worker + # thread (e.g. a ThreadPoolExecutor tool/LLM call blocked with no timeout) cannot block interpreter + # finalization (Py_FinalizeEx joins all non-daemon threads, incl. concurrent.futures' _python_exit) + # and strand the gateway half-shut down with the supervisor unable to restart it (#53107). + # SystemExit is caught explicitly: start_gateway raises it on the clean-fatal-config (#51228), + # planned-restart, and service-restart paths, all of which complete teardown first. Routing those + # codes through the same os._exit backstop means EVERY exit path is wedge-proof, not just the + # boolean-return ones. success = asyncio.run(start_gateway(config)) exit_code = 0 if success else 1 except SystemExit as e: @@ -5009,7 +5408,20 @@ def _exit_after_graceful_shutdown(exit_code: int) -> None: """Flush stdio, release the PID file + runtime lock, then hard-exit. ``os._exit`` (not ``sys.exit``): SystemExit runs ``Py_FinalizeEx``, which joins every non-daemon thread — exactly the hang a wedged worker causes. It bypasses ``atexit``, so PID/lock release and the - bounded log drain (file handlers sit behind a ``QueueListener`` thread) are done here explicitly.""" + bounded log drain (file handlers sit behind a ``QueueListener`` thread) are done here explicitly. + + Graceful teardown is already complete by the time this runs, so there is nothing left that needs a clean + interpreter shutdown. See #53107. + ``os._exit`` bypasses ``atexit`` handlers, so we cannot rely on the ``atexit``-registered + ``remove_pid_file`` / ``release_gateway_runtime_lock`` (registered in ``start_gateway``) to run. The + full-shutdown path releases both explicitly in ``_stop_impl``, but the EARLY exit paths — + clean-fatal-config (#51228) and startup-aborted-before-running — raise ``SystemExit`` right after + ``runner.start()`` without going through ``_stop_impl``, so on those paths ``atexit`` was the only thing + releasing them. Now that those paths are routed through this backstop (#53107), release both here + explicitly. Both calls are idempotent — ``remove_pid_file`` only unlinks a PID file that belongs to this + process, and ``release_gateway_runtime_lock`` no-ops when the lock is already released — so this is a + no-op on the normal shutdown path and the actual cleanup on the early-exit paths. + """ for stream in (sys.stdout, sys.stderr): with suppress(Exception): stream.flush() diff --git a/gateway/run_adapters.py b/gateway/run_adapters.py index 2e619d83d0..cb59cb8919 100644 --- a/gateway/run_adapters.py +++ b/gateway/run_adapters.py @@ -77,7 +77,11 @@ class GatewayAdapterLifecycleMixin: async def _bounded_adapter_teardown(self, adapter, platform, *, profile: Optional[str] = None) -> None: """Tear down one adapter on the shutdown path with bounded awaits (never raises). Unbounded, a half-dead transport stalls past systemd's ``TimeoutStopSec``; the SIGKILL skips ``atexit`` - PID-file cleanup and the next start dies with "PID file race lost".""" + PID-file cleanup and the next start dies with "PID file race lost". + + Both ``cancel_background_tasks()`` and ``disconnect()`` can block indefinitely when a platform's + network state is half-dead (e.g. a wedged Feishu/Lark WebSocket thread waiting on I/O). See #14128. + """ timeout = self._adapter_disconnect_timeout_secs() suffix = f" (profile: {profile})" if profile else "" started_at = time.monotonic() @@ -123,7 +127,12 @@ class GatewayAdapterLifecycleMixin: def _platform_connect_timeout_secs(self, platform=None, *, initial: bool = False) -> float: """Per-platform connect timeout. Telegram's full 180s is NOT spent at cold start (it would - hold the gateway out of ``running``); the watcher retries with the full budget.""" + hold the gateway out of ``running``); the watcher retries with the full budget. + + ``initial=True`` marks the cold-start connect awaited before the gateway reaches ``running``. The + cold-start wait is capped and the platform is handed to the reconnect watcher, which retries with + the full budget (and ``is_reconnect=True``, preserving the offline update queue — #46621). + """ from gateway.run import ( _PLATFORM_CONNECT_TIMEOUT_SECS_DEFAULT, _TELEGRAM_CONNECT_TIMEOUT_SECS_DEFAULT, _TELEGRAM_INITIAL_CONNECT_TIMEOUT_SECS_DEFAULT, @@ -139,7 +148,15 @@ class GatewayAdapterLifecycleMixin: self, adapter, platform, *, is_reconnect: bool = False, initial: bool = False ) -> bool: """Connect with a bound so one platform can't block others. ``is_reconnect``: cold boot - drops the stale server-side queue, a reconnect keeps it. ``initial``: capped budget.""" + drops the stale server-side queue, a reconnect keeps it. ``initial``: capped budget. + + ``is_reconnect`` is forwarded to ``adapter.connect()`` so platform adapters can distinguish a cold + first boot (drop any stale server-side queue) from a watcher reconnect after a prolonged outage + (preserve the queue so messages sent during the outage are delivered rather than silently dropped — + #46621). + ``initial`` selects the capped cold-start budget for platforms whose full connect budget is too long + to spend before the gateway reaches ``running`` (#85993 — Telegram's 180s). + """ timeout = self._platform_connect_timeout_secs(platform, initial=initial) if timeout <= 0: return await adapter.connect(is_reconnect=is_reconnect) @@ -192,6 +209,8 @@ class GatewayAdapterLifecycleMixin: """Queue a retryable fatal adapter for background reconnection (True when newly queued). Must not await: callers run this BEFORE any disconnect so a wedged close can't strand it. + + Idempotent if already queued. See #80598. """ if not adapter.fatal_error_retryable: return False @@ -201,12 +220,24 @@ class GatewayAdapterLifecycleMixin: if adapter.platform in self._failed_platforms: # Already queued is exactly when the watcher may have died (supervision gave up); without # this backstop nothing retries and the stranded check treats "queued" as safe. + # Nothing to enqueue -- but "already queued" is precisely the state in which the watcher has had + # time to die, and the enqueue branch below holds the ONLY call to + # _ensure_reconnect_watcher_running(). _spawn_supervised auto-restarts the watcher after a crash + # (#71758), but only _MAX_SUPERVISED_RESTARTS times in rapid succession; past that it logs + # "giving up restarts" and the watcher stays dead forever. _ensure_reconnect_watcher_running is + # the documented backstop for exactly that budget exhaustion (#70344) -- and it was unreachable + # for a platform already in the queue, which is the only kind of platform the watcher can have + # been retrying long enough to exhaust it on. The result is a silent permanent outage: nothing + # retries, and the stranded check in _handle_adapter_fatal_error_detached deliberately treats a + # queued platform as safe, so the process never restarts either (#90386). self._ensure_reconnect_watcher_running() return False self._failed_platforms[adapter.platform] = self._reconnect_queue_entry( adapter.platform, adapter, platform_config, attempts=0, delay=0.0, ) logger.info("%s queued for background reconnection", adapter.platform.value) + # Ensure the reconnect watcher is alive — if it died (e.g. from exhausting its restart budget), + # respawn it so queued platforms are not permanently stranded (#70344). self._ensure_reconnect_watcher_running() return True @@ -217,6 +248,9 @@ class GatewayAdapterLifecycleMixin: # Outer hard deadline: the stranded check in ``finally`` only runs when we return. timeout = self._adapter_disconnect_timeout_secs() if timeout <= 0: + # Outer hard deadline (#80598): even with queue-before-disconnect, a hang anywhere in the + # impl (status write side effects, detach races, etc.) must not leave this task wedged + # forever — the stranded check in ``finally`` only runs when we return. await self._handle_adapter_fatal_error_impl(adapter) else: # Disconnect budget + proportional bookkeeping overhead (tests shrink the timeout). @@ -228,6 +262,10 @@ class GatewayAdapterLifecycleMixin: "Fatal-error handling for %s timed out after %.1fs; " "ensuring reconnect queue is populated", adapter.platform.value, outer, ) + # Best-effort queue before re-raising: a cancelled fatal handler must not strand a + # retryable platform (#80598). + # Best-effort queue so an unexpected raise mid-handler cannot leave a retryable platform + # permanently deaf (#80598). self._queue_retryable_fatal_platform(adapter) except asyncio.CancelledError: # A cancelled or raising fatal handler must not strand a retryable platform. @@ -291,6 +329,11 @@ class GatewayAdapterLifecycleMixin: self._queue_retryable_fatal_platform(adapter) if existing is adapter: # Bounded by the shutdown-path timeout so this always returns to the stranded check. + # Queue retryable failures BEFORE any disconnect await (#80598). A half-dead transport can wedge + # native close() (or swallow CancelledError inside it) so the previous "disconnect then queue" + # order left platforms permanently deaf inside a live process even after the network recovered. + # Populate the queue first so the reconnect watcher always has work; teardown is best-effort + # after. await self._safe_adapter_disconnect(adapter, adapter.platform) if not self.adapters and not self._failed_platforms: self._exit_reason = adapter.fatal_error_message or "All messaging adapters disconnected" @@ -342,7 +385,16 @@ class GatewayAdapterLifecycleMixin: failures (counter resets after ``_SUPERVISED_HEALTHY_SECS`` healthy). Fresh ``Context`` per spawn (an inherited delegated-child marker would make the Kanban dispatcher reject its own writes). ``on_spawn`` fires on EVERY spawn incl. respawns — handle trackers MUST pass it or a - respawn leaves a stale handle and a SECOND watcher; ``on_give_up(name)`` fires at budget end.""" + respawn leaves a stale handle and a SECOND watcher; ``on_give_up(name)`` fires at budget end. + + ``on_give_up`` (optional) is invoked with ``name`` when supervision is abandoned — the restart + budget is spent and this task will never be respawned by the supervisor again. Supervision being + finite is correct; having no owner of the invariant afterwards is not. A task that still has queued + work depending on it needs somewhere to hand that fact to, and before this hook existed the only + thing standing between budget exhaustion and a permanent silent outage was a *later, unrelated + event* happening to call ``_ensure_...`` (#90386). This is the supervisor telling its caller "I am + done; the invariant is yours now", which is a thing only the supervisor knows. + """ # Spawn timestamp lets ``_done`` tell a rapid crash-loop from a healthy-run-then-crash. _started = time.monotonic() # No create_task kwargs (test doubles mock a narrow signature); Context().run isolates instead. @@ -446,6 +498,15 @@ class GatewayAdapterLifecycleMixin: continue # INVARIANT (do not weaken): created inside _profile_runtime_scope but RUNS after it # exits; it sees the profile scope only because ensure_future copies the Context. + # Positional, not keyword: the watcher's existing unit tests bind a stand-in + # ``_process_handoff(row)`` with no second parameter, and a keyword call would TypeError + # into the failure branch — turning a passing suite into a silent no-op watcher. Arity is + # probed above. It still sees the profile's home and secret scope only because + # ``set_hermes_home_override`` and ``set_secret_scope`` are ContextVar-based — ensure_future + # copies the current Context into the Task. If either seam is ever migrated to a + # thread-local or module global, secondary- profile handoffs silently regress to + # primary-config delivery (the exact bug fixed in #91217) while still recording + # handoff_state='completed'. inflight[session_id] = asyncio.ensure_future( _dispatch(row, session_id, session_db, profile_name) ) @@ -482,7 +543,16 @@ class GatewayAdapterLifecycleMixin: def _on_reconnect_watcher_gave_up(self, name: str = "") -> None: """Own the reconnect invariant once supervision gives up: while running with queued platforms, a watcher is live or a bounded respawn is scheduled (no later event can notice - a dead watcher). Slow-tier exhaustion logs loudly; deliberately NOT a process restart.""" + a dead watcher). Slow-tier exhaustion logs loudly; deliberately NOT a process restart. + + Before this, the only thing that noticed a dead watcher was a *later fatal error from some other + platform* reaching ``_queue_retryable_fatal_platform``. That is event-coupled recovery: it needs an + event that, by construction, may never come. #81036 moved queue publication ahead of disconnect and + drops the failed adapter from the live map, so once the watcher's budget is spent there may be no + adapter left that can emit the event recovery was waiting on. The platform stays queued, nothing + retries it, and the stranded check in ``_handle_adapter_fatal_error_detached`` treats a queued + platform as safe — so the process is never restarted either. + """ if not getattr(self, "_running", False): return if getattr(self, "_failed_platforms", None): @@ -535,7 +605,15 @@ class GatewayAdapterLifecycleMixin: def _ensure_reconnect_watcher_running(self) -> None: """Respawn a dead reconnect watcher (called on BOTH _queue_retryable_fatal_platform paths: - the re-fatal of an already-queued platform is the only case that exhausts the budget).""" + the re-fatal of an already-queued platform is the only case that exhausts the budget). + + If the tracked reconnect watcher task has died (e.g. from exhausting its restart budget, or a + terminal exception that _spawn_supervised could not recover), respawns it so platforms queued for + reconnection are not permanently stranded. Called from _queue_retryable_fatal_platform on BOTH paths + (#70344, #90386): after a new enqueue, and after a re-fatal for a platform that is already queued -- + the latter being the only case in which the watcher can have been retrying long enough to exhaust + its supervised restart budget. + """ task = getattr(self, "_reconnect_watcher_task", None) if not getattr(self, "_running", False) or (task is not None and not task.done()): return # not running, or already alive @@ -624,6 +702,7 @@ class GatewayAdapterLifecycleMixin: attempt = info["attempts"] + 1 # Empty-token primary configs can never reconnect; drop them so multiplex setups # where a secondary profile owns the bot do not spin forever. + # See #64674. if not _platform_has_bot_credential(platform, platform_config): self._drop_from_reconnect_queue(platform, "no bot credential on queued config") return @@ -646,6 +725,11 @@ class GatewayAdapterLifecycleMixin: platform.value, adapter.fatal_error_message, ) # Never installed on self.adapters: dispose here or its __init__ resources leak ~2 fds each. + # The adapter is about to be dropped from the queue without ever being installed on + # self.adapters, so nothing else will call disconnect() on it. We must dispose it here, + # otherwise the resource owners it constructed in __init__ (ResponseStore for + # APIServerAdapter, etc.) leak 2 fds each. The gateway hits the 2560-fd limit after ~12h of + # failed reconnects at the 300s backoff cap (#37011). await _dispose_unused_adapter(adapter) del self._failed_platforms[platform] else: @@ -655,6 +739,10 @@ class GatewayAdapterLifecycleMixin: adapter.fatal_error_message or "failed to reconnect", ) logger.info("Reconnect %s failed, next retry in %ds", platform.value, backoff) + # Same fd-leak concern as the non-retryable branch above: the adapter failed to connect and + # is being thrown away. Without an explicit dispose call, the resources it opened in + # __init__ stay open until the next GC pass — and aiohttp/SQLite handles don't get GC'd + # promptly, so 2 fds/retry leak at 300s backoff cap = ~12 fds/hour (#37011). await _dispose_unused_adapter(adapter) except Exception as e: if adapter is not None: @@ -931,6 +1019,7 @@ class GatewayAdapterLifecycleMixin: continue profile_map[platform] = adapter # Restore persisted /voice state for this bot (primary startup and reconnects do too). + # See #84872. self._sync_voice_mode_state_to_adapter(adapter) for claim in (credential_claim, listener_claim): if claim is not None: @@ -971,6 +1060,8 @@ class GatewayAdapterLifecycleMixin: _set_owner = getattr(adapter, "set_owner_profile", None) if callable(_set_owner): _set_owner(profile_name) + # Voice transcripts from this bot's channels dispatch through THIS adapter (primary wiring lives at + # connect time; see #75198). text_modes = getattr(self, "_busy_text_modes_by_profile", None) self._wire_adapter_handlers( adapter, @@ -988,6 +1079,7 @@ class GatewayAdapterLifecycleMixin: # Voice transcripts from this bot's channels dispatch through THIS adapter. self._bind_voice_input_callback(adapter) # Secondary adapters carry their profile so prune paths namespace topic bindings correctly. + # See #76423. adapter._hermes_profile_name = profile_name async def _secondary_reconnect_attempt(self, profile_name: str, platform: Platform): @@ -1007,6 +1099,8 @@ class GatewayAdapterLifecycleMixin: if profile_config is None or not profile_config.enabled: return None, None # Startup credential gate mirror: a removed credential must not rebuild. + # Mirrors the startup credential gate (#84079): a credential removed from this profile's scope + # must not rebuild an adapter that would fan out turns. if not _platform_has_bot_credential(platform, profile_config): logger.info( "Secondary %s reconnect skipped: no bot credential (profile: %s)", @@ -1097,6 +1191,8 @@ class GatewayAdapterLifecycleMixin: if is_global_startup_conflict(getattr(adapter, "fatal_error_code", None)): # A live foreign token holder is an ownership conflict, not a blip: park it fatal. logger.error( + # Park it fatal (like ``duplicate_credential``) instead of retry-storming the token every + # backoff (#83183). "[MULTIPLEX] Profile '%s': %s credential is held by another " "gateway (%s) — parked, not retried. %s", profile_name, platform.value, adapter.fatal_error_code, adapter.fatal_error_message or "", @@ -1368,7 +1464,11 @@ class GatewayAdapterLifecycleMixin: """Platform-bound auth callback for adapters (prompt-injection mitigation for fetched context); delegates to :meth:`_is_user_authorized`. ``profile_name`` binds a secondary to its scope; for the shared primary (None) the routed profile is stamped so its pairing store - is consulted while allowlist reads stay under the transport home.""" + is consulted while allowlist reads stay under the transport home. + + Without this an inline-button caller approved only in the routed profile's pairing store was denied + (#86296), because the adapter's callback source was never route-stamped. + """ from gateway.run import get_hermes_home transport_home = Path(get_hermes_home()) if self._multiplex_on() and profile_name is None else None diff --git a/gateway/run_agent_cache.py b/gateway/run_agent_cache.py index 742fa0dd9b..a7845d8eba 100644 --- a/gateway/run_agent_cache.py +++ b/gateway/run_agent_cache.py @@ -103,7 +103,18 @@ class GatewayAgentCacheMixin: ) -> str: """Stable key from agent config: change → cached AIAgent rebuilt; unchanged → reused (frozen prompt + schemas for cache hits). ``user_id`` / ``user_id_alt`` participate because Honcho - freezes them at init; omitting them in shared-thread keys would cross-attribute messages.""" + freezes them at init; omitting them in shared-thread keys would cross-attribute messages. + + ``user_id`` and ``user_id_alt`` are the runtime user identities carried by the current message's + gateway source. They participate in the cache key because the Honcho memory provider freezes them + into ``HonchoSessionManager`` at first-message init (see + ``plugins/memory/honcho/__init__.py::_do_session_init``). Without them in the signature, a + shared-thread session_key (one in which ``build_session_key`` intentionally omits the participant + ID, e.g. ``thread_sessions_per_user=False``) would reuse the cached AIAgent across distinct users, + causing the second user's messages to be attributed to the first user's resolved Honcho peer. This + broke #27371's per-user-peer contract in multi-user gateways. Per-user agent rebuilds in shared + threads trade prompt-cache warmth for correct memory attribution. + """ import hashlib, json as _j # Fingerprint the FULL credential, not a short prefix: OAuth/JWT-style tokens often share a # common prefix (e.g. "eyJhbGci"), so a prefix would give false cache hits across auth switches. @@ -287,7 +298,14 @@ class GatewayAgentCacheMixin: compression-exhausted reset). New conversation-scoped dicts go in _CONVERSATION_SCOPED_STATE so every boundary picks them up. Turn-scoped state (_running_agents/_ts, slot leases, turn- lease tokens) is owned by _release_running_agent_state and NOT cleared. Idle agent-cache - eviction is NOT a boundary (a resumed turn rebuilds from these). getattr-guarded.""" + eviction is NOT a boundary (a resumed turn rebuilds from these). getattr-guarded. + + Why a funnel: these boundaries used to each carry a hand-copied pop-list of the per-session dicts, + and the lists drifted every time a new dict was added (#48031, #58403, #10702, #35809 were all + "boundary X forgot dict Y" bugs — e.g. /new cleared the /model override but not the /model --once + restore snapshot). Adding a new conversation-scoped dict now means adding its attribute name to + _CONVERSATION_SCOPED_STATE below; every boundary picks it up automatically. + """ from gateway.run import _CONVERSATION_SCOPED_STATE if not session_key: return @@ -333,6 +351,7 @@ class GatewayAgentCacheMixin: if not session_key: return 0 persistent = self._session_state(session_key).persistent + # Monotonic by design (#28686): incremented here, NEVER reset. persistent.run_generation = int(persistent.run_generation) + 1 return persistent.run_generation @@ -409,13 +428,18 @@ class GatewayAgentCacheMixin: # so on a hung/still-draining run the flag survives and silently kills the session's NEXT # message (interrupted=True, api_calls=0, empty response). Like /new and /model, the next # message rebuilds from history; the old agent keeps its flag so a hung drain still dies. + # See #44212. self._evict_cached_agent(session_key) async def _refresh_agent_cache_message_count(self, session_key: str, session_id: Optional[str]) -> None: """Re-baseline a cached agent's stored message_count after THIS turn — the coherence guard rebuilds on mismatch, so without this every turn would rebuild and destroy prompt caching. Only the count is refreshed, only if the same agent is still cached. DB errors leave the - snapshot as-is (one spare rebuild).""" + snapshot as-is (one spare rebuild). + + But the snapshot is taken at agent-BUILD time — before this turn writes its own user + assistant (+ + tool) rows — and the cache entry is never rewritten on a reuse. See #45966. + """ from gateway.run import _AGENT_PENDING_SENTINEL _cache_lock = getattr(self, "_agent_cache_lock", None) _cache = getattr(self, "_agent_cache", None) @@ -542,7 +566,14 @@ class GatewayAgentCacheMixin: holds reference cycles; without it RSS grows across /new). Soft = frees clients and child subagents but PRESERVES terminal sandbox / browser / bg processes since the session may resume; true boundaries call ``_cleanup_agent_resources`` first. Cleanup runs on a daemon - thread so ``_agent_cache_lock`` never spans slow socket teardown.""" + thread so ``_agent_cache_lock`` never spans slow socket teardown. + + Pops the entry AND soft-releases the evicted agent's LLM client pool so the httpx connection + (sockets + held buffers) is freed promptly rather than waiting on CPython GC — AIAgent holds + reference cycles (callbacks, tool state) that delay refcount collection, so a manual release is + required to keep gateway RSS flat across many /new, /model, undo and reset operations (#29298, same + leak class as #25315). + """ from gateway.run import _AGENT_PENDING_SENTINEL # Prompt-stability state rides the agent-cache lifecycle: a fresh agent must re-render its # session-context bytes (the pin) and re-see the current voice-channel state once. @@ -671,6 +702,11 @@ class GatewayAgentCacheMixin: throttles. Above the anonymous-RSS budget this soft-evicts LRU agents (transcript rebuilt from the persisted session next turn). Never touched: agents mid-turn, the most recently used sessions, and transcripts not yet on disk. + + A gateway serving many chats therefore holds every warm transcript indefinitely: agents that took a + turn within the TTL are never idle-swept, and the sweep additionally defers finalizable sessions + until they expire. RSS climbs until the cgroup throttles and SIGTERM can no longer flush inside + systemd's stop timeout (#80764). """ from gateway.run import _AGENT_PENDING_SENTINEL from gateway.agent_cache_pressure import ( diff --git a/gateway/run_busy.py b/gateway/run_busy.py index f200e640e4..0604a67a39 100644 --- a/gateway/run_busy.py +++ b/gateway/run_busy.py @@ -83,6 +83,8 @@ class GatewayBusySessionMixin: a NON-busy session the oldest orphan runs as THIS turn, the next is staged into the slot so arrival order holds, and the caller enqueues the incoming event behind it. The returned event is REMOVED from both stores, else the post-turn dequeue would run it twice. + + See #28503. """ try: overflow = self._overflow_queue(session_key) @@ -186,6 +188,10 @@ class GatewayBusySessionMixin: "user_id": getattr(source, "user_id", "") or "", # Writer identity: a leaked lease from this process is re-acquired by the next # turn rather than fencing it out forever (pruning only reclaims dead PROCESSES). + # Writer identity for re-entrancy (#94595): if this process leaks a lease for this + # session (exception path skipped release), the next turn re-acquires its own entry + # instead of being fenced out of it forever — pruning only reclaims entries whose + # PROCESS died. "live_session_id": str(session_key), }, ) @@ -214,7 +220,12 @@ class GatewayBusySessionMixin: async def _session_has_compression_in_flight(self, session_key: str) -> bool: """True when a compression lock is held for this session's id (callers demote interrupt → queue, else a follow-up against the pre-rotation parent orphans compression siblings). - Both blocking reads run in a worker thread so a large state.db never freezes the loop.""" + Both blocking reads run in a worker thread so a large state.db never freezes the loop. + + Context compression is interrupt-protected (#23975) but gateway ``interrupt`` busy-input mode can + still start a follow-up turn against the pre-rotation parent while compression is mid-flight, + producing orphaned compression siblings (#56391). + """ session_store = getattr(self, "session_store", None) if not session_key or session_store is None: return False @@ -241,6 +252,7 @@ class GatewayBusySessionMixin: holder = await asyncio.to_thread(raw_db.get_compression_lock_holder, str(session_id)) # Production returns Optional[str]. Reject non-strings so a MagicMock auto-attr (or any # unexpected truthy) cannot look like a held lock and skip hygiene. + # See #96953. return isinstance(holder, str) and bool(holder) except (AttributeError, TypeError): return False @@ -270,6 +282,9 @@ class GatewayBusySessionMixin: # FIFO so each follow-up gets its own turn in arrival order (the single pending slot used to # be silently OVERWRITTEN). Photo bursts still merge into the head slot (album semantics). pending_slot = getattr(adapter, "_pending_messages", None) + # #28503 — Previously this called ``merge_pending_message_event`` with the default + # ``merge_text=False``, which silently OVERWROTE the single pending slot when consecutive text + # messages arrived in ``busy_input_mode: queue``. existing = pending_slot.get(session_key) if isinstance(pending_slot, dict) else None same_security_context = existing is not None and ( getattr(existing, "internal", False) == getattr(event, "internal", False) @@ -368,6 +383,18 @@ class GatewayBusySessionMixin: # has_blocking_approval so a conversational "yes" never fires a command. try: from tools.approval import has_blocking_approval + # --- Approval response routing (#46866) --- When the agent is blocked waiting for a + # dangerous-command approval, plain-text responses like "yes" or "approve" must be routed to the + # approval handler instead of being steered/queued/interrupted. Slash forms (/approve, /deny) + # already bypass to the runner at the base-adapter guard. This handles the bare-word forms + # (Signal/SMS users naturally type "yes" rather than "/approve"). Gating on + # has_blocking_approval(session_key) is the disambiguator that keeps a conversational "yes" from + # triggering a dangerous command when no approval is actually pending (design intent — see + # run.py "Pending exec approvals are handled by /approve and /deny" note). We reuse the + # canonical /approve and /deny handlers rather than re-deriving the resolution + i18n messaging: + # they resolve the waiting thread, resume typing, AND return a localized confirmation string. + # The busy-handler path does not auto-send that return, so we deliver it ourselves (mirroring + # the draining-case send above). if event.allow_gateway_control and has_blocking_approval(session_key): _raw_text = (event.text or "").strip().lower() _match = self._PLAINTEXT_APPROVAL_WORDS.get(_raw_text) @@ -424,6 +451,9 @@ class GatewayBusySessionMixin: if effective_mode == "steer": steer_text = await self._prepare_busy_steer_text(event) # Steerable: plain text, OR every attachment is voice media folded into steer_text. + # A follow-up qualifies for steering when it is plain text, OR when every attachment is + # STT-eligible voice media whose transcript was just folded into steer_text — otherwise a voice + # note in steer mode silently degrades to queue mode (#58780). _steer_media_urls = getattr(event, "media_urls", None) or [] _steer_all_voice = bool(_steer_media_urls) and ( len(self._pending_event_audio_paths(event)) == len(_steer_media_urls) @@ -573,6 +603,7 @@ class GatewayBusySessionMixin: # Same authorization gate as the cold path, else unauthorized users in shared threads # inject messages into a session they don't own. from gateway.run import _AGENT_PENDING_SENTINEL + # See #17775. if not self._is_user_authorized(event.source): logger.warning( "Dropping message from unauthorized user in active session: " @@ -613,6 +644,15 @@ class GatewayBusySessionMixin: # the run and must NOT replay). FIFO gives each text its own turn (raw merge would join them). if not _steer.steered and not redirected: self._queue_or_replace_pending_event(session_key, event) + # Store the message so it's processed as the next turn after the current run finishes (or is + # interrupted). Skip this for a successful steer — the text already landed inside the run and must + # NOT also be replayed as a next-turn user message. Route through _queue_or_replace_pending_event + # (the same FIFO infrastructure used by busy queue-mode and /queue) rather than a raw + # merge_pending_message_event(merge_text=True). The raw merge newline-joins consecutive TEXT + # follow-ups into a SINGLE pending turn, destroying message boundaries — so two separate user + # messages sent while the agent was busy (interrupt mode, or a steer that fell back to queue) + # arrived as one mashed-together turn (#43066 sub-bug 2). The FIFO path gives each text its own turn + # in arrival order while still preserving photo-burst / album merge semantics for media. is_queue_mode = effective_mode == "queue" is_steer_mode = effective_mode == "steer" is_redirect_mode = effective_mode == "interrupt" and redirected @@ -706,6 +746,9 @@ class GatewayBusySessionMixin: registered as Discord slash commands) would interrupt the agent AND get silently discarded by the slash-command safety net, producing a zero-char response. See #5057, #6252, #10370. + + 1. ``busy_handler`` — special mid-run variant (e.g. /goal's control-verb whitelist, /queue's FIFO + enqueue, /model's custom reject text). 2. 3. See #5057, #6252, #10370. """ name = cmd_def.name policy = getattr(cmd_def, "busy_policy", "reject") @@ -773,6 +816,8 @@ class GatewayBusySessionMixin: # /reset and /new bypass the running-agent guard (else they'd queue as user text and replay # into the same broken history); clear pending messages so the old text doesn't replay. from gateway.run import _INTERRUPT_REASON_RESET + # Interrupt the agent first, then clear the adapter's pending queue so the stale "/reset" text + # doesn't get re-processed as a user message after the interrupt completes. See #2170. await self._interrupt_and_clear_session( quick_key, source, interrupt_reason=_INTERRUPT_REASON_RESET, invalidation_reason="new_command", ) @@ -929,6 +974,10 @@ class GatewayBusySessionMixin: # ONLY when this process booted from a chat /restart AND is within a short post-boot # window; consume the flag one-shot so a later legitimate /restart is honored. if ( + # Belt-and-suspenders for when the dedup marker goes missing (manually cleaned up, or + # the previous cycle's write failed). Without a marker the update_id comparison below + # can't run, so a redelivered /restart would sail through and re-restart the gateway — + # an infinite loop (issue #18528). getattr(self, "_booted_from_restart", False) and time.time() - getattr(self, "_startup_time", 0.0) < 60 ): diff --git a/gateway/run_config_loaders.py b/gateway/run_config_loaders.py index d209f9c702..8d5b00fc1e 100644 --- a/gateway/run_config_loaders.py +++ b/gateway/run_config_loaders.py @@ -130,6 +130,11 @@ class GatewayConfigLoadersMixin: config on every call (callers run inside ``_profile_runtime_scope``, so routed multiplex profiles get their own personality/system_prompt and ``/personality`` edits apply next turn). Legacy ``channel_prompts`` are applied separately via ``event.channel_prompt`` in ``run_sync``. + + Callers run inside ``_profile_runtime_scope`` (``run_sync`` under ``_run_agent``), so a routed + multiplex profile gets its own ``display.personality`` / ``agent.system_prompt`` instead of a + boot-time snapshot of the launch profile's (#89161); ``/personality`` edits take effect on the next + turn for the same reason. """ override = self._channel_override(platform, chat_id, thread_id, parent_id) if override and override.system_prompt: @@ -142,6 +147,8 @@ class GatewayConfigLoadersMixin: Per-model override > global ``agent.reasoning_effort``; YAML False = disabled. Empty ``model`` uses ``model.default``. + + Closes #21256. """ from gateway.run import _load_gateway_runtime_config from hermes_constants import resolve_reasoning_config @@ -343,7 +350,10 @@ class GatewayConfigLoadersMixin: @classmethod def _load_cron_drain_timeout(cls) -> float: - """The cron-only floor under the stop()/drain wait.""" + """The cron-only floor under the stop()/drain wait. + + See #82161. + """ return cls._load_env_or_agent_cfg_timeout( "HERMES_CRON_DRAIN_TIMEOUT", "cron_drain_timeout", parse_cron_drain_timeout, DEFAULT_GATEWAY_CRON_DRAIN_TIMEOUT, @@ -410,6 +420,11 @@ class GatewayConfigLoadersMixin: Lets a chain edited after startup reach messaging sessions (cron already re-reads per job). A TRANSIENT read/parse failure (user mid-edit, non-atomic write) keeps the last known-good chain; only a successful read that genuinely lacks the key clears it. + + Cron already does this per job via ``get_fallback_chain``; the gateway previously froze + ``self._fallback_model`` at process start, so a chain configured (or changed) after ``hermes + gateway`` was running never reached messaging sessions even though the same process's cron jobs fell + back correctly. Fixes #60955. """ from gateway.run import _hermes_home try: @@ -443,6 +458,9 @@ class GatewayConfigLoadersMixin: Skips the rewrite while a cooldown holds the agent on an activated fallback provider (``restore_primary_runtime`` owns that lifecycle); otherwise replaces the chain so mid-uptime ``fallback_providers`` edits apply without a restart. + + When primary is active (or cooldown expired), replace the chain so mid-uptime ``fallback_providers`` + edits take effect without requiring a gateway restart (#60955). """ if agent is None: return @@ -458,6 +476,7 @@ class GatewayConfigLoadersMixin: # A config edit means the user changed something — drop the session-scoped unavailability # memo so re-configured entries (e.g. credentials added mid-uptime) get retried. Only on real # content change, so the per-message no-op refresh keeps the memo's rate-limiting benefit. + # See #60955. if new_chain != old_chain: unavailable = getattr(agent, "_unavailable_fallback_keys", None) if unavailable: diff --git a/gateway/run_goals.py b/gateway/run_goals.py index 4b451316e2..27a31fbd3b 100644 --- a/gateway/run_goals.py +++ b/gateway/run_goals.py @@ -373,6 +373,11 @@ class GatewayGoalsMixin: wakeup = await self._run_in_executor_with_context(mgr.fire_tick) if not wakeup: return + # #85957: after the parent turn's event.complete the CLIENT owns the next turn on this stateless + # surface. Persist the completion as a durable delivery row — never self-post it as a new role=user + # prompt. + # #85957: same client-owns-the-turn rule as the raw-key branch above — persist the completion as a + # delivery row, never self-post it as a new role=user prompt. try: logger.info( "loop wakeup #%s — injecting for %s chat=%s thread=%s", diff --git a/gateway/run_inbound.py b/gateway/run_inbound.py index 9633c98a28..a0ce1e9587 100644 --- a/gateway/run_inbound.py +++ b/gateway/run_inbound.py @@ -154,6 +154,7 @@ class GatewayInboundMixin: # and session setup — so an ignored channel can never reach pairing/auth/session state. _chat_id = getattr(source, "chat_id", None) if ( + # See #51899. not is_internal and getattr(source, "platform", None) == Platform.SLACK and _is_slack_ignored_channel(_config, _chat_id) @@ -455,6 +456,10 @@ class GatewayInboundMixin: lived (``ws_orphan_reap`` / ``agent_close``): otherwise the fast-path queues every next message into the dead runtime. The cold path re-attaches via ``get_or_create_session``.""" try: + # #99106: durable-reaped guard. This is the live-gateway variant of #54878 and the #632 + # detached/ 405 suppressions in production. Evict the stale slot so the next message falls + # through to the cold path and re-attaches or creates a fresh session; /status then correctly + # shows 代理运行中: 否 before the heal and a live turn after. _reap_store = getattr(self, "session_store", None) # Public, lock-held accessors: peek_session_id returns a non-str on stubbed stores in # bare test runners — the isinstance() / ``is True`` gates keep this inert unless a @@ -950,6 +955,8 @@ class GatewayInboundMixin: # Quick commands are slash capabilities too — and type:exec ones run a shell command in # the gateway process. They are never in the registry, so the early gate never fires for # them; apply the same admin/user policy to the raw typed name here. + # The early gate above only fires for registry-known commands, so quick commands (never in the + # registry) would otherwise reach this dispatch sink unchecked. (#44727) _denied = self._check_slash_access(source, command) if _denied is not None: return True, _denied, command @@ -996,6 +1003,7 @@ class GatewayInboundMixin: # Pass the platform explicitly: bundle skill loading bypasses get_skill_commands()' # scan-time disabled filter, and one gateway process serves several platforms, so # env-var platform resolution can't be trusted here. + # Mirrors the stacked-skill gate (#58888). bundle_result = build_bundle_invocation_message( bundle_key, event.get_command_args().strip(), task_id=_quick_key, platform=source.platform.value if source.platform else None, @@ -1139,6 +1147,13 @@ class GatewayInboundMixin: oldest orphan runs as THIS turn and the incoming event is parked behind the chain. Skipped for control commands and internal events.""" try: + # ── FIFO orphan rescue (#99882) ──────────────────────────────── If this session went idle with + # a populated overflow (queued during a busy window whose post-turn drain never promoted — e.g. + # a compression-demoted follow-up after the compression window ended through an exit that + # skipped the promotion site), those events were silently orphaned. We are starting the next + # turn for this session NOW: re-stage the orphans in FIFO order and enqueue the incoming event + # behind them, so arrival order (#28503) holds: oldest orphan runs as this turn, the rest drain + # in order, the new message last. _orphan_adapter = self._adapter_for_source(source) if _orphan_adapter is None or getattr(event, "internal", False) or event.get_command(): return event, source, is_internal @@ -1261,6 +1276,12 @@ class GatewayInboundMixin: self._release_running_agent_state(_quick_key) # Turn lease is keyed by (routing key, run generation) so this unwind can only free # the lease its own turn acquired, never a newer turn's. + # Unconditional release covers every exit path. _release_running_agent_state is idempotent + # (pop-on-absent is harmless) and, called without a run_generation guard, always clears the slot + # regardless of which generation it holds. This evicts the zombie left when session_reset bumps + # the generation (N -> N+1) mid-flight: gen-N's guarded release inside _run_agent returns False, + # and the old sentinel-only check here missed the leftover real agent — locking the session out + # forever (#28686). self._release_turn_lease(_quick_key, _run_generation) def _restore_moa_one_shot(self, event: "MessageEvent", quick_key: str) -> None: @@ -1299,6 +1320,7 @@ class GatewayInboundMixin: _safe_user_name = neutralize_untrusted_inline_text(source.user_name) # Slack: expose the CURRENT speaker's verifiable `<@U...>` id so "mention me again" has a # trusted target (display names are ambiguous). user_id comes from the envelope, not user-editable. + # See #17916. if source.platform == Platform.SLACK and source.user_id: _safe_user_name = f"{_safe_user_name} | Slack user <@{source.user_id}>" message_text = f"[{_safe_user_name}] {message_text}" @@ -1900,6 +1922,7 @@ class GatewayInboundMixin: transcript = result["transcript"] # STT may return success=True with an empty/whitespace transcript (silence, cut-off); # empty quotes make the agent reply to nothing and can loop, so emit a sentinel note. + # See #41603. if not (transcript or "").strip(): return None, ( "[The user sent a voice message but it came through " diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py index 7928eb27a7..7dc18f0366 100644 --- a/gateway/run_notifications.py +++ b/gateway/run_notifications.py @@ -244,7 +244,11 @@ class GatewayNotificationsMixin: EXPLICIT-ONLY, unlike the non-streaming path in ``gateway/platforms/base.py``: a bare local path in a streamed reply is shown text or stale inspected content, and promoting it sent files the model never asked for. MEDIA tags are NOT deduped against prior turns (a final-reply - directive is a deliberate attach); stale auto-appended tags are deduped upstream.""" + directive is a deliberate attach); stale auto-appended tags are deduped upstream. + + Only ``MEDIA:`` directives — the explicit attachment contract — trigger post-stream uploads. See + #20834. + """ from urllib.parse import quote as _quote with _log_suppressed(logging.WARNING, "Post-stream media extraction failed: %s"): # Capture [[as_document]] before extract_media strips it: images then go via send_document. @@ -253,6 +257,13 @@ class GatewayNotificationsMixin: media_files, cleaned = adapter.extract_media(response) media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files) # Strip image URLs (parity with the non-streaming chain); no extract_local_files here. + # Do NOT deduplicate explicit MEDIA tags against prior turns here (#73771). This rescan is + # already EXPLICIT-ONLY (see docstring): a MEDIA: directive in the final streamed reply is the + # model deliberately attaching a file — including a user-requested resend. Stale auto-appended + # tags are deduped upstream in _collect_auto_append_media_tags with history_media_paths. Mirrors + # the same filter removal on the non-streaming path in gateway/platforms/base.py. Bare local + # paths in an already-streamed reply are text the user has seen (or stale inspected content), + # not an attachment request. adapter.extract_images(cleaned) _thread_meta = ( dict(thread_metadata) @@ -704,6 +715,8 @@ class GatewayNotificationsMixin: When SessionDB init fails at gateway startup, messages may flow but nothing is persisted — /resume, /history, and session_search all silently break. Best-effort: failures are logged, not raised. + + See #88235. """ error = getattr(self, "_session_db_init_error", None) if not error: @@ -804,6 +817,8 @@ class GatewayNotificationsMixin: The queue is ALWAYS drained (so watch events don't rot or requeue-spin) but injection is skipped entirely when ``display.background_process_notifications`` is ``off``. + + See #9290. """ from gateway.run import _drain_gateway_watch_events, _format_gateway_process_notification watch_events = _drain_gateway_watch_events(completion_queue) @@ -1039,6 +1054,10 @@ class GatewayNotificationsMixin: parent_session_id = str(evt.get("parent_session_id") or "").strip() if not parent_session_id: return claim + # Pre-flight (#65838-class): adapter acceptance is NOT proof of delivery — the inner #55578 resolver + # can still fail closed inside the message pipeline AFTER the adapter accepted, which would falsely + # acknowledge the durable row as delivered. Verify the target here, before acceptance, and give + # drops an honest durable disposition. verdict = await self._classify_completion_target(parent_session_id) if verdict == "terminal": if evt_type == "async_delegation": @@ -1338,6 +1357,9 @@ class GatewayNotificationsMixin: _pr.completion_queue.put(evt) # A fan-out finishing together yields N completions for one session; group by full route + # parent session so each group becomes ONE consolidated turn. + # A same-tick drain often carries several completions for the SAME originating session (a + # fan-out of background subagents finishing together). Events for different sessions never + # coalesce. See #70300. groups: dict[tuple[str, ...], list[dict]] = {} for evt in async_events: self._enrich_async_delegation_routing(evt) @@ -1388,6 +1410,9 @@ class GatewayNotificationsMixin: _raw = redact_terminal_output(_raw, _command) # Keep the last ~2000 chars snapped to a line boundary, with a marker when cut. _LIMIT = 2000 + # Truncate at line boundaries so notifications never start mid-line (fixes #23284). Keep the last + # ~2000 chars but snap to the nearest preceding newline, then prepend a truncation marker when + # output was cut. if len(_raw) > _LIMIT: _tail = _raw[-_LIMIT:] _nl = _tail.find("\n") diff --git a/gateway/run_shutdown.py b/gateway/run_shutdown.py index be40f06b47..0f6997072c 100644 --- a/gateway/run_shutdown.py +++ b/gateway/run_shutdown.py @@ -140,11 +140,27 @@ class GatewayShutdownMixin: @staticmethod def _running_cron_job_count() -> int: + # The FULL work aggregate, not _running_agent_count(): cron jobs run on the scheduler's own thread + # pool and API-server runs live on the adapter — both outside _running_agents (the #60432 blind + # spot), so counting agents alone let a suspend land mid-cron-job. Fail-AWAKE accounting: the shared + # shutdown-drain counters (_active_cron_job_count/_active_api_run_count) swallow exceptions to 0, + # which is fine for a drain but unsafe for a suspend predicate — a transient read failure would make + # live work look idle and reopen the mid-job freeze. Here an unreadable source counts as work + # (sentinel 1) so the machine stays awake until the source is readable again. from cron.scheduler import get_running_job_ids return len(get_running_job_ids()) def _active_cron_job_count(self) -> int: - """Cron jobs currently executing — they run outside ``_running_agents``; 0 if cron can't import.""" + """Cron jobs currently executing — they run outside ``_running_agents``; 0 if cron can't import. + + Cron jobs run through a standalone ``AIAgent`` on the scheduler's own thread pool + (``cron/scheduler.py::run_job``), entirely outside ``self._running_agents`` — the dict every OTHER + active-work check on this class (``_running_agent_count``, ``_drain_active_agents``) reads. Without + this, the shutdown drain is structurally blind to in-flight cron work: it can report + ``active_at_start=0`` and proceed straight to killing tool subprocesses while a cron job's terminal + command is still running (#60432). Best-effort: returns 0 if the cron module can't be imported (e.g. + a minimal test double for this class). + """ try: return self._running_cron_job_count() except Exception: @@ -191,6 +207,7 @@ class GatewayShutdownMixin: workers.pop(done_future, None) # Workers that outlive their starting coroutine have no later waiter: consume the # terminal exception so asyncio emits no unhandled-future warning. + # See #98973. if not done_future.cancelled(): with suppress(Exception): done_future.exception() @@ -583,6 +600,11 @@ class GatewayShutdownMixin: if not self._running_agents and not (_cron0 or _api0 or _deferred0): return snapshot, False # Cron has its own deadline: a chat turn is announced+resumable; a killed cron run is a permanent failure. + # ``timeout`` (``restart_drain_timeout``) defaults to 0 because interrupting a chat turn is + # announced and resumable; a cron run killed mid-flight is recorded in jobs.json as a permanent + # failure nobody is waiting on. Sharing one budget meant the default config could report + # ``timed_out=True`` after 0.00s with a cron job in flight and kill it — the drain never even + # entered this loop (#82161). started = loop.time() deadline = started + timeout cron_deadline = started + (timeout if cron_timeout is None else cron_timeout) @@ -626,6 +648,9 @@ class GatewayShutdownMixin: from gateway.run import _AGENT_PENDING_SENTINEL reason = "restart_timeout" if self._restart_requested else "shutdown_timeout" marked: list[str] = [] + # Pre-mark sessions as resume_pending BEFORE the drain wait. If the process is killed by the service + # manager during the drain, the durable marker is already written so the next gateway boot can + # recover in-flight sessions (#27856). for _sk, _agent in list(self._running_agents.items()): if _agent is _AGENT_PENDING_SENTINEL: continue @@ -653,6 +678,15 @@ class GatewayShutdownMixin: The cron worker can't (its thread reaches ``_deliver_result`` after teardown closed the transport), so this runs post-interrupt while adapters are still connected. Best-effort. + + Its thread reaches ``_deliver_result`` asynchronously, and by then ``_bounded_adapter_teardown`` has + closed the transport — so the notice never leaves the process, and ``_consume_interrupted_flag`` + discards the resulting ``delivery_error`` along with it. The run's only trace is a line in jobs.json + nobody reads (#82232). + Must therefore be called from the post-interrupt phase, while adapters are still connected — the + same window ``_notify_active_sessions_of_shutdown`` relies on for chat sessions, which is blind to + cron work because cron runs on the scheduler's own thread pool rather than ``self._running_agents`` + (#60432). """ if not job_ids: return 0 @@ -671,6 +705,7 @@ class GatewayShutdownMixin: continue # deliver=local / unresolvable-origin jobs resolve to zero targets and stay silent (no home- # channel fallback). Interrupted notices are failure-category status: honor failure_deliver. + # See #43014. targets = _resolve_delivery_targets(job, for_failure=True) except Exception as e: logger.debug("Cron interrupt targets unresolved for %s: %s", job_id, e) @@ -843,6 +878,16 @@ class GatewayShutdownMixin: tool rounds would vanish on resume. Idempotent; gracefully finished agents re-flush nothing. """ with _log_suppressed(logging.DEBUG, "Shutdown transcript flush failed: %s"): + # Persist any in-flight transcript to the SQLite session store before teardown (#13121). An + # agent forcibly interrupted by the drain-timeout escalation may never reach + # ``turn_finalizer.finalize_turn`` (the only place that flushes the turn to state.db) — e.g. it + # was blocked in a tool call that did not abort within the post-interrupt grace window. Its + # in-flight tool rounds live only in the in-memory ``_session_messages`` (refreshed per tool + # round in ``conversation_loop`` but never written to SQLite mid-turn), so the immediate + # pre-restart turn is silently dropped from ``load_transcript()`` on resume. Flushing here + # closes that gap; the resume_pending / fresh-tool-tail branches in + # ``_handle_message_with_agent`` already expect a transcript whose tail may be a pending tool + # result. _flush = getattr(agent, "_flush_messages_to_session_db", None) _session_messages = getattr(agent, "_session_messages", None) if not (callable(_flush) and isinstance(_session_messages, list) and _session_messages): @@ -880,7 +925,12 @@ class GatewayShutdownMixin: def _should_emit_long_running_notification( self, session_key: Optional[str], agent: Any, executor_task: Optional[Any], ) -> bool: - """Emit the heartbeat only while this task still owns the live run (not after ``/new`` rebinds).""" + """Emit the heartbeat only while this task still owns the live run (not after ``/new`` rebinds). + + Guards against a stale ``running: delegate_task`` heartbeat outliving the run that started it: stop + once the executor finishes, the agent is gone, or the session key has been rebound to a different + live agent (e.g. the user sent ``/new`` and a fresh agent took the slot mid-run, #12029). + """ if agent is None or (executor_task is not None and executor_task.done()): return False if session_key: @@ -957,12 +1007,25 @@ class GatewayShutdownMixin: if hasattr(agent, "shutdown_memory_provider"): # Drain queued memory writes BEFORE teardown (shutdown_all() gives the worker only ~5s, so a # /reset or rotation could drop them). Bounded; a failure never blocks teardown. + # The memory manager persists per-turn sync and end-of-session extraction on a single + # serialized background worker. shutdown_memory_provider() -> shutdown_all() only gives that + # worker a ~5s bounded drain and abandons (cancels) anything still queued past it, so a + # /reset — or any gateway session rotation that reaches this cleanup path — could silently + # drop writes the session had already handed off. The next session then loads stale memory + # (#73297). Give pending work a bounded head start through the manager's own barrier first, + # mirroring the CLI exit path (cli.py). Best-effort: a flush failure must never block + # teardown. _mm = getattr(agent, "_memory_manager", None) if _mm is not None and hasattr(_mm, "flush_pending"): with suppress(Exception): _mm.flush_pending(timeout=10) # Pass the real transcript so ``on_session_end`` hooks don't see the empty default. # ``_session_messages`` may be absent on ``object.__new__`` test stubs, hence getattr. + # ``_session_messages`` is set on ``AIAgent`` (run_agent.py:1518) and refreshed at the end + # of every ``run_conversation`` turn via ``_persist_session``; on an agent built through + # ``object.__new__`` (test stubs) the attribute may be absent, so ``getattr`` with a + # ``None`` default keeps the call signature-compatible with the pre-fix behaviour + # (``shutdown_memory_provider(messages=None)``). See #15165. session_messages = getattr(agent, "_session_messages", None) if isinstance(session_messages, list): agent.shutdown_memory_provider(session_messages) @@ -1062,6 +1125,9 @@ class GatewayShutdownMixin: project_root = Path(__file__).resolve().parent.parent # Console python under CREATE_NO_WINDOW: nothing flashes. NOT pythonw.exe — a console-less # watcher makes every console-subsystem descendant allocate a visible conhost (#54220/#56747). + # The watcher runs sys.executable (console python) under the CREATE_NO_WINDOW detach kwargs below: + # it owns one hidden console, inherited by the `hermes gateway restart` child, so nothing flashes. + # See #54220, #56747. watcher_python = sys.executable venv_dir = Path(watcher_env.get("VIRTUAL_ENV") or project_root / "venv") site_packages = venv_dir / "Lib" / "site-packages" @@ -1237,6 +1303,13 @@ class GatewayShutdownMixin: await self.stop(restart=True, detached_restart=detached, service_restart=via_service) # NOT in _background_tasks: _stop_impl cancels those, which would skip _shutdown_event.set() / exit 75. + # _run_restart is a short-lived self-terminating task (calls stop() then returns). Don't add it to + # _background_tasks — _stop_impl cancels all entries in that set, which would cancel _run_restart + # while it's awaiting _stop_task, propagating CancelledError into _stop_impl and preventing + # _shutdown_event.set() / _exit_code = 75. See #12875. We still hold a strong reference in + # self._restart_task: a bare asyncio.create_task() keeps only a weak reference, so the event loop + # may garbage-collect a still-pending task mid-flight. The cancel loop in _stop_impl explicitly + # skips _restart_task for the same reason it skips _stop_task. self._restart_task = asyncio.create_task(_run_restart()) return True @@ -1296,6 +1369,9 @@ class GatewayShutdownMixin: def _mark_cron_interrupted() -> list: # kill_all() is global: a cron job mid-dispatch lost its tool subprocess and its agent thread may # still emit a plausible response from truncated output — mark it interrupted, never success. + # Any cron job still dispatched at this instant just had its tool subprocess killed above + # (kill_all() has no per-job-ID targeting — it's a global sweep). No-op when no cron job is in + # flight. See #60432. from cron.scheduler import mark_running_jobs_interrupted _interrupted = mark_running_jobs_interrupted( f"Gateway shutdown ({phase}) killed the job's tool subprocess before the run finished." @@ -1437,6 +1513,8 @@ class GatewayShutdownMixin: logger.info("Shutdown phase: post-interrupt tool kill done at +%.2fs", ctx.elapsed()) # Last window with the transport up (the cron worker's own notice arrives after teardown). with _log_suppressed(logging.DEBUG, "Cron interrupt notification failed: %s"): + # The cron worker whose run we just killed will try to deliver its own "interrupted" notice, but + # it gets there after the adapter teardown below and the message is lost (#82232). await self._notify_interrupted_cron_jobs(_interrupted_cron_jobs) logger.info("Shutdown phase: cron interrupt notices done at +%.2fs", ctx.elapsed()) @@ -1481,6 +1559,9 @@ class GatewayShutdownMixin: if _task is self._stop_task or _task is self._restart_task: continue _task.cancel() + # The restart orchestration task is awaiting _stop_task right now; cancelling it would propagate + # CancelledError into this _stop_impl and skip _shutdown_event.set() / _exit_code = 75 (#12875). It + # self-terminates anyway. self._background_tasks.clear() self.adapters.clear() for _session_key in list(self._running_agents): @@ -1510,6 +1591,11 @@ class GatewayShutdownMixin: logger.info("Shutdown phase: final-cleanup tool kill done at +%.2fs", ctx.elapsed()) # Reap the auxiliary-client cache: clients bound to dead worker-thread loops leak httpx transports. def _reap_aux_clients() -> None: + # Reap the process-global auxiliary-client cache once at the very end of teardown. Per-turn + # cleanup runs in _cleanup_agent_resources for each active agent, but clients bound to + # worker-thread loops that died with their ThreadPoolExecutor (notably cron ticks) only get + # swept here. Without this, long-running gateways accumulate async httpx transports until they + # hit EMFILE on macOS's default RLIMIT_NOFILE=256. See #14210. from agent.auxiliary_client import shutdown_cached_clients shutdown_cached_clients() @@ -1521,6 +1607,18 @@ class GatewayShutdownMixin: # Quiesce the thread pool BEFORE closing session DBs: a late executor write after # SessionDB.close() checkpointed the WAL reopens the handle and splits the WAL generation # (close-time corruption). Clamped to the remaining watchdog leash minus 1s for the close. + # This used to run *after* the close block below, which left two holes: (a) `_executor_closing` was + # still False during the close, so any coroutine reaching `_run_in_executor_with_context` minted a + # brand-new pool and ran more blocking DB work against handles that had just been closed; (b) + # cancelling `self._background_tasks` above does not stop a `run_in_executor` future that already + # started — the task dies, the worker thread keeps writing. Either way a write lands after + # `SessionDB.close()`, which has already checkpointed the WAL and let SQLite unlink the sidecar. The + # late write silently reopens the handle (#94736) and mints a fresh WAL generation behind that + # checkpoint, so teardown checkpoints the same file a second time from a connection the shutdown log + # never accounts for — the close-time page-write damage in #101093 and the split WAL generation in + # #101064. The wait is bounded and clamped to what is left of the shutdown watchdog leash (minus a + # second for the close itself), so a stuck worker can never cost us the post-close cleanup window + # (#82161). _exec_quiesce_budget = max( 0.0, min(_EXECUTOR_QUIESCE_TIMEOUT, resolve_shutdown_watchdog_delay(timeout) - ctx.elapsed() - 1.0), ) @@ -1553,6 +1651,8 @@ class GatewayShutdownMixin: def _close_shared() -> None: # Shared SessionDB instances still held by the process-wide registry (tools, cron, mirror). + # This is the safety net that guarantees no WAL write lock survives past gateway shutdown + # (#90837). from hermes_state import close_shared_session_dbs closed = close_shared_session_dbs() if closed: @@ -1632,6 +1732,9 @@ class GatewayShutdownMixin: from gateway.run import GatewayRunner # Thread-based watchdog (asyncio timeouts cannot recover a frozen loop): dumps stacks and # os._exit past drain+grace so the service manager revives us. Skipped under pytest. + # Arm a plain OS thread at the start of stop(); if teardown never finishes within drain+grace it + # dumps faulthandler stacks and os._exit so KeepAlive/systemd can revive. Skip under pytest so + # stop()-driving unit tests don't get a delayed hard-exit in the worker. See #66892. _watchdog_done = threading.Event() self._shutdown_watchdog_done = _watchdog_done # Shutdown-path doubles may lack the deferred-worker counter. diff --git a/gateway/run_startup.py b/gateway/run_startup.py index 0ee6bd31e7..cc20ba6245 100644 --- a/gateway/run_startup.py +++ b/gateway/run_startup.py @@ -198,7 +198,11 @@ class GatewayStartupMixin: flood-control sleep must not freeze inbound on every platform): same bounded wait as the resume gate, sends finish in the background on timeout. The ledger claim + ``resume_pending`` clear run INLINE before the send task exists — deferring it let a hung notification expire the - gate with zero rows claimed, so answered turns were replayed AND redelivered.""" + gate with zero rows claimed, so answered turns were replayed AND redelivered. + + ``_send_restart_notification`` and ``_redeliver_pending_obligations`` used to be awaited inline + *before* ``_finish_startup_restore`` released the gate. See #91969. + """ from gateway.run import _clear_planned_restart_notification, _startup_restore_drain_timeout_secs claimed = await self._claim_pending_obligations() @@ -249,7 +253,13 @@ class GatewayStartupMixin: work, no sends). Must run INLINE BEFORE ``_schedule_resume_pending_sessions`` and the abandonable boot-send task: these sessions already produced their answer, so the resume path must not re-run (re-pay for) the turn however long the sends take. Mid-send / rejected rows - carry a visible recovered-reply marker (gateway/delivery_ledger.py). Returns the rows.""" + carry a visible recovered-reply marker (gateway/delivery_ledger.py). Returns the rows. + + A session with a recoverable obligation already produced its answer — the turn completed and only + delivery is owed — so clearing ``resume_pending`` here prevents the resume path from re-running (and + re-paying for) a turn whose output we hold, regardless of how long the sends ahead of redelivery + take (#91969). + """ try: from gateway.delivery_ledger import ledger_enabled, sweep_recoverable if not await asyncio.to_thread(ledger_enabled): @@ -275,6 +285,8 @@ class GatewayStartupMixin: if not claimed: return [] # Clear resume_pending for EVERY claimed row before any send: the answer is in the ledger. + # Claiming already spent one of the row's redelivery attempts — the answer is in the ledger, so the + # resume path must never re-run these turns (#91969). await self._clear_resume_pending_for_claimed_obligations(claimed) return claimed @@ -507,7 +519,10 @@ class GatewayStartupMixin: def _start_loop_liveness_guards(self, loop: asyncio.AbstractEventLoop) -> None: """Arm the selector floor and out-of-loop watchdog before adapters. Disabled entirely with - ``gateway.loop_watchdog: false`` in config.yaml (config-only knob).""" + ``gateway.loop_watchdog: false`` in config.yaml (config-only knob). + + See #69089. + """ from gateway.run import _arm_loop_floor_timer, start_loop_liveness_watchdog config = getattr(self, "config", None) if config is not None and not getattr(config, "loop_watchdog", True): @@ -600,7 +615,10 @@ class GatewayStartupMixin: def _start_loop_heartbeat_task(self) -> None: """Start the loop-liveness heartbeat task (idempotent, best-effort). An asyncio task so a - frozen loop stops refreshing ``state/gateway.heartbeat``; cancelled with the others in stop().""" + frozen loop stops refreshing ``state/gateway.heartbeat``; cancelled with the others in stop(). + + See #66892. + """ with _log_suppressed(logging.DEBUG, "Failed to start gateway loop heartbeat", exc_info=True): _existing_hb = getattr(self, "_loop_heartbeat_task", None) if _existing_hb is not None and not _existing_hb.done(): @@ -628,6 +646,9 @@ class GatewayStartupMixin: """Enable faulthandler (stderr or a log file) plus the SIGUSR2 stack-dump hook.""" # sys.stderr may be None (Windows VBS / pythonw / detached service): fall back to a log file. try: + # Enable faulthandler for stack dumps on freezes/crashes (#70344). Falls back to a log file when + # sys.stderr is None (Windows VBS / pythonw / detached service) — otherwise the gateway would + # die here and take every adapter offline. See #71671. faulthandler.enable() except (RuntimeError, ValueError, OSError): with _log_suppressed(logging.DEBUG, "faulthandler.enable() unavailable", exc_info=True): @@ -666,6 +687,7 @@ class GatewayStartupMixin: # Warn prominently when redaction is opted out; the redactor snapshots its state at import time, # so this line is the source of truth for the process lifetime. with suppress(Exception): + # Redaction status: ON by default (#17691). _redact_raw = os.getenv("HERMES_REDACT_SECRETS", "true") if _redact_raw.lower() in {"1", "true", "yes", "on"}: logger.info( @@ -846,6 +868,7 @@ class GatewayStartupMixin: ) # Stuck-loop detection: a session active across 3+ consecutive restarts is auto-suspended. with _log_suppressed(logging.DEBUG, "Stuck-loop detection failed: %s"): + # Auto-suspend it so the user gets a clean slate on the next message. See #7536. stuck = self._suspend_stuck_loop_sessions() if stuck: logger.warning("Auto-suspended %d stuck-loop session(s)", stuck) @@ -865,6 +888,10 @@ class GatewayStartupMixin: continue # Multiplex: a platform enabled in the shared config.yaml may hold its token only in a # secondary profile's .env; an empty primary would queue a reconnect loop that never heals. + # Starting that primary adapter with an empty token fails immediately and queues an infinite + # reconnect loop that can never heal (#64674). Secondary profiles still start their own adapters + # under _profile_runtime_scope with the real token -- skip the empty primary instead of failing + # loudly. if _multiplex_on and not _platform_has_bot_credential(platform, platform_config): logger.info( "Skipping %s on default profile: no bot credential in this " @@ -1033,6 +1060,8 @@ class GatewayStartupMixin: self._platform_lock_takeover_on_start = False # A platform skipped on the primary should have been picked up by a secondary owning the token; # if none did it is enabled yet silently unserved — say so loudly. + # If none did, the platform is enabled in config.yaml yet silently unserved — surface it loudly so + # the operator sees a config problem instead of a quiet dead channel (#64674 follow-up). for _skipped in _multiplex_skipped_platforms: if not any(_skipped in _profile_map for _profile_map in self._profile_adapters.values()): logger.warning( @@ -1059,6 +1088,14 @@ class GatewayStartupMixin: # Mixed (some fatal, some transient): exiting 78 would take the gateway PERMANENTLY down # over a blip. Log the fatal side loudly and fall through to the degraded/retry path. logger.error( + # WhatsApp enabled but never paired) while others hit merely transient errors (e.g. Telegram + # TimedOut during polling startup). Exiting with GATEWAY_FATAL_CONFIG_EXIT_CODE here is + # wrong in both supervision worlds: under supervisors that honor the exit-78 contract + # (systemd RestartPreventExitStatus, s6 finish→125 since #51228) the gateway goes + # PERMANENTLY down over a network blip; under anything else it crash-loops. Either way the + # retryable platforms never get their retry. Log the fatal side loudly, then fall through to + # the degraded/retry path below: the reconnect watcher recovers the retryable platforms; the + # non-retryable ones remain fatal-parked and visible in runtime status. "%d platform(s) fatally misconfigured and parked: %s. " "Staying alive so retryable platforms can recover.", len(startup_nonretryable_errors), "; ".join(startup_nonretryable_errors), @@ -1077,6 +1114,10 @@ class GatewayStartupMixin: # No adapter for any enabled platform: fleet nodes share one config.yaml but hold a subset of # credentials, so degrade gracefully. logger.warning( + # Fall through to the normal "running" state — reconnect watcher takes it from here. In fleet + # deployments the same config.yaml is shared across nodes that may only have credentials for a + # subset of platforms. Rather than failing hard, degrade gracefully and allow cron jobs to run + # (#5196). "No adapter could be created for any of the %d configured platform(s). " "Check that required dependencies are installed and credentials are set. " "Gateway will continue for cron job execution.", enabled_platform_count, @@ -1128,6 +1169,8 @@ class GatewayStartupMixin: self._booted_from_restart = True # Boot-path adapter.send() calls must not pin the inbound restore gate (a Telegram flood- # control sleep here once froze every platform). + # Restart notification, home-channel startup notice, and obligation redelivery all call + # adapter.send(). Bound them the same way _finish_startup_restore bounds resume turns. See #91969. await self._await_startup_boot_sends( planned_restart_notification_pending=_planned_restart_notification_pending(), ) @@ -1136,6 +1179,7 @@ class GatewayStartupMixin: self._schedule_resume_pending_sessions() await self._finish_startup_restore() # Surface state.db init failures to messaging platforms before the user loses data. + # See #88235. await self._send_session_db_warning_notifications() # Resume recovered process watchers. Detach the batch atomically (fresh list, not clear(): a # concurrent append during the yield must not be lost); yield every 100 to keep the loop live. @@ -1349,6 +1393,8 @@ class GatewayStartupMixin: try: store = getattr(self.async_session_store, "_store", self.async_session_store) resolver = getattr(store, "_resolve_profile_for_key", None) + # Resolve the bound text channel's channel_prompt so voice input gets the same per-channel + # context as typed messages (#50149). if callable(resolver): resolved = resolver(dest.source) if isinstance(resolved, str) and resolved.strip(): diff --git a/gateway/run_topics.py b/gateway/run_topics.py index c507ab4250..719517dc1c 100644 --- a/gateway/run_topics.py +++ b/gateway/run_topics.py @@ -44,7 +44,10 @@ class GatewayTopicThreadsMixin: @staticmethod def _telegram_topic_profile_name(source: SessionSource) -> str: """Profile namespace for topic-mode rows: the profile stamped on the routed event, never the - process-global one (under multiplex that mis-attributes state across bots sharing state.db).""" + process-global one (under multiplex that mis-attributes state across bots sharing state.db). + + See #76423. + """ return str(getattr(source, "profile", None) or "").strip() or "default" def _sync_session_db(self): @@ -90,7 +93,10 @@ class GatewayTopicThreadsMixin: def _telegram_topic_cooldown_key(self, source: SessionSource) -> Optional[str]: """Cooldown key (profile, chat_id): profiles sharing a Telegram private chat_id under - multiplex must not suppress each other's lobby reminders / capability hints.""" + multiplex must not suppress each other's lobby reminders / capability hints. + + See #76423. + """ chat_id = str(source.chat_id or "") return f"{self._telegram_topic_profile_name(source)}:{chat_id}" if chat_id else None @@ -180,7 +186,11 @@ class GatewayTopicThreadsMixin: def _sync_telegram_topic_binding(self, source: SessionSource, session_entry, *, reason: str) -> None: """Update the topic binding to ``session_entry.session_id``: a stale binding after a mid-turn - compression rotation reloads the oversized parent next message, retriggering compression.""" + compression rotation reloads the oversized parent next message, retriggering compression. + + Telegram topic lanes persist a (chat_id, thread_id) -> session_id row so reopening a topic in a + fresh process resumes the right Hermes session. See #20470, #29712, #33414. + """ if not self._is_telegram_topic_lane(source): return try: @@ -551,6 +561,7 @@ class GatewayTopicThreadsMixin: logger.exception("Failed to disable Telegram topic mode") return f"Failed to disable topic mode: {exc}" # Reset per-profile+chat debounce state so the next activation doesn't see a stale cooldown. + # See #76423. cooldown_key = self._telegram_topic_cooldown_key(source) for attr in ("_telegram_lobby_reminder_ts", "_telegram_capability_hint_ts") if cooldown_key else (): store = getattr(self, attr, None) diff --git a/gateway/run_turn.py b/gateway/run_turn.py index 2363d1a48c..eb65754aa0 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -208,6 +208,12 @@ class GatewayTurnMixin: row = db.get_session(session_id) if not row: return + # Legacy backfill: canonical Bot Chats created BEFORE the follow_profile_config contract existed + # carry no marker, yet they are still the plugin-owned forever-DM. The plugin's own identity + # rule is "the profile's session titled exactly 'Bot Chat'" (UNIQUE(title) makes that an exact + # registry, and pre-policy rows may be visible OR hidden), so mirror that rule here. Without + # this, every Bot Chat that already exists in the field stays pinned to its stale stored + # provider until the user deletes it — the exact live-report shape (#89497 / #94818). raw_config = row.get("model_config") config = {} with suppress(Exception): @@ -319,6 +325,8 @@ class GatewayTurnMixin: bound_session_id = stored_session_id # A binding pointing at a pre-compression parent is walked forward to the tip so the next # message resumes the compressed child instead of reloading the oversized parent. + # Returns the input unchanged when the session isn't a compression parent, so this is cheap and + # safe. See #20470, #29712, #33414. if bound_session_id and self._session_db is not None: try: canonical_session_id = await self._session_db.get_compression_tip(bound_session_id) @@ -344,10 +352,16 @@ class GatewayTurnMixin: """Consume auto-reset / fresh-reset flags and emit ``session:start`` for new sessions. Returns ``(_was_auto_reset, _is_new_session)``.""" # Consume was_auto_reset immediately so it cannot re-fire and wipe overrides set between turns. + # Capture and immediately consume was_auto_reset so it does not re-fire on subsequent messages — + # preventing the cleanup from wiping model/reasoning overrides set between turns (Closes #48031). _was_auto_reset = getattr(session_entry, "was_auto_reset", False) if _was_auto_reset: # Conversation boundary: the funnel clears every conversation-scoped dict; evict the cached # agent so context_compressor._previous_summary cannot leak into new summaries. + # Treat auto-reset as a full conversation boundary — clear every conversation-scoped per-session + # dict in one funnel call so the fresh session does not inherit the previous conversation's + # model/reasoning overrides, a queued "/model switched" note, or a stale resolved-model cache + # (#48031, #58403). See _CONVERSATION_SCOPED_STATE. self._clear_conversation_scope(session_key, reason="auto_reset") self._evict_cached_agent(session_key) session_entry.was_auto_reset = False @@ -356,6 +370,7 @@ class GatewayTurnMixin: _is_new_session = session_entry.created_at == session_entry.updated_at or _was_auto_reset or _is_fresh_reset # Consume is_fresh_reset so it doesn't leak onto later messages in the same session. if _is_fresh_reset: + # See #6508. session_entry.is_fresh_reset = False if _is_new_session: await self.hooks.emit("session:start", { @@ -373,6 +388,8 @@ class GatewayTurnMixin: context_note = _AUTO_RESET_CONTEXT_NOTES.get(reset_reason, _AUTO_RESET_CONTEXT_NOTES["idle"]) # Long-lived channels: point the agent at the prior same-channel session for session_search. try: + # Returns None (appends nothing) for other platforms or when there's no prior activity to + # recall. Deterministic — no extra API/DB calls (#36220). continuity_note = build_channel_continuity_note(session_entry, source) except Exception: continuity_note = None @@ -384,6 +401,16 @@ class GatewayTurnMixin: policy = self.session_store.config.get_reset_policy( platform=source.platform, session_type=getattr(source, 'chat_type', 'dm'), ) + # Check pairing store. A pairing entry is a first-class authorization grant, created only by a + # trusted operator approving a pairing code (hermes gateway pairing approve / the authenticated + # dashboard) — an inbound sender can never reach approve_code, so this is not an + # attacker-controlled path. Honored as a UNION with the allowlist: a paired user is authorized + # regardless of the allowlist, and when an allowlist IS configured, operator approval also + # writes the user into that allowlist (see PairingStore._approve_user), keeping a single + # operator-visible source of truth. (#23778: the original bypass was the inbound + # message/approval-button gate, not this gate; that gate is fixed separately.) In multiplex + # gateways, route to the per-profile PairingStore so each profile's whitelist is isolated; falls + # back to the global store when the source has no profile or the profile isn't registered. platform_name = source.platform.value if source.platform else "" # Suspended / restart-recovery-expired sessions always notify (the user must learn they # can /resume); idle/daily resets respect policy.notify + excluded platforms + activity. @@ -601,6 +628,8 @@ class GatewayTurnMixin: if _needs_compress: # DB-backed cooldown (shared with context_compressor.py): survives gateway restarts, so a # failing compression is not re-triggered on every restart. + # The in-memory dict was reset on every restart, re-triggering the same failing compression and + # wedging session storage (#74136). _session_db = getattr(self, "_session_db", None) _getter = getattr(getattr(_session_db, "_db", _session_db), "get_compression_failure_cooldown", None) if _getter is not None: @@ -928,6 +957,20 @@ class GatewayTurnMixin: ) _hyg_rotated = False _compressed = history + # Only rewrite the transcript when rotation produced a NEW session id. In-place compaction does NOT + # need a rewrite: archive_and_compact() has already soft-archived the previous active rows and + # inserted the compacted messages as the new active set inside _compress_context(). Calling + # rewrite_transcript() after in-place compaction would invoke replace_messages(active_only=False) + # which DELETEs ALL rows — including the archived turns that archive_and_compact() deliberately + # preserved (silent data loss, #61145). The danger this guards against (mirrors the /compress fix + # #44794/#39704): if _compress_context returns a summary but neither rotates nor completes + # archive_and_compact(), the session_id is unchanged for a FAILURE reason, and an unconditional + # rewrite_transcript() would DELETE the original messages and replace them with only the compressed + # summary (permanent data loss, #21301). Write-before-repoint (mirrors manual /compress): if we + # repointed session_entry onto the child SID and rewrite_transcript then failed (lock/ENOSPC), the + # live entry would already reference a brand-new empty session while the turn continues — the + # conversation silently vanishes. Persist the child transcript first; only then rebind the live + # entry. if _hyg_rotated: if not await self.async_session_store.rewrite_transcript(_hyg_new_sid, _compressed): logger.error( @@ -1039,6 +1082,8 @@ class GatewayTurnMixin: would destroy the live thread (next turn starts blank), so use the cached agent's thread/compact/start and KEEP it cached.""" from gateway.run import run_codex_hygiene_compaction + # codex app-server runtime: the model's real context is the app-server's server-side thread, not the + # transcript mirror. See #73503. _hyg_codex_auto = "native" _hyg_comp_cfg = hs.data.get("compression") if isinstance(hs.data, dict) else None if isinstance(_hyg_comp_cfg, dict): @@ -1114,6 +1159,11 @@ class GatewayTurnMixin: attempt.commit_fence = _hyg_commit_fence attempt.future = loop.run_in_executor( None, + # But it MUST run inside the caller's contextvars: under multiplex_profiles the profile + # secret scope / HERMES_HOME override live in ContextVars, and a bare run_in_executor worker + # starts with an empty Context — the summary model's get_secret(_API_KEY) then + # fails closed (UnscopedSecretError) and every hygiene compaction silently degrades to a + # lossy truncation (#100849 bundle). copy_context().run, lambda: _hyg_agent._compress_context( _hyg_msgs, "", approx_tokens=plan.approx_tokens, commit_fence=_hyg_commit_fence, @@ -1177,6 +1227,8 @@ class GatewayTurnMixin: ) except HygieneTurnHoldExceeded: # Availability boundary, not a failure — already logged at INFO by the turn-hold handler. + # Must not hit the generic "auto-compress failed" warning below: that log is how thinking-model + # deployments read as permanently broken (#97963; surfaced by @686f6c61 in PR #99657). pass except Exception as e: logger.warning("Session hygiene auto-compress failed: %s", e) @@ -1327,6 +1379,7 @@ class GatewayTurnMixin: logger.debug("clear_resume_pending failed for %s: %s", session_key, _e) # Normalize empty responses: surface errors, partial failures, and work-without-text. + # Fix for #18765. if not _intentional_silence: response = _normalize_empty_agent_response(agent_result, response, history_len=len(history)) response = _sanitize_gateway_final_response(source.platform, response) @@ -1459,6 +1512,15 @@ class GatewayTurnMixin: Context-overflow failures must NOT persist the user message (session would grow and reproduce the failure forever); transient failures (429/timeout/5xx) DO.""" from gateway.run import _is_gateway_hidden_reasoning_incomplete_turn + # Save the full conversation to the transcript, including tool calls. This preserves the complete + # agent loop (tool_calls, tool results, intermediate reasoning) so sessions can be resumed with full + # context and transcripts are useful for debugging and training data. IMPORTANT: For + # context-overflow failures (compression exhausted, generic 400 on large sessions) we must NOT + # persist the user's message — doing so would grow the session further and cause the same failure on + # the next attempt, an infinite loop. (#1630, #9893) Transient failures (429, timeout, connection + # error, provider 5xx) are different: the session is not oversized, and silently dropping the user + # message causes severe context loss on retry — the agent forgets what was just asked. Persist the + # user turn so the conversation is preserved. (#7100) agent_failed_early = bool(agent_result.get("failed")) hidden_reasoning_incomplete = _is_gateway_hidden_reasoning_incomplete_turn(agent_result) _err = str(agent_result.get("error", "")).lower() @@ -1492,6 +1554,9 @@ class GatewayTurnMixin: replaying the oversized context forever. Never on a lock-contended defer — that is the OPPOSITE case (a concurrent path holds the lock and is shrinking it). Returns ``(response, session_entry)``.""" + # When compression is exhausted, the session is permanently too large to process. (#9893) Never wipe + # the session for that — retry-next-message semantics apply (#69870 lock-skip consumer; salvaged + # from #49874). if agent_result.get("compression_deferred"): logger.info( "Compression deferred for session %s — the compression " @@ -1509,6 +1574,13 @@ class GatewayTurnMixin: # Re-point the Telegram topic binding at the fresh session, or the binding-heal walk # switches the next message back onto the bloated child and re-triggers exhaustion # forever. No-op on non-topic lanes. + # Compression rotated session_entry.session_id to the oversized compressed child earlier + # this turn (the agent-result sync above), and that _sync also rewrote the (chat_id, + # thread_id) -> bloated-child binding. reset_session swaps in a clean, parentless session, + # but without re-syncing the binding the next inbound message in this topic gets + # switch_session'd back onto the bloated child by the binding-heal walk, reloads the + # oversized transcript, and re-triggers compression exhaustion forever (#35809 — regression + # of the #9893/#10063 auto-reset). session_entry = new_entry await asyncio.to_thread( self._sync_telegram_topic_binding, source, session_entry, reason="compression-exhausted-reset", @@ -1522,6 +1594,11 @@ class GatewayTurnMixin: @staticmethod def _hmwa_user_transcript_entry(event, prepared, ts): """Transcript row for the inbound user turn (clean text + event time when captured).""" + # Transient failure (429/timeout/5xx): persist only the user message so the next message can load a + # transcript that reflects what was said. Skip the assistant error text since it's a + # gateway-generated hint, not model output. Hidden- reasoning-only incomplete turns follow the same + # persistence rule so peer-agent channels don't ingest them as completed assistant turns. (#7100, + # #51628) _user_entry = { "role": "user", "content": ( @@ -1550,6 +1627,13 @@ class GatewayTurnMixin: history = prepared.history # The agent already persisted this turn's rows (codex app-server reports agent_persisted=True # too); skip the DB write. Default = a session DB exists; non-persisting runtimes pass False. + # The agent already persisted these messages to SQLite via _flush_messages_to_session_db(), so skip + # the DB write here to prevent the duplicate-write bug (#860 / #42039). This holds for the codex + # app-server runtime too: although it early-returns and bypasses conversation_loop's per-step + # flushes, it flushes its own projected assistant/tool messages before returning and reports + # agent_persisted=True (see agent/codex_runtime.py). Reading the flag (default = self._session_db is + # not None) keeps the persistence contract explicit and lets any future non-persisting runtime opt + # into a gateway-side write by returning False. agent_persisted = agent_result.get("agent_persisted", self._session_db is not None) _user_row = self._hmwa_user_transcript_entry(event, prepared, ts) @@ -2076,6 +2160,7 @@ class GatewayTurnMixin: )}, session_db=getattr(self._session_db, "_db", self._session_db), # Reload from disk — do not reuse the startup snapshot. + # See #60955. fallback_model=self._refresh_fallback_model(), ) try: @@ -2164,7 +2249,10 @@ class GatewayTurnMixin: """Disconnect, reconnect, and notify MCP tool changes (shared by button / text / no-confirm paths). Under multiplex the reload runs inside the requesting profile's runtime scope (entered here - when the caller did not) and only that profile's servers are torn down and rediscovered.""" + when the caller did not) and only that profile's servers are torn down and rediscovered. + + See #95518. + """ from gateway.run import _profile_runtime_scope multiplex = bool(getattr(self.config, "multiplex_profiles", False)) if multiplex and not get_hermes_home_override(): @@ -2265,6 +2353,9 @@ class GatewayTurnMixin: if _buffer_only: _effective_cursor = "" # Fresh-final applies to Telegram only (others edit in place cheaply). + # Fresh-final applies to Telegram only — other platforms either edit in place cheaply (Discord, + # Slack) or don't have the timestamp-on-edit / edit-timestamp-stays-stale problem. (Ported from + # openclaw/openclaw#72038.) _fresh_final_secs = ( float(getattr(scfg, "fresh_final_after_seconds", 0.0) or 0.0) if source.platform == Platform.TELEGRAM else 0.0 @@ -2293,6 +2384,9 @@ class GatewayTurnMixin: """Platform stream consumer for the proxy path when streaming is enabled, else ``None``.""" from gateway.run import _load_gateway_config, _platform_config_key _scfg = getattr(getattr(self, "config", None), "streaming", None) + # #60671 — streaming TTS consumer is created on the outer event-loop thread before run_sync + # launches. run_sync only reads it via ``streaming_tts_consumer_holder[0]`` for delta callback + # wiring. if _scfg is None: from gateway.config import StreamingConfig _scfg = StreamingConfig() @@ -2552,6 +2646,10 @@ class GatewayTurnMixin: "thinking_progress", default=False, require_platform_override_for={Platform.MATTERMOST}, ) != "off" # Slack-native task cards need the progress queue even with text tool_progress off. + # Slack-native task cards (#29483): when the Slack adapter's opt-in is set, tool progress renders as + # native plan/task cards via chat.startStream — the progress queue is needed even though Slack keeps + # ordinary text tool_progress off by default (requiring both flags would silently leave the native + # feature inactive). _native_slack_task_cards = False if source.platform == Platform.SLACK and hasattr(adapter, "native_task_cards_enabled"): try: @@ -2691,6 +2789,10 @@ class GatewayTurnMixin: self._thread_metadata_for_progress( source, event_message_id, _progress_thread_id, _relay_prospective_thread_id, ), + # Freshness-gate stale resume_pending zombies (#46934) — but honor an explicit + # ``session_reset.mode: none``: the user opted out of ALL automatic resets, so an expired resume + # marker must fall through to a normal resume of the preserved transcript, never a silent fresh + # session (#61052). platform=source.platform, ) if _native_slack_task_cards: @@ -2766,6 +2868,10 @@ class GatewayTurnMixin: Created on the gateway loop thread (not run_sync's executor); an inactive consumer leaves the holder None so the whole-file fallback path runs.""" + # Skip when streaming TTS already delivered audio for this turn (#60671). + # This avoids a cross-scope NameError: the outer interrupt / finalisation paths reference the + # consumer via ``streaming_tts_consumer_holder[0]``. Gates: voice input, auto-TTS enabled for this + # chat, adapter supports streaming, and a usable streaming TTS provider configured. See #60671. _stts_adapter = self._adapter_for_source(source) _is_voice_input = ( message_type is not None @@ -2849,6 +2955,14 @@ class GatewayTurnMixin: agent.interrupt(pending_text) _interrupt_detected.set() # Abort streaming TTS on barge-in. + # See #60671. + # See #60671. + # See #60671. + # Finalize the streaming-TTS consumer (#60671). finish() is called from the outer event-loop thread + # (not the executor worker) so early returns from run_sync are also finalised. wait_complete() + # drains queued audio; on timeout the consumer is aborted unconditionally — if audio was audible, + # suppression is preserved so the gateway does not replay from the beginning; if no audio was + # audible, the whole-file fallback path is permitted. _stts = streaming_tts_consumer_holder[0] if _stts is not None: _stts.abort("barge-in") @@ -2944,6 +3058,9 @@ class GatewayTurnMixin: # background=true processes survive a turn: reap only children created by THIS turn on timeout. _turn_task_id = turn_ctx.session_id or "" + # The daemon watchdog is independent of asyncio: cgroup memory reclaim may starve the event loop + # that runs the normal timeout poll, but it need not also postpone cleanup until the loop recovers + # (#76115). _turn_process_baseline = process_registry.snapshot_running_ids(_turn_task_id) turn_ctx.process_task_id = _turn_task_id turn_ctx.process_baseline = _turn_process_baseline @@ -2968,6 +3085,10 @@ class GatewayTurnMixin: # `.turn.agent` stays reachable until the *next* turn is claimed; clearing the # ownership markers now means a /stop on the finished turn no longer reaps background # work it left running. + # `.turn.agent` on the session state is only reset to _AGENT_PENDING_SENTINEL when the + # *next* turn is claimed (see _session_state(...).turn.agent = ... at claim time), so a + # stale reference to this exact agent instance stays reachable from + # _interrupt_and_clear_session() until then. See #76115. _finished_agent = agent_holder[0] if agent_holder else None if _finished_agent is not None: _finished_agent._gateway_turn_process_task_id = "" @@ -3277,6 +3398,7 @@ class GatewayTurnMixin: _active[session_key].clear() # Cap recursion depth (user keeps sending while the agent keeps failing). + # (#816) if _interrupt_depth >= self._MAX_INTERRUPT_DEPTH: logger.warning( "Interrupt recursion depth %d reached for session %s — " @@ -3297,6 +3419,7 @@ class GatewayTurnMixin: next_source, next_message, next_session_key = source, pending, session_key # message_type is carried into the recursive call so queued voice turns can stream TTS. next_message_id = next_channel_prompt = next_message_type = None + # See #60671. if pending_event is not None: next_source = getattr(pending_event, "source", None) or source if self._is_goal_continuation_event(pending_event) and not self._goal_still_active_for_session(session_id): @@ -3324,6 +3447,7 @@ class GatewayTurnMixin: next_message_type = getattr(pending_event, "message_type", None) # Clear the prior turn's streaming-TTS completion marker so the recursive turn isn't suppressed. + # See #60671. _clear_adapter = self._adapter_for_source(source) _completed_turns = getattr(_clear_adapter, "_streaming_tts_completed_turns", None) _prior_key = getattr(_clear_adapter, "_streaming_tts_turn_key", None) @@ -3339,6 +3463,15 @@ class GatewayTurnMixin: # Re-baseline the cached agent's message_count before recursing, else the coherence guard # rebuilds on OUR OWN flushed rows (the outer handler re-baselines only after the chain). + # Re-baseline the cached agent's message_count snapshot before recursing into the in-band queued + # (/queue) follow-up turn. The first turn has completed and flushed its own user + assistant rows to + # the SessionDB, so the cross-process coherence guard (#45966) — which this recursive _run_agent + # call re-enters — would otherwise see the grown on-disk count against the stale build-time snapshot + # and rebuild the agent on THIS process's OWN writes, destroying the prompt-cache prefix #46237 was + # merged to preserve. The existing re-baseline in _handle_message_with_agent only runs after the + # whole _run_agent chain unwinds — too late for the in-band follow-up. Use the same (session_key, + # session_id) the recursive call runs under so the snapshot matches exactly what the follow-up's + # guard will consult. Fail-safe in helper. await self._refresh_agent_cache_message_count(session_key, session_id) followup_result = await self._run_agent( @@ -3371,6 +3504,7 @@ class GatewayTurnMixin: # Abort + bounded wait for streaming TTS: covers paths where normal finalisation was skipped. _stts_finally = turn_ctx.streaming_tts_consumer_holder[0] + # See #60671. if _stts_finally is not None and not _stts_finally.done: _stts_finally.abort("cleanup") with suppress(Exception): @@ -3426,8 +3560,17 @@ class GatewayTurnMixin: _final = response.get("final_response") or "" _is_empty_sentinel = not _final or _final == "(empty)" # response_previewed: only suppress if that EXACT text was delivered, not unrelated commentary. + # Unrelated commentary/progress must not be mistaken for the final response (#14238). _previewed = bool(response.get("response_previewed")) _content_delivered = bool(_sc and getattr(_sc, "final_content_delivered", False)) + # #71643: a *successful* finalize edit can still carry only the last preview snapshot — deltas + # generated between that edit and stream completion never reach any API call, and both suppression + # flags are set from the call's success rather than its content. Reconcile the consumer's recorded + # turn-final payload against the completed response: on a demonstrable mismatch (False) neither + # final_response_sent nor final_content_delivered may suppress the normal final send. False also + # covers payload-less multi-message split delivery (#78541). None (no record on a non-split legacy + # path) keeps legacy trust; the failed-finalize family (#51828 / #33793) is unaffected because those + # paths leave the flags False or record the complete fallback payload. _stale_finalized = False if _content_delivered and not _is_empty_sentinel: _matcher = getattr(_sc, "delivered_final_matches", None) diff --git a/gateway/run_turn_runner.py b/gateway/run_turn_runner.py index 9e03a6b938..a7c0696cca 100644 --- a/gateway/run_turn_runner.py +++ b/gateway/run_turn_runner.py @@ -119,6 +119,11 @@ class TurnRunner: if ( not ctx.tool_progress_enabled or event_type != "tool.started" + # The adapter's send_clarify IS the user-facing rendering (interactive buttons or the + # numbered-text fallback), so a progress bubble is pure duplication — and in verbose mode it + # dumps the raw tool-call args JSON ({"question": ..., "choices": [...]}) into the chat. Because + # the progress queue drains on a background task, that raw JSON typically lands right underneath + # the rendered prompt (#52374). or tool_name == "clarify" or self._agent_interrupted() ): @@ -365,7 +370,10 @@ class TurnRunner: async def _send_native_task_card_progress(self, adapter) -> None: """Drain the progress queue into Slack-native plan/task cards; on any native failure, fall - back to an editable in-thread message so progress stays live.""" + back to an editable in-thread message so progress stays live. + + See #29483. + """ ctx = self._ctx st = self._TaskCardState(adapter) try: @@ -651,6 +659,10 @@ class TurnRunner: ctx = self._ctx return bool(ctx.progress_queue) and ctx._run_still_current() and not self._agent_interrupted() + # ── Slack-native task cards: ID-bearing lifecycle callbacks (#29483) ── These ride + # agent.tool_start_callback / agent.tool_complete_callback so start/completion events correlate by the + # REAL tool-call id — the name-correlated text events in progress_callback would duplicate cards and + # mispair concurrent calls to the same tool. def native_tool_start_callback(self, call_id, tool_name, args): """Queue an ID-correlated native progress start from the agent thread.""" if not self._native_card_gate(): @@ -953,6 +965,7 @@ class TurnRunner: gateway_session_key=ctx.session_key, session_db=getattr(runner._session_db, "_db", runner._session_db), # Reload from disk — do not reuse the startup snapshot. + # See #60955. fallback_model=self._runner._refresh_fallback_model(), skip_context_files=skip_context_files, # Keep the persona even with minimal context: soul identity is one small file. @@ -1287,6 +1300,8 @@ class TurnRunner: # FTS write-corruption guard: if persistence failed silently the reloaded transcript is stale # while the SAME cached agent still holds the live conversation (same-session amnesia). Only # for a reused agent bound to this exact session_id. + # Replacing the live transcript with that shorter copy causes immediate same-session amnesia. See + # #50502. if reused_cached_agent and getattr(agent, "session_id", None) == ctx.session_id: selected = _select_cached_agent_history(agent_history, getattr(agent, "_session_messages", None)) if selected is not agent_history: @@ -1556,6 +1571,19 @@ class TurnRunner: from gateway.run import _collect_auto_append_media_tags if "MEDIA:" in final_response: return final_response + # Scan tool results for MEDIA: tags that need to be delivered as native audio/file + # attachments. The TTS tool embeds MEDIA: tags in its JSON response, but the model's final text + # reply usually doesn't include them. We collect unique tags from tool results and append any that + # aren't already present in the final response, so the adapter's extract_media() can find and + # deliver the files exactly once. Scope the scan to THIS turn's tool results only. ``agent_history`` + # was passed into run_conversation as ``conversation_history``, so the agent's returned ``messages`` + # list is ``agent_history`` followed by the messages produced this turn. Slicing at + # ``len(agent_history)`` isolates the current turn precisely, so a stale MEDIA: path emitted by a + # tool several turns earlier (still present in the full message list) can never leak onto a later + # text-only reply. (Fixes #34608) Path-based deduplication against _history_media_paths (collected + # before run_conversation) is retained as a secondary guard. It is also the sole guard on the + # fallback branch taken when mid-run context compression shrinks the message list below the original + # history length, preserving the compression-safe behaviour of #160. media_tags, has_voice_directive = _collect_auto_append_media_tags( result.get("messages", []), history_offset=len(agent_history), history_media_paths=history_media_paths, ) @@ -1575,6 +1603,16 @@ class TurnRunner: ctx = self._ctx runner = self._runner # Platform.LOCAL ("local") maps to the "cli" hint key the agent understands. + # session_key is propagated via contextvars in _set_session_env() (_SESSION_KEY) and via + # set_current_session_key() (_approval_session_key) below — both concurrency-safe and inherited by + # tool worker threads. We deliberately do NOT write os.environ["HERMES_SESSION_KEY"] here: + # os.environ is process-global, so concurrent gateway sessions (e.g. two Discord threads) would + # clobber each other's value, and a tool thread whose contextvar is unset would fall back to + # os.environ and read the wrong session key — misrouting command-approval prompts to the wrong + # thread (#24100). The non-gateway surfaces don't depend on this write: CLI and cron bind the + # session via contextvars (set_current_session_key / session context), and only the TUI slash-worker + # *subprocess* exports HERMES_SESSION_KEY (from its own --session-key argv, a separate process) — so + # removing this in-process gateway write does not affect any of them. platform_key = "cli" if ctx.source.platform == Platform.LOCAL else ctx.source.platform.value combined_ephemeral = self._combined_ephemeral_prompt() max_iterations = _current_max_iterations() @@ -1604,6 +1642,7 @@ class TurnRunner: self._finish_stream_consumer(result, agent_history, stream_consumer) # The streaming-TTS consumer's finish() runs on the outer loop thread after the executor # returns, so early run_sync returns are also finalised. + # See the outer finally/completion section below. See #60671. final_response = result.get("final_response") # Actual token counts from the agent instance used for this run. agent = ctx.agent_holder[0] diff --git a/gateway/run_voice.py b/gateway/run_voice.py index efa7e0c14b..718d242c98 100644 --- a/gateway/run_voice.py +++ b/gateway/run_voice.py @@ -31,7 +31,12 @@ _VOICE_MODES = {"off", "voice_only", "all"} class GatewayVoiceMixin: def _voice_key(self, platform: Platform, chat_id: str, profile: Optional[str] = None) -> str: """``::`` under multiplexing (else two bots in one channel - share a key and one ``/voice`` flips the other's); default keeps ``:``.""" + share a key and one ``/voice`` flips the other's); default keeps ``:``. + + Under multiplexing the key is additionally namespaced by the profile whose bot speaks in the chat + (``::``); the default profile keeps the historical + ``:`` shape so persisted state stays valid. See #75198. + """ base = f"{platform.value}:{chat_id}" profile = profile.strip() if isinstance(profile, str) else "" return base if not profile or profile == "default" else f"{profile}:{base}" @@ -239,6 +244,10 @@ class GatewayVoiceMixin: if not text_ch_id: return source = self._voice_input_source(adapter, guild_id, user_id, text_ch_id) + # Validate the session owner against the current allowlist before auto-resuming. A session created + # before TELEGRAM_ALLOWED_USERS (or equivalent) was configured, or before the owner was removed from + # it, must not silently receive a full agent response on gateway restart just because it has a + # resume-pending marker (issue #23778). if not self._is_user_authorized(source): logger.debug("Unauthorized voice input from user %d, ignoring", user_id) return diff --git a/gateway/run_watchers.py b/gateway/run_watchers.py index e5e4d5f9e9..ec7493eea4 100644 --- a/gateway/run_watchers.py +++ b/gateway/run_watchers.py @@ -120,6 +120,7 @@ class GatewaySessionWatchersMixin: # persisted flag also drops the /model override: finalization is a conversation boundary. self._evict_cached_agent(key) self._clear_conversation_scope(key, reason="expiry_finalized") + # See #9006. await self.async_session_store.set_expiry_finalized(entry) logger.debug("Session expiry finalized for %s", entry.session_id) @@ -133,6 +134,10 @@ class GatewaySessionWatchersMixin: logger.debug("Idle agent sweep failed: %s", e) # Neither LRU cap nor idle TTL knows what a cached transcript costs in memory. try: + # Neither the LRU cap nor the idle TTL is aware of how much memory a cached transcript costs, so + # a busy gateway keeps every warm session's tool output resident until RSS hits the cgroup limit + # (#80764). Shed LRU transcripts once the heap is over budget; they reload from the persisted + # session on the next turn. self._sweep_agent_cache_under_pressure() except Exception as e: logger.debug("Agent cache pressure sweep failed: %s", e) @@ -153,7 +158,10 @@ class GatewaySessionWatchersMixin: return _float_env("HERMES_SESSION_STALL_TIMEOUT", 300) def _session_activity_for_stall(self, session_key: str) -> Optional[dict]: - """Stall-progress snapshot from ``AIAgent.get_activity_summary()`` only; no other clocks.""" + """Stall-progress snapshot from ``AIAgent.get_activity_summary()`` only; no other clocks. + + See #72039. + """ from gateway.run import _AGENT_PENDING_SENTINEL agent = (getattr(self, "_running_agents", None) or {}).get(session_key) if agent is None or agent is _AGENT_PENDING_SENTINEL: @@ -234,6 +242,7 @@ class GatewaySessionWatchersMixin: # Re-read pending state + activity IMMEDIATELY before delivery: the snapshot ages while # earlier candidates await sends; an agent that progressed (or drained its queue) must not # get a false stall notice. Abort with the latch un-set so the next tick re-evaluates. + # See #76354. still_pending = ( (getattr(adapter, "_pending_messages", None) or {}).get(session_key) is not None or bool((getattr(self, "_queued_events", None) or {}).get(session_key)) @@ -291,7 +300,11 @@ class GatewaySessionWatchersMixin: async def _session_stall_watcher(self, interval: float = 30.0): """Pending-inbound + stale-activity stall watchdog. Progress comes only from ``get_activity_summary()``; pending inbound is a notify policy gate, not a progress clock. - Notify-only: never kills the turn (contrast ``gateway_timeout`` / ``shutdown_watchdog``).""" + Notify-only: never kills the turn (contrast ``gateway_timeout`` / ``shutdown_watchdog``). + + See #72016. + See #72039. + """ # Short initial delay so startup reconnect noise does not false-fire. await asyncio.sleep(min(30.0, max(1.0, float(interval)))) while self._running: diff --git a/gateway/session.py b/gateway/session.py index de42e7e7f6..925e8ac5e8 100644 --- a/gateway/session.py +++ b/gateway/session.py @@ -494,14 +494,25 @@ class SessionEntry: prev_session_id: Optional[str] = None # replaced by auto-reset; feeds the continuity note # Explicit /new or /reset; consumed once to re-inject topic/channel skills. Distinct from # was_auto_reset, whose "expired due to inactivity" notice is wrong for a manual reset. + # Set by reset_session() when the user explicitly sends /new or /reset. Consumed once by + # _handle_message_with_agent to trigger topic/channel skill re-injection on the first message of the new + # session. We can't reuse was_auto_reset for this because that flag fires the "session expired due to + # inactivity" user-facing notice and a misleading context-note prepend — both wrong for an explicit + # manual reset. See issue #6508. is_fresh_reset: bool = False # Set by the expiry watcher after finalizing; persisted so restarts don't re-run finalization. expiry_finalized: bool = False # Next get_or_create_session() auto-resets; set by /stop to break stuck-resume loops. + # When True the next call to get_or_create_session() will auto-reset this session (create a new + # session_id) so the user starts fresh. See #7536. suspended: bool = False # Interrupted by a restart/drain timeout, recovery expected: unlike ``suspended`` the # session_id is kept so the agent auto-continues. Cleared after the next successful turn; # escalation to ``suspended`` is the runner's ``.restart_failure_counts`` job. + # Unlike ``suspended``, ``resume_pending`` preserves the existing session_id on next access — the user + # stays on the same transcript and the agent auto-continues from where it left off. Escalation to + # ``suspended`` is handled by the existing ``.restart_failure_counts`` stuck-loop counter (#7536), not + # by a parallel counter on this entry. resume_pending: bool = False resume_reason: Optional[str] = None # e.g. "restart_timeout" last_resume_marked_at: Optional[datetime] = None @@ -769,6 +780,16 @@ class SessionStore( # SQLite handles are cached per path and resolved through ``_db`` per call, never bound # once: a multiplexed gateway serves every profile from ONE process and a handle frozen to # the root home would land every profile's rows in the root state.db. + # Initialize SQLite session database. A multiplexed gateway serves every profile from a SINGLE + # process, so a handle bound during __init__ is frozen to the process's own root home; every + # profile's rows then land in the root state.db even though ``_profile_runtime_scope`` has already + # redirected ``get_hermes_home()`` for the turn (its docstring lists "sessions" among what it + # scopes). The row still carries the right ``profile_name``, so the damage is invisible in the data + # and shows up only as the desktop listing a profile's session under the default bot -- + # ``_open_session_db_for_profile`` reads ``profiles//state.db``, which never received the + # write. See #88532. Priming the handle for the current scope here keeps the startup diagnostics + # exactly where they were: the live-DB isolation guard still raises during construction, and the + # JSONL-fallback warning is still printed once at startup rather than on first use. self._db_pinned = _DB_UNPINNED self._db_handles: Dict[Path, Any] = {} self._db_handles_lock = threading.Lock() @@ -1028,7 +1049,12 @@ class SessionStore( def set_session_metadata(self, session_key: str, key: str, value: Any) -> bool: """Persist a small JSON-serializable metadata value. Deliberately does NOT advance - ``updated_at``: a background write must not make an idle session look fresh.""" + ``updated_at``: a background write must not make an idle session look fresh. + + Metadata writes are internal bookkeeping and deliberately do NOT advance ``updated_at``: it is the + user-activity clock that drives idle/daily reset policy and the restart-resume freshness gate + (#85709), and a background write must not make an idle session look fresh. + """ return self._update_entry(session_key, lambda e: e.metadata.__setitem__(key, value)) def set_model_override(self, session_key: str, override: Optional[Dict[str, Any]]) -> None: @@ -1084,6 +1110,9 @@ class SessionStore( self._save() return new_entry + # Compression repoint is store bookkeeping, not user activity — leave ``updated_at`` alone so a + # background compression on an idle session cannot make it look fresh to reset policy or the + # restart-resume freshness gate (#85709). def switch_session(self, session_key: str, target_session_id: str) -> Optional[SessionEntry]: """Point a session key at an existing session ID (``/resume``): ends the current row and reopens the target so resume matches the CLI.""" diff --git a/gateway/session_context.py b/gateway/session_context.py index 4032c62356..7a3861ed5d 100644 --- a/gateway/session_context.py +++ b/gateway/session_context.py @@ -176,7 +176,10 @@ def session_is_messaging_surface() -> bool: def declare_stateless_channel() -> None: """Declare that this session cannot receive an async background completion. Unlike ``set_session_vars(async_delivery=False)`` this does NOT latch ``_session_context_engaged`` - (flipping the subprocess env bridge), which a one-shot CLI must not do as a side effect.""" + (flipping the subprocess env bridge), which a one-shot CLI must not do as a side effect. + + See NousResearch/hermes-agent#53027 and #63142. + """ _SESSION_ASYNC_DELIVERY.set(False) diff --git a/gateway/session_lifecycle.py b/gateway/session_lifecycle.py index 7738ddba46..78fb67e3ae 100644 --- a/gateway/session_lifecycle.py +++ b/gateway/session_lifecycle.py @@ -70,6 +70,9 @@ class SessionLifecycleMixin: entry.model_override = None self._save() # Background caller never entered ``_profile_runtime_scope``: resolve the store by key. + # The expiry watcher calls this from a background task that never entered + # ``_profile_runtime_scope``, so resolve the store from the key rather than from the ambient scope + # (#66887). _db = self._db_for_key(entry.session_key) if not _db: return @@ -123,7 +126,17 @@ class SessionLifecycleMixin: def _is_session_ended_in_db(self, session_id: str) -> bool: """True iff state.db has this session with a non-null end_reason (same staleness test as ``_prune_stale_sessions_locked``; no DB/row or DB error -> False). Lets routing self-heal a - session ended while the gateway stays alive. Store resolved from the owning profile.""" + session ended while the gateway stays alive. Store resolved from the owning profile. + + Used by ``get_or_create_session`` to self-heal at routing time: ``_prune_stale_sessions_locked`` + only runs at startup, so a session ended in the DB while the gateway stays alive (any path that + finalizes the row without clearing sessions.json) would otherwise be reused as a live routing key + and silently swallow every subsequent message until the next restart (#54878 — the live-gateway + variant of #52804/FM9). DB errors are non-fatal — never block routing on a failed lookup. + The store is resolved from the row's owning profile rather than the ambient scope: an unscoped + background writer keeps its own copy of the same session, and comparing against that copy reports a + live session as ended (#66887). + """ db = self._db_for_session_id(session_id) if not db or not session_id: return False @@ -186,7 +199,10 @@ class SessionLifecycleMixin: return changed def suspend_session(self, session_key: str) -> bool: - """Mark a session suspended so it auto-resets on next access (/stop). True if it existed.""" + """Mark a session suspended so it auto-resets on next access (/stop). True if it existed. + + Used by ``/stop`` to prevent stuck sessions from being resumed after a gateway restart (#7536). + """ return self._update_entry(session_key, lambda e: setattr(e, "suspended", True)) def _set_turn_marker_locked(self, session_key: str, entry: SessionEntry, token, started_at) -> None: @@ -322,7 +338,13 @@ class SessionLifecycleMixin: def suspend_recently_active(self, max_age_seconds: int = 120) -> int: """Mark sessions active within *max_age_seconds* as ``resume_pending`` after a crash/fast - restart (already-pending and suspended entries are skipped). Returns the number marked.""" + restart (already-pending and suspended entries are skipped). Returns the number marked. + + Called on gateway startup after a crash or fast restart to preserve in-flight sessions instead of + destroying their conversation history (#7536). Only marks sessions updated within *max_age_seconds* + to avoid touching long-idle sessions. Sets ``resume_pending=True`` so the next incoming message on + the same session_key auto-resumes from the existing transcript. + """ cutoff = _now() - timedelta(seconds=max_age_seconds) def _mark(entry: SessionEntry) -> bool: diff --git a/gateway/session_persistence.py b/gateway/session_persistence.py index 851c0f2563..197c0fad6e 100644 --- a/gateway/session_persistence.py +++ b/gateway/session_persistence.py @@ -46,7 +46,11 @@ class SessionPersistenceMixin: """SessionDB for the active profile scope. ``db_path`` pins the store; otherwise ``_default_db_path()`` follows the context-local HERMES_HOME (resolved per call so multiplexed profiles reach their own store). Handles are cached per path; failed opens enter - a bounded backoff during which callers keep using the JSONL fallback.""" + a bounded backoff during which callers keep using the JSONL fallback. + + Resolving here rather than once in ``__init__`` is the whole fix for #88532: it lets the scoping + that the multiplexed inbound path already performs actually reach session storage. + """ from hermes_state import _default_db_path, get_shared_session_db path = Path(db_path) if db_path is not None else Path(_default_db_path()) @@ -84,7 +88,13 @@ class SessionPersistenceMixin: flat dict holding every profile's keys, so it must persist to ONE file (``_routing_home``), not whichever profile is scoped — otherwise a mid-turn rewrite and the unscoped startup load see different copies and crash markers under a secondary profile go unrecovered. A - pinned handle still wins; bare test instances lacking the handle cache report no DB.""" + pinned handle still wins; bare test instances lacking the handle cache report no DB. + + Reading it through ``_db`` made that file whichever profile happened to be scoped at the time: a + whole-index rewrite during one profile's turn copied every other profile's routing rows into that + profile's store, and startup — which runs unscoped — then loaded a different copy than the one the + last writer produced. See #66887. + """ pinned = self._pinned_db() if pinned is not _DB_UNPINNED: return pinned @@ -132,7 +142,14 @@ class SessionPersistenceMixin: """The SessionDB holding *session_key*'s rows, whatever scope is active (the owning profile is encoded in the key). ``_db`` follows the ambient HERMES_HOME that only the inbound message path installs; unscoped background work (expiry watcher) would otherwise write profile rows - into the ROOT store until the stale-route self-heal drops a live conversation.""" + into the ROOT store until the stale-route self-heal drops a live conversation. + + Background work runs unscoped while operating on every profile's keys out of the single process-wide + ``_entries`` dict — ``_session_expiry_watcher`` is the clearest case — so it reads and writes the + ROOT store for rows that actually live under ``profiles//state.db``. The two writers then + drift apart on the same logical session until the routing index disagrees with the row and the + #54878 self-heal drops a live conversation (#66887). + """ pinned = self._pinned_db() if pinned is not _DB_UNPINNED: return pinned @@ -239,7 +256,12 @@ class SessionPersistenceMixin: def _ensure_loaded_locked(self) -> None: """Load the routing index (lock held). state.db ``gateway_routing`` is primary; - sessions.json is the legacy import for keys the DB lacks (persisted on the next _save).""" + sessions.json is the legacy import for keys the DB lacks (persisted on the next _save). + + Read order (#9006 follow-up): the ``gateway_routing`` table in state.db is the primary source; + sessions.json is the legacy import path for pre-migration installs (its entries are folded in for + keys the DB doesn't have, then persisted to the DB on the next _save). + """ if self._loaded: self._reconcile_recovered_routing_locked() return @@ -341,6 +363,9 @@ class SessionPersistenceMixin: return recovered_entry # Same-id recovery == successful resume: keep the ORIGINAL entry object (the recovered one # is rebuilt minimal and would drop counters, model_override, resume markers, metadata). + # A non-None recovery with the SAME session id is a successful resume (all recovery gates passed, + # row reopened): keep the routing entry — it is proven valid, not a dead route (#95957). Nothing in + # sessions.json changes, so no save is needed for this branch. if recovered_entry is not None: logger.info( "gateway.session: reopened ended session %s for sessions.json entry %r " diff --git a/gateway/session_recovery.py b/gateway/session_recovery.py index a35efa7842..ffe341c8fb 100644 --- a/gateway/session_recovery.py +++ b/gateway/session_recovery.py @@ -69,7 +69,13 @@ class SessionRecoveryMixin: ) -> bool: """Prevent a gateway from reviving another profile's row. Single-profile: the row's namespace must match the ACTIVE profile. Multiplexed: it must match the requested key's - namespace (the active profile is meaningless there). Keyless rows stay adoptable.""" + namespace (the active profile is meaningless there). Keyless rows stay adoptable. + + Multiplexed: several profiles serve traffic at once, so the active profile is meaningless — the + requested key carries the profile the turn was routed to, and the recovered row must sit in the same + ``agent::`` namespace (#74285). Rows with no key namespace stay adoptable in both modes + (legacy/keyless data owned by this store). + """ recovered_key = str(recovered.get("session_key") or "") if not recovered_key or recovered_key == requested_session_key: return True @@ -179,6 +185,9 @@ class SessionRecoveryMixin: recoverable, or when the recovered session is already overdue under the reset policy — the row is then durably promoted to a reset boundary instead of resurrected.""" entry, migrated_legacy = self._query_recoverable_row( + # The legacy (pre-workspace) Slack key fallback happens INSIDE _query_recoverable_session + # (#20583/#66398 design): it performs the exact-key legacy lookup, claims the key once per + # process, and rewrites the peer row to the scoped key on success. session_key=session_key, source=source, now=now, raise_on_lookup_error=raise_on_lookup_error) if entry is None: diff --git a/gateway/session_stall.py b/gateway/session_stall.py index b2d06312b1..413176b09c 100644 --- a/gateway/session_stall.py +++ b/gateway/session_stall.py @@ -35,7 +35,10 @@ def should_clear_session_stall_notification( def format_session_stall_notification(idle_seconds: float) -> str: - """User-facing stall warning (ASCII minutes).""" + """User-facing stall warning (ASCII minutes). + + See #72016. + """ mins = max(1, int(idle_seconds // 60)) return f"⚠️ Agent session appears stalled (last activity {mins} min ago). Try /new to reset." @@ -56,7 +59,10 @@ def resolve_session_idle_seconds_from_activity( ) -> Optional[float]: """Idle seconds from a shared activity snapshot: a finite ``seconds_since_activity``, else derived from ``last_activity_at`` / ``last_activity_ts``. None when there is no usable - progress timestamp — callers must not fall back to turn-start or inbound clocks.""" + progress timestamp — callers must not fall back to turn-start or inbound clocks. + + See #72039. + """ if not activity: return None idle = _finite_float(activity.get("seconds_since_activity")) diff --git a/gateway/session_state.py b/gateway/session_state.py index bb8b6e14a8..ba66caf3b2 100644 --- a/gateway/session_state.py +++ b/gateway/session_state.py @@ -66,6 +66,18 @@ class PersistentState: run_generation: int = 0 # monotonic; NEVER reset (stale-run detection depends on it) # Consecutive hygiene compression failures (the in-agent ladder is unreachable: hygiene builds # a FRESH AIAgent per run). Reset on success; process-local, mirrored to the DB by run.py. + # Monotonic run-generation counter (#28686). NEVER reset: clearing it would break stale-run detection. + # The in-agent compressor escalates repeat timeouts via ContextCompressor._consecutive_timeout_failures, + # but hygiene builds a FRESH AIAgent per run and bind_session_state() zeroes that counter, so the + # in-agent ladder is structurally unreachable from the gateway. Tracking the streak here — outside the + # per-run agent — lets hygiene escalate its cooldown instead of retrying on a flat interval forever. + # Reset on a successful compression, not by turn/boundary resets. PROCESS-LOCAL, deliberately: + # `PersistentState` means "survives turn and boundary resets", NOT "survives a restart" — this field has + # no disk flush (unlike `pending_command_text` above, #72680), so a gateway restart drops escalation + # back to rung 1 while the DB-backed deadline itself survives (#74136). Keying on `session_key` rather + # than `session_id` is what buys correctness across compaction ROTATION (the sid changes, the chat does + # not). gateway.run mirrors this value to the DB keyed by session_key so the same semantics also survive + # gateway restarts. hygiene_failure_streak: int = 0 diff --git a/gateway/session_transcript.py b/gateway/session_transcript.py index 82ad454391..c36e5856b6 100644 --- a/gateway/session_transcript.py +++ b/gateway/session_transcript.py @@ -126,6 +126,10 @@ class SessionTranscriptMixin: """Queue *message* (retry lock held); evicts + spools the oldest past the cap.""" pending = self._dirty_transcripts.setdefault(session_id, []) pending.append(dict(message)) + # Cap pending messages per session to avoid unbounded memory growth when the DB is persistently + # broken. Spool the evicted oldest message to the on-disk pending spool (same machinery + # flush_pending_to_file uses at shutdown) so a runtime cap rotation does not silently discard it + # (#78182); it is replayed on the next successful transcript flush. if len(pending) > self._MAX_PENDING_PER_SESSION: spool_path = _spool_dropped(session_id, pending.pop(0)) if spool_path is not None: @@ -214,7 +218,11 @@ class SessionTranscriptMixin: def _append_to_transcript_serialized(self, session_id: str, message: Dict[str, Any]) -> None: """Append a message to a session's transcript (SQLite), draining the per-session retry - queue.""" + queue. + + Args: skip_db: When True, skip the SQLite write. Used when the agent already persisted messages to + SQLite via its own _flush_messages_to_session_db(), preventing the duplicate-write bug (#860). + """ with self._transcript_retry_lock: pending = self._enqueue_transcript_message(session_id, message) msg = pending[0] @@ -297,6 +305,7 @@ class SessionTranscriptMixin: msg = pending[0] if queue_empty: # Backlog clear: replay cap-dropped messages spooled to disk. + # See #78182. self._drain_spooled_drops(session_id) return continue @@ -341,6 +350,8 @@ class SessionTranscriptMixin: # persistence path or the next replay diverges. api_content=extract_api_content_sidecar(message), # Presentation typing (e.g. "internal_notification"); DB-only. + # "internal_notification" for self-injected async-delegation/background notification turns, + # #82888). DB-only; stripped from provider-bound payloads. display_kind=message.get("display_kind"), display_metadata=message.get("display_metadata"), ) @@ -350,7 +361,12 @@ class SessionTranscriptMixin: """True only when the failure is provably scoped to the FTS index. A bare SQLITE_CORRUPT can mean structural B-tree damage; only errors naming ``messages_fts`` or carrying FTS provenance (``SessionDB._is_fts_write_corruption_error``) may authorize the one-shot - rebuild-and-retry; everything else takes the retry path.""" + rebuild-and-retry; everything else takes the retry path. + + A generic ``database disk image is malformed`` (bare SQLITE_CORRUPT) can mean structural damage to + canonical B-trees, not just the FTS shadow tables — treating it as FTS-only here made the store + rebuild the index and retry transcript writes against a structurally corrupt database (#97940). + """ if "messages_fts" in str(exc).lower(): return True import sqlite3 @@ -390,7 +406,11 @@ class SessionTranscriptMixin: self._transcript_append_failures.pop(session_id, None) def has_platform_message_id(self, session_id: str, platform_message_id: str) -> bool: - """Whether a message with this platform_message_id is persisted (False without a DB).""" + """Whether a message with this platform_message_id is persisted (False without a DB). + + Thin wrapper over SessionDB.has_platform_message_id(). Returns False when no DB is available + (in-memory sessions). Used by the gateway's transient-failure dedupe guard (#47237). + """ db = self._db_for_session_id(session_id) if not db: return False @@ -414,6 +434,17 @@ class SessionTranscriptMixin: return True with self._get_transcript_drain_lock(): try: + # Even when the current agent doesn't "own" persistence, the session on disk may already + # carry compaction-archived rows — e.g. after a model switch or a /restore, both of which + # mint a fresh agent with _session_db_created=False (so the check above is False) yet leave + # the durable archived transcript in place. A full-history replace would DELETE those + # archived rows just like the owned-agent case. Guard against it by replacing ONLY the live + # (active=1) set unconditionally: on a fresh create/fork every row is active=1, so + # active-only replace is behaviorally identical to the full replace — and when archived rows + # DO exist they survive. An existence probe here (has_archived_messages) would fail OPEN + # into the destructive replace on any DB error and can race a concurrent archive_and_compact + # — the same probe failure mode #80216's /retry fix (gateway/slash_commands.py) deliberately + # avoids. db.replace_messages( session_id, messages, active_only=active_only, reject_active_turn_lease=reject_active_turn_lease) diff --git a/gateway/shutdown_flush.py b/gateway/shutdown_flush.py index 1f58fa1b84..960463c23a 100644 --- a/gateway/shutdown_flush.py +++ b/gateway/shutdown_flush.py @@ -25,6 +25,7 @@ logger = logging.getLogger(__name__) # Reason tag for transcript messages dropped by the in-memory pending cap during live # operation. Payloads carry the full transcript message dict for verbatim replay. +# See #78182. TRANSCRIPT_CAP_DROP_REASON = "transcript_cap_drop" # Monotonic tiebreaker so same-second spool files replay in drop order. _TRANSCRIPT_SPOOL_SEQ = itertools.count() @@ -112,7 +113,12 @@ def flush_overflow_to_file(overflow_by_session: Dict[str, Any], *, reason: str = def spool_dropped_transcript_message(session_id: str, message: Dict[str, Any]) -> Optional[Path]: - """Spool a cap-evicted transcript message; ``None`` on failure (callers degrade to drop+log).""" + """Spool a cap-evicted transcript message; ``None`` on failure (callers degrade to drop+log). + + Uses the same on-disk pending spool as :func:`flush_pending_to_file` (one atomic JSON payload per + message under ``/pending_messages/``), so a runtime cap rotation no longer silently + discards user data while the process stays up (#78182). + """ try: return _write_payload(_get_flush_dir(), { "session_key": session_id, "reason": TRANSCRIPT_CAP_DROP_REASON, "ts": int(time.time()), @@ -227,6 +233,8 @@ def recover_pending_to_db(session_db=None) -> int: def _recover_one_payload(session_db, path: Path, payload: Dict[str, Any]) -> bool: """Append one flush payload to ``session_db``; False (file kept) when structurally invalid.""" + # Cap-dropped transcript payloads carry the full message dict keyed by session_id — replay directly + # (#78182). This handles spool files that were never drained before a restart. if payload.get("reason") == TRANSCRIPT_CAP_DROP_REASON: # Cap-dropped payloads carry the full message dict keyed by session_id — replay directly. data = payload.get("data", {}) or {} diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 19801e1a11..547c471957 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -225,6 +225,8 @@ class GatewaySlashCommandsMixin( correct for the write-back round-trip (merged defaults must not be persisted back to the user's file); the cached agent is dropped so the setting takes effect next message.""" from gateway.run import _gateway_config_home + # Persist to config (default) unless --session opted out, mirroring the text /model command path + # above so a picked model survives across sessions like a typed one (#49066). from hermes_cli.config import read_user_config_raw config_path = _gateway_config_home() / "config.yaml" session_key = self._session_key_for_source(event.source) @@ -233,6 +235,10 @@ class GatewaySlashCommandsMixin( user_config = read_user_config_raw(config_path) user_config.setdefault(section, {})["write_approval"] = bool(enabled) atomic_config_write(config_path, user_config) + # Evict any cached agent for this session so the next message rebuilds with the correct + # session_id end-to-end — mirrors /branch and /reset. Without this, the cached AIAgent (and its + # memory provider, which cached `_session_id` during initialize()) keeps writing into the wrong + # session's record. See #6672. self._evict_cached_agent(session_key) return _set_approval @@ -429,6 +435,7 @@ class GatewaySlashCommandsMixin( # No running agent anywhere for this scope. A platform status indicator can still be stuck — # e.g. Slack's persistent assistant.threads.setStatus survives a gateway restart or a turn # that died without a final send. + # Best-effort clear so /stop always dismisses a phantom "is thinking...". See #32295. adapter = getattr(self, "adapters", {}).get(source.platform) try: if adapter and hasattr(adapter, "_stop_typing_with_metadata"): @@ -535,6 +542,9 @@ class GatewaySlashCommandsMixin( from gateway.restart import is_container_restart_context, is_gateway_supervisor_process via_service = is_gateway_supervisor_process() or is_container_restart_context() self.request_restart(detached=not via_service, via_service=via_service) + # Track sessions that were active at shutdown for stuck-loop detection (#7536). On each restart, the + # counter increments for sessions that were running. If a session hits the threshold (3 consecutive + # restarts while active), the next startup auto-suspends it — breaking the loop. if active_agents: return t("gateway.draining", count=active_agents) return EphemeralReply(t("gateway.restart.restarting")) @@ -603,6 +613,7 @@ class GatewaySlashCommandsMixin( # Voice state belongs to the (bot, chat) pair: resolve the adapter that received the # command and key the mode by its owning profile so two multiplexed bots in one chat keep # independent /voice state. + # See #75198. voice_key = self._voice_key_for_source(event.source) adapter = self._adapter_for_source(event.source) @@ -1127,7 +1138,11 @@ class GatewaySlashCommandsMixin( async def _handle_deny_command(self, event: MessageEvent) -> str: """Handle /deny — reject pending dangerous command(s) with a definitive BLOCKED result, as in - the CLI. ``/deny`` denies the oldest; ``/deny all`` denies everything.""" + the CLI. ``/deny`` denies the oldest; ``/deny all`` denies everything. + + ``/deny `` (or ``/deny all ``) attaches a one-line reason that is relayed back to + the agent so it can adapt instead of only hearing "denied". Ported from qwibitai/nanoclaw#2832. + """ from tools.approval import resolve_gateway_approval session_key, stale = self._blocking_approval_or_stale(event, "gateway.deny.stale", "gateway.deny.no_pending") diff --git a/gateway/slash_commands_model.py b/gateway/slash_commands_model.py index 1a42095bfa..4337db0c82 100644 --- a/gateway/slash_commands_model.py +++ b/gateway/slash_commands_model.py @@ -234,6 +234,7 @@ class GatewayModelCommandsMixin: """Persist a committed switch: session DB, next-turn note, override map, config write-through.""" from hermes_cli.model_switch import format_model_for_display + # Persist the new model to the session DB so the dashboard shows the updated model (#34850). _sess_db = getattr(self, "_session_db", None) if _sess_db is not None: # so the dashboard shows the updated model try: @@ -241,6 +242,7 @@ class GatewayModelCommandsMixin: # Typed path: consume the auto-reset flag so the next message's cleanup does not # wipe the override stored below. if not picker and getattr(_sess_entry, "was_auto_reset", False): + # See #48031. _sess_entry.was_auto_reset = False await _sess_db.update_session_model( _sess_entry.session_id, result.new_model, provider=result.target_provider, @@ -273,6 +275,13 @@ class GatewayModelCommandsMixin: self._pending_one_turn_model_restores.pop(ctx.session_key, None) # Non-secret write-through so the override survives a restart (api_key/api_mode are # re-resolved on rehydration); a --once override must NOT outlive a restart. + # Write-through the non-secret parts (model/provider/base_url) to the session store so the override + # survives a gateway restart. api_key/api_mode are never persisted — they are re-resolved via + # runtime provider resolution on rehydration. /model --once is intentionally EXCLUDED from the + # write-through: a one-turn override must never survive a restart. The persisted value stays at the + # pre-once state (the prior session override, or nothing), which is exactly what the finally-restore + # reverts the in-memory dict to. (#29923 review defect: the original implementation wrote through, + # so a crash before the restore rehydrated the once-model permanently.) if not one_turn: try: await self.async_session_store.set_model_override( @@ -355,6 +364,10 @@ class GatewayModelCommandsMixin: is session-key-normalized so the picker's thread metadata lands where the next turn reads.""" from hermes_cli.model_switch import list_picker_providers try: # off-loop: listing can hit a synchronous HTTP fetch on a stale cache + # Offload blocking provider-listing (can fall through to a synchronous urllib HTTP fetch on a + # stale cache) off the event loop so the gateway doesn't freeze. See #41289. + # Offload blocking provider-listing off the event loop so the gateway doesn't freeze on a + # stale-cache HTTP fetch. See #41289. providers = await asyncio.to_thread( list_picker_providers, max_models=50, include_moa=True, **listing_kwargs ) @@ -466,9 +479,22 @@ class GatewayModelCommandsMixin: clear_provider_models_cache() # Normalize like a message turn (Telegram DM topic recovery) before deriving the override # key, so the override lands under the key the next turn reads. + # Check for session override. See #30479. source = await asyncio.to_thread(self._normalize_source_for_session_key, event.source) session_key = self._session_key_for_source(source) ctx = _ModelSwitchContext( + # Gateway routing columns — forward ALL of them at CREATE time, same fix as the + # compression-rotation bug in agent/conversation_compression.py. Without these, the branched + # child row has NULL routing columns until switch_session() below calls + # _record_gateway_session_peer() — a crash/kill anywhere between here and there (most plausibly + # mid-history-copy, since each append_message call a few lines down is independently + # best-effort) leaves the branch permanently unroutable: unreachable by chat/thread lookup, and + # unreachable via /resume's IDOR guard too (which requires the row's chat_id/thread_id to match + # the caller's). user_id is critical for the fallback lookup path (hermes_state.py:1994-2009) + # that searches by the complete peer tuple when session_key doesn't match. origin_json and + # display_name complete the identity (same shape as the reset path's db_create_kwargs in + # gateway/session.py, #82633) so consumers that read routing/presentation data from state.db + # (mcp_serve, mirror, channel directory) see the branch row fully formed with zero backfill gap. session_key=session_key, source=source, config_path=(profile_home or _hermes_home) / "config.yaml", @@ -638,6 +664,7 @@ class GatewayModelCommandsMixin: raw_args = event.get_command_args().strip() args, persist_global = self._parse_reasoning_command_args(raw_args) # Normalize (Telegram DM topic recovery) so the override key matches the next turn's. + # See #30479. _reasoning_source = await asyncio.to_thread(self._normalize_source_for_session_key, event.source) session_key = self._session_key_for_source(_reasoning_source) self._show_reasoning = self._load_show_reasoning() diff --git a/gateway/slash_commands_session.py b/gateway/slash_commands_session.py index 8e3c7764d4..e48850a79e 100644 --- a/gateway/slash_commands_session.py +++ b/gateway/slash_commands_session.py @@ -157,6 +157,10 @@ class GatewaySessionCommandsMixin: # drops all later messages. Idempotent, so the run's finally calling it again is harmless. self._release_running_agent_state(session_key) # Snapshot the old entry so on_session_finalize can report the expiring session id. + # Evict the running-agent slot now that the generation is bumped. The in-flight run's own guarded + # release (run_generation=old) will return False and leave its dead agent behind; clearing here + # keeps the slot from becoming a zombie that silently drops all later messages (#28686). Idempotent, + # so the run's finally calling it again is harmless. old_entry = self.session_store._entries.get(session_key) await self._cleanup_old_agent_for_reset(session_key) self._evict_cached_agent(session_key) @@ -401,6 +405,8 @@ class GatewaySessionCommandsMixin: # archives it and inserts the pure scaffold atomically, reselecting the latest carrier # on the same snapshot so a concurrent newer turn is never removed for stale text. try: + # Plain turns keep the existing rewrite path below; #84078 owns its separate + # archive_dropped/prefix-CAS semantics. rewind_result = await self.async_session_store.rewind_session( session_entry.session_id, 1, require_retryable_composite=True) except ValueError as exc: @@ -458,7 +464,10 @@ class GatewaySessionCommandsMixin: async def _compress_codex_app_server_session(self, session_key: str, session_id: str) -> str: """Manual /compress for codex_app_server sessions: compacts the LIVE cached agent's app-server thread (``force=True`` bypasses the ``codex_app_server_auto`` gate) and keeps it - cached. A temporary agent or a mirror rewrite cannot shrink the server-side thread.""" + cached. A temporary agent or a mirror rewrite cannot shrink the server-side thread. + + See #73503. + """ from gateway.run import _AGENT_PENDING_SENTINEL agent = self._cached_agent_for(session_key) @@ -554,6 +563,8 @@ class GatewaySessionCommandsMixin: tmp_agent = await self._build_manual_compression_agent(session_entry.session_id, model, runtime_kwargs) try: # Estimate with system prompt + tool schemas (real request pressure); needs the built agent. + # Must be computed after tmp_agent is built so _cached_system_prompt/tools are populated. See + # #6217. _sys_prompt = getattr(tmp_agent, "_cached_system_prompt", "") or "" _tools = getattr(tmp_agent, "tools", None) or None approx_tokens = estimate_request_tokens_rough(msgs, system_prompt=_sys_prompt, tools=_tools) @@ -844,6 +855,7 @@ class GatewaySessionCommandsMixin: return t("gateway.resume.not_found", name=name) # Follow compression continuations to the live transcript (matches CLI /resume). try: + # Follow that chain so gateway /resume matches CLI behavior (#15000). target_id = await self._session_db.resolve_resume_session_id(target_id) except Exception as e: logger.debug("Failed to resolve resume continuation for %s: %s", target_id, e) @@ -901,6 +913,9 @@ class GatewaySessionCommandsMixin: if not new_entry: return t("gateway.resume.switch_failed") # Conversation boundary: all conversation-scoped state + security state in one funnel call. + # Conversation boundary: clear ALL conversation-scoped per-session state (model/reasoning overrides + # #10702, one-turn restores, model notes, last-resolved cache #58403, /queue overflow) + security + # state in one funnel call. See _CONVERSATION_SCOPED_STATE in gateway/run.py. self._clear_conversation_scope(session_key, reason="resume") # Evict so the next turn rebuilds with the right session_id — the cached AIAgent's memory # provider cached _session_id at initialize() and would keep writing to the wrong session. @@ -1018,6 +1033,7 @@ class GatewaySessionCommandsMixin: parent_session_id = current_entry.session_id # Full parent origin (same shape as the reset path in gateway/session.py); the live entry's # origin may hold richer metadata than the triggering event's source. + # See #82633. _branch_origin_json = None with contextlib.suppress(Exception): _branch_origin_json = _json.dumps((current_entry.origin or source).to_dict()) @@ -1040,6 +1056,9 @@ class GatewaySessionCommandsMixin: # Chunked transactions; best-effort — a failed copy still yields a usable (partial) branch. with contextlib.suppress(Exception): + # Copy conversation history to the new session in bounded-chunk transactions (see #23254): one + # txn per row was the removed write-amplification pattern, and a history can be hundreds of + # rows. await self._session_db.append_messages_batch( new_session_id, [_branch_row(msg) for msg in history], chunk_rows=500) with contextlib.suppress(Exception): diff --git a/gateway/slash_commands_status.py b/gateway/slash_commands_status.py index 90915903cb..c467971c52 100644 --- a/gateway/slash_commands_status.py +++ b/gateway/slash_commands_status.py @@ -398,6 +398,8 @@ class GatewayStatusCommandsMixin: if hasattr(task, "done") and not task.done()] # Background (async) delegations — delegate_task(background=true). + # Live per-child activity comes from the registry's progress sampler (#51690): api calls, current + # tool, seconds since last activity. from tools.async_delegation import list_async_delegations delegations = [d for d in _quiet_sync(list_async_delegations, []) if d.get("status") in ("running", "stalling", "finalizing")] diff --git a/gateway/status.py b/gateway/status.py index 11277079d8..440093e6a9 100644 --- a/gateway/status.py +++ b/gateway/status.py @@ -215,7 +215,12 @@ def terminate_pid( ) -> None: """Terminate a PID; POSIX SIGTERM/SIGKILL, Windows taskkill /T /F for force. Identity guard: Windows ``force`` REQUIRES a matching ``expected_start_time`` (taskkill on a recycled PID has - killed svchost.exe); POSIX optional, but a provided mismatch refuses the kill everywhere.""" + killed svchost.exe); POSIX optional, but a provided mismatch refuses the kill everywhere. + + On POSIX an expectation is optional, but when the caller provides one and it no longer matches the live + process, the kill is refused on every platform — a mismatched fingerprint always means the PID was + recycled. See #89614. + """ if force and (_IS_WINDOWS or expected_start_time is not None): if expected_start_time is None: raise OSError(f"refusing to force-kill PID {pid} without a process start-time guard") @@ -421,7 +426,13 @@ def _build_pid_record() -> dict: def _get_code_identity_fields() -> dict[str, Any]: """Code identity of THIS process for ``gateway_state.json`` (restart picked up new code?). - Lazy import keeps ``gateway.status`` free of ``hermes_cli`` at import time. Never raises.""" + Lazy import keeps ``gateway.status`` free of ``hermes_cli`` at import time. Never raises. + + A gateway keeps serving the module versions it imported at startup, so stamping the identity into + ``gateway_state.json`` lets `hermes update` (and the dashboard) prove whether a running gateway actually + picked up new code after the restart phase — instead of assuming it did (#88654, #69754). Never raises; + degrades to absent fields. + """ try: from hermes_cli.build_info import get_code_identity identity = get_code_identity() @@ -557,6 +568,14 @@ def _pid_exists(pid: int) -> bool: import psutil # type: ignore # Best-effort zombie check: status-read failures fall through to pid_exists(). try: + # A zombie (defunct) process is still in the process table, so ``psutil.pid_exists()`` returns + # True for it — but it is already dead: SIGKILL has no effect and it cannot be a running + # gateway. Treating a zombie as alive makes ``--replace`` wait for the old PID to die (it never + # does, until its parent reaps it), then abort with exit 1 — a silent crash loop under systemd + # ``Restart=always``, which respawns the gateway before reaping the previous process (issue + # #42126). Report zombies as dead so the takeover proceeds. Best-effort: any failure to read + # status (partial/stub psutil, access denied, transient race) falls through to the authoritative + # ``pid_exists()`` below rather than raising. if psutil.Process(pid).status() == psutil.STATUS_ZOMBIE: return False except getattr(psutil, "NoSuchProcess", ()): @@ -586,6 +605,13 @@ def _posix_is_zombie(pid: int) -> bool: return len(stat_fields) > 2 and stat_fields[2] == "Z" except FileNotFoundError: with contextlib.suppress(Exception): + # --compile-bytecode: uv does NOT write __pycache__ by default (pip does), so without it the + # first `import ` in the foreground of a user request recompiles every module of the + # backend *and* its transitive deps (#100461). This covers the whole install; + # _warm_installed_bytecode below is the belt-and-braces pass for the spec's own roots on any + # tier. + # CREATE_NO_WINDOW on Windows — under the desktop GUI's windowless parent, this spawn otherwise + # flashes a console (#56747). r = subprocess.run( ["ps", "-o", "state=", "-p", str(pid)], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=5, @@ -892,7 +918,15 @@ def resolve_gateway_liveness( in another container); (3) LOCAL runtime status PID validated against the live process table with ``expected_home`` (a recycled PID of another profile never counts; pass ``runtime`` if already read). ``*_probe``/``runtime_reader`` are the dashboard's injection/test seam. A rung - that raises degrades to the next (never 500 a status endpoint) and sets ``probe_error``.""" + that raises degrades to the next (never 500 a status endpoint) and sets ``probe_error``. + + Before this existed, ``/api/status`` and ``/api/messaging/platforms`` each open-coded their own ladder + and disagreed on the same page load — the sidebar read "running" while the Channels page rendered "The + gateway is not running." Three deployments hit it: a cross-container gateway (only ``/api/status`` ran + the HTTP health probe), a profile-scoped dashboard (only ``/api/status`` passed the profile's paths, so + messaging borrowed another profile's runtime state — issue #71211), and a launch-service-managed gateway + with no PID file (only some callers used the runtime-status fallback). + """ _pid_probe = pid_probe or (get_running_pid_cached if use_cache else get_running_pid) _runtime_reader = runtime_reader or read_runtime_status _runtime_pid_probe = runtime_pid_probe or get_runtime_status_running_pid @@ -1024,6 +1058,12 @@ def acquire_scoped_lock( existing_pid = _pid_from_record(existing) # Our own PID: always self-reacquire. start_time guards reuse of OTHER PIDs; requiring # equality here rejects reconnects when the on-disk record has start_time null. + # Same live PID as this process: always self-reacquire. ``start_time`` is a PID-reuse guard for + # *other* PIDs; it cannot distinguish two processes that share the caller's own PID (impossible + # while we are alive). Requiring start_time equality here falsely rejects reconnects when the + # on-disk record has ``start_time: null`` (older writers / psutil failure at first write) while the + # freshly built record has a real value — the gateway then reports itself as the foreign squatter of + # its own token (#81468). if existing_pid == os.getpid(): _write_json_file(lock_path, record) return True, existing @@ -1080,6 +1120,8 @@ def release_all_scoped_locks( # SIGTERM; the target's shutdown handler treats a matching marker as a planned takeover and # exits 0. Unlinked once consumed, so a stale one can grief at most one future shutdown on # the same PID, within _TAKEOVER_MARKER_TTL_S. +# When a new gateway starts with ``--replace``, it SIGTERMs the existing gateway so it can take over the bot +# token. ``hermes.service`` + ``hermes- gateway.service``). See #5646. _TAKEOVER_MARKER_FILENAME = ".gateway-takeover.json" _TAKEOVER_MARKER_TTL_S = 60 # Marker older than this is treated as stale _PLANNED_STOP_MARKER_FILENAME = ".gateway-planned-stop.json" @@ -1137,6 +1179,7 @@ def _consume_pid_marker_for_self(path: Path, *, ttl_s: int) -> bool: # Cross-profile guard: new markers name the verified TARGET home, which permits a deliberate # cross-HERMES_HOME --replace while ignoring a marker accidentally written into another # profile's directory. Legacy markers have no target field: keep the same-replacer-home rule. + # See #29092. our_home = _get_process_hermes_home() target_home = record.get("target_hermes_home") if target_home is not None: diff --git a/gateway/stream_consumer.py b/gateway/stream_consumer.py index 8918e0ad27..872f002068 100644 --- a/gateway/stream_consumer.py +++ b/gateway/stream_consumer.py @@ -68,6 +68,9 @@ class StreamConsumerConfig: buffer_only: bool = False # >0: final goes out as a fresh message once the preview has been visible this # long (timestamp reflects completion); 0 = always edit in place. + # This makes the platform's visible timestamp reflect completion time instead of first-token time for + # long-running responses (e.g. reasoning models that stream slowly). Ported from + # openclaw/openclaw#72038. The gateway enables this selectively per-platform. fresh_final_after_seconds: float = 0.0 # "auto"/"draft": native drafts when adapter+chat support it, else "edit" # (progressive editMessageText). "off" is handled by the gateway. @@ -140,6 +143,9 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi # Every real preview id on screen this response (fresh-final deletes them all); # the per-segment set holds only the active segment so failure recovery never # deletes an earlier finalized preamble/commentary. + # Wall-clock timestamp (time.monotonic) when ``_message_id`` was first assigned from a successful + # first-send. Used by the fresh-final logic to detect long-lived previews whose edit timestamps + # would be stale by completion time. Ported from openclaw/openclaw#72038. self._preview_message_ids: "set[str]" = set() self._already_sent = False self._edit_supported = True # False once progressive edits stop working @@ -199,10 +205,17 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi may carry a stale preview); None = legacy trust. A payload-less ``_turn_split_delivery`` must NOT inherit legacy trust; ``_delivery_ambiguous`` (a full-final send timed out but MAY have landed) is the only case that does.""" + # #29346: a tool/segment boundary means what we delivered was an interim preamble, not the final + # answer — clear the flags so a premature setter can't fool the gateway. Safe: got_done returns + # before any reset, and run.py reads these only after the consumer task exits. self._final_response_sent = False self._final_content_delivered = False # content landed even if the cosmetic edit failed self._delivered_final_text: Optional[str] = None self._turn_split_delivery = False + # True when a full-final send timed out in a way that MAY have reached the platform + # (``_send_empty_fallback_final`` → "ambiguous"). The only case where a payload-less delivery flag + # keeps legacy trust in ``delivered_final_matches`` (#95382 tightening) — re-sending there risks a + # duplicate rather than recovering a loss. self._delivery_ambiguous = False def _stream_is_message(self) -> bool: @@ -284,6 +297,12 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi def _mark_final_delivered(self, record: Optional[str] = None) -> None: """Set both turn-final flags; ``record`` also records the delivered payload.""" self._final_response_sent = True + # Only claim final delivery if the sealed chunks and final tail actually landed. ``_already_sent`` + # may be True from prior progress/fallback state (#10748). + # The final clean-up edit failed, but the complete answer is already visible from the last streaming + # frame (usually with only the cursor still stuck on screen). Mark the content delivered so the + # gateway suppresses its normal full final send; otherwise users see the same long answer twice when + # Telegram/Discord rate-limit this cosmetic final edit (#36965, #25349). self._final_content_delivered = True if record is not None: self._record_turn_final_payload(record) @@ -318,6 +337,13 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi # No recorded payload: judge against the FINAL content, not the flag. # ``_already_sent`` gates the match: draft frames set ``_last_sent_text`` but # deliberately not ``_already_sent``. + # #95382 / #98552 class fix: a delivery flag with NO recorded payload must still be judged against + # the FINAL content, not trusted blindly. Every internal flag-setting site records a payload; a + # record-less consumer whose visible/streamed text does not contain the completed response has + # demonstrably NOT delivered it (first-edit prefix, mid-stream truncation) — the flag alone must not + # suppress the corrective send. ``_already_sent`` gates the visible-text match: draft frames set + # ``_last_sent_text`` for dedupe but are ephemeral (they deliberately do not set ``_already_sent``), + # so draft-only visibility must not count as durable delivery. if self._already_sent and self.has_delivered_text(final_text): return True # Only a timed-out full-final send that MAY have landed keeps legacy trust. @@ -583,6 +609,11 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi return self._use_native_streaming = False self._use_draft_streaming = self._resolve_draft_streaming() + # Native draft streaming: bump the draft_id so the next text segment animates as a fresh preview + # below the tool-progress bubbles, not over the prior segment's already-finalized draft. This is how + # we avoid the "inter-tool-call text leak" failure mode openclaw documented in their issue #32535 — + # each text block becomes its own visible message via the finalize, then a new draft animates for + # the next one. if self._use_draft_streaming: self._bump_draft_id() logger.debug("Stream consumer using native-draft transport (chat=%s draft_id=%s)", @@ -898,6 +929,11 @@ class GatewayStreamConsumer(StreamTransportMixin, StreamFallbackMixin, StreamThi self._signal_flush(item[1]) @staticmethod + # Strip MEDIA: tags before display. Uses the shared anchored MEDIA_TAG_CLEANUP_RE from + # gateway/platforms/base.py — only tags whose path ends in a deliverable extension are removed, so an + # unknown-extension path stays visible instead of being silently dropped (issue #34517). Streaming and + # non-streaming paths share the same regex, so a tag is treated identically whichever path delivered the + # text. def _clean_for_display(text: str) -> str: """Hide MEDIA: / [[audio_as_voice]] directives; media is delivered post-stream.""" return _BasePlatformAdapter.strip_media_directives_for_display(text) diff --git a/gateway/stream_consumer_fallback.py b/gateway/stream_consumer_fallback.py index 1439277ef9..e836280923 100644 --- a/gateway/stream_consumer_fallback.py +++ b/gateway/stream_consumer_fallback.py @@ -151,6 +151,10 @@ class StreamFallbackMixin: # "failed" lets the gateway perform its normal final send. self._final_content_delivered = delivery in {"ambiguous", "preview"} if delivery == "preview": + # This branch is only reached when the ACKed preview already shows the complete final text + # (final_text == _visible_prefix()), so record it as the turn-final payload: the gateway's + # reconciliation then confirms delivery instead of re-sending a second bubble next to the + # never-deleted preview (#71047 Problem B). self._record_turn_final_payload(final_text) elif delivery == "ambiguous": self._delivery_ambiguous = True @@ -327,6 +331,9 @@ class StreamFallbackMixin: if result.success: self._notify_new_message() # Lets run.py confirm whether an interim send carried the final. + # Record the exact delivered text so run.py can confirm whether an interim "preview" + # actually carried the final response, vs. unrelated commentary delivered during a session + # split (#14238). self._delivered_commentary_texts.append(text) return result.success except Exception as e: @@ -353,6 +360,8 @@ class StreamFallbackMixin: return cap return base + # Fresh send carried exactly ``text`` — record it so the gateway can reconcile the flag against the + # completed response (#71643/#95382 content-vs-flag contract). async def _suppress_silence_marker(self) -> None: """Retract any streamed preview when the final reply is a bare silence marker. Flags stay False: the gateway's whole-response filter turns the marker into "" so no diff --git a/gateway/stream_consumer_transport.py b/gateway/stream_consumer_transport.py index 2cfed0a686..c01bd34d61 100644 --- a/gateway/stream_consumer_transport.py +++ b/gateway/stream_consumer_transport.py @@ -109,6 +109,10 @@ class StreamTransportMixin: try: deleted = await delete_fn(self.chat_id, stale_id) if retry_on_false and deleted is False: + # Telegram's delete_message reports failure by returning False, not raising. The same + # flood window that broke the finalize edit can reject this delete too, leaving the + # preview bubble next to the fresh final (#71047 Problem B). One short bounded retry + # clears the common transient case; a second failure stays best-effort. await asyncio.sleep(1.0) await delete_fn(self.chat_id, stale_id) except Exception as e: @@ -196,7 +200,10 @@ class StreamTransportMixin: return bool(self._message_id) and self._message_id != "__no_edit__" def _should_send_fresh_final(self) -> bool: - """True when fresh-final is enabled and a real preview has been visible ≥ threshold.""" + """True when fresh-final is enabled and a real preview has been visible ≥ threshold. + + Ported from openclaw/openclaw#72038. + """ threshold = getattr(self.cfg, "fresh_final_after_seconds", 0.0) or 0.0 if threshold <= 0 or not self._has_real_preview() or self._message_created_ts is None: return False @@ -242,7 +249,13 @@ class StreamTransportMixin: async def _try_fresh_final(self, text: str, *, is_turn_final: bool = True) -> bool: """Send ``text`` fresh and best-effort delete the preview(s); False on any failure so - the caller falls back to edit. ``is_turn_final=False`` leaves the delivery flag unset.""" + the caller falls back to edit. ``is_turn_final=False`` leaves the delivery flag unset. + + ``is_turn_final`` is False when finalizing an interim segment at a tool boundary (a preamble) rather + than the turn-final answer; the final-delivery flag is then left unset so the gateway still delivers + the real answer from the next API call (#29346). + Ported from openclaw/openclaw#72038. + """ # Replacing every preview is only sound while ``text`` holds the whole answer; # after a split, deleting sealed heads would erase delivered text. if self._turn_split_delivery: @@ -482,6 +495,10 @@ class StreamTransportMixin: # twice, and record the on-screen payload. self._final_content_delivered = True self._record_turn_final_payload(text) + # ``text`` is already cleaned/fence-closed here and equals the visible prefix — the on-screen + # content IS this finalize payload (#71643). Record it on split turns too: post-#78541 an unrecorded + # split reads as a mismatch and would re-send this already-visible answer, reintroducing the + # duplicate #45517 fixed (#36965 / #25349). raw_response = getattr(result, "raw_response", None) if isinstance(raw_response, dict) and raw_response.get("partial_overflow"): # Some overflow chunks landed but not the whole response: preserve the diff --git a/gateway/turn_context.py b/gateway/turn_context.py index 4a0f6cc84e..bfe807ca65 100644 --- a/gateway/turn_context.py +++ b/gateway/turn_context.py @@ -54,6 +54,7 @@ class TurnContext: persist_user_message: Optional[Any] = None persist_user_timestamp: Optional[float] = None # display_kind of the persisted user row for a self-injected turn; DB-only, never sent. + # "internal_notification" for async-delegation/background notifications (#82888). persist_user_display_kind: Optional[str] = None user_config: Any = None enabled_toolsets: Any = None @@ -84,6 +85,10 @@ class TurnContext: _event_callback_sync: Optional[Callable] = None _status_callback_sync: Optional[Callable] = None # Slack-native task cards (opt-in); ID-bearing callbacks correlate start/complete by call ID + # --- Slack-native task-card progress (opt-in; #29483) ------------------ True when the Slack adapter's + # ``native_task_cards_enabled()`` opt-in is set for this turn's platform. The ID-bearing lifecycle + # callbacks are published by TurnRunner (like voice_ack_callback above) so tool starts and completions + # correlate by real tool-call ID instead of tool name. _native_slack_task_cards: bool = False native_tool_start_callback: Optional[Callable] = None native_tool_complete_callback: Optional[Callable] = None diff --git a/gateway/wake.py b/gateway/wake.py index aa4bc84b98..6c383a46e8 100644 --- a/gateway/wake.py +++ b/gateway/wake.py @@ -74,7 +74,16 @@ def _delegation_display_metadata(evt: dict) -> dict: async def persist_delegation_delivery(adapter: Any, *, text: str, session_id: str, evt: Optional[dict] = None) -> None: """Persist an async-delegation completion as a durable DELIVERY row (see module docstring) WITHOUT running any agent turn. Raises on failure so the caller can release the durable claim - and retry.""" + and retry. + + 85957: on stateless api_server sessions the client owns the turn after ``event.complete`` — a completion + must never become a new ``role=user`` prompt via the self-post (that starts an unauthorized agent turn + and can cross a pending human-confirmation gate). Instead, append the completion to the session + transcript as a timeline bookkeeping row (``display_kind="async_delegation_complete"`` + display + metadata — the exact shape the TUI/desktop delivery path persists), WITHOUT running any agent turn. + Clients polling ``GET /api/sessions/{id}/messages`` see it immediately; the pre-request repair belt + folds it into the next real client turn as context. See #85957. + """ if not session_id: raise ValueError("persist_delegation_delivery: raw session id required to persist " "the completion on the api_server session transcript") diff --git a/hermes_cli/_early_recovery.py b/hermes_cli/_early_recovery.py index ad29b2ce14..1eb0d21ed1 100644 --- a/hermes_cli/_early_recovery.py +++ b/hermes_cli/_early_recovery.py @@ -20,6 +20,7 @@ from pathlib import Path # metadata but wiped import files. ``module`` is probed via a real import; ``attr`` guards against # an empty/stub module. main.py's marker-recovery path reuses these tables — keep them here so # both layers probe and repair the same set. +# See #57828. LAZY_REFRESH_IMPORT_PROBES: tuple[tuple[str, str], ...] = ( ("yaml", "SafeDumper"), ("dotenv", "load_dotenv"), ("click", "Command"), ("certifi", "contents"), ("rich", "print"), ("cryptography", "__version__"), @@ -36,6 +37,10 @@ LAZY_REFRESH_REPAIR_PACKAGES: dict[str, str] = { # leaves no ``hermes`` on PATH, and the command that would repair it IS ``hermes update``. The # updater, the early-recovery installer and the startup orphan sweep all restore through this one # stdlib-only helper so the retry ladder and the recovery wording cannot drift apart again. +# --- Windows entry-point shim quarantine ----------------------------------- They used to be separate +# one-shot renames with swallowed errors; the two that had messages had already drifted apart. The logic +# lives here, in the one stdlib-only module all of them can import, so the ladder and the recovery wording +# stay in lockstep. See #75584. QUARANTINE_RESTORE_BACKOFF_MS: tuple[int, ...] = (0, 100, 250, 500, 1000) @@ -280,6 +285,8 @@ def _base_interpreter_is_externally_managed() -> bool: Those ship an ``EXTERNALLY-MANAGED`` marker next to their stdlib (PEP 668), so ``python -m pip install`` aborts with ``externally-managed-environment``; the early repair must then go through uv (or explicitly override pip) or the venv stays broken. + + See #83569. """ try: import sysconfig @@ -371,6 +378,9 @@ def recover_if_needed(project_root: Path | None = None, argv: list[str] | None = # updater inside the marker-to-install window — never race it. A dead owner MUST be # recovered even when this launch is itself `hermes update`: CLI and Desktop retries keep # that argv, and skipping solely on argv recreates the self-lock loop. + # Bounded retries: a persistently failing install must not hammer every launch, so attempts past the + # ceiling are left for main.py's post-import recovery path (which can safely probe-import after this + # process already holds whatever extensions it needs). See #83569. if core_marker.exists(): if _marker_owner_is_live(core_marker): return @@ -446,6 +456,12 @@ def _complete_pending_core_install(root: Path, core_marker: Path) -> bool: Never raises: any failure leaves the marker for the post-import path and returns ``False``. Returns ``True`` only after the install succeeds. + + ``recover_if_needed`` invokes this when ``.update-incomplete`` exists — a prior ``hermes update`` (or + the self-lock preflight, #83569) left the dependency sync deliberately unfinished. Completing it here + matters on Windows: the deferral exists precisely because the process that wrote the marker had a native + venv extension mapped; this process, running before ``hermes_cli.main``'s third-party imports, maps + nothing yet, so the installer can replace ``.pyd`` files without hitting the lock. """ try: from hermes_cli import _install_repair as ir diff --git a/hermes_cli/_install_repair.py b/hermes_cli/_install_repair.py index 763e16c521..743407e691 100644 --- a/hermes_cli/_install_repair.py +++ b/hermes_cli/_install_repair.py @@ -80,6 +80,8 @@ def _resolve_install_target(root: Path) -> tuple[list[str], dict | None]: def _venv_scripts_dir(root: Path) -> Path | None: """Project venv Scripts/bin dir, when present (hermes_constants is stdlib-only).""" + # hermes_constants is stdlib-only, so the canonical layout helpers are safe to use from this + # corrupted-venv repair path (#76105: never open-code the Scripts/bin split). from hermes_constants import project_venv_dir, venv_bin_dir venv_dir = project_venv_dir(root) @@ -163,6 +165,12 @@ def ensure_windows_bin_launchers( canonical managed binary dir (only when *root* is the managed clone, so source checkouts elsewhere never gain launchers) and the legacy ``\bin`` (only while the user PATH still points at it). Never raises. + + The canonical launcher home is the managed binary dir — the default Hermes root's ``bin`` + (``%LOCALAPPDATA%\\hermes\\bin``, next to the managed uv) — which lives OUTSIDE the git checkout so no + git operation can ever touch it. It is a per-machine dir shared by every profile: ``get_hermes_home()`` + would point inside ``profiles\\`` under ``hermes -p``, so the anchor here is + :func:`hermes_constants.get_default_hermes_root`. See #83797. """ if windows is None: windows = _is_windows() @@ -265,6 +273,8 @@ def migrate_windows_bin_path( absolute launcher paths keep working and the dir is git-ignored. Registry writes preserve the stored value type and raw ``%VARS%``. Never raises; True when the canonical layout is in place. *read_user_path*/*write_user_path* are injectable for tests. + + See #83797. """ if windows is None: windows = _is_windows() @@ -299,6 +309,7 @@ def migrate_windows_bin_path( _normalize_windows_path(root / "bin"), # The old installer put the venv's Scripts dir itself on PATH, always at the literal # `venv` layout (never `.venv`) — match what it wrote then, not where the venv lives now. + # See #83797. _normalize_windows_path(venv_bin_dir(root / "venv", windows=True)), } home_bin_key = _normalize_windows_path(home_bin) @@ -336,6 +347,8 @@ class ShimQuarantineError(RuntimeError): Raised BEFORE the install command runs. Callers catch it like any install failure: the update-incomplete marker survives and a later launch retries once the holder exits — the contended venv is never mutated. + + See #87331. """ def __init__(self, failed_shims: list[str]): @@ -351,6 +364,9 @@ def _quarantine_running_hermes_exe( Windows blocks REPLACE on a running .exe but allows RENAME. Best-effort: silently skips anything that cannot be renamed (names appended to *failed_out*). Returns (original, quarantined) pairs. The console-script set comes from pyproject ``[project.scripts]`` (fallback: well-known trio). + + ``failed_out``: when provided, names of shims that could not be renamed are appended so the caller can + refuse instead of mutating a contended venv (#87331 fail-closed). """ if not _is_windows(): return [] @@ -373,7 +389,13 @@ def _quarantine_running_hermes_exe( def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None: - """Put quarantined shims back when the installer did not replace them (shared retry ladder).""" + """Put quarantined shims back when the installer did not replace them (shared retry ladder). + + Delegates to the shared helper in the stdlib-only ``_early_recovery`` module: one retry ladder and one + recovery message for every restore site, instead of the near-identical copies that had already drifted + (#75584). Warnings land on stderr — this module runs in the early-recovery path and ``hermes acp`` + speaks JSON-RPC on stdout. + """ _er.restore_quarantined_shims(moved) @@ -384,6 +406,8 @@ def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None: installer would die partway on the same locks — raise :class:`ShimQuarantineError` WITHOUT running it. Raises CalledProcessError on install failure (callers implement the per-extra fallback ladder). + + The caller's marker-keeping failure handling turns that into "retry next launch". See #87331. """ scripts_dir = _venv_scripts_dir(root) if _is_windows() else None failed: list[str] = [] @@ -398,6 +422,7 @@ def _run_install_cmd(cmd: list[str], *, env: dict | None, root: Path) -> None: # entirely (uv audits an already-satisfied editable install as a no-op), which would leave # the shims renamed aside and `hermes` gone from PATH. Restore only renames back when the # installer did NOT write a fresh shim, so this is safe in both cases. + # See #75584. if scripts_dir is not None: _restore_quarantined_exes(moved) diff --git a/hermes_cli/_parser.py b/hermes_cli/_parser.py index 3788a2418e..e87c5e6a74 100644 --- a/hermes_cli/_parser.py +++ b/hermes_cli/_parser.py @@ -32,6 +32,9 @@ def top_level_value_flag_sets() -> tuple[frozenset[str], frozenset[str]]: ``main.py`` (``_first_positional_argv``, ``_apply_profile_override``) can never drift from the argparse surface — the drift that made ``hermes --reasoning high chat …`` misread ``high`` as the subcommand and forced eager plugin discovery. + + Mirrors the ``update_cmd._holder_value_flags`` precedent, including the handwritten-snapshot fallback + for a broken parser import. Cached per process. See #93530. """ try: parser = build_top_level_parser()[0] diff --git a/hermes_cli/_scan_venv_blockers.py b/hermes_cli/_scan_venv_blockers.py index ab8768c2df..c628547dfd 100644 --- a/hermes_cli/_scan_venv_blockers.py +++ b/hermes_cli/_scan_venv_blockers.py @@ -29,6 +29,8 @@ def _probe_fail_json(diagnostic: str = "probe failed") -> str: ``ok: false`` plus ``probe_failed: true`` means the detector itself could not run — this is *not* a clear scan. Callers must treat ``ok is not True`` / non-zero exit as probe failure, never as ``blocked: false`` "clear". + + See #83149. """ return json.dumps({"ok": False, "probe_failed": True, "blocked": False, "processes": [], "error": diagnostic}) @@ -161,7 +163,17 @@ def _is_pausable_gateway(cmdline: str) -> bool: def _is_updater_owned_backend(pid: int, cmdline: str) -> bool: - """True when *pid* is a Hermes backend the CLI updater can stop (positive ledger identity).""" + """True when *pid* is a Hermes backend the CLI updater can stop (positive ledger identity). + + The gateway exemption above keeps ``gateway run`` holders out of the blocker list because the updater's + own pause machinery stops and resumes them. ``hermes serve`` / ``hermes dashboard`` backends had no such + deferral, so a leaked serve child (or a Desktop-owned backend the teardown lost track of) dead-ended the + hand-off with ``venv-blocked`` — or, worse, survived the hand-off and made the shim quarantine fail with + ``os error 32`` (#98336) — even though the updater downstream owns exactly this case with its ledger + rungs (`_ledger_reapable_backend_pids` reaps dead-spawner orphans; `_ledger_manual_serve_holders` stops + manual serves and relaunches them on their recorded host/port). + Positive identity only — never name/substring matching (#90778, and the 99558 identity-guard contract): + """ return _updater_owned_backend_entry(pid, cmdline) is not None @@ -170,6 +182,8 @@ def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None: Returning the entry lets ``main()`` emit sanitized decision evidence — structured identity fields only, never argv, which can carry tokens or private endpoints. + + See #98350. """ try: from hermes_cli.update_cmd import _hermes_holder_subcommand # noqa: PLC0415 @@ -199,7 +213,12 @@ def _updater_owned_backend_entry(pid: int, cmdline: str) -> dict | None: def _deferred_backend_evidence(entries: list[dict]) -> list[dict]: - """Sanitized evidence (pid, purpose, recorded port — never argv) for deferred backends.""" + """Sanitized evidence (pid, purpose, recorded port — never argv) for deferred backends. + + Structured ledger fields only — pid, purpose, recorded port — never the command line, which can carry + tokens or private endpoints. Lets the scan result explain *why* a holder disappeared from ``processes`` + without echoing argv (#98350). + """ return [{"pid": entry.get("pid"), "purpose": entry.get("purpose"), "port": entry.get("port")} for entry in entries if isinstance(entry.get("pid"), int)] @@ -253,6 +272,7 @@ def main() -> None: if deferred_entry is not None: # Ledger-verified backend the updater's own rungs stop (and relaunch) downstream — # reporting it here would dead-end the hand-off before that machinery can run. + # See #98336. deferred_entries.append(deferred_entry) continue # Truncate for display AFTER the gateway exemption has seen the full cmdline (long @@ -267,7 +287,11 @@ def main() -> None: "blocked": bool(processes), "processes": processes, "pausable_gateways": exempted_gateways, + # Diagnostic only: ledger-verified serve/dashboard backends deferred to the updater's stop/relaunch + # rungs (#98336). "deferred_backends": len(deferred_entries), + # Diagnostic only: sanitized evidence (structured ledger identity, never argv) explaining which + # holders the deferral consumed (#98350). "deferred_backend_evidence": _deferred_backend_evidence(deferred_entries), } print(json.dumps(data)) diff --git a/hermes_cli/_subprocess_compat.py b/hermes_cli/_subprocess_compat.py index c4e0729890..e3830efd25 100644 --- a/hermes_cli/_subprocess_compat.py +++ b/hermes_cli/_subprocess_compat.py @@ -84,6 +84,11 @@ def split_command_line(line: str) -> list[str]: ``shlex.split`` (posix=True) treats every backslash as an escape, mangling Windows paths. On Windows use ``posix=False`` and strip one layer of matching quotes per token; on POSIX this is exactly ``shlex.split``. Raises ValueError on unbalanced quotes. + + ``shlex.split(line)`` (posix=True) treats every backslash as an escape character, so Windows paths are + silently mangled: ``C:\\Users\\me\\out.txt`` becomes ``C:Usersmeout.txt`` — no error, just a wrong path + that then "succeeds" against a mangled relative filename (#83934) or makes a valid hook script report + "not executable" (#78293). """ import shlex @@ -119,6 +124,7 @@ _CREATE_NEW_PROCESS_GROUP = 0x00000200 # CREATE_NO_WINDOW child instead OWNS a hidden console all descendants inherit (A/B verified on # Windows 11 by the desktop backend fix, commit aa2ae36c3f: with per-site hide flags neutered, # naive git/gh/cmd spawns don't flash under a hidden-console parent and do under a console-less one). +# 1. Combining them means DETACHED_PROCESS governs and the no-window bit is dead. 2. See #54220, #56747. _CREATE_NO_WINDOW = 0x08000000 # Escape any Win32 job object the parent belongs to. Without this a detached child inherits the # parent's job, and when that parent (Electron, Tauri, Windows Terminal, the Desktop bootstrap @@ -137,6 +143,17 @@ def windows_detach_flags() -> int: (#54220/#56747) at every spawn; CREATE_BREAKAWAY_FROM_JOB escapes Electron/Tauri job objects. A job that forbids breakaway yields PermissionError from Popen — callers catch OSError and fall back to :func:`windows_detach_flags_without_breakaway`. + + Rationale: This both detaches it from the parent's console lifetime (closing the launching terminal + doesn't CTRL_CLOSE it) AND gives every console-subsystem descendant (git, gh, cmd, node, …) a console to + inherit, so they don't allocate visible flashing ones. This deliberately replaces the old + ``DETACHED_PROCESS`` approach: MSDN specifies CREATE_NO_WINDOW is *ignored* when combined with + DETACHED_PROCESS, and a truly console-less daemon re-creates the per-descendant console-flash bug + (#54220/#56747) at every spawn — see the note on ``_DETACHED_PROCESS`` above. Electron (Desktop app) and + Tauri (bootstrap installer) wrap their children in job objects; without breakaway, those children die + when the parent process exits even though they have their own console. This was the missing flag that + made the post-update gateway respawn watcher silently die alongside the Tauri updater after the Electron + Desktop's update flow finished. """ if not IS_WINDOWS: return 0 @@ -221,6 +238,14 @@ def noninteractive_git_env(base: "Mapping[str, str] | None" = None) -> dict[str, *working* askpass helper or ssh-agent should still succeed non-interactively. Pair with ``stdin=subprocess.DEVNULL``. Internal plumbing only — the agent-facing terminal tool has its own policy layer and visible PTY. + + Hermes shells out to git from many non-interactive contexts — MCP catalog installs, plugin + install/update, profile distribution staging, worktree base fetches, desktop review-pane fetch/push. + When the remote is private, misconfigured, or requires auth, git's default behavior is to prompt on the + inherited terminal (or via an askpass helper), which silently hangs the operation until its timeout — or + forever at call sites without one. Ported from openai/codex#34540 / #34612 ("detach non-interactive + subprocesses from stdin"): a background tool invocation must fail fast with a readable error, not wait + for input nobody can type. """ env = dict(base if base is not None else os.environ) env["GIT_TERMINAL_PROMPT"] = "0" @@ -313,6 +338,17 @@ def kill_process_tree(proc: "subprocess.Popen") -> None: ``proc.kill()`` alone only terminates the direct child. This is cleanup on an already-failing path whose contract is to fail open, so every failure (access denied, already reaped) is swallowed rather than escaping the caller's ``except``. + + On Windows a suspended descendant (e.g. ``git.exe``) can survive holding duplicates of the captured pipe + handles, which keeps the pipes from reaching EOF and leaks two reader threads + the process per fired + timeout — ``taskkill /T /F`` takes the whole tree down so the bounded drain that follows can actually + reach EOF. On POSIX the same class exists: killing the launcher leaves descendants (credential helpers, + ``git-remote-https``, hook children) running and holding the pipe write ends. Callers spawn the child in + its own process group (``process_group=0``, Python ≥3.11), so when — and only when — the child leads its + own group (``pgid == pid``), the entire group is signalled with ``os.killpg``. The ownership check means + a fallback spawn that shares our group can never cause us to kill unrelated processes. Ported from + openai/codex#36793 ("Terminate timed-out Git process trees"); generalized for the shell-hook runner via + openai/codex#37527 ("Terminate timed-out hook process trees"). """ try: from agent.deadline import kill_process_tree as _deadline_kill_tree @@ -364,6 +400,13 @@ def bounded_probe_run( Returns a ``CompletedProcess`` when the child finished within *timeout* (any exit code), or ``None`` on spawn failure or timeout. + + Why not ``subprocess.run``: on Windows, ``run()``'s post-timeout cleanup calls an *unbounded* + ``communicate()`` after killing the direct child. Killing it can leave a descendant (``git.exe`` under a + launcher shim, ``conhost.exe`` under wmic/powershell) holding duplicates of the captured stdout/stderr + handles, so the pipes never reach EOF and the reader-thread join blocks forever. The wmic / + ``Get-CimInstance Win32_Process`` gateway scan hit exactly this during ``hermes update`` on slow-WMI + machines (#87134); the git probes hit it first (#68609 / #66037). """ _popen_kwargs: dict = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {"process_group": 0} try: @@ -400,6 +443,18 @@ def bounded_git_probe(argv: Sequence[str], *, timeout: float) -> str: repo-configured ``core.fsmonitor`` program. Every probe therefore runs under :func:`noninteractive_git_env`; diff-rendering callers additionally pass :data:`NO_DRIVER_DIFF_FLAGS` (attribute-scoped drivers can't be disabled via env). + + Killing the PATH-resolved launcher can leave a suspended descendant ``git.exe`` holding duplicates of + the captured stdout/stderr handles, so the pipes never reach EOF and the reader-thread join blocks + forever. On the Desktop agent-build path (``_start_agent_build → _session_info → branch() → run_git``) + that turned an optional branch label into ``agent initialization timed out`` (issues #68609 / #66037). + The normal-path spawn contract mirrors the previous ``run`` call byte-for-byte: PIPE/PIPE/DEVNULL, + ``text`` with UTF-8 ``errors="replace"`` decoding, and the hidden-window ``creationflags`` on Windows + only. On POSIX the probe is additionally placed in its own process group (``process_group=0``, Python + ≥3.11) so timeout cleanup can take down descendants — credential helpers, ``git-remote-https``, hook + children — with the launcher instead of orphaning them (see :func:`_kill_git_process_tree`; port of + openai/codex#36793). ``process_group`` only changes which group the child belongs to; it does not detach + the terminal or alter the fast path. """ result = bounded_probe_run(argv, timeout=timeout, env=noninteractive_git_env()) if result is None or result.returncode != 0: diff --git a/hermes_cli/active_sessions.py b/hermes_cli/active_sessions.py index b61cabbf7a..a7ba1b7074 100644 --- a/hermes_cli/active_sessions.py +++ b/hermes_cli/active_sessions.py @@ -105,6 +105,9 @@ MAX_CONCURRENT_SESSIONS = "MAX_CONCURRENT_SESSIONS" # Ownership could not be PROVEN (registry unreadable/corrupt). Deliberately distinct # from SESSION_NOT_OWNED: treating "can't tell" as a go-ahead is the fail-open hole # that let two writers share one session. +# Distinct from SESSION_NOT_OWNED on purpose -- "someone else owns this" and "I cannot tell who owns this" +# call for different operator action, and collapsing the second into a silent go-ahead is exactly the +# fail-open hole that let two writers share one session (#94595 review, blocker 2). SESSION_COORDINATION_UNAVAILABLE = "SESSION_COORDINATION_UNAVAILABLE" # Advertised through the gateway. A module constant, not a config flag: it holds @@ -360,6 +363,7 @@ class ActiveSessionLease: # Pinned at acquisition: a lease taken under the root HERMES_HOME must release # against the same registry even inside a profile-home override, or phantom # leases fill the session cap. + # See #85431. state_path: Optional[Path] = None lock_path: Optional[Path] = None track_liveness: bool = False @@ -438,6 +442,9 @@ def try_acquire_active_session( owner per stored session); ``max_concurrent_sessions`` is resource POLICY, applied only when configured. ``registry_home`` lets profile-scoped backends share the owning profile's registry. Ownership uncertainty fails CLOSED (SESSION_COORDINATION_UNAVAILABLE). + + Liveness tracking keeps richer desktop lifecycle semantics; ``registry_home`` lets profile-scoped + backends share the owning profile's registry even when launched from another home. See #94595. """ max_sessions = resolve_max_concurrent_sessions(config) lease_id = uuid.uuid4().hex @@ -522,6 +529,7 @@ def try_acquire_active_session( def release_active_session(lease: ActiveSessionLease) -> None: # Prefer the registry the lease was acquired against: the caller may be # running under a profile HERMES_HOME override. + # See #85431. state_path, lock_path = _lease_paths(lease) with _FileLock(lock_path): if lease.released: @@ -584,6 +592,7 @@ def transfer_active_session( # the file lock and the server attaches the lease to its session record only after that # returns. A concurrent finalize that snapshotted its live ids in between would otherwise # read the brand-new lease as an orphan and drop it. Real orphans are minutes old. +# See #101415. _SELF_ORPHAN_GRACE_SECONDS = 30.0 diff --git a/hermes_cli/auth.py b/hermes_cli/auth.py index d7075b2762..7a7cf5880f 100644 --- a/hermes_cli/auth.py +++ b/hermes_cli/auth.py @@ -293,7 +293,11 @@ except Exception: def get_anthropic_key() -> str: """First usable Anthropic credential (``.env`` preferred over a stale shell export), or ``""``. - Order mirrors ``PROVIDER_REGISTRY["anthropic"].api_key_env_vars``.""" + Order mirrors ``PROVIDER_REGISTRY["anthropic"].api_key_env_vars``. + + Checks both the ``.env`` file and the process environment, preferring ``~/.hermes/.env`` so a deliberate + key rotation isn't shadowed by a stale shell export (matches the api-key resolution path — see #20591). + """ from hermes_cli.config import get_env_value_prefer_dotenv env_vars = PROVIDER_REGISTRY["anthropic"].api_key_env_vars return next((v for v in (get_env_value_prefer_dotenv(var) or "" for var in env_vars) if v), "") @@ -317,6 +321,7 @@ def has_usable_secret(value: Any, *, min_length: int = 4) -> bool: # Known API-key prefixes per provider. Only listed providers get prefix validation; everyone else # is fail-open. Keeps an obviously malformed key in .env (truncated paste, wrong provider's key) # from silently shadowing a valid credential-pool entry and producing opaque 401s. +# See #93593. KNOWN_PROVIDER_KEY_PREFIXES: Dict[str, tuple] = { "openrouter": ("sk-or-",), # all OpenRouter keys are sk-or-... (currently sk-or-v1-) } @@ -362,6 +367,8 @@ def _resolve_api_key_provider_secret(provider_id: str, pconfig: ProviderConfig) for env_var in pconfig.api_key_env_vars: val = _usable_declared_secret(provider_id, get_env_value_prefer_dotenv(env_var), env_var) if val: + # A provably malformed key (declared prefix mismatch) must not shadow a valid credential-pool + # entry (#93593). Warn and keep looking instead of returning it. return val, env_var # Fallback: credential pool (e.g. zai key stored via auth.json). Prefer the pool's own @@ -505,7 +512,10 @@ def _is_same_auth_store(left: Path, right: Path) -> bool: """True when two auth paths name ONE store rather than two copies. ``_same_path`` resolves symlinks and ``..``; ``samefile`` adds hardlinks and bind-mounts (same inode under two resolved names). Used by the forked-grant heal: a shared store has - no "other side" to consolidate.""" + no "other side" to consolidate. + + See #101356. + """ if _same_path(left, right): return True try: @@ -696,6 +706,9 @@ def _save_auth_store(auth_store: Dict[str, Any], target_path: Optional[Path] = N """Atomically persist *auth_store* (0o600, parent tightened to 0o700) to the active store, or to an explicit *target_path* (e.g. the global-root write-through for rotating xAI OAuth grants).""" auth_file = target_path if target_path is not None else _auth_file_path() + # Tighten parent dir to 0o700 so siblings can't traverse to creds. No-op on Windows (POSIX mode bits not + # enforced); ignore failures. secure_parent_dir refuses to chmod /, top-level dirs, or the hermes-agent + # install tree (#25821, #93050). auth_store["version"] = AUTH_STORE_VERSION auth_store["updated_at"] = datetime.now(timezone.utc).isoformat() _write_private_file_atomic(auth_file, json.dumps(auth_store, indent=2) + "\n", fsync_dir=True) @@ -1099,6 +1112,7 @@ def _pool_entry_is_explicit(entry: Any) -> bool: if source.startswith("env:"): # A stale env-seeded entry survives in auth.json after the user deletes the env var: only # count it when the referenced var still resolves to a usable secret NOW. + # See #55790. env_var = entry.get("source", "").split(":", 1)[1].strip() return bool(env_var and _env_secret(env_var)) return bool(source) and (source in _EXPLICIT_POOL_SOURCES or source.startswith("manual:")) @@ -1282,6 +1296,9 @@ def _scoped_key_env_reader() -> Callable[[str], str]: ONLY ImportError: any other auxiliary_client failure must propagate rather than silently falling back to os.getenv (a traceless fail-open).""" try: + # Scope-aware key reads: under multiplex a secondary profile's API keys live only in its secret + # scope, not os.environ — a bare getenv here would find nothing and auto-resolution would report "No + # LLM provider configured" for every secondary profile (same class as #86905). from agent.auxiliary_client import _scoped_key_env return _scoped_key_env except ImportError: @@ -1300,6 +1317,11 @@ def _openrouter_auto_detected(scoped_key_env: Callable[[str], str]) -> bool: if any(has_usable_secret(scoped_key_env(v)) for v in ("OPENAI_API_KEY", "OPENROUTER_API_KEY")): return True try: + # Auto-detect an OpenRouter credential added via `hermes auth add openrouter` (manual pool entry, no + # env var). Without this, a key that only lives in the credential pool is invisible to + # auto-detection — the user sees `hermes auth list` showing the credential while requests go out + # with no Authorization header ("HTTP 401: Missing Authentication header"). The env-var check above + # only covers keys exported as OPENROUTER_API_KEY / OPENAI_API_KEY. See issue #42130. from agent.credential_pool import load_pool as _load_pool return bool(_load_pool("openrouter").has_credentials()) except Exception as e: @@ -1352,6 +1374,9 @@ def _env_key_auto_detected( if has_usable_secret(scoped_key_env(env_var)): if oauth_active and oauth_active != pid: logger.warning( + # An exported API key now wins over a logged-in OAuth provider (the #29285 fix). + # Surface that so a user who deliberately uses OAuth but has a stale key in + # ~/.hermes/.env isn't silently switched without knowing why. "Provider resolved to %r via %s, preempting your " "logged-in OAuth provider %r. If you meant to use the " "OAuth login, unset %s or set `model.provider` " @@ -1371,7 +1396,11 @@ def resolve_provider( "auto" priority (explicit intent beats a stale OAuth login): 1. CLI api_key/base_url -> "openrouter"; 2. config.yaml ``model.provider``; 3. OPENAI_API_KEY / OPENROUTER_API_KEY -> "openrouter"; 4. OpenRouter pool; 5. provider env keys; 6. auth.json ``active_provider``; - 7. AWS Bedrock chain; 8. AuthError(no_provider_configured).""" + 7. AWS Bedrock chain; 8. AuthError(no_provider_configured). + + 1. 3. 4. 5. Provider-specific API keys (GLM, Kimi, MiniMax, ...) -> that provider 7. 8. Error (no + provider configured) See #29285. + """ normalized = (requested or "auto").strip().lower() normalized = _plugin_aliases().get(normalized, normalized) @@ -1404,6 +1433,8 @@ def resolve_provider( # Logged-in OAuth provider is a LAST-RESORT fallback (it used to sit above the env/config # checks, so a stale login silently overrode explicit intent). + # Logged-in OAuth provider (auth.json `active_provider`) — a LAST-RESORT fallback, chosen only when the + # user expressed no other preference above. Demoted here so explicit intent always wins. See #29285. if _oauth_active: if isinstance(_model_cfg, dict) and _model_cfg and not _model_cfg.get("provider"): logger.warning( diff --git a/hermes_cli/auth_codex.py b/hermes_cli/auth_codex.py index 34a20ec0de..ac5f5e9591 100644 --- a/hermes_cli/auth_codex.py +++ b/hermes_cli/auth_codex.py @@ -111,6 +111,13 @@ def _sync_codex_pool_entries( access_token equals the PREVIOUS singleton token — a legacy alias of the singleton; an entry with its own token material is an independent account and must be left alone. ``manual:api_key`` and any other source are independent credentials and are never overwritten by a re-auth. + + See #33000, #39236. + The original #33538 fix refreshed every ``manual:device_code`` entry unconditionally. That worked when + ``manual:device_code`` only meant "legacy alias of the singleton", but the same source string is now + also produced by independent-account additions, and the broad sync silently clobbered distinct accounts + with the latest-authenticated token pair. The access_token-match check distinguishes the two cases + without changing the source-string contract. """ access_token = tokens.get("access_token") if not access_token: @@ -220,6 +227,11 @@ def _codex_http_client(**kwargs: Any) -> "httpx.Client": A host advertising AAAA records but blackholing IPv6 makes each serial connect eat the full timeout before IPv4 is tried (same failure mode as the chat transport). Best-effort: if the racing backend can't be installed (mocked client in tests), serial connect behavior remains. + + Same broken-IPv6 failure mode as the chat transport (#13834): a host that advertises AAAA records but + blackholes IPv6 makes each serial connect attempt eat the full connect timeout before IPv4 is tried, so + token refresh / device login / usage probes time out where the official Codex CLI (which races families + per RFC 8305) works. """ client = httpx.Client(**kwargs) with suppress(Exception): @@ -371,6 +383,12 @@ def resolve_codex_runtime_credentials( Falls back to the credential pool when the singleton (``providers.openai-codex.tokens``) has no usable access_token but the pool (``credential_pool.openai-codex``) does. + + This closes the divergence between the chat path (singleton-only via this function) and the auxiliary + path (pool-first via ``_read_codex_access_token``). Without this fallback, a user whose tokens live only + in the pool — for example after a manual pool seed, a partial re-auth, or pool-only restoration from a + backup — gets a bare HTTP 401 ``Missing Authentication header`` from the wire instead of a usable + credential. See issue #32992. """ from hermes_cli.auth import ( _auth_store_lock, _codex_access_token_is_expiring, _probe_codex_quota_restored, @@ -655,6 +673,10 @@ def _login_openai_codex(args, pconfig: ProviderConfig, *, force_new_login: bool def _codex_login_rate_limited_error(response: "httpx.Response", *, during: str = "") -> AuthError: """AuthError for a 429 from OpenAI's device-auth endpoints (throttle, not credential fault).""" + # Upstream rate-limit / usage-quota exhaustion on the token endpoint. The stored refresh token is still + # valid here — re-authenticating cannot lift a quota cap. Classify distinctly from auth failures so + # callers surface a "retry later" notice instead of a misleading "run hermes auth" prompt (see issue + # #32790). retry_after = _parse_retry_after_seconds(getattr(response, "headers", None)) wait_hint = ( f" Try again in about {retry_after}s." if retry_after is not None diff --git a/hermes_cli/auth_device_flow.py b/hermes_cli/auth_device_flow.py index 56cf7b6058..1951869388 100644 --- a/hermes_cli/auth_device_flow.py +++ b/hermes_cli/auth_device_flow.py @@ -43,7 +43,14 @@ _REMOTE_IDE_ENV_VARS = ( def _is_remote_session() -> bool: - """Detect environments where loopback OAuth can't reach the local browser.""" + """Detect environments where loopback OAuth can't reach the local browser. + + Historically only SSH was checked, but #26923 surfaced that **browser-only remote consoles** (GCP Cloud + Shell, GitHub Codespaces, AWS EC2 Instance Connect, Gitpod, Replit, etc.) hit the exact same problem — + the user has a browser on their laptop but the loopback listener is bound on the remote VM that the + laptop's browser can't reach. These environments typically don't set ``SSH_CLIENT`` / ``SSH_TTY``, so + the SSH-only check left them with no guidance and no fallback. + """ return bool( os.getenv("SSH_CLIENT") or os.getenv("SSH_TTY") or any(os.getenv(var) for var in _REMOTE_IDE_ENV_VARS)) @@ -176,7 +183,12 @@ def _request_device_code( def _nous_device_auth_timeout_message(portal_base_url: str) -> str: - """Actionable timeout text: the usual cause is Portal sign-in failing in the browser tab.""" + """Actionable timeout text: the usual cause is Portal sign-in failing in the browser tab. + + A bare "Timed out waiting for device authorization" gives the user nothing to act on. The most common + cause is Portal sign-in failing in the opened browser tab (including the server-side CAPTCHA loop from + 20605), so point at the Portal login page and the retry command. See #20605. + """ portal = (portal_base_url or DEFAULT_NOUS_PORTAL_URL).rstrip("/") return ( "Timed out waiting for device authorization.\n" diff --git a/hermes_cli/auth_oauth_grants.py b/hermes_cli/auth_oauth_grants.py index 707b270b68..2e4dc8da66 100644 --- a/hermes_cli/auth_oauth_grants.py +++ b/hermes_cli/auth_oauth_grants.py @@ -379,6 +379,7 @@ class _HealPass: from hermes_cli.auth import _is_same_auth_store if self.root_singleton is not None and _is_same_auth_store(profile_singleton, self.root_singleton): return # an aliased singleton pair is one shared grant, not a fork: never self-compare/unlink + # See #101356. p_single = _singleton_as_row(profile_singleton) root_has_grant = bool(self.r_oauth) or self.root_singleton_row is not None # Otherwise root has NO grant for this provider (or the file is not a grant): the @@ -479,6 +480,7 @@ def _heal_forked_single_use_oauth_grants(provider_id: str) -> Optional[Dict[str, # itself, and the strip would write through the alias and delete the shared credential. # Nothing to consolidate; the mtime mark keeps this off the per-call hot path. _oauth_heal_clean_marks[provider_id] = fingerprint + # See #101356. logger.debug("%s: forked-OAuth heal skipped, %s is the root store", provider_id, profile_path) return None diff --git a/hermes_cli/auth_xai.py b/hermes_cli/auth_xai.py index b983fe68cc..dd96bdd3f0 100644 --- a/hermes_cli/auth_xai.py +++ b/hermes_cli/auth_xai.py @@ -322,6 +322,10 @@ def refresh_xai_oauth_pure( suffix = f" Response: {detail}" if detail else "" # 403 is almost always a tier/entitlement gate; re-login won't fix it, so use a separate # code and format_auth_error skips the re-authenticate hint. + # ``403`` from xAI's token endpoint is almost always a tier / entitlement gate (the OAuth grant + # exists but the account isn't on the allowlist for API access). Re-running ``hermes model`` won't + # fix that — surface a separate error code so ``format_auth_error`` doesn't append a misleading + # re-authenticate hint, and point users at the ``XAI_API_KEY`` fallback. See #26847. if response.status_code == 403: raise _xai_err( "xAI token refresh failed with HTTP 403." + suffix @@ -390,6 +394,9 @@ def _quarantine_xai_oauth_tokens(exc: AuthError) -> None: tokens = dict(state.get("tokens") or {}) tokens.pop("access_token", None) tokens.pop("refresh_token", None) + # Capture the previous singleton tokens BEFORE overwriting them. The pool-sync step uses this to + # distinguish legacy singleton-aliases (which should be refreshed) from independent accounts that + # ``hermes auth add openai-codex`` created (which must not be overwritten — see #39236). state["tokens"] = tokens state["last_auth_error"] = _last_auth_error_marker( "xai-oauth", exc, reason="runtime_refresh_failure", default_code="xai_refresh_failed", diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py index 113356fc1d..5af0a41f5b 100644 --- a/hermes_cli/backup.py +++ b/hermes_cli/backup.py @@ -317,6 +317,10 @@ def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = F """True when *path* looks like the #68474 zeroed-state.db signature. Only regular files qualify: probing a FIFO/device/socket could block indefinitely. + + Signature: no ``SQLite format 3`` header and no data — either empty (size 0, the total-loss case, + #97568) or first *probe_bytes* all NUL. Used at SessionDB open and for snapshot diagnostics so a silent + all-zero file becomes a guided recovery instead of a generic failure. """ try: if not path.is_file(): @@ -338,6 +342,9 @@ _SQLITE_HEADER = b"SQLite format 3\0" # Above this size ``PRAGMA integrity_check`` (walks every b-tree page — minutes of pegged CPU on a # 30 GB state.db, reading as a hung ``hermes update``) is replaced by the O(1) header+schema probe. +# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour of the (O(1)) header + +# structural probe. Sessions databases in the tens of GB are normal for heavy users, so the size-unbounded +# check is never an acceptable default on the update path. See #70553. DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB @@ -1281,6 +1288,10 @@ def restore_cron_jobs_if_emptied(snapshot_id: str, hermes_home: Optional[Path] = Conservative: restores only when the snapshot had MORE jobs than the live file (a user who deleted jobs is never second-guessed); an unreadable live file is left so corruption surfaces. + + Config-version migrations have been observed to leave ``cron/jobs.json`` valid-but-empty after an + update, silently dropping every scheduled job (issue #34600). The desktop scheduler can also overwrite + the file with its own small set of internally-tracked crons, causing partial loss (issue 52144). """ if not snapshot_id: return None diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index 4547fa45b2..672ea88174 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -301,6 +301,7 @@ def _check_via_local_git(repo_dir: Path) -> Optional[int]: # if it already shows HEAD behind, that is sound evidence an update exists. Return the positive # stale count; None (inconclusive) otherwise so the caller doesn't cache a false "up to date". if is_shallow: + # (#82166, review #92578) if not fetch_ok: return None # No history across the shallow boundary. `origin/main` may not be a tracking ref in a diff --git a/hermes_cli/browser_connect.py b/hermes_cli/browser_connect.py index 201a5ca71c..fed82e3b20 100644 --- a/hermes_cli/browser_connect.py +++ b/hermes_cli/browser_connect.py @@ -276,6 +276,9 @@ def _launchservices_https_handler(dump: str) -> str | None: # Strip the nested LSHandlerPreferredVersions block first: on macOS 26 it carries # a VERSION NUMBER (LSHandlerRoleAll = "7559.97";), not the "-" placeholder older # releases used, and the role regex below would return it instead of the bundle id. + # Left in, the role regex below would match that version before the real bundle id sitting at the + # entry's own level and return "7559.97" — which maps to no browser, so detection fails on a machine + # whose default IS Chrome (PR #95620 review). low = re.sub(r"lshandlerpreferredversions\s*=\s*\{[^}]*\}\s*;", "", low) # The real bundle id is the first non-"-" role value at this level. roles = re.findall(r'lshandlerrole(?:all|viewer)\s*=\s*"?([a-z0-9.\-]+)"?\s*;', low) diff --git a/hermes_cli/claw.py b/hermes_cli/claw.py index def2dee09b..773299754f 100644 --- a/hermes_cli/claw.py +++ b/hermes_cli/claw.py @@ -102,6 +102,7 @@ def _detect_openclaw_processes() -> list[str]: if sys.platform == "win32": # bounded_probe_run: plain subprocess.run(timeout=...) can hang forever on Windows when a # conhost.exe descendant holds duplicated pipe handles — a hang is not an exception. + # See #87134. from hermes_cli._subprocess_compat import bounded_probe_run try: for exe in ("openclaw.exe", "clawd.exe"): @@ -372,6 +373,7 @@ def _cmd_cleanup(args): print() return print_success("No OpenClaw directories found. Nothing to clean up.") # Archiving while the service is active makes it recreate an empty skeleton directory. + # See #8502. running = _detect_openclaw_processes() if running and not _warn_running( auto_yes, "OpenClaw appears to be still running:", running, diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index e6b93b8afd..1807c8378a 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -17,7 +17,11 @@ def _single_query_clarify_callback(question: str, choices=None, multi_select=Fal A -q turn never builds the prompt_toolkit app, so the interactive clarify modal can never be painted or answered — the CLI callback would poll until ``agent.clarify_timeout`` while the caller sees a silent hang. Mirror the oneshot - path and answer immediately instead.""" + path and answer immediately instead. + + The oneshot path answers immediately via ``_oneshot_clarify_callback``; single-query turns need the same + headless behavior (#94943). + """ prefix = f"[single-query mode: no user available to answer {question!r}. " if choices: what = "subset" if multi_select else "option" @@ -249,6 +253,7 @@ class CLIAgentSetupMixin: pass # Normalize model for the resolved provider (e.g. swap non-Codex models on openai-codex). + # Fixes #651. model_changed = self._normalize_model_for_provider(resolved_provider) # AIAgent/OpenAI client holds auth at init, so rebuild on key/routing/model change. @@ -296,7 +301,10 @@ class CLIAgentSetupMixin: """Silently probe whether any inference provider can be resolved. Never prints or mutates CLI state, so the interactive first-run path can route a - keyless install into onboarding before the user types into a chat that can't work.""" + keyless install into onboarding before the user types into a chat that can't work. + + See #62935. + """ from hermes_cli.runtime_provider import resolve_runtime_provider try: runtime = resolve_runtime_provider( @@ -409,6 +417,7 @@ class CLIAgentSetupMixin: session_meta = self._session_db.get_session(self.session_id) # Quiet mode (tool_progress_mode == "off") routes resume status lines to # stderr so stdout stays machine-readable for `$(hermes chat -Q --resume ...)`. + # Without this, the resume banner pollutes captured stdout. See #11793. _quiet_mode = getattr(self, "tool_progress_mode", "full") == "off" def _say(plain: str, rich: str) -> None: @@ -496,6 +505,7 @@ class CLIAgentSetupMixin: # -q never builds the prompt_toolkit app, so the clarify modal can't be # answered — answer headless instead of polling until clarify_timeout. clarify_callback = ( + # See #94943. _single_query_clarify_callback if getattr(self, "_single_query_mode", False) else self._clarify_callback) @@ -538,10 +548,14 @@ class CLIAgentSetupMixin: # Reference for atexit memory-provider shutdown: ``_run_cleanup`` in cli.py # reads ``cli._active_agent_ref``, so this MUST write the ``cli`` module's # global — a ``global`` statement here would bind this module's namespace. + # When this code lived in cli.py a bare ``global _active_agent_ref`` worked; after the god-file + # extraction into this mixin a ``global`` here would bind *this module's* namespace, leaving + # ``cli._active_agent_ref`` None forever — so memory shutdown never ran on /exit (#49287). import cli as _cli _cli._active_agent_ref = self.agent # Route agent status output through prompt_toolkit so ANSI escapes # aren't garbled by patch_stdout's StdoutProxy. + # See #2262. self.agent._print_fn = _cprint # Hydrate credits notices at session OPEN (parity with the TUI) so a depletion # warning shows before the first message. Idempotent + fail-open in the helper. diff --git a/hermes_cli/cli_chat_turn_mixin.py b/hermes_cli/cli_chat_turn_mixin.py index dae6b6f948..daea3f0bda 100644 --- a/hermes_cli/cli_chat_turn_mixin.py +++ b/hermes_cli/cli_chat_turn_mixin.py @@ -29,6 +29,10 @@ class CLIChatTurnMixin: ``_pending_input`` so process_loop and the interrupt monitor never compete); an interrupting message is re-queued as the next turn. ``voice_input`` gates the concise voice-response prefix. + + Args: message: The user's message (str or multimodal content list) images: Optional list of Path + objects for attached images voice_input: True when the message came from voice transcription (gates + the concise voice-response prefix, #65827) """ from cli import ChatConsole, _ChatTurn, _DIM, _RST, _accent_hex, _cprint, set_secret_capture_callback # Single-query and direct chat callers do not go through run(). @@ -361,6 +365,7 @@ class CLIChatTurnMixin: except queue.Empty: # Flush the StdoutProxy buffer: it otherwise only flushes on input-triggered # renderer passes, so on macOS the CLI looks frozen until the user types. + # Force prompt_toolkit to flush any pending stdout output from the agent thread. (#1624) self._invalidate(min_interval=0.15) continue if not interrupt_msg: diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 19903745fd..e86a0aecb7 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -332,6 +332,19 @@ def _sync_agent_to_session(cli, session_id: str, *, parent_session_id: str, reas cli.agent._invalidate_system_prompt() with suppress(Exception): _mm = getattr(cli.agent, "_memory_manager", None) + # Notify memory providers that session_id rotated to a fresh conversation. reset=True signals + # providers to flush accumulated per-session state (_session_turns, _turn_counter, _document_id). + # Fires BEFORE the plugin on_session_reset hook (shell hooks only see the new id; Python providers + # see the transition). See #6672. When the old session has history, end-of-session extraction + # (LLM-bound, seconds) and this switch are queued as ONE task on the memory manager's serialized + # worker — end strictly before switch, without blocking /new (#16454). With no history there is + # nothing to extract; switch inline as before. + # Notify memory providers that session_id rotated to a resumed session. reset=False — the provider's + # accumulated state is still valid; it just needs to target the new session_id for subsequent + # writes. See #6672. + # Notify memory providers that session_id forked to a new branch. reset=False — the branched session + # carries the transcript forward, so provider state tracks the lineage. parent_session_id links the + # branch back to the original. See #6672. if _mm is not None: _mm.on_session_switch( session_id, parent_session_id=parent_session_id or "", reset=False, reason=reason) @@ -616,6 +629,11 @@ class CLICommandsMixin: # No checkpoints for this dir → cross-project view (writes may sit under the session cwd). checkpoints = mgr.list_checkpoints(cwd) if not checkpoints: + # List checkpoints — fall back to the cross-project view when the current directory has none + # (#10505, reapply of PR #10633 by @nightq). The Aug 2026 QA sweep hit this live: writes + # landed checkpoints under the session cwd (/tmp/qa-repo) while bare /rollback searched only + # TERMINAL_CWD's project and reported "No checkpoints found" despite fresh checkpoints + # existing. all_checkpoints = mgr.list_all_checkpoints() if all_checkpoints: print(f" No checkpoints for {cwd} — showing all directories.") @@ -870,7 +888,10 @@ class CLICommandsMixin: # ---- /stop, /agents ------------------------------------------------------------------- def _handle_stop_command(self): """Handle /stop — kill all running background processes and background (async) delegations. - Separate from interrupt (stop the current turn), as in Codex.""" + Separate from interrupt (stop the current turn), as in Codex. + + See #14602. + """ from tools.process_registry import process_registry running = [p for p in process_registry.list_sessions() if p.get("status") == "running"] # Background subagents live in their own registry, not the process registry. @@ -905,6 +926,7 @@ class CLICommandsMixin: status = d.get("status", "?") line = f" {d.get('delegation_id', '?')} · {status} · {(d.get('goal') or '')[:60]}" # Live-status detail for in-flight delegations. + # See #51690. if status == "stalling": quiet = d.get("stalled_after_quiet_seconds") if quiet is not None: @@ -994,6 +1016,7 @@ class CLICommandsMixin: # Over SSH native tools write the REMOTE clipboard; OSC 52 reaches the user's terminal. # Locally, OSC 52 is the fallback when native tools are unavailable/fail (SSH/tmux). if is_remote_shell_session() or not write_clipboard_text(text): + # Fixes #31528. self._write_osc52_clipboard(text) _cp(f" Copied assistant response #{idx + 1} via OSC 52 (terminal support required)") else: @@ -1189,6 +1212,7 @@ class CLICommandsMixin: f" Resume it on this CLI later with: /resume {session_title}", "") # _run_cleanup must NOT finalize the row on exit: the gateway owns it now, and an # end_reason set under it would drop the handoff leg from session history/search. + # See #88234. from cli import _handed_off_session_ids _handed_off_session_ids.add(self.session_id) self._should_exit = True # same exit semantics as /quit @@ -1240,6 +1264,10 @@ class CLICommandsMixin: if self._show_recent_sessions(reason="resume"): # Arm a one-shot bare-number selection; must be the same list the table showed # and the numbered branch resolves (all use _list_recent_sessions(limit=10)). + # Arm a one-shot pending-resume selection so the user can type just the number (`3`) on the + # next line instead of having to retype `/resume 3`. The list here must match the one shown + # by _show_recent_sessions and used for index resolution below — all three go through + # _list_recent_sessions(limit=10). See #34584. self._pending_resume_sessions = self._list_recent_sessions(limit=10) return return _cp(" Tip: Use /history or `hermes sessions list` to find sessions.") @@ -1278,6 +1306,11 @@ class CLICommandsMixin: _cp(f" ↻ Resumed session {target_id}{title_part} — no messages, starting fresh.") # Same contract as startup --resume: retarget the tool cwd, restore the persisted YOLO # bypass (approval session key changed) and the model/provider (else config default). + # Retarget the process + tool cwd to where the session was started, so a mid-chat /resume (and + # /sessions , which delegates here) lands in the same directory as a startup `hermes + # -c`/`--resume`. The startup resume paths already call this; without it, the terminal/code-exec + # tools and relative-path resolution keep operating in the wrong repo. Idempotent and a no-op when + # the session recorded no cwd. See #38562. self._restore_session_cwd(session_meta) self._restore_session_yolo(session_meta) self._restore_session_model(session_meta) @@ -1301,6 +1334,8 @@ class CLICommandsMixin: return _cp(f" Session not found: {target}", " Use /sessions or `hermes sessions list` to see available sessions.") try: + # If the target is the empty head of a compression chain, redirect to the descendant that + # actually holds the transcript. See #15000. resolved_id = self._session_db.resolve_resume_session_id(target_id) except Exception: resolved_id = target_id @@ -1550,6 +1585,11 @@ class CLICommandsMixin: if not concept: # prompt_toolkit owns stdin on this daemon thread — raw input() never renders and eats # keystrokes; prefer the thread-aware helper (None when prompting isn't safe). + # Bare /hatch is dispatched from the process_loop daemon thread while prompt_toolkit owns stdin + # — a raw input() here types into a prompt that never renders and swallows the next keystrokes + # (same class as #23185; found in the Aug 2026 full-surface CLI QA sweep: bare /hatch left the + # session eating input until Ctrl+C). Route through the thread-aware prompt helper, which uses + # run_in_terminal on the main thread and cancels cleanly (None) when prompting isn't safe. prompt_helper = getattr(self, "_prompt_text_input", None) try: concept = ((prompt_helper or input)("(o_o) Describe your pet: ") or "").strip() @@ -1802,6 +1842,10 @@ class CLICommandsMixin: if store is None: # No live agent store (e.g. Desktop GUI): use a fresh on-disk store, as the gateway # does — same MEMORY/USER.md, same configured char limits. + # Apply against a freshly loaded on-disk store, mirroring the gateway path + # (gateway/slash_commands.py): it persists to the same MEMORY/USER.md and creates MEMORY.md on + # the first approved write. Without this the shared handler returns "memory store unavailable". + # See #46783. from tools.memory_tool import load_on_disk_store store = load_on_disk_store() out = handle_pending_subcommand( @@ -1897,6 +1941,9 @@ class CLICommandsMixin: if not self._agent_running: self._spinner_text = text if self._app: + # Display result in the CLI (thread-safe via patch_stdout). Force a TUI refresh + # first so spinner/status bar don't overlap with the output (fixes #2718). + # Same TUI refresh pattern as success path (#2718) self._app.invalidate() bg_agent.thinking_callback = _bg_thinking @@ -2190,6 +2237,15 @@ class CLICommandsMixin: _cp(f" ▶ Goal resumed: {state.goal}") # Resume must restart work, not just flip state: queue the continuation prompt the same # way /goal queues its kickoff. + # Resume must restart work, not just flip persisted state (#75362): enqueue the canonical + # continuation through the adapter FIFO — the same path the post-turn judge uses — so the next turn + # fires as soon as this reply is delivered. A real user message already queued still preempts + # naturally, and pause/clear's stale-continuation cleanup recognizes it. + # See #75362. + # An `exec` result is display-only — nothing would re-enter the conversation loop until the user + # typed another message. Return a `send` dispatch carrying the canonical continuation prompt so the + # client fires the next turn immediately; `display` keeps the transcript showing the concise + # invocation instead of the model-facing scaffolding. See #75362. prompt = mgr.next_continuation_prompt() if prompt and self._kick_goal(prompt): _cp(_dim_line('Continuing now — taking the next step.')) diff --git a/hermes_cli/cli_info_mixin.py b/hermes_cli/cli_info_mixin.py index 154fa33022..834aa5535a 100644 --- a/hermes_cli/cli_info_mixin.py +++ b/hermes_cli/cli_info_mixin.py @@ -27,6 +27,9 @@ _TOOL_PROGRESS_CYCLE = ["off", "new", "all", "verbose"] # Raw ANSI (not Rich markup): _cprint routes through prompt_toolkit's renderer, while Rich markup # written to stdout gets mangled by patch_stdout's StdoutProxy ('?[33mTool progress: NEW?[0m'). _TOOL_PROGRESS_LABELS = { + # Use raw ANSI codes via _cprint so the output is routed through prompt_toolkit's renderer. + # self.console.print() with Rich markup writes directly to stdout which patch_stdout's StdoutProxy + # mangles into garbled sequences like '?[33mTool progress: NEW?[0m' (#2262). "off": f"{_Colors.DIM}Tool progress: OFF{_Colors.RESET} — silent mode, just the final response.", "new": f"{_Colors.YELLOW}Tool progress: NEW{_Colors.RESET} — show each new tool (skip repeats).", "all": f"{_Colors.GREEN}Tool progress: ALL{_Colors.RESET} — show every tool call.", @@ -770,6 +773,11 @@ class CLIInfoMixin: When opted out it only notifies and points at ``/reload-mcp`` — every reload rebuilds the tool surface and INVALIDATES the provider prompt cache (next message re-sends the full prefix), so silent reloads are wrong when external tooling rewrites config.yaml often. + + Instead it notifies the user that the config changed and that they can apply it with ``/reload-mcp`` + — while warning that ``/reload-mcp`` rebuilds the tool surface and **invalidates the provider prompt + cache** (the next message re-sends the full input prefix, expensive on long-context / high-reasoning + models). See #1474. """ import yaml as _yaml diff --git a/hermes_cli/cli_loops_mixin.py b/hermes_cli/cli_loops_mixin.py index 65a45314b7..259ac5b45d 100644 --- a/hermes_cli/cli_loops_mixin.py +++ b/hermes_cli/cli_loops_mixin.py @@ -34,6 +34,7 @@ class CLILoopsMixin: # /exit --delete also removes the session's transcripts + SQLite history. from cli import _DIM, _RST, _cprint, _slash_args _args = _slash_args(cmd_original).lower() + # Ported from google-gemini/gemini-cli#19332. if _args in {"--delete", "-d"}: self._delete_session_on_exit = True elif _args: diff --git a/hermes_cli/cli_modal_mixin.py b/hermes_cli/cli_modal_mixin.py index 407c4aef7e..b74c0bca12 100644 --- a/hermes_cli/cli_modal_mixin.py +++ b/hermes_cli/cli_modal_mixin.py @@ -218,7 +218,14 @@ class CLIModalMixin: ``run_in_terminal`` only works on the main-thread loop; on the ``process_loop`` daemon thread a bare ``input()`` would block forever on loop-owned stdin, so with an app running - off-main we cancel cleanly (None) — mirroring ``_stdin_fallback`` in the modal prompt.""" + off-main we cancel cleanly (None) — mirroring ``_stdin_fallback`` in the modal prompt. + + Mirrors the thread-aware guard in ``_run_curses_picker``: ``run_in_terminal`` returns a coroutine + that must be awaited by the prompt_toolkit event loop, which only exists on the main thread. Slash + commands are dispatched from the ``process_loop`` daemon thread (see issue #23185), so calling + ``run_in_terminal`` from there orphans the coroutine — ``_ask`` never runs, and user keystrokes leak + into the composer instead. Fall back to a direct ``input()`` when we're off the main thread. + """ result = [None] def _ask(): @@ -228,6 +235,11 @@ class CLIModalMixin: pass in_main_thread = threading.current_thread() is threading.main_thread() + # Slash-worker guard (#23185 / billing auto-reload hang): when a prompt_toolkit app is running but + # we're on a non-main thread (the process_loop / TUI slash-worker daemon thread), stdin is owned by + # the event loop / JSON-RPC pipe. A bare input() there blocks forever until the worker's 45s timeout + # fires. We cannot safely prompt off the main thread, so cancel cleanly (None) instead of hanging — + # mirrors the _stdin_fallback discipline in _prompt_text_input_modal. if self._app and not in_main_thread: self._invalidate() return None @@ -282,7 +294,14 @@ class CLIModalMixin: prompt_toolkit's stdin ownership: prompt above the TUI, Enter read as EOF). All platforms drive the modal via ``self._app.loop`` + ``call_soon_threadsafe``; raw ``input()`` is kept only for the safe cases (no app, no loop, scheduling failure) — on Windows a non-main-thread - input() deadlocks against prompt_toolkit, so that case cancels instead.""" + input() deadlocks against prompt_toolkit, so that case cancels instead. + + **Platform note (Windows — issue #33961):** Earlier code bypassed the modal on ``sys.platform == + "win32"`` and fell back to a raw ``input()`` prompt. When the confirm was triggered from the + ``process_loop`` daemon thread (the normal case) that ``input()`` ran off the main thread and + deadlocked against prompt_toolkit's stdin ownership — the user saw a frozen cursor and Ctrl-C was + swallowed (bare ``/reset`` froze; ``/reset now`` worked only because it skips the prompt entirely). + """ if not choices: return None if not getattr(self, "_app", None): @@ -295,6 +314,9 @@ class CLIModalMixin: in_main_thread = threading.current_thread() is threading.main_thread() def _stdin_fallback() -> str | None: + # On native Windows a raw input() from a non-main thread deadlocks against prompt_toolkit's + # stdin ownership (#33961). With an app running we cannot safely prompt off the main thread, so + # cancel cleanly (None) rather than hang the terminal. if sys.platform == "win32" and not in_main_thread: self._invalidate() return None @@ -497,7 +519,15 @@ class CLIModalMixin: """Confirm a destructive slash command (``/clear``, ``/new``/``/reset``, ``/undo``): returns ``"once"``, ``"always"`` (persists the opt-out) or ``None`` (cancelled). Gate off → "once" silently; ``now`` / ``--yes`` / ``-y`` in ``cmd_original`` bypasses the modal (callers strip - the tokens via :meth:`_split_destructive_skip`).""" + the tokens via :meth:`_split_destructive_skip`). + + Inline-skip: if ``cmd_original`` contains ``now``, ``--yes``, or ``-y`` as an argument (e.g. + ``/reset now``, ``/new --yes My title``), the modal is bypassed and ``"once"`` is returned + immediately. This is an escape hatch for non-interactive use and for the degraded path where the + modal can't be marshaled onto the app loop (native Windows itself now drives the modal normally — + see #33961). Callers are responsible for stripping the skip tokens from any remaining argument + parsing (see :meth:`_split_destructive_skip`). + """ if cmd_original and self._split_destructive_skip(cmd_original)[1]: return "once" return _gated_confirm( @@ -546,7 +576,10 @@ class CLIModalMixin: open-ended questions) and block until the key bindings answer or the timeout dismisses it (the agent is then told to decide). ``multi_select`` shows checkboxes (Space toggles). A non-empty ``questions`` list switches to the batch panel and returns - ``{"answers": {qid: raw}}`` (plus ``"timed_out": True`` on a partial deadline expiry).""" + ``{"answers": {qid: raw}}`` (plus ``"timed_out": True`` on a partial deadline expiry). + + The single-question path below is unchanged. See #18450. + """ from cli import CLI_CONFIG, _DIM, _RST, _cprint from tools.clarify_gateway import resolve_clarify_timeout @@ -580,6 +613,7 @@ class CLIModalMixin: _cprint(f"\n{_DIM}(clarify timed out after {timeout}s — agent will decide){_RST}") return _CLARIFY_TIMEOUT_REPLY + # --- Batch clarify (multi-question, issue #18450) ----------------------- def _clarify_batch_set_active(self, state, index) -> None: """Point the batch clarify panel at question ``index``: mirror it into the flat keys the single-question keybindings/renderer read so ↑/↓/Space/number keys work unchanged; diff --git a/hermes_cli/cli_model_switch_mixin.py b/hermes_cli/cli_model_switch_mixin.py index 7b0ce6403e..5176c75c15 100644 --- a/hermes_cli/cli_model_switch_mixin.py +++ b/hermes_cli/cli_model_switch_mixin.py @@ -35,6 +35,9 @@ def _heal_bare_custom_provider(provider, *, base_url, model): if str(provider or "").strip().lower() != "custom": return provider try: + # Heal bare "custom" persisted by older builds / gateway turns: it's the resolved billing class, not + # a routable identity. (Stricter than the TUI gateway's recovery, which keeps bare "custom" when a + # base_url exists — the CLI's resolve path would hard-fail on it, #14676.) from hermes_cli.runtime_provider import canonical_custom_identity return canonical_custom_identity(base_url=base_url or None, model=model or None) or None except Exception: @@ -168,6 +171,12 @@ def _persist_global_switch(cli, result) -> None: HermesCLI._clear_persisted_context_for_model_switch(cli, result) save_config_value("model.default", result.new_model) save_config_value("model.provider", result.target_provider) + # base_url/api_mode were previously never persisted here, so a global switch left the OLD provider's + # endpoint/wire-protocol in config.yaml. result.base_url/api_mode are always freshly resolved for the + # target provider (see model_switch.py), so sync them every time; None clears a value the new provider + # doesn't need (#25106). + # See _apply_model_switch_result above for why base_url/api_mode must be synced on every global switch + # (#25106). save_config_value("model.base_url", result.base_url or None) save_config_value("model.api_mode", result.api_mode or None) @@ -300,6 +309,12 @@ class CLIModelSwitchMixin: ``gateway_runtime`` (CLI --resume) and top-level keys (TUI session.resume) — from one or-None dict so stale keys are DELETED (``_merge_model_config_json`` only deletes on explicit None) and the shapes never diverge. + + Writes the model column plus the runtime route so ``--resume`` (CLI, reads ``gateway_runtime``) and + ``session.resume`` (TUI/desktop, reads top-level ``model_config`` keys via + ``_stored_session_runtime_overrides``) both restore the switched provider instead of recombining the + model with the ambient default (#79536). Mirrors the gateway's ``update_session_model()`` call. + getattr: tests drive the switch paths with ``object.__new__`` stubs. """ from cli import logger db = getattr(self, "_session_db", None) @@ -310,6 +325,12 @@ class CLIModelSwitchMixin: "provider": _heal_bare_custom_provider( result.target_provider, base_url=result.base_url, model=result.new_model, ) or None, + # Both shapes use the same or-None discipline so stale keys from a previous switch are deleted + # (not merely omitted) in BOTH the nested gateway_runtime dict (CLI reader) and the top-level + # keys (TUI gateway reader). _merge_model_config_json only deletes on explicit None, so falsy + # values must be converted, not filtered. Deriving the top-level from **route guarantees the two + # shapes can never diverge — the asymmetry that caused the original stale-key bug (#85261 + # simplify-code review). "base_url": result.base_url or None, "api_mode": result.api_mode or None} try: @@ -563,6 +584,11 @@ class CLIModelSwitchMixin: api_key=result.api_key, base_url=result.base_url, api_mode=result.api_mode, capabilities=getattr(result, "runtime_capabilities", None)) except Exception as exc: + # The agent rolled itself back to the old working model/client. Roll the CLI's own staged + # fields back too and abort the rest of the commit (note + success print) so a failed switch + # is a no-op rather than a dead session (#50163). + # Agent rolled itself back; roll the CLI back too and abort so a failed switch is a no-op + # rather than a dead session (#50163). for _k, _v in _cli_snapshot.items(): setattr(self, _k, _v) _cprint( diff --git a/hermes_cli/cli_session_mixin.py b/hermes_cli/cli_session_mixin.py index dc3df7e1dd..91f8b7cda2 100644 --- a/hermes_cli/cli_session_mixin.py +++ b/hermes_cli/cli_session_mixin.py @@ -512,6 +512,12 @@ class CLISessionMixin: # the current turn to the OLD session before rotating or it is silently lost. if self.agent: with contextlib.suppress(Exception): + # Flush any un-persisted messages from the current turn to the old session *before* + # rotating. /new can be called mid-turn when _flush_messages_to_session_db() has not + # yet run — without this, messages generated during the current turn are silently lost + # on session rotation (#47202). + # See #47202. + # See #47202. self.agent._flush_messages_to_session_db( self.conversation_history, conversation_history=self.conversation_history) with contextlib.suppress(Exception): @@ -530,6 +536,8 @@ class CLISessionMixin: self.reasoning_config = _parse_reasoning_config( CLI_CONFIG["agent"].get("reasoning_effort", "")) # Session-scoped overrides (/model --session, /fast, one-turn restores) don't carry over. + # Re-derive model/provider and service tier from config.yaml so a session-only switch never leaks + # into the next session (#48055, #23131). self._pending_one_turn_model_restore = None self.service_tier = _parse_service_tier_config(CLI_CONFIG["agent"].get("service_tier", "")) _reset_model_to_config_default(self, silent) @@ -591,6 +599,8 @@ class CLISessionMixin: chance to be a bare session number. The pending state is one-shot — cleared on the first input regardless of outcome, so a stray later number is never hijacked. Returns True if the input was consumed (caller must not treat it as chat). + + See #34584. """ from cli import _cprint pending = self._pending_resume_sessions @@ -678,6 +688,11 @@ class CLISessionMixin: f.write(content) label = {"json": "JSON", "md": "Markdown", "html": "HTML"}[fmt] print(f"(^_^)v Conversation saved to: {path} ({label})") + # #76354 review F5: the worker thread also rebound the session ContextVar inside its own + # (copied) context, which the caller never sees — and get_session_env() prefers an already-bound + # ContextVar over os.environ. Rebind in the CALLER's context so post-compression + # tools/subprocesses on this thread resolve HERMES_SESSION_ID to the child id after an + # out-of-place rotation (idempotent when no rotation happened). if self.session_id: print(f" Resume the live session with: hermes --resume {self.session_id}") except Exception as e: @@ -881,6 +896,7 @@ class CLISessionMixin: self._publish_truncated_history(truncated, invalidate_prompt=True) # Same hook /branch fires; rewound=True invalidates per-turn document caches. _mm = getattr(self.agent, "_memory_manager", None) + # See #21910, #6672. if _mm is not None and self.session_id: with contextlib.suppress(Exception): _mm.on_session_switch(self.session_id, parent_session_id="", reset=False, rewound=True) @@ -1080,6 +1096,9 @@ class CLISessionMixin: # system_message=None so _compress_context rebuilds the prompt from scratch; # passing _cached_system_prompt duplicated the identity block. + # Passing _cached_system_prompt caused duplication because _build_system_prompt appends + # system_message to prompt_parts which already contain the agent identity — resulting in the + # identity block appearing twice (issue #15281). compressed, _ = self.agent._compress_context( head, None, approx_tokens=approx_tokens, focus_topic=focus_topic or None, force=True, defer_context_engine_notification=True) @@ -1222,6 +1241,15 @@ class CLISessionMixin: self._write_terminal_breadcrumb() try: + # Create the DB session row now that _cached_system_prompt is populated, so the persisted + # snapshot is written non-NULL on the first turn (Issue #45499). Idempotent: + # _ensure_db_session() no-ops once the row exists. Must run BEFORE preflight compression: + # in-place compaction inserts message rows referencing this session (archive_and_compact), and + # rotation creates a child with parent_session_id pointing at it — with PRAGMA foreign_keys=ON, + # a missing parent row fails both INSERTs on a fresh oversized first turn. The user-turn crash + # persist itself runs LATER (after memory prefetch / pre_llm_call), so the row is written once + # with its final api_content — both steps take the same per-agent persist lock as CLI close + # persistence. if persist_lock is None: _snapshot_and_persist() else: @@ -1232,9 +1260,20 @@ class CLISessionMixin: def _print_exit_summary(self, clear_screen: bool = True): """Print session resume info on exit. ``clear_screen`` (interactive TUI teardown) - wipes screen + scrollback first; single-query mode passes False to keep the answer.""" + wipes screen + scrollback first; single-query mode passes False to keep the answer. + + Args: clear_screen: When True (default), clear the terminal screen and scrollback before printing + the summary. See #38252, #53009. + """ from cli import datetime if clear_screen: + # Clear the screen + scrollback before printing the summary so the live bottom chrome (status + # bar, input box, separator rules) and the rest of the session transcript don't get stranded + # above the exit summary (#38252). By this point app.run() has returned and prompt_toolkit has + # restored terminal modes, so writing raw escapes to stdout is safe. ESC[3J clears scrollback, + # ESC[2J clears the visible screen, ESC[H homes the cursor — so the summary prints at a clean + # top-left. Falls back to the platform clear command if stdout isn't a TTY-capable stream. + # Honors NO_COLOR/dumb terminals by skipping silently when there's no real console. self._clear_terminal_on_exit() print() msg_count = len(self.conversation_history) diff --git a/hermes_cli/cli_status_bar_mixin.py b/hermes_cli/cli_status_bar_mixin.py index 00d0a72198..8950ec82cd 100644 --- a/hermes_cli/cli_status_bar_mixin.py +++ b/hermes_cli/cli_status_bar_mixin.py @@ -481,7 +481,15 @@ class CLIStatusBarMixin: """Full viewport width for printed scrollback box rules, floored at 32 cols so tiny terminals never hit negative ``'─' * (w - 2)`` math. (The old 56-col clamp against reflow-on-shrink is gone: the ``_output_screen_diff`` patch keeps chrome out of - scrollback, and reflow of already-printed borders is a cosmetic artifact.)""" + scrollback, and reflow of already-printed borders is a cosmetic artifact.) + + Previously this clamped to ``max(32, min(width, 56))`` as a defense against terminal-emulator reflow + on column-shrink (#25975, salvaging 24403). That clamp made response/reasoning borders look stubby + on any modern wide terminal. We now trust the prompt_toolkit ``_output_screen_diff`` monkey-patch + landed in #26137 (salvaging 25981) to keep chrome out of scrollback in the first place, and accept + that an aggressive column-shrink may visually reflow already printed Panel borders — that's a + cosmetic artifact of stamped scrollback history, not a live-render bug. + """ if width is None: try: width = shutil.get_terminal_size((80, 24)).columns @@ -891,7 +899,10 @@ class CLIStatusBarMixin: """The push-to-talk key label every voice-facing hint advertises. Cached at startup (``set_voice_record_key_cache``) because the prompt_toolkit binding is registered once — re-reading config per render could advertise a chord that isn't bound — and this sits on - the hot render path.""" + the hot render path. + + Two reasons (Copilot round-13 on 19835): See #19835. + """ return getattr(self, "_voice_record_key_display_cache", None) or "Ctrl+B" def set_voice_record_key_cache(self, raw_key: object) -> None: diff --git a/hermes_cli/cli_stream_mixin.py b/hermes_cli/cli_stream_mixin.py index 60c719522b..7d1075c8a9 100644 --- a/hermes_cli/cli_stream_mixin.py +++ b/hermes_cli/cli_stream_mixin.py @@ -198,6 +198,7 @@ class CLIStreamMixin: # try/except rather than path.exists(): the paste file may be deleted between # check and read (TOCTOU), silently dropping the input. try: + # See #17666. return path.read_text(encoding="utf-8") except (OSError, IOError): logger.warning("Paste file gone or unreadable, returning placeholder: %s", path) diff --git a/hermes_cli/cli_tui_mixin.py b/hermes_cli/cli_tui_mixin.py index c5aa0021fb..c26ea1de68 100644 --- a/hermes_cli/cli_tui_mixin.py +++ b/hermes_cli/cli_tui_mixin.py @@ -1957,6 +1957,13 @@ class CLITuiMixin: mid-session. """ from cli import logger + # Voice push-to-talk key: configurable via config.yaml (voice.record_key) Default: Ctrl+B (avoids + # conflict with Ctrl+R readline reverse-search). Config spellings (ctrl/control/alt/option/opt) are + # normalized to prompt_toolkit's c-x / a-x format via + # ``normalize_voice_record_key_for_prompt_toolkit`` so the same config value binds identically in + # the TUI and CLI (Copilot round-9 review on #19835). ``super``/``win``/``windows`` configs silently + # fall back to the default here since prompt_toolkit has no super modifier — log a warning so users + # notice the TUI/CLI split instead of a silent mismatch (round-11). _raw_key: object = "ctrl+b" try: from hermes_cli.config import load_config @@ -1977,6 +1984,10 @@ class CLITuiMixin: _raw_key) except Exception: _voice_key = "c-b" + # Cache the UI label here — same ``_raw_key`` that drives the prompt_toolkit binding below. Every + # status / placeholder / recording-hint render reads this cached value so display can never drift + # from the live keybinding even if the user edits voice.record_key mid-session (Copilot round-13 on + # #19835). self.set_voice_record_key_cache(_raw_key) return pt_key_to_sequence(_voice_key) diff --git a/hermes_cli/cli_voice_mixin.py b/hermes_cli/cli_voice_mixin.py index c6ec8aaff8..252e8f1090 100644 --- a/hermes_cli/cli_voice_mixin.py +++ b/hermes_cli/cli_voice_mixin.py @@ -524,6 +524,7 @@ class CLIVoiceMixin: self._tts_lease_async(True) # warm the engine so the first reply isn't dead air # Startup-pinned label so the advertised shortcut always matches the live # prompt_toolkit binding (live config would drift after a mid-session edit). + # See #19835. _cprint(f"\n{_ACCENT}Voice mode enabled{tts_status}{_RST}") _cprint(f" {_DIM}{self._voice_record_key_label()} to start/stop recording{_RST}") # Spoken-stop hint from voice.stop_phrases (first entry); "" when disabled. @@ -540,7 +541,11 @@ class CLIVoiceMixin: def _typed_voice_stop(self, user_input) -> bool: """Typed bare stop phrase during an active voice chat ends the chat (mirrors the spoken one; outside voice mode "stop" passes through to the agent). Exact-match via - ``is_voice_stop_phrase``, so longer messages containing "stop" are never swallowed.""" + ``is_voice_stop_phrase``, so longer messages containing "stop" are never swallowed. + + Saying "stop" ends the voice chat (PR #73106); TYPING the same bare stop phrase while voice mode is + on must behave identically instead of sending "stop" to the agent as a turn. + """ from cli import _DIM, _RST, _cprint if not isinstance(user_input, str): return False @@ -837,6 +842,7 @@ class CLIVoiceMixin: _cprint(f" Recording: {'YES' if self._voice_recording else 'no'}") # Startup-pinned label so /voice status always matches the live prompt_toolkit # binding (live config would drift after a mid-session config edit). + # See #19835. _cprint(f" Record key: {self._voice_record_key_label()}") _cprint(f"\n {_BOLD}Requirements:{_RST}") for line in reqs["details"].split("\n"): diff --git a/hermes_cli/codex_models.py b/hermes_cli/codex_models.py index f3f477115c..178800b2e7 100644 --- a/hermes_cli/codex_models.py +++ b/hermes_cli/codex_models.py @@ -30,6 +30,14 @@ DEFAULT_CODEX_MODELS: List[str] = [ # availability, not Codex availability, so fetch/cache paths must not filter on it. "gpt-5.3-codex-spark"] +# gpt-5.3-codex-spark is in research preview and is exposed *only* via the Codex CLI / OAuth backend +# (chatgpt.com/backend-api/codex/models) for ChatGPT Pro subscribers. It is NOT available in the public +# OpenAI API, so it intentionally stays out of the "openai" provider catalog in hermes_cli/models.py — only +# the openai-codex (OAuth) provider surfaces it. The Codex backend reports ``supported_in_api: false`` for +# this slug; that flag describes API availability, not Codex backend availability, so the fetch/cache code +# paths below intentionally do not filter on it. PR #12994 removed this entry on the assumption it was +# unsupported — that was wrong; restored here. Keep it in the curated fallback so Pro users still see Spark +# in `/model` when live discovery is unavailable (offline first run, transient API failure). _FORWARD_COMPAT_TEMPLATE_MODELS: List[tuple[str, tuple[str, ...]]] = [ ("gpt-5.6-sol", ("gpt-5.5", "gpt-5.4")), ("gpt-5.6-terra", ("gpt-5.5", "gpt-5.4")), diff --git a/hermes_cli/codex_runtime_plugin_migration.py b/hermes_cli/codex_runtime_plugin_migration.py index e712f18e91..0b379b043c 100644 --- a/hermes_cli/codex_runtime_plugin_migration.py +++ b/hermes_cli/codex_runtime_plugin_migration.py @@ -304,6 +304,11 @@ def _query_codex_plugins( for plugin in plugins if isinstance(plugins, list) else (): if not isinstance(plugin, dict) or not plugin.get("installed", False): continue + # Skip plugins codex itself reports as unavailable (broken install, missing OAuth, removed from + # marketplace, etc.). Cf. openclaw/openclaw#80815 — OpenClaw learned to gate migration on app + # readiness to avoid writing config that would fail at activation time. Our migration writes to + # codex's config.toml directly, so a broken plugin would surface as a codex error on first use. + # Skipping it here keeps the migrated config clean and the user's first codex turn from failing. availability = str(plugin.get("availability") or "").upper() if availability and availability != "AVAILABLE": logger.debug("skipping plugin %s: availability=%s", plugin.get("name"), @@ -343,6 +348,14 @@ def _build_hermes_tools_mcp_entry() -> dict: """ import sys env: dict[str, str] = {} + # HERMES_HOME passes through IF SET so the MCP subprocess sees the same config / auth / sessions DB as + # the parent CLI. Read from os.environ (not get_hermes_home()) on purpose: when the env var is unset we + # want codex's subprocess to inherit whatever HERMES_HOME its launcher sets at runtime (systemd unit, + # gateway, kanban dispatcher, custom shell), rather than burning the migrate-time resolved default into + # config.toml — that would override the launcher's HERMES_HOME and pin the subprocess to the wrong + # profile. The pytest-tempdir guard below catches the issue #26250 Bug C scenario: a sibling test's + # monkeypatch.setenv("HERMES_HOME", tmp_path) would otherwise leak a transient pytest tempdir into the + # user's real ~/.codex/config.toml and silently brick codex once the tempdir is GC'd. hermes_home = os.environ.get("HERMES_HOME") or "" if hermes_home and not _looks_like_test_tempdir(hermes_home): env["HERMES_HOME"] = hermes_home diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index b37ac4d89f..13173f05f1 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -407,7 +407,10 @@ def should_bypass_active_session(command_name: str | None) -> bool: /insights, /title, /resume, /retry, /undo, /compress, /usage, /reload-mcp, /sethome, /reset) would silently interrupt the agent AND get discarded — a zero-char response. See issue #5057 / PRs #6252, #10370, #4665. ACTIVE_SESSION_BYPASS_COMMANDS remains the subset with - explicit Level-2 handlers; the rest fall through to the catch-all.""" + explicit Level-2 handlers; the rest fall through to the catch-all. + + See #10370, #4665, #5057, #6252. + """ return resolve_command(command_name) is not None if command_name else False diff --git a/hermes_cli/commands_completion.py b/hermes_cli/commands_completion.py index d04cc5215b..30e5d33624 100644 --- a/hermes_cli/commands_completion.py +++ b/hermes_cli/commands_completion.py @@ -131,6 +131,10 @@ def _tools_completions(sub_text: str, sub_lower: str): from hermes_cli.tools_config import ( CONFIGURABLE_TOOLSETS, _get_platform_tools, _get_plugin_toolset_keys) # Readonly loader: per keystroke and never mutates, so skip load_config()'s deepcopy. + # Read-only path: the completer only inspects the config (toolset enable state + MCP server names) — it + # never mutates it. Use the readonly loader so the per-keystroke completion doesn't pay the defensive + # deepcopy (perf(agent) #74322 converted 29 call sites to the readonly loader; this per-keystroke site + # was missed). config = load_config_readonly() enabled = _get_platform_tools(config, "cli", include_default_mcp_servers=False) mcp_servers = config.get("mcp_servers") or {} diff --git a/hermes_cli/commands_platforms.py b/hermes_cli/commands_platforms.py index d9972ca3c3..b30f965209 100644 --- a/hermes_cli/commands_platforms.py +++ b/hermes_cli/commands_platforms.py @@ -302,6 +302,17 @@ def discord_skill_commands_by_category( ``(name, description, cmd_key)``, names clamped to 32 chars, descriptions to 100. No per-group cap (the caller flattens into one autocomplete callback); ``hidden_count`` only reports 32-char clamp collisions against reserved names or earlier skills. + + Scan roots include the local ``SKILLS_DIR`` **and** any configured ``skills.external_dirs`` — matching + the widened filter applied to the flat ``discord_skill_commands()`` collector in #18741. Without this + parity, external-dir skills are visible via ``hermes skills list`` and the agent's ``/skill-name`` + dispatch but silently absent from Discord's ``/skill`` autocomplete. + The legacy 25-group × 25-subcommand caps (from the old nested ``/skill `` layout) are + **not** applied — the live caller (``_register_skill_group`` in ``gateway/platforms/discord.py``, + refactored in PR #11580) flattens these results and feeds them into a single autocomplete callback, + which scales to thousands of entries without any per-command payload concerns. ``hidden_count`` is + retained in the return tuple for backward compatibility and still reports skills dropped for other + reasons (32-char clamp collision vs a reserved name). """ categories: dict[str, list[tuple[str, str, str]]] = {} uncategorized: list[tuple[str, str, str]] = [] diff --git a/hermes_cli/config.py b/hermes_cli/config.py index 21615522c5..78848329ac 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -52,7 +52,10 @@ class InvalidUserConfigError(RuntimeError): def _backup_corrupt_config(config_path: Path) -> Optional[Path]: """Copy an unparseable ``config.yaml`` to a timestamped ``.corrupt.*.bak``; None on skip/failure. Symlinks are not followed (never clobber whatever a malicious symlink points at). A sibling - backup of the same size means this corruption was already snapshotted — skip to avoid churn.""" + backup of the same size means this corruption was already snapshotted — skip to avoid churn. + + Returns the backup path on success, else ``None``. See #21541. + """ try: if config_path.is_symlink(): return None @@ -90,7 +93,12 @@ _PARSE_FAILURE_DEFAULTS_MSG = ( def _warn_config_parse_failure( config_path: Path, exc: Exception, *, fallback: str = "defaults") -> None: """Surface a config.yaml parse failure to log and stderr (once per file signature). - Silent fallback to ``DEFAULT_CONFIG`` drops every user override, so this must be loud.""" + Silent fallback to ``DEFAULT_CONFIG`` drops every user override, so this must be loud. + + ``fallback`` selects the message wording: ``"defaults"`` (fresh process, nothing else to serve) or + ``"last-known-good"`` (in-process retention of the previously loaded config — see the codex#31188 port + in ``_load_config_impl``). + """ try: st = config_path.stat() key = (str(config_path), st.st_mtime_ns, st.st_size) @@ -198,6 +206,11 @@ _LAST_EXPANDED_CONFIG_BY_PATH: Dict[str, Any] = {} # -> new mtime_ns) so no explicit invalidation is needed. The managed-file signature is folded # in so editing the managed-scope config.yaml invalidates, and the env snapshot invalidates # when a referenced ${VAR} changes value (late .env load, in-process rotation). +# (path, mtime_ns, size) -> cached expanded config dict. load_config() returns a deepcopy of the cached +# value when the file hasn't changed since the last load, skipping yaml.safe_load + _deep_merge + +# _normalize_* + _expand_env_vars (~13 ms/call). save_config() + migrate_config() write via +# atomic_yaml_write which produces a fresh inode, so stat() sees a new mtime_ns and the next load +# repopulates automatically — no explicit invalidation hook. See #58514. _LOAD_CONFIG_CACHE: Dict[str, Tuple[int, int, int, int, Dict[str, Any], Dict[str, Optional[str]]]] = {} # path -> (mtime_ns, size, raw yaml dict) for read_raw_config() (no defaults merged in). _RAW_CONFIG_CACHE: Dict[str, Tuple[int, int, Dict[str, Any]]] = {} @@ -317,7 +330,14 @@ def detect_install_method(project_root: Optional[Path] = None) -> str: The stamp lives next to the code because HERMES_HOME is shared data: a container and a host install can bind-mount the same home, so a home-scoped ``docker`` stamp would make the host ``hermes update`` refuse to run. A legacy ``docker`` value is therefore ignored unless we are - really inside a container, and being in a container alone never implies 'docker'.""" + really inside a container, and being in a container alone never implies 'docker'. + + The supported installs self-identify via the code-scoped stamp: - the curl installer + (scripts/install.sh, the README/website install command) git-clones the repo and stamps ``git`` next to + the code; - the published ``nousresearch/hermes-agent`` image bakes a ``docker`` stamp into + ``/opt/hermes`` at build time. An unsupported manual install dropped into a container (no stamp) falls + through to the ``.git`` checks and behaves like any off-path install. See issue #34397. + """ # The stamp is a property of the running code tree (parent of hermes_cli/), NOT of $HERMES_HOME, # so it survives two installs sharing a home. root = project_root if project_root is not None else get_project_root() @@ -519,7 +539,11 @@ def get_project_root() -> Path: def _resolve_hermes_uid_gid() -> tuple[Optional[int], Optional[int]]: """Read HERMES_UID / HERMES_GID (set by Docker deployments); (None, None) if unset/invalid/Windows. The entrypoint chowns HERMES_HOME once, but subdirs created at runtime (``profiles//``) - need the same chown or they land root:root and block later uid-mapped workers.""" + need the same chown or they land root:root and block later uid-mapped workers. + + Docker containers running Hermes commonly set these to map the in-container user to a host user so + volume-mounted state files end up with the right ownership. See #34107. + """ if sys.platform == "win32": return None, None @@ -534,7 +558,11 @@ def _resolve_hermes_uid_gid() -> tuple[Optional[int], Optional[int]]: def _chown_to_hermes_uid(path) -> None: """Chown ``path`` to ``HERMES_UID:HERMES_GID`` when set; EPERM/ENOENT are non-fatal (the - entrypoint's startup chown -R fixes ownership on the next restart).""" + entrypoint's startup chown -R fixes ownership on the next restart). + + Used by :func:`_secure_dir` to keep ownership consistent across all directories created by + :func:`ensure_hermes_home` on Docker deployments. See #34107. + """ uid, gid = _resolve_hermes_uid_gid() if uid is None and gid is None: return @@ -547,7 +575,12 @@ def _chown_to_hermes_uid(path) -> None: def _secure_dir(path): """chmod a directory owner-only (0700) and apply HERMES_UID/GID ownership. No-op when managed. HERMES_HOME_MODE (e.g. 0701) overrides the mode so a web server can traverse HERMES_HOME to - a served subdirectory without directory listings.""" + a served subdirectory without directory listings. + + Also applies ``HERMES_UID``/``HERMES_GID``-based ownership when those env vars are set (#34107 — Docker + deployments need this so profile subdirs created at runtime by kanban workers don't land as root:root + and block subsequent uid-mapped workers). + """ if is_managed(): return try: @@ -702,7 +735,12 @@ def get_missing_env_vars(required_only: bool = False) -> List[Dict[str, Any]]: def _split_key_path(key: str) -> list[str]: """Split a dotted config-key path, honoring backslash-escaped dots (``a\\.b`` -> ``a.b``). - Backslashes before any other character are preserved verbatim.""" + Backslashes before any other character are preserved verbatim. + + ``hermes config set`` uses ``.`` as the nesting separator, so a key that itself contains a literal dot + (e.g. provider names like ``qwen3.5-397b-wafer``) was silently split into bogus nested segments + (#84064). + """ parts: list[str] = [] current: list[str] = [] i = 0 @@ -724,7 +762,15 @@ def _split_key_path(key: str) -> list[str]: def _greedy_literal_match(container: dict, parts: list) -> Optional[Tuple[str, int]]: """Return ``(literal_key, n_consumed)`` for the longest dotted literal key present in - *container*, or None. With no multi-segment literal this is the historic plain-split walk.""" + *container*, or None. With no multi-segment literal this is the historic plain-split walk. + + Dots in config key names are the norm, not the exception — model IDs (``grok-4.6``, ``glm-5.3``), Matrix + room IDs (``!room:chat.example.cc``), and versioned provider names all embed dots. Users typing + ``providers.myprov.models.grok-4.6.context_length`` do not know the escape syntax exists, so when + navigating an EXISTING mapping we prefer an existing literal key equal to the dot-join of the next N + path segments (longest match wins) over blindly splitting. See #84064 / #80006 / 91095 / #91607 / + #99124. + """ if not isinstance(container, dict) or not parts: return None return next( @@ -735,7 +781,10 @@ def _greedy_literal_match(container: dict, parts: list) -> Optional[Tuple[str, i def _phantom_sibling(container: dict, part: str) -> Optional[str]: """Existing literal dotted key that creating an intermediate mapping ``part`` would shadow (``grok-4`` beside ``grok-4.5``) — the write would produce a phantom sibling the runtime never - reads, so callers fail loudly instead.""" + reads, so callers fail loudly instead. + + Called when a write is about to CREATE a new intermediate mapping named ``part``. See #84064. + """ if not isinstance(container, dict): return None prefix = part + "." @@ -744,7 +793,17 @@ def _phantom_sibling(container: dict, part: str) -> Optional[str]: def _set_nested(config, dotted_key: str, value): """Set a value at a dotted key path, creating intermediate dicts on demand. - Numeric segments index lists; the index must already exist (lists are never grown).""" + Numeric segments index lists; the index must already exist (lists are never grown). + + Guards against #17876: before this fix the code unconditionally replaced any non-dict value (including + lists) with ``{}``, silently destroying list-typed config like ``custom_providers`` whenever a caller + used an indexed path. + Dotted key names (#84064 family): when navigating an existing mapping, an existing literal key equal to + the dot-join of the next N segments is preferred over blind splitting (see ``_greedy_literal_match``), + so ``models.grok-4.6.supports_vision`` lands on the real ``grok-4.6`` entry. And when a write WOULD + create a new intermediate mapping that shadows an existing dotted sibling (``grok-4`` beside + ``grok-4.5``), it raises ``ValueError`` instead of silently writing a phantom the runtime never reads. + """ parts = _split_key_path(dotted_key) current = config i = 0 @@ -848,7 +907,12 @@ def _locate_nested(config, parts: list): def _get_nested(config, dotted_key: str): """Return a dotted-path value (``_MISSING`` when absent); same navigation as ``_set_nested`` - so ``models.grok-4.6.context_length`` reads the real ``grok-4.6`` entry.""" + so ``models.grok-4.6.context_length`` reads the real ``grok-4.6`` entry. + + Mirrors ``_set_nested``'s navigation: honors backslash-escaped dots and prefers an existing literal + dotted key over blind splitting, so ``config get providers.p.models.grok-4.6.context_length`` reads the + real ``grok-4.6`` entry instead of reporting the key unset (#84064). + """ loc = _locate_nested(config, _split_key_path(dotted_key)) if loc is None: return _MISSING @@ -858,7 +922,11 @@ def _get_nested(config, dotted_key: str): def _unset_nested(config, dotted_key: str) -> bool: """Remove a dotted-path value; True if it existed. Empty dict containers left behind are - dropped, while user-authored empty lists and non-empty sibling branches are preserved.""" + dropped, while user-authored empty lists and non-empty sibling branches are preserved. + + Same escape-aware, greedy-literal navigation as ``_set_nested`` / ``_get_nested`` (#84064): unsetting an + unescaped dotted key removes the real literal entry rather than a phantom sibling. + """ loc = _locate_nested(config, _split_key_path(dotted_key)) if loc is None: return False @@ -1126,6 +1194,7 @@ def _validate_fallback_model(fb: Any, issues: List[ConfigIssue]) -> None: def _validate_web_backends(config: Dict[str, Any], issues: List[ConfigIssue]) -> None: """A stale web backend selection otherwise fails only at the first web_search/web_extract call with a generic "no registered provider" error; warn at startup instead.""" + # See #99199. web_cfg = config.get("web") if not isinstance(web_cfg, dict): return @@ -1340,6 +1409,8 @@ def _disable_suspicious_mcp_servers(results: Dict[str, Any], quiet: bool) -> Non """Post-migration: disable exfiltration-shaped MCP stdio entries (hand-edited or from older installs). The stanza is preserved for auditability but marked disabled.""" config = read_raw_config() + # Preserve the stanza for auditability but mark it disabled so the next startup will not spawn it. + # (#45620) raw_mcp_servers = config.get("mcp_servers") if not isinstance(raw_mcp_servers, dict): return @@ -1459,7 +1530,12 @@ def _merge_partial_save(raw: dict, override: dict) -> dict: def _deep_merge(base: dict, override: dict) -> dict: """Recursively merge *override* into *base*: dict-over-dict recurses (so overriding one leaf - keeps sibling defaults), and ``None`` over a dict section is ignored.""" + keeps sibling defaults), and ``None`` over a dict section is ignored. + + An empty section key in config.yaml (``terminal:`` with no value) parses as YAML ``None``; treating that + as an override would replace the entire default dict with ``None`` and crash every downstream consumer + that expects a mapping (#58277). + """ result = base.copy() for key, value in override.items(): over_dict = isinstance(result.get(key), dict) @@ -1489,7 +1565,14 @@ _ENV_REF_RE = re.compile(r"\${([^}]+)}") def _env_ref_lookup(name: str) -> Optional[str]: """Resolve the env var behind a ``${VAR}`` / ``${env:VAR}`` ref — plain ``os.environ`` outside - a profile secret scope (legacy behavior for the default profile).""" + a profile secret scope (legacy behavior for the default profile). + + Inside a scope (a multiplexed gateway turn, a secondary profile's config load, a cron job) the read goes + through ``agent.secret_scope.get_secret`` so the ref resolves against *that* profile's ``.env``: under + multiplexing a miss is a miss, never another profile's ``os.environ`` value (#84079 — every profile + "had" the default profile's ``${MATRIX_ACCESS_TOKEN}`` and fanned out). Same policy as + ``gateway.config._getenv`` and ``get_env_value``. + """ try: from agent.secret_scope import current_secret_scope, get_secret as _get_secret except Exception: @@ -1556,7 +1639,10 @@ def _env_ref_snapshot(obj, snapshot=None): """Map each env-sourced ``${...}`` ref in *obj* to its current value. Stored with cached ``load_config()`` results so a cache hit can detect that the expansion was made against a different environment (load before ``load_hermes_dotenv()``, in-process - rotation) — file mtime/size alone cannot see either.""" + rotation) — file mtime/size alone cannot see either. + + See #58514. + """ if snapshot is None: snapshot = {} if isinstance(obj, str): @@ -1680,7 +1766,24 @@ def _normalize_root_model_keys(config: Dict[str, Any]) -> Dict[str, Any]: ``model`` only when the corresponding ``model.*`` key is empty — never overriding. ``api_base`` (the OpenAI-SDK/LiteLLM name users reach for) is an alias for ``base_url``; the runtime reads only ``model.base_url``. A dict-valued ``default``/``model``/``name`` is flattened so no reader - sees a nested dict, and the id is canonicalized to ``default``.""" + sees a nested dict, and the id is canonicalized to ``default``. + + Also aliases ``api_base`` → ``base_url`` (issue #8919). ``api_base`` is the intuitive name OpenAI-SDK / + LiteLLM users reach for, and ``hermes config set`` blindly accepts any dotted key — so + ``model.api_base`` got written, confirmed, and then silently ignored by the runtime resolver (which + reads only ``model.base_url``), causing requests to fall back to OpenRouter. We migrate the alias to the + canonical key (fallback-only — never override an explicit ``base_url``) and drop the alias so it can't + confuse later loads. + Finally, canonicalizes the model-id key to ``model.default`` (issue #34500). The runtime resolver and + ~14 other readers select the chat model via ``model.default``; ``model.model`` was already aliased + inline at some sites but ``model.name`` was not, so a custom-provider config like ``model: {name: , + provider: }`` resolved to an empty model and the API request went out with ``model=`` (HTTP 400 + from OpenAI-compatible backends) — while display paths (``hermes status``/``dump``) read ``name`` and + *showed* the model, making the failure silent. Normalizing here (the single load/save chokepoint) means + every reader, present and future, sees a populated ``default`` and the stale alias is migrated out of + config.yaml on the next save. Precedence: ``default`` > ``model`` > ``name`` (never overrides an + explicit ``default``, so existing configs are unaffected). + """ model_in = config.get("model") needs_model_work = isinstance(model_in, dict) and ( model_in.get("api_base") @@ -2056,6 +2159,10 @@ def _last_known_good_fallback(config_path: Path, path_key: str, cache_sig, exc: A parse failure must not silently replace the effective config with defaults — that drops EVERY user override, including security-critical ``approvals.deny`` rules, when a gateway user mid-edits config.yaml into broken YAML. Keep serving the last good config until fixed.""" + # Falling through to DEFAULT_CONFIG here drops EVERY user override — including security-critical + # ``approvals.deny`` rules, which are supposed to block commands even under yolo. Within a running + # process we still have the last successfully loaded config — keep serving it until the file is fixed. + # See #31188. lkg = _LAST_EXPANDED_CONFIG_BY_PATH.get(path_key) _warn_config_parse_failure( config_path, exc, fallback="last-known-good" if lkg is not None else "defaults") @@ -2102,6 +2209,8 @@ def _load_config_impl(*, want_deepcopy: bool) -> Dict[str, Any]: # Signatures match, but the cached expansion is only valid if every ${VAR} it was # expanded against still has the same value — otherwise a load before # load_hermes_dotenv() pins unexpanded literals for the process lifetime. + # Without this, a load_config() that ran before load_hermes_dotenv() pins unexpanded literals + # (e.g. auxiliary..api_key) for the life of the process (#58514). env_snapshot = cached[5] if len(cached) > 5 else {} if all(_env_ref_lookup(k) == v for k, v in env_snapshot.items()): return copy.deepcopy(cached[4]) if want_deepcopy else cached[4] @@ -2418,13 +2527,20 @@ def _quote_env_value(value: str) -> str: def _env_line_defines_key(line: str, key: str, *, is_windows: Optional[bool] = None) -> bool: """True when a .env line assigns ``key`` — plain, ``export``-prefixed, or ``KEY = value``. Must match exactly the shapes ``load_env()`` parses; otherwise a hand-added line is invisible - to save (duplicate appended) and remove (line survives -> the value resurrects on next load).""" + to save (duplicate appended) and remove (line survives -> the value resurrects on next load). + + ``load_env()`` accepts the bash-compatible ``export KEY=value`` form (#6659), so the writers must + recognise the same shape. + """ stripped = line.strip() if stripped.startswith("export "): stripped = stripped[7:].lstrip() assigned_key, separator, _value = stripped.partition("=") if not separator: return False + # load_env() strips whitespace around the parsed name, so `KEY = value` IS a live assignment. The + # writers must match the same shape, or a hand-edited spaced line is invisible to save (duplicate + # appended) and remove (line survives -> value resurrects on next load). #67488. return _env_var_policy_name( assigned_key.strip(), is_windows=is_windows ) == _env_var_policy_name(key, is_windows=is_windows) @@ -2434,7 +2550,12 @@ def _publish_env_value(key: str, value: Optional[str]) -> None: """Publish a just-persisted ``.env`` change to the live process. Under a multiplexed gateway a routed profile's write must not land in the SHARED ``os.environ`` where every profile sees it; the installed scope mapping is updated instead so - same-turn reads see the change. All other callers keep the legacy ``os.environ`` publish.""" + same-turn reads see the change. All other callers keep the legacy ``os.environ`` publish. + + ``save_env_value`` / ``remove_env_value`` already target the right file (``get_env_path()`` honors the + profile-home override), but the in-process mirror historically went straight to ``os.environ``. See + #77490, #88441. + """ try: from agent.secret_scope import current_secret_scope, is_multiplex_active @@ -2556,6 +2677,9 @@ def save_env_value_secure(key: str, value: str) -> Dict[str, Any]: value and lifts a prior env-source suppression).""" from hermes_cli.credential_lifecycle import save_provider_env_credential + # Route through the unified credential lifecycle so a rotation via the secret-capture path also + # refreshes any config.yaml mirror of the old value and lifts a prior env-source suppression (#62269 fix + # family). save_provider_env_credential(key, value) return {"success": True, "stored_as": key, "validated": False} @@ -2594,7 +2718,16 @@ def _scoped_environ_get(key: str) -> Optional[str]: def get_env_value(key: str) -> Optional[str]: - """Get a value from ``os.environ`` (scope-aware) or ``~/.hermes/.env``.""" + """Get a value from ``os.environ`` (scope-aware) or ``~/.hermes/.env``. + + The ``os.environ`` read routes through ``agent.secret_scope.get_secret`` so that, under an active + profile scope (multiplexed gateway turn), this is scope-checked rather than leaking another profile's + raw ``os.environ`` value. ``get_secret`` encodes the whole policy: global vars pass through; scope is + authoritative under multiplexing (miss -> None, no environ fallthrough); when multiplexing is off it + behaves exactly like the legacy ``os.environ`` read. Its siblings ``get_env_value_prefer_dotenv`` and + ``gateway.config._getenv`` already work this way — this was the last scope-blind reader of the trio + (#67027). + """ val = _scoped_environ_get(key) return load_env().get(key) if val is None else val @@ -3095,7 +3228,14 @@ def _suggest_closest_key(key: str, candidates: set[str], cutoff: float = 0.6) -> def _validate_config_key(key: str) -> tuple[bool, Optional[str]]: - """Validate a dotted config-key path against the known schema -> ``(is_known, suggestion)``.""" + """Validate a dotted config-key path against the known schema -> ``(is_known, suggestion)``. + + Headline case from #34067: ``gateway.discord.gateway_restart_notification`` was silently written, even + though ``gateway`` only has 4 known sub-keys (``strict``, ``media_delivery_allow_dirs``, + ``trust_recent_files``, ``trust_recent_files_seconds``). The correct path is + ``discord.gateway_restart_notification`` (platform configs live at the top level, not under a + ``platforms`` namespace). + """ if not key: return False, None @@ -3221,7 +3361,12 @@ def _redirect_platform_display_key(key: str) -> tuple[str, Optional[str]]: The gateway resolves per-platform display settings (streaming, show_reasoning, ...) from ``display.platforms``; the top-level ``platforms.`` block holds only connection config. Only known display settings (``OVERRIDEABLE_KEYS``) are redirected. Returns ``(key, note)``; - the gateway import is guarded so the CLI works where the gateway package is unavailable.""" + the gateway import is guarded so the CLI works where the gateway package is unavailable. + + Before #71047 a write such as ``hermes config set platforms.telegram.streaming false`` landed on a key + the gateway never reads: ``config get`` echoed the new value back while the runtime kept the old + ``display.platforms`` one — a silent no-op that looks like a duplicated key to the user. + """ segs = _split_key_path(key) if len(segs) != 3 or segs[0] != "platforms": return key, None @@ -3337,6 +3482,8 @@ def set_config_value(key: str, value: str, force: bool = False): # Unified lifecycle: also rotates any config.yaml mirror of the old value. from hermes_cli.credential_lifecycle import save_provider_env_credential + # Unified lifecycle: also rotates any config.yaml mirror of the old value so a stale + # higher-precedence copy can't win (#62269). save_provider_env_credential(key.upper(), value) print(f"✓ Set {key} in {get_env_path()}") return @@ -3346,6 +3493,11 @@ def set_config_value(key: str, value: str, force: bool = False): # for skills/external apps) but get a post-write "did you mean" hint. key, _redirect_note = _redirect_platform_display_key(key) if _redirect_note: + # Unknown-key notice (#34067): the key is still written (arbitrary keys are supported — top-level + # scalars are bridged into os.environ for skills and external apps), but a plausible-but-wrong + # dotted path like ``gateway.discord.gateway_restart_notification`` previously reported bare success + # and left the user debugging behavior that never changed. Warn after the write so the user gets + # immediate feedback plus a "did you mean" hint, without blocking legitimate unknown keys. print(_redirect_note) is_known, suggestion = _validate_config_key(key) @@ -3365,6 +3517,9 @@ def set_config_value(key: str, value: str, force: bool = False): _exit_invalid(f"✗ {e}") # api_base -> base_url alias at set-time too (mirrors _normalize_root_model_keys). if key.strip().lower() in ("model.api_base", "api_base"): + # Normalize the api_base → base_url alias at set-time too (issue #8919), so a fresh `hermes config + # set model.api_base ...` lands on the canonical key the runtime resolver actually reads, instead of + # being silently ignored. user_config = _normalize_root_model_keys(user_config) key = "model.base_url" print(" (note: 'api_base' is an alias — saved as model.base_url)") @@ -3386,6 +3541,8 @@ def set_config_value(key: str, value: str, force: bool = False): print(f"✓ Set {key} = {_display_value} in {config_path}") warn_unpinned_cron_jobs_after_model_config_change(key, value, user_config) + # Post-write unknown-key notice (#34067): value IS saved, but tell the user the runtime may never read + # it and suggest the likely-intended path. if not is_known and not force: _print_unknown_key_notice(key, suggestion) @@ -3397,6 +3554,7 @@ def get_config_value(key: str, *, as_json: bool = False): value = _MISSING if env_value is None else env_value else: # Mirror set_config_value: read the canonical display.platforms path. + # See #71047. key, _ = _redirect_platform_display_key(key) value = _get_nested(load_config(), key) @@ -3416,6 +3574,7 @@ def unset_config_value(key: str): if _is_env_config_key(key): # Unified lifecycle: also prunes env-seeded credential_pool entries and model-cache rows so # the provider is fully removed instead of left resurrectable. + # See #51071. from hermes_cli.credential_lifecycle import remove_provider_env_credential if not remove_provider_env_credential(key.upper()).get("found"): @@ -3428,6 +3587,7 @@ def unset_config_value(key: str): key, _redirect_note = _redirect_platform_display_key(key) if _redirect_note: + # Mirror set_config_value's display.platforms canonicalization (#71047). print(_redirect_note.replace("saved as", "resolved as")) removed = _unset_nested(user_config, key) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index b156d1a461..2aa681e7eb 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -84,15 +84,21 @@ DEFAULT_CONFIG = { # the next message, but an interrupted cron run is recorded as a permanent failure, so it # must not inherit restart_drain_timeout's 0. Clamped to the shutdown-watchdog leash minus # teardown headroom (~50s unless TimeoutStopSec is raised). 0 = opt out. + # A chat turn interrupted by a restart is announced to the user and resumed on their next message; + # an interrupted cron run is written to jobs.json as a permanent failure that nobody is waiting on, + # so it must not inherit restart_drain_timeout's 0 (#82161). "cron_drain_timeout": 30, # In-band restart (/restart, SIGUSR1): refuse new work, then wait up to this many seconds # for in-flight agents/cron/api runs to finish before stop(). 0 = enter stop() at once. 30 # min is a safety valve for wedged agents, not a target; raise for long unattended turns. + # Default 30 min is a safety valve for wedged agents, not a target latency — an interactive `hermes + # gateway restart` must never block for hours on a turn that wedged (#79133). "restart_after_turn_timeout": 1800, # Max seconds a submitted prompt waits for the deferred agent build (MCP discovery, model # metadata, skills scan) before failing visibly. The prompt is delivered as soon as the # build completes (progress notice past 30s), so this only fires on a hung build. Raise for # many slow/unreachable MCP servers. + # See #63078. "build_wait_timeout": 600, # Hermes-level retry attempts for API errors (connection drops, timeouts, 5xx) wrapping the # whole call; the OpenAI SDK also retries transient errors (max_retries=2). Set 1 for fast @@ -178,6 +184,9 @@ DEFAULT_CONFIG = { # with "[user did not respond within Xm]". CLI clarify blocks indefinitely and ignores this. # 1h because users step away and a shorter value evicted the entry mid-think so a later # button tap hit a dead entry. Lower it to free the running-agent guard sooner. + # Maximum time (seconds) the gateway will block an agent waiting for a clarify-tool response from + # the user. Tradeoff: a higher value holds the gateway's running-agent guard longer for a genuinely + # abandoned prompt — lower it if a single session must free up the guard sooner. See #32762. "clarify_timeout": 3600, # "Still working" status interval (seconds); 0 = off. Lower = faster feedback, more noise; # 180 catches spinning weak-model runs before users /restart. @@ -187,11 +196,13 @@ DEFAULT_CONFIG = { # (ignores startup restore, build sentinels, leases, debounce, other processes; scan cadence # per AIAgent). Notify-only: tells the user to try /new. Distinct from gateway_timeout # (kills the turn) and gateway_notify_interval. 0 = disable. + # See #76354. "session_stall_timeout": 300, # Transcript-sanitiser heal escalation: after this many pre-send heal passes within a # 10-minute window, log one ERROR and queue a ONE-TIME out-of-band notice pointing at /debug # share or `hermes doctor` (status channel only; prompt cache untouched). 0 = no escalation # (per-window WARNINGs still fire). + # See #96870. "sanitizer_heal_escalation_threshold": 3, # Seconds of continuous reconnect failure before a platform gets needs_attention flagged in # gateway status (`hermes status` / fleet monitoring). Retries never stop — a signal, not a @@ -268,6 +279,8 @@ DEFAULT_CONFIG = { # immediately kills the delivery (e.g. Bot Mode handoff replies via message_agent / # bot_relay). Plain background processes without notify_on_complete are never waited on. 0 # disables. + # Bounded linger (seconds) for one-shot CLI runs (-q/-Q/-z) that exit while background processes + # spawned with notify_on_complete=true are still running. See #90879. "oneshot_completion_wait_seconds": 600.0, # Env vars passed into sandboxed terminal/execute_code (skill-declared # required_environment_variables pass through automatically). @@ -517,6 +530,7 @@ DEFAULT_CONFIG = { # threshold_tokens: absolute token cap — compression triggers at the lower of the ratio # threshold and this count. Clamped to the model's context length. "threshold_tokens": None, + # "progress_notices": False, # opt-in (#52995): when True, routine compression "target_ratio": 0.20, # fraction of threshold to preserve as recent tail # tail_mode: "lean" = clamped 2.5%-of-window tail (10K floor / 25K cap) plus chunked # digests, anchor index, verbatim user messages and session_search pointers in the summary @@ -690,6 +704,9 @@ DEFAULT_CONFIG = { # prefer_fast_model opts in to the provider fast tier; auto otherwise = main model. "title_generation": { "enabled": True, + # Note: session_search no longer uses an auxiliary LLM (PR #27590 — single-shape tool returns DB + # content directly). The old ``auxiliary.session_search.*`` block was removed here. Existing + # values in user config.yaml files are harmless leftovers and ignored. "provider": "auto", "model": "", "prefer_fast_model": False, @@ -811,6 +828,7 @@ DEFAULT_CONFIG = { # Seconds between idle prompt_toolkit redraws in the classic CLI; keeps wall-clock # status-bar read-outs ticking and the bottom chrome from going stale. 0 disables it if it # fights terminal auto-scroll in non-fullscreen mode. + # See #45592. "cli_refresh_interval": 1.0, "user_message_preview": { # CLI: submitted user-message lines echoed to scrollback "first_lines": 2, @@ -919,11 +937,15 @@ DEFAULT_CONFIG = { # A detached RUNNING turn is only interrupted once its activity clock (API waits, stream # tokens, tool heartbeats) has been idle this many seconds; an active turn runs to # completion. Default = agent.turn_liveness.timeout_s. 0 = interrupt at grace. + # See #100325, #98028. "ws_orphan_activity_stale_s": 600.0, # On gateway boot, close tui/desktop/subagent rows orphaned by a dead gateway (start AND # newest message older than HERMES_TUI_SESSION_TTL_S, default 6h) with # end_reason='startup_orphan_reap'; otherwise they stay phantom "active" forever. # Messaging-gateway and live sessions are never touched; swept rows stay resumable. + # The ws-orphan grace timer above is in-process, so a gateway restart (update, crash, systemd) + # leaves disconnected sessions ``ended_at IS NULL`` forever — phantom "active" rows in /resume and + # dashboards. See #65194. "startup_orphan_sweep": True, # OAuth gate (engaged when --host is set and --insecure is not), read by the Nous Portal # plugin. Env HERMES_DASHBOARD_OAUTH_CLIENT_ID / HERMES_DASHBOARD_PORTAL_URL win when @@ -1257,6 +1279,11 @@ DEFAULT_CONFIG = { "trace_dir": "", # PII/credential redaction of advisor outputs: "" off | "display" (UI reference blocks + # traces only; aggregator sees raw) | "full" (also the aggregator prompt). + # Advisors can echo PII from the conversation (emails, formatted phone numbers) and credential + # shapes into reference blocks, traces, and the aggregator prompt. Modes ('' = off, the default): + # "display" — redact user-visible surfaces only (reference blocks shown in the UI + saved MoA trace + # records); the aggregator still sees raw advisor text. "full" — additionally redact the advisor + # text injected into the aggregator prompt (issue #59959). "privacy_filter": "", "presets": { "default": { @@ -1306,6 +1333,7 @@ DEFAULT_CONFIG = { # Audit ledger: every skill mutation appends to ~/.hermes/skills/.curator_ledger.jsonl with # before/after hashes (blobs under ~/.hermes/.curator_backups/blobs/); powers `hermes # curator ledger` / `rollback `. Never a gate — failures can't block. + # See #79686. "ledger": True, }, # Curator — background maintenance of AGENT-CREATED skills (never hub-installed): marks @@ -1386,6 +1414,7 @@ DEFAULT_CONFIG = { # Opt-in DM role auth: DISCORD_ALLOWED_ROLES normally authorizes guild messages only (DMs # need DISCORD_ALLOWED_USERS). A guild ID here also authorizes DMs from that guild's members # holding the allowed role. Unset / "" / 0 = off. + # See #12136. "dm_role_auth_guild": "", # discord / discord_admin tools: allowed actions (comma string or YAML list; empty = all, # subject to bot intents; unknown names dropped with a warning): list_guilds, server_info, @@ -1467,6 +1496,19 @@ DEFAULT_CONFIG = { # timeout: seconds before an unanswered prompt fails closed (CLI and gateway). 60s # proved too tight for Telegram/Discord push notifications, hence 300. "approvals": { + # single_query_mode — what to do when a single-query (-q) session hits a dangerous command. -q runs + # export HERMES_INTERACTIVE=1 (for interactive sudo prompts) but have NO user waiting to answer + # approval prompts — an unanswered prompt just waits the full timeout then fails closed, so the + # agent is forced to work around the block (often via execute_code). This setting makes that intent + # explicit: deny — block the command and let the agent find another way (default, safe; mirrors + # cron_mode deny) approve — auto-approve all dangerous commands in single-query mode These surfaces + # bind a session platform like chat gateways do, but have no send_exec_approval and no /approve + # channel — a pending approval there just blocks for the full timeout with nobody to answer (#37284, + # #87509): deny — block the command instantly and let the agent find another way (default, safe; + # mirrors cron_mode deny) approve — auto-approve all dangerous commands on unattended platforms + # Shared by the CLI prompt and gateway/messaging waits. Messaging approvals arrive as a push + # notification the user may not see immediately — 60s proved too tight on Telegram/Discord (the + # prompt expired before the user reached their phone), so the default is 300. "mode": "smart", "timeout": 300, "cron_mode": "deny", @@ -1656,10 +1698,20 @@ DEFAULT_CONFIG = { # Global cap: positive int = the HOST never has more than N tasks 'running' across all # boards and both dispatch lanes. None = ~MemTotal / 512 MiB clamped to [2, 8]; where # MemTotal is unreadable (macOS/Windows) None means no cap. + # Global concurrency cap (#33488): when set to a positive int, the HOST never has more than N tasks + # in 'running' at once — counted across every active board and across both the ready and review + # dispatch lanes (workers are OS processes sharing one machine's memory, so the cap bounds the + # machine, not each board; OOF-30). Unset (None) means "derive from system memory" (OOF-30/OOF-77): + # the dispatcher caps concurrency at roughly MemTotal / 512 MiB, clamped to [2, 8] — e.g. 2 workers + # on a 1 GiB VM. On hosts where total memory can't be read (macOS/Windows), unset falls back to no + # cap. Set an explicit value to override the derived default in either direction. "max_in_progress": None, # Per-profile cap: positive int = no single profile runs more than N workers even if the # global caps allow; blocked tasks defer to the next tick. None = no per-profile cap. Useful # when fan-out would saturate one profile's model/API quota/browser pool. + # Unset (None) means "no per-profile cap" — backward-compatible with existing installs. Useful for + # fan-out workflows that would otherwise saturate one profile's local model / API quota / browser + # pool while leaving other profiles idle. See #21582. "max_in_progress_per_profile": None, # Auto-run the decomposer on Triage tasks every tick. False = manual via `hermes kanban # decompose ` or the dashboard's Decompose button. @@ -1764,6 +1816,9 @@ DEFAULT_CONFIG = { # defaults (200K context, tools on, vision/reasoning off) and get patched. Provider keys: Hermes # or models.dev id; model ids match case-insensitively. Example: {"custom:my-local-vllm": # {"my-llava-model": {"context_window": 8192}}} + # Semantics: 1. NOTE: an explicit model.context_length (global) and a custom_providers per-model + # context_length are user settings at other layers and are consulted in the resolution chain order + # documented in agent/model_metadata.py. 2. See #84482, #8731. "model_overrides": {}, # models.dev registry (context windows, capabilities, pricing, modalities): fetched on startup, # served from cache, refreshed by a background daemon with ETag conditional GET. Override `url` @@ -1814,10 +1869,16 @@ DEFAULT_CONFIG = { # Seconds to wait for one platform to connect at startup/reconnect; raise on "discord # connect timed out" loops (many slash commands to sync). 0/negative = wait forever. Bridged # to HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT, which wins if set explicitly. + # Seconds the gateway waits for a single messaging platform to finish connecting during startup (and + # on reconnect). Discord in particular can blow past the old fixed 30s when an account has many + # slash commands to sync (#19776: 90-173 skills → ~28-31s sync). Raise this if your gateway hits + # "discord connect timed out" / "Timeout waiting for connection to Discord" restart loops. ``0`` or + # negative disables the timeout entirely (wait indefinitely). "platform_connect_timeout": 30, # Event-loop liveness watchdog: a daemon thread probes the asyncio loop; after consecutive # missed probes it dumps all-thread stacks and hard-exits with the service-restart code so # systemd/launchd revives the process instead of leaving a wedged-but-alive zombie. + # Set to false to disable. See #69089. "loop_watchdog": True, # Watchdog tuning (defaults mirror gateway/shutdown_watchdog.py): probe_interval = seconds # between probes; probe_timeout = seconds before an unprocessed probe counts as a miss; @@ -1920,6 +1981,13 @@ DEFAULT_CONFIG = { # full retention window before removal. "auto_prune": True, # Inactive days of ended-session history to keep (= `hermes sessions prune`). + # When true, prune ENDED sessions inactive for retention_days once per (roughly) min_interval_hours + # at CLI/gateway/cron startup. Activity is the latest message timestamp, falling back to creation + # time for empty sessions. Sessions that are still open, pinned, or mid-turn are never deleted — the + # only open rows the sweep touches are stale automation sessions (cron/kanban/subagent/one-shot CLI) + # whose process died without closing them; those are *closed*, not deleted, and get a further full + # retention window before removal. Default true since #54189: without it state.db grows without + # bound (multi-GB installs reported within weeks). "retention_days": 90, # Auto-archive (soft-hide, never delete) sessions with no activity for auto_archive_days, # once per min_interval_hours. Pinned sessions are exempt. @@ -1929,6 +1997,8 @@ DEFAULT_CONFIG = { # VACUUM after a prune that deleted rows (SQLite never reclaims disk on DELETE). VACUUM # blocks writes (~seconds per 100MB), so it runs only at startup, only when ≥1 session was # deleted AND freelist/page_count > 25%. + # SQLite does not reclaim disk space on DELETE — freed pages are just reused on subsequent INSERTs — + # so without VACUUM the file stays bloated even after pruning. See #54189. "vacuum_after_prune": True, # Minimum days between VACUUM rewrites; pruning keeps its normal cadence. "min_vacuum_interval_days": 30, @@ -1997,6 +2067,9 @@ DEFAULT_CONFIG = { # /backups/ (``hermes import`` restores; slow on large homes; ``--backup`` # forces once). off = none (``--no-backup`` forces once). Legacy booleans: true -> full, # false -> off. + # Pre-update safety backup — ONE consolidated mechanism, three modes: Files over 1 GiB (e.g. a + # bloated state.db) are skipped with a warning so the snapshot stays fast. This is the #48200 + # (wrong-path wipe) safety net. "pre_update_backup": "quick", # Full backup zips to retain (older pruned after each success; floored to 1 so the newest is # always kept). The quick snapshot always keeps exactly 1. @@ -2118,6 +2191,9 @@ DEFAULT_CONFIG = { # Disable cua-driver's cursor overlay, which can peg a core when idle (macOS redraw loop; # Linux/WSL2 idle spin). None = auto (off on macOS + headless/ WSL2 Linux, on elsewhere); # True = always disable; False = always enable. + # The overlay shows where agent actions land but can peg a core when idle (macOS vImage redraw loop + # #47032; Linux/WSL2 idle spin #28152). cua-driver ≥ 0.6.x supports --no-overlay; Hermes also calls + # set_agent_cursor_enabled(false) after start_session when this is on. "no_overlay": None, # standard = cua-driver's own approval boundary; bounded = no runtime prompts, anything # outside capability_manifest fails closed. `unrestricted` is NOT accepted here: it stays on @@ -2170,6 +2246,7 @@ DEFAULT_CONFIG = { # Chromium default; x11 = XWayland, for compositors that ignore always-on-top for Wayland # clients (e.g. COSMIC) — also puts the HUD on the solid-window input path; wayland = force # a native Wayland surface. + # See #84011. "ozone_platform_hint": "auto", # Bridged to HERMES_DESKTOP_DISABLE_GPU: auto = disable GPU only on remote displays # (SSH/VNC/RDP); true = always software rendering (no-GPU VMs where the GPU path hangs); diff --git a/hermes_cli/config_providers.py b/hermes_cli/config_providers.py index ca1202c532..3425c60fb2 100644 --- a/hermes_cli/config_providers.py +++ b/hermes_cli/config_providers.py @@ -38,6 +38,7 @@ def _warn_once_per_provider(provider_key: str, signature: str, msg: str, *args: # fell through to hostname-based guessing, so ``api_mode: openai`` could flip to # ``codex_responses`` after an update and break the provider. _API_MODE_ALIASES = { + # See #66543. "openai": "chat_completions", "openai_chat": "chat_completions", "openai-chat": "chat_completions", @@ -130,6 +131,10 @@ def _pick_provider_base_url(entry: Dict[str, Any], provider_key: str) -> str: if not (isinstance(raw_url, str) and raw_url.strip()): continue candidate = raw_url.strip() + # Accept URLs containing unresolved placeholder tokens — both ``${ENV_VAR}`` env-refs and bare + # ``{region}``-style templates — without URL validation. They are expanded at runtime, so a caller + # reaching this normalizer with raw (un-expanded) config would otherwise see the provider silently + # dropped (#14457). if re.search(r"\{[^}]+\}", candidate): return candidate parsed = urlparse(candidate) @@ -477,7 +482,11 @@ def get_custom_provider_context_length( base_url: str, custom_providers: Optional[List[Dict[str, Any]]] = None, config: Optional[Dict[str, Any]] = None) -> Optional[int]: - """Per-model ``context_length`` override from a route-matching entry, or ``None``.""" + """Per-model ``context_length`` override from a route-matching entry, or ``None``. + + Before this helper existed, the lookup was duplicated in ``run_agent.py``'s startup path only; every + other path (notably ``/model`` switch) fell back to the 128K default. See #15779. + """ from hermes_cli.config import get_compatible_custom_providers if not model or not base_url: return None diff --git a/hermes_cli/console_engine.py b/hermes_cli/console_engine.py index 90fff06534..66dc711bcc 100644 --- a/hermes_cli/console_engine.py +++ b/hermes_cli/console_engine.py @@ -109,6 +109,8 @@ def _table_summary(summary: str, *, limit: int = 76) -> str: def _split_line(line: str) -> list[str]: # Windows-safe splitter: plain shlex posix=True eats backslashes in paths. + # See #78293. + # See #83934. from hermes_cli._subprocess_compat import split_command_line try: return split_command_line(line) diff --git a/hermes_cli/container_boot.py b/hermes_cli/container_boot.py index c50dc09bdf..9d189aa919 100644 --- a/hermes_cli/container_boot.py +++ b/hermes_cli/container_boot.py @@ -24,6 +24,20 @@ _AUTOSTART_STATES = frozenset({"running"}) # hard-killed in one of them with no `desired_state` would otherwise stay DOWN on every later boot # (observed: staging stranded at `draining`); map them to `running`, mirroring gateway/run.py. # `starting` / `startup_failed` are excluded: auto-restarting a mid-boot death is the crash-loop. +# A gateway only ever reaches these while it is up and serving, so they are NOT an operator stop and NOT a +# failed boot: - `draining` — written by the drain watcher / scale-to-zero go-dormant path when an +# in-flight quiesce begins (gateway/run.py). - `degraded` — written when the gateway comes up with some +# platforms queued for retry, then "falls through to the normal running state" (gateway/run.py #5196): the +# process is up, serving cron + whatever platforms connected, and the reconnect watcher takes the rest from +# there. When a gateway is hard-killed *while in one of these states* (a container/VM recreate SIGTERMs it +# before `_stop_impl` reaches its terminal-state persist), the last value left in gateway_state.json is the +# transient sub-state. With no explicit `desired_state` to fall back to, treating that literal value as the +# autostart intent would leave the gateway DOWN on every subsequent boot — the gateway never comes back, the +# dashboard is up but messaging stays dark (observed on a relay-opted-in staging instance stranded at +# `draining`, 2026-06; `degraded` is the same wedge class). Map these transient sub-states to `running` so a +# stranded marker reads as the run-intent it actually represents. This mirrors gateway/run.py's #42675 +# handling, which persists `running` (not the mid-shutdown `draining`) when an unexpected signal tears the +# gateway down — extended here to the case where the gateway died before it could persist anything at all. _TRANSIENT_RUNNING_STATES = frozenset({"draining", "degraded"}) # Container-namespaced state is garbage post-restart (an equal PID is a different process). _STALE_RUNTIME_FILES = ("gateway.pid", "processes.json") @@ -57,6 +71,10 @@ def reconcile_profile_gateways( Always registers a ``gateway-default`` slot for the root profile (the implicit profile at the top of ``$HERMES_HOME``): ``hermes_cli.gateway`` maps an empty profile suffix to it, so it is what ``hermes gateway start`` (no ``-p``) targets. + + Without it, bare ``hermes gateway start`` inside the container would land on ``s6-svc -u + /run/service/gateway-default`` → uncaught ``CalledProcessError`` → traceback to the user (PR #30136 + review). """ actions: list[ReconcileAction] = [] # Under a multiplexing root gateway named slots are still registered but must not boot from @@ -230,6 +248,11 @@ def _register_service(scandir: Path, profile: str, *, start: bool) -> None: Mirrors ``S6ServiceManager.register_profile_gateway`` but sets start state via the ``down`` marker (cont-init.d runs before s6-svscan has a control socket). Built in a sibling temp dir and ``Path.replace``d into place so an interrupted write never leaves a half-built dir. + + This matches :meth:`S6ServiceManager.register_profile_gateway` (PR #30136 review item O4) — even though + cont-init.d runs before s6-svscan starts scanning, an atomic publication keeps the contract uniform + between the two registration paths and protects against a half-populated dir if the script is + interrupted mid-write. """ import shutil @@ -271,7 +294,14 @@ _LOG_ROTATE_BYTES = 256 * 1024 def _write_reconcile_log(hermes_home: Path, actions: list[ReconcileAction]) -> None: """Append one line per profile to $HERMES_HOME/logs/container-boot.log (rotated to ``.1``) — - a separate greppable file for "why didn't my profile come back up".""" + a separate greppable file for "why didn't my profile come back up". + + Size-bounded: when the file exceeds ``_LOG_ROTATE_BYTES`` (defaults to 256 KiB ≈ 3000 reconcile lines), + the current file is renamed to ``container-boot.log.1`` (replacing any previous rotation) before the new + entries are appended. This gives long- lived containers a soft cap of ~512 KiB across the two files + without pulling in logrotate or s6-log machinery just for this one append-only file (PR #30136 review + item O3). + """ import time log_dir = hermes_home / "logs" log_dir.mkdir(parents=True, exist_ok=True) diff --git a/hermes_cli/credential_lifecycle.py b/hermes_cli/credential_lifecycle.py index 1f4f6f769e..9a9bfd2027 100644 --- a/hermes_cli/credential_lifecycle.py +++ b/hermes_cli/credential_lifecycle.py @@ -149,6 +149,8 @@ def purge_env_credential_references( Prunes env-seeded pool entries and (optionally) the affected ``provider_models_cache.json`` rows so the model picker stops advertising a provider whose key is gone. + + See #59761. """ pruned = _prune_env_pool_entries(env_var) providers = sorted(set(pruned) | set(_providers_for_env_var(env_var))) @@ -168,6 +170,15 @@ def save_provider_env_credential(env_var: str, value: str) -> Dict[str, Any]: config.yaml mirrors of the PREVIOUS value are updated so a stale higher-precedence copy cannot shadow the rotation, and ``load_pool()`` runs now so the env-seeded ``credential_pool`` entry lands in ``auth.json`` (a ``.env``-only write left env-backed providers 401'ing). + + Suppressed ``env:`` pool sources are re-enabled so a deliberate re-add through the UI behaves like + ``hermes auth add``. See #62269. + The save also forces an immediate ``load_pool()`` for every provider registered against this env var so + the env-seeded ``credential_pool`` entry is materialized to ``auth.json`` right now — the live runtime + reads from the pool, and before #96058 the Desktop "Save" action only touched ``.env`` while + ``auth.json``'s mtime stayed unchanged, so an OpenCode Go (or any other env-backed provider) request + kept 401'ing until the user ran ``hermes auth add --type api-key`` separately. This makes the + Desktop save's effect on disk match what ``hermes auth add`` does. """ from hermes_cli.config import load_env, save_env_value diff --git a/hermes_cli/cron.py b/hermes_cli/cron.py index ad21d2d631..f8b0288de6 100644 --- a/hermes_cli/cron.py +++ b/hermes_cli/cron.py @@ -49,6 +49,10 @@ def _builtin_gateway_liveness() -> Optional[bool]: The builtin ticker only runs inside the gateway process, so a scheduled job with no live gateway can never fire; non-builtin providers fire jobs without the gateway. + + Chronos) fire through their own machinery and are deliberately exempt — a missing gateway process means + nothing for them, so they report active. ``None`` = probe failed; callers must not claim either way. See + #87033. """ try: if _active_cron_provider_name() != "builtin": @@ -72,6 +76,11 @@ def _warn_if_gateway_not_running() -> None: """Warn that scheduled jobs won't fire unless the gateway is running (the #1 cron report). False is the only warn-worthy liveness state (None = unknown). + + The cron ticker only runs inside the gateway (``_start_cron_ticker`` in gateway/run.py); there is no + standalone cron daemon. Without a running gateway, ``next_run_at`` passes but jobs never fire and + ``last_run_at`` stays null — the most common cron support report (#51038). Surfacing this at create/list + time, when the user is right there, prevents it. """ if _builtin_gateway_liveness() is not False: return @@ -101,6 +110,8 @@ def _dispatch_display(dispatch: dict) -> Optional[str]: On-time dispatches render dim; late/catch-up dispatches render loudly so a run fired long after gateway downtime doesn't look like an ordinary success. + + See #99879. """ if not isinstance(dispatch, dict): return None @@ -176,6 +187,9 @@ def _job_rows(job: Dict[str, Any]) -> List[tuple[str, str]]: # `repeat` / `deliver` may be present-but-null (dict-default only covers a missing key). repeat_info = job.get("repeat") or {} repeat_times = repeat_info.get("times") + # `deliver` may be present-but-null in the job record (same pitfall as `repeat` above), so coalesce to + # the default rather than relying on the dict-default, which only applies to a missing key. A null value + # would otherwise reach `", ".join(None)` and crash the whole listing (#32896). deliver = job.get("deliver") or ["local"] skills = job.get("skills") or ([job["skill"]] if job.get("skill") else []) monitor_source = job.get("monitor_script") or job.get("monitor_url") @@ -234,6 +248,8 @@ def cron_tick(): return 1 except OSError as exc: # Real lock-acquisition failures (EMFILE, EACCES) propagate; they are not contention. + # For the one-shot CLI surface, report cleanly instead of dumping a traceback; the gateway ticker + # loop handles its own retry. See #87644. print(color(f"✗ Cron tick failed: {exc}", Colors.RED)) print(" Check `hermes cron status` and the gateway log for details.") return 1 @@ -318,6 +334,7 @@ def _print_ticker_health(pids: list) -> None: The ticker THREAD can die silently or stay alive while every tick fails, so check both the liveness heartbeat and the last-successful-tick marker before saying "will fire". """ + # See #32612, #32895. from cron.jobs import ( get_ticker_heartbeat_age, get_ticker_last_error, get_ticker_success_age, TICKER_INTERVAL_SECONDS) @@ -348,6 +365,9 @@ def _print_ticker_health(pids: list) -> None: last_error = get_ticker_last_error() if last_error: # WHY ticks fail: root-rewritten jobs.json (PermissionError) or fd exhaustion. + # Show WHY ticks fail — e.g. a root-rewritten jobs.json (PermissionError) that silently locked + # out the ticker's uid for ~14h in the field (#68483), or fd exhaustion (EMFILE) that used to + # stall the scheduler invisibly (#87644). print(color(f" Last tick error: {last_error}", Colors.RED)) if "Permission denied" in last_error: print(color(_PERMISSION_HINT, Colors.YELLOW)) @@ -383,6 +403,9 @@ def cron_status(): # The pid scan transiently misses a live gateway right after a restart; the runtime # lock proves the process is alive. Declare "not running" only when both agree. with contextlib.suppress(Exception): + # Same false-alarm class the cronjob tool fixed (#95947): the pid scan can transiently miss + # a live gateway (just after a restart) while the runtime lock — held for exactly the + # gateway's lifetime — proves the ticker's process is alive. from gateway.status import get_running_pid, is_gateway_runtime_lock_active gateway_alive_via_lock = is_gateway_runtime_lock_active() lock_pid = get_running_pid() if gateway_alive_via_lock else None @@ -616,6 +639,10 @@ def _job_action(action: str, job_id: str, success_verb: str) -> int: # stuck 'claimed'. Declaring the channel stateless forces a synchronous run; scoped to # this call so in-process callers (tests, embedding apps) are not tainted. with contextlib.suppress(Exception): + # The background path in ``_try_dispatch_background_run`` triggers when the CLI inherits a + # gateway/desktop session env (HERMES_SESSION_KEY); declare the channel stateless so + # ``async_delivery_supported()`` gates it off and the run executes synchronously to completion + # instead. See #86721. from gateway.session_context import _SESSION_ASYNC_DELIVERY _stateless_token = _SESSION_ASYNC_DELIVERY.set(False) try: diff --git a/hermes_cli/dashboard_auth/__init__.py b/hermes_cli/dashboard_auth/__init__.py index d08372826c..17bd8a075e 100644 --- a/hermes_cli/dashboard_auth/__init__.py +++ b/hermes_cli/dashboard_auth/__init__.py @@ -6,6 +6,9 @@ from hermes_cli.dashboard_auth.base import ( DashboardAuthProvider, Session, TokenPrincipal, LoginStart, InvalidCodeError, InvalidCredentialsError, ProviderError, RefreshExpiredError, assert_protocol_compliance, classify_jwks_lookup_error) +# Dashboard-auth providers are persistent host-owned registrations that deliberately survive a routine +# manager unload (#91701), so the "clean slate" reset must drop the process-global auth registry explicitly +# — otherwise a provider auto-registered during one test leaks into the next. from hermes_cli.dashboard_auth.registry import ( register_provider, get_provider, list_providers, list_token_providers, list_session_providers, clear_providers) diff --git a/hermes_cli/dashboard_auth/base.py b/hermes_cli/dashboard_auth/base.py index 7aee171215..681d555d33 100644 --- a/hermes_cli/dashboard_auth/base.py +++ b/hermes_cli/dashboard_auth/base.py @@ -63,7 +63,14 @@ def classify_jwks_lookup_error(exc: BaseException) -> Exception: :class:`ProviderError` (503, never forces logout). A non-JWT bearer (``DecodeError``), a JWKS with no key for this ``kid`` (``PyJWKSetError``) or any other invalid token is simply not verifiable by this provider -> :class:`InvalidCodeError` (``verify_session`` returns ``None``). - Folding "cannot parse" into "cannot reach" once made every opaque bearer a fast 503.""" + Folding "cannot parse" into "cannot reach" once made every opaque bearer a fast 503. + + * ``jwt.DecodeError`` — the bearer is not a JWT at all (an opaque peer key, a legacy session token, + garbage). #94558: hosted agents answered every non-JWT bearer with a fast 503 ``Auth provider 'nous' + unreachable`` even though Portal was healthy, because "cannot parse" and "cannot reach" were folded into + one branch. * ``jwt.PyJWKSetError`` — the JWKS was fetched fine but holds no key for this token's + ``kid`` (rotated/foreign key). + """ try: import jwt except Exception: # pragma: no cover - jwt is a hard dep of these providers diff --git a/hermes_cli/dashboard_auth/cookies.py b/hermes_cli/dashboard_auth/cookies.py index 7e33273bc5..a315b385e0 100644 --- a/hermes_cli/dashboard_auth/cookies.py +++ b/hermes_cli/dashboard_auth/cookies.py @@ -92,7 +92,12 @@ def set_session_cookies( use_https: bool, prefix: str = "", provider: str = "") -> None: """``access_token_expires_in`` is seconds (the provider's reported TTL). An empty ``refresh_token`` means "don't persist the RT cookie" — a literal empty cookie would be dead - state at best, attack surface at worst.""" + state at best, attack surface at worst. + + Nous Portal issues a 24h rotating refresh token (hermes #37247); a provider that omits it returns + ``Session.refresh_token == ""`` and we simply don't persist the RT cookie — the session then behaves as + access-token-only until the AT expires. No other branch changes between the two cases. + """ _set(response, SESSION_AT_COOKIE, access_token, max_age=access_token_expires_in, use_https=use_https, prefix=prefix) if refresh_token: @@ -174,7 +179,14 @@ def parse_pkce_payload(raw: str) -> dict[str, str]: CSRF state, broker binding). Compatibility ladder for cookies minted by an older server mid-upgrade: 1. base64url(JSON); 2. flat form with raw ``;`` delimiters, split WITHOUT unquoting (the ``next`` segment carries its own URL-encoding); 3. URL-encoded flat form, - unquote once then split. A NEW cookie hitting an OLD server fails the state check.""" + unquote once then split. A NEW cookie hitting an OLD server fails the state check. + + 1. **base64url(JSON)** (current): the wire value is pure urlsafe base64 that decodes to a JSON object. + Legacy forms can never match — they always contain ``%`` (URL-encoded, #99176) or a raw ``;`` (oldest + flat form), both outside the base64url alphabet. Split as-is WITHOUT unquoting the payload — the + ``next`` segment carries its own single URL-encoding, and unquoting here would turn a ``%3B`` inside it + into a bogus delimiter and truncate the post-login target. Neither newer format can contain a raw ``;``. + """ if _B64URL_RE.match(raw): try: padded = raw + "=" * (-len(raw) % 4) diff --git a/hermes_cli/dashboard_auth/registry.py b/hermes_cli/dashboard_auth/registry.py index 43bbcbc1ac..2a84f33c9f 100644 --- a/hermes_cli/dashboard_auth/registry.py +++ b/hermes_cli/dashboard_auth/registry.py @@ -98,7 +98,10 @@ def register_global_provider(provider: DashboardAuthProvider) -> None: across every profile one dashboard process serves, so these outlive any per-home plugin manager: always targets ``_providers`` (never a per-home overlay) and *replaces* a same-name entry instead of raising, so a forced plugin re-discovery (e.g. after a password change) - rotates the provider in place. Pairs with ``unregister_global_provider``.""" + rotates the provider in place. Pairs with ``unregister_global_provider``. + + Pairs with ``unregister_global_provider`` for teardown of the exact object still current (#91701). + """ assert_protocol_compliance(type(provider)) with _lock: _providers[provider.name] = provider diff --git a/hermes_cli/dashboard_procs.py b/hermes_cli/dashboard_procs.py index 8c22a1e6a9..51e0f35d48 100644 --- a/hermes_cli/dashboard_procs.py +++ b/hermes_cli/dashboard_procs.py @@ -43,6 +43,15 @@ def _iter_process_table() -> list[tuple[int, str]]: # errors="ignore": wmic may emit the system code page. bounded_probe_run, not run(): # run()'s post-timeout cleanup joins pipe readers unbounded and a conhost descendant # holding duplicated handles wedges it forever. + # In text mode, subprocess output decoding depends on Python's configuration (locale-dependent by + # default, or UTF-8 in UTF-8 mode). The important protection here is errors="ignore": it prevents a + # reader-thread UnicodeDecodeError from leaving result.stdout=None and turning the later .split() + # into an AttributeError (#17049). bounded_probe_run (rather than subprocess.run with a timeout) + # keeps a slow scan from wedging the caller forever: run()'s post-timeout cleanup joins the pipe + # reader threads unbounded, and a conhost.exe descendant holding duplicated pipe handles blocks that + # join indefinitely (#87134). It also passes CREATE_NO_WINDOW: this scan can run from the windowless + # pythonw.exe desktop/gateway backend during an update, where a bare wmic spawn would pop a console + # window. from hermes_cli._subprocess_compat import bounded_probe_run result = bounded_probe_run( ["wmic", "process", "get", "ProcessId,CommandLine", "/FORMAT:LIST"], @@ -73,6 +82,12 @@ def _scan_dashboard_processes(*, exclude_pids: set[int] | None = None) -> list[t A forgotten dashboard keeps the old Python backend against the new JS bundle after ``hermes update`` (every API call 401s). *exclude_pids* (Desktop's HERMES_DESKTOP_CHILD_PID backends) are never returned. + + *exclude_pids* is an optional set of PIDs that must never be returned. This is used by the Hermes + Desktop Electron app to protect its own backend child process: when the desktop spawns ``hermes serve`` + as a backend and triggers an auto-update, the update must not kill the backend that the desktop itself + manages. The desktop sets the environment variable ``HERMES_DESKTOP_CHILD_PID`` on the spawned backend + process; ``_kill_stale_dashboard_processes`` reads it and passes it here. (#37532) """ skip = {os.getpid(), *(exclude_pids or ())} try: @@ -83,6 +98,10 @@ def _scan_dashboard_processes(*, exclude_pids: set[int] | None = None) -> list[t # Spawn-ledger augmentation: substring patterns miss profiled launches (`hermes --profile p # serve`); the ledger holds live-verified pids. Unavailable ledger → scan-only. with contextlib.suppress(Exception): + # Every serve/ dashboard registers itself in the machine spawn ledger at startup with live-verified + # (pid, create_time), so ledger rows are positive identity, not argv guessing. Add any live ledger + # serve/dashboard the scan missed; prefer the ledger's recorded argv (full launch args) over the + # scan's truncated view. See #81564. from hermes_cli.process_identity import ledger_entries seen = {pid for pid, _ in found} | skip for entry in ledger_entries(): @@ -125,7 +144,10 @@ def _profile_flag_value(argv: list[str]) -> str | None: def _is_ephemeral_port_zero_backend(argv: list[str]) -> bool: """True for Desktop-style ``serve|dashboard --port 0`` backends — replaying them after - ``hermes update`` multiplies listening backends because ``--port 0`` binds a fresh port.""" + ``hermes update`` multiplies listening backends because ``--port 0`` binds a fresh port. + + See #78821. + """ if _dashboard_subcommand_index(argv) is None: return False return any((tok == "--port" and i + 1 < len(argv) and str(argv[i + 1]) == "0") @@ -160,7 +182,10 @@ def _resolved_home(home: str) -> Path: def _normalized_home_for_compare(home: str) -> str: - """Install-identity key for *home*: symlinked / differently-spelled roots compare equal.""" + """Install-identity key for *home*: symlinked / differently-spelled roots compare equal. + + See #94030. + """ return os.path.normcase(str(_resolved_home(home))) @@ -169,6 +194,8 @@ def _profile_key_for_respawn(argv: list[str], hermes_home: str | None = None) -> A home ending in ``profiles/`` → ``profile:`` (shares a cap with an explicit ``--profile``); other homes keep a ``home:`` key so unrelated installs never collapse. + + See #78821. """ if hermes_home: parts = _resolved_home(hermes_home).parts @@ -188,6 +215,14 @@ def _filter_dashboard_respawn_candidates( and steal the foreign install's fixed port → EADDRINUSE crash-loop; unreadable ``None`` stays eligible); dedupe by normalized cmdline; one backend per profile / home. PPID-1 is NOT skipped: a prior respawn detaches, so fixed-port manual backends sit under init. + + 1. Never resurrect Desktop ephemeral ``serve|dashboard --port 0`` backends — Desktop + (``HERMES_DESKTOP_CHILD_PID``) owns their lifecycle. These are also the PPID-1 orphans that previously + multiplied across updates because ``--port 0`` always binds a fresh free port. 2. A foreign install's + backend is owned by that install's supervisor/user. 3. 4. See #78821, #94030. + Intentionally does **not** blanket-skip every PPID-1 process: a prior ``hermes update`` respawn detaches + with ``start_new_session=True``, so fixed-port manual backends are reparented to init and must still be + eligible for the next update's #40449 restart. """ if own_home is None: try: @@ -289,6 +324,13 @@ def _kill_stale_dashboard_processes( kill (systemd treats our SIGTERM as a clean stop, so ``Restart=on-failure`` never fires) and manual PIDs are respawned from captured argv. PIDs owned by *already_restarted_units* (no ``.service`` suffix) are left untouched, not killed twice. + + Manually-started dashboards are not auto-restarted because we don't know the original launch args + (--host, --port, --insecure, --tui, --no-open). See #68934. + *already_restarted_units* names units (no ``.service`` suffix) the caller already restarted directly — + e.g. ``hermes update``'s systemd fleet-restart loop, which restarts ``hermes-serve*`` units before this + function runs. Without excluding them, a Serve-only install's freshly restarted process is found again + here and restarted a second time for no benefit (review on #83595). """ if restart_managed and _m()._restart_managed_dashboard_service(reason): # The dashboard unit is handled but other backends (e.g. hermes-serve.service) are not: @@ -315,6 +357,9 @@ def _kill_stale_dashboard_processes( pid_service[pid] = _m()._get_systemd_service_for_pid(pid) if not pid_service[pid] and (cmdline := _m()._dashboard_cmdline_for_pid(pid)): # Manual process: exact argv + HERMES_HOME for the respawn and its profile cap. + # Manually-started process: preserve its exact argv so we can respawn it after the update + # (#40449, #68934). Snapshot HERMES_HOME before the kill so per-profile caps still work + # after the process is gone (#78821). pid_cmdline[pid] = cmdline pid_home[pid] = _hermes_home_for_pid(pid) if already_restarted_units: @@ -345,6 +390,10 @@ def _restart_killed_backends( pid_cmdline: dict[int, list[str]], pid_home: dict[int, str | None]) -> list[int]: """Update path: restart systemd units, respawn manual argv (detached, headless, logged to logs/dashboard-restart.log; one per profile, no ``--port 0``). Returns PIDs not brought back.""" + # Two categories: Without this, a remote backend (hermes serve) under Restart=on-failure never comes + # back after our clean SIGTERM, and the Desktop can't reconnect (#68934). Filtered so Desktop + # ``serve|dashboard --port 0`` backends are not resurrected and duplicates collapse to one per profile + # (#78821). unrecovered: list[int] = [] failed_restarts: list[tuple[str, str]] = [] seen_services: set[str] = set() diff --git a/hermes_cli/debug.py b/hermes_cli/debug.py index 43ff19d3fe..c5bb966b3b 100644 --- a/hermes_cli/debug.py +++ b/hermes_cli/debug.py @@ -354,6 +354,7 @@ def collect_debug_report( # In-process sanitiser heal counters: populated only inside a process that ran agent turns # (gateway /debug share); a fresh CLI's errors.log tail carries the same escalation lines. with contextlib.suppress(Exception): + # See #96870. from agent.agent_runtime_helpers import get_sanitizer_heal_stats heal_stats = get_sanitizer_heal_stats() if heal_stats: diff --git a/hermes_cli/default_soul.py b/hermes_cli/default_soul.py index 0f4c13a217..3c99d404dd 100644 --- a/hermes_cli/default_soul.py +++ b/hermes_cli/default_soul.py @@ -4,6 +4,8 @@ # seeds this into SOUL.md on first run, so it is the text virtually every real user gets. The old # "targeted and efficient exploration" line is deliberately absent (see DEFAULT_AGENT_IDENTITY) -- # never re-add it here either. +# DEFAULT_AGENT_IDENTITY only serves sessions with no SOUL.md at all (e.g. skip_context_files), which is not +# the common case. See #95681. DEFAULT_SOUL_MD = ( "You are Hermes Agent, built by Nous Research. Be direct: match the length of your reply to the weight of " "the ask — a one-line question gets a one-line answer, and finished work gets a short report of what " @@ -58,6 +60,13 @@ def _normalize_soul(text: str) -> str: def is_legacy_template_soul(text: str) -> bool: - """True if ``text`` is a non-customized, auto-seeded SOUL.md (see ``_LEGACY_TEMPLATE_SOULS``).""" + """True if ``text`` is a non-customized, auto-seeded SOUL.md (see ``_LEGACY_TEMPLATE_SOULS``). + + Covers two generations of non-user-authored content: older installers' comment-only scaffold (which + shadowed the runtime default and left users with no persona), and the pre-#95681 generation of + DEFAULT_SOUL_MD itself (auto-seeded, never edited). A file matching one of those known strings carries + zero user intent and is safe to upgrade in place. Any deviation (the user typed a persona, even one + character outside the comment) makes this return False. + """ normalized = _normalize_soul(text) return any(normalized == _normalize_soul(t) for t in _LEGACY_TEMPLATE_SOULS) diff --git a/hermes_cli/dep_ensure.py b/hermes_cli/dep_ensure.py index 3464f4f668..f2b7a4cf3a 100644 --- a/hermes_cli/dep_ensure.py +++ b/hermes_cli/dep_ensure.py @@ -37,7 +37,10 @@ def _has_system_browser() -> bool: def _has_npx_agent_browser() -> bool: """agent-browser resolves lazily via npx on the default install, invisible to the PATH/managed-dir probes above. Mirror tools.browser_tool.check_browser_requirements's Termux carve-out so this - check can't diverge from what browser tools actually find.""" + check can't diverge from what browser tools actually find. + + See #43564. + """ try: from tools.browser_tool import _find_agent_browser, _is_npx_agent_browser_sentinel, _requires_real_termux_browser_install browser_cmd = _find_agent_browser(validate=False) diff --git a/hermes_cli/doctor_config.py b/hermes_cli/doctor_config.py index 5fad927840..53ca085ecc 100644 --- a/hermes_cli/doctor_config.py +++ b/hermes_cli/doctor_config.py @@ -211,6 +211,7 @@ def _provider_has_credentials(runtime_provider: str) -> bool: def _validate_model_config(config_path, issues: list) -> None: """Validate model.provider / model.default against the provider registry (raw file).""" + # Detect stale root-level model keys (known bug source — PR #4329) from hermes_cli.config import read_user_config_raw cfg = read_user_config_raw(config_path) model_section = cfg.get("model") or {} @@ -334,6 +335,12 @@ def _drift_max_iterations_ghost(f: Finding, should_fix: bool, config_path) -> No .env FILE (load_env), not get_env_value/os.environ, which the bridge may have overridden already. """ from hermes_cli.doctor import _DHH + # Detect stale HERMES_MAX_ITERATIONS ghost in .env shadowing agent.max_turns in config.yaml (issue + # #17534). The setup wizard used to dual-write the iteration budget to both stores; users who later edit + # only config.yaml are left with a .env ghost. The gateway bridge normally derives HERMES_MAX_ITERATIONS + # from agent.max_turns at startup, but if that bridge bails (any earlier config-parse error), the stale + # .env value silently wins and the agent runs at the wrong budget — e.g. config says 400 but the + # activity line reads N/90. from hermes_cli.config import load_env, read_user_config_raw, remove_env_value raw_config = read_user_config_raw(config_path) agent_cfg = raw_config.get("agent") diff --git a/hermes_cli/doctor_platform.py b/hermes_cli/doctor_platform.py index 2318114fa2..59892cdc6c 100644 --- a/hermes_cli/doctor_platform.py +++ b/hermes_cli/doctor_platform.py @@ -258,6 +258,8 @@ def check_macos_tcc_grants() -> None: rebuild, so grants silently stop matching while the Settings toggle stays ON; identifier-pinned builds survive rebuilds, but grants made to older binaries stay stale until re-granted once. TCC.db needs Full Disk Access, so the DR string is the only readable signal (a proxy for the signing class, not DR wording). + + See #86385. """ from hermes_cli.doctor import _desktop_app_bundle, _macos_desktop_dr app = _desktop_app_bundle() if sys.platform == "darwin" else None @@ -298,7 +300,10 @@ def _macos_desktop_dr(app: Path) -> str | None: def check_macos_tcc_anchor(should_fix: bool = False) -> None: """Report (and with --fix install) the dylib-complete TCC anchor; silent on non-macOS / non-uv interpreters. - Never raises. Install is gated by the module's pre-install boot probe, so ``--fix`` cannot brick the CLI.""" + Never raises. Install is gated by the module's pre-install boot probe, so ``--fix`` cannot brick the CLI. + + See #95596. + """ with warn_on_error("macOS TCC anchor check failed"): from hermes_cli import macos_tcc_anchor as tcc status, detail = tcc.tcc_anchor_state() @@ -317,6 +322,12 @@ def check_macos_full_disk_access() -> None: Probe: listdir of ``~/Library/Application Support/com.apple.TCC`` — FDA-gated, and probing it does NOT trigger a prompt (the TCC dir just returns EPERM). A missing dir / other error is indeterminate: stay silent. + + macOS TCC prompts per-category (Desktop, then Downloads, then Documents, ...), so first-run agents + drip-feed permission dialogs as they touch each folder. ONE Full Disk Access grant covers all of them, + permanently — and with the stable signing identities now in place (#73681/#95091/#95131), it survives + updates too. This check probes whether the terminal context already has FDA and, when it doesn't, prints + the exact one-switch setup with the System Settings deep link. """ if sys.platform != "darwin": return @@ -377,9 +388,16 @@ def _check_python_environment(should_fix: bool, f: Finding) -> None: check_info(f"SQLite source id: {(src[:48] + '…') if len(src) > 48 else src}") _report_database_journal_modes() check_bool(sys.prefix != sys.base_prefix, "Virtual environment active", ("Not in virtual environment", "(recommended)")) + # macOS TCC interpreter anchor (#95596): dylib-complete re-land of the mechanism reverted in #95563. + # Silent on non-macOS. check_macos_tcc_anchor(should_fix=should_fix) + # macOS Full Disk Access (issue #52010 follow-up): one grant silences every per-folder prompt + # permanently. Silent on non-macOS. check_macos_full_disk_access() _check_version_consistency(f.issues) + # macOS TCC grant persistence (issue #86385): a locally-built desktop bundle whose DR is cdhash-pinned + # loses every permission grant on each rebuild; a post-#73681 identifier-pinned DR survives, but grants + # made to older binaries stay stale (toggle shows ON while macOS re-prompts). check_macos_tcc_grants() diff --git a/hermes_cli/doctor_state.py b/hermes_cli/doctor_state.py index 3550cb07d4..200ad5e0f8 100644 --- a/hermes_cli/doctor_state.py +++ b/hermes_cli/doctor_state.py @@ -193,6 +193,8 @@ def _state_db_health(f: Finding, should_fix: bool, state_db_path: Path, _DHH: st # COUNT(*) succeeds even when the FTS index is corrupt and every write fails through the triggers; # _db_opens_cleanly drives a rolled-back write to surface that. from hermes_state import _db_opens_cleanly + # `_db_opens_cleanly` now drives a rolled-back write so this otherwise-silent corruption class is + # surfaced (and repaired in place with --fix). See #50502. _write_reason = _db_opens_cleanly(state_db_path) if _write_reason is not None: check_warn(f"{_DHH}/state.db fails a write-health probe (FTS index may be corrupt)", f"({_write_reason})") diff --git a/hermes_cli/doctor_tools.py b/hermes_cli/doctor_tools.py index fc0c59b0cd..4a9c250aeb 100644 --- a/hermes_cli/doctor_tools.py +++ b/hermes_cli/doctor_tools.py @@ -63,6 +63,8 @@ def _doctor_web_capability_rows() -> list[tuple[str, str, str]]: Uses the same active-provider resolvers as the tools but reports ``is_available()`` readiness, so an explicitly selected but unconfigured backend does not look healthy. + + See #78412. """ rows: list[tuple[str, str, str]] = [] try: @@ -250,6 +252,8 @@ def _check_agent_browser(should_fix: bool) -> bool: install) so doctor can't diverge from the tools; validate=False keeps it a cheap, side-effect-free check. """ try: + # agent-browser is no longer a root package.json dependency (#43564) — it resolves lazily via npx + # (or a global/Hermes-managed install) at first use. from tools.browser_tool import _find_agent_browser, _is_npx_agent_browser_sentinel resolved = _find_agent_browser(validate=False) except Exception: @@ -390,6 +394,12 @@ def _check_npm_audit(should_fix: bool, f: Finding) -> None: npm_bin = _safe_which("npm") if npm_bin: try: + # Each entry: (cwd, label, extra_audit_args) PROJECT_ROOT is audited with --workspaces=false so + # that the apps/* glob (which pulls in Electron, node-pty, etc.) is never resolved for a routine + # security check. The web and ui-tui workspaces are audited separately via --workspace flags. + # See #38772. The WhatsApp bridge may live under a writable HERMES_HOME mirror instead of the + # (possibly read-only) install tree in Docker — resolve it through the shared helper so we audit + # the dir that actually holds node_modules. See #49561. from gateway.platforms.whatsapp_common import resolve_whatsapp_bridge_dir whatsapp_bridge_dir = resolve_whatsapp_bridge_dir() except Exception: @@ -418,6 +428,7 @@ def _check_tool_availability(should_fix: bool, f: Finding) -> None: # Web is split into search/extract readiness rows so an explicitly # selected but unconfigured backend cannot look healthy. web_rows = [] + # See #78412. if "web" in available or any(item.get("name") == "web" for item in unavailable): web_rows = _doctor_web_capability_rows() if web_rows: diff --git a/hermes_cli/dump.py b/hermes_cli/dump.py index bc4b058bfd..fa9e5b6053 100644 --- a/hermes_cli/dump.py +++ b/hermes_cli/dump.py @@ -15,7 +15,13 @@ from agent.skill_utils import is_excluded_skill_path def _dotenv_key_names() -> set[str]: """Env-var names assigned a non-empty value in ~/.hermes/.env — what the managed backends (launchd / - systemd / desktop ``serve``) load, as opposed to the shell exports ``os.getenv`` reflects here.""" + systemd / desktop ``serve``) load, as opposed to the shell exports ``os.getenv`` reflects here. + + ``hermes debug share`` runs in a terminal, so ``os.getenv`` reflects the shell's environment, which can + include exported keys the managed backend never sees. Comparing against this set lets the dump flag that + mismatch (the exact trap behind #48504-style "no web_search" reports: key exported in the shell, absent + from .env, invisible to the launchd backend). + """ try: text = get_env_path().read_text(encoding="utf-8", errors="ignore") except (OSError, UnicodeError): @@ -200,6 +206,9 @@ def _api_key_lines(show_keys: bool) -> list[str]: if val and env_var not in dotenv_keys: display += " (shell only — not in .env; managed/desktop backend may not see it)" # `hermes auth add openrouter` credentials live in the pool, not env — don't read "not set". + # A credential added via `hermes auth add openrouter` lives in the credential pool, not as an env + # var — surface it so the dump doesn't misleadingly read "not set" while `hermes auth list` shows it + # (#42130). if not val and label == "openrouter": try: from agent.credential_pool import load_pool as _load_pool diff --git a/hermes_cli/env_loader.py b/hermes_cli/env_loader.py index a7dcb86ea5..66441058b9 100644 --- a/hermes_cli/env_loader.py +++ b/hermes_cli/env_loader.py @@ -69,7 +69,11 @@ def _env_keys_defined_in_dotenv(path: Path) -> set[str]: def _clear_known_keys_missing_from_dotenv(path: Path) -> None: """After ``.env`` loaded with override, delete inherited ``_PROFILE_MANAGED_ENV_KEYS`` it does not - define. Deliberately NARROW: only keys that change *which provider path* is used.""" + define. Deliberately NARROW: only keys that change *which provider path* is used. + + Does **not** run when the ``.env`` file does not exist (bare-profile case, which follows ``#66930`` / + ``#67027`` semantics). + """ if not path.exists(): return defined = _env_keys_defined_in_dotenv(path) @@ -119,6 +123,8 @@ def _hydrate_profile_secret_sources(home: Path) -> dict[str, str]: local_env.update(load_env_file(home / ".env")) # Mirror load_hermes_dotenv()'s .op.env bootstrap (1Password token lives in gitignored .op.env) # or cold profiles fail 1Password hydration. .env wins. + # Without seeding it here a cold profile configured for the supported .op.env flow fails 1Password + # hydration (sweeper review on #74549). .env values win — never override an existing key. op_env = home / ".op.env" if op_env.exists(): for _name, _value in load_env_file(op_env).items(): @@ -186,7 +192,12 @@ def _format_offending_chars(value: str, limit: int = 3) -> str: def _sanitize_loaded_credentials() -> None: - """Strip non-ASCII from credential env vars (``_CREDENTIAL_SUFFIXES``) so the codebase never sees them.""" + """Strip non-ASCII from credential env vars (``_CREDENTIAL_SUFFIXES``) so the codebase never sees them. + + Emits a one-line warning to stderr when characters are stripped. Silent stripping would mask copy-paste + corruption (Unicode lookalike glyphs from PDFs / rich-text editors, ZWSP from web pages) as opaque + provider-side "invalid API key" errors (see #6843). + """ for key, value in list(os.environ.items()): if not any(key.endswith(suffix) for suffix in _CREDENTIAL_SUFFIXES): continue @@ -245,6 +256,8 @@ def _sanitize_env_file_if_needed(path: Path) -> None: # ORDER MATTERS: BOM_UTF32_LE (FF FE 00 00) startswith BOM_UTF16_LE (FF FE); UTF-16 first would mangle it. force_utf8_rewrite = False if raw.startswith(codecs.BOM_UTF32_LE) or raw.startswith(codecs.BOM_UTF32_BE): + # Lazy import keeps the module import block identical to #65124's codecs/io additions so the two PRs + # auto-merge either order. path_key = str(path.resolve()) if path_key not in _WARNED_UTF32_PATHS: _WARNED_UTF32_PATHS.add(path_key) @@ -355,6 +368,13 @@ def load_hermes_dotenv( # self-lock preflight exit 2 again. from hermes_cli import _early_recovery + # External secret sources are skipped in two updater situations: 1. ``load_external_secrets=False`` — + # the caller is an ``update`` invocation that must not import optional secret-manager libraries + # (Bitwarden → cryptography → ``_rust.pyd``) into the process that replaces that same environment on + # Windows (#73381, #86735). 2. A fresh ``hermes update`` retry just completed a deferred dependency + # install before importing this module. Do not remap native secret-source dependencies in that same + # updater process or the self-lock preflight will recreate the marker and exit 2 again. Dotenv and + # managed env still load in both cases; only external source resolution is unnecessary for the updater. if load_external_secrets and not _early_recovery._should_skip_external_secret_sources(): _apply_external_secret_sources(home_path) _apply_managed_env() @@ -362,6 +382,12 @@ def load_hermes_dotenv( # config.yaml owns terminal.*, but the override=True loads above let a stale TERMINAL_ENV=docker in # ~/.hermes/.env win on every reload and flip the backend mid-session in long-lived processes. # Re-apply the explicit terminal keys LAST, after the managed overlay, so the merged config lands. + # config.yaml is the documented source of truth for terminal.* settings, but the dotenv loads above run + # with override=True — so a stale TERMINAL_ENV=docker left in ~/.hermes/.env (e.g. written by an older + # `hermes setup` before the user switched terminal.backend in config.yaml) silently wins again on every + # reload. Startup launchers bridge config→env once, but long-lived processes (gateway per-turn reload, + # cron standalone runs) call load_hermes_dotenv() repeatedly and used to flip the effective backend back + # to the stale .env value mid-session (#29186, #67323). _reapply_terminal_config_bridge(home_path) return loaded @@ -414,6 +440,7 @@ def _apply_external_secret_sources(home_path: Path) -> None: try: cfg = _load_secrets_config(home_path) except Exception: # noqa: BLE001 — config errors must not block startup + # See #40597. return if not cfg: return @@ -443,8 +470,14 @@ def _apply_external_secret_sources(home_path: Path) -> None: # Marking AFTER the attempt keeps the earlier failure paths retryable. _APPLIED_HOMES.add(home_key) + # A real fetch attempt happened (success OR error). Mark the home now so the 3-5 import-time + # load_hermes_dotenv() calls per startup don't re-fetch / re-print — error retries within one process + # are opt-in via reset_secret_source_cache(). Marking AFTER the attempt (not before, see #40597) is what + # lets the earlier failure paths stay retryable. if report.applied_any: _sanitize_loaded_credentials() # vault values carry the same copy-paste corruption risk as .env + # Re-run the ASCII sanitization pass: vault values are user-supplied and might have the same + # copy-paste corruption as a manually edited .env (see #6843). values: dict[str, str] = {} for name, applied in report.provenance.items(): _SECRET_SOURCES[name] = applied.source diff --git a/hermes_cli/fallback_cmd.py b/hermes_cli/fallback_cmd.py index b006226e63..79a3d3d927 100644 --- a/hermes_cli/fallback_cmd.py +++ b/hermes_cli/fallback_cmd.py @@ -147,6 +147,8 @@ def cmd_fallback_add(args) -> None: # Same deployment as the primary → nothing to add. Identity semantics are owned by # agent.backend_identity: same provider+model on a DIFFERENT explicit base_url is a different # backend (multi-endpoint pool) and a legitimate fallback. + # Picker picked the same thing that's already the primary → nothing changed, and there's nothing useful + # to add as a fallback to itself. See #54250, #57584, #62984. from agent.backend_identity import same_deployment new_ident = _identity(new_entry) primary_entry = _extract_fallback_from_model_cfg(model_before) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 90728a21c5..02a98184ab 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -122,6 +122,14 @@ def _get_service_pids(all_profiles: bool = False) -> set: Relies on the service manager committing the new PID before the restart command returns. ``all_profiles`` widens the current profile's unit/label to the whole ``hermes-gateway*`` / ``ai.hermes.gateway*`` fleet so update/reaper never kill a sibling's service gateway as "manual". + + ``all_profiles`` widens the launchd branch to every installed ``ai.hermes.gateway*`` LaunchAgent — the + update path needs the whole fleet excluded from its sweep (#41403, #73626): sibling-profile launchd + gateways found by the (BSD-fixed) ps scan must not be misclassified as manual processes and killed. + Default-scope callers (``gateway status``, cron checks) keep seeing only the current profile's service; + the orphan reaper passes all_profiles=True for the same friendly-fire reason. The systemd branch mirrors + this: default scope filters to the current profile's exact unit name; ``all_profiles=True`` widens to + the ``hermes-gateway*`` fleet glob. """ pids: set = set() @@ -130,6 +138,11 @@ def _get_service_pids(all_profiles: bool = False) -> set: pattern = "hermes-gateway*" if all_profiles else get_service_name() for scope_args in [["systemctl", "--user"], ["systemctl"]]: try: + # Belt-and-suspenders for the EXCLUDE use case (#74075): a bare ``launchctl list`` prefix + # scan also catches ai.hermes.gateway* agents the label derivation can't map (renamed + # profiles, other installs sharing this user). Over-inclusion is safe here — these PIDs are + # only ever protected from the kill sweep, never targeted. Restart paths use the + # label-derived set only. result = subprocess.run( scope_args + ["list-units", pattern, "--plain", "--no-legend", "--no-pager"], @@ -160,6 +173,10 @@ def _get_service_pids(all_profiles: bool = False) -> set: labels = {get_launchd_label()} if all_profiles: # Whole fleet, mirroring the systemd ``hermes-gateway*`` glob above. + # Every gateway LaunchAgent, not just the invoking profile's — mirrors the systemd branch's + # ``hermes-gateway*`` pattern above. The update path restarts the whole fleet, and its + # stale-process sweep must not mistake a sibling service's fresh PID for a manual gateway it + # should kill (#41403). labels.update(launchd_gateway_labels_for_install()) for label in sorted(labels): try: @@ -291,6 +308,34 @@ def _wait_for_pid_exit(pid: int, timeout: float) -> bool: # treated as alive; never escalate on ambiguity. Legacy payloads (no ``loop_tick_socket`` flag) # wrote on-loop, so staleness alone remains proof. +# --- Wedged-gateway detection + bounded escalation (#81642) ----------------- A gateway whose asyncio loop +# is stalled (e.g. an in-loop compression pass, #72707) cannot process SIGTERM/SIGUSR1 shutdown: the drain +# wait then burns the full drain budget (180s by default), warns "still running after 180.0s — restart may +# fail", and `hermes update` can deadlock behind it. The loop publishes a liveness signal precisely for this +# case: an asyncio task rewrites ``state/gateway.heartbeat`` every 30s (#66892), so a frozen loop stops +# refreshing the file while a busy-but-alive loop keeps refreshing it. Since #90502 the heartbeat write runs +# on a thread (a stalling filesystem must not be able to block the loop the watchdog watches), which costs +# the file its status as *proof*: a stalled write or a saturated executor can age the file while the loop +# runs, and an off-loop write can land after the loop froze, keeping the file fresh for a dead loop. The +# loop therefore also arms a second witness — ``state/gateway.loop-tick..sock``, a UNIX socket answered +# by the loop itself — and records whether it is armed in the heartbeat payload (``loop_tick_socket``). +# ``probe_gateway_loop_liveness`` reads both signals (a local stat + JSON read + a bounded socket ping, +# repeated up to ``tick_strikes`` times when a wedge is suspected — worst case ~3.4s, still far inside the +# 10s query tier of the subprocess timeout doc) and classifies the gateway BEFORE any drain wait begins: - +# ``alive`` — the loop answered the tick socket, or the file is fresh and the loop is not contradicted by +# the socket. Callers must take the normal graceful-drain path, which honours the in-flight cron drain floor +# (#86684). - ``wedged`` — the heartbeat belongs to this PID, is stale well past several missed beats, AND +# the tick socket is armed but stays silent across a sustained window of consecutive misses (default 3): +# both witnesses agree, sustained, that the loop is provably dead. One silent probe is never destructive +# authority — a transient synchronous stall can outlast a single recv timeout, so a lone miss falls to +# ``unknown``. Draining is pointless for a provably dead loop (nothing can run the drain), so callers may +# escalate immediately via ``_escalate_wedged_gateway``. - ``unknown`` — no heartbeat / unreadable / PID +# mismatch / witness conflict (fresh file with a silent loop, armed socket unreachable). Treated like +# ``alive``: never escalate on ambiguity. The distinction matters: only a *provably dead* loop may bypass +# the cron drain floor. A merely busy gateway still answers the probe (socket ping) and keeps its full drain +# budget — even when the filesystem is stalling the heartbeat write (the incident that motivated #90502). +# Legacy gateways (no ``loop_tick_socket`` flag in the payload) wrote the file on-loop, so their staleness +# remains proof and the old single-witness contract is unchanged. GATEWAY_LOOP_ALIVE = "alive" GATEWAY_LOOP_WEDGED = "wedged" GATEWAY_LOOP_UNKNOWN = "unknown" @@ -348,7 +393,15 @@ def _probe_loop_tick_socket_sustained( ) -> bool | None: """Probe the tick socket up to ``strikes`` times, ``gap_s`` apart: True once answered, False if a node stayed silent the whole window, None if the node vanished (not evidence). One silent probe is not - destructive evidence — a transient synchronous stall can outlast one recv timeout.""" + destructive evidence — a transient synchronous stall can outlast one recv timeout. + + A single silent probe is NOT destructive evidence: the loop may be in a short transient synchronous + stall (a reconnect storm, a heavy synchronous callback, scheduler delay) that outlasts one recv timeout. + Killing a gateway on that would be a false wedge — the exact class of false positive #90502 exists to + prevent. Destructive authority therefore requires the loop to fail to answer across a bounded window of + ``strikes`` consecutive misses, ``gap_s`` apart; any answer inside the window proves the loop is + dispatching and returns ``True``. + """ total = max(int(strikes), 0) for attempt in range(total): if tcp_port is not None: @@ -371,7 +424,15 @@ def probe_gateway_loop_liveness( ) -> str: """Classify a gateway PID's event loop as alive / wedged / unknown (see block comment above). Stale heartbeat is ``wedged`` only when the payload declares the tick socket armed AND it stays - silent across ``tick_strikes`` misses; any answer is ``alive``; ambiguity is ``unknown``.""" + silent across ``tick_strikes`` misses; any answer is ``alive``; ambiguity is ``unknown``. + + - the loop-tick socket (``state/gateway.loop-tick..sock``): answered by the gateway loop itself, so + a reply is direct proof that the loop is dispatching. It is never refreshed by the heartbeat executor + thread and never stalled by a filesystem that is slow to fsync. - the heartbeat file + (``state/gateway.heartbeat``): rewritten every 30s on a thread since #90502, so freshness alone is no + longer proof of loop schedulability — a stalled write (measured at 112.6s max on the incident box) or a + saturated executor can age the file while the loop runs, and a write can land after the loop froze. + """ try: stale_budget = max(float(stale_after), 0.0) except (TypeError, ValueError): @@ -404,6 +465,7 @@ def probe_gateway_loop_liveness( if witness is True: # Loop answered: a stale file is a stalled write, not a wedge. return GATEWAY_LOOP_ALIVE + # The loop answered a ping — it is dispatching right now. See #90502. age = time.time() - mtime if age <= stale_budget: if witness is False: @@ -420,6 +482,9 @@ def probe_gateway_loop_liveness( return GATEWAY_LOOP_UNKNOWN if witness is False: # First miss. The probe above is miss #1, so ``tick_strikes - 1`` more attempts follow. + # One silent probe is NOT destructive authority: a short transient synchronous stall can outlast a + # single recv timeout, and killing a live gateway on it would be the exact false wedge #90502 exists + # to prevent. sustained = _probe_loop_tick_socket_sustained( pid, home, timeout=tick_timeout, strikes=tick_strikes - 1, gap_s=tick_gap_s, tcp_port=tcp_port_int ) @@ -434,7 +499,10 @@ def probe_gateway_loop_liveness( def _escalate_wedged_gateway(pid: int, *, term_grace: float = 5.0, kill_wait: float = 5.0) -> bool: """Bounded stop (SIGTERM, ``term_grace``, SIGKILL, ``kill_wait``) for a provably dead loop; True once gone. Callers MUST have classified ``GATEWAY_LOOP_WEDGED`` first: escalating a merely busy gateway - bypasses the cron drain floor and SIGKILLs live work.""" + bypasses the cron drain floor and SIGKILLs live work. + + See #86684. + """ from gateway.status import get_process_start_time expected_start_time = get_process_start_time(pid) try: @@ -452,7 +520,12 @@ def _escalate_wedged_gateway(pid: int, *, term_grace: float = 5.0, kill_wait: fl def _get_ancestor_pids() -> set[int]: - """PIDs of this process and its ancestors, so scans never count the invoking ``hermes`` CLI as a gateway.""" + """PIDs of this process and its ancestors, so scans never count the invoking ``hermes`` CLI as a gateway. + + Walks from the current PID up to PID 1 (init) so that process-table scans never match the calling CLI + process or any of its parents. This prevents ``hermes gateway status`` from falsely counting the + ``hermes`` CLI that invoked it as a running gateway instance (see #13242). + """ ancestors: set[int] = set() pid = os.getpid() for _ in range(64): @@ -490,6 +563,8 @@ def _scan_gateway_pids( exclude_pids: set[int], all_profiles: bool = False, include_restart_managers: bool = False ) -> list[int]: """Best-effort process-table scan for gateway PIDs (backs up a stale/missing PID file; ``--all`` sweeps).""" + # Exclude the entire ancestor chain so the CLI process that invoked this scan (e.g. ``hermes gateway + # status``) is never mistaken for a running gateway. See #13242. exclude_pids = exclude_pids | _get_ancestor_pids() pids: list[int] = [] # Strict matcher shared with gateway.status: requires a real ``gateway run`` argv, so @@ -595,6 +670,12 @@ def _windows_process_listing() -> str | None: ``bounded_probe_run``, NOT ``subprocess.run(timeout=...)``: run()'s post-timeout cleanup joins pipe readers unbounded and a conhost.exe holding duplicated handles wedges the caller forever; it also hides the console window this windowless pythonw backend would flash.""" + # Prefer wmic when present (fast, stable output format). On modern Windows 11 / Win 10 late builds, wmic + # has been removed as part of the WMIC deprecation — fall back to PowerShell's Get-CimInstance. A spawn + # failure or timeout (result is None) trips the fallback. ``hermes update`` hung exactly there on + # slow-WMI machines where the full Win32_Process scan exceeds its budget (#87134). bounded_probe_run + # also hides the console window: this scan runs inside the windowless pythonw.exe gateway/desktop + # backend, so a bare wmic/powershell spawn would flash a conhost window on every watchdog probe. from hermes_cli._subprocess_compat import bounded_probe_run wmic_path = shutil.which("wmic") result = None @@ -830,7 +911,15 @@ def _capture_gateway_argv(pid: int) -> list[str] | None: def _prepare_profile_gateway_update_restart(profile: str, pid: int) -> str | None: """Choose who relaunches a profile gateway after ``hermes update``: ``--external-supervisor`` gateways exit back to their manager (a detached watcher would race its replacement); otherwise arm the - profile-derived detached watcher, falling back to replaying the captured command line.""" + profile-derived detached watcher, falling back to replaying the captured command line. + + When the profile-derived relaunch cannot be armed -- typically because ``_gateway_run_args_for_profile`` + cannot rebuild a run argv for this profile -- fall back to replaying the process's own captured command + line, which is what ``launch_detached_gateway_restart_by_cmdline`` exists for and what the Windows + post-update path already does for its unmapped gateways. Without this the caller has no way to relaunch + the process and (before #88654) silently left it running pre-update modules against post-update code on + disk. ``argv`` is already captured above, so the fallback costs nothing extra. + """ argv = _capture_gateway_argv(pid) if argv and "--external-supervisor" in argv: return "external-supervisor" @@ -865,6 +954,7 @@ def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool: # normalizes the interpreter and captures a stable cwd + env overlay (HERMES_HOME, # VIRTUAL_ENV, PYTHONPATH) so the respawn doesn't depend on the watcher's cwd. No-op on POSIX. respawn_cwd = "" + # See gateway_windows.windowless_gateway_restart_spec. See #54220, #56747. respawn_env_overlay: dict[str, str] = {} if sys.platform == "win32": try: @@ -1250,7 +1340,13 @@ def _parse_launchd_pid_from_print_output(output: str) -> int | None: def _launchd_print_service_pid(domain: str, label: str) -> tuple[bool, int | None]: """``(loaded, pid)`` for ``domain/label`` via ``launchctl print`` (domain-explicit; ``launchctl list`` - infers it from caller context). ``TimeoutExpired`` propagates: a wedged launchctl is not "unloaded".""" + infers it from caller context). ``TimeoutExpired`` propagates: a wedged launchctl is not "unloaded". + + Domain-explicit on purpose: legacy ``launchctl list`` infers its domain from the caller's execution + context, which is exactly the ambiguity that sank the first fleet-restart attempt (#41403 review). + ``TimeoutExpired`` propagates — fleet-restart callers own per-label failure accounting (a wedged + launchctl call must be reported, not read as "unloaded"). + """ try: result = subprocess.run(["launchctl", "print", f"{domain}/{label}"], timeout=5, **_CAPTURE_TEXT) except FileNotFoundError: @@ -1437,6 +1533,10 @@ def kill_gateway_processes(force: bool = False, exclude_pids: set | None = None, expected_start_time = None if force: # Re-verify the LIVE cmdline at kill time: a PID recycled since the scan must never be tree-killed. + # Re-verify at kill time, not just scan time: the cmdline match inside find_gateway_pids() + # is stale by the time we get here, and a recycled PID could otherwise be tree-killed + # (#89614 class). _capture_gateway_argv re-reads the LIVE cmdline and returns None for + # anything that no longer looks like a gateway — refuse those. if _capture_gateway_argv(pid) is None: continue from gateway.status import get_process_start_time @@ -1459,7 +1559,15 @@ def _reaper_candidate_is_supervisor_owned(pid: int) -> bool: """True when ``pid``'s parent chain reaches ``services.exe`` (Task Scheduler-owned gateway). Windows-only reaper backstop: ``_get_service_pids()`` is empty there, so a Scheduled-Task gateway with a stale pidfile would look like an orphan. Fail-open once the Task's bootstrap parent exits. Not applied - on POSIX, where everything descends from PID 1 and would look supervised.""" + on POSIX, where everything descends from PID 1 and would look supervised. + + See #83683, #86098. + This check is deliberately NOT applied on POSIX: there, every process has PID 1 (launchd / init / + systemd) in its ancestry — and a genuine orphan is *reparented directly to PID 1* — so supervisor-name + ancestry carries zero signal and would spare every orphan the reaper exists to kill (#51325, 75936). + POSIX supervised gateways are already covered pidfile- independently by the ``_get_service_pids()`` + exclusion. + """ if not is_windows(): return False try: @@ -1490,6 +1598,11 @@ def _reap_unsupervised_gateway_orphans(extra_exclude: set | None = None) -> bool return False # Task Scheduler is a supervisor too; its state beats a parent-chain walk (broken once the bootstrap exits). + # A Scheduled Task gateway whose conhost/VBS bootstrap has already exited is invisible to + # `_reaper_candidate_is_supervisor_owned` (the parent chain breaks before services.exe, fail-open), yet + # it is alive and supervised. After that launcher exits the task is typically Ready, not Running — + # treating only Running as supervised still kills the detached gateway on every desktop serve start + # (#86098, #87001). if is_windows(): try: from hermes_cli.gateway_windows import get_task_name # profile-aware task name @@ -1512,6 +1625,11 @@ def _reap_unsupervised_gateway_orphans(extra_exclude: set | None = None) -> bool return False # Pin each orphan's start time now: the delayed SIGKILL must never hit a recycled PID. + # Pin each orphan's identity NOW: the cmdline scan above matched at scan-time only, and the SIGKILL + # escalation below fires seconds later. A PID recycled inside that window must never be force-killed + # (#89614 class). Fingerprint capture is best-effort — SIGTERM below proceeds regardless (it targets the + # process verified by the scan an instant ago), but the delayed SIGKILL requires a still-matching + # fingerprint. orphan_identity: dict[int, int] = {} for pid in orphans: start = get_process_start_time(pid) @@ -1546,6 +1664,15 @@ def _reaper_exclusion_pids(extra_exclude: set | None) -> set[int]: # Service-managed gateways are never orphans (on macOS supports_systemd_services() is False, so a # launchd gateway would otherwise be SIGTERM'd); all_profiles because the scan sees siblings too. with contextlib.suppress(Exception): + # This covers macOS launchd (supports_systemd_services() is False there, so without this the launchd + # gateway looks like an unsupervised orphan and gets SIGTERM'd, causing launchd to restart it — or + # leaving it down under KeepAlive.SuccessfulExit=false) and any systemd unit reachable from a host + # that got past the gate above (#83683, #85344). + # all_profiles=True: the reaper's process scan sees every profile's gateway (and on macOS the + # now-working ps fallback surfaces sibling launchd gateways, #73626), so the service exclusion must + # cover the whole ai.hermes.gateway* fleet — not just the current profile's label — or a sibling + # profile's launchd gateway is misclassified as an unsupervised orphan and reaped. Same class as the + # update-sweep fix in #74075. own |= _get_service_pids(all_profiles=True) # Exempt the recorded gateway PID and its parent chain (on Windows the Scheduled-Task bootstrap's # ``gateway run`` argv matches the scan; killing it takes the gateway down). Use the RAW pidfile + @@ -1628,7 +1755,12 @@ def _mark_planned_stop(pid: int | None = None) -> None: def stop_profile_gateway() -> bool: """Stop only this profile's gateway via its PID file; True if a process was stopped. Without a supervisor the pidfile can be stale while a live orphan holds the webhook port, so fall back to - the orphan-aware scan rather than stacking a duplicate.""" + the orphan-aware scan rather than stacking a duplicate. + + Even when the pid file is valid and points to the current gateway, older orphans may linger from prior + restarts that overwrote the pid file before the old process exited. After killing the recorded PID, also + sweep for any remaining orphans so each restart produces at most one live gateway (#75936). + """ try: from gateway.status import get_running_pid, remove_pid_file except ImportError: @@ -1659,6 +1791,8 @@ def stop_profile_gateway() -> bool: # Reap orphans from prior restarts whose pidfile entry was overwritten; skip the PID just killed. try: + # Exclude the PID we just killed so the sweep doesn't double-kill a process that's still tearing + # down — _reap_unsupervised_gateway_orphans already excludes our own PID. See #75936. _reap_unsupervised_gateway_orphans(extra_exclude={pid} if pid else None) except Exception as exc: logger.debug("orphan reap after stop_profile_gateway failed: %s", exc) @@ -1713,6 +1847,8 @@ def _gw_windows(): # Task Scheduler states meaning "still supervised" (Ready = steady state after the launcher exits). +# Task Scheduler states that mean "this profile still has an official supervisor". Queued is a rare +# in-between. Disabled / MISSING are not supervisors. See #87001. _WINDOWS_TASK_SUPERVISOR_STATES = frozenset({"Running", "Ready", "Queued"}) @@ -1739,7 +1875,14 @@ def _windows_scheduled_task_state(task_name: str) -> str | None: def _windows_scheduled_task_supervises(task_name: str) -> bool: """True when Task Scheduler still owns this profile's gateway (Ready counts: the task is Ready, not - Running, after bootstrap exits). Any failure returns False so callers fall back to pidfile / parent-chain.""" + Running, after bootstrap exits). Any failure returns False so callers fall back to pidfile / parent-chain. + + Used to treat Task Scheduler as a gateway supervisor on Windows: the orphan-reap sweep must not kill a + gateway that a scheduled task launched and left detached. After the bootstrap exits the task is Ready, + not Running; a Running-only check still writes the planned-stop marker, the gateway exits cleanly with + code 0, and the scheduler never restarts it — silently killing A2A/messaging on every desktop-app launch + (#86098, #87001). + """ return _windows_scheduled_task_state(task_name) in _WINDOWS_TASK_SUPERVISOR_STATES @@ -1873,7 +2016,14 @@ def _user_systemd_private_socket_path() -> Path: def _path_exists_safe(path: Path) -> bool: """``Path.exists()`` treating an inaccessible path as absent: a leaked ``XDG_RUNTIME_DIR`` from - another user (``/run/user/0`` is 0700) would otherwise crash the preflight with EACCES.""" + another user (``/run/user/0`` is 0700) would otherwise crash the preflight with EACCES. + + ``Path.exists()`` only swallows a subset of ``OSError`` (ENOENT/ENOTDIR/ EBADF/ELOOP); ``EACCES`` still + propagates. When ``XDG_RUNTIME_DIR`` leaks from another user — the classic ``su``/``sudo -u`` from a + root shell case, where ``/run/user/0`` is ``0700 root:root`` — stat-ing a socket underneath it raises + ``PermissionError`` that escapes the systemd preflight as a raw traceback (#86558). An unreadable path + is, for our purposes, not reachable. + """ try: return path.exists() except OSError: # e.g. EACCES on another user's runtime dir @@ -1896,7 +2046,12 @@ def _user_systemd_socket_ready() -> bool: def _ensure_user_systemd_env() -> None: """Set XDG_RUNTIME_DIR / DBUS_SESSION_BUS_ADDRESS so ``systemctl --user`` works on headless (SSH) - hosts; an XDG_RUNTIME_DIR leaked from another user is replaced with our own ``/run/user/{uid}``.""" + hosts; an XDG_RUNTIME_DIR leaked from another user is replaced with our own ``/run/user/{uid}``. + + An ``XDG_RUNTIME_DIR`` that leaked from another user (``su``/``sudo -u`` from root, where the env still + points at ``/run/user/0``) is dropped in favour of our own ``/run/user/{uid}`` so ``systemctl --user`` + targets the right instance instead of an unreadable foreign socket (#86558). + """ uid = os.getuid() # windows-footgun: ok — POSIX systemd helper, never invoked on Windows xdg = os.environ.get("XDG_RUNTIME_DIR") if (not xdg or not _runtime_dir_is_ours(xdg)) and _runtime_dir_is_ours(f"/run/user/{uid}"): @@ -2060,7 +2215,12 @@ def _legacy_unit_search_paths() -> list[tuple[bool, Path]]: def _find_legacy_hermes_units() -> list[tuple[str, Path, bool]]: """``[(unit_name, unit_path, is_system)]`` for legacy gateway units (e.g. ``hermes.service``), which fight the current unit for the bot token (SIGTERM flap loop). Explicit name allowlist + ExecStart - marker check so profile/third-party units never match; no mutation.""" + marker check so profile/third-party units never match; no mutation. + + Detects unit files installed by older Hermes versions that used a different service name (e.g. When both + a legacy unit and the current ``hermes-gateway.service`` are active, they fight over the same bot token + — the PR #5646 signal-recovery change turns this into a 30-second SIGTERM flap loop. + """ results: list[tuple[str, Path, bool]] = [] for is_system, base in _legacy_unit_search_paths(): for name in _LEGACY_SERVICE_NAMES: @@ -2868,13 +3028,20 @@ def _agent_timeout_setting(env_var: str, key: str, parse) -> float: def _get_cron_drain_timeout() -> float: - """Return the configured cron-only drain floor in seconds.""" + """Return the configured cron-only drain floor in seconds. + + See #82161. + """ return _agent_timeout_setting("HERMES_CRON_DRAIN_TIMEOUT", "cron_drain_timeout", parse_cron_drain_timeout) def _get_restart_exit_wait_budget() -> float: """CLI wait for gateway exit after SIGUSR1 / self-restart (#77184).""" return resolve_restart_exit_wait_budget( + # TimeoutStopSec must cover the full stop budget, not just restart_drain_timeout. Cron work can + # legally wait cron_drain_timeout plus cleanup reserve before interrupt/teardown, and systemd + # SIGKILLs if the unit's deadline is shorter (#94759). 30s of post-drain headroom is preserved on + # top, with a 60s floor. _get_restart_drain_timeout(), _agent_timeout_setting( "HERMES_RESTART_AFTER_TURN_TIMEOUT", "restart_after_turn_timeout", parse_restart_after_turn_timeout @@ -3027,6 +3194,15 @@ def systemd_restart(system: bool = False): if pid is not None and probe_gateway_loop_liveness(pid) == GATEWAY_LOOP_WEDGED: # Event loop provably dead: SIGUSR1 can't drain it, so bounded SIGTERM → SIGKILL and let systemd relaunch. print( + # Health probe says the event loop is provably dead (#81642): SIGUSR1 can never drain it, so the + # graceful wait below would burn the full budget. A busy-but-alive gateway (fresh heartbeat) + # never takes this path — its in-flight work, including the #86684 cron drain floor, keeps the + # full graceful budget. + # Health probe says the event loop is provably dead (#81642): the gateway cannot process a + # graceful shutdown, so waiting the full drain budget only stalls the restart (and `hermes + # update` behind it) for 180s. Bounded escalation instead: SIGTERM grace → SIGKILL → proceed, + # ~10s worst case. Never taken for a busy-but-alive gateway — a fresh heartbeat keeps the drain + # path (and the #86684 cron drain floor) fully intact. f"⚠ Gateway PID {pid} event loop is unresponsive — " "skipping graceful drain and forcing a bounded stop..." ) @@ -3051,6 +3227,17 @@ def _systemd_graceful_restart_action(system: bool, pid: int) -> str | None: """SIGUSR1-drain the live gateway ``pid``; return the follow-up ``systemctl`` verb (``"start"`` / ``"restart"``) the caller must still issue, or None when systemd already owns the relaunch.""" scope_label = _service_scope_label(system).capitalize() + # Graceful in-band restart, mirroring the systemd branch. Previously this sent a bare SIGTERM and waited + # ``_get_restart_drain_timeout()`` — which defaults to 0, so the wait could never succeed and every + # restart fell through to ``kickstart -k``. A bare SIGTERM also leaves ``restart_requested`` False, so + # the gateway exits 1 instead of 75 and reports itself to chat as "shutting down" rather than + # "restarting", losing the resume_pending handoff. SIGUSR1 is the drain-aware path: refuse new turns, + # wait for in-flight work (``agent.restart_after_turn_timeout``), then stop() within + # ``agent.restart_drain_timeout``. The wait budget must cover BOTH phases plus headroom (#77184) — the + # raw drain timeout covers only the second. Announce the wait BEFORE it runs: it can last the full + # budget while the old gateway finishes in-flight agent runs, and it streams into surfaces with no other + # feedback — the desktop updater's live output most of all, where a silent stop here reads as "update + # stuck" (#44515). wait_budget = _get_restart_exit_wait_budget() print( f"⏳ {scope_label} service restarting gracefully (PID {pid}) — " @@ -3224,7 +3411,10 @@ def _probe_launchd_domain_for_label(label: str) -> str: def _launchd_domain() -> str: - """Domain managing the current profile's gateway; cached per process so start/stop/restart agree.""" + """Domain managing the current profile's gateway; cached per process so start/stop/restart agree. + + See #40831, #23387. + """ global _resolved_launchd_domain if _resolved_launchd_domain is None: _resolved_launchd_domain = _probe_launchd_domain_for_label(get_launchd_label()) @@ -3238,6 +3428,10 @@ _LAUNCHD_JOB_UNLOADED_EXIT_CODES = frozenset({3, 113, 125}) # 5 (EIO) / persistent 125 mean either a stale still-registered label (recoverable: bootout + # bootstrap, which `_launchctl_bootstrap()` tries first) or a domain that genuinely can't manage # services (macOS 26+). Only when the retry ALSO fails do callers degrade to a detached process. +# launchctl returns 5 ("Input/output error") or a persistent 125 in two very different situations, so exit 5 +# is NOT on its own proof the domain is broken: 1. See #42914. 2. Here launchd cannot supervise the gateway +# at all and we degrade to a detached background process (the `nohup hermes gateway run` workaround). See +# #23387. _LAUNCHCTL_DOMAIN_UNSUPPORTED_CODES = frozenset({5, 125}) @@ -3363,7 +3557,19 @@ def _timestamped_stderr_gateway_command(error_log: Path, *, external_supervisor: """Wrap gateway run so raw stderr lines are timestamped before file write. ``external_supervisor`` (launchd ProgramArguments only) adds ``--external-supervisor`` so ``hermes update`` hands back to launchd, and drops ``--replace``: KeepAlive respawns would re-arm takeover, so two profiles sharing - a token would kill each other forever.""" + a token would kill each other forever. + + ``external_supervisor=True`` is for launchd ProgramArguments only: the inner ``gateway run`` must carry + ``--external-supervisor`` so ``hermes update`` sees the flag on the live grandchild argv and hands the + process back to launchd instead of starting a detached watcher (#86893 / #87005). The detached nohup + fallback stays unmarked. + Supervised starts also drop ``--replace`` (issue #79048): a launchd service is respawned by KeepAlive, + so takeover authority would be re-armed on every respawn — two profiles legitimately sharing one + platform token would each terminate the sibling, and launchd would revive the victim forever. Bounded + replacement is the lifecycle commands' job (``launchctl kickstart -k``, drain in ``launchd_restart()``, + bootout+bootstrap in install/refresh), which run before supervision resumes. Mirrors + ``generate_systemd_unit``, whose ExecStart also runs ``gateway run`` without ``--replace``. + """ inner = _gateway_run_command() if external_supervisor: inner = [part for part in inner if part != "--replace"] @@ -3374,7 +3580,13 @@ def _timestamped_stderr_gateway_command(error_log: Path, *, external_supervisor: def _spawn_detached_gateway() -> bool: """Launch the gateway detached (launchd fallback for macOS 26+). CLI-managed nohup equivalent: - stdout → gateway.log, timestamped stderr → gateway.error.log, PID via gateway.pid so stop/status work.""" + stdout → gateway.log, timestamped stderr → gateway.error.log, PID via gateway.pid so stop/status work. + + Used when launchctl can no longer bootstrap/kickstart the gateway on macOS 26+ (issue #23387). Mirrors + the `nohup hermes gateway run --replace` workaround but keeps it CLI-managed: stdout goes to + gateway.log, stderr is timestamped into gateway.error.log, and the PID is tracked via the gateway.pid + file that `run_gateway` writes, so stop/status/restart keep working. + """ from hermes_cli._subprocess_compat import windows_detach_popen_kwargs log_dir = get_hermes_home() / "logs" log_dir.mkdir(parents=True, exist_ok=True) @@ -3565,6 +3777,13 @@ def _spawn_deferred_launchd_reload( ) try: # `launchctl submit` rather than setsid: setsid does NOT leave the launchd coalition that bootout kills. + # Spawn the reload helper via `launchctl submit` (a transient launchd one-shot job) instead of + # `start_new_session=True`. `start_new_session=True` only calls setsid(2), which creates a new POSIX + # session but does NOT move the child outside the launchd job's process coalition. When `launchctl + # bootout` fires on the gateway label, launchd terminates ALL processes in that coalition — + # including a setsid-detached child (#69098). `launchctl submit` creates a wholly independent + # transient launchd job that launchd manages separately from the gateway, so bootout of the gateway + # job cannot reach the helper. subprocess.Popen( [ "launchctl", "submit", "-l", submit_label, "-o", str(reload_log_path), "-e", str(reload_log_path), @@ -3748,6 +3967,9 @@ def launchd_stop(): subprocess.run(["launchctl", "bootout", target], check=True, timeout=90) except subprocess.CalledProcessError as e: # Job already unloaded (3/113/125) or domain unmanageable (5/125): fall through to the PID-based kill. + # Job already unloaded (3/113/125), or the domain can't be managed at all (5/125, macOS 26+ + # detached-fallback process, issue #23387) — in both cases just fall through to the PID-based kill + # below. if not (_launchd_error_indicates_unloaded(e) or _launchctl_domain_unsupported(e.returncode)): raise _wait_for_gateway_exit(timeout=10.0, force_after=5.0) @@ -3869,7 +4091,17 @@ def wait_for_launchd_gateway_supervision( ) -> bool: """Poll launchd until it supervises a live gateway; True at once if the detached fallback is active. ``launchd_restart`` returns once the restart is *requested* (asynchronous), so it can't see a helper - dying before bootstrap or a ``launchctl bootstrap`` that exits 0 without registering.""" + dying before bootstrap or a ``launchctl bootstrap`` that exits 0 without registering. + + The ``_request_gateway_self_restart`` branch hands the work to the running gateway and returns + immediately, and a plist reload is handed to a detached helper. Both are asynchronous, so a caller that + reads "returned without raising" as "the service is up" cannot see a helper that dies before its first + bootstrap (#88848) — nor a ``launchctl bootstrap`` that exits 0 without registering, which the reporter + measured on macOS 26.6.1. + Judge the outcome the way #80491 taught the helper to judge it: by a live supervised pid, never by an + exit code. :func:`_launchctl_label_supervising_process` is already that predicate, so this only adds + the wait. + """ if _launchd_unsupported_marker_exists(): return True @@ -3970,7 +4202,10 @@ def _running_under_gateway_supervisor() -> bool: def named_profile_served_by_running_multiplexer(profile_name: str | None = None) -> bool: """True when a live default multiplexer already ticks this named profile (a satellite profile has no - gateway.pid; the multiplexer fires its jobs and serves its platforms). Defaults to the current profile.""" + gateway.pid; the multiplexer fires its jobs and serves its platforms). Defaults to the current profile. + + See #97120. + """ try: suffix = profile_name if profile_name is not None else _profile_suffix() except Exception: @@ -4055,13 +4290,24 @@ def _guard_named_profile_under_multiplexer(force: bool = False) -> None: # (Restart=always, StartLimitIntervalSec=0) relies on RestartPreventExitStatus=78 as its only # backstop — exit 1 turned a correct refusal into an unbounded restart loop; s6 maps 78 to # "permanent failure" too. + # This refusal is decided entirely by configuration (multiplex_profiles plus the allowlist), so it is + # permanent: no number of retries can change the answer. Exiting 1 made it look transient to a service + # manager -- and the systemd unit this module generates pairs Restart=always/RestartSec=5 with + # StartLimitIntervalSec=0, deliberately trading systemd's generic start-rate limiter for the specific + # RestartPreventExitStatus=GATEWAY_FATAL_CONFIG_EXIT_CODE backstop declared beside it. Returning 1 left + # that backstop unarmed with the limiter already off, so a correct refusal became an unbounded restart + # loop. 78 also reaches the s6 finish script's 125 "permanent failure" translation (see #51228), the + # same path the other fatal-config exits take. sys.exit(GATEWAY_FATAL_CONFIG_EXIT_CODE) def _guard_supervised_gateway_conflict(force: bool = False) -> None: """Refuse a foreground gateway when a service manager already supervises one: a shell-launched run becomes a second dispatcher that escapes the cgroup, survives ``systemctl restart``, and writes the - shared kanban DB concurrently (multi-writer SQLite WAL corruption). ``--force`` starts anyway.""" + shared kanban DB concurrently (multi-writer SQLite WAL corruption). ``--force`` starts anyway. + + See #35240. + """ if force or _running_under_gateway_supervisor(): return try: @@ -5445,6 +5691,17 @@ def _maybe_redirect_run_to_s6_supervision(args) -> bool: # SIGTERMs it). Prefer `sleep infinity` (frees the interpreter); execvp only returns by raising # (ENOENT with a clobbered PATH / no `sleep`), which used to crash containers. try: + # The supervised gateway's lifetime is independent of this process — s6-supervise restarts it on + # crash, and we don't want the container to exit when the gateway flaps. The CMD process keeps /init + # alive until `docker stop` sends SIGTERM, at which point /init runs stage 3 shutdown (which tears + # down the supervised gateway cleanly). Prefer `sleep infinity` (matches the static main-hermes + # service's pattern in docker/s6-rc.d/main-hermes/run, and frees the Python interpreter — the + # heartbeat is a tiny `sleep` process, not a resident interpreter). But `os.execvp` does a PATH + # lookup for the `sleep` binary and historically crashed the whole container with FileNotFoundError + # when PATH was empty/truncated/clobbered at this point — e.g. after user customizations rewrote + # PATH, or on minimal images without `sleep` on PATH (issue #36208). Fall back to an in-process + # block (no external binary, can't fail on PATH) so the container keeps running instead of dying + # during boot. os.execvp("sleep", ["sleep", "infinity"]) except OSError: print( @@ -5459,7 +5716,14 @@ def _maybe_redirect_run_to_s6_supervision(args) -> bool: def _block_until_terminated() -> None: """Heartbeat when ``execvp("sleep")`` fails. SIGTERM exits 128+signum so ``docker stop`` is clean; - ``Event().wait()`` covers platforms without ``signal.pause()``.""" + ``Event().wait()`` covers platforms without ``signal.pause()``. + + Fallback heartbeat for when ``os.execvp("sleep", ...)`` can't run (``sleep`` missing from PATH — issue + #36208). Installs a SIGTERM handler that exits with the conventional 128+signum code so ``docker stop`` + produces a clean, expected exit, then blocks on ``signal.pause()``. Windows) — although this path only + runs inside the s6 Linux container image, the fallback keeps the helper safe to import and unit-test + anywhere. + """ signal.signal(signal.SIGTERM, lambda signum, _frame: sys.exit(128 + signum)) pause = getattr(signal, "pause", None) if pause is not None: diff --git a/hermes_cli/gateway_windows.py b/hermes_cli/gateway_windows.py index a561c7c376..a11ab462b2 100644 --- a/hermes_cli/gateway_windows.py +++ b/hermes_cli/gateway_windows.py @@ -169,7 +169,10 @@ def _launch_elevated_gateway_command(command: str, extra_args: list[str] | None """Launch an elevated gateway subcommand via UAC and return True on handoff. The child is console ``python.exe`` with ``SW_HIDE``: it owns one hidden console its subprocesses (schtasks, taskkill) inherit — no visible window and no per-descendant conhost flashes (the console-less pythonw.exe - alternative re-created #54220/#56747 for every descendant).""" + alternative re-created #54220/#56747 for every descendant). + + All operator decisions are already collected in the parent shell before this point. See #54220, #56747. + """ _assert_windows() args = ["-m", "hermes_cli.main", *_current_profile_cli_args(), "gateway", command, *(extra_args or [])] params = subprocess.list2cmdline(args) @@ -328,6 +331,15 @@ def _build_gateway_vbs_script(python_path: str, working_dir: str, hermes_home: s groups, killing a cmd-hosted gateway with STATUS_CONTROL_C_EXIT, which Task Scheduler treats as a user cancel (``RestartOnFailure`` never fires). wscript has no console; python.exe runs with window style 0 so descendants inherit one hidden console instead of flashing their own (#54220/#56747). + + Why: issue #45599 root cause #1. + ``wscript.exe`` is a GUI-subsystem executable with no console, so this launcher receives no console + control events. It ``Run``s the console ``python.exe`` with window style 0 (hidden): the gateway owns a + single hidden console — never shown, never CTRL_CLOSE'd at logon, and inherited by every + console-subsystem descendant (git, gh, node, …) so none of them allocate a visible flashing conhost + (#54220/#56747; the previous console-less pythonw.exe gateway forced exactly that per-descendant flash). + No cmd.exe anywhere in the chain. Mirrors ``_build_gateway_cmd_script`` (same env + argv via + ``_resolve_detached_python``). """ python_exe_path, venv_dir, extra_pythonpath = _resolve_detached_python(python_path) # list2cmdline gives CreateProcess-correct quoting for WScript.Shell.Run. @@ -382,6 +394,8 @@ def _write_task_script() -> Path: settings = _launcher_settings() script_path = get_task_script_path() _atomic_write(script_path, _build_gateway_cmd_script(*settings), script_path.with_suffix(".tmp")) + # Also render the console-less .vbs launcher used by Scheduled Task and the Startup-folder fallback via + # wscript.exe (issue #45599 fix A). The .cmd wrapper stays as a generated helper/compatibility artifact. vbs_path = script_path.with_suffix(".vbs") _atomic_write(vbs_path, _build_gateway_vbs_script(*settings), vbs_path.with_name(vbs_path.name + ".tmp")) return script_path @@ -408,7 +422,10 @@ def _resolve_task_user() -> str | None: def _build_scheduled_task_xml(task_name: str, launcher_path: Path, user: str | None) -> str: """Task Scheduler XML with safe long-running defaults. ``launcher_path`` is the console-less - ``.vbs`` run via ``wscript.exe`` (see ``_build_gateway_vbs_script`` for why not cmd.exe).""" + ``.vbs`` run via ``wscript.exe`` (see ``_build_gateway_vbs_script`` for why not cmd.exe). + + See #45599. + """ user_principal = f"\n {escape(user)}" if user else "" return f""" @@ -474,6 +491,7 @@ def _install_scheduled_task(task_name: str, script_path: Path) -> tuple[bool, st launcher_path = script_path.with_suffix(".vbs") # the task launches the console-less .vbs xml_path = launcher_path.with_suffix(".task.xml") xml_path.write_text(_build_scheduled_task_xml(task_name, launcher_path, user), encoding="utf-16", newline="") + # Immediate manual starts use _spawn_detached(). See #45599. base = ["/Create", "/F", "/TN", task_name, "/XML", str(xml_path)] variants = [[*base, "/RU", user, "/NP", "/IT"], base] if user else [base] last_code, last_err = 1, "" @@ -509,7 +527,27 @@ def _install_startup_entry(script_path: Path) -> Path: def _resolve_detached_python(python_exe: str) -> tuple[str, Path, list[str]]: """Return (hidden_console_python, venv_dir, extra_pythonpath) for detached runs. ``extra_pythonpath`` - is always empty now; the tuple shape is kept so every call site stays unchanged.""" + is always empty now; the tuple shape is kept so every call site stays unchanged. + + Returns the venv's **console** ``python.exe`` — deliberately NOT ``pythonw.exe``. Every detached launch + path pairs this interpreter with a hidden-console mechanism (``CREATE_NO_WINDOW`` creationflags, or + ``WScript.Shell.Run`` window style 0), so the daemon owns a single hidden console that all of its + console-subsystem descendants (git, gh, cmd, node, wmic, powershell, …) inherit instead of each + allocating a visible flashing one. A GUI-subsystem ``pythonw.exe`` daemon has NO console, which is what + made every descendant spawn flash (#54220/#56747) and forced the endless per-call-site CREATE_NO_WINDOW + sweep. Root cause isolated + A/B verified on Windows 11 by the desktop backend fix (commit aa2ae36c3f). + - uv venv launcher: ``venv\\Scripts\\python.exe`` under ``CREATE_NO_WINDOW`` re-execs the base + interpreter *windowless* — the child inherits the shim's hidden console, so no conhost flashes (the + #52239 concern). The historical "CREATE_NO_WINDOW cannot suppress the second window" observations were + made while ``DETACHED_PROCESS`` was in the flag bundle, where MSDN specifies CREATE_NO_WINDOW is IGNORED + — the hide bit was dead, not ineffective. The base-interpreter + PYTHONPATH-overlay detour is therefore + unnecessary; the venv shim resolves imports itself. - Console python restores stdout/stderr, so daemon + logs flow normally. + Legacy normalization: launchers and argv snapshots from pre-aa2ae36c3f installs lead with + ``pythonw.exe``. When the sibling console ``python.exe`` exists, swap to it so respawns and regenerated + launchers get the hidden-console design instead of resurrecting the console-less daemon (the + #54220/#56747 flash class, plus the ``sys.stderr is None`` startup-crash class from #71671). + """ p = Path(python_exe) if p.name.lower() in ("pythonw.exe", "pythonw"): sibling = p.with_name("python.exe" if p.suffix else "python") @@ -549,7 +587,16 @@ def _build_gateway_argv() -> tuple[list[str], str, dict[str, str]]: def windowless_gateway_restart_spec(run_argv: list[str]) -> tuple[list[str], str, dict[str, str]]: """(argv, cwd, env overlay) for a hidden-console gateway respawn; arguments after the interpreter - are preserved verbatim. Non-Windows or a non-python argv[0] → argv unchanged, empty overlay.""" + are preserved verbatim. Non-Windows or a non-python argv[0] → argv unchanged, empty overlay. + + The post-update restart paths build their respawn command from ``get_python_path()`` (the venv's console + ``python.exe``). That is the right interpreter: the watcher launches it with ``CREATE_NO_WINDOW`` detach + flags, so the respawned gateway owns a single hidden console that all of its descendants inherit — + nothing flashes (#54220/#56747; the old pythonw.exe rewrite here produced a console-less gateway whose + every console-subsystem child allocated a visible conhost). This helper now only normalizes the + interpreter via ``_resolve_detached_python`` and supplies the stable cwd + env overlay (HERMES_HOME, + VIRTUAL_ENV, PYTHONPATH) so the respawn doesn't depend on the watcher's transient working directory. + """ if not run_argv or sys.platform != "win32": return run_argv, "", {} from hermes_cli.gateway import PROJECT_ROOT @@ -576,7 +623,14 @@ def _spawn_detached(script_path: Path | None = None) -> int: gets reaped when the shell exits. Flags: CREATE_NEW_PROCESS_GROUP (no Ctrl+C from our group), CREATE_NO_WINDOW (hidden console descendants inherit, so nothing flashes — #54220/#56747; the old DETACHED_PROCESS made every descendant spawn flash), CREATE_BREAKAWAY_FROM_JOB - (escape a parent Job Object — some Windows Terminal versions wrap children in one).""" + (escape a parent Job Object — some Windows Terminal versions wrap children in one). + + With ``CREATE_NO_WINDOW`` the gateway gets its OWN hidden console instead of inheriting ours, so it + survives our shell closing, and every console-subsystem descendant it spawns inherits that hidden + console instead of flashing a visible one (#54220/#56747 — this is why we don't use console-less + pythonw.exe here). Combined with CREATE_NEW_PROCESS_GROUP + DEVNULL stdin + a fresh env, the resulting + process is independent of whichever shell started it. + """ _assert_windows() argv, working_dir, env_overlay = _build_gateway_argv() env = {**os.environ, **env_overlay} @@ -758,7 +812,14 @@ def install( def _confirm_gateway_stable(initial_pids: list[int], confirm_s: float, interval_s: float, all_profiles: bool = False) -> list[int]: """Re-check a freshly detected gateway for ``confirm_s`` seconds: one process-table hit proves - the child was *created*, not that it survived startup (or a parent Job Object teardown).""" + the child was *created*, not that it survived startup (or a parent Job Object teardown). + + A single process-table hit only proves the child was *created*, not that it survived startup — a gateway + that crashes moments after spawn (or is reaped by the parent shell's Job Object, #91675/#84185) passes a + first-hit poll and then dies. Require the gateway to stay visible for the whole confirmation window + before we vouch for it. Returns the last observed PID list, or ``[]`` if the gateway vanished + mid-window. + """ if confirm_s <= 0: return initial_pids from hermes_cli.gateway import find_gateway_pids diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index f96b8d38e6..8d0b24f679 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -637,7 +637,14 @@ def clear_goal(session_id: str) -> None: def migrate_goal_to_session(old_session_id: str, new_session_id: str, *, reason: str = "") -> bool: """Carry a persistent /goal from a parent session to its continuation. Best-effort, never raises - (a failure here must not block compression). Returns True when a goal was migrated.""" + (a failure here must not block compression). Returns True when a goal was migrated. + + Context compression rotates ``session_id`` to a fresh child session, but ``load_goal`` does a flat + ``goal:`` lookup with no parent-lineage walk — so an active goal silently dies at the + compaction boundary (#33618). Copy the goal onto the new session and archive the old row as ``cleared`` + so exactly one active goal row exists per logical conversation (avoids the "two active goals" hazard of + a pure copy). + """ if not old_session_id or not new_session_id or old_session_id == new_session_id: return False try: @@ -829,6 +836,8 @@ def _render_background_block(background_processes: Optional[List[Dict[str, Any]] def _call_goal_judge_llm(call_llm, system_prompt: str, user_prompt: str, timeout: Optional[float]) -> str: """Route through call_llm so auxiliary.goal_judge.* config (provider/model, extra_body, reasoning_effort, retries) all apply. Returns the raw reply text.""" + # See #35566. + # Route through call_llm — same #35566 fix as the judge call above. resp = call_llm( task="goal_judge", messages=[{"role": "system", "content": system_prompt}, {"role": "user", "content": user_prompt}], @@ -920,6 +929,9 @@ def draft_contract(objective: str, *, timeout: Optional[float] = None) -> Option if not objective: return None if timeout is None: + # The declared default for this path is the config key, not the module constant — see + # _goal_judge_timeout (#91022). + # Same config-backed default as judge_goal (#91022). timeout = _goal_judge_timeout() try: @@ -1360,6 +1372,8 @@ class GoalManager: # BLOCKED is NOT done: pause so the user sees the judge's reason and can re-scope or override, # instead of burning turns on an unachievable goal or waving it through as complete. + # BLOCKED verdict: the judge ruled the goal genuinely cannot be satisfied as stated (impossible, out + # of scope, needs user input). See #100954. if verdict == "blocked": return self._pause_decision( f"judged unachievable: {reason}", "blocked", reason, @@ -1525,6 +1539,7 @@ def run_kanban_goal_loop( if verdict == "blocked": # Unachievable is NOT done: block the card with the judge's reason now instead of # re-poking an impossible goal, and never let it land in done. + # The judge ruled the goal cannot be satisfied at all — this is NOT done (#100954). _log(f"kanban goal loop: task {task_id} judged unachievable; blocking") _block(f"Goal-mode judge ruled the goal unachievable: {reason}") return _result("blocked_unachievable", f"judge verdict blocked: {reason}") diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index 7f3ecd613a..9f910f8162 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -167,6 +167,11 @@ def _strip_aggregator_overlaps(rows: list[dict]) -> None: for row in rows: if row.get("is_user_defined") or not is_routing_aggregator(row.get("slug", "")): continue + # Only strip overlaps from TRUE routing aggregators (OpenRouter, custom:* proxies). Flat-namespace + # resellers (opencode-go / opencode-zen) serve every listed model as a first-party model, so their + # rows must keep models that a user's proxy happens to share a name with — otherwise a subscription + # provider's own catalog (minimax-m3, glm-5, deepseek-v4-flash, ...) is silently gutted in the + # picker. (#47077) original = row.get("models") or [] filtered = [m for m in original if m.lower() not in user_models] if len(filtered) < len(original): @@ -203,7 +208,15 @@ def build_aux_picker_rows( """Provider rows for any auxiliary-task picker (vision, compression, …). Honours ``excluded_providers``; exhausted-pool providers stay visible (``for_picker``); only the active custom endpoint is probed. ``moa`` is excluded: auxiliary_client unwraps it to its aggregator - anyway, so offering it would be a choice silently rewritten.""" + anyway, so offering it would be a choice silently rewritten. + + Aux pickers kept re-deriving their own kwargs and each one silently dropped a different slice of the + user's configuration. Two independent contributor PRs landed against the same two call sites for exactly + this: 52642 (user ``providers:`` / ``custom_providers:`` entries never appeared) and #66624 (providers + with an exhausted credential pool were hidden). Both were per-site kwarg patches, so the next aux picker + would have reintroduced the same gap. Routing through one function makes the correct behaviour the + default that a new caller cannot forget: + """ ctx = load_picker_context().with_overrides( current_provider=current_provider, current_model=current_model, current_base_url=current_base_url, ) @@ -345,7 +358,12 @@ def _apply_featured(rows: list[dict]) -> None: def _apply_custom_aliases(rows: list[dict]) -> None: """Attach the accepted identity set to each user-defined row: ``model.options`` reports the canonical - ``custom:`` while rows carry the bare key as ``slug``, so GUI exact-match never finds the row.""" + ``custom:`` while rows carry the bare key as ``slug``, so GUI exact-match never finds the row. + + GUI pickers compare the two to decide which row is active; exact equality never matches for custom + providers (#87035). Exposing ``aliases`` — every current and legacy spelling from + :func:`hermes_cli.providers.custom_provider_aliases` — lets the frontend do a membership check instead. + """ from hermes_cli.providers import custom_provider_aliases for row in rows: diff --git a/hermes_cli/kanban.py b/hermes_cli/kanban.py index d1d24b1aba..758494ba31 100644 --- a/hermes_cli/kanban.py +++ b/hermes_cli/kanban.py @@ -96,6 +96,11 @@ def _check_dispatcher_presence(hermes_home: Optional[Path] = None) -> tuple[bool alive for this HERMES_HOME with ``kanban.dispatch_in_gateway`` on, else False + human guidance. Fails OPEN (probe/config errors -> ``(True, "")``) — a missed warning beats crying wolf. ``hermes_home`` scopes the probe to a profile dir (dashboard backend); CLI callers pass None. + + The dashboard plugin API passes it because the dashboard backend process can be running under a + different HERMES_HOME than the profile the request targets, which otherwise produced a "no gateway is + running" warning against a perfectly healthy profile gateway (#71211). CLI callers leave it ``None`` and + keep the existing process-level behavior. """ try: from gateway.status import resolve_gateway_liveness # type: ignore @@ -612,6 +617,11 @@ def _rows_by_task(conn, table: str, ids: list[str]) -> dict[str, list]: def _cmd_diagnostics(args: argparse.Namespace) -> int: """List active diagnostics on the board via the same rule engine the dashboard uses.""" from hermes_cli import kanban_diagnostics as kd + # Honour kanban.default_assignee as the fallback for unassigned ready tasks (#27145), + # kanban.max_in_progress as the global concurrency cap (#33488), kanban.max_in_progress_per_profile as + # the per-profile cap (#21582), and kanban.max_spawn as the per-tick spawn limit (#28805). Same + # semantics as the gateway dispatch path so behavior matches whether the user runs the CLI directly or + # relies on the gateway-embedded dispatcher. from hermes_cli.config import load_config diag_config = kd.config_from_runtime_config(load_config()) @@ -784,6 +794,11 @@ def _goal_mode_handoff_rejection(task: Optional[kb.Task], evidence: str): Returns ``(verdict, reason_or_None)``: ``"done"`` allows; ``"blocked"`` = judge ruled the goal unachievable; ``"continue"``/``"wait"`` reject with the judge's reason. Judge failures allow the handoff (logged). + + See #100954. + ``{"done", None}`` means the judge allows the handoff; anything else is a rejection whose verdict + disambiguates the guidance the caller gives the worker (``continue`` = not done yet, ``blocked`` = + judged unachievable — see #100954). """ if task is None or not task.goal_mode: return ("done", None) diff --git a/hermes_cli/kanban_boards.py b/hermes_cli/kanban_boards.py index 6724dd53ac..98b4017b9f 100644 --- a/hermes_cli/kanban_boards.py +++ b/hermes_cli/kanban_boards.py @@ -93,6 +93,7 @@ def _cmd_boards_create(args: argparse.Namespace) -> int: def _cmd_boards_rm(args: argparse.Namespace) -> int: # `boards delete ` (alias) never sets args.delete because --delete belongs to the 'rm' # subparser only; treat the alias as `rm --delete`. + # See #23139. force_delete = getattr(args, "delete", False) or getattr(args, "boards_action", "") == "delete" try: res = kb.remove_board(args.slug, archive=not force_delete) diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py index 3688790822..8fd8e6e7e9 100644 --- a/hermes_cli/kanban_db.py +++ b/hermes_cli/kanban_db.py @@ -228,7 +228,13 @@ def _fire_dispatch_tick_hook( result: "DispatchResult", *, board: Optional[str] = None, dry_run: bool = False, ) -> None: """``on_kanban_dispatch_tick`` — strictly AFTER ``_dispatch_tick_lock`` is - released so a slow subscriber cannot stall a sibling dispatcher.""" + released so a slow subscriber cannot stall a sibling dispatcher. + + Re-port of PR #56066 per the #64231 batch disposition: renamed to the taxonomy form and called by + ``dispatch_once`` strictly AFTER ``_dispatch_tick_lock`` has been released — the original fired inside + the lock, so a slow subscriber could extend the single-writer critical section and stall a sibling + dispatcher's tick. Observer-only and fully best-effort: any subscriber failure is swallowed. + """ if not _kanban_observer_consumed("on_kanban_dispatch_tick"): return try: @@ -259,6 +265,11 @@ DEFAULT_CLAIM_TTL_SECONDS = 15 * 60 # A live PID with a heartbeat older than this is wedged and reclaimed anyway # (``_touch_activity`` keeps genuinely active workers fresh). +# If a worker's PID is still alive but its ``last_heartbeat_at`` is older than this when +# ``release_stale_claims`` runs, treat the worker as wedged and reclaim regardless of PID liveness (#29747 +# gap 3). This catches the logic-loop case where the process is technically running but not making +# observable progress. ``_touch_activity`` bridges chunk-level liveness into ``last_heartbeat_at`` via +# #31752, so any genuinely active worker keeps its heartbeat fresh as a side effect of normal API traffic. DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS = 60 * 60 # Grace when a host-local worker survived termination (e.g. parked in D state @@ -1413,6 +1424,9 @@ def _inherit_notify_subs( Single owner of inheritance (create_task, link_tasks, decompose). It must copy EVERY routing/delivery column: dropping ``chat_type`` made DM-originated completions wake a fresh group session instead of the originating DM. + + Omitting columns here silently degrades routing: a DM-originated child completion falls back to + chat_type='group' and wakes a fresh group-scoped session instead of the originating DM (issue #73030). """ parent_ids = tuple(dict.fromkeys(p for p in parents if p)) if not parent_ids: @@ -1626,6 +1640,9 @@ def _linked_ids(conn: sqlite3.Connection, want: str, where: str, task_id: str) - return [r[want] for r in rows] +# Dependency edge removed — re-evaluate promotion eligibility for the child immediately. Matches the +# contract of complete_task and unblock_task; without this the child stays stuck in todo until the next +# dispatcher tick or a manual `hermes kanban recompute` (issue #22459). def parent_ids(conn: sqlite3.Connection, task_id: str) -> list[str]: return _linked_ids(conn, "parent_id", "child_id", task_id) @@ -1949,7 +1966,13 @@ def _has_sticky_block(conn: sqlite3.Connection, task_id: str) -> bool: """True when the newest ``blocked``/``unblocked`` event is ``blocked`` — an explicit ``kanban_block`` that must wait for an operator. A breaker trip emits ``gave_up`` (not ``blocked``) and so auto-recovers, as does a task - with no such event at all (direct DB edit).""" + with no such event at all (direct DB edit). + + See #28712. + Returns ``False`` when there is no such event at all (e.g. the task was set to ``status='blocked'`` by + the circuit breaker or by direct DB manipulation) — preserves the pre-#28712 auto-recover semantics for + that path. + """ row = conn.execute( "SELECT kind FROM task_events " "WHERE task_id = ? AND kind IN ('blocked', 'unblocked') " @@ -1996,6 +2019,9 @@ def recompute_ready(conn: sqlite3.Connection, failure_limit: int = None) -> int: ``consecutive_failures`` reached the limit (else the breaker could never trip). Limit order matches ``_record_task_failure``: ``max_retries`` > ``failure_limit`` > ``DEFAULT_FAILURE_LIMIT``. + + 1. The most recent block event was a worker-initiated ``kanban_block`` — those stay blocked until an + explicit ``kanban_unblock`` (#28712). """ if failure_limit is None: failure_limit = DEFAULT_FAILURE_LIMIT @@ -2052,6 +2078,8 @@ def recompute_ready(conn: sqlite3.Connection, failure_limit: int = None) -> int: def _parents_satisfied(conn: sqlite3.Connection, task_id: str) -> bool: """Return whether every direct parent is terminal for dependency gating.""" return conn.execute( + # Check if this task has children that still need the workspace. If any child is not yet + # done/archived, defer cleanup so the child can read handoff artifacts from the workspace (#33774). "SELECT 1 FROM task_links l " "JOIN tasks p ON p.id = l.parent_id " "WHERE l.child_id = ? " @@ -2260,6 +2288,18 @@ def release_stale_claims(conn: sqlite3.Connection, *, signal_fn=None) -> int: heartbeat) — unless ``last_heartbeat_at`` is older than ``DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS`` (wedged; ``_touch_activity`` keeps any genuinely active worker fresh). Safe to call often. + + Reclaiming a live worker mid-flight produces the spawn- then-immediately-reclaim loop seen on slow + models that spend longer than ``DEFAULT_CLAIM_TTL_SECONDS`` inside a single tool-free LLM call (#23025): + no tool calls means no ``kanban_heartbeat``, even though the subprocess is healthy. + Backstop (#29747 gap 3): if the worker's PID is still alive but its ``last_heartbeat_at`` is stale by + more than ``DEFAULT_CLAIM_HEARTBEAT_MAX_STALE_SECONDS`` (1h), the worker has been making no observable + progress and we reclaim anyway — even if ``_pid_alive`` is still true. This catches the + wedged-in-a-logic-loop case where the process is technically running but accomplishing nothing. + ``_touch_activity`` (run_agent.py) bridges chunk-level liveness into ``last_heartbeat_at`` via #31752, + so any genuinely active worker keeps its heartbeat fresh as a side effect of normal API traffic. + ``enforce_max_runtime`` and ``detect_crashed_workers`` remain the upper bounds for genuinely wedged or + dead workers. """ now = int(time.time()) reclaimed = 0 @@ -2621,6 +2661,11 @@ def _completed_event_payload( notifiers / dashboard WS render without a second round-trip; verified cards; and ``metadata["artifacts"]`` promoted so the notifier can upload them as native attachments without fetching the run row.""" + # Mirror CLI's _show_voice_status: include STT/TTS provider availability so the user can tell at a + # glance *why* voice mode isn't working ("STT provider: MISSING ..." is the common case). ``record_key`` + # mirrors the configured ``voice.record_key`` so the TUI can both bind it (frontend + # ``isVoiceToggleKey``) and display it in /voice status — previously the TUI hardcoded Ctrl+B and + # ignored the config (#18994). payload: dict = { "result_len": len(result) if result else 0, "summary": _first_line(event_summary, 400) or None, @@ -3872,6 +3917,10 @@ def _ctx_comments(lines: list[str], comments: list[Comment], now: int) -> None: if omitted_note: lines.append(omitted_note) for c in shown: + # Render author with explicit "comment from worker" framing so operator-controlled HERMES_PROFILE + # values like "hermes-system" or "operator" can't be misread by the next worker as a system + # directive above the (attacker-influenceable) comment body. Defense-in-depth — the LLM-controlled + # author-forgery surface was already closed in #22435. See #22452. safe_author = (c.author or "").replace("`", "") lines.append(f"comment from worker `{safe_author}` at {_ctx_stamp(c.created_at, now)}:") lines.append(_ctx_cap(c.body, _CTX_MAX_COMMENT_BYTES)) diff --git a/hermes_cli/kanban_db_connect.py b/hermes_cli/kanban_db_connect.py index 9817c2eaf5..048d534272 100644 --- a/hermes_cli/kanban_db_connect.py +++ b/hermes_cli/kanban_db_connect.py @@ -169,6 +169,13 @@ def _dispatch_tick_lock(db_path: Path): ``_guard_supervised_gateway_conflict``. Non-blocking on purpose: the gateway's async watcher must never stall; the loser retries next interval. Without ``fcntl``/``msvcrt`` it degrades to a no-op (yields ``True``). + + Motivation (issue #35240): a ``hermes gateway run --replace`` / ``gateway restart`` invoked from a shell + on a systemd/launchd host can leave an orphan gateway whose dispatcher escapes the service cgroup, + survives ``systemctl restart``, and becomes a *second* long-lived writer on the same ``kanban.db``. The + startup guard (``_guard_supervised_gateway_conflict``) blocks the common way an orphan is born, but this + lock is the defense-in-depth that prevents two dispatchers from ever writing concurrently *regardless of + how the second one got there*. """ lock_path = db_path.with_name(db_path.name + ".dispatch.lock") handle = None @@ -207,6 +214,8 @@ def _dispatch_tick_lock(db_path: Path): # PASSIVE never takes the exclusive checkpoint lock; WAL size is bounded by # ``journal_size_limit`` (set at connection init) on the writer's natural # post-checkpoint reset. Best-effort, keyed per resolved DB path. +# Once per coarse interval the dispatcher issues an explicit ``wal_checkpoint(PASSIVE)``. Best-effort: a +# busy/locked checkpoint is logged at DEBUG and retried next interval. See #44795, #45383, #80255. _WAL_CHECKPOINT_INTERVAL_SECONDS = 300.0 _LAST_WAL_CHECKPOINT: dict[str, float] = {} _WAL_CHECKPOINT_LOCK = threading.Lock() @@ -681,6 +690,8 @@ def connect(db_path: Optional[Path] = None, *, board: Optional[str] = None) -> s # rest of the process's life. Drop the stale entry and re-init. conn.close() with _INIT_LOCK: + # Drop the stale cache entry and fall through to the full init path, which re-runs the header + # and integrity probes and the schema script under the cross-process lock. See #83445. _INITIALIZED_PATHS.discard(resolved) _kb._log.warning( "kanban DB %s lost its schema after this process initialized it " @@ -691,6 +702,7 @@ def connect(db_path: Optional[Path] = None, *, board: Optional[str] = None) -> s with _kb._cross_process_init_lock(path): # Read-only file/sidecar preflight first, so a stray read-only kanban.db # fails actionably instead of "attempt to write a readonly database". + # See #12508. from hermes_state import preflight_db_writability preflight_db_writability(path, db_label=f"kanban.db ({path.name})") # Cheap byte-level header check before any sqlite connection, then the @@ -716,7 +728,10 @@ def connect_closing(db_path: Optional[Path] = None, *, board: Optional[str] = No """Open a kanban DB connection and guarantee it is closed on exit. Use instead of ``with kb.connect() as conn:`` — sqlite3's context manager only commits/rolls back, it does NOT close the fd, so long-lived processes - (gateway, dashboard) leak FDs until ``[Errno 24] Too many open files``.""" + (gateway, dashboard) leak FDs until ``[Errno 24] Too many open files``. + + See #33159 for the production incident. + """ conn = _kb.connect(db_path=db_path, board=board) try: yield conn @@ -943,6 +958,13 @@ def _backfill_legacy_inflight_runs(conn: sqlite3.Connection) -> None: # type, so drift requires a rebuild. Each entry pairs the canonical CREATE # TABLE with the indexes DROP TABLE takes down with it. # ``test_rebuilt_schema_matches_fresh`` guards this against SCHEMA_SQL drift. +# The current schema uses ``INTEGER PRIMARY KEY AUTOINCREMENT`` / ``INTEGER NOT NULL DEFAULT 0``. ``CREATE +# TABLE IF NOT EXISTS`` skips existing tables regardless of schema and ``_add_column_if_missing`` only adds +# columns, so neither can fix a drifted column type — the table must be rebuilt. See #35096. Each entry +# pairs the canonical CREATE TABLE with the CREATE INDEX statements that DROP TABLE would otherwise take +# down with it (including ``idx_events_run``, added by the additive pass above). To guard against this list +# drifting from SCHEMA_SQL, ``test_rebuilt_schema_matches_fresh`` asserts a rebuilt legacy DB is +# byte-identical to a fresh one. _REBUILD_SPECS = { "task_events": ( "CREATE TABLE task_events (" @@ -1013,6 +1035,9 @@ def _rebuild_drifted_tables(conn: sqlite3.Connection) -> None: the first post-migration tick replays history once — safe for a feature that was already fully broken. One transaction under ``connect()``'s init locks so an interruption can't leave a table half-renamed. Idempotent. + + Each affected table is rebuilt with the standard SQLite pattern — CREATE new → INSERT shared columns → + DROP old → RENAME — recreating its indexes too (DROP TABLE takes them down). See #35096. """ drifted = [t for t in _REBUILD_SPECS if _table_has_drifted(conn, t)] if not drifted: diff --git a/hermes_cli/kanban_db_dispatch.py b/hermes_cli/kanban_db_dispatch.py index bebd257421..eced13d741 100644 --- a/hermes_cli/kanban_db_dispatch.py +++ b/hermes_cli/kanban_db_dispatch.py @@ -76,7 +76,18 @@ _RESPAWN_GUARD_PR_URL_RE = re.compile( @dataclass class DispatchResult: - """Outcome of a single ``dispatch`` pass.""" + """Outcome of a single ``dispatch`` pass. + + ``kanban.default_assignee`` applied this tick before spawning (#27145). Surfaces the auto-assignment to + telemetry / CLI / dashboard so the operator can see when the dispatcher is acting on the fallback rule + ``kanban.max_in_progress_per_profile`` (#21582). Each entry is ``(task_id, assignee, + current_running_count)``. NOT an operator-actionable failure — the task will be picked up on a + subsequent tick when the assignee has capacity. Separate bucket so telemetry / dashboards can show "this + profile is busy" vs + the board's dispatch lock (issue #35240). A losing dispatcher does no DB writes this tick — the lock + holder is making progress on the same board. This is the steady-state signal that a single-writer guard + is + """ reclaimed: int = 0 promoted: int = 0 @@ -709,6 +720,10 @@ def _protocol_violation_streak(conn: sqlite3.Connection, task_id: str) -> int: _PROTOCOL_VIOLATION_ERROR = ( + # Worker subprocess returned 0 but its task is still ``running`` in the DB — it exited without calling + # ``kanban_complete`` / ``kanban_block``. Overwhelmingly the work itself succeeded and only the + # paperwork was skipped, so a retry usually completes; the corrective sentence below is surfaced to the + # retry worker via the prior-attempt error in ``build_worker_context`` (guidance approach from #61817). "worker exited cleanly (rc=0) without calling " "kanban_complete or kanban_block — protocol violation. " "If the prior run already did the work, verify it and " @@ -946,6 +961,22 @@ def detect_crashed_workers(conn: sqlite3.Connection) -> list[str]: for hook_fields in sweep.exited_hook_payloads: hook_fields = dict(hook_fields) _kb._fire_kanban_lifecycle_hook( + # Kanban worker-lifecycle, task-mutation, and dispatcher-tick observers (RFC #58548, + # accepted as the design basis in the #64231 batch disposition; on_kanban_dispatch_tick is + # the re-port of PR #56066). All five are observers only: return values are ignored, and + # every fire site is fully best-effort, so a broken callback can never break dispatch or a + # task mutation. Cost rule: every call site short-circuits on has_hook(), so when nothing + # subscribes no payload is built and the hot paths (each dispatcher tick, each task write) + # pay one dict probe. WHICH PROCESS: worker spawn/exit/stale-claim and the dispatch tick + # fire in the DISPATCHER process (gateway-embedded dispatcher or ``hermes kanban + # dispatch``); on_kanban_task_updated fires in whichever process committed the mutation + # (CLI, worker, or the gateway-embedded dashboard API). Common kwargs (task-scoped hooks): + # task_id: str, profile_name: str, board: str | None, assignee: str | None, run_id: int | + # None. on_kanban_worker_spawned fires after ``spawn_fn`` returns AND the worker PID (when + # one was reported) is durably persisted, per the RFC timing contract; like + # kanban_task_claimed it runs inside the board's dispatch lock, so callbacks must stay fast. + # Adds: worker_pid: int | None, workspace_path: str. Privacy: workspace_path is a filesystem + # path and may reveal project layout or usernames. "on_kanban_worker_exited", hook_fields.pop("task_id"), board=_board, @@ -1514,6 +1545,12 @@ def _dispatch_lane_task( if guard_reason is not None: result.respawn_guarded.append((task_id, guard_reason)) # Event so ``hermes kanban tail`` shows why the task looks stuck. + # Honour kanban.default_assignee: when the dispatcher hits an unassigned ready task and an + # operator-configured fallback exists, persist the assignment and proceed. This removes the + # dashboard footgun where a task created without an assignee parks in 'ready' forever even though + # the operator's intent ("default") was perfectly clear (#27145). Mutating the row (not just the + # in-memory view) keeps diagnostics and the board state consistent: the task is now legitimately + # owned by ``kanban.default_assignee``, not "unassigned but secretly routed". if not dry_run: with _kb.write_txn(conn): _kb._append_event(conn, task_id, "respawn_guarded", {"reason": guard_reason}) @@ -1720,6 +1757,9 @@ def _resolve_default_assignee(default_assignee: Optional[str]) -> Optional[str]: return name +# The dispatch lock has been released here. Fire the tick observer strictly OUTSIDE the single-writer +# critical section (#56066 sweeper finding / #64231 disposition): a slow subscriber must never extend the +# lock hold and stall a sibling dispatcher's tick. def _dispatch_once_locked( conn: sqlite3.Connection, *, @@ -1765,6 +1805,10 @@ def _dispatch_once_locked( # Per-profile cap. Deferred tasks go to skipped_per_profile_capped, not # skipped_unassigned — "busy, retry later" differs from "needs routing". per_profile_cap = max_in_progress_per_profile if ( + # Per-profile concurrency cap (#21582): when set, track how many workers each assignee already has + # in flight, and refuse to spawn when this would push that assignee past the cap. Prevents fan-out + # workloads from melting a single profile's local model / API quota / browser pool while leaving + # other profiles idle. isinstance(max_in_progress_per_profile, int) and max_in_progress_per_profile > 0 ) else None @@ -2179,6 +2223,13 @@ def _default_spawn(task: Task, workspace: str, *, board: Optional[str] = None) - # build_context_files_prompt; without it relative writes land in the gateway # user's home and workers load the gateway's AGENTS.md. file_tools rejects # relative / sentinel values, so only set a real absolute directory. + # Pin TERMINAL_CWD to the task's workspace so the worker's file tools and context-file loader anchor on + # the workspace, not whatever cwd the dispatching gateway happened to export. The worker subprocess is + # already launched with cwd=workspace, but TERMINAL_CWD takes precedence over the process cwd in both + # file_tools._resolve_base_dir (#41312 — relative write_file paths were landing in the gateway user's + # home) and build_context_files_prompt (#34619 — workers loaded the dispatching gateway's AGENTS.md + # instead of the task's). Setting it to the workspace fixes both: the workspace is where the task's work + # actually happens. if workspace and os.path.isabs(workspace) and os.path.isdir(workspace): env["TERMINAL_CWD"] = workspace if task.branch_name: diff --git a/hermes_cli/kanban_db_notify.py b/hermes_cli/kanban_db_notify.py index 755d6ca495..546dbecbfa 100644 --- a/hermes_cli/kanban_db_notify.py +++ b/hermes_cli/kanban_db_notify.py @@ -265,6 +265,14 @@ def purge_stale_done_notify_subs(conn: sqlite3.Connection, *, max_age_days: int abandoned (unlike ``backlog``/``ready``) so it reaps on the same clock. Age = latest event, else ``completed_at``, else ``created_at`` — any activity, including a reopen, exempts the sub. + + The notifier keeps subscriptions alive through ``done`` because a completed task can be reopened (review + corrections, continuation) and the reopened cycle must still notify its origin session. On boards that + never archive, that retention would otherwise accumulate subscription rows forever — each one scanned + every notifier tick. This GC bounds that: a task that has been ``done`` with no new events for the + retention window is treated as settled and its subscriptions are purged. ``blocked`` tasks + (circuit-breaker trips, dead workers) are reaped on the same clock — they are abandoned, not idle, + unlike a ``backlog``/``ready`` card that is merely waiting for pickup (#100955). """ try: days = int(max_age_days) diff --git a/hermes_cli/kanban_db_workspace.py b/hermes_cli/kanban_db_workspace.py index e38ee463a9..91c96f904f 100644 --- a/hermes_cli/kanban_db_workspace.py +++ b/hermes_cli/kanban_db_workspace.py @@ -104,6 +104,8 @@ def _is_managed_scratch_path(p: Path) -> bool: ``rmtree`` outside managed storage — a board ``default_workdir`` on a real source tree paired with ``workspace_kind='scratch'`` would otherwise make task completion delete user data. + + See #28818. """ return _managed_scratch_path_info(p)[0] @@ -123,6 +125,7 @@ def _cleanup_workspace(conn: sqlite3.Connection, task_id: str) -> None: if kind not in _REMOVABLE_KINDS or not path: # Not removable itself, but completing may still unblock a deferred # parent scratch cleanup (e.g. a 'dir' child of a scratch parent). + # See #33774. _try_cleanup_parent_workspaces(conn, task_id) return # Defer while any child is not yet terminal so it can still read @@ -146,6 +149,7 @@ def _cleanup_workspace(conn: sqlite3.Connection, task_id: str) -> None: # Containment guard: a board's ``default_workdir`` can pair # ``workspace_kind='scratch'`` with a user path pointing at a real # source tree; without this, completion would rmtree the user's data. + # See #28818. if _is_managed_scratch_path(wp): shutil.rmtree(wp, ignore_errors=True) _kb._log.debug("Removed scratch workspace: %s", wp) @@ -159,6 +163,8 @@ def _cleanup_workspace(conn: sqlite3.Connection, task_id: str) -> None: # Kill the owning worker's tmux session if it is now dead, then let any # parent whose children are all done run its deferred cleanup. _cleanup_worker_tmux(conn, task_id) + # After cleaning up this task's workspace, check if any parent tasks now have all children done — + # their deferred cleanup can proceed (#33774). _try_cleanup_parent_workspaces(conn, task_id) except Exception: pass # best-effort — never block completion @@ -213,7 +219,10 @@ def _cleanup_worktree_workspace( def _try_cleanup_parent_workspaces(conn: sqlite3.Connection, task_id: str) -> None: """Run the deferred cleanup of any parent scratch/worktree workspace whose children are now all done/archived/failed/cancelled (called after each - child completes).""" + child completes). + + See #33774. + """ try: parents = conn.execute( "SELECT parent_id FROM task_links WHERE child_id = ?", diff --git a/hermes_cli/kanban_diagnostics.py b/hermes_cli/kanban_diagnostics.py index f957491556..577339acb1 100644 --- a/hermes_cli/kanban_diagnostics.py +++ b/hermes_cli/kanban_diagnostics.py @@ -566,7 +566,12 @@ def _rule_block_unblock_cycling(task, events, runs, now, cfg) -> list[Diagnostic """>= cfg["block_cycle_threshold"] (default 3) blocked-after-unblocked cycles within cfg["block_cycle_window_seconds"] (default 24h). Complements ``_rule_stuck_in_blocked``, whose timer any unblock resets, so fast cyclers - are invisible to it.""" + are invisible to it. + + ``_rule_stuck_in_blocked`` resets its timer on any ``commented`` / ``unblocked`` event, so a task that + cycles every few minutes is invisible to it regardless of how many times it cycles (#29747 gap 1). This + rule complements that one by counting block→unblock cycles in a sliding window. + """ threshold = _positive_int(cfg.get("block_cycle_threshold"), 3) window_seconds = float(cfg.get("block_cycle_window_seconds", 24 * 3600)) cycle_cutoff = now - window_seconds diff --git a/hermes_cli/kanban_specify.py b/hermes_cli/kanban_specify.py index db2206ac7f..7ff7176f21 100644 --- a/hermes_cli/kanban_specify.py +++ b/hermes_cli/kanban_specify.py @@ -78,6 +78,8 @@ class SpecifyOutcome: def _truncate(text: str, limit: int) -> str: + # Stored history is untrusted for display — remove escape sequences and control chars so a recap line + # can't clear the screen / retitle the window when echoed to a terminal (openai/codex#31494 bug class). if len(text) <= limit: return text return text[: limit - 1] + "…" @@ -154,6 +156,9 @@ def _call_aux(verb: str, task_id: str, *, aux_task: str, system: str, user: str, log.debug("%s: auxiliary client import failed: %s", verb, exc) return None, "auxiliary client unavailable" try: + # Route through call_llm so auxiliary.triage_specifier.* config (provider/model/base_url, + # extra_body, reasoning_effort, retries) all apply — the direct-create path dropped extra_body + # (#35566). resp = call_llm( task=aux_task, messages=[{"role": "system", "content": system}, {"role": "user", "content": user}], diff --git a/hermes_cli/linux_desktop_entry.py b/hermes_cli/linux_desktop_entry.py index 5bdd833111..ebd8fcc3f7 100644 --- a/hermes_cli/linux_desktop_entry.py +++ b/hermes_cli/linux_desktop_entry.py @@ -49,6 +49,10 @@ def _running_interpreter() -> str: (uv, pyenv, conda). ``resolve()`` follows it out of the venv, and CPython discovers ``pyvenv.cfg`` from the *lexical* argv[0] — so a dereferenced path boots without the venv's site-packages. Keep the lexical form when any ancestor holds a ``pyvenv.cfg``. + + See #80547, #90292. + Idea credit: the lexical-preservation rule was independently proposed in #92516/#94115/#94544 and by + nosliwhtes' review of this PR; the pyvenv.cfg-detection refinement here keeps both properties. """ lexical = os.path.abspath(sys.executable) path = Path(lexical) @@ -67,6 +71,8 @@ def _can_import_hermes_cli(interpreter: Path) -> bool: the answer matches a cold desktop environment. Cached per process; an unprobeable interpreter (missing binary, spawn failure, timeout) is assumed capable and deliberately NOT cached, so one transient hiccup doesn't freeze the assumption for the session. + + Probe design per @nosliwhtes' isolated-mode capability check (#92122 lineage, commit 4150501f641). """ key = str(interpreter) if key in _probe_cache: @@ -98,6 +104,12 @@ def resolve_exec_command(project_root: Optional[Path] = None) -> str: if not _can_import_hermes_cli(Path(interpreter)): # Persisting an interpreter that can't import the CLI writes a dead entry (the DE spawns # Exec in a cold environment where exactly this import must succeed). + # The candidate interpreter cannot actually import hermes_cli.main (checked in isolated mode from a + # neutral cwd — so the probe can't be fooled by a checkout cwd or an inherited PYTHONPATH). Fall + # back to the module form under the RUNNING interpreter, which by definition has the CLI importable. + # Probe design follows the isolated-mode capability check proposed by @nosliwhtes (#92122 review + # lineage, commit 4150501f641) — cached here per-process so a desktop launch pays the subprocess + # cost at most once. interpreter = _running_interpreter_fallback() argv = [interpreter, "-m", "hermes_cli.main", "desktop"] if bin_path: @@ -106,6 +118,7 @@ def resolve_exec_command(project_root: Optional[Path] = None) -> str: # with `#!/usr/bin/env python3`) would die silently on the first third-party import under # Terminal=false — run it under the venv interpreter explicitly. prefix = [interpreter] if _needs_interpreter(resolved) else [] + # See #90292. argv = [*prefix, str(resolved), "desktop"] return " ".join(_quote_exec_arg(a) for a in argv) @@ -113,7 +126,11 @@ def resolve_exec_command(project_root: Optional[Path] = None) -> str: def _is_interpreter(candidate: Path) -> bool: """A python interpreter binary (``bin/python*``), not a launcher: strict basename match (rejects ``python3-config``, ``pythonw``) inside a bin/Scripts dir (rejects a stray script - named ``python`` elsewhere).""" + named ``python`` elsewhere). + + Regex approach proposed independently in 94051; kept here with the parent-dir guard so a script named + ``python`` outside a bin/Scripts tree is not misclassified. See #94051. + """ return bool(re.fullmatch(r"python[23]?(\d+)?(\.\d+)?", candidate.name.lower())) and ( candidate.parent.name in {"bin", "scripts"} ) @@ -156,6 +173,8 @@ def _resolve_hermes_bin_for_desktop_entry( the entry depend on how the previous launch happened (a bootstrap loop). Skip such candidates and fall through to PATH, then to the installer's known wrapper locations. ``resolve_fn`` is injectable for tests. + + See #90492. """ if resolve_fn is None: from hermes_cli.relaunch import resolve_hermes_bin as resolve_fn @@ -177,6 +196,13 @@ def _resolve_hermes_bin_for_desktop_entry( if primary and not _inside_checkout(primary, checkout_root, original_argv0): return primary + # A primary that is NOT checkout-internal and not the invoking interpreter is an external launcher (e.g. + # /opt/.../bin/hermes from another install method, or a venv console script). It must be evaluated + # BEFORE any known-location probing: probing first could silently switch the entry to a different + # installation (#94443 review case 3). + # Only reroute when argv[0] actually drove the resolution: re-run the resolver with argv[0] hidden and + # compare. If PATH yields nothing, keep the resolver's original answer (its fallback chain stays + # authoritative; #90492 semantics preserved). sys.argv[0] = "" try: rerouted = resolve_fn() @@ -330,6 +356,16 @@ def _shebang_escapes_running_env(shebang: str) -> bool: ``/bin``. ``env`` shebangs ALWAYS escape — ``env`` resolves through the DE's cold PATH, not the shell that installed the venv — except the rare ``env -S `` form, which is judged by its absolute target. + + Tokenizes the shebang (interpreter path plus any flags) and compares PATH COMPONENTS, never substrings: + ``/bin-extra/python`` is not inside ``/bin`` even though it starts with it + (sibling-directory confusion; independently surfaced in nosliwhtes' #92122 hardening ``b96427d0`` — + reimplemented here with two extensions). + The comparison uses the LEXICAL interpreter directory (abspath, not resolve()): on uv venvs the resolved + parent is the base interpreter's dir, which makes a valid ``.venv/bin/python`` shebang look foreign + (#94443 review case 1). Both sides use the SAME case operation (``.lower()``): interpreter paths + legitimately carry uppercase (conda env names, usernames, uv's ephemeral build dirs) and an asymmetric + compare would flag the venv's own console script as foreign. """ tokens = _shebang_tokens(shebang) if not tokens: @@ -538,6 +574,8 @@ def install_desktop_entry(project_root: Path) -> Optional[Path]: return entry_path # Atomic replace: an interrupted plain write leaves a zero-byte entry, which permanently # breaks the taskbar pin (nothing later rewrites a file that exists at the right path). + # The temp+rename dance in utils.atomic_write_text is the codebase's shared implementation — ported + # from #80547, which closed unmerged with this piece unlanded. from utils import atomic_write_text atomic_write_text(entry_path, contents, create_mode=0o755) diff --git a/hermes_cli/loops.py b/hermes_cli/loops.py index 0d7b98f49a..7242004d5a 100644 --- a/hermes_cli/loops.py +++ b/hermes_cli/loops.py @@ -242,7 +242,11 @@ def _meta_key(session_id: str) -> str: def _get_session_db() -> Optional[Any]: """The goals module's cached SessionDB, so goals/loops/heartbeats share one connection and - its off-loop bootstrap (a cold cache on the loop thread never runs ``SessionDB()`` inline).""" + its off-loop bootstrap (a cold cache on the loop thread never runs ``SessionDB()`` inline). + + The previous copy here did, which froze the loop for the init duration and dropped the first ``loop:*`` + write (the /goal bug class, #88965). + """ try: from hermes_cli.goals import _get_session_db as _goals_db except Exception as exc: # pragma: no cover @@ -322,6 +326,9 @@ def migrate_loop_to_session(old_session_id: str, new_session_id: str, *, reason: Context compression rotates ``session_id`` to a fresh child; without this the loop silently dies at the compaction boundary. + + Copies the loop onto the new session and archives the old row as ``cleared`` so exactly one active loop + row exists per logical conversation. See #33618. """ if not old_session_id or not new_session_id or old_session_id == new_session_id: return False diff --git a/hermes_cli/macos_tcc_anchor.py b/hermes_cli/macos_tcc_anchor.py index 75afd8e920..82327deb79 100644 --- a/hermes_cli/macos_tcc_anchor.py +++ b/hermes_cli/macos_tcc_anchor.py @@ -150,6 +150,8 @@ def _provision_libpython(venv_dir: Path, source_file: Path, *, refresh: bool = F Provision-if-present: a surplus hardlink on a statically-linked build is free; a missed detection is the only way the dylib-not-found crash returns. + + See #95425. """ src_lib = _store_root(source_file) / "lib" if not src_lib.is_dir(): @@ -319,6 +321,8 @@ def ensure_tcc_anchor(project_root: Path | None = None) -> Path | None: No-op (None) on non-macOS, without a venv interpreter, or when the interpreter is not uv-managed. Idempotent. Best-effort — None (and logs) if the copy or boot-gate fails; callers must never depend on success. + + See #95596. """ found = _managed_venv(project_root) if isinstance(found, str): diff --git a/hermes_cli/main.py b/hermes_cli/main.py index 8dcea5753a..9bd9fe36b5 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -43,6 +43,8 @@ from hermes_cli import _startup_fast # noqa: E402 # import; the marker lifecycle stays with the full recovery path. Its own # import is unguarded on purpose: same package dir, so if IT can't import # nothing in hermes_cli can. +# It is also the canonical home of the probe/repair tables reused by the full recovery path below. See +# #57828. from hermes_cli import _early_recovery as _early_recovery_mod try: @@ -82,6 +84,8 @@ def _exit_after_oneshot(rc: object) -> None: file logging, then ``os._exit`` past finalization. The ``atexit`` chain is deliberately skipped — several handlers re-enter native code that may be the abort source; stateful cleanup lives in ``_cleanup_oneshot_runtime``. + + See #30387, #43055. """ for stream in (sys.stdout, sys.stderr): try: @@ -171,6 +175,7 @@ def _run_and_exit_oneshot( finally: # Even an interrupt during cleanup must not fall back into interpreter # finalization, where the native SIGABRT occurs. + # The hard exit is the safety boundary for #43055. _exit_after_oneshot(rc) @@ -549,6 +554,12 @@ _apply_profile_override() # AFTER the profile override on purpose — no hermes module may import before # profiles resolve; the helper anchors on the DEFAULT root, so profile # sessions heal the same shared dir. +# That dir lives OUTSIDE the git checkout precisely because an earlier layout staged the copies at +# ``\bin``, where ``hermes update``'s autostash (``git stash push --include-untracked``) swept +# them off disk; with the desktop updater's ``--keep-stash`` nothing restored them and ``hermes`` stopped +# resolving in every new terminal (venv\Scripts itself must stay off PATH — it shadows the user's +# ``python``, #83797). Costs a few stat calls when healthy; gates fail toward inaction so source checkouts +# are untouched. if sys.platform == "win32": try: from hermes_cli import _install_repair as _install_repair_mod @@ -566,6 +577,9 @@ from hermes_cli.env_loader import load_hermes_dotenv # replaces the environment: on Windows Bitwarden's cryptography import maps # ``_rust.pyd`` and the parent updater then blocks its own child installer. # Profile flags are already stripped, so argv[1] is the authoritative subcommand. +# Profile flags have already been stripped above, so the first remaining argument is the authoritative +# argparse subcommand. Dotenv/managed config still loads; only external secret fetches are unnecessary for +# installation maintenance. See #73381. load_hermes_dotenv( project_env=PROJECT_ROOT / ".env", load_external_secrets=sys.argv[1:2] != ["update"], @@ -1044,6 +1058,8 @@ def _dotenv_has_provider_key(env_file: Path, provider_env_vars: set) -> bool: if line.startswith("#") or "=" not in line: continue if line.startswith("export "): + # Strip the bash-compatible ``export `` prefix so lines like ``export API_KEY=...`` parse as + # ``API_KEY`` rather than being stored under the wrong key ``"export API_KEY"`` (#6659). line = line[7:] key, _, val = line.partition("=") if key.strip() in provider_env_vars and val.strip().strip("'\""): @@ -1446,6 +1462,10 @@ def _create_titled_session(title: str) -> Optional[str]: Same timestamp+uuid id shape the CLI uses; the title is recorded with user provenance so auto-titling never overwrites it. + + Used by ``chat -c --create-if-missing`` (#86794): programmatic callers (plugins, scripts) that + want "send to this named thread, making it if needed" get a deterministic outcome instead of a silent + no-op. """ db = None try: @@ -1462,6 +1482,7 @@ def _create_titled_session(title: str) -> Optional[str]: # Programmatic callers rely on --create-if-missing being deterministic; # swallow the failure but log the cause so it lands in errors.log # (DB lock, I/O error, import error — all otherwise invisible). + # See #86794. logger.exception("Failed to create titled session %r", title) return None finally: @@ -1479,6 +1500,8 @@ def _resolve_continue_arg(args, *, use_tui: bool) -> None: 1) so programmatic callers see it even under quiet mode, or with ``--create-if-missing`` create a fresh titled session. Bare ``-c``: this terminal's breadcrumb session if valid, else the MRU session. + + Handles both forms: See #86794. """ continue_val = getattr(args, "continue_last", None) if continue_val and not getattr(args, "resume", None): @@ -1489,6 +1512,8 @@ def _resolve_continue_arg(args, *, use_tui: bool) -> None: elif getattr(args, "create_if_missing", False): # "send to this named thread, making it if needed" — without it # a quiet send to a not-yet-existing session silently no-ops. + # --create-if-missing: no session matches the title — create a new session with that title + # and proceed. See #86794. new_sid = _create_titled_session(continue_val) if new_sid: args.resume = new_sid @@ -2341,6 +2366,10 @@ def _clear_bytecode_cache(root: Path) -> int: def _finalize_update_receipt(code: int, reason: str) -> None: """Best-effort receipt close at the command boundary; no-op if already finalized.""" try: + # Receipt boundary (#91283 review): the impl has many early sys.exit paths (concurrent-instance + # preflight, venv-holder refusal, head-pinned no-op, fetch failure) that never reach an inner + # finalize. Persist any still-open receipt with the real exit code, then let the exit proceed + # unchanged. No-op when an inner path already finalized (exactly-once by construction). from hermes_cli.update_receipt import finalize_pending_update_receipt finalize_pending_update_receipt(code, reason) @@ -2360,6 +2389,8 @@ def _update_preflight_handled(args) -> bool: # docker/nix/apt refusal gates: on an image/package-managed install the # plan itself reports "not updatable in place" plus the right mechanism. if getattr(args, "plan", False): + # Read-only plan phase (#91277 Phase 2): inventory every running Hermes runtime across profiles, its + # supervisor, and its running code version — without mutating anything. Safe on a live fleet. from hermes_cli.update_inventory import ( collect_runtime_inventory, print_update_plan, @@ -2371,6 +2402,12 @@ def _update_preflight_handled(args) -> bool: # Image/package-managed admission gate: baked provenance marker first # (fail-closed on malformed), then docker/nix/apt heuristics. Records a # `refused` receipt and exits 2 (refused-by-contract, distinct from errors). + # Image-managed / package-managed admission gate (#91277 Phase 3): one shared decision for every + # mutation surface. Prints the real update command, records a `refused` receipt so fleet tooling sees + # the blocked attempt, and exits 2 (refused-by-contract, distinct from exit 1 errors). + # Shared admission gate (#91277 Phase 3): same marker-first decision as the apply path, so --check can + # never report git state for an install whose real update mechanism is an image pull. + # The response keeps the pre-existing per-kind error codes the dashboard UI already keys on. See #91277. from hermes_cli.update_contract import ( evaluate_update_admission, record_refusal_receipt, @@ -2444,6 +2481,9 @@ def cmd_update(args): # is durable. Every durable step is done by now, so on the hand-off # path only (marker env set solely by # _reexec_dependency_sync_off_windows_shim) flush and exit hard. + # By this point every durable step is done (receipt finalized above, lock released, stdio restored), + # so on the hand-off path only, flush and exit hard instead of waiting for the interpreter to unwind + # — the same treatment #79040's cron workaround applies. if _update_handoff_exit_code is not None and os.environ.get(_UPDATE_REEXEC_ENV) == "1": logger.debug( "Update hand-off child %s exiting via os._exit(%s)", @@ -2598,6 +2638,8 @@ def _dashboard_prepare_runtime(args, headless_backend) -> bool: # this those consumers saw an unset TERMINAL_ENV and ran every command on # the host even under `terminal.backend: docker` (#63141, #54449). try: + # PTY chat spawns already bridge their child env copy; this covers the in-process consumers. See + # #61115, #65696. from hermes_cli.config import apply_terminal_config_to_env apply_terminal_config_to_env() @@ -2803,6 +2845,10 @@ def _resolve_deferred_platform_cli_command(command_name: str | None) -> None: on import; ``discover_plugins()`` alone leaves ``hermes photon`` failing with ``invalid choice``. Importing just the matching platform keeps startup cheap. + + On the unknown-top-level-command slow path, ``discover_plugins()`` records the deferred loader but does + not import it, so the CLI registration never happens and ``hermes photon`` fails with argparse ``invalid + choice`` (issue #54678). """ if not command_name: return @@ -2865,6 +2911,7 @@ def _prepare_agent_startup(args) -> None: # below imports tools.approval, which freezes _YOLO_MODE_FROZEN at import. # main() sets it earlier too, but other launchers (Termux fast-CLI) reach # here directly, so the guarantee lives where the import is triggered. + # See #7994. if getattr(args, "yolo", False): os.environ["HERMES_YOLO_MODE"] = "1" _apply_safe_mode(args) @@ -3233,6 +3280,10 @@ def _advertise_agent_env() -> None: value must be our id in the public agent-harness registry (``hermes-agent``) — matching is exact. ``HERMES_AGENT`` is the Hermes-specific marker. setdefault: never clobber an outer harness. + + ``AI_AGENT`` is the emerging cross-agent standard (huggingface_hub's agent detection reads it; pi and + other agents set it — earendil-works/pi#7493) so generic tooling can attribute subprocesses to the + harness that spawned them. Hermes running inside another agent's terminal). """ os.environ.setdefault("AI_AGENT", "hermes-agent") os.environ.setdefault("HERMES_AGENT", "true") @@ -3271,6 +3322,7 @@ def _register_plugin_cli_commands(subparsers) -> None: discover_plugins() # The invoked platform may still be a deferred entry; import it so its # register_cli_command side effect runs before we read _cli_commands. + # See #54678. _resolve_deferred_platform_cli_command(_first_positional_argv()) for cmd_info in get_plugin_manager()._cli_commands.values(): if cmd_info["name"] not in seen_plugin_commands: @@ -3466,6 +3518,7 @@ def main(): # substring match is deliberately loose: over-matching (``hermes skills # install update``) only defers recovery one launch; under-matching # (``hermes -p work update``) would race. Never raises. + # See #95294. if "update" not in sys.argv[1:]: try: _recover_from_interrupted_install() diff --git a/hermes_cli/main_dashboard.py b/hermes_cli/main_dashboard.py index 90058f828f..14db9b4022 100644 --- a/hermes_cli/main_dashboard.py +++ b/hermes_cli/main_dashboard.py @@ -243,7 +243,10 @@ def _dashboard_cmdline_for_pid(pid: int) -> list[str] | None: def _respawn_dashboard_processes(commands: list[list[str]]) -> list[list[str]]: """Respawn manually-started dashboards after ``hermes update``, detached, logging to ``logs/dashboard-restart.log``; returns the argvs that failed to spawn. Callers pre-filter via - ``_filter_dashboard_respawn_candidates`` (no Desktop ``--port 0`` backends, capped per profile).""" + ``_filter_dashboard_respawn_candidates`` (no Desktop ``--port 0`` backends, capped per profile). + + See #78821. + """ from hermes_constants import get_hermes_home respawned: list[list[str]] = [] failed: list[tuple[list[str], str]] = [] @@ -383,6 +386,9 @@ def _report_dashboard_status() -> int: Serve-mode backends are INCLUDED: ``--stop`` kills them, so hiding them from ``--status`` let an operator kill what they couldn't see. + + Ledger-registered serves (profiled launches the argv scan can't match) surface via the spawn-ledger + augmentation in _scan_dashboard_processes. See #81564. """ from hermes_cli.main import _dashboard_listening, _self from gateway.status import _pid_exists @@ -551,6 +557,7 @@ def _read_ssh_session_token_file(path: str) -> str: # desktop-ssh, independent of HERMES_HOME and the active profile. Anchor # validation there, NOT get_hermes_home(): a non-default profile or a Docker # /opt/data root re-homes get_hermes_home() and would reject every token. + # See #69551. token_root = Path.home() / ".hermes" / "desktop-ssh" try: relative = Path(path).relative_to(token_root) @@ -726,6 +733,8 @@ def _resolve_dashboard_web_dist(args, _headless_backend: bool) -> None: sys.exit(1) elif skip_build: _dist_root = ( + # --build-mode skip trusts the caller to have pre-built the web UI. Verify the dist actually + # exists; otherwise the server will start and serve 404s with no obvious cause (issue #23817). Path(os.environ["HERMES_WEB_DIST"]) if "HERMES_WEB_DIST" in os.environ else PROJECT_ROOT / "hermes_cli" / "web_dist" @@ -734,6 +743,9 @@ def _resolve_dashboard_web_dist(args, _headless_backend: bool) -> None: # Only the default dist location is recoverable (desktop launches with # --build-mode skip after a wipe of web_dist); a custom HERMES_WEB_DIST # is a caller-managed directory the build cannot populate. + # The caller promised a pre-built dist but there isn't one. Instead of hard-failing (issue + # #59288 — desktop launches with --build-mode skip after a wipe of web_dist), warn and attempt + # ONE recovery build through the normal build path. _recoverable = "HERMES_WEB_DIST" not in os.environ if _recoverable: print(f"⚠ --skip-build was passed but no web dist found at: {_dist_root}") @@ -751,6 +763,10 @@ def _resolve_dashboard_web_dist(args, _headless_backend: bool) -> None: else: # HERMES_WEB_DIST without --skip-build: the env var points at a # caller-managed dist, so validate it like the --skip-build branch. + # HERMES_WEB_DIST is set without --skip-build: the build is skipped (the env var points at a + # caller-managed dist), so validate it the same way the --skip-build branch does — otherwise the + # server starts and serves 404s with no obvious cause (same failure mode as #23817, via the env-var + # path). _dist_root = Path(os.environ["HERMES_WEB_DIST"]).expanduser() if not (_dist_root / "index.html").exists(): print(f"✗ HERMES_WEB_DIST is set but no web dist found at: {_dist_root}") diff --git a/hermes_cli/main_desktop.py b/hermes_cli/main_desktop.py index 25367944a9..1796cf525a 100644 --- a/hermes_cli/main_desktop.py +++ b/hermes_cli/main_desktop.py @@ -133,7 +133,11 @@ def _desktop_packaged_executable(desktop_dir: Path) -> Optional[Path]: def _desktop_packaged_executable_in(release_dir: Path) -> Optional[Path]: - """The unpacked Electron app executable under *release_dir* (live ``release`` or a staging dir).""" + """The unpacked Electron app executable under *release_dir* (live ``release`` or a staging dir). + + *release_dir* is electron-builder's ``directories.output`` — the live ``apps/desktop/release`` or a + stage-and-swap staging dir (#86443). + """ if sys.platform == "darwin": candidates = list(release_dir.glob("mac*/Hermes.app/Contents/MacOS/Hermes")) elif sys.platform == "win32": @@ -152,6 +156,10 @@ def _desktop_packaged_executable_in(release_dir: Path) -> Optional[Path]: # A stale win-arm64-unpacked next to the real win-unpacked: picking by # mtime can hand a wrong-architecture Hermes.exe to the launcher. Prefer # candidates whose PE machine matches the host; mtime when none parse. + # Multiple unpacked trees can coexist (e.g. a stale win-arm64-unpacked left behind by a cross-arch + # experiment next to the real win-unpacked). Picking purely by mtime can then hand a + # wrong-architecture Hermes.exe to the launcher, which Windows rejects with "This app can't run on + # your computer" (#69179). expected = _expected_windows_pe_machines() matching = [p for p in existing if _pe_machine_or_none(p) in expected] if matching: @@ -159,6 +167,13 @@ def _desktop_packaged_executable_in(release_dir: Path) -> Optional[Path]: return max(existing, key=lambda p: p.stat().st_mtime) +# ─── Desktop stage-and-swap pack (#86443) ─────────────────────────────────── electron-builder packs IN +# PLACE: before-pack.mjs wipes ``release/<platform>- unpacked`` (or the mac ``Hermes.app``) and the Electron +# unpack + asar + rename then rebuild it. Any failure after that wipe — corrupt cached zip, blocked +# download, missing dep, disk full — leaves the user with NO app, and ``hermes update`` used to report +# "partially complete" over an empty release/. Fix the class, not the predicate: build into a STAGING output +# dir next to release/, verify the staged result, and only then swap it over the live tree with renames. On +# any failure the live app is untouched. _DESKTOP_STAGING_PREFIX = ".staging-" _DESKTOP_PREVIOUS_SUFFIX = ".previous" @@ -219,6 +234,15 @@ def _discard_desktop_staging(staging_dir: Path) -> None: shutil.rmtree(staging_dir, ignore_errors=True) +# ─── Desktop exe integrity gate (#69179) ──────────────────────────────────── The desktop self-update chain +# (Desktop → hermes-setup --update → `hermes update` → `hermes desktop --build-only` → relaunch) rebuilds +# Hermes.exe on the end user's machine and used to verify only that the file EXISTS before declaring +# success. A corrupt cached Electron zip whose extraction produced a truncated electron.exe, an interrupted +# rcedit resource rewrite, a disk-full pack, or a wrong-arch unpacked tree therefore shipped a broken binary +# that Windows refuses to load ("This app can't run on your computer" / 此应用无法在你的电脑上运行). These helpers parse +# the PE header — no signature infrastructure required — so a structurally broken or wrong-architecture +# Hermes.exe is caught BEFORE the updater replaces the working app, and the previous build can be restored +# from the .bak tree that apps/desktop/scripts/before-pack.mjs now preserves. _PE_MACHINE_I386 = 0x014C _PE_MACHINE_AMD64 = 0x8664 _PE_MACHINE_ARM64 = 0xAA64 @@ -241,7 +265,15 @@ def _kernel32(): def _windows_native_machine_from_iswow64() -> Optional[str]: """IsWow64Process2's OS-native machine, or None. HANDLE types are bound explicitly: ctypes' - default ``c_int`` truncates the ``(HANDLE)-1`` pseudo-handle → ``ERROR_INVALID_HANDLE`` on Win64.""" + default ``c_int`` truncates the ``(HANDLE)-1`` pseudo-handle → ``ERROR_INVALID_HANDLE`` on Win64. + + ctypes defaults ``GetCurrentProcess``'s restype to ``c_int``, so the current-process pseudo-handle + ``(HANDLE)-1`` is truncated to ``0xFFFFFFFF`` and zero-extended into a 64-bit invalid handle. On Win64 + that makes ``IsWow64Process2`` fail with ``ERROR_INVALID_HANDLE`` (6), which is exactly the residual + Windows-on-ARM failure after #71218: the gate fell through to ``PROCESSOR_ARCHITECTURE=AMD64`` (the + emulated process arch) and rejected a correctly-built ARM64 ``Hermes.exe``. Binding + ``restype``/``argtypes`` to ``wintypes.HANDLE`` keeps the full ``0xFFFFFFFFFFFFFFFF`` pseudo-handle. + """ import ctypes from ctypes import wintypes kernel32 = _kernel32() @@ -283,7 +315,15 @@ def _windows_native_machine() -> str: """The Windows host's NATIVE machine, upper-cased: ``IsWow64Process2`` (the only API that tells the truth from an emulated x64 process on ARM64), then ``PROCESSOR_ARCHITEW6432`` / ``PROCESSOR_ARCHITECTURE``, then ``platform.machine()`` (which lies under emulation). - ``GetNativeSystemInfo`` is NOT used: it also returns emulated details.""" + ``GetNativeSystemInfo`` is NOT used: it also returns emulated details. + + ``platform.machine()`` reports the PROCESS architecture, which lies under emulation: the desktop update + chain runs an x64 hermes-setup.exe (and thus x64 Python) on Windows-on-ARM devices, where + ``platform.machine()`` returns ``AMD64`` even though the OS is ARM64. The #71119 integrity gate then + rejected the CORRECT ARM64 rebuild as an "architecture mismatch" (#69179 follow-up report). Probe order: + 1. ``IsWow64Process2`` with a correctly-typed current-process HANDLE (#71218 + HANDLE-truncation fix). + 2. 3. + """ if sys.platform == "win32": try: name = _windows_native_machine_from_iswow64() @@ -417,7 +457,10 @@ def _rollback_desktop_from_backup(packaged_executable: Path) -> Optional[Path]: def _ensure_desktop_exe_launchable(desktop_dir: Path, packaged_executable: Optional[Path]) -> tuple: """Windows post-build integrity gate → ``(verified_exe_or_None, rolled_back)``: pass → ``(exe, False)``; corrupt with backup restored → ``(old_exe, True)``; nothing restorable → - ``(None, False)``. Failure purges the cached zip + stamp so the retry re-downloads.""" + ``(None, False)``. Failure purges the cached zip + stamp so the retry re-downloads. + + See #69179. + """ from hermes_cli.main import _desktop_stamp_path, _purge_electron_build_cache if packaged_executable is None or sys.platform != "win32": return packaged_executable, False @@ -430,6 +473,9 @@ def _ensure_desktop_exe_launchable(desktop_dir: Path, packaged_executable: Optio # Only the exe's OWN output dir is purged (a staging dir), never the live # release/ tree that still holds the last working app. + # Self-heal setup for the retry: drop the (likely corrupt) cached Electron zip and the content stamp so + # the next rebuild is a genuine re-download + re-stage rather than a replay of the same broken + # extraction. See #86443. _purge_electron_build_cache(desktop_dir, release_dir=packaged_executable.parent.parent) with contextlib.suppress(OSError): _desktop_stamp_path().unlink() @@ -489,6 +535,11 @@ def _purge_electron_build_cache(desktop_dir: Path, release_dir: Optional[Path] = zip_path.unlink() removed.append(zip_path) + # Drop the half-written unpacked dir too: an interrupted prior pack leaves a partial tree that poisons + # the rename even after the zip is fixed. (before-pack.cjs also handles this, but clearing it here makes + # the retry robust even if the hook is somehow skipped.) ``release_dir`` lets a stage-and-swap caller + # point this at its STAGING output so a mid-retry purge never touches the live app under ``release/`` + # (#86443). if release_dir is None: release_dir = desktop_dir / "release" if release_dir.is_dir(): @@ -502,6 +553,7 @@ def _purge_electron_build_cache(desktop_dir: Path, release_dir: Optional[Path] = # Last-resort Electron mirror after GitHub download fails. Only used when the # user hasn't pinned ELECTRON_MIRROR. +# See #47266. _ELECTRON_FALLBACK_MIRROR = "https://npmmirror.com/mirrors/electron/" @@ -515,7 +567,12 @@ def _electron_dir(project_root: Path) -> Path: def _electron_dist_binary(project_root: Path) -> Path: - """The Electron main binary inside the installed package — the exact file ``electronDist`` needs.""" + """The Electron main binary inside the installed package — the exact file ``electronDist`` needs. + + electron-builder reads the binary from ``build.electronDist`` since #38673, so this is the exact file + whose absence makes a pack fail with "The specified electronDist does not exist". The basename differs + per OS (the platform Electron is named for the host the build runs on). + """ dist = _electron_dir(project_root) / "dist" if sys.platform == "darwin": return dist / "Electron.app" / "Contents" / "MacOS" / "Electron" @@ -810,6 +867,8 @@ def _desktop_macos_relaunchable_fixup( os.environ.get("CSC_LINK") or os.environ.get("APPLE_SIGNING_IDENTITY")) if publisher_signing_configured: return True + # ``release_dir`` (stage-and-swap, #86443): sign the STAGED bundle before it is promoted, so the live + # app is never touched mid-sign. exe = _desktop_packaged_executable_in(release_dir or (desktop_dir / "release")) if exe is None: return True @@ -876,6 +935,7 @@ def _macos_create_signing_identity( # with "MAC verification failed". `-legacy` restores the accepted # RC2/SHA-1 format but only exists on OpenSSL 3 — so try plain first and # fall back to `-legacy` when the IMPORT fails with that signature. + # (Verified E2E on macOS 26.3.1 / OpenSSL 3.6.3 by @ctaylor86 on PR #77189.) def _export_p12(extra_args: list) -> None: subprocess.run( [ @@ -1235,6 +1295,12 @@ def _run_desktop_pack_with_recovery( build_result = _pack(npm_build_env) if build_result.returncode != 0 and staging_dir is not None and _staged_exe() is None: + # Corrupt cached Electron zip → partial unpack → ENOENT on rename. stdlib zipfile won't catch the + # common concat-junk case, so purge and retry once; @electron/get SHASUM is the real gate. Gate on a + # MISSING packaged executable: that is the signature of the corrupt-download class this recovery + # exists for. A late failure such as macOS code signing leaves the executable in place — + # redownloading Electron can't repair it, so the purge + retry would only add another slow, + # identical failure (#40187). purged: list[Path] = [] restored = False if not _electron_dist_ok(PROJECT_ROOT): @@ -1308,6 +1374,7 @@ def _build_desktop_app(desktop_dir: Path, *, source_mode: bool, npm: str, env: d # release/<unpacked> first, so a pack that fails afterwards used to leave # the user with NO app. Build into a staging dir; the live release/ tree is # only replaced — by rename — after the staged result verifies. + # See #86443. staging_dir: Optional[Path] = None build_cmd = [npm, "run", build_script] if not source_mode: diff --git a/hermes_cli/main_install_repair.py b/hermes_cli/main_install_repair.py index f8f3d45235..7cccbc589a 100644 --- a/hermes_cli/main_install_repair.py +++ b/hermes_cli/main_install_repair.py @@ -90,6 +90,7 @@ def _load_installable_optional_extras(group: str = "all") -> list[str]: # ``.lazy-refresh-incomplete`` — lazy-backend refresh may have corrupted packages; # cleared only after import-probe repair confirms healthy (never on indeterminate). # Narrow lazy probes must NEVER clear the generic core marker. +# See #58004. def _update_marker_path() -> Path: from hermes_cli.main import PROJECT_ROOT return PROJECT_ROOT / ".update-incomplete" @@ -251,6 +252,8 @@ def _recover_core_update_marker_locked() -> None: # Windows: a ``hermes.exe`` launch has the launcher as an ancestor; the quarantined full # reinstall can still replace it. Package-only repair is first aid and NEVER clears the marker. + # Full editable reinstall uses quarantine so the live shim can still be replaced. Package-only import + # repair may help as first aid but must NEVER clear this core marker on its own (#58004 review). self_locked = _windows_running_hermes_launcher_locked() if self_locked: install_prefix, install_env = _default_venv_install_target() @@ -306,7 +309,10 @@ def _windows_shim_in_process_chain() -> Path | None: process lifetime, so an editable install run from one can never rewrite it. Two probes, since either can come up empty: own launch paths (argv[0], ``__main__`` file/spec origin — runpy/ zipapp puts ``<shim>\\__main__.py`` there) and psutil ancestry. Candidates are intersected - with the project venv's own shims so a foreign ``hermes.exe`` never matches.""" + with the project venv's own shims so a foreign ``hermes.exe`` never matches. + + See #88838, #89599. + """ from hermes_cli.main import _hermes_exe_shims, _is_windows, _venv_scripts_dir if not _is_windows(): return None @@ -370,7 +376,18 @@ def _reexec_dependency_sync_off_windows_shim() -> bool: ``.update_exit_code``. The child re-runs ``hermes update`` so the sync and its tail happen exactly once; ``_UPDATE_REEXEC_ENV`` stops it spawning again and stops the "already up to date" early return from swallowing the sync. ``.update-incomplete`` is already written, so - a child that dies mid-install is finished by the next launch's recovery.""" + a child that dies mid-install is finished by the next launch's recovery. + + Called at the dependency-sync boundary, NOT at the top of the command — the same placement rule as the + native-module deferral beside it, and for the same reason (#86735): a hand-off that fires before the + fetch detaches every run, including the ``Already up to date!`` no-op that never touches the venv at + all, and it takes the interactive prompts with it. By the time we reach here the code swap is done and + every question — stash, branch switch, config migration — has already been asked and answered in the + user's own console. + ``venv\\Scripts\\hermes.exe`` is a launcher that runs the interpreter with the shim as its script and + holds it open without ``FILE_SHARE_DELETE`` for the whole command, so the quarantine rename is refused + and uv fails to replace it with os error 32 (#88838, #89599). + """ from hermes_cli.main import _UPDATE_REEXEC_ENV, _windows_shim_in_process_chain if os.environ.get(_UPDATE_REEXEC_ENV) == "1": return False @@ -523,6 +540,8 @@ def _quarantine_running_hermes_exe( naming the likely culprit. Returns ``(original, quarantined)`` pairs for rollback; ``failed_out`` collects shims whose rename failed every attempt so the update dependency sync can refuse instead of stranding a half-broken venv. + + See #87331. """ from hermes_cli.main import _hermes_exe_shims, _is_windows moved: list[tuple[Path, Path]] = [] @@ -616,14 +635,21 @@ def _cleanup_pending_shim_renames(scripts_dir: Path) -> int: def _restore_quarantined_exes(moved: list[tuple[Path, Path]]) -> None: """Roll back ``_quarantine_running_hermes_exe`` if uv didn't write replacements. Safety- critical: a failed quarantine only aborts an update; a failed restore leaves no ``hermes`` - on PATH. Delegates to the stdlib-only retrying helper shared with ``_install_repair``.""" + on PATH. Delegates to the stdlib-only retrying helper shared with ``_install_repair``. + + The outbound rename already retries a lock, so this one must too rather than swallow the first + ``OSError`` in silence. See #75584. + """ _early_recovery_mod.restore_quarantined_shims(moved) class ShimQuarantineError(RuntimeError): """A live ``hermes*.exe`` shim could not be renamed aside. Raised by :func:`_run_quarantined_install` in ``strict_quarantine`` mode BEFORE the install runs: a - process holds the venv hard enough that the sync would die partway — refuse, don't warn.""" + process holds the venv hard enough that the sync would die partway — refuse, don't warn. + + See #87331. + """ def __init__(self, failed_shims: list[str]): self.failed_shims = list(failed_shims) @@ -642,6 +668,8 @@ def _run_quarantined_install( retry proves a hard venv hold — the install WILL hit the same lock on .pyd files — so roll back and raise :class:`ShimQuarantineError` without installing. Non-strict callers already mutated the venv, so refusing buys nothing. ``scripts_dir is None`` is a pass-through. + + See #87331. """ from hermes_cli.main import ShimQuarantineError, _quarantine_running_hermes_exe, _restore_quarantined_exes, _run_install_with_heartbeat moved: list[tuple[Path, Path]] = [] @@ -662,6 +690,12 @@ def _run_quarantined_install( # A quarantine file younger than this may belong to an update running RIGHT NOW in # another process, whose restore step still needs it — the only copy of that shim. +# Restore shims when the installer didn't write replacements — on FAILURE (install died before the +# entry-points step) and on SUCCESS too: uv audits an already-satisfied editable install as a no-op and +# rewrites no entry points, which would otherwise leave the shims quarantined aside and `hermes` missing +# from PATH after a green install (#75584). _restore_quarantined_exes skips any shim the installer actually +# replaced, so this never clobbers fresh output. Errors are not swallowed — the finally re-raises whatever +# escaped. _QUARANTINE_GRACE_SECONDS = 15 * 60 @@ -684,6 +718,8 @@ def _cleanup_quarantined_exes(scripts_dir: Path | None = None) -> None: the same retry-and-report helper the update-time restore uses. (2) concurrency — a fresh quarantine file may belong to an update in flight in another process; leave anything inside the grace window alone. Silent no-op on non-Windows, nothing to do, or locked/permission errors. + + Deleting it converts a one-rename recovery into a full reinstall. See #75584. """ from hermes_cli.main import _QUARANTINE_GRACE_SECONDS, _cleanup_pending_shim_renames, _is_windows, _quarantine_stamp_ms, _venv_scripts_dir if not _is_windows(): @@ -718,6 +754,7 @@ def _cleanup_quarantined_exes(scripts_dir: Path | None = None) -> None: # Import probes for venv corruption after a failed lazy ``uv pip install`` (metadata can # look fine while ``.py`` files were removed mid-install). Canonical tables live in the # stdlib-only ``_early_recovery`` module so the early and full recovery layers never drift. +# See #57828. _LAZY_REFRESH_IMPORT_PROBES: tuple[tuple[str, str], ...] = ( _early_recovery_mod.LAZY_REFRESH_IMPORT_PROBES) _LAZY_REFRESH_REPAIR_PACKAGES: dict[str, str] = _early_recovery_mod.LAZY_REFRESH_REPAIR_PACKAGES @@ -725,7 +762,10 @@ _LAZY_REFRESH_REPAIR_PACKAGES: dict[str, str] = _early_recovery_mod.LAZY_REFRESH def _run_package_only_install(cmd: list[str], *, env: dict[str, str] | None = None) -> None: """Package-only pip/uv install — no shim quarantine: ``--force-reinstall <pkg>`` never rewrites - ``hermes.exe``, and the quarantine path would rename shims uv then never recreates.""" + ``hermes.exe``, and the quarantine path would rename shims uv then never recreates. + + See #57828. + """ from hermes_cli.main import _run_install_with_heartbeat _run_install_with_heartbeat(cmd, env=env) @@ -829,6 +869,8 @@ def _repair_venv_via_import_probes( but ``.py`` files were wiped mid-install. Package-only reinstall — never rewrites ``hermes.exe``. Never raises. Returns ``"healthy"``, ``"repaired"``, ``"failed"`` (repair did not confirm clean) or ``"indeterminate"`` (probes could not run; NOT healthy). + + See #57828. """ from hermes_cli.main import _detect_broken_lazy_refresh_imports, _repair_broken_lazy_refresh_imports broken = _detect_broken_lazy_refresh_imports(install_cmd_prefix, env=env) @@ -873,7 +915,10 @@ def _insert_python_pin(args: list[str]) -> list[str]: def _interpreter_scripts_dir() -> Path | None: """Scripts/bin dir of ``sys.executable``: on a site-packages install ``PROJECT_ROOT/venv`` does not exist and the shims uv rewrites live next to the interpreter. Layout via the - canonical ``venv_bin_dir`` (hand-rolling Scripts/bin is lint-tested against).""" + canonical ``venv_bin_dir`` (hand-rolling Scripts/bin is lint-tested against). + + See #76105. + """ from hermes_cli.main import _is_windows from hermes_constants import venv_bin_dir exe = Path(sys.executable) @@ -894,6 +939,10 @@ def _install_python_dependencies_with_optional_fallback( ``VIRTUAL_ENV`` that does not exist (pip / site-packages install), ``uv pip`` fails with ``Failed to inspect Python interpreter from active virtual environment`` before doing any work — pin the install at the running interpreter instead. + + Pin the install at the running interpreter instead so the update/recovery path succeeds on those + installs (#71510 fixed the ZIP path, #83335 fixed lazy-deps; this closes the shared helper for the + remaining callers). """ from hermes_cli.main import _insert_python_pin, _interpreter_scripts_dir, _is_windows, _load_installable_optional_extras, _run_quarantined_install, _venv_scripts_dir, _verify_console_scripts_installed, _verify_core_dependencies_installed scripts_dir = _venv_scripts_dir() if _is_windows() else None @@ -915,6 +964,8 @@ def _install_python_dependencies_with_optional_fallback( args = _insert_python_pin(args) # strict_quarantine: this is the UPDATE dependency sync; ShimQuarantineError propagates # to the sync boundary, which defers via the update-incomplete marker instead. + # A shim that cannot be renamed aside proves a hard venv hold; running uv anyway is how installs + # strand half-updated (#87331). _run_quarantined_install( install_cmd_prefix + args, env=env, scripts_dir=scripts_dir, strict_quarantine=True) @@ -959,7 +1010,11 @@ def _verify_console_scripts_installed( On Windows ``uv pip install -e .`` can register ``hermes.exe`` in the wheel RECORD while the file never lands (live shim locked, launcher write skipped), so ``hermes`` drops off PATH - after a "successful" install. Missing shims get ``--reinstall -e .`` under quarantine.""" + after a "successful" install. Missing shims get ``--reinstall -e .`` under quarantine. + + The symptom is ``hermes-agent.exe`` and ``hermes-acp.exe`` present but ``hermes.exe`` missing, so + ``hermes`` drops off PATH even though the install reported success (issue #52931). + """ from hermes_cli.main import _is_windows, _run_quarantined_install, _venv_scripts_dir if not _is_windows(): return @@ -1120,7 +1175,10 @@ def _resolve_node_runtime_npm() -> str | None: On WSL, PATH interop can hand back a Windows npm that fails with EISDIR / symlink errors over ``\\\\wsl.localhost\\...`` UNC paths. Refuse it on a POSIX host and re-scan PATH minus the - ``/mnt/*`` drive mounts. ``None`` when no suitable npm is reachable.""" + ``/mnt/*`` drive mounts. ``None`` when no suitable npm is reachable. + + On WSL/Linux ``shutil.which("npm")`` may resolve a Windows npm exposed through PATH interop. See #30271. + """ from hermes_cli.main import _is_windows from hermes_constants import find_node_executable npm = find_node_executable("npm") diff --git a/hermes_cli/main_provider_setup.py b/hermes_cli/main_provider_setup.py index 08f4cad61b..06ad533e51 100644 --- a/hermes_cli/main_provider_setup.py +++ b/hermes_cli/main_provider_setup.py @@ -432,7 +432,10 @@ def _save_custom_provider(base_url, api_key="", model="", context_length=None, n key_env=""): """Save a custom endpoint to ``custom_providers`` in config.yaml, deduplicated by base_url (an existing entry gets model / context_length / api_mode updated). *key_env* set means the caller - already wrote the key to ``.env``; the entry references it instead of inlining the secret.""" + already wrote the key to ``.env``; the entry references it instead of inlining the secret. + + See #69449. + """ from hermes_cli.config import load_config, save_config cfg = load_config() providers = cfg.get("custom_providers") or [] diff --git a/hermes_cli/main_tui_launch.py b/hermes_cli/main_tui_launch.py index 8c0e63601e..f7f0c78358 100644 --- a/hermes_cli/main_tui_launch.py +++ b/hermes_cli/main_tui_launch.py @@ -140,6 +140,11 @@ def _npm_lock_workspace_closure(packages: dict, starts) -> Optional[set]: every OTHER workspace's deps (``apps/desktop``, ``web``) as missing and reinstall on every launch. Names resolve by walking up ``node_modules`` ancestors; ``link: true`` entries are followed to their real package. + + The launch install is scoped with ``npm install --workspace ui-tui`` (see ``_make_tui_argv``), so only + the ui-tui workspace's dependency closure is written to the hidden ``.package-lock.json``. On Termux it + additionally selects ui-tui's child ``packages/*`` workspaces, so their devDependencies join the closure + too. See #66978. """ start_set = {starts} if isinstance(starts, str) else {s for s in starts if s} present = [s for s in start_set if s in packages] @@ -239,6 +244,8 @@ def _tui_need_npm_install(root: Path) -> bool: # Shared workspace checkout: the launch install is scoped to ui-tui (+ child # packages on Termux), so limit the comparison to that closure. Standalone / # own-lockfile layouts do a full install and keep the full comparison. + # Limit the comparison to the same selected-workspace closure so unrelated workspace deps (apps/desktop, + # web, …) don't force a reinstall every launch (#66978). closure: Optional[set] = None if ws_root != root: selected = _tui_selected_workspace_keys(root, ws_root) @@ -251,6 +258,10 @@ def _tui_need_npm_install(root: Path) -> bool: if name not in installed: # Workspace link entries are never materialized by a partial # `npm install --workspace ui-tui`; don't force a reinstall for them. + # Workspace link entries (`"link": true`, paths outside node_modules/ like `apps/desktop`, + # `node_modules/web`) are never materialized by a partial `npm install --workspace ui-tui` — + # they're deliberately skipped (see #38772) and would otherwise force a reinstall on every + # launch. if pkg.get("optional") or pkg.get("peer") or pkg.get("link"): continue if not name.startswith("node_modules/"): @@ -362,7 +373,14 @@ def _find_bundled_tui(hermes_cli_dir: Path | None = None) -> Path | None: def _restore_tui_workspace(tui_dir: Path) -> bool: """Best-effort ``git restore`` of a missing ``ui-tui/`` (Windows AV/NTFS filters can delete - tracked files after ``hermes update``); True when the directory exists afterwards.""" + tracked files after ``hermes update``); True when the directory exists afterwards. + + On Windows an antivirus / NTFS filter driver can leave tracked ``ui-tui/`` files deleted in the working + tree after ``hermes update`` (HEAD stays intact; the files just vanish — see issue #49145). Those files + are tracked, so ``git restore`` puts them back deterministically. Best-effort: returns False (rather + than raising) when git is unavailable, this isn't a checkout, or the restore leaves the directory still + missing — the caller then prints the manual-recovery message. + """ git = shutil.which("git") if not git or not (tui_dir.parent / ".git").exists(): return False @@ -377,7 +395,13 @@ def _restore_tui_workspace(tui_dir: Path) -> bool: def _ensure_tui_workspace(tui_dir: Path) -> None: """Ensure ``ui-tui/`` exists before it is used as a subprocess cwd (else ``NotADirectoryError`` - / ``WinError 267`` with no usable message): git-restore first, then abort with recovery steps.""" + / ``WinError 267`` with no usable message): git-restore first, then abort with recovery steps. + + Without this, a missing workspace falls through to ``subprocess.run(..., cwd=<missing ui-tui>)``, which + crashes with ``NotADirectoryError`` (``WinError 267`` on Windows) instead of a usable message (#49145). + We first try to self-heal via ``git restore``; only if that can't recover the directory do we abort with + concrete manual-recovery steps. + """ if tui_dir.is_dir(): return @@ -460,6 +484,10 @@ def _install_tui_dependencies(tui_dir: Path, *, termux_startup: bool) -> None: if not os.environ.get("HERMES_QUIET"): print("Installing TUI dependencies…") npm_cwd = _workspace_root(tui_dir) + # --workspace ui-tui avoids resolving apps/desktop (Electron + node-pty). See #38772. When ui-tui/ has + # its own package-lock.json (e.g. curl install), _workspace_root() returns tui_dir itself. Passing + # --workspace in that case fails because npm cannot find a workspace named "ui-tui" inside ui-tui/. See + # #42973. npm_workspace_args: tuple[str, ...] = () if npm_cwd == tui_dir else ("--workspace", "ui-tui") if termux_startup: npm_cwd, npm_workspace_args = _termux_workspace_install_context(tui_dir, include_child_workspaces=True) @@ -508,6 +536,11 @@ def _make_tui_argv(tui_dir: Path, tui_dev: bool) -> tuple[list[str], Path]: # 1. Prebuilt bundle (nix / packaged release / Docker image): just run it. # Must run BEFORE _ensure_tui_workspace(): a prebuilt install ships # hermes_cli/tui_dist/entry.js but never ui-tui/ (git checkouts only). + # 1. A prebuilt install (Docker image, Nix build, or prior `npm run build`) ships + # hermes_cli/tui_dist/entry.js but never ships ui-tui/ at all (that directory only exists in a git + # checkout) — so requiring the workspace to exist first made every prebuilt dashboard Chat tab + # connection hard-exit before it ever got a chance to try the bundled entry.js it already has. See + # #56665. if not tui_dev: if ext_dir: p = Path(ext_dir) @@ -778,7 +811,14 @@ def _launch_tui( def _pin_kanban_board_env() -> None: """Pin the active kanban board into ``HERMES_KANBAN_BOARD`` so in-process tools and shelled-out - ``hermes kanban`` calls agree even if a concurrent ``boards switch`` flips the file mid-turn.""" + ``hermes kanban`` calls agree even if a concurrent ``boards switch`` flips the file mid-turn. + + Without this, in-process tools (``kanban_*``) and shelled-out CLI calls (``hermes kanban …``) resolve + the board on different paths: the env-pin if set, otherwise the global ``<root>/kanban/current`` file. A + concurrent ``hermes kanban boards switch`` from another session can flip the file mid-turn, so the same + chat sees its tool calls hit board A while its shell calls hit board B (#20074). Pinning at chat boot + mirrors what the dispatcher already does for spawned workers. + """ if os.environ.get("HERMES_KANBAN_BOARD"): return with contextlib.suppress(Exception): diff --git a/hermes_cli/main_web_build.py b/hermes_cli/main_web_build.py index 7cd47ae20a..d085b94d43 100644 --- a/hermes_cli/main_web_build.py +++ b/hermes_cli/main_web_build.py @@ -53,6 +53,12 @@ def _sweep_stale_bytecode_if_checkout_changed() -> None: Update-time clears can't close the stale-bytecode class: ``hermes update`` runs the PRE-pull updater code and manual pulls never run it. Cheap file reads, no git subprocess. Never raises. + + The stale-bytecode bug class (issues #6207, #60242; Dhruv's WhatsApp ``cannot import name + 'parse_model_flags_detailed'`` report) has one shared shape: the checkout's ``.py`` files change (git + pull inside ``hermes update``, a manual ``git pull``, a ZIP update, a file-sync restore) while + ``__pycache__`` retains bytecode from the previous revision, and a later process trusts the stale + ``.pyc`` instead of the fresh source. """ from hermes_cli.main import PROJECT_ROOT, _clear_bytecode_cache, _read_git_revision_fingerprint, _record_bytecode_fingerprint try: @@ -207,7 +213,17 @@ def _run_with_idle_timeout( env: dict[str, str] | None = None) -> subprocess.CompletedProcess: """Stream a subprocess, killing it after *idle_timeout_seconds* of silence (a silent captured Vite build on a low-memory host looks like a hang and users reboot mid-install). Returns merged - stdout, empty stderr, rc 124 if terminate raced a clean exit; never raises on idle timeout.""" + stdout, empty stderr, rc 124 if terminate raced a clean exit; never raises on idle timeout. + + Issue #33788: ``npm run build`` (Vite) was invoked with ``capture_output=True`` and no timeout. On + low-memory hosts (notably WSL2 with the default 4 GB cap) the build can stall or sit silent for minutes; + users see a frozen terminal, assume the update is hung, and reboot — leaving the editable install in a + half-state with the ``hermes`` launcher present but ``hermes_cli`` not importable. + This helper fixes both halves: stdout is streamed (so the user sees progress), and if no bytes have + appeared on stdout/stderr for ``idle_timeout_seconds``, the process is terminated and the call returns + with a non-zero ``returncode``. The caller's existing stale-dist fallback (#23817) takes over from + there. + """ merged_chunks: list[str] = [] last_output_ts = _time.monotonic() lock = threading.Lock() @@ -307,6 +323,11 @@ def _run_npm_install_deterministic( is forced: an inherited ``NODE_ENV=production`` / ``omit=dev`` silently skips the build toolchain and the build dies with ``tsc: not found``. An npm outside ``engines.npm`` fails every command, so it gets one engine-repair retry. + + ``--no-save`` on the ``npm install`` fallback keeps it true to this function's contract: never mutate + ``package-lock.json``. Without it, an out-of-sync lockfile gets rewritten by the fallback, which drifts + the committed lockfile and makes every future ``npm ci`` fail — a self-reinforcing cycle where web + devDeps never install and a stale dist is served on every update (PR #65595). """ # CI=1 no-ops unicode-animations' postinstall that animates to /dev/tty. run_env = _npm_lifecycle_env(env) @@ -424,6 +445,16 @@ def _web_npm_install_context(web_dir: Path) -> tuple[Path, tuple[str, ...]]: if _is_termux_startup_environment(): return _termux_workspace_install_context(web_dir) npm_cwd = _workspace_root(web_dir) + # Scope the install to the web workspace only so that the full workspace graph (including apps/desktop + # with its Electron + node-pty deps) is never resolved here. Without --workspace the root package.json's + # apps/* glob would pull in desktop on every web build. See #38772. When web/ has its own + # package-lock.json, _workspace_root() returns web_dir itself and --workspace would fail. See #42973. + # When running from the workspace root, this must name the SAME closure as `hermes update`'s + # _update_node_dependencies() (ui-tui + web + --include-workspace-root): the helper prefers `npm ci`, + # which deletes node_modules before reifying the requested tree, so a narrower closure here silently + # prunes everything the update step just installed (root devDependencies and the ui-tui workspace) while + # still exiting 0 — and since the manifests digest was already recorded, later no-op updates skip the + # repair. See #43564/#64354. if npm_cwd == web_dir: return npm_cwd, () args: tuple[str, ...] = ("--workspace", "web", "--include-workspace-root") @@ -482,6 +513,10 @@ def _do_build_web_ui(web_dir: Path, *, fatal: bool = False) -> bool: # interrupted link step); a plain retry would keep `tsc: not found` # forever. Reinstall non-silently first, then one delayed retry for # boot-time races (antivirus scanning Node, npm cache not ready). + # First attempt — stream output via idle-timeout helper (issue #33788). capture_output=True on a + # long Vite build looks identical to a hang; users react by rebooting, which leaves the editable + # install in a half-state. Streaming + idle-kill makes failures observable AND recoverable (the + # stale-dist fallback below handles the kill path). missing_tool = _missing_web_build_tool((r2.stdout or "") + (r2.stderr or "")) if missing_tool: _console_print(f" ⚠ Build could not resolve {missing_tool} — reinstalling web dependencies...") diff --git a/hermes_cli/managed_uv.py b/hermes_cli/managed_uv.py index 2ae682ce6f..a61f15e145 100644 --- a/hermes_cli/managed_uv.py +++ b/hermes_cli/managed_uv.py @@ -351,7 +351,13 @@ _MAX_PATCH_RETRIES = 5 def _list_available_patches( uv_bin: str, minor: str, *, cwd: Path, env: dict) -> list[tuple[int, int, int]]: """Known patch versions for ``minor`` (e.g. "3.11"), newest first; [] on any failure - (network, parse), in which case callers fall back to the bare-minor request.""" + (network, parse), in which case callers fall back to the bare-minor request. + + Queries ``uv python list --all-versions`` rather than trusting the bare minor-line request to resolve to + the newest patch (issue #71250: on some hosts/uv versions, the resolved candidate for a bare "3.11" + request can be an older cached/indexed patch that still links a vulnerable SQLite, even when a newer + non-vulnerable patch is available). + """ try: result = subprocess.run( [ @@ -454,6 +460,11 @@ def _retry_explicit_patches( the fix and the downgrade guard rejects the rest; on a stale uv catalog the newest indexed patch can be the installed one, and the loop would burn every retry walking backwards. """ + # The bare minor-line request resolved to a still-vulnerable (or otherwise rejected) candidate. Rather + # than giving up immediately, query which patches on this minor line uv actually knows about and retry + # with explicit newer versions, newest-first -- this handles the case where the default resolution for a + # bare request picks an older cached/indexed patch even though a newer, non-vulnerable one is available + # (issue #71250). env_for_list = managed_python_env(project_root, install_dir=python_root) patches = _list_available_patches(uv_bin, request, cwd=project_root, env=env_for_list) attempts = 0 @@ -513,6 +524,7 @@ def _install_safe_python_generation( # All patches on the current minor line are vulnerable or rejected. Fall forward to the next # supported minor (e.g. 3.11 → 3.12) so the user isn't stuck on every `hermes update`. The # requires-python window (>=3.11,<3.14) and the import smoke-test gate compatibility. + # See #76106. cur_major, cur_minor = current.python_version[:2] fb_tried: set[tuple[int, int, int]] = set(tried_versions) for next_minor in range(cur_minor + 1, 14): # up to 3.13 @@ -710,6 +722,12 @@ def _windows_runtime_self_lock(live: Path) -> tuple[bool, str]: ``_detect_venv_python_processes`` excludes the calling process and its ancestors on purpose (``hermes update`` itself runs from the venv python), which is correct for the dependency-sync path where only a *loaded* ``.pyd`` image blocks the rewrite and a fresh child dodges it. + + For the whole-venv park rename that exemption is fatal: Windows keeps the image of any executable a + running process was started from mapped until that process exits, so a directory containing the + updater's own ``python.exe`` (or a waiting ``hermes.exe`` launcher ancestor) can never be renamed from + inside the updater. The retry loop in ``_cut_over_candidate`` cannot help against that — the lock is + structural, not transient (#93032). """ if platform.system() != "Windows": return False, "" @@ -761,7 +779,16 @@ def _uv_version_string(uv_bin: str) -> str: def _refresh_managed_uv_catalog(uv_bin: str) -> bool: """Re-bootstrap the managed uv binary to refresh its Python catalog (the only supported - refresh path for unmanaged installs). A caller-supplied foreign uv path is left alone.""" + refresh path for unmanaged installs). A caller-supplied foreign uv path is left alone. + + The managed uv is installed with ``UV_UNMANAGED_INSTALL``, which disables ``uv self update`` by design — + so its embedded python-build-standalone download catalog stays frozen at bootstrap age. + python-build-standalone re-releases existing CPython patch versions with newer SQLite (e.g. the 3.11.15 + build was re-cut with SQLite 3.53.x), so a stale catalog can make every provisioning attempt resolve to + a vulnerable build even though a fixed build of the SAME patch version exists (issue #72093). The + patch-retry loop cannot recover from that: the fixed build carries no newer version number to retry + with. + """ managed = managed_uv_path() try: if Path(uv_bin).resolve() != managed.resolve(): @@ -794,6 +821,10 @@ def _sweep_stale_runtime_backups( On POSIX this is safe while an older process still maps files from the tree (open FDs/mmaps keep their inodes). ``min_age_seconds`` avoids racing a concurrent repair whose fresh backup may still be its rollback path; ``keep`` exempts the backup this repair just created. + + A successful runtime repair parks the previous venv as ``<live>.stale.runtime-<token>``; historically + nothing ever reclaimed those, so each repair leaked a full venv (~1 GB) at the project root forever + (issue #73109). """ try: candidates = list(live.parent.glob(f"{live.name}.stale.runtime-*")) @@ -830,6 +861,7 @@ def _repair_windows_preflight( # staged for a cutover that can never run only leaks an incomplete generation. for line in ( f" ⚠ SQLite runtime repair deferred: {self_detail}.", + # See #93032. " Retrying `hermes update` from inside this venv cannot help: " "the mapped executable is released only when this process exits.", " To complete the repair, run the updater from an interpreter " @@ -862,6 +894,7 @@ def _repair_under_lock( # versions with fixed SQLite, but a frozen catalog keeps resolving the old vulnerable build # and the patch-retry loop has no newer number to try. Refresh the binary and retry once. if provisioned is None and _refresh_managed_uv_catalog(uv_bin): + # See #72093. print(" → Managed uv refreshed; retrying provisioning...") provisioned = _install_safe_python_generation(uv_bin, project_root=root, current=current) if provisioned is None: @@ -916,6 +949,7 @@ def repair_vulnerable_runtime( # Already fixed: any venv.stale.runtime-* markers next to the live venv are leftovers # from a past repair and will never be rolled back to. Sweep them so they don't leak # ~1 GB each forever. Age-gated to avoid racing an in-flight repair in a sibling process. + # See #73109. _sweep_stale_runtime_backups(live, root=root) return _result("safe", current, sqlite_after=current.sqlite_version_string) deferred = _repair_windows_preflight(root, live, current) diff --git a/hermes_cli/mcp_config.py b/hermes_cli/mcp_config.py index 679be96d60..7e0d956f00 100644 --- a/hermes_cli/mcp_config.py +++ b/hermes_cli/mcp_config.py @@ -153,6 +153,8 @@ def _strip_bearer_prefix(token: str) -> str: The header template already stores ``Authorization: Bearer ${MCP_X_API_KEY}``; a token pasted with its own prefix would send ``Bearer Bearer <jwt>`` and get a 401. + + Normalize on save. (#37792) """ if not isinstance(token, str): return token @@ -231,6 +233,12 @@ def _resolve_mcp_server_config(config: dict) -> dict: Mirrors ``_load_mcp_config()`` in ``tools/mcp_tool.py``; without it the discovery probe sent literal placeholders in header templates and auth-requiring servers returned 401. + + Mirrors ``_load_mcp_config()`` in ``tools/mcp_tool.py``: load ``~/.hermes/.env`` into ``os.environ`` and + recursively interpolate any ``${VAR}`` placeholders. The CLI builds header templates like + ``Authorization: Bearer ${MCP_X_API_KEY}`` but the probe path never resolved them, so the discovery + probe sent the literal placeholder and auth-requiring servers (e.g. n8n) returned 401 — while runtime + tool loading worked because it interpolates. (#37792) """ from tools.mcp_tool import _interpolate_env_vars from agent.secret_scope import current_secret_scope @@ -696,6 +704,9 @@ def cmd_mcp_reauth(args): """Re-authenticate one OAuth MCP server, or all of them sequentially. Serial-by-design: a human can only complete one browser OAuth flow at a time. + + This is the self-service fix for the recurring stale-client ritual in GH#36767 (and avoids the startup + popup storm when several servers go stale at once). """ servers = _get_mcp_servers() name = getattr(args, "name", None) diff --git a/hermes_cli/mcp_security.py b/hermes_cli/mcp_security.py index ca4d43cdeb..8365ea42e3 100644 --- a/hermes_cli/mcp_security.py +++ b/hermes_cli/mcp_security.py @@ -92,6 +92,9 @@ def validate_mcp_server_entry(name: str, entry: dict[str, Any]) -> list[str]: Intentionally not a whitelist — custom commands, Python scripts, npx, uvx stay legal. Only three narrow shapes are blocked: (1) a known IOC anywhere in command/args/env, (2) a shell interpreter with network egress in its inline script, (3) a shell interpreter writing an OS persistence surface. + + * a shell interpreter whose inline script writes to an OS persistence surface (June 2026 hermes-0day + SSH/PAM/sudoers/cron shape). See #45620. """ if not isinstance(entry, dict): return [] diff --git a/hermes_cli/mcp_startup.py b/hermes_cli/mcp_startup.py index 8a2c587e47..11620e359c 100644 --- a/hermes_cli/mcp_startup.py +++ b/hermes_cli/mcp_startup.py @@ -98,6 +98,8 @@ def start_background_mcp_discovery(*, logger, thread_name: str) -> None: # Re-install the caller's context-local HERMES_HOME override (multi-profile dashboard/desktop # backends) inside the thread: ContextVars don't propagate into bare threads, so a session # switched to profile X would otherwise discover the LAUNCH profile's mcp_servers. + # The config gate above already runs on the caller's thread, so it sees the same override. See + # #67605. home_override = get_hermes_home_override() def _discover() -> None: @@ -218,6 +220,10 @@ def mcp_discovery_in_flight() -> bool: Mirrors ``tui_gateway.entry.mcp_discovery_in_flight``; surfaces that start discovery here (desktop, dashboard sidecar) populate this thread, so the late-refresh scheduler consults both. + + Those processes populate THIS module's ``_mcp_discovery_thread``, not ``tui_gateway.entry``'s, so the + late-refresh scheduler must consult both to decide whether a slow server's tools are still pending (see + #51587). """ thread = _mcp_discovery_thread return thread is not None and thread.is_alive() diff --git a/hermes_cli/memory_setup.py b/hermes_cli/memory_setup.py index 96d6da338a..07f4d52666 100644 --- a/hermes_cli/memory_setup.py +++ b/hermes_cli/memory_setup.py @@ -25,6 +25,10 @@ def _provider_pip_dependencies(provider_name: str, declared: list) -> list: ``plugin.yaml`` declares the baseline bridge packages; some providers add mode-dependent extras at setup time that the manifest can't express. + + Hindsight's ``local_embedded`` mode installs ``hindsight-all`` (daemon + embedder + client) during + ``hermes memory setup`` — if the update-time refresh only reinstalled the declared ``hindsight-client``, + the embedded daemon would stay broken after a venv rebuild stripped ``hindsight-embed`` (#70636). """ deps = list(declared or []) if provider_name == "hindsight": @@ -83,6 +87,11 @@ def _install_dependencies(provider_name: str, *, force: bool = False) -> None: With ``force`` every declared dependency goes to the installer even if it imports (the resolver no-ops when nothing drifted) — how ``hermes update`` heals a provider after a venv rebuild. + + When ``force`` is true, every declared dependency is handed to the installer even if its import + currently succeeds — the resolver then reinstalls anything missing or version-drifted and no-ops on + satisfied ranges. This is how ``hermes update`` heals the active memory provider after a venv + rebuild/sync removed or downgraded its bridge packages (#53272, #70636). """ import subprocess from plugins.memory import find_provider_dir diff --git a/hermes_cli/moa_cmd.py b/hermes_cli/moa_cmd.py index 46a28b3518..eb2939f622 100644 --- a/hermes_cli/moa_cmd.py +++ b/hermes_cli/moa_cmd.py @@ -27,6 +27,11 @@ def _prompt_choice(title: str, rows: list[str], default: int = 0) -> int: def _model_options() -> list[dict[str, Any]]: payload = build_models_payload( + # Keep the profile override inside the worker thread so the full sync picker build (config load, + # pricing, refresh probes) runs off the event loop under the requested profile. Use + # _config_profile_scope (contextvar only, no skill-module lock) — the payload build can block for + # 15s on a models.dev cache miss, and _profile_scope's RLock held across that block starves + # concurrent /api/config and freezes the server (#58576). load_picker_context(), # Slot pickers must only offer providers the user can actually call. # Including setup-only rows makes an unconfigured canonical provider diff --git a/hermes_cli/moa_config.py b/hermes_cli/moa_config.py index 46e6508617..0e35a866d0 100644 --- a/hermes_cli/moa_config.py +++ b/hermes_cli/moa_config.py @@ -58,7 +58,12 @@ def _coerce_reference_timeout(value: Any) -> float | None: def _coerce_fanout(value: Any) -> str: """Normalize the fan-out cadence to ``per_iteration`` | ``user_turn`` | ``every_n:<N>`` (N >= 2); the mapping form ``{mode: every_n, n: N}`` becomes the string, ``every_n:1`` collapses to - ``per_iteration``, anything unparseable falls back to ``user_turn`` (the cheapest cadence).""" + ``per_iteration``, anything unparseable falls back to ``user_turn`` (the cheapest cadence). + + The ``every_n`` cadence also accepts the mapping form ``{mode: every_n, n: N}`` from hand-edited YAML + and normalizes it to the canonical string, so the rest of the pipeline (presets, flattened view, + runtime) only ever sees one shape. See #67199. + """ def _every_n(n: int) -> str: return f"every_n:{n}" if n >= 2 else ("per_iteration" if n == 1 else "user_turn") @@ -82,7 +87,13 @@ def coerce_privacy_filter(value: Any) -> str: ``false``/``None``/unknown values land on '' so a hand-edited config degrades to prior behavior. 'display' redacts user-visible surfaces only (reference blocks in the UI and saved - MoA trace records).""" + MoA trace records). + + - ``''`` (empty string): filter off — the default. The aggregator still sees raw advisor text, so answer + quality is unaffected. - ``'full'``: additionally redact the advisor text injected into the aggregator + prompt (issue #59959's literal ask). A hand-edited boolean ``true`` maps here because the issue framed + the toggle as "redact before passing to the aggregator". + """ if value is True: return "full" if value is None or value is False: @@ -168,7 +179,12 @@ def validate_moa_payload(raw: Any) -> list[str]: """Return the problems ``normalize_moa_config`` would silently paper over (empty = safe to save). Read-time tolerance (a hand-edited config degrades to defaults instead of crashing) is a - corruption engine at write time: a half-filled slot would silently replace the whole preset.""" + corruption engine at write time: a half-filled slot would silently replace the whole preset. + + ``normalize_moa_config`` is deliberately tolerant: at *read* time a hand-edited config must degrade to + defaults rather than crash the agent. API write paths call this first and reject invalid payloads loudly + instead of saving something the user never chose. See #64156. + """ if not isinstance(raw, dict): return ["MoA config must be an object"] @@ -229,6 +245,14 @@ _FLAT_PRESET_KEYS = ( "fanout", "enabled") +# When the reference fan-out runs. "user_turn" (default) runs the advisors ONCE per user turn (the original +# MoA shape, and the cheapest cadence — #67199): the aggregator gets their upfront plan-level advice, then +# acts alone for the rest of the tool loop. "per_iteration" re-runs the advisors whenever the advisory view +# changes — i.e. every tool iteration, so advice tracks live task state at the cost of multiplying advisor +# spend by tool-loop depth. "every_n:<N>" (N >= 2) is the middle ground: advisors run on the first iteration +# of each user turn and every Nth tool iteration after it; in-between iterations reuse the cached guidance +# from the last advisor run. Also accepts the mapping form {mode: every_n, n: N}, normalized to the +# canonical string. def normalize_moa_config(raw: Any) -> dict[str, Any]: """Return validated MoA config with named presets.""" if not isinstance(raw, dict): @@ -280,7 +304,10 @@ def exact_moa_preset_name(config: Any, text: str) -> str | None: Used by the no-explicit-provider switch path for a bare ``/model <preset>``. Because the match is implicit it honors the per-preset ``enabled`` opt-out: a plain model switch that collides with a disabled preset's name must not silently pivot onto the MoA provider. Explicit - ``--provider moa`` / picker selection bypasses this, so disabled presets stay reachable.""" + ``--provider moa`` / picker selection bypasses this, so disabled presets stay reachable. + + See #55187. + """ wanted = str(text or "").strip() if not wanted: return None diff --git a/hermes_cli/model_normalize.py b/hermes_cli/model_normalize.py index 67636d2470..3454a15282 100644 --- a/hermes_cli/model_normalize.py +++ b/hermes_cli/model_normalize.py @@ -48,6 +48,13 @@ _AUTHORITATIVE_NATIVE_PROVIDERS: frozenset[str] = frozenset({ _MATCHING_PREFIX_STRIP_PROVIDERS: frozenset[str] = frozenset({ "zai", "kimi-coding", + # Providers whose endpoint does not accept image input, even though the provider's broader ecosystem has + # vision models available elsewhere. When `auxiliary.vision.provider: auto` sees one of these as the + # main provider, it must skip straight to the aggregator chain instead of returning a client that will + # 404 on every vision request. kimi-coding / kimi-coding-cn: the Kimi Coding Plan routes through + # api.kimi.com/coding (Anthropic Messages wire) which Kimi's own docs describe as having no image_in + # capability. Vision lives on the separate Kimi Platform (api.moonshot.ai, OpenAI-wire, pay-as-you-go). + # See #17076. "kimi-coding-cn", "minimax", "minimax-oauth", @@ -226,6 +233,7 @@ def normalize_model_for_provider(model_input: str, target_provider: str) -> str: # Copilot's own normalizer knows the alias table (vendor stripping, dash-to-dot repair for Claude) # and live-catalog lookups; without it dash-notation Claude ids hit HTTP 400 model_not_supported. + # See issue #6879. if provider in {"copilot", "copilot-acp"}: try: from hermes_cli.models import normalize_copilot_model_id diff --git a/hermes_cli/model_setup_flows.py b/hermes_cli/model_setup_flows.py index 7632dc0107..b065a652d1 100644 --- a/hermes_cli/model_setup_flows.py +++ b/hermes_cli/model_setup_flows.py @@ -694,6 +694,9 @@ def _model_flow_stepfun(config, current_model=""): model = _finish_model(selected, provider_id, f"Default model set to: {selected} (via {pconfig.name})", base_url=effective_base, drop_api_mode=True) if model is not None: + # Sync the caller's config dict so the setup wizard's final save_config(config) preserves our model + # settings. Without this, the wizard overwrites model.provider/base_url with the stale values from + # its own config dict (#4172). config["model"] = dict(model) diff --git a/hermes_cli/model_setup_flows_common.py b/hermes_cli/model_setup_flows_common.py index 02bccb26ea..8a655a890e 100644 --- a/hermes_cli/model_setup_flows_common.py +++ b/hermes_cli/model_setup_flows_common.py @@ -225,6 +225,7 @@ def _prune_replaced_custom_model_config_credentials(base_url: str, *, provider_n # A keyed ``providers.<key>`` endpoint stores under the durable slug while # legacy pools keep ``custom:<display-name>``; every identity the active # endpoint may occupy must be skipped or its own legacy pool gets pruned. + # See #100413. active_pool_keys = { str(key).strip().lower() for key in custom_provider_pool_key_candidates(base_url, provider_name=provider_name or None)} diff --git a/hermes_cli/model_setup_flows_custom.py b/hermes_cli/model_setup_flows_custom.py index 49599d342b..25111710da 100644 --- a/hermes_cli/model_setup_flows_custom.py +++ b/hermes_cli/model_setup_flows_custom.py @@ -139,6 +139,7 @@ def _model_flow_custom(config): # The key goes to .env and config.yaml only references it. Keyed on host:port # so two servers on one machine keep separate credentials. + # See #69449. custom_key_env = "" if effective_key: _parsed = urllib.parse.urlparse(effective_url) @@ -306,6 +307,8 @@ def _model_flow_named_custom(config, provider_info): # ``discover_models: false`` (default True) uses the configured ``models:`` list # verbatim and skips the live probe, so operators can restrict the picker to the # subset their plan serves. Same semantics as the slash-command picker. + # This lets operators restrict the picker to the subset their plan actually serves instead of the + # endpoint's full catalog (#18726: Baidu Qianfan returns 100+ models for a 2-3 model plan). discover = provider_info.get("discover_models", True) if isinstance(discover, str): discover = discover.lower() not in {"false", "no", "0"} diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 32e9f4a515..00cdb5c3dc 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -180,7 +180,10 @@ class DirectAlias(NamedTuple): ``api_key`` / ``key_env`` carry the alias endpoint's OWN credential. Without them the switch would keep the *default* provider's key, which 401s against the alias host and sends that provider's secret to an unrelated third party. Both default so positional - ``DirectAlias(model, provider, base_url)`` keeps working.""" + ``DirectAlias(model, provider, base_url)`` keeps working. + + See #83612. + """ model: str provider: str base_url: str @@ -207,7 +210,10 @@ def _load_direct_aliases() -> dict[str, DirectAlias]: ``api_key`` — literal or ``"${VAR}"`` — / ``key_env``); with neither credential field the key is resolved from the alias HOST, never from the previously active provider. ``model.aliases`` never overrides ``model_aliases``; its string entries (``ds-flash: deepseek/deepseek-v4-flash``) - take the provider from the ``provider/`` prefix, else the current provider.""" + take the provider from the ``provider/`` prefix, else the current provider. + + See #83612. + """ merged = dict(_BUILTIN_DIRECT_ALIASES) try: from hermes_cli.config import load_config @@ -316,7 +322,10 @@ def direct_alias_runtime_request(alias: DirectAlias) -> tuple[str, Optional[str] would otherwise reach that provider's explicit-runtime branch and put the live vendor token on the foreign wire. Bare ``custom`` is host-gated, so an authoritative URL still resolves its vendor key and a foreign one resolves none. An alias with no base_url keeps its label — - there is no foreign host, and the label is the only routing information.""" + there is no foreign host, and the label is the only routing information. + + See #28660. + """ return ("custom" if alias.base_url else (alias.provider or "custom")), direct_alias_api_key(alias) or None @@ -370,6 +379,10 @@ def resolve_startup_model_route( # model/base_url only. return StartupModelRoute(model=direct.model, provider=explicit_provider, base_url=direct.base_url) # Same owner as the interactive /model and oneshot paths: credential for the alias HOST. + # Resolve through the SAME owner the interactive /model and oneshot paths use: a URL-bearing alias + # must resolve its credential for the alias HOST, never for its provider label — a label like + # ``anthropic`` on a foreign URL would otherwise reach that provider's explicit-runtime branch and + # put the live vendor token on the foreign wire (#28660). alias_provider, alias_key = direct_alias_runtime_request(direct) return StartupModelRoute( model=direct.model, provider=alias_provider, base_url=direct.base_url, api_key=alias_key or "") @@ -490,7 +503,16 @@ def resolve_persist_behavior( the pick does not evaporate into whatever ``*_API_KEY`` is lying around on the next launch; ``--provider`` without a persist flag -> False (exploratory); else ``model.persist_switch_by_default`` (default False). A flat-string ``model`` IS a configured - default; an unreadable config -> False.""" + default; an unreadable config -> False. + + 1. ``--once`` explicitly opts out → ``False`` (next turn only). 2. ``--session`` explicitly opts out → + ``False`` (this session only). 3. 4. Applies to every surface (CLI, gateway, Desktop picker) so no + client has to hardcode ``--global``. 5. Provider switches are typically exploratory — the user is trying + a different backend for this conversation, not reconfiguring the default. 6. Otherwise defer to + ``model.persist_switch_by_default`` in ``config.yaml`` (defaults to ``False``: a plain ``/model <name>`` + affects only the current session). Users who want the old persist-by-default behavior can set the key to + ``true``; a one-off ``--global`` always persists. See #86414. + """ if is_once or is_session: return False if is_global: @@ -769,7 +791,12 @@ def resolve_display_context_length( models.dev reports per-vendor context but provider-enforced limits can be lower (Codex OAuth caps gpt-5.5 at 272k), so ``agent.model_metadata.get_model_context_length`` is authoritative (it also honors ``custom_providers[].models.<id>.context_length``); ``model_info.context_window`` - is the fallback. A ``config_context_length`` pin is dropped when the route changed.""" + is the fallback. A ``config_context_length`` pin is dropped when the route changed. + + When ``custom_providers`` is provided, per-model ``context_length`` overrides from + ``custom_providers[].models.<id>.context_length`` are honored — this closes #15779 where ``/model`` + switch ignored user-set overrides. + """ if config_context_length is not None and (configured_model or configured_provider or configured_base_url): try: from hermes_cli.route_identity import should_clear_context_pin @@ -808,7 +835,11 @@ def _configured_provider_matches( """``{provider_slug: canonical_model_id}`` for every configured provider whose declared models (``models``, ``model``, ``default_model`` — exact, case-insensitive, never fuzzy) contain ``model_name``, so a typed name routes to the provider that declares it instead of being - soft-accepted by the current provider (openai-codex) as an unknown hidden model.""" + soft-accepted by the current provider (openai-codex) as an unknown hidden model. + + Used by :func:`switch_model` to route a *typed* model name to the provider that actually declares it in + user/custom provider config, instead of leaving it on the current provider. See #45006. + """ if not model_name or not model_name.strip(): return {} target = model_name.strip().lower() @@ -1352,6 +1383,14 @@ def _copilot_api_mode(provider: str, model: str, api_key: str) -> str: def _opencode_api_mode(provider: str, model: str, api_key: str) -> str: + # Re-derive api_mode from the effective model rather than the persisted api_mode: the opencode providers + # serve both anthropic_messages and chat_completions models, so the previous session's mode must not + # leak across /model switches. Refs #16878. + # opencode-zen/go must always re-derive api_mode from the target model (not the stale persisted + # api_mode), because the same provider serves both anthropic_messages (e.g. minimax-m2.7) and + # chat_completions (e.g. deepseek-v4-flash) and switching models via /model would otherwise carry the + # previous mode forward, stripping /v1 from base_url for chat_completions models and 404'ing. Refs + # #16878. from hermes_cli.models import opencode_model_api_mode return opencode_model_api_mode(provider, model) diff --git a/hermes_cli/model_switch_providers.py b/hermes_cli/model_switch_providers.py index a80b6b6392..d673407546 100644 --- a/hermes_cli/model_switch_providers.py +++ b/hermes_cli/model_switch_providers.py @@ -448,6 +448,10 @@ def _extend_unique(target: list, items) -> None: def _norm_url(url: Any) -> str: + # Effective base URLs of every built-in row we emit (normalized lower+rstrip). Section 4 uses this to + # hide ``custom_providers`` entries that point at the same endpoint as a built-in (e.g. a user-defined + # "my-dashscope" on https://coding-intl.dashscope.aliyuncs.com/v1 collides with the built-in + # alibaba-coding-plan row when DASHSCOPE_API_KEY is present). Fixes #16970. return str(url or "").strip().rstrip("/").lower() @@ -710,6 +714,9 @@ def _overlay_has_creds(b: _PickerBuild, pid: str, hermes_slug: str, overlay) -> has_creds = _overlay_has_env_creds(pid, hermes_slug, overlay, os.environ.get) # External-process providers (copilot-acp) hold no key/token/pool entry by design — the # spawned ACP subprocess brings its own auth. "Configured" means the executable resolves. + # "Configured" means the executable resolves, which is exactly what get_auth_status() reports for them; + # without this branch the has_creds filter below unconditionally hides the provider from every picker + # (#63662). if not has_creds and overlay.auth_type == "external_process": try: from hermes_cli.auth import get_auth_status @@ -1039,6 +1046,7 @@ def list_authenticated_providers( pass # PyYAML parses unquoted numeric names (`provider: 2070`) as int. + # seen_slugs: set = set() # lowercase-normalized to catch case variants (#9545) current_provider = coerce_provider_id(current_provider) current_base_url = str(current_base_url or "").strip() current_model = str(current_model or "").strip() diff --git a/hermes_cli/models.py b/hermes_cli/models.py index c44afab89b..57a709eadf 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -58,6 +58,12 @@ from hermes_cli.models_reasoning_caps import ( # noqa: F401 (re-exported; test _seed_reasoning_caps, nous_catalog_url, nous_model_reasoning_capabilities, + # Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models + # entries advertise reasoning support via supported_parameters + a reasoning object, which covers every + # routed vendor without a hand-maintained prefix list. The static prefix allowlist below repeatedly went + # stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before + # it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached; + # unknown (catalog unreachable / unlisted model) falls back to the static list. openrouter_model_reasoning_capabilities, parse_openrouter_reasoning_capabilities, refresh_reasoning_caps_async, @@ -469,7 +475,10 @@ def _openrouter_model_is_free(pricing: Any) -> bool: def _openrouter_model_supports_tools(item: Any) -> bool: """True when ``supported_parameters`` advertises ``tools`` (hermes-agent is tool-calling-first). Permissive when the field is absent/malformed: some OpenRouter-compatible gateways (Nous Portal, - private mirrors) don't populate it, and the picker must not silently empty for them.""" + private mirrors) don't populate it, and the picker must not silently empty for them. + + Ported from Kilo-Org/kilocode#9068. + """ params = item.get("supported_parameters") if isinstance(item, dict) else None return "tools" in params if isinstance(params, list) else True @@ -550,6 +559,9 @@ def fetch_openrouter_models( # Hide models without tool-calling support — selecting one fails at the first tool call. if live_item is None or not _openrouter_model_supports_tools(live_item): continue + # Hide models that don't advertise tool-calling support — hermes-agent requires it and surfacing + # them leads to immediate runtime failures when the user selects them. Ported from + # Kilo-Org/kilocode#9068. if preferred_id == silent_default: desc = "default" # keep the silent-default badge through the live refresh else: @@ -902,7 +914,11 @@ def _configured_provider_ids() -> set[str]: def _resolve_provider_prefix(model_name: str) -> Optional[tuple[str, str]]: """Route an explicit ``vendor/model`` prefix (``nous/deepseek-v4-pro``, ``ollama/qwen3.5:4b``) to - a provider the user defined in ``providers:`` (by raw name or alias) instead of the default.""" + a provider the user defined in ``providers:`` (by raw name or alias) instead of the default. + + ``nous/deepseek-v4-pro`` or ``ollama/qwen3.5:4b`` should route to the named provider instead of falling + back to the configured default (which silently sends non-default models to the wrong endpoint, #87189). + """ if "/" not in model_name: return None vendor, model = model_name.split("/", 1) @@ -1226,6 +1242,12 @@ def _openai_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]] # Custom OpenAI-compatible endpoints serve a small curated catalog — use it verbatim. Official # OpenAI hosts (canonical and data-residency regional) return 120+ embeddings/whisper/tts/… # entries, so intersect with the curated agentic catalog so ``/model`` matches ``hermes model``. + # Model not in live /v1/models — check the curated catalog before rejecting. Providers may omit models + # from their live listing that are still valid (stale cache, partial rollout, gated previews). Use the + # pure-catalog helper (no extra live fetch) so we only accept models Hermes actually ships. (#46850) + # Their /v1/models listing is access-scoped and authoritative — a model absent from it is one this key + # CANNOT serve, so the curated soft-accept would manufacture a selection that 400s at first use. Custom + # OpenAI-compatible proxies keep the fallback (incomplete listings are common there). from hermes_cli.providers import is_official_openai_host try: @@ -1342,6 +1364,14 @@ def provider_model_ids(provider: Optional[str], *, force_refresh: bool = False) if models is not None: return models + # Merge static curated list with live API results so models that the live endpoint omits (stale cache, + # partial rollout) still appear in the picker. Single providers (kimi, zai) use curated-first (commit + # 658ac1d86) to surface newest models even when live API lags (#46309). OpenCode Zen / Go are different: + # their live API is the authoritative catalog, so they merge live-first — live entries lead and stale + # curated entries no longer pollute the top of the picker. (#49129) Plugin providers with no static + # _PROVIDER_MODELS entry fall back to the profile's curated fallback_models so their agentic picks lead + # the picker instead of whatever the live catalog happens to return first (e.g. Fireworks lists an image + # model, flux-*, ahead of its chat models). curated_static = list(_PROVIDER_MODELS.get(normalized, [])) if normalized not in _MODELS_DEV_PREFERRED: return curated_static @@ -1912,7 +1942,14 @@ _OPENCODE_FAMILIES = ("opencode-free", "opencode-go", "opencode-zen") def opencode_provider_family(provider_id: Optional[str]) -> Optional[str]: - """Resolve a provider id (canonical or prefixed) to its OpenCode family, or None.""" + """Resolve a provider id (canonical or prefixed) to its OpenCode family, or None. + + Returns ``"opencode-zen"`` or ``"opencode-go"`` for the built-in providers AND for custom providers + whose name extends a family slug (e.g. ``opencode-go-bridge`` pointing at + ``https://opencode.ai/zen/go/v1``, issue #85589). Matching is case-insensitive. Custom family providers + need the same per-model api_mode routing and /v1 base-url normalization as the built-ins — this + predicate is the single owner of that family-membership question; do not re-implement it inline. + """ raw = str(provider_id or "").strip().lower() if not raw: return None diff --git a/hermes_cli/models_validate.py b/hermes_cli/models_validate.py index 18192737a5..31466cf59e 100644 --- a/hermes_cli/models_validate.py +++ b/hermes_cli/models_validate.py @@ -289,6 +289,12 @@ def _static_catalog(normalized: str) -> list[str]: _STATIC_FAMILY_PREFIXES = { + # Plausibility gate (#45006): the soft-accept (#16172 / #19729) exists for entitlement-gated *hidden* + # slugs the curated listing hasn't caught up with — but those are always the provider's own family + # (openai-codex -> gpt-*; xai-oauth -> grok-*). Accepting an unrelated typed name (e.g. `qwen3.5-4b`, + # `llama-3.1-8b`) here turns what should be an actionable "did you mean --provider <x>?" error into a + # confusing success that 400s on the next turn. Only soft- accept names that share the provider's family + # prefix; reject the rest with guidance to pin the right provider. "openai-codex": ("gpt-", "codex-", "o1", "o3", "o4"), "xai-oauth": ("grok-",), } diff --git a/hermes_cli/nous_billing.py b/hermes_cli/nous_billing.py index 3e383bb3a5..ecb37b27b7 100644 --- a/hermes_cli/nous_billing.py +++ b/hermes_cli/nous_billing.py @@ -208,7 +208,15 @@ _ERRORS_BY_STATUS: dict[int, tuple[type[BillingError], str]] = { def _raise_for_error(status: int, payload: dict[str, Any], headers: Any = None) -> None: - """Map an HTTP error response to the right typed :class:`BillingError` (see tables above).""" + """Map an HTTP error response to the right typed :class:`BillingError` (see tables above). + + Recognizes the Remote-Spending gate contract (NAS PR #481): 403 ``remote_spending_revoked`` (this + terminal's spend revoked → reconnect), 401 ``session_revoked`` (full logout → re-login), 503 + ``temporarily_unavailable`` (gate fail-closed → back off, NOT revoked). The business-denial codes + (``cli_billing_disabled`` + dual ``code:remote_spending_disabled``, ``role_required``, + ``idempotency_conflict``, …) flow through as a generic BillingError carrying + ``error``/``code``/``recovery`` for the surface to map. + """ p = payload if isinstance(payload, dict) else {} error = p.get("error") common = { diff --git a/hermes_cli/nous_subscription.py b/hermes_cli/nous_subscription.py index a136e9da73..c50755c945 100644 --- a/hermes_cli/nous_subscription.py +++ b/hermes_cli/nous_subscription.py @@ -203,6 +203,7 @@ def _has_agent_browser() -> bool: # from the *probe* process's PATH); local node_modules/.bin (PATHEXT-aware ``shutil.which`` so # Windows picks the ``.cmd`` shim). The hit must also run: a dangling symlink is reported by # ``which`` but fails at exec. + # See #48521. from hermes_constants import with_hermes_node_path local_bin_dir = Path(__file__).parent.parent / "node_modules" / ".bin" @@ -531,6 +532,9 @@ def _get_gateway_direct_credentials() -> Dict[str, bool]: audio_direct = bool(resolve_openai_audio_api_key()) return { "web": _any_env("FIRECRAWL_API_KEY", "FIRECRAWL_API_URL", "PARALLEL_API_KEY", "TAVILY_API_KEY", "EXA_API_KEY", "SEARXNG_URL"), + # Env-configured keyless local backend: a reachable self-hosted SearXNG is a working web setup even + # with no stored selection (the autodetect cascade in tools/web_tools.py picks it up), so it must + # not be classified "unconfigured" and pre-checked (#92647). "image_gen": fal_direct, "video_gen": fal_direct, "tts": audio_direct or _any_env("ELEVENLABS_API_KEY"), @@ -604,6 +608,8 @@ def prompt_enable_tool_gateway(config: Dict[str, object], *, force_fresh: bool = # Unconfigured tools first (pre-checked for new users), then tools with the user's own key # (unchecked). Tools previously offered and left unchecked are recorded in # ``tool_gateway_declined_tools`` and never pre-checked again (no re-fire on every model swap). + # Acceptance used to be sticky while refusal was not, so the identical pre-checked checklist re-fired on + # every Nous model swap. See #92647. declined_raw = config.get("tool_gateway_declined_tools") declined: set[str] = {str(k) for k in declined_raw} if isinstance(declined_raw, list) else set() offer_keys: list[str] = list(unconfigured) + list(has_direct) diff --git a/hermes_cli/npm_engine.py b/hermes_cli/npm_engine.py index b15b34fcc4..2c7e71b0a5 100644 --- a/hermes_cli/npm_engine.py +++ b/hermes_cli/npm_engine.py @@ -138,6 +138,8 @@ def upgrade_managed_npm(npm: str, npm_range: str, *, prefix: Path, quiet: bool = print(f"→ Upgrading Hermes-managed npm to satisfy {npm_range}…", flush=True) # The desktop app's Node processes execute from this tree; an in-place upgrade while in use # fails with PermissionError on npm.cmd. Defer — the upgrade re-triggers on the next resolution. + # Defer instead of forcing the write — the upgrade re-triggers on the next resolution (e.g. the next + # update once the app is closed). See #80926. if managed_node_tree_in_use(): if not quiet: print( diff --git a/hermes_cli/oneshot.py b/hermes_cli/oneshot.py index 91a1b75a7c..59afaa5a64 100644 --- a/hermes_cli/oneshot.py +++ b/hermes_cli/oneshot.py @@ -242,6 +242,7 @@ def run_oneshot( if response: # Lone UTF-16 surrogates would raise UnicodeEncodeError on a real stdout and abort with # exit 1 after the turn already completed — scrub to U+FFFD first. + # Model text can contain lone UTF-16 surrogates (invalid in UTF-8). See #80366. from agent.message_sanitization import _sanitize_surrogates response = _sanitize_surrogates(response) @@ -371,6 +372,9 @@ def _run_agent( # Oneshot builds AIAgent directly, bypassing cli.py's MCP background discovery and # _init_agent's wait, so the construction-time tool snapshot would miss late MCP servers. # Idempotent start + bounded wait with the single-query bound (there is no later turn). + # Ensure MCP tools are discovered before building the agent. This helper starts discovery if needed + # (idempotent) and bounded-waits with the larger single-query bound (default 15s) because there is only + # ONE turn and no between-turns late-binding refresh (#38448). from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build ensure_mcp_discovery_before_agent_build(logger=logging.getLogger(__name__), single_query=True) @@ -421,6 +425,10 @@ def _quietly(what: str, fn) -> None: def _linger_for_background_completions() -> None: + # Linger (bounded) for background processes this turn spawned with notify_on_complete=true BEFORE + # agent.close(): close() calls process_registry.kill_all(task_id) and the dying parent owns the + # children's stdout pipes, so exiting now destroys in-flight deliveries — including Bot Mode handoff + # replies dispatched from a short-lived recipient (#90879). from tools.process_registry import process_registry process_registry.wait_for_pending_completions(None) diff --git a/hermes_cli/pairing.py b/hermes_cli/pairing.py index 47b3298220..0a154835f9 100644 --- a/hermes_cli/pairing.py +++ b/hermes_cli/pairing.py @@ -69,6 +69,7 @@ def _cmd_approve(store, platform: str, code: str): print(" They'll be recognized automatically on their next message.\n") elif store._is_locked_out(platform): # approve_code returns None for both invalid codes and lockout — say which. + # Tell the operator it's lockout so they don't chase a "wrong code" rabbit hole (#10195). import time as _time lockout_until = store._load_json(store._rate_limit_path()).get(f"_lockout:{platform}", 0) mins = max(0, int(lockout_until - _time.time())) // 60 diff --git a/hermes_cli/plugin_packs.py b/hermes_cli/plugin_packs.py index 093f918040..2d67912c83 100644 --- a/hermes_cli/plugin_packs.py +++ b/hermes_cli/plugin_packs.py @@ -358,6 +358,8 @@ def install_pack_plugins( _save_enabled_set(enabled) _save_disabled_set(disabled) + # Per-plugin capability consent — the SAME flow as a single install (#64228). A pack never + # bulk-grants capabilities. declared = _declared_capabilities_from_manifest(manifest, installed_name) if declared: _run_capability_consent(console, installed_name, declared, context="install") diff --git a/hermes_cli/plugins.py b/hermes_cli/plugins.py index 13cc9b11e6..34c11150db 100644 --- a/hermes_cli/plugins.py +++ b/hermes_cli/plugins.py @@ -162,11 +162,28 @@ VALID_HOOKS: Set[str] = { # gateway_platform_event: normalized envelopes only, never raw SDK objects or adapter handles. # Kwargs: platform, event_type, payload (event_type-local; see hooks.md). New event types land # only together with real fire-sites. + # on_kanban_dispatch_tick fires once per dispatcher tick in dispatch_once, strictly AFTER the board's + # single-writer dispatch lock has been released (the #56066 original fired inside the lock — the #64231 + # disposition mandates the post-lock re-port), so a slow subscriber can never extend the writer critical + # section. Kwargs: board: str | None, profile_name: str, dry_run: bool, outcome: "ok" | "skipped_locked" + # | "idle", result: hermes_cli.kanban_db.DispatchResult (spawned, reclaimed, promoted, + # reconciled_orphans, crashed, stale, timed_out, auto_blocked, rate_limited, auto_assigned_default, + # respawn_guarded, skipped_per_profile_capped, skipped_unassigned, skipped_nonspawnable, + # skipped_locked). Privacy: result carries task ids, assignees, and workspace paths. + # Gateway platform-boundary observer hooks (#64176). Observer-only; each callback isolated by + # invoke_hook. This surface grants no adapter handles or platform actions. Fired today: Telegram + # "reaction" + "message_edited"; Discord "message_edited", "message_deleted", "thread_created", + # "thread_renamed". Each event type carries its own event-local additive payload contract (see + # hooks.md). Other event types and hook names land here only together with real fire-sites and payload + # contracts; no inert VALID_HOOKS surface is registered ahead of implementation. "gateway_platform_event", # pre_command: BEFORE a recognized slash command's handler on CLI and gateway canonical dispatch; # returns IGNORED in v1. Deliberately NOT fired for the gateway's running-agent intercept path # (/stop, /approve, busy_policy) — a slow/hostile plugin must not touch the operator's escape # hatches. Kwargs: surface, command (canonical), alias_used, args_raw, session_key, platform. + # Slash-command dispatch observer (#64204, observer-first per #64182 ground rule 3). Return values are + # IGNORED in v1 — a plugin returning a directive-shaped dict gets a debug log so future block/rewrite + # adopters are discoverable once the middleware variant ships against the #64231 taxonomy. "pre_command", } @@ -209,7 +226,10 @@ class PluginContext: def has_plugin(self, plugin_id: str) -> bool: """Return True when another plugin is loaded and enabled (runtime probe for advisory - ``requires_plugins``). Matches on registry key or manifest name.""" + ``requires_plugins``). Matches on registry key or manifest name. + + See #64165. + """ return any( loaded.enabled and (key == plugin_id or loaded.manifest.name == plugin_id) for key, loaded in self._manager._plugins.items() @@ -434,7 +454,12 @@ class PluginContext: """Register a tool in the global registry and track it as plugin-provided. ``override=True`` replaces a same-named built-in (without it a name claimed by another toolset is rejected) and needs operator opt-in via ``plugins.entries.<plugin_id>.allow_tool_override: true`` — otherwise - any enabled plugin could silently replace a privileged built-in like ``write_file``.""" + any enabled plugin could silently replace a privileged built-in like ``write_file``. + + ``override=True`` against a built-in tool requires the operator to opt in via + ``plugins.entries.<plugin_id>.allow_tool_override: true`` in config.yaml — mirrors the trust gate + pattern used for ``ctx.llm`` provider/model overrides (#23194). + """ if override and not self._tool_override_allowed(name): raise PluginToolOverrideError( f"Plugin {self.manifest.name!r} cannot override built-in tool {name!r}. Set " @@ -465,6 +490,7 @@ class PluginContext: " (override)" if override else "") return handle + # -- capability probing (#64228) ----------------------------------------- def has_capability(self, capability: str) -> bool: """True when *capability* is live for this plugin (probe, then degrade gracefully). Bundled plugins are trusted for ``tools.override``; otherwise granted_capabilities or the legacy @@ -480,7 +506,12 @@ class PluginContext: """Call ``tool`` on MCP ``server`` synchronously through :mod:`tools.mcp_tool`'s native client (same trust gates, breaker, reconnect — never a parallel connection). Servers not in ``plugins.entries.<plugin_id>.mcp_allowlist`` raise ``PermissionError`` (default-deny). ``timeout`` - clamps to 1–600s; results over ~64KB are truncated with a marker.""" + clamps to 1–600s; results over ~64KB are truncated with a marker. + + This is a per-server grant, deliberately not ambient authority over every configured server. + TODO(#64228): swap the per-server allowlist for the declared capability model once it lands + (per-tool grants, expiry, ro/rw). + """ if server not in self._mcp_allowlist(self.plugin_id): raise PermissionError( f"Plugin {self.manifest.name!r} is not allowed to call MCP " @@ -537,7 +568,16 @@ class PluginContext: def _tool_override_allowed(self, tool_name: str) -> bool: """Whether this plugin may override built-in tools: bundled plugins are trusted (a maintainer choice, not privilege escalation); others need ``tools.override`` via - :func:`plugin_capability_granted` (granted_capabilities OR legacy ``allow_tool_override: true``).""" + :func:`plugin_capability_granted` (granted_capabilities OR legacy ``allow_tool_override: true``). + + Bundled plugins (shipped with Hermes core) are trusted by default — an override there is a + deliberate maintainer choice, not a third-party plugin trying to elevate privilege. For every other + source, the canonical check is :func:`plugin_capability_granted` with the ``tools.override`` + capability — satisfied by EITHER the consent-flow grant + (``plugins.entries.<plugin_id>.granted_capabilities``) OR the deprecated legacy key + ``allow_tool_override: true`` (still honored for backward compatibility; #64228 reference + migration). + """ if self.manifest.source == "bundled": return True try: @@ -550,6 +590,9 @@ class PluginContext: # active profile's consent state instead. return plugin_capability_granted(self.plugin_id, "tools.override", config=cfg) + # Fail-closed by construction: any failure to read consent state inside plugin_capability_granted + # returns False. The profile-scoped config is passed through so a multi-profile process consults THIS + # manager's home, never the active profile's (#65593 constraint). def inject_message( self, content: str, role: str = "user", *, session_key: str | None = None, ) -> bool: @@ -707,6 +750,11 @@ class PluginContext: # per-home manager teardown emptied it for the WHOLE process and disabled sign-in until # restart — so upsert and keep it out of reverse-order teardown (``persistent=True``). try: + # A per-home manager is torn down routinely (profile-scoped dashboard activity, force + # re-discovery), and disposing this registration on that teardown emptied the auth registry for + # the WHOLE process, permanently disabling sign-in until restart (#91701). The handle still + # disposes explicitly (identity- conditional), and a forced re-discovery rotates the provider in + # place via the upsert. register_global_provider(provider) except (TypeError, ValueError) as e: logger.warning("Plugin '%s' failed to register dashboard-auth provider %r: %s", @@ -1112,8 +1160,17 @@ class PluginManager(PluginLoaderMixin, PluginDispatchMixin, PluginLedgerMixin): # is keyed per (hermes_home, plugin_id) and every inverse is identity-conditional — one # profile's unload can never clear another's. Persistent registrations that survived an # unload-all park in ``_persistent_carryover`` until force re-discovery evicts the stale ones. + # Registration handles are kept both per plugin (ownership lookup) and globally (reverse-order + # teardown for overrides spanning plugins). Registry overlays keyed by scope_key (see + # tools/registry.py and gateway/platform_registry.py) carry the profile dimension; anything still + # process-global is guarded by the identity checks. TODO(#64178): extend explicit profile keying to + # any remaining process-global slots when the symmetric force-reload lands. self._ownership_ledger: Dict[str, List[PluginRegistration]] = {} self._registration_order: List[PluginRegistration] = [] + # Force re-discovery drains this via _evict_stale_persistent_registrations(): entries whose plugin + # re-registered the same (kind, key) are kept (the upsert rotated them in place), the rest are + # disposed so a disabled/removed auth plugin's provider does not outlive its plugin (#91701 + # follow-up). self._persistent_carryover: List[PluginRegistration] = [] # Deferred platforms whose client tools registered at discovery (see # _register_deferred_platform_tools): imported package (don't re-execute on materialize) @@ -1160,12 +1217,19 @@ class PluginManager(PluginLoaderMixin, PluginDispatchMixin, PluginLedgerMixin): self._discover_and_load_inner() # Persistent registrations survived the unload-all; now that plugins re-registered, # dispose the ones whose plugin did not come back. + # Now that plugins have had their chance to re-register, dispose the ones whose plugin did + # not come back (disabled, removed, or omitted from this discovery pass) so e.g. a disabled + # auth plugin's provider does not stay live process-wide until restart. See #91701. self._evict_stale_persistent_registrations() # load_hermes_dotenv() ran at import, before plugin secret sources existed: re-pull. + # Plugin secret sources register during discover; the initial load_hermes_dotenv() already + # ran at import time. Re-pull so the first process sees plugin backends (tracking #64177). self._refresh_secret_sources_after_discovery() if force: # config.yaml shell hooks / outbound webhooks live in ``_hooks`` but are # config-owned; unload() wiped them and cannot restore them. + # Re-register so force-reload is symmetric (#60036; tracking #64178 — salvaged from PR + # #64188; outbound webhooks added per #92682 review). self._re_register_config_hooks_after_force() except BaseException: self._discovered = False @@ -1573,7 +1637,13 @@ def get_portable_mcp_server_names_nowait() -> "set[str]": def _delivery_manager() -> PluginManager: """Active manager, lazily discovering if it never ran — delivery must not depend on WHICH surface imported us (dashboards/TUI/cron never import model_tools). ``getattr`` default - ``True`` leaves test doubles untouched.""" + ``True`` leaves test doubles untouched. + + Hook/middleware delivery must not depend on WHICH surface imported us: dashboards, TUI slash workers, + query mode, and cron delivery paths never import ``model_tools`` (whose import side-effect is the + discovery trigger on the interactive CLI path), so hooks registered by user plugins were silently dead + on those surfaces (#50776, #67597, #67890, #50937; tracking #64178 — salvaged from PR #64188). + """ manager = get_plugin_manager() if not getattr(manager, "_discovered", True): _join_background_discovery() @@ -1582,7 +1652,16 @@ def _delivery_manager() -> PluginManager: def invoke_hook(hook_name: str, **kwargs: Any) -> List[Any]: - """Invoke a lifecycle hook (lazy-discovers first); return non-``None`` callback results.""" + """Invoke a lifecycle hook (lazy-discovers first); return non-``None`` callback results. + + Hot-path / observer hooks in ``_HOOK_TIMEOUT_BOUNDED_HOOKS`` and the policy hook ``pre_tool_call`` are + bounded by ``plugins.hook_callback_timeout`` (default 30s). On timeout the worker is abandoned (not + joined) so we do not reintroduce the #6622 hang. Timed-out or still-running ``pre_tool_call`` callbacks + fail closed with a block directive; other bounded hooks fail open (skip). + Ensures plugins are discovered on first invocation so callers in processes that never explicitly call + ``discover_plugins()`` (gateway platform events, TUI slash workers, query mode, cron) still fire + callbacks registered by user plugins (tracking #64178). + """ return _delivery_manager().invoke_hook(hook_name, **kwargs) @@ -1592,13 +1671,21 @@ def render_system_prompt_sections(session_info: Mapping[str, Any]) -> List[Rende def invoke_middleware(kind: str, **kwargs: Any) -> List[Any]: - """Invoke registered middleware callbacks (lazy-discovers like :func:`invoke_hook`).""" + """Invoke registered middleware callbacks (lazy-discovers like :func:`invoke_hook`). + + Lazy-discovers plugins on first use — same delivery-parity guarantee as :func:`invoke_hook` (tracking + #64178). + """ return _delivery_manager().invoke_middleware(kind, **kwargs) def has_middleware(kind: str) -> bool: """True when middleware is registered for ``kind``; lazy-discovers first since callers gate - :func:`invoke_middleware` on it.""" + :func:`invoke_middleware` on it. + + Lazy-discovers first: callers use this as a gate before :func:`invoke_middleware`, so a pre-discovery + ``False`` here would silently skip delivery on surfaces that never ran discovery (#64178). + """ manager = _delivery_manager() method = getattr(manager, "has_middleware", None) if callable(method): @@ -1607,7 +1694,10 @@ def has_middleware(kind: str) -> bool: def has_hook(hook_name: str) -> bool: - """True when a loaded plugin handles a hook (lazy-discovers first, like :func:`has_middleware`).""" + """True when a loaded plugin handles a hook (lazy-discovers first, like :func:`has_middleware`). + + Lazy-discovers first — same gate-before-invoke rationale as :func:`has_middleware` (tracking #64178). + """ return _delivery_manager().has_hook(hook_name) @@ -1807,7 +1897,17 @@ def get_plugin_error_classification( """Consult ``transform_api_error_classification`` hooks BEFORE the built-in classifier. Run-all-then-pick-first: the first valid result in registration order wins, losing valid results warn (conflicts visible, not shadowed). Returns a sanitized dict (``reason`` -> ``FailoverReason``, - hint flags -> bool, ``message`` capped at 500) or ``None``. Privacy: inputs may be unredacted.""" + hint flags -> bool, ``message`` capped at 500) or ``None``. Privacy: inputs may be unredacted. + + A callback returns ``None`` to decline, or a dict with a required ``"reason"`` (a + :class:`agent.error_classifier.FailoverReason` member or its string name) plus optional recovery-hint + overrides. Dispatch is run-all-then-pick-first: ``invoke_hook`` runs every registered callback with + failures isolated, then the first result carrying a valid reason wins in registration order — mirroring + :func:`get_pre_tool_call_block_message`, invalid or irrelevant returns are silently ignored so a + misbehaving plugin degrades to a no-op. When more than one callback returns a valid classification, the + losing results are skipped with a runtime warning (the #64714 skipped-transform rule) so conflicting + provider plugins are visible in logs instead of silently shadowed. + """ from agent.error_classifier import FailoverReason hook_results = invoke_hook( "transform_api_error_classification", provider=provider, model=model, diff --git a/hermes_cli/plugins_cmd.py b/hermes_cli/plugins_cmd.py index b64792c7e7..16a78d336b 100644 --- a/hermes_cli/plugins_cmd.py +++ b/hermes_cli/plugins_cmd.py @@ -338,7 +338,11 @@ def _missing_env_specs(manifest: dict) -> list[dict]: def _print_python_dependencies(manifest: dict, console) -> None: """Print declared ``python_dependencies`` with an install hint — Hermes never auto-installs - plugin pip dependencies.""" + plugin pip dependencies. + + See #64165. + See #15220, #64165. + """ deps = manifest.get("python_dependencies") or [] if not isinstance(deps, list): return @@ -809,6 +813,7 @@ def cmd_update(name: str) -> None: # Re-consent when the new version declares capabilities the granted set lacks or the # declared set changed; additions stay ungranted until the user says yes (fail closed). + # See #64228. updated_manifest = _read_manifest(target) plugin_id = updated_manifest.get("name") or target.name declared_caps = _declared_capabilities_from_manifest(updated_manifest, plugin_id) @@ -849,6 +854,8 @@ def _rescan_after_update(target: Path, name: str, console) -> None: def _post_pull_housekeeping(target: Path, console) -> None: """After ``git pull``: drop stale ``__pycache__`` and copy any new ``.example`` files.""" + # Same stale-bytecode class as the main checkout (#6207/#60242): the pull just changed .py files under + # this plugin dir, so drop any __pycache__ compiled from the previous revision. _clear_plugin_bytecode(target) _copy_example_files(target, console) @@ -1021,6 +1028,7 @@ def cmd_enable(name: str, allow_tool_override: Optional[bool] = None) -> None: return # When the manifest declares capabilities the consent screen is the canonical grant path # (it covers tools.override too); the legacy prompt then only runs on an explicit flag. + # See #64228. declared_caps = _declared_capabilities_for_key(key) if declared_caps: _run_capability_consent(console, key, declared_caps, context="enable") @@ -1033,6 +1041,7 @@ def cmd_enable(name: str, allow_tool_override: Optional[bool] = None) -> None: # ── Capability consent flow ────────────────────────────────────────────────── +# ── Capability consent flow (#64228) ───────────────────────────────────────── def _declared_capabilities_from_manifest(manifest: dict, plugin_name: str = "?") -> list: """Extract + normalize the ``capabilities:`` declaration from a manifest.""" from hermes_cli.plugin_capabilities import parse_declared_capabilities @@ -1462,6 +1471,7 @@ def cmd_toggle() -> None: # key (``web/firecrawl``) while the name may differ (``web-firecrawl``); persisting the bare # name let plugins.disabled drift so "explicit disable wins" kept a plugin off forever. plugin_keys = [entry[5] for entry in entries] + # Keys keep every surface aligned. See #40190. plugin_labels = [ (f"{name} \u2014 {description}" if description else name) + (" [bundled]" if source == "bundled" else "") for name, _version, description, source, _d, _key in entries @@ -1490,6 +1500,9 @@ def _persist_plugin_selection(plugin_keys, chosen, disabled) -> tuple[bool, set] them) under the canonical key ONLY, so the list can't drift from what ``cmd_enable`` clears. Re-checking also drops any stale legacy bare-leaf disable. """ + # See #40190. + # Persist by canonical key only — never the bare manifest name — so the disabled-list stays aligned with + # cmd_enable / PluginManager (#40190). new_enabled: set = set() new_disabled: set = set(disabled) # preserve existing disabled state for unseen plugins for i, key in enumerate(plugin_keys): @@ -1807,7 +1820,10 @@ def dashboard_update_user_plugin(name: str) -> dict[str, Any]: def _clear_plugin_bytecode(target: Path) -> int: """Remove ``__pycache__`` dirs under a just-updated plugin checkout. Plugin dirs sit outside the repo, so the launch-time bytecode sweep never covers them and stale bytecode after a pull - can ImportError in the next process. Never raises.""" + can ImportError in the next process. Never raises. + + See #60242, #6207. + """ removed = 0 try: for cache_dir in target.rglob("__pycache__"): @@ -1868,7 +1884,14 @@ def _autostash_dirty_tree(git_exe: str, target: Path) -> tuple[bool, str]: def _git_pull_plugin_dir(target: Path) -> tuple[bool, str]: """``git pull --ff-only`` a plugin checkout, autostashing local edits (users patch installed - plugins in place, and a plain ff-only pull would then refuse forever).""" + plugins in place, and a plain ff-only pull would then refuse forever). + + Users tweak installed plugins in place (config constants, small patches), and a plain ``pull --ff-only`` + then aborts with "Your local changes ... would be overwritten by merge" — making the plugin permanently + un-updatable until they hand-run git. Same UX class Factory Droid fixed in v0.188 ("Updating a plugin + marketplace now succeeds when its checkout has local changes"), and the same autostash approach ``hermes + update`` already uses for the main checkout (PR #70161). + """ git_exe = _resolve_git_executable() if not git_exe: return False, "git is not installed or not in PATH." diff --git a/hermes_cli/plugins_dispatch.py b/hermes_cli/plugins_dispatch.py index 2db2d309cb..737708af0e 100644 --- a/hermes_cli/plugins_dispatch.py +++ b/hermes_cli/plugins_dispatch.py @@ -27,6 +27,17 @@ logger = logging.getLogger("hermes_cli.plugins") # Intentionally unbounded: on_session_finalize/reset (last-chance flush — abandon can lose state); # subagent_start (observer); pre_gateway_dispatch (policy gate — neither fail mode is acceptable); # pre/post_approval_* (approval UX has its own timeout); kanban_* (own heartbeat/stale reclaim). +# The goal is to stop a hung Python plugin callback from wedging the conversation loop (#76821) without +# joining the worker (avoids the #6622 ThreadPoolExecutor shutdown hang). Hooks not listed below run +# synchronously to completion. (on_session_start/end stay bounded — they sit on the common session-boundary +# path.) - subagent_start — observer only; blocking delegation belongs in pre_tool_call. Lower frequency +# than tool/LLM hooks. Abandoning is unsafe either way (fail-open skips auth-like checks; fail-closed can +# drop legitimate messages). Prefer finish-or-exception fallthrough. - pre_approval_request / +# post_approval_response — observers only (cannot veto); the approval UX already has its own timeout; not on +# the tool loop hot path. - kanban_task_* — fire after the board DB commit, observers only, in +# dispatcher/worker processes; kanban has its own heartbeat/stale reclaim. Abandon-without-join also leaves +# a daemon thread that may still mutate shared state — safer for value-returning observers than for +# gates/flushes. _HOOK_TIMEOUT_BOUNDED_HOOKS: Set[str] = { "post_tool_call", "transform_terminal_output", "transform_tool_result", "transform_llm_output", "pre_llm_call", "post_llm_call", "pre_api_request", "post_api_request", "api_request_error", @@ -228,6 +239,7 @@ class PluginDispatchMixin: thread.start() if not done.wait(timeout=timeout): # do not join — that would reintroduce the hang with self._hook_timeout_lock: + # See #6622. self._hook_timeout_suppressed_until[callback_key] = ( time.monotonic() + self._hook_timeout_suppression_seconds) logger.warning( @@ -246,7 +258,12 @@ class PluginDispatchMixin: def _remove_plugin_subscriptions(self, owner: str) -> int: """Remove every subscription owned by *owner*; return the count. Queued envelopes re-check - membership per callback, so this also cancels already-snapshotted deliveries.""" + membership per callback, so this also cancels already-snapshotted deliveries. + + TODO(#64229): when the central plugin ownership ledger / registration handles land, route this + owner-tagged bookkeeping through that ledger so per-plugin unload cancels event subscriptions + alongside every other registration surface. This method is the integration seam. + """ removed = 0 with self._event_lock: for event in list(self._subscriptions): diff --git a/hermes_cli/plugins_ledger.py b/hermes_cli/plugins_ledger.py index b4381901b0..32d4759d04 100644 --- a/hermes_cli/plugins_ledger.py +++ b/hermes_cli/plugins_ledger.py @@ -30,6 +30,7 @@ class PluginRegistration: # Process-global host infrastructure (e.g. dashboard-auth providers): kept out of ``_registration_order`` # so unload-all cannot dispose it, but still disposed by a *targeted* unload and evicted on force # re-discovery when the plugin no longer re-registers it. + # See #91701. persistent: bool = False _disposed: bool = field(default=False, init=False, repr=False) _on_dispose: Optional[Callable[["PluginRegistration"], None]] = field(default=None, init=False, repr=False) @@ -58,7 +59,10 @@ class PluginLedgerMixin: ) -> PluginRegistration: """Record one registration under its canonical plugin key. ``persistent`` ones (process-global host infrastructure) stay in the ownership ledger for attribution but NOT in ``_registration_order``, so a - routine unload cannot dispose them; the handle still releases on explicit ``dispose()``.""" + routine unload cannot dispose them; the handle still releases on explicit ``dispose()``. + + See #91701. + """ registration = PluginRegistration( kind=kind, key=key, release=release, plugin_key=manifest_key(manifest), persistent=persistent) registration._on_dispose = lambda disposed: self._forget_registrations([disposed]) @@ -89,7 +93,12 @@ class PluginLedgerMixin: def _evict_stale_persistent_registrations(self) -> None: """After re-discovery, dispose parked persistent handles whose plugin did not re-register the same ``(kind, key)``. Re-registered ones are dropped WITHOUT disposing — the same object re-registered would - pass the identity check and evict the live entry.""" + pass the identity check and evict the live entry. + + Persistent registrations (process-global host infrastructure such as dashboard-auth providers) + survive an unload-all by design (#91701); ``_unload_scoped`` parks their handles in + ``_persistent_carryover``. After a re-discovery pass, three cases exist for each parked handle: + """ if not self._persistent_carryover: return parked, self._persistent_carryover = self._persistent_carryover, [] @@ -197,6 +206,7 @@ class PluginLedgerMixin: # Persistent registrations are absent from _registration_order (unload-all keeps them), but a # *targeted* unload is the disable/uninstall path: a disabled auth plugin's provider must NOT stay # live process-wide. + # See #91701. registrations.extend( r for key in target_keys for r in self._ownership_ledger.get(key, []) if r.persistent and r.active ) diff --git a/hermes_cli/plugins_loader.py b/hermes_cli/plugins_loader.py index 176a4ec94e..e725737f91 100644 --- a/hermes_cli/plugins_loader.py +++ b/hermes_cli/plugins_loader.py @@ -132,16 +132,38 @@ class PluginLoaderMixin: """Register a deferred platform's *client* tools without its adapter. Deferring the plugin would otherwise defer its outbound tools too, so CLI/TUI processes (which never materialize platforms) would miss them in ``hermes tools`` / ``platform_toolsets``. Opt-in is explicit via ``provides_tools``; - tools live in a ``tools`` submodule so ``__init__`` stays import-light.""" + tools live in a ``tools`` submodule so ``__init__`` stays import-light. + + A platform plugin can ship two independent things: an inbound adapter (heavy — it imports the + platform SDK) and outbound client tools the agent calls like any other tool. Deferring the plugin + defers both, so in a CLI/TUI process the client tools never register at all: ``resolve_toolset()`` + returns ``[]``, the toolset is missing from the ``hermes tools`` checklist, and even an explicit + ``platform_toolsets`` entry is dropped because the key is unknown. The same tools work in + gateway/web processes only because those materialize every platform at startup (issue #78050). + Opting in is explicit: the manifest must declare ``provides_tools`` (the field the plugin list and + web server already read to name a plugin's tools, per #78538). Keying off the mere presence of a + ``tools.py`` would opt a plugin in by accident — a platform is free to put internal helpers there — + and would leave the contract invisible to anyone reading the manifest. ``tools.py`` remains where + the code is imported from; ``provides_tools`` is what asks for it. A platform that does not declare + the field is untouched and stays fully deferred. + """ from hermes_cli.plugins import PluginContext, _PLUGINS_DEBUG if not manifest.provides_tools: return lookup_key = manifest_key(manifest) + # Never let a client-tool import break discovery — the platform stays deferred and behaves exactly + # as it did before. But a broken tools.py produces the #78050 symptom itself (declared tools missing + # from the session), so this has to be visible without turning on debug logging to find it. Where it + # failed is the first thing an operator needs: nothing registered points at the import or the module + # body, a partial run points at one tool's definition, and a full run that still raised points past + # the registrations entirely. declared = list(manifest.provides_tools) plugin_dir = Path(manifest.path) if manifest.path else None if plugin_dir is None or not (plugin_dir / "tools.py").is_file(): # Declared but undeliverable — staying quiet reproduces the very symptom this fixes. logger.warning( + # Staying quiet here reproduces the exact symptom this path exists to fix — tools the + # manifest promises, silently absent from the session (#78050) — so say so. "Plugin '%s' declares provides_tools %s but has no tools.py; " "those tools will not be available in CLI/TUI sessions.", lookup_key, declared, ) @@ -195,7 +217,14 @@ class PluginLoaderMixin: ) def _warn_python_dependencies(self, manifest: PluginManifest) -> None: - """Warn about missing declared pip dependencies with an install hint — NEVER auto-install.""" + """Warn about missing declared pip dependencies with an install hint — NEVER auto-install. + + See #64165. + python_dependencies is a declaration seam ONLY: Hermes validates and prints the requirements with an + install hint but NEVER auto-installs them. The isolation design (constraints installs vs. vendored + dirs vs. conflict-detection-and-refusal) is an explicitly deferred follow-up — see the round-2 + review on #64165 and #15220. + """ deps = manifest.python_dependencies if not deps: return @@ -212,7 +241,10 @@ class PluginLoaderMixin: logger.debug("Plugin %s python_dependencies satisfied: %s", key, ", ".join(deps)) def _validate_plugin_config_schema(self, manifest: PluginManifest) -> None: - """Warn (never block) on plugins.entries.<id> settings that violate config_schema.""" + """Warn (never block) on plugins.entries.<id> settings that violate config_schema. + + See #64165. + """ if not manifest.config_schema: return plugin_id = manifest_key(manifest) @@ -251,6 +283,7 @@ class PluginLoaderMixin: self._track_tool_override_policy(manifest, module_name) try: # Reuse a deferred platform's already-imported package so its body doesn't run twice. + # See #78050. module = self._predeclared_modules.pop(plugin_key, None) if module is None and manifest.source in {"user", "project", "bundled"}: module = self._load_directory_module(manifest, module_name=module_name) @@ -276,6 +309,9 @@ class PluginLoaderMixin: logger.warning("Failed to load plugin '%s': %s", manifest.name, exc, exc_info=_PLUGINS_DEBUG) # The failure path swept this plugin's whole ledger (not just the registration_start slice), so # discovery-time pre-registrations are gone too. + # There is no live tool left to credit — attribution and the registry agree at zero. Only the + # success path pops _predeclared_tools, so drop the entry here rather than let the bookkeeping + # outlive the load attempt (#78050). if not loaded.enabled: self._predeclared_tools.pop(plugin_key, None) self._plugins[plugin_key] = loaded diff --git a/hermes_cli/plugins_manifest.py b/hermes_cli/plugins_manifest.py index dbb2752d62..fec0acfbad 100644 --- a/hermes_cli/plugins_manifest.py +++ b/hermes_cli/plugins_manifest.py @@ -27,6 +27,7 @@ _VALID_PLUGIN_KINDS: Set[str] = {"standalone", "backend", "exclusive", "platform # Unknown plugin.yaml fields are forward-compat surface: warn (debug for v1 files, warning for v2+) # and continue loading. ``capabilities``/``emits``/``listens``/``hermes``/``depends`` are reserved. +# ── Manifest v2 (#64165) parsing helpers ────────────────────────────────── _KNOWN_MANIFEST_FIELDS: Set[str] = { "name", "version", "description", "author", "requires_env", "provides_tools", "provides_hooks", "kind", "hooks", "label", "optional_env", "platforms", "external_dependencies", @@ -106,7 +107,10 @@ def _manifest_int(raw: object, key: str, warn: str, fallback: Optional[int]) -> def _parse_manifest_v2_fields(data: Mapping, key: str) -> Dict[str, Any]: - """Validate/normalize manifest v2 fields into PluginManifest kwargs (warnings, never failures).""" + """Validate/normalize manifest v2 fields into PluginManifest kwargs (warnings, never failures). + + See #64165. + """ # manifest_version — absent means v1 (supported forever); api_version is the independent API generation. mv = _manifest_int(data.get("manifest_version", 1), key, "Plugin %s: manifest_version %r is not an integer; treating as 1", 1) @@ -161,7 +165,10 @@ def _parse_manifest_v2_fields(data: Mapping, key: str) -> Dict[str, Any]: def validate_config_schema(plugin_id: str, schema: Mapping, settings: Mapping) -> List[str]: - """Return actionable warning strings for settings vs config_schema mismatches (never raises).""" + """Return actionable warning strings for settings vs config_schema mismatches (never raises). + + Never raises; schema mismatches must not block plugin load (#64165). + """ warnings: List[str] = [] if not isinstance(schema, Mapping) or not isinstance(settings, Mapping): return warnings @@ -192,7 +199,10 @@ def validate_config_schema(plugin_id: str, schema: Mapping, settings: Mapping) - def resolve_plugin_load_order(manifests: Mapping[str, "PluginManifest"]) -> List[str]: """Return plugin keys in dependency order: B before A when A requires B; alphabetical ties. A cycle warns and falls back to alphabetical order for all; a missing dependency warns once but never removes the - dependent plugin (loads never hard-fail on advisory deps).""" + dependent plugin (loads never hard-fail on advisory deps). + + See #64165. + """ import graphlib keys = sorted(manifests.keys()) by_name: Dict[str, str] = {} @@ -330,14 +340,18 @@ class PluginManifest: skill_namespace: str = "" # Declared capability ids, normalized to KNOWN ids. Declaration is consent metadata, NOT a grant: live # only via plugins.entries.<id>.granted_capabilities or the legacy allow_* key. + # See #64228. capabilities: List[str] = field(default_factory=list) # Manifest v2 fields — all optional and additive. manifest_version versions the FILE FORMAT (v1 supported # forever); api_version is the runtime plugin API generation (None = current). + # Absent (v1) manifests are fully supported forever. See #64165. manifest_version: int = 1 api_version: Optional[int] = None # Advisory deps [{"id", "version_range"}]: missing ones warn but load; they order the load. requires_plugins: List[Dict[str, Any]] = field(default_factory=list) # Declared pip deps — VALIDATED AND SURFACED ONLY, never auto-installed. + # VALIDATED AND SURFACED ONLY — Hermes never auto-installs these (isolation design for the install seam + # is a deferred follow-up; see #64165 round-2 review and #15220). python_dependencies: List[str] = field(default_factory=list) # Schema for plugins.entries.<id>.settings; mismatches warn, never fail. config_schema: Dict[str, Any] = field(default_factory=dict) diff --git a/hermes_cli/plugins_state.py b/hermes_cli/plugins_state.py index 3d85afcb1c..5e7c7dbb1a 100644 --- a/hermes_cli/plugins_state.py +++ b/hermes_cli/plugins_state.py @@ -23,7 +23,10 @@ _PLUGIN_STATE_LOCKS_GUARD = threading.Lock() def _plugin_relative_segments(key: str) -> tuple[str, ...]: """Validate/split a plugin-relative settings key; global paths, traversal, and core roots are rejected - before any config read.""" + before any config read. + + The public API accepts only relative keys (``endpoint`` or ``retry.policy``). See #64227. + """ if not isinstance(key, str): raise ValueError("Expected a plugin-relative config key string") segments = tuple(key.split(".")) diff --git a/hermes_cli/process_identity.py b/hermes_cli/process_identity.py index c4bdafb426..4247f74cd3 100644 --- a/hermes_cli/process_identity.py +++ b/hermes_cli/process_identity.py @@ -136,7 +136,11 @@ def _ledger_path() -> Path: def _read_ledger(path: Path) -> Optional[list[dict]]: - """Entries list, ``[]`` for empty/missing, ``None`` for CORRUPT (never silently an empty roster).""" + """Entries list, ``[]`` for empty/missing, ``None`` for CORRUPT (never silently an empty roster). + + Mirrors the #89298 contract: corrupt is a distinct state that must never be silently treated as an empty + roster. + """ try: text = path.read_text(encoding="utf-8") except FileNotFoundError: @@ -245,6 +249,8 @@ def _append_entry(entry: LedgerEntry) -> bool: Serialized under ``_LEDGER_LOCK`` with an atomic tmp+replace; no writer touches the file outside this function. + + See #91660. """ path = _ledger_path() with _LEDGER_LOCK: @@ -294,7 +300,13 @@ def register_child(pid: int, purpose: str, *, project_root: Optional[Path] = Non def ledger_entries(*, project_root: Optional[Path] = None) -> list[dict]: - """Live-verified ledger entries for THIS install (a corrupt ledger is quarantined, read as empty).""" + """Live-verified ledger entries for THIS install (a corrupt ledger is quarantined, read as empty). + + Entries whose ``(pid, create_time)`` no longer matches a live process are excluded (PID reuse reads as + dead, thanks to the create-time pair). A corrupt ledger is quarantined and read as empty — identical + philosophy to the backend-ownership fix (#89298): never let corruption erase or fake a roster; never let + it block the caller either. + """ want_install = install_id(project_root) with _LEDGER_LOCK: entries = _read_ledger_or_quarantine(_ledger_path()) diff --git a/hermes_cli/profile_cmd.py b/hermes_cli/profile_cmd.py index 1157d95ae7..62759140ee 100644 --- a/hermes_cli/profile_cmd.py +++ b/hermes_cli/profile_cmd.py @@ -37,6 +37,8 @@ def _env_file_has_key(env_path: Path, key: str) -> bool: if not env_path.is_file(): return False try: + # .env is written as UTF-8 everywhere in the codebase, but a Notepad-edited file can carry a BOM — + # read as utf-8-sig so the first key isn't hidden behind U+FEFF (#62617). for raw in env_path.read_text(encoding="utf-8-sig").splitlines(): line = raw.strip() if line and not line.startswith("#") and line.split("=", 1)[0].strip() == key: diff --git a/hermes_cli/profile_describer.py b/hermes_cli/profile_describer.py index f21e0a3854..f0eda23821 100644 --- a/hermes_cli/profile_describer.py +++ b/hermes_cli/profile_describer.py @@ -149,6 +149,7 @@ def describe_profile(profile_name: str, *, overwrite: bool = False, timeout: Opt try: # call_llm applies auxiliary.profile_describer.* config (provider/model/base_url, # extra_body, reasoning_effort, retries); the direct-create path dropped extra_body. + # See #35566. resp = call_llm( task="profile_describer", messages=[{"role": "system", "content": _SYSTEM_PROMPT}, {"role": "user", "content": user_msg}], diff --git a/hermes_cli/profiles.py b/hermes_cli/profiles.py index 8e104449ff..035cbfd3f8 100644 --- a/hermes_cli/profiles.py +++ b/hermes_cli/profiles.py @@ -100,6 +100,7 @@ def _clone_all_copytree_ignore(source_dir: Path): # infrastructure (``state.db``, ``logs/``, ``auth.*``, other profiles) is deliberately # absent so the export stays a portable, credential-free snapshot. Add new artifacts here # when introduced in ``hermes_constants``. +# See #58394. _DEFAULT_EXPORT_INCLUDE_ROOT = frozenset({ # Configuration / persona "config.yaml", "SOUL.md", "MEMORY.md", "USER.md", "todo.json", @@ -168,7 +169,10 @@ def _missing_profile_error(canon: str) -> FileNotFoundError: def normalize_profile_name(name: str) -> str: """Canonical profile id used on disk and in ``-p`` argv: lowercase, ``default`` matched case-insensitively. Dashboards/tools may pass title-cased labels — normalize before - validation, assignment, and subprocess spawn.""" + validation, assignment, and subprocess spawn. + + Named profiles are stored lowercase under ``profiles/<id>/``. See #18498. + """ if not isinstance(name, str): name = str(name) stripped = name.strip() @@ -181,7 +185,12 @@ def normalize_profile_name(name: str) -> str: def validate_profile_name(name: str) -> None: """Raise ``ValueError`` unless *name* is a valid profile id (strict as-given lowercase — - normalize mixed-case input first) and not in ``_RESERVED_NAMES``; ``default`` passes.""" + normalize mixed-case input first) and not in ``_RESERVED_NAMES``; ``default`` passes. + + Callers that accept mixed-case or title-cased input from users (dashboard UI, CLI args) should call + :func:`normalize_profile_name` first. This separation keeps validate honest about what the on-disk + directory name must look like, while ingress-point normalization handles UX flexibility (see #18498). + """ if name == "default": return # special alias for ~/.hermes if not _PROFILE_ID_RE.match(name): @@ -535,7 +544,10 @@ def _check_gateway_running(profile_dir: Path) -> bool: def _served_by_running_multiplexer(profile_name: str) -> bool: """True when the live default gateway multiplexes ``profile_name`` (such a profile has no - gateway.pid of its own, so ``_check_gateway_running`` alone reports it stopped).""" + gateway.pid of its own, so ``_check_gateway_running`` alone reports it stopped). + + Single shared lookup with the named-profile start guard and cron liveness (#97120). + """ try: from hermes_cli.gateway import named_profile_served_by_running_multiplexer return named_profile_served_by_running_multiplexer(profile_name) @@ -628,6 +640,7 @@ def write_profile_meta( existing.pop("display_name", None) # Atomic write: bare open("w") truncates before the dump, and the read path swallows # parse errors as {}, so a crashed write would silently drop unspecified fields. + # See #51356. from utils import atomic_yaml_write atomic_yaml_write(path, existing, sort_keys=False) @@ -904,7 +917,15 @@ def seed_profile_skills(profile_dir: Path, quiet: bool = False) -> Optional[dict def backfill_profile_envs(quiet: bool = False) -> List[str]: """Give every named profile predating per-profile ``.env`` one (copy of the default's, or - the placeholder header). Never overwrites an existing profile ``.env``.""" + the placeholder header). Never overwrites an existing profile ``.env``. + + Profiles created before the dashboard/CLI started seeding a ``.env`` (PR #44792) have none, so once the + Channels/Keys endpoints became profile-scoped those profiles stopped inheriting the root install's + credentials and showed everything as unconfigured. To avoid breaking anyone on update, copy the DEFAULT + install's ``.env`` into each named profile that lacks one — that preserves the effective credentials + those profiles were already running with (they previously read the root ``.env`` via the process + environment). Users can then diverge per profile from there. + """ backfilled: List[str] = [] default_env = _get_default_hermes_home() / ".env" for entry in _iter_named_profile_dirs(): @@ -1151,6 +1172,7 @@ def delete_profile(name: str, yes: bool = False) -> Path: # deliberately not stopped above; on Windows its handles fail rmtree with WinError 32. # Inside serve (DELETE /api/profiles/<name>) the handles live here; from the CLI no-op. with contextlib.suppress(Exception): # best-effort: never block the delete on the release path + # 2c. See #88347. from plugins.memory.holographic.store import MemoryStore as _MemoryStore _released = _MemoryStore.release_all_under(profile_dir) if _released: @@ -1192,7 +1214,15 @@ def _s6_runtime_manager(): def _maybe_register_gateway_service(profile_name: str) -> None: """Register a profile's gateway with s6 inside the container. Best-effort: profile - creation must not fail over a supervision-tree hiccup; `gateway start` re-registers.""" + creation must not fail over a supervision-tree hiccup; `gateway start` re-registers. + + Port selection: each supervised profile gateway loads its own ``HERMES_HOME`` and binds the port + resolved by ``gateway/config.py`` from that profile's environment — ``API_SERVER_PORT`` (or + ``platforms.api_server.extra.port`` in the profile's ``config.yaml``), defaulting to 8642. There is no + ``[gateway] port`` key and no Python-side allocator (PR #30136 review item I5 retired the + SHA-256-derived range [9200, 9800) as dead code), so two profiles that both leave the port at its + default will both try to bind 8642 — give each profile a distinct ``API_SERVER_PORT`` in its ``.env``. + """ mgr = _s6_runtime_manager() if mgr is None: return @@ -1373,6 +1403,7 @@ def _profile_export_directory() -> Path: # Fail closed: writing a secret-bearing archive into a source tree is the incident this # helper prevents; a stderr warning would not stop a scripted export. raise ValueError( + # See #92457. "No safe automatic export destination: every candidate directory is " "inside a Git checkout. Provide an explicit output path outside the " "checkout (CLI: -o /path/outside/repo/profile.tar.gz)." @@ -1404,7 +1435,14 @@ def get_profile_export_path(name: str, *, timestamp: Optional[str] = None) -> Pa def _default_export_ignore(root_dir: Path): """copytree ignore for the default-profile export: root-level allow-list (``_DEFAULT_EXPORT_INCLUDE_ROOT``) plus universal exclusions. Surviving text files are - then force-redacted by :func:`_scrub_export_secrets`.""" + then force-redacted by :func:`_scrub_export_secrets`. + + * **Root-level allow-list** — only entries whose name appears in ``_DEFAULT_EXPORT_INCLUDE_ROOT`` + survive. Everything else (such as an unrelated ``x11-dev/`` directory in a Docker deployment where + HERMES_HOME equals the cwd) is excluded. Blacklisting was tried first and proved unable to anticipate + every non-Hermes file the user may have lying alongside HERMES_HOME (#58394). * **Universal exclusions + at any depth** — ``__pycache__``, sockets, temp files; plus npm lockfiles, which may appear at the root. + """ def _ignore(directory: str, contents: list) -> set: # Universal exclusions (any depth) plus npm lockfiles that can appear at root. @@ -1641,7 +1679,14 @@ def rename_profile(old_name: str, new_name: str) -> Path: def resolve_profile_env(profile_name: str) -> str: """Resolve a profile name to a HERMES_HOME path string. Called early in the CLI entry - point, before hermes modules are imported, to set HERMES_HOME.""" + point, before hermes modules are imported, to set HERMES_HOME. + + When HERMES_HOME is already set, the configured spelling IS the launch root (it may be a + junction/symlink alias of the platform default). Keep that spelling so profile re-home does not destroy + the launcher's lexical provenance -- the subprocess sanitizer needs it to match Hermes-owned PYTHONPATH + entries written in the same spelling (#82581 junction follow-up). Physically the paths are identical + (junction-transparent); only the spelling is preserved. + """ canon = _canon_valid(profile_name) env_home = os.environ.get("HERMES_HOME", "").strip() if env_home: diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 3e91e69f14..20edd8837a 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -261,7 +261,13 @@ def is_official_openai_host(base_url: str) -> bool: """True when *base_url* points at OpenAI's official API host family. Hostname-parsed matching only — never substring — so lookalike hosts (``api.openai.com.attacker.test``) and path-segment spoofs (``proxy.test/api.openai.com/v1``) are rejected; a genuine ``*.api.openai.com`` - subdomain requires control of openai.com DNS.""" + subdomain requires control of openai.com DNS. + + A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS, so the dot-suffix match + does not reopen the #32243 spoofing hole. Delegates to ``utils.base_url_host_matches``, which owns the + exact-or-dot-suffix hostname contract (userinfo/port stripped, lowercased, trailing dot removed) — one + implementation, not two. + """ return base_url_host_matches(base_url, "api.openai.com") @@ -283,6 +289,9 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: return None url_lower = base_url.rstrip("/").lower() hostname = base_url_hostname(base_url) + # Exact-hostname matching only — never bare substring — so lookalike hosts + # (api.openai.com.attacker.test) and path-segment spoofs (proxy.test/api.openai.com/v1) are NOT treated + # as the real endpoint. (#32243) if hostname == "api.kimi.com" and "/coding" in url_lower: return "anthropic_messages" if hostname == "api.anthropic.com" or url_lower.endswith("/anthropic"): @@ -290,6 +299,9 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: # Official OpenAI host family (canonical + us./eu. data-residency hosts) mandates Responses; # the shared predicate keeps this in lockstep with catalog filtering and listing authority. if is_official_openai_host(base_url) or hostname in _RESPONSES_NATIVE_HOSTS: + # Ramp Router (api.router.com) is Responses-native: reasoning-effort validation, reasoning + # summaries, and prompt caching live on /v1/responses, and /v1/chat/completions is only a minimal + # compatibility shim (docs.router.com/api/endpoint). Exact-hostname match per #32243. return "codex_responses" if hostname.startswith("bedrock-runtime.") and base_url_host_matches(base_url, "amazonaws.com"): return "bedrock_converse" @@ -374,6 +386,8 @@ def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, if not requested or not custom_providers or not isinstance(custom_providers, list): return None first_valid: Optional[ProviderDef] = None + # If the stored provider is the bare string "custom" (corrupt state from a prior model-switch bug), fall + # back to the first custom provider entry so existing configs self-heal. (GH #17478) for entry in custom_providers: if not isinstance(entry, dict): continue diff --git a/hermes_cli/pt_input_extras.py b/hermes_cli/pt_input_extras.py index 120e28fa1b..220e0baf82 100644 --- a/hermes_cli/pt_input_extras.py +++ b/hermes_cli/pt_input_extras.py @@ -6,6 +6,7 @@ from __future__ import annotations # on: CapsLock=64, NumLock=128, both=192. Every fixed-modifier CSI-u (and legacy CSI-tilde / # CSI-letter) registration therefore needs lock-offset twins, or those events leak into the prompt # as literal text. The xterm modifyOtherKeys ``ESC[27;N;CP~`` encoding never carries lock bits. +# See #88221, #89651. _LOCK_BIT_OFFSETS = (0, 64, 128, 192) @@ -57,6 +58,13 @@ def _install(build, *, overwrite: bool) -> int: def install_keypress_data_normalization() -> int: """Normalize KeyPress data for extended-key aliases that map to a single plain character (Shift+Space → ``' '``, Shift+letter → uppercase, keypad digits/operators). + + Root cause of #88071: ``Vt100Parser._call_handler`` builds ``KeyPress(key, match.group(0))`` — the *key* + is correctly remapped by ``ANSI_SEQUENCES``, but the *data* field still carries the full raw escape text + (e.g. ``"\\x1b[32;2u"``). prompt_toolkit's default character-insert binding (``self-insert``, + ``basic.py``) inserts ``event.data``, so the raw CSI bytes land in the prompt buffer. For a plain space + both fields are ``' '`` so it is invisible; for any mapped extended sequence the escape text is what + gets inserted. """ try: import prompt_toolkit.input.vt100_parser as _vt100_mod @@ -104,7 +112,13 @@ def install_shift_enter_alias() -> int: def install_ctrl_enter_alias() -> int: - """Map Ctrl+Enter to (Escape, ControlM); otherwise Kitty/mintty/xterm over SSH insert raw CSI.""" + """Map Ctrl+Enter to (Escape, ControlM); otherwise Kitty/mintty/xterm over SSH insert raw CSI. + + Stock prompt_toolkit maps only the tilde form ``\\x1b[27;5;13~`` (to plain ``Keys.ControlM``, which this + deliberately overwrites — same bug-fix rationale as install_shift_enter_alias). Without this alias, + Kitty/mintty/xterm-with-modifyOtherKeys users over SSH never get a Ctrl+Enter newline — the keystroke + arrives as a raw CSI sequence that falls through to the default character-insert handler. See #22379. + """ return _install_enter_alias(5) @@ -154,6 +168,28 @@ def install_modify_other_keys_aliases() -> int: Ctrl+A/C/D/... leak as text. Installs Ctrl/Alt/Shift letters, digits, symbols, multi-modifier combos, lock-bit variants, CSI-u Esc, modified Enter/Tab/Backspace/Space and Kitty functional keys. ``setdefault`` semantics: existing mappings (incl. the Shift/Ctrl+Enter aliases) win. + + (#56684, #86866, #87390). + * **Ctrl+letter** (a–z): ``ESC[27;5;<codepoint>~`` and ``ESC[<codepoint>;5u`` → ``Keys.ControlA`` .. * + **Ctrl+digit** (0–9): same formats → ``Keys.Control0`` .. * **Ctrl+symbol** (``[`` ``\\`` ``]`` ``^`` + ``_`` `` `` ``@``): same formats → the same ``Keys`` value the raw control byte maps to. * + **Alt+letter** (a–z, A–Z): ``ESC[27;3;<codepoint>~`` and ``ESC[<codepoint>;3u`` → ``(Keys.Escape, + <letter>)`` — matching how prompt_toolkit handles a bare ``ESC`` followed by a character. * + **Shift+letter** (a–z): → the uppercase character. * **Multi-modifier letters** (Shift+Alt=4, + Ctrl+Shift=6, Ctrl+Alt=7, Ctrl+Alt+Shift=8): normalized onto the same targets — Ctrl-bearing combos + behave as the Ctrl key (Alt adds an ``Escape`` prefix), matching how dte/kakoune normalize these + protocols. * **Lock-bit variants**: every CSI-u mapping above is also installed with the CapsLock (64) + and NumLock (128) bits ORed into the modifier parameter — kitty/ghostty include them while a lock is on, + and without the variants every key combo dies with the lock enabled (``ESC[99;133u`` instead of + ``ESC[99;5u``, #89651). * **Esc key**: ``ESC[27u`` / ``ESC[27;<mod>u`` (Kitty disambiguate mode reports + Esc this way, #56684) → ``Keys.Escape``. * **Modified Enter/Tab/Backspace/Space**: Alt+Enter → the + Alt+Enter newline tuple; Shift+Tab → ``BackTab``; Ctrl+Tab → plain Tab; Ctrl/Alt+Backspace → ``(Escape, + ControlH)`` (backward-kill-word, matching the Ink TUI and Desktop, #78285); Shift+Backspace → plain + backspace; Shift+Space → a plain space (#86866); Alt+Space → ``(Escape, " ")``. * **Kitty functional + keys** (Private Use Area codepoints): keypad keys → their non-keypad equivalents (KP_ENTER → Enter, KP_4 + → '4', KP_LEFT → Left, …); F13–F24 → ``Keys.F13``..``F24``; lock/media/ modifier-event keys → + ``Keys.Ignore`` so they are consumed instead of leaking as literal text. kitty emits these CSI-u forms + even in legacy mode for keys that have no legacy encoding. """ return _install(_modify_other_keys_aliases, overwrite=False) @@ -163,6 +199,11 @@ def _modify_other_keys_aliases(ANSI_SEQUENCES: dict, Keys) -> dict[str, object]: aliases: dict[str, object] = {} _put = aliases.setdefault + # Kitty CSI-u encodes CapsLock/NumLock state as extra modifier bits (caps=64, num=128) ORed into the + # parameter: with NumLock on, Ctrl+C arrives as ESC[99;133u (5 + 128) instead of ESC[99;5u. Terminals + # that report these bits (kitty, ghostty) break every key combo while a lock is on (#89651) unless the + # lock variants are mapped too. The xterm modifyOtherKeys encoding never carries the lock bits, so only + # the CSI-u form needs them. def _install_paired(modifier: int, mapping: dict) -> None: """Both modifyOtherKeys (ESC[27;N;CP~, never for mod 1) and CSI-u (ESC[CP;Nu + lock twins).""" for codepoint, key_val in mapping.items(): diff --git a/hermes_cli/pty_session.py b/hermes_cli/pty_session.py index 4c2a606d07..d5cb58a517 100644 --- a/hermes_cli/pty_session.py +++ b/hermes_cli/pty_session.py @@ -110,6 +110,7 @@ class PtySession: pass try: # bridge.close() joins the child — blocking; keep it off the event loop. + # See #53227. await asyncio.to_thread(self.bridge.close) except Exception: pass @@ -148,6 +149,7 @@ class PtySessionRegistry: if len(self._sessions) >= self._max: self._reap_one_idle_or_raise() # PTY spawn does blocking fork/exec work — keep it off the event loop. + # See #53227. bridge = await asyncio.to_thread(spawn) session = PtySession(key, bridge, buffer_cap=self._buffer_cap, read_timeout=self._read_timeout) await session.start() diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index a072a33d03..cc8c6e2ff7 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -75,7 +75,10 @@ def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider Custom while ``model.provider`` still names a previous provider, so non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a local-server alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint falls through to OpenRouter): a stale OpenRouter/Z.ai base_url - cannot hijack local sessions.""" + cannot hijack local sessions. + + See #14676. + """ cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() return bool(bu) and (cfg_provider_norm == "custom" or _resolves_to_custom(cfg_provider_norm) @@ -101,7 +104,18 @@ def _detect_api_mode_for_url(base_url: str) -> Optional[str]: """Auto-detect api_mode from the resolved base URL, or None. Exact-hostname matches reject lookalike subdomains (api.anthropic.com.attacker.test) and path-segment spoofing (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. us./eu. data-residency hosts) - need Responses for GPT-5.x tool calls with reasoning.""" + need Responses for GPT-5.x tool calls with reasoning. + + - Direct api.anthropic.com endpoints must use the native Messages API (``/v1/messages``). Anthropic also + exposes an OpenAI-compat ``/chat/completions`` shim on the same host, but Pro/Max OAuth subscriptions + are only billed against the native Messages route; hitting the shim accounts against a separate "extra + usage" pool that is empty by default and surfaces as HTTP 400 "You're out of extra usage." See issue + #32243. - Third-party Anthropic-compatible gateways (MiniMax, Zhipu GLM, LiteLLM proxies, etc.) + conventionally expose the native Anthropic protocol under a ``/anthropic`` suffix — treat those as + ``anthropic_messages`` transport instead of the default ``chat_completions``. - Kimi Code's + ``api.kimi.com/coding`` endpoint also speaks the Anthropic Messages protocol (the /coding route accepts + Claude Code's native request shape). + """ normalized = (base_url or "").strip().lower().rstrip("/") hostname = base_url_hostname(base_url) mandated = _HOST_MANDATED_API_MODES.get(hostname) or ("codex_responses" if is_official_openai_host(base_url) else None) @@ -109,6 +123,10 @@ def _detect_api_mode_for_url(base_url: str) -> Optional[str]: return mandated path = urlparse(normalized).path.rstrip("/") if path.endswith(("/anthropic", "/anthropic/v1")) or (hostname == "api.kimi.com" and "/coding" in normalized): + # Direct native Anthropic host: realign with providers.determine_api_mode, which already maps this + # host to anthropic_messages. The exact-hostname match rejects lookalike subdomains + # (api.anthropic.com.attacker.test) and path-segment spoofing (proxy.test/api.anthropic.com/v1). + # (#32243) return "anthropic_messages" return None @@ -199,6 +217,10 @@ def _api_key_provider_api_mode(provider: str, model_cfg: Dict[str, Any], api_key if provider == "copilot": return _copilot_runtime_api_mode(model_cfg, api_key, target_model=effective_model) if provider in ("xai", "actual"): + # Ramp Router: Responses-native host — /v1/chat/completions is only a minimal compatibility shim, + # while reasoning and caching support live on /v1/responses (docs.router.com/api/endpoint). Mirrors + # the host_mandated_api_mode clause in hermes_cli/providers.py so the runtime resolver stays in + # lockstep. Exact hostname per #32243. return "codex_responses" return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=opencode_by_model) @@ -814,6 +836,10 @@ def _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, targe yield _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) yield _tag(_resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model), requested_provider) + # If provider is "auto" (or unset) but config.yaml has an explicit base_url pointing at a custom/local + # endpoint (e.g. Ollama at localhost:11434), route through the OpenAI-compatible resolver instead of + # letting resolve_provider() pick up an ANTHROPIC_API_KEY or OPENAI_API_KEY from the environment and + # send the request to a cloud API. Fixes #3846. if not explicit_base_url and not explicit_api_key: yield _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index 400d4bc204..2ce619a3ab 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -134,6 +134,9 @@ def _resolve_openrouter_runtime( ) base_url = ((explicit_base_url or "").strip() or env_custom_base_url or (cfg_base_url.strip() if use_config_base_url else "") or env_openrouter_base_url or OPENROUTER_BASE_URL).rstrip("/") + # Choose API key based on whether the resolved base_url targets OpenRouter. When hitting OpenRouter, + # prefer OPENROUTER_API_KEY (issue #289). When hitting a custom endpoint (e.g. Z.ai, local LLM), prefer + # OPENAI_API_KEY so the OpenRouter key doesn't leak to an unrelated provider (issues #420, #560). is_openrouter_url = base_url_host_matches(base_url, "openrouter.ai") # Explicitly-configured OpenRouter mirrors (OPENROUTER_BASE_URL + provider=openrouter) still # count as OpenRouter for key selection. diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 48f9f64428..29eb851a69 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -348,6 +348,10 @@ def _try_resolve_from_custom_pool( # Legacy configs used short placeholder keys ('123', 'm') for local no-auth # services; has_usable_secret's 4-char floor rejects them. Every other path # substitutes "no-key-required" for a loopback endpoint — this was the one gap. + # Every OTHER resolution path in this file already substitutes "no-key-required" for a + # loopback endpoint with no usable secret (the config-based custom_providers fallback a few + # hundred lines below, and the "actual" provider's local-offline exemption further down) -- + # this pool path was the one gap (issue #86864). pool_api_key = "no-key-required" return rp._runtime(provider_label, api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, pool_api_key, source=f"pool:{pool_key}", credential_pool=pool) @@ -435,6 +439,10 @@ def _resolve_direct_alias_runtime(requested_provider: str, explicit_api_key: Opt def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optional[str]: """OpenCode family by provider name, else by opencode.ai host (``/zen/go`` => opencode-go).""" + # Custom providers in the OpenCode family (name extends opencode-go/zen, or base_url hosted on + # opencode.ai) serve models behind different API surfaces per model — a static api_mode 503s for + # /v1/responses-only models like grok-4.5 (#85589). Re-derive api_mode from the effective model and + # normalize the /v1 suffix, exactly like the built-in opencode-zen/go paths do. from hermes_cli.models import opencode_provider_family family = opencode_provider_family(requested_provider) if family is not None: @@ -455,6 +463,11 @@ def _resolve_named_custom_runtime(*, requested_provider: str, explicit_api_key: llamacpp alias with no explicit base_url resolves to the managed server first; an explicit base_url always wins.""" rp = _rp() + # Bare `provider="custom"` with an explicit base_url (e.g. propagated from a `model_aliases:` + # direct-alias resolution) — build a runtime directly so the alias's base_url actually takes effect. + # GitHub #27132: provider aliases that resolve to "custom" at runtime (ollama, vllm, llamacpp, …) are + # treated identically here, so a YAML `provider: ollama` with a LAN/WireGuard `base_url` doesn't + # silently fall through to OpenRouter. requested_norm = (requested_provider or "").strip().lower() if requested_norm in _LLAMACPP_ALIASES and not explicit_base_url: return _resolve_llamacpp_runtime(requested_provider, explicit_api_key) diff --git a/hermes_cli/secrets_cli.py b/hermes_cli/secrets_cli.py index ecf6ef08b5..2398b1e5dd 100644 --- a/hermes_cli/secrets_cli.py +++ b/hermes_cli/secrets_cli.py @@ -20,6 +20,7 @@ from rich.table import Table # parse-time from ``hermes_cli.main``, so the backend import stays lazy (nothing touches ``bw`` # until a handler runs) and ``_BWS_VERSION`` is duplicated here for the ``install --help`` text. # ``agent.secret_sources.bitwarden._BWS_VERSION`` is the source of truth; bump both together. +# See #86781. _BWS_VERSION = "2.0.0" from hermes_cli._secrets_common import ( @@ -53,7 +54,13 @@ def _load_bw(): def __getattr__(name: str): """PEP 562 lazy ``bw`` attribute (tests monkeypatch ``secrets_cli.bw``); an eager binding - would re-import ``cryptography`` at import time.""" + would re-import ``cryptography`` at import time. + + Existing callers (and upstream tests) monkeypatch attributes on ``hermes_cli.secrets_cli.bw`` directly. + Resolving that attribute at module-import time would re-import ``cryptography`` eagerly — the very + self-lock we are preventing (#86781). Defer the backend import until the first actual attribute access, + so ``import hermes_cli.secrets_cli`` stays crypto-free while ``secrets_cli.bw.find_bws`` still resolves. + """ if name == "bw": return _load_bw() raise AttributeError(f"module {__name__!r} has no attribute {name!r}") diff --git a/hermes_cli/service_manager.py b/hermes_cli/service_manager.py index c338a6bfc7..5b924fd3a4 100644 --- a/hermes_cli/service_manager.py +++ b/hermes_cli/service_manager.py @@ -83,6 +83,12 @@ def _s6_running() -> bool: (``resolve()`` silently yields the literal ``exe``), which made runtime registration inert in production. Probe the world-readable ``/proc/1/comm`` AND ``/run/s6/basedir`` — either alone can false-positive. + + The obvious probe — ``Path('/proc/1/exe').resolve()`` — only works as root: for any other UID, the + symlink at ``/proc/1/exe`` is unreadable and ``resolve()`` silently returns the path unchanged, so the + resolved name is the literal ``"exe"`` and detection always fails. Since every Hermes runtime call + inside the container drops to hermes via ``s6-setuidgid``, that silent failure made the entire + service-manager runtime-registration path inert in production (PR #30136 review). """ try: comm = Path("/proc/1/comm").read_text(encoding="utf-8").strip() @@ -303,6 +309,15 @@ def _seed_supervise_skeleton(svc_dir: Path) -> None: its chown/chmod fix-up, so seeding before ``s6-svscanctl -a`` makes s6-supervise inherit our ownership. ``log/`` gets the same skeleton (its own supervise instance) or unregister teardown EACCESes on the logger. Idempotent: existing entries (possibly live FIFOs) are left untouched. + + The PR #30136 review surfaced this as a real product gap: the entire S6ServiceManager lifecycle + (``register/start/stop/unregister _profile_gateway``) was inert in production because every operation is + dispatched as the hermes user. + Reference --------- Discussed at length on the skarnet `skaware` mailing list in 2020 + (`<http://skarnet.org/lists/skaware/1424.html>`_); see also just-containers/s6-overlay#130. The + pre-creation pattern was historically called out as forward-compatibility-fragile, but the EEXIST + handling in s6-supervise has been stable since 2015 — it's the same pattern ``s6-svperms`` and + ``fix-attrs.d`` rely on. """ def _mkdir_owned(path: Path, mode: int) -> None: @@ -393,6 +408,17 @@ class S6ServiceManager: profile and ``-p default`` would look up ``profiles/default/``. Port comes from the profile's own env (``API_SERVER_PORT``, default 8642); two profiles that both leave it unset collide. + + Port selection: the gateway binds the port resolved by ``gateway/config.py`` from the profile's own + environment — ``API_SERVER_PORT`` (or ``platforms.api_server.extra.port`` in that profile's + ``config.yaml``), defaulting to 8642. There is no ``[gateway] port`` key and no Python-side + allocator: because each supervised profile gateway loads its own ``HERMES_HOME``, two profiles that + both leave the port unset will both try to bind 8642 — give each profile a distinct + ``API_SERVER_PORT`` in its ``.env``. Previously this method took a ``port`` parameter that was + passed in but never substituted into the rendered script (carried for "API parity" with a + deterministic SHA-256 allocator in ``hermes_cli.profiles._allocate_gateway_port``). PR #30136 review + item I5 retired both the allocator and the parameter because they were dead code through the entire + stack. """ lines = [ "#!/command/with-contenv sh", @@ -432,7 +458,14 @@ class S6ServiceManager: def _render_finish_script() -> str: """Finish script: exit 78 (EX_CONFIG, fatal config) and clean exit 0 (intentional stop — restarting would turn every normal exit into a reconnect storm) both map to 125 so s6 stops - restarting; only other non-zero exits let s6 restart normally.""" + restarting; only other non-zero exits let s6 restart normally. + + When the gateway exits with EX_CONFIG (78) — a fatal configuration error such as a token collision + or no messaging platforms — we tell s6-supervise to stop restarting by exiting 125 (permanent + failure). A clean exit 0 is an intentional stop, not a crash: restarting after it turns any normal + gateway exit into a reconnect loop (the ashriel-discord storm in #76435 — 1,000+ connections and a + provider token reset). See #51228, #76435. + """ from gateway.restart import GATEWAY_FATAL_CONFIG_EXIT_CODE code = GATEWAY_FATAL_CONFIG_EXIT_CODE return ( @@ -465,6 +498,7 @@ class S6ServiceManager: # chown/unlink hermes-writable volume paths from this restartable root-context script: # an unprivileged user can race a pathname op through a symlink swap (CWE-59/CWE-367). # Parent logs/gateways is seeded hermes-owned at stage2 boot (test_log_dir_seed.py). + # See #45258. f'if [ "$(id -u)" = 0 ]; then\n' f' s6-setuidgid hermes mkdir -p "$log_dir"\n' f' s6-setuidgid hermes rm -f "$log_dir/lock"\n' diff --git a/hermes_cli/session_recovery.py b/hermes_cli/session_recovery.py index 9a1f1d4e4f..db92ba3a9b 100644 --- a/hermes_cli/session_recovery.py +++ b/hermes_cli/session_recovery.py @@ -35,6 +35,7 @@ def _init_delivery_ledger_schema(conn: sqlite3.Connection) -> None: # state.db tables created lazily by a gateway module (base ``SessionDB`` never creates them on a fresh # destination) -> the initializer owning their DDL. Recovery creates them before copying so owed rows # don't silently vanish from a "complete" salvage. Register new lazy tables HERE, not as ``if table ==``. +# See #100313, #86236. _AUXILIARY_TABLE_SCHEMAS: dict[str, Callable[[sqlite3.Connection], None]] = { "delivery_obligations": _init_delivery_ledger_schema, } @@ -180,6 +181,11 @@ def _copy_source_bundle(source: Path, snapshot_dir: Path) -> tuple[Path, list[st The whole copy runs inside ``offline_file_access`` (holds the connection-lifecycle lock). Recovery normally runs as its own CLI process against an offline file, so the refusal should never fire; the guard keeps this path consistent with ``hermes_state._backup_db_file``. + + Checking for a live connection and *then* copying would be a check/use race: a connection could open in + that window, and the copy's ``close()`` would cancel its POSIX advisory locks -- the failure class + ``hermes_cli.sqlite_safe_read`` exists to prevent (see #71724). Holding the lock means no connection can + appear mid-copy, across the main file and every sidecar. """ from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access snapshot_source = snapshot_dir / source.name @@ -1011,6 +1017,7 @@ def _recover_via_lost_and_found( verification.update(loss_detected=True, complete=False) # Structural checks cannot see a positional mis-mapping: every row still inserts, so integrity/FK/FTS # stay green. A systematic timestamp violation is the semantic tell — never report such a salvage as verified. + # See #101409. plausibility_conn = sqlite3.connect(str(output), isolation_level=None) try: plausibility_errors = _lost_and_found_plausibility_errors(plausibility_conn) diff --git a/hermes_cli/sessions_cmd.py b/hermes_cli/sessions_cmd.py index 3b5133ec2a..94b2647a72 100644 --- a/hermes_cli/sessions_cmd.py +++ b/hermes_cli/sessions_cmd.py @@ -551,7 +551,11 @@ _NEVER_ACTIVE_DEFAULT_DAYS = 30.0 def _prune_never_active_keyed(db, args): """`prune --never-active`: drop keyed gateway rows opened and never used (mostly escaped test fixtures). Separate from the shared prune/archive selector, which is pinned to `ended_at IS NOT - NULL` — never-closed rows sit outside it by construction.""" + NULL` — never-closed rows sit outside it by construction. + + The population is dominated by escaped test fixtures (#82770), which the hermetic-isolation guard can + only stop from being *created* — rows already written to a developer's state.db need a sweep to leave. + """ from hermes_cli.session_filters import format_epoch, parse_duration_seconds older_than = getattr(args, "older_than", None) days = _NEVER_ACTIVE_DEFAULT_DAYS @@ -695,6 +699,11 @@ def _cmd_rename(db, args): def _cmd_pin(db, args, pinning): """Durable "keep" flag (exempt from sessions.auto_archive, always listed); every surface shares the store.""" failures = 0 + # Pinned sessions are exempt from the sessions.auto_archive stale sweep and always surface in listings; + # until now only the Desktop sidebar could write the flag. Inspired by Perplexity Computer's + # conversational session management (pin/archive from any surface): pin state is operational + # infrastructure, so every surface — GUI, TUI, CLI, scripts — needs read/write access to the same store. + # See #52955. for raw_id in args.session_ids: resolved = db.resolve_session_id(raw_id) if resolved and db.set_session_pinned(resolved, pinning): diff --git a/hermes_cli/setup.py b/hermes_cli/setup.py index d50888ac70..ec96988c83 100644 --- a/hermes_cli/setup.py +++ b/hermes_cli/setup.py @@ -461,6 +461,8 @@ def setup_agent_settings(config: dict): _info(f" Guide: {_DOCS_BASE}/user-guide/configuration", None) # ── Max Iterations ── (config.yaml is authoritative; never surface a stale legacy .env value) + # If a legacy .env entry is still around (from pre-PR#18413 setups), prefer the config value so we don't + # surface a stale number to the user. current_max = str(cfg_get(config, "agent", "max_turns", default=90)) _info("Maximum tool-calling iterations per conversation.", "Higher = more complex tasks, but costs more tokens.", @@ -737,6 +739,7 @@ def _run_setup_wizard_impl(args): quick_requested = bool(getattr(args, "quick", False)) config = load_config() hermes_home = get_hermes_home() + # Back up existing config before setup modifies it (#3522) config_path = get_config_path() _backup_path = _backup_config_file(config_path) diff --git a/hermes_cli/setup_platforms.py b/hermes_cli/setup_platforms.py index 05004b89d8..596e33ef41 100644 --- a/hermes_cli/setup_platforms.py +++ b/hermes_cli/setup_platforms.py @@ -180,6 +180,9 @@ def _setup_telegram(): _save_prompted("TELEGRAM_HOME_CHANNEL", "Home channel ID (or leave empty to set later with /set-home in Telegram)") +# _setup_slack and _write_slack_manifest_and_instruct moved to the slack plugin: +# plugins/platforms/slack/adapter.py::interactive_setup (registered via setup_fn and dispatched through the +# plugin path). #41112 / #3823. def _setup_bluebubbles(): """Configure BlueBubbles iMessage gateway.""" from hermes_cli.setup import _info, print_header, print_success, prompt, prompt_yes_no diff --git a/hermes_cli/setup_quick.py b/hermes_cli/setup_quick.py index c3d68465cd..f9373bab28 100644 --- a/hermes_cli/setup_quick.py +++ b/hermes_cli/setup_quick.py @@ -135,7 +135,10 @@ def _run_first_time_quick_setup(config: dict, hermes_home, is_existing: bool): def _print_macos_fda_tip() -> None: """One-time macOS tip: one Full Disk Access grant kills every per-folder prompt. Same prompt-free probe as doctor's check_macos_full_disk_access (the FDA-gated TCC dir never - triggers a dialog); silent on non-macOS and when FDA is granted or indeterminate.""" + triggers a dialog); silent on non-macOS and when FDA is granted or indeterminate. + + every per-folder permission prompt, permanently (issue #52010 follow-up). + """ from hermes_cli.setup import _info if sys.platform != "darwin": return @@ -172,6 +175,9 @@ def _blank_slate_minimal_toolsets(config: dict): for k, tdef in TOOLSETS.items(): if k.startswith("hermes-") or (isinstance(tdef, dict) and (tdef.get("includes") or tdef.get("posture"))): continue + # selections made by agent/coding_context.py — not permanent user-facing disables. Adding them + # here causes model_tools to subtract their tools (terminal, read_file, …) from the minimal + # Blank Slate surface (#57315). all_keys.add(k) disabled = sorted(all_keys - keep) if disabled: diff --git a/hermes_cli/skills_config.py b/hermes_cli/skills_config.py index ebc44c534c..e7bcc08789 100644 --- a/hermes_cli/skills_config.py +++ b/hermes_cli/skills_config.py @@ -11,7 +11,10 @@ PLATFORMS = {k: info.label for k, info in _PLATFORMS.items() if k != "api_server def _normalize_skill_names(values) -> Set[str]: """Config value -> set of skill names (mirrors ``agent.skill_utils._normalize_string_set``): - ``None`` (YAML null) is empty and a bare scalar is a single-item list, NOT its characters.""" + ``None`` (YAML null) is empty and a bare scalar is a single-item list, NOT its characters. + + See #13026. + """ if values is None: return set() if isinstance(values, str): diff --git a/hermes_cli/skills_hub.py b/hermes_cli/skills_hub.py index 1712fe290a..2f906088c5 100644 --- a/hermes_cli/skills_hub.py +++ b/hermes_cli/skills_hub.py @@ -823,7 +823,15 @@ def _has_local_edits(installed: dict) -> bool: def do_update(name: Optional[str] = None, console: Optional[Console] = None, force: bool = False) -> None: """Update hub-installed skills. Locally edited ones are skipped unless ``force`` — the - update rmtree-replaces the user's work, so that must be an explicit choice.""" + update rmtree-replaces the user's work, so that must be an explicit choice. + + Skills whose on-disk content no longer matches the hash recorded at install time have been edited + locally; updating them would silently destroy the user's work (``do_install(force=True)`` + rmtree-replaces the directory). Those are skipped by default and only overwritten when ``force=True``. + Mirrors the user-modified protection bundled skills already get from ``hermes update`` (ported from + paperclipai/paperclip#10978's explicit-merge-mode rule: destructive replacement must be an explicit + caller choice, never a rerun default). + """ from tools.skills_hub import HubLockFile, check_for_skill_updates c = console or _console lock = HubLockFile() diff --git a/hermes_cli/skin_cmd.py b/hermes_cli/skin_cmd.py index ea0a220723..5af5668e4f 100644 --- a/hermes_cli/skin_cmd.py +++ b/hermes_cli/skin_cmd.py @@ -61,6 +61,7 @@ def _skin_set(key: str, value: str, skin: str | None) -> int: data.setdefault("name", target) # Atomic write: write_text truncates with no fsync; safe_load("") → None → {} would # permanently lose the palette on the next set. + # See #51356. from utils import atomic_yaml_write atomic_yaml_write(path, data, sort_keys=False) if target != name: diff --git a/hermes_cli/sqlite_safe_read.py b/hermes_cli/sqlite_safe_read.py index a8a6d12aba..fecbd03efb 100644 --- a/hermes_cli/sqlite_safe_read.py +++ b/hermes_cli/sqlite_safe_read.py @@ -96,6 +96,9 @@ class _TrackingMixin: def close(self) -> None: # type: ignore[misc] with _live_lock: path = getattr(self, "_hermes_tracked_path", None) + # Close first; untrack only once the descriptor is actually gone. Untracking before a failing + # close (e.g. cross-thread ProgrammingError) leaves the FD open while the byte-probe guard + # thinks nothing is live — see #75629. super().close() # type: ignore[misc] if path is not None: self._hermes_tracked_path = None diff --git a/hermes_cli/sqlite_util.py b/hermes_cli/sqlite_util.py index bd5e7779b4..e1ed4bc09f 100644 --- a/hermes_cli/sqlite_util.py +++ b/hermes_cli/sqlite_util.py @@ -8,7 +8,11 @@ import sqlite3 def add_column_if_missing(conn: sqlite3.Connection, table: str, column: str, ddl: str) -> bool: """``ALTER TABLE <table> ADD COLUMN <ddl>``, idempotent across races: True when this call added - it, False on the ``duplicate column name`` a concurrent migrator caused.""" + it, False on the ``duplicate column name`` a concurrent migrator caused. + + ``column`` is the human-readable name for the call site; ``ddl`` carries the actual definition. See + #21708. + """ try: conn.execute(f"ALTER TABLE {table} ADD COLUMN {ddl}") return True diff --git a/hermes_cli/subcommands/login.py b/hermes_cli/subcommands/login.py index 359b9a898f..06912c7983 100644 --- a/hermes_cli/subcommands/login.py +++ b/hermes_cli/subcommands/login.py @@ -12,6 +12,9 @@ def build_login_parser(subparsers, *, cmd_login: Callable) -> None: ``invalid choice``. Registered WITHOUT ``help=`` so it is omitted from ``hermes --help`` (``help=SUPPRESS`` leaks ``==SUPPRESS==`` for top-level subparsers on 3.12+). ``--provider`` takes ANY value (no ``choices=``) so the handler is reached rather than argparse erroring. + + This hides a command that no longer works (#24756) without the ``help=argparse.SUPPRESS`` + ``==SUPPRESS==`` leak that argparse emits for a top-level subparser on Python 3.12+. """ login_parser = subparsers.add_parser( "login", diff --git a/hermes_cli/subcommands/secrets.py b/hermes_cli/subcommands/secrets.py index ba6c8c0e32..9a597f596e 100644 --- a/hermes_cli/subcommands/secrets.py +++ b/hermes_cli/subcommands/secrets.py @@ -22,6 +22,8 @@ def build_secrets_parser(subparsers) -> None: # Lazy import: secrets_cli pulls cryptography's native extension, which on Windows # maps into the updater process and defers its self-lock preflight. secrets_cli # defers its backend import to first use, so register_cli here costs no crypto load. + # Lazy-import secrets_cli: the module imports agent.secret_sources.bitwarden which loads + # cryptography._rust.pyd. See #86781. from hermes_cli import secrets_cli as _secrets_cli from hermes_cli import onepassword_secrets_cli as _op_secrets_cli diff --git a/hermes_cli/tools_config.py b/hermes_cli/tools_config.py index acfc7988ed..b0dc86d113 100644 --- a/hermes_cli/tools_config.py +++ b/hermes_cli/tools_config.py @@ -479,6 +479,11 @@ def _default_off_toolsets(platform: str, explicitly_configured: bool) -> Set[str default_off = set(_DEFAULT_OFF_TOOLSETS) if platform in default_off and platform not in _TOOLSET_PLATFORM_RESTRICTIONS: default_off.remove(platform) + # Home Assistant is already runtime-gated by its check_fn (requires HASS_TOKEN to register any tools). + # When a user has configured HASS_TOKEN, they've explicitly opted in — don't also strip it via + # _DEFAULT_OFF_TOOLSETS, which would silently drop HA from platforms (e.g. cron) that run through + # _get_platform_tools without an explicit saved toolset list. Without this, Norbert's HA cron jobs + # regressed after #14798 made cron honor per-platform tool config. if "homeassistant" in default_off and _homeassistant_credentials_present(): default_off.remove("homeassistant") if explicitly_configured: @@ -548,6 +553,8 @@ def _get_platform_tools(config: dict, platform: str, *, include_default_mcp_serv toolset_names = platform_toolsets.get(platform) # An explicitly saved list (even a composite like ``hermes-discord``) is an opt-in to the platform's # native default-off toolsets — see _default_off_toolsets. + # Track whether the user explicitly saved a toolset list for this platform (vs. falling back to the + # platform default). See #35527. explicitly_configured = isinstance(toolset_names, list) if not explicitly_configured: toolset_names = [_platform_default_toolset(platform)] @@ -558,6 +565,9 @@ def _get_platform_tools(config: dict, platform: str, *, include_default_mcp_serv plugin_ts_keys = _get_plugin_toolset_keys() platform_default_keys = _platform_default_keys() # Plugin toolsets are first-class on a saved list: ``[hermes-cli, a2a]`` must survive filtering. + # Plugin-provided toolsets are first-class on a platform-toolsets list — explicit config like + # ``[hermes-cli, a2a]`` must survive filtering just like a built-in configurable toolset would. See + # issue #81163. explicit_known_keys = configurable_keys | plugin_ts_keys if any(ts in explicit_known_keys for ts in toolset_names): @@ -581,6 +591,8 @@ def _get_platform_tools(config: dict, platform: str, *, include_default_mcp_serv # agent.disabled_toolsets is a global suppression list and runs LAST so it overrides everything above. It # may arrive as a JSON-array string ("['memory']") from `hermes config set` or a JSON-mode editor save. disabled_toolsets = (config.get("agent") or {}).get("disabled_toolsets") + # Honor agent.disabled_toolsets from config.yaml — allows users to globally suppress specific toolsets + # (e.g. "memory") across all platforms without per-platform toolset configuration. See #86661. if disabled_toolsets: from agent.skill_utils import parse_config_string_list enabled_toolsets -= {name.strip() for name in parse_config_string_list(disabled_toolsets) if name.strip()} @@ -671,6 +683,7 @@ def _save_platform_tools(config: dict, platform: str, enabled_toolset_keys: Set[ # listed there stays OFF no matter what this writes (Blank Slate installs pre-populate ~27 entries, making # the desktop Toolsets UI unable to re-enable anything). Only toolsets just explicitly enabled FOR THIS # PLATFORM are cleared, so the list keeps working as a cross-platform suppression list for everything else. + # See #49995. agent_cfg = config.get("agent") newly_enabled = enabled_toolset_keys - preserved_entries if isinstance(agent_cfg, dict) and agent_cfg.get("disabled_toolsets") and newly_enabled: diff --git a/hermes_cli/tools_config_cua.py b/hermes_cli/tools_config_cua.py index dd73f58260..5509f498ae 100644 --- a/hermes_cli/tools_config_cua.py +++ b/hermes_cli/tools_config_cua.py @@ -23,10 +23,16 @@ logger = logging.getLogger("hermes_cli.tools_config") # One upstream-installer run must outlive the installer's own stale-lock recovery (_install-rust.sh # force-releases a dead holder's lock only after LOCK_STALE_AFTER_SECONDS=600; a shorter timeout # kills every run before that fires — a permanent wedge). 660s = 600s + 60s headroom. +# With a shorter Python-side timeout, a stale lock means every run gets killed before the installer's +# recovery can fire — a permanent "always times out" wedge (issue #58762). 660s = 600s lock window + 60s +# headroom for the actual download/swap. _CUA_INSTALLER_TIMEOUT = 660 # Bounded pipe drain after a timeout kill: the kill is best-effort (_reap_after_timeout), and a # surviving descendant holding the inherited stdout would otherwise block the read on an EOF that # never comes. A successful kill closes the pipe at once, so this costs nothing. +# Grace period for draining the installer's pipes after a timeout kill. A successful kill closes the pipe +# immediately, so this costs nothing in the normal case; it only caps how long a failed one can stall the +# update. See #87703. _CUA_INSTALLER_DRAIN_GRACE = 15 # Quiet ``hermes update`` refreshes stay bounded even when upstream waits on Read-Host / a consent # prompt (explicit ``install --upgrade`` keeps the full ceiling); safe because the lock/network @@ -443,7 +449,12 @@ def _clear_stale_cua_install_lock() -> None: def _cua_install_lock_held() -> bool: """True when the upstream installer's lock is held by a LIVE process. Called after ``_clear_stale_cua_install_lock()``: anything provably stale is already gone, so a - surviving lock artifact means a concurrent (or orphaned-but-alive) install owns it.""" + surviving lock artifact means a concurrent (or orphaned-but-alive) install owns it. + + Upstream waits up to ``LOCK_STALE_AFTER_SECONDS=600`` on a held lock before probing — unattended + refreshes must not eat that wait (the 11-minute hang class, 87703): they skip instead. Best-effort: + unreadable state reports not-held so a probe failure can never block an install. See #87703. + """ try: if sys.platform != "win32": return _cua_install_lock_dir().is_dir() @@ -586,11 +597,20 @@ def _kill_installer_tree(proc, *, is_windows: bool) -> None: def _reap_after_timeout(proc, *, is_windows: bool) -> None: """Kill the installer tree, then drain its pipes under a deadline. An unbounded drain blocks on an EOF that only arrives when someone kills a surviving descendant - by hand, so ``_CUA_INSTALLER_TIMEOUT`` would stop bounding anything.""" + by hand, so ``_CUA_INSTALLER_TIMEOUT`` would stop bounding anything. + + Bound the drain instead: a kill that landed closes the pipe at once, and one that did not costs + ``_CUA_INSTALLER_DRAIN_GRACE`` rather than forever. The caller re-raises the original ``TimeoutExpired`` + either way, so the manual re-run hint still prints and the update unwinds. Losing the tail of a + timed-out installer's log is the cheaper half of that trade. See #87703. + """ _kill_installer_tree(proc, is_windows=is_windows) try: drained_out, _ = proc.communicate(timeout=_CUA_INSTALLER_DRAIN_GRACE) # Partial output names WHERE the installer was stuck (lock wait, consent prompt, download). + # Diagnosability (#87703 post-mortem): the partial output names WHERE the installer was stuck (lock + # wait, consent prompt, download) — before this, the answer died with the process and the timeout + # line was unactionable. if drained_out: logger.warning("cua-driver installer timed out; last output before kill:\n%s", drained_out[-2000:]) @@ -714,9 +734,17 @@ def _run_cua_driver_installer(label: str = "Installing", verbose: bool = True, installer_env["CUA_DRIVER_RS_VERSION"] = pin_version # A previous timed-out install can leave upstream's concurrent-install lock behind; clear it # when provably stale so the refresh doesn't wedge waiting on a dead holder. + # See #58762. _clear_stale_cua_install_lock() # Unattended refreshes (installer_timeout set by `hermes update`) preflight and may skip. + # Unattended refreshes (installer_timeout set by `hermes update`) fail FAST on the two conditions that + # otherwise consume the whole ceiling: 1. Install lock held by a live process — upstream would poll it + # for up to LOCK_STALE_AFTER_SECONDS=600 before probing the holder. That is the 11-minute silent hang + # class (#87703; observed live 2026-08-25: "cua-driver refreshing timed out after 660s"). 2. Release + # host unreachable (outage/DNS/firewall) — the installer would die slowly inside its own retries. A 5s + # HEAD answers now. Explicit `computer-use install --upgrade` runs keep upstream's full lock-recovery + # semantics — a human is watching and can wait or Ctrl-C. if installer_timeout is not None: install_cmd = _unattended_installer_preflight(install_cmd, is_windows) if install_cmd is None: diff --git a/hermes_cli/tools_config_post_setup.py b/hermes_cli/tools_config_post_setup.py index 7f96955407..9f0a9e2722 100644 --- a/hermes_cli/tools_config_post_setup.py +++ b/hermes_cli/tools_config_post_setup.py @@ -96,6 +96,18 @@ def _post_setup_agent_browser(post_setup_key: str) -> None: _ensure_browser_use_cli() try: # Lazy import so the tools_config UI doesn't pull in browser_tool at import time. + # agent-browser resolves lazily via npx on the default install (#43564), invisible to the + # PATH/node_modules probes above. Mirror the rung hermes_cli.doctor uses so this probe can't diverge + # from it, including the Termux carve-out (bare npx is too fragile to advertise as ready there — see + # check_browser_requirements). + # agent-browser is no longer a root package.json dependency (#43564) — it resolves lazily via npx + # for most installs, which a bare PATH + node_modules probe can't see. Mirror the local-CLI tail of + # :func:`tools.browser_tool.check_browser_requirements` (same cascade, same Termux carve-out) so the + # setup/status surfaces can't diverge from what browser tools actually find at runtime; + # validate=False keeps this a cheap existence check with no subprocess spawn. + # agent-browser is no longer a root package.json dependency (#43564) — it resolves lazily via npx + # (or a global/Hermes-managed install) instead of a local `npm install`, so there's no node_modules/ + # population step here anymore. from tools.browser_tool import ( _chromium_installed, _running_in_docker, _find_agent_browser, _resolve_npx_bin, _is_npx_agent_browser_sentinel, AGENT_BROWSER_NPX_SPEC) diff --git a/hermes_cli/update_abort_recovery.py b/hermes_cli/update_abort_recovery.py index d8ccce9f68..186053f259 100644 --- a/hermes_cli/update_abort_recovery.py +++ b/hermes_cli/update_abort_recovery.py @@ -150,7 +150,13 @@ def _recover_gateway_restart_after_abort( plan, *, gateway_mode: bool, skip_profiles: set[str] | None = None, skip_units: set[str] | None = None) -> dict[str, list]: """Retry supervised gateway restarts from a clean Python process (the in-process restart ran - in the pre-``git pull`` interpreter). Only inventory-classified supervisor-owned profiles.""" + in the pre-``git pull`` interpreter). Only inventory-classified supervisor-owned profiles. + + ``skip_units`` names the units the aborted phase already settled, as ``<scope>/<unit>``. The scope is + part of the identity, not decoration: ``hermes-serve.service`` can exist in both the user and the system + manager as two different processes, and an unqualified token would let a settled one suppress recovery + of a stale one (review on #96235). + """ from hermes_cli.update_cmd import _gateway_recovery_partition candidates, skipped = _gateway_recovery_partition(plan, skip_profiles=skip_profiles) profiles = sorted(candidates) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 8c8fb7c342..d08f123438 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -259,7 +259,10 @@ def _called_process_error_is_python_dep_install(exc: subprocess.CalledProcessErr def _format_update_failure_stage(exc: subprocess.CalledProcessError) -> str: """Name the failed stage: git pull and dep install share one ``try``, and calling every CalledProcessError a git failure misled users and keyed the ZIP overlay on exception - *type* rather than on git actually failing.""" + *type* rather than on git actually failing. + + See #85840, #87304. + """ if _called_process_error_is_python_dep_install(exc): return "Python dependency install failed" if _called_process_error_is_git(exc): @@ -284,7 +287,10 @@ def _refuse_update_for_contended_shims(exc: BaseException) -> None: """Fail closed when live shims could not be quarantined: a rename failing every retry proves a holder without FILE_SHARE_DELETE, and installing anyway strands the venv between versions. The code swap is already committed; only the dep install is deferred (via the - update-incomplete marker). Exits 2 so the receipt records a refusal, not a failure.""" + update-incomplete marker). Exits 2 so the receipt records a refusal, not a failure. + + See #87331. + """ print("✗ Cannot continue the update: live Hermes launcher(s) could not be") print(" moved aside:") for name in getattr(exc, "failed_shims", []) or ["hermes.exe"]: @@ -419,6 +425,10 @@ def _log_only_write(text: str) -> None: def _run_logged_subprocess(cmd, *, cwd=None, env=None): """Run ``cmd`` with combined output captured into update.log only; returns the ``CompletedProcess`` so the caller can surface the output on failure.""" + # Check if there are updates. On shallow checkouts `rev-list --count` walks the truncated graph and can + # report the entire remote ancestry (e.g. "Found 9980 new commit(s)" on a depth-1 install — #53479). The + # zero/nonzero gate is still sound (HEAD == origin/<branch> counts 0), so keep it, but treat the shallow + # NUMBER as unknown and recover the real one via the GitHub compare API when possible. result = subprocess.run( cmd, cwd=cwd, env=env, check=False, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, encoding="utf-8", errors="replace") @@ -538,6 +548,10 @@ def _repair_venv_on_current_checkout( """Reinstall ``.[all]`` + lazy/tool deps into an unhealthy (or handed-off) venv; returns whether the checkout can be reported complete.""" # Self-lock deferral: the repair rewrites the venv too (same mapped-extension hazard). + # See #86735. + # Self-lock deferral (relocated preflight — #86735): if THIS process holds a native extension the sync + # must rewrite, defer NOW — after the code swap, so only the dependency install is pending and the next + # fresh launch completes it via the marker. _m()._abort_dependency_sync_if_self_locked(_windows_gateway_resume) _write_update_incomplete_marker() from hermes_cli.managed_uv import ensure_uv @@ -552,6 +566,9 @@ def _repair_venv_on_current_checkout( _m()._install_python_dependencies_with_optional_fallback(repair_prefix, env=repair_env, group="all") _m()._refresh_active_lazy_features(repair_prefix, env=repair_env, features=active_lazy_features) _m()._restore_active_tool_dependencies(active_tool_dependencies, repair_prefix, env=repair_env) + # Core ``.[all]`` install finished. Clear the generic core breadcrumb before the lazy-refresh phase — + # that phase uses its own marker so a later lazy failure cannot be "healed" by clearing the core marker + # based on a narrow 7-package import probe (#58004 review). _m()._clear_update_incomplete_marker() healthy_after, detail_after = _venv_core_imports_healthy() if not healthy_after: @@ -559,6 +576,7 @@ def _repair_venv_on_current_checkout( print(" Close all Hermes windows/gateways and re-run: hermes update") return False print("✓ Dependencies repaired!") + # Check for config migrations (#91360). _check_and_apply_config_migration( assume_yes=assume_yes, gateway_mode=gateway_mode, pre_update_snapshot_id=pre_update_snapshot_id) # The hand-off child never reaches the commits-pulled rebuild; do it here. @@ -573,6 +591,10 @@ def _pip_install_prefix(uv_bin) -> tuple[list[str], dict | None]: """``(install prefix, env)``: ``uv pip`` isolated from third-party UV env vars (so a foreign UV_PYTHON_INSTALL_DIR can't hijack it), else ``sys.executable -m pip`` (avoids PEP 668 errors).""" if uv_bin: + # Same third-party UV-env isolation as the main update path (#83914): a user-level + # UV_PYTHON_INSTALL_DIR / UV_PYTHON from unrelated software must not steer which interpreter uv + # resolves here. + # See #83914. from hermes_cli.managed_uv import managed_python_env env = managed_python_env() env["VIRTUAL_ENV"] = str(_m().PROJECT_ROOT / "venv") @@ -721,6 +743,9 @@ def _pull_updates( post-pull syntax error in a critical file rolls back. Exits on failure; returns pre-pull SHA.""" update_succeeded = False # Pre-pull SHA for auto-rollback (stray conflict markers once bricked every updater). + # Capture the pre-pull SHA so we can auto-roll-back if the new code has a syntax error in a + # critical-path file (PR #28452 incident: orphan merge-conflict markers in hermes_cli/config.py bricked + # every user who ran ``hermes update`` for the 7 minutes between the bad commit and the fix landing). pre_pull_sha = _capture_head_sha(git_cmd, _m().PROJECT_ROOT) try: # merge --ff-only the already-fetched ref instead of `git pull`, which would do a @@ -853,6 +878,12 @@ def _prepare_checkout_for_update( # A fork can match origin yet trail upstream, so the sync can move HEAD with # commit_count == 0; detect that BEFORE the no-update return so deps, restarts AND the # fleet matrix still run (it used to live in the early-return branch and verified nothing). + # The sync can therefore advance HEAD even though the origin comparison found no commits. Detect that + # BEFORE taking the no-update return so dependency refreshes, gateway restarts, AND the fleet version + # matrix still run for the pulled code (#73108 — previously the sync lived inside the commit_count == 0 + # branch, which returns immediately after: an update that pulled hundreds of upstream commits printed + # "Already up to date!" and verified nothing). Non-fork checkouts have no upstream question: origin IS + # the official repo, so "Already up to date!" is fully verified there. upstream_checked = True if commit_count == 0 and is_fork and branch == "main": pre_sync_sha = _capture_head_sha(git_cmd, _m().PROJECT_ROOT) @@ -893,6 +924,10 @@ def _resolve_update_options(args, gateway_mode: bool) -> _UpdateOptions: active_tool_dependencies = _m()._capture_active_tool_dependencies() # Captured before any pull so the completion line can report the transition. + # Snapshot the pre-update version before files are replaced so the completion line can report the + # transition (prime-agent#630 port). + # Snapshot the pre-update version before any code is pulled so the completion line can report the + # transition (prime-agent#630 port). pre_update_version = _read_project_version() gw_input_fn = ( (lambda prompt, default="": _gateway_prompt(prompt, default)) if gateway_mode else None) @@ -902,6 +937,7 @@ def _resolve_update_options(args, gateway_mode: bool) -> _UpdateOptions: keep_stash = bool(getattr(args, "keep_stash", False)) # --switch-branch: prefer switching over an in-place merge so an update never writes the # branch's history; only meaningful with parked_branch_strategy "update_in_place". + # See #89507. switch_branch = bool(getattr(args, "switch_branch", False)) # Interactive terminals always stash-and-ask; only non-interactive updates consult @@ -925,11 +961,17 @@ def _begin_update_receipt_and_plan(args): holds the venv shim.""" # Structured receipt: record what this run discovers/does/skips so silent failures are diagnosable. with _best_effort('Update receipt unavailable: %s'): + # See #74973, #81193, #85753, #88848, #91277. from hermes_cli.update_receipt import begin_update_receipt begin_update_receipt() # Plan phase: snapshot runtimes/supervisors/version (read-only; probe failure records # nothing). Re-read AFTER the restart phase to reconcile — the plan is the worklist. + # Plan phase (#91277 Phase 2): snapshot the pre-update fleet — every running Hermes runtime, its + # supervisor, and its running code version — into the receipt, so a post-mortem can compare what the + # update SAW against what it did. ``_pre_update_plan`` is read again AFTER the restart phase to + # reconcile every planned runtime against the phase's bookkeeping (restart via declared mechanism — the + # plan is the worklist, not just a printout). _pre_update_plan = None with _best_effort('Update plan phase failed: %s'): from hermes_cli.update_inventory import collect_runtime_inventory, record_plan_in_receipt @@ -943,6 +985,13 @@ def _begin_update_receipt_and_plan(args): # Windows: another hermes.exe holding the venv shim means WinError 32 spam and a # deferred-rename leftover or silent ZIP fallback. Positively identified gateways are # paused/restarted by the update instead; anything else still aborts. + # Continuing would result in a string of WinError 32 warnings and then either a deferred-rename leftover + # or a failed git-pull fast path that silently falls back to the slower ZIP route. See issue #26670. + # Exception (#37039): when every concurrent instance is a gateway runtime, the pause machinery a few + # lines below (``_pause_windows_gateways_for_update``) stops it before any file mutation, and the + # post-update restart phase brings it back. Aborting just to make the user run the same kill manually is + # friction without benefit. Anything not positively identified as a gateway (TUI shell, Desktop backend + # child, unreadable cmdline) still aborts exactly as before. if _m()._is_windows() and not getattr(args, "force", False): scripts_dir = _m()._venv_scripts_dir() concurrent = _m()._detect_concurrent_hermes_instances(scripts_dir) if scripts_dir is not None else [] @@ -968,6 +1017,7 @@ def _prepare_git_command() -> tuple[bool, list, bool]: _git_run(git_cmd, ["config", "windows.appendAtomically", "false"]) # A broken Git-for-Windows trampoline refuses every call with a "BUG (fork bomb)" guard; # swap in a real binary up front so git survives instead of degrading to ZIP. + # See #87876. git_cmd = _ensure_non_trampoline_git(git_cmd) # Before stash/branch logic: npm rewrites package-lock.json non-deterministically and @@ -991,6 +1041,13 @@ def _verify_head_after_pull( """Return the post-pull HEAD SHA; ``sys.exit(1)`` if the pull was a no-op or landed off-branch.""" # A detached checkout pinned to a SHA can report "N new commit(s)" and a successful # merge --ff-only yet stay put; surface the no-op instead of claiming "Code updated!". + # Verify HEAD actually moved (issue #79678). ``merge --ff-only`` succeeding only means the merge + # completed, not that the update applied: a checkout that is pinned to a raw SHA (detached HEAD) can + # report "N new commit(s)" against origin yet still sit on the old commit afterward (the branch-switch + # step re-detaches to the SHA). Before this guard, ``hermes update`` printed "✓ Code updated!" and + # reinstalled deps + rebuilt the desktop app against the stale tree — no error, no warning, ``hermes + # doctor`` healthy. Compare pre-pull and post-pull HEAD; if they match, surface the no-op instead of + # claiming success. post_pull_sha = _capture_head_sha(git_cmd, _m().PROJECT_ROOT) if pre_pull_sha and post_pull_sha == pre_pull_sha: print() @@ -1095,6 +1152,10 @@ def _finish_already_up_to_date( _m()._resume_windows_gateways_after_update(_windows_gateway_resume) # A prior pull may still owe the fleet a restart; catch up here too, BEFORE the exit # gate so a partial outcome can't strand the fleet on stale code. + # Catch up even on the "Already up to date" path — that early return is what left the gateway on stale + # code for two days. Runs BEFORE the runtime-verification exit gate below: a vulnerable SQLite runtime + # demotes the outcome to partial, but must not strand the fleet on stale code (#91277 fleet contract — + # the pending-restart check always executes). _apply_pending_fleet_restart_catchup() if not current_checkout_complete: if gateway_mode: @@ -1116,6 +1177,7 @@ def _apply_pulled_update( # Gateways still serve pre-pull modules until the restart phase; an interrupt before a # completed restart leaves this marker so the next update catches up even when git is # current. Distinct from ``.update-incomplete`` (venv/install repair). + # See #95294. _write_fleet_restart_pending_marker(expected_sha=post_pull_sha or "") # Stale .pyc would ImportError on gateway restart when new source references new names. _sweep_bytecode_after_update(branch) @@ -1190,6 +1252,14 @@ def _cmd_update_impl(args, gateway_mode: bool): _clear_windows_venv_holders_or_exit(args, gateway_mode, _windows_gateway_resume) # After every fail-closed venv guard, before either path can remove the release tree. + # Self-lock deferral moved: the venv-holder sweep above excludes this process by design (a CLI `hermes + # update` IS the venv python), and an updater that has imported a native venv extension cannot rewrite + # its own mapped .pyd (#83569). That check used to run HERE — before the fetch — but firing pre-fetch + # meant a deferral stranded the user on the OLD checkout, and any startup path that eagerly loaded + # cryptography turned every Windows update into an exit-2 loop (#86735/#86780/#86781). It now runs via + # _abort_dependency_sync_if_self_locked() after the code swap, immediately before the dependency sync — + # the only phase the lock can actually break — and only when the sync would truly rewrite the loaded + # distribution. desktop_dir = _m().PROJECT_ROOT / "apps" / "desktop" had_desktop_app_before_update = _desktop_app_present(desktop_dir) @@ -1219,6 +1289,8 @@ def _cmd_update_impl(args, gateway_mode: bool): print(" (removed %d aborted-fetch pack temp file(s))" % len(swept)) # Surface autostashes left by earlier updates (--keep-stash, failed restores). + # Surface autostash entries left behind by earlier updates (#63717 problem 6) — parked --keep-stash + # runs and failed restores preserve the stash but nothing ever mentioned it again. _m()._warn_orphaned_update_autostashes(git_cmd, _m().PROJECT_ROOT) print("→ Fetching updates...") @@ -1264,6 +1336,7 @@ def _cmd_update_impl(args, gateway_mode: bool): _windows_gateway_resume=_windows_gateway_resume) except _shim_quarantine_error_type() as e: # Strict quarantine refused BEFORE any installer ran — defer via marker, exit 2, no ZIP. + # See #87331. _refuse_update_for_contended_shims(e) except subprocess.CalledProcessError as e: _handle_update_called_process_error(e, args, gateway_mode, had_desktop_app_before_update) diff --git a/hermes_cli/update_cmd_config.py b/hermes_cli/update_cmd_config.py index f453e991e4..53cc4b73d2 100644 --- a/hermes_cli/update_cmd_config.py +++ b/hermes_cli/update_cmd_config.py @@ -50,7 +50,12 @@ def _migrate_sibling_profile_configs() -> list[tuple[str, int, int]]: """Migrate every SIBLING profile's config.yaml (the shared checkout serves all profiles). Per sibling (active skipped): scope via the context-local HERMES_HOME override (never ``os.environ``) and run the NON-INTERACTIVE quiet migration — prompt-requiring settings wait for that profile's - own session. Returns ``[(name, from_version, to_version), ...]``; never raises.""" + own session. Returns ``[(name, from_version, to_version), ...]``; never raises. + + 91277 Phase 2 (fleet-wide config migration; #20438/#54926/#79048): the shared checkout serves every + profile, but ``hermes update`` historically migrated only the active profile's config — siblings drifted + versions until their gateway hit a config the new code couldn't read. + """ from hermes_cli.update_cmd import _run_config_check_fresh, _run_migrate_config_fresh migrated: list[tuple[str, int, int]] = [] with _best_effort('Sibling profile enumeration failed: %s'): @@ -103,6 +108,10 @@ def _restore_snapshot_safety_nets(pre_update_snapshot_id) -> None: f"{', '.join(r['keys'])} from pre-update snapshot {r['snapshot_id']}.") with _best_effort("Cron jobs auto-restore check failed: %s"): + # Safety net: config-version migrations have been observed to leave cron/jobs.json valid-but-empty, + # silently dropping every scheduled job (issue #34600). The desktop scheduler can also overwrite + # with its own small set, causing partial loss (issue #52144). If the live file now has fewer jobs + # than the pre-update snapshot, restore it and warn loudly. from hermes_cli.backup import restore_cron_jobs_if_emptied cron_restore = restore_cron_jobs_if_emptied(pre_update_snapshot_id) if cron_restore: @@ -155,7 +164,10 @@ def _check_and_apply_config_migration( ) -> None: """Check/apply config migrations with freshly-reloaded modules. Runs on EVERY completion path (post-pull, venv-repair, Node-deps repair on ``commit_count == 0``) so an interrupted update - that already pulled code doesn't strand an old config version.""" + that already pulled code doesn't strand an old config version. + + See #91360. + """ from hermes_cli.update_cmd import ( _migrate_sibling_profile_configs, _reload_config_modules, _run_config_check_fresh, _run_migrate_config_fresh) @@ -166,6 +178,7 @@ def _check_and_apply_config_migration( from hermes_cli.config import get_missing_env_vars, get_missing_config_fields # A config-check failure must not break an otherwise-successful update. try: + # Log, point at the manual command, and return. See #91360. missing_env = get_missing_env_vars(required_only=True) missing_config = get_missing_config_fields() current_ver, latest_ver = _run_config_check_fresh() @@ -189,6 +202,8 @@ def _check_and_apply_config_migration( print(" ✓ Config format updated (no new settings to configure)") # quiet=True also mutes steps that RESET/REMOVE a setting; re-surface them so an # unattended update never silently changes config (config_added holds only mutations here). + # In this branch missing_config is empty, so config_added can only contain migration-step + # mutations, not missing-key listings. See #81946, #86656. for _note in _mig_results.get("config_added") or []: print(f" ℹ {_note}") for _warn in _mig_results.get("warnings") or []: diff --git a/hermes_cli/update_cmd_deps.py b/hermes_cli/update_cmd_deps.py index 6d5c0f7193..b573b22b4a 100644 --- a/hermes_cli/update_cmd_deps.py +++ b/hermes_cli/update_cmd_deps.py @@ -137,7 +137,12 @@ def _web_build_toolchain_ready(*roots: Path) -> bool: def _web_toolchain_roots(web_dir: Path) -> tuple[Path, ...]: """Roots whose ``node_modules/.bin`` can satisfy the web build: ``npm run build`` searches the - package and each ancestor, so hoisted and package-local shims are equally valid.""" + package and each ancestor, so hoisted and package-local shims are equally valid. + + ``npm run build`` prepends ``node_modules/.bin`` for the package and each of its ancestors, so shims + hoisted to the workspace root and shims nested under a package that owns its lockfile (#42973) are + equally valid. + """ return (web_dir, web_dir.parent) @@ -155,7 +160,10 @@ def _ensure_venv_pip(pip_cmd: list, python_exe: str) -> None: def _upgrade_pip_before_lazy_refresh( install_cmd_prefix: list[str], *, env: dict[str, str] | None = None) -> None: """Upgrade pip before lazy refreshes: older pip can fail setuptools source builds and - leave a partially-written venv. Never raises.""" + leave a partially-written venv. Never raises. + + See #57828. + """ from hermes_cli.update_cmd import _m try: _m()._run_package_only_install(install_cmd_prefix + ["install", "--upgrade", "pip"], env=env) @@ -254,7 +262,10 @@ def _refresh_active_lazy_features( """Refresh previously-activated lazy backends (cold ones untouched): the core install never touches them, so a bumped :data:`LAZY_DEPS` pin would leave them stale forever. Returns True when the venv is safe (refreshed / nothing active / import repair succeeded), False when a - failed lazy install left broken core imports repair couldn't fix. Never raises.""" + failed lazy install left broken core imports repair couldn't fix. Never raises. + + See #57828. + """ from hermes_cli.update_cmd import _m try: from tools import lazy_deps @@ -313,6 +324,7 @@ def _refresh_active_lazy_features( # Import-based recovery: metadata-only verifiers miss dist-info intact but import files # wiped. Unavailable probes are indeterminate, not healthy — keep the lazy marker. + # See #57828. status = _m()._repair_venv_via_import_probes(install_cmd_prefix, env=env) if status == "repaired": print(" Lazy backend(s) keep their previous version until refresh succeeds.") @@ -329,7 +341,12 @@ def _refresh_active_lazy_features( def _refresh_active_memory_provider_dependencies() -> None: """Refresh pip deps for the configured external memory provider: its bridge packages live in ``plugin.yaml`` (not Hermes extras / ``LAZY_DEPS``), so the core reinstall can strip them; - re-run the ACTIVE provider's install last so its writes land last. Never raises.""" + re-run the ACTIVE provider's install last so its writes land last. Never raises. + + Re-run the provider's declared install for the ACTIVE provider only, after the core install and lazy + refresh, so the last write to any shared package is the one the active provider needs. See #53272, + #70636. + """ try: from hermes_cli.config import load_config cfg = load_config() @@ -493,7 +510,10 @@ def _repair_node_deps_on_current_checkout( had_desktop_app_before_update: bool = False) -> bool: """Repair Node deps on the ``commit_count == 0`` path: a failed npm install says "re-run hermes update" but the early return used to skip the refresh. ``_update_node_dependencies`` self-gates - on the hash recorded only after a SUCCESSFUL install, so this is a cheap no-op when healthy.""" + on the hash recorded only after a SUCCESSFUL install, so this is a cheap no-op when healthy. + + See #77211. + """ from hermes_cli.update_cmd import ( _check_and_apply_config_migration, _m, _rebuild_desktop_after_update, _update_node_dependencies) node_failures = _update_node_dependencies() @@ -508,9 +528,11 @@ def _repair_node_deps_on_current_checkout( assume_yes=assume_yes, gateway_mode=gateway_mode, pre_update_snapshot_id=pre_update_snapshot_id) # A current checkout can still owe a Desktop rebuild (e.g. the Windows hand-off child # never reaches the commits-pulled rebuild). Self-gates on the build stamp. + # Skipping it leaves a stale desktop app behind a successful-looking update. See #97343. if not _rebuild_desktop_after_update( _m().PROJECT_ROOT / "apps" / "desktop", had_desktop_app_before_update=had_desktop_app_before_update): # Retry hint already printed; withhold success rather than claim completion. + # See #88251. print_completion( "⚠ Update partially complete — the desktop app was not rebuilt " "and is still on the previous build.") @@ -520,7 +542,10 @@ def _repair_node_deps_on_current_checkout( def _update_node_dependencies() -> list[str]: """Refresh Node deps for ui-tui and web. Returns labels whose npm install failed (empty on - success) so the caller reports a partial update instead of ``Update complete!``.""" + success) so the caller reports a partial update instead of ``Update complete!``. + + See #30271. + """ from hermes_cli.update_cmd import _m if not (_m().PROJECT_ROOT / "package.json").exists(): return [] @@ -532,6 +557,14 @@ def _update_node_dependencies() -> list[str]: from hermes_constants import is_wsl path_npm = shutil.which("npm") if is_wsl() and path_npm and _m()._is_windows_npm_path(path_npm): + # Root package.json has no dependencies of its own (agent-browser and @streamdown/math were + # moved out — see #43564): agent-browser resolves at runtime via `npx agent-browser` + # (tools/browser_tool.py), and @streamdown/math is a desktop-only import now declared in + # apps/desktop/package.json. That means a plain workspace-scoped install can never prune + # anything root-only, so we only need to name the workspaces the CLI/TUI/web build actually + # requires. apps/desktop pulls in Electron as a devDependency with a ~200MB postinstall + # download, so it's deliberately never named here — desktop deps install on demand (see + # _desktop_build_needed). print("→ Updating Node.js dependencies...") print(" ⚠ Skipped: only a Windows npm is reachable from this WSL shell.") print(" Install Node.js inside the WSL distro (nvm, or your distro's") @@ -547,6 +580,8 @@ def _update_node_dependencies() -> list[str]: # Best-effort npx cache warm before the lockfile-unchanged early return. Can block # ~11s on a cold cache — print first so it doesn't look like a hang. + # Runs before the lockfile-unchanged early return below since that's the common `hermes update` case. + # See #43564. print("→ Warming npx cache for agent-browser...") with suppress(Exception): from tools.browser_tool import warm_agent_browser_npx_cache @@ -572,6 +607,8 @@ def _update_node_dependencies() -> list[str]: # capture_output=False is deliberate: postinstall scripts print download progress and # capturing makes a long download look hung. + # The chatty npm-deprecation noise during `hermes update` comes from the *desktop* build, not this step; + # that one is captured to update.log. See #18840. result = _m()._run_npm_install_deterministic( npm, _m().PROJECT_ROOT, extra_args=tuple(install_args), capture_output=False, env=nixos_env) if result.returncode == 0: @@ -637,6 +674,21 @@ def _venv_core_imports_healthy() -> tuple[bool, str]: # every CLI process, so the guard must be HONEST: fire only when the sync would actually REWRITE # the dist, and only AFTER the code swap so a deferral leaves new code with just the install # pending. Keys are ``sys.modules`` prefixes; values are ``(display name, PyPI dist)``. +# If the updater process itself has any of these loaded, the dependency sync below cannot rewrite the +# backing ``.pyd``/``.dll`` — Windows blocks REPLACE on a mapped image — and the update dies with ``os error +# 5`` between uninstall and reinstall, stranding the venv half-updated (#83569). ``cryptography`` is the +# canonical case: ``hermes_cli.main`` used to import it at startup while resolving external secret sources; +# ``PyYAML``'s ``_yaml`` C extension is loaded by every CLI process (config parsing). Keep this guard as +# defence-in-depth against future eager imports (new secret sources, plugins absorbed into core, refactors +# of the startup order) — but the guard must be HONEST (#86735/#86780/#86781: a preflight that fired on +# every run, before the fetch, re-bricked the exact flow it was meant to protect). Two honesty gates: 1. It +# only fires when the dependency sync would actually REWRITE the loaded distribution +# (``_dependency_sync_would_rewrite``): if the installed version already satisfies the on-disk pyproject +# pins, uv/pip will not touch the mapped ``.pyd``, so there is no lock to trip. 2. It runs AFTER the code +# swap (git pull / ZIP commit), immediately before the venv rewrite — so the on-disk pyproject is the NEW +# one (gate 1 compares against the right target) and a deferral no longer strands the user on the old +# checkout: the next launch's marker recovery completes the dependency install against the already-updated +# pyproject. _SELF_LOCKING_NATIVE_MODULES: dict[str, tuple[str, str]] = { "cryptography.hazmat.bindings._rust": ("cryptography (_rust.pyd)", "cryptography"), "yaml._yaml": ("PyYAML (_yaml.pyd)", "pyyaml")} @@ -646,7 +698,10 @@ def _dependency_sync_would_rewrite(dist_name: str) -> bool | None: """Whether the ``.[all]`` install would replace *dist_name*'s files, judged against every applicable pin in on-disk ``pyproject.toml`` (base + extras). False: all pins satisfied; True: pin unsatisfied or dist missing; None: undeterminable. Never raises. Callers treat - None as fail-OPEN — PyYAML is in every process, so deferring on uncertainty always fires.""" + None as fail-OPEN — PyYAML is in every process, so deferring on uncertainty always fires. + + See #86735. + """ from hermes_cli.update_cmd import _m try: from importlib import metadata as _ilmd @@ -689,7 +744,13 @@ def _dependency_sync_would_rewrite(dist_name: str) -> bool | None: def _detect_self_loaded_native_modules() -> list[str]: """Display names of native venv extensions loaded into THIS process that the sync would rewrite. Empty off Windows (POSIX keeps an unlinked inode usable). Modules whose installed version - already satisfies the pins are NOT reported — no swap at risk. Never raises.""" + already satisfies the pins are NOT reported — no swap at risk. Never raises. + + Returns display names (empty off Windows — POSIX lets a running process keep using an unlinked inode, so + self-locking is a Windows-only hazard). A loaded module whose installed version already satisfies the + on-disk pyproject pins is NOT reported: the dependency sync will not touch its files, so there is no + swap at risk (#86735 — the always-firing variant of this preflight bricked every Windows update). + """ from hermes_cli.update_cmd import _m if not _m()._is_windows(): return [] @@ -706,7 +767,10 @@ def _abort_dependency_sync_if_self_locked(gateway_resume=None) -> None: code swap, so a deferral leaves NEW code with only the install pending). Two hazards: a mapped ``.pyd`` -> exit 2, next launch's marker recovery finishes; the ``hermes.exe`` shim we run from -> every launch is the shim so the marker would defer forever: hand the - install to a child under the venv interpreter and exit 0.""" + install to a child under the venv interpreter and exit 0. + + See #88838, #89599. + """ from hermes_cli.update_cmd import _m locked = _m()._detect_self_loaded_native_modules() if locked: @@ -749,7 +813,10 @@ def _rebuild_desktop_after_update( desktop_dir: Path, *, had_desktop_app_before_update: bool) -> bool: """Rebuild an installed Desktop app when its source or artifact changed. Returns ``False`` only when a rebuild was attempted and failed (caller withholds ``✓ Update complete!`` and - writes a failing ``.update_exit_code`` in gateway mode); every other outcome is ``True``.""" + writes a failing ``.update_exit_code`` in gateway mode); every other outcome is ``True``. + + See #88251. + """ from hermes_cli.update_cmd import _m # The release tree is git-ignored and can vanish mid-update; pre-update presence suffices. # Never make people who never used Desktop pay for an Electron build. @@ -810,6 +877,11 @@ def _venv_foreign_owned_paths(venv_root, limit: int = 5) -> list: deleted — never mutate a venv we can't safely mutate. Deliberately BOUNDED (venv root, ``venv/bin``, first site-packages top level, ``*.dist-info`` children; ~2000 stats). POSIX-only: ``[]`` on Windows and as root; ``[]`` on any surprise — must NEVER raise or add latency. + + See #83529. + A later normal ``hermes update`` then dies mid-mutation inside ``uv pip install -e .`` ("Permission + denied (os error 13)") with ``venv/bin/hermes`` already deleted — the CLI is bricked. Same philosophy as + the contended-venv gate (#87331): a venv we cannot safely mutate is never mutated at all. """ from hermes_cli.update_cmd import _path_uid try: @@ -867,7 +939,10 @@ def _venv_foreign_owned_paths(venv_root, limit: int = 5) -> list: def _refuse_update_if_venv_foreign_owned(project_root) -> None: """Refuse-before-mutate ownership gate, run after the pull and before the first venv mutation: foreign-owned files would brick the install mid-mutation, so refuse with the recovery command - while the venv is intact. No subprocess calls — tests mock ``subprocess.run`` with sequenced effects.""" + while the venv is intact. No subprocess calls — tests mock ``subprocess.run`` with sequenced effects. + + See #83529. + """ foreign = _venv_foreign_owned_paths(Path(project_root) / "venv") if not foreign: return @@ -892,6 +967,10 @@ def _sync_python_dependencies_after_pull( from hermes_cli.update_cmd import ( _m, _pip_install_prefix, _sweep_bytecode_after_update, _validate_critical_modules_import, _write_lazy_refresh_incomplete_marker, _write_update_incomplete_marker) + # Reinstall Python dependencies. Prefer .[all], but if one optional extra breaks on this machine, keep + # base deps and reinstall the remaining extras individually so update does not silently strip working + # capabilities. Ownership preflight (#83529): refuse before the first venv mutation if the venv contains + # foreign-owned files (sudo-pip residue) — the install below would die mid-mutation and brick the CLI. _refuse_update_if_venv_foreign_owned(_m().PROJECT_ROOT) # Self-lock deferral: if THIS process holds a native extension the sync must rewrite, defer # NOW (after the code swap) so only the install is pending for the next launch's marker. @@ -945,6 +1024,7 @@ def _sync_python_dependencies_after_pull( _m()._reload_updated_runtime_modules() # Stale pip can fail source builds and leave partially-written packages. + # See #57828. _write_lazy_refresh_incomplete_marker() _m()._upgrade_pip_before_lazy_refresh(install_prefix, env=lazy_env) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index 972ed1d3bb..eebb6732a6 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -21,6 +21,8 @@ logger = logging.getLogger("hermes_cli.update_cmd") # Under HERMES_HOME (not next to the venv): records the fleet-restart obligation # after a pull advanced HEAD; cleared only when the restart completes or nothing ran. +# The existing ``.update-incomplete`` / ``.lazy-refresh-incomplete`` markers gate dependency/venv repair; +# this one is the fleet-restart obligation after a git pull that advanced HEAD (#95294). _FLEET_RESTART_PENDING_NAME = "fleet_restart_pending" _FRESH_RESTART_SUPERVISORS = frozenset({"systemd", "launchd", "service", "s6"}) @@ -91,6 +93,8 @@ def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: Prefer the post-restart ``fleet`` matrix. ``plan.runtimes[].code_sha`` is captured *before* the pull, so a finished update's plan always looks stale and must not retrigger a restart; consult it only for an unfinished receipt. + + See #95294. """ from hermes_cli.update_cmd import _current_checkout_sha try: @@ -127,7 +131,10 @@ def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: def _pending_fleet_restart_needed() -> bool: - """True when a prior pull still owes the fleet a restart.""" + """True when a prior pull still owes the fleet a restart. + + See #95294. + """ with suppress(OSError): if _fleet_restart_pending_marker_path().is_file(): return True @@ -199,11 +206,17 @@ def _run_pending_fleet_restart() -> bool: """Catch-up restart for gateways left on pre-update code. Never raises. True when the restart completed or nothing was running; False if incomplete. + + See #95294. """ from hermes_cli.update_cmd import _m print("→ Restarting gateways left on pre-update code...") with suppress(Exception): _m()._purge_stale_hermes_modules() + # Warn if legacy Hermes gateway unit files are still installed. When both hermes.service (from a + # pre-rename install) and the current hermes-gateway.service are enabled, they SIGTERM-fight for the + # same bot token (see PR #11909). Flagging here means every `hermes update` surfaces the issue until the + # user migrates. try: from hermes_cli.gateway import ( find_gateway_pids, is_macos, is_windows, kill_gateway_processes, supports_systemd_services, @@ -225,8 +238,13 @@ def _run_pending_fleet_restart() -> bool: failed: list = [] try: + # --- Systemd services (Linux) --- Discover all hermes-gateway* units (default + profiles) plus + # hermes-serve* units (the Desktop app's backend, #83438). if supports_systemd_services(): _restart_systemd_gateway_units_best_effort(failed) + # --- Launchd services (macOS) --- Restart EVERY ai.hermes.gateway* LaunchAgent, not only the + # invoking profile's — parity with the systemd branch above (#41403). Per-label TimeoutExpired + # isolation happens inside. if is_macos(): try: _restart_macos_launchd_gateways([], failed, 45.0) @@ -298,6 +316,8 @@ def _is_hermes_gateway_unit(unit: str) -> bool: """Exact base unit or hyphenated profile family only: ``startswith("hermes-serve")`` would accept ``hermes-server.service``.""" return ( + # list-units is already pattern-filtered, but keep the name gate so a stray non-gateway/serve line + # cannot enter the restart path. See #83595. unit == "hermes-gateway.service" or unit.startswith("hermes-gateway-") or unit == "hermes-serve.service" @@ -310,6 +330,8 @@ def _for_each_systemd_gateway_unit(list_units_stdout: str, *, process_unit, on_u ``TimeoutExpired`` from ``process_unit`` is isolated per unit via ``on_unit_timeout`` so one wedged systemctl call cannot abort the rest of the fleet. + + See #68523. """ for line in (list_units_stdout or "").strip().splitlines(): parts = line.split() @@ -332,6 +354,8 @@ def _service_unit_supports_graceful_sigusr1_restart(svc_name: str) -> bool: kill ``hermes-serve*`` and burn the drain budget, so those go straight to the blunt restart. Same exact/hyphenated shape as ``_for_each_systemd_gateway_unit`` so a near-prefix unit like ``hermes-gatewayd`` is never signalled. + + See #83438. """ return svc_name == "hermes-gateway" or svc_name.startswith("hermes-gateway-") @@ -349,6 +373,7 @@ def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None: if is_macos(): # A label lands here when launchd wasn't supervising a live process after # the restart — likely deregistered, which `launchctl kickstart` can't revive. + # See #88848. print(" Listed services may be deregistered from launchd, or still") print(" running pre-update code (mixed sys.modules). Recover with:") print(" hermes gateway status") @@ -374,6 +399,13 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) -> ``launchd_restart()`` always runs; every failure path is loud with a manual recovery command. Returns ``(restarted_labels, failed_labels)``; with ``supervision_verify`` success also requires a fresh supervised PID ("the call returned" is not "supervised"). + + 74973 (salvage #75021 by @jeff-mettel): the restart used to be gated on ``launchctl list <label>`` + exiting 0. A *booted-out* job — plist present, definition deregistered from launchd (crashed helper, + manual bootout, failed prior update) — fails that check, so the whole branch silently skipped: no + restart, no message, ``KeepAlive`` unable to revive a definition launchd no longer knows, and the update + still printed "Update complete!". + See #88848. """ from hermes_cli.gateway import ( get_launchd_label, get_launchd_plist_path, launchd_restart, wait_for_launchd_gateway_supervision, @@ -396,6 +428,7 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) -> # A plist exists, so a gateway is SUPPOSED to be supervised; a broken/wedged # launchctl is not proof nothing needs restarting. Count it, tell the operator. print( + # The old code `pass`ed here (#74973's second silent variant); count it and tell the operator. " ⚠ Could not restart the gateway " f"({e.__class__.__name__}: {e}).\n" " Recover manually: hermes gateway restart" @@ -409,6 +442,8 @@ def _restart_launchd_gateway_after_update(*, supervision_verify: bool = True) -> # before first bootstrap, or a bootstrap exiting 0 without registering (macOS 26.6.1), # would otherwise reach "Update complete!" unsupervised. Verified domain-agnostically: # domain locate fails on macOS-26 per-user domains. + # launchd_restart() returning is only "restart REQUESTED" — the self-restart branch hands work to the + # running gateway, a plist reload to a detached helper; both asynchronous. See #88848. if wait_for_launchd_gateway_supervision(label=current_label): return [current_label], [] print( @@ -426,6 +461,11 @@ def _restart_macos_launchd_gateways(restarted_services: list, failed_or_stale_un Invoking profile uses ``launchd_restart()``; siblings get the same drain-first sequence with their domain (``gui/<uid>`` vs ``user/<uid>``) resolved per label so none is kickstarted in the wrong domain. ``TimeoutExpired`` is isolated per label. + + See #41403. + The invoking profile keeps the existing ``launchd_restart()`` treatment (self-restart request → graceful + drain → kickstart). ``subprocess.TimeoutExpired`` is isolated per label so one wedged launchctl call + cannot leave the rest of the fleet on old code (#68523). """ from hermes_cli.gateway import ( get_launchd_label, launchd_gateway_labels_for_install, _graceful_restart_via_sigusr1, _launchd_kickstart, @@ -561,6 +601,10 @@ def _warn_gateway_restart_phase_aborted(exc: BaseException, pids) -> None: Previously a blanket debug-logged ``except Exception`` erased every drain/restart line, so "Update complete!" exited 0 while the gateway kept serving pre-update modules and died on the next turn with an ImportError. + + Issue #78574: the gateway auto-restart phase was wrapped in a blanket ``except Exception`` that only + logged at debug level, so an early failure (e.g. importing ``hermes_cli.gateway`` from the freshly + pulled checkout) erased every drain/restart line from the update output. """ print() print(f"⚠ Update incomplete — gateway auto-restart failed: {exc}") @@ -585,6 +629,10 @@ def _drain_or_signal_gateway_for_update(pid: int, drain_budget: float, label: st So fire-and-forget: signal restart and return; it completes once THIS process exits. 2. Event loop provably wedged: SIGUSR1 can never drain it; bounded SIGTERM→SIGKILL. 3. Live out-of-tree gateway: graceful SIGUSR1 drain up to ``drain_budget``. + + The wedged-loop probe cannot break it: the cron session posts activity every ~180s (process-tool poll + return), so it is "actively waiting forever" and never marked wedged — the gateway burns the full + force-drain cap (1800s) before killing its own updater's session. See #86684. """ from hermes_cli.gateway import ( GATEWAY_LOOP_WEDGED, _escalate_wedged_gateway, _graceful_restart_via_sigusr1, @@ -760,6 +808,7 @@ def _restart_systemd_gateway_units(restarted_services, failed_or_stale_units, re # later gateway and leave the fleet on mixed code. failed_or_stale_units.append(svc_name) print( + # See #68523. f" ⚠ systemctl timed out restarting {svc_name} " f"({exc.cmd if exc.cmd else 'unknown command'}); " f"continuing with remaining gateways" @@ -833,6 +882,10 @@ def _restart_manual_gateways(out: _GatewayRestartOutcome, _drain_budget) -> None } # Profile gateways we couldn't arm a relaunch for must NOT keep running stale: # the unmapped sweep below stops them and lists them under "Restart manually". + # These must NOT be left running: their modules are the pre-update ones and every lazy import from here + # on mixes versions against the new code on disk (#88654). Handing them to the unmapped sweep below + # stops them and surfaces them in the "Stopped N manual gateway process(es) / Restart manually" summary, + # which is the contract already used for gateways with no profile mapping. unrestartable_pids = set() for pid, proc in profile_processes.items(): restart_mode = _prepare_profile_gateway_update_restart(proc.profile, pid) @@ -891,6 +944,10 @@ def _force_kill_stuck_gateways(killed_pids) -> None: moment, then SIGKILL remaining pre-update PIDs.""" with _best_effort('Post-restart survivor sweep failed: %s'): from hermes_cli.gateway import find_gateway_pids, _get_service_pids + # --- Post-restart survivor sweep ----------------------------- Issue #17648: some gateways ignore + # SIGTERM (stuck drain, blocked I/O, PID dead but zombie). The detached profile watchers wait 120s + # for the old PID to exit — if it never does, no respawn happens and the user keeps hitting + # ImportError against a stale sys.modules. _time.sleep(3.0) _surviving = find_gateway_pids(exclude_pids=_get_service_pids(all_profiles=True), all_profiles=True) # Only PIDs we already tried to kill; newer ones are left alone. @@ -920,6 +977,12 @@ def _recover_after_restart_phase_abort( # Restart output never printed: assume stale unless provably no gateway runs. # Empty ``_surviving`` proves safety only if nothing ran beforehand; a gone # pre-restart gateway was stopped without verified replacement → fail closed. + # An exception escaping the whole phase means the drain/restart output the user relies on never printed. + # Don't let that pass for a clean update: surface it and treat the fleet as stale unless we can + # positively prove no gateway is running (#78574). A positive-empty ``_surviving`` is only + # proof-of-safety when nothing was running before we touched anything. If a gateway was discovered + # pre-restart and none survive now, it was stopped and its replacement was never verified — the same + # fail-open contract this fix closes — so we must still fail closed on ``[]``. _surviving = _surviving_gateway_pids_after_failed_restart() _planned_gateway_profiles = { runtime.profile @@ -991,6 +1054,13 @@ def _restart_gateway_fleet_after_update(_pre_update_plan, gateway_mode: bool): # Scope-qualified twin (``user/hermes-serve`` vs ``system/hermes-serve`` are different # processes; abort recovery needs WHICH settled). Bare names stay in # ``restarted_services`` for the fleet probe, receipt and summary. + # Snapshot of gateways running before we touch anything. Stays empty until we successfully import the + # probe and are about to stop/drain — so an exception raised before we touch any gateway keeps this + # empty (nothing to fail closed on), while a failure after we have stopped a discovered gateway lets the + # handler fail closed on an empty survivor probe rather than reporting a clean update (#78574). + # Declared outside the restart try/except below (and never reset to None) so it's always safe to read + # afterwards even if that block raises before reaching its own restart bookkeeping — needed to forward + # already-restarted units to ``_finish_dashboard_update_cleanup`` (review on #83595). restarted_scoped_units: set = set() # Purge stale cached Hermes modules FIRST: the import below loads new gateway @@ -1116,6 +1186,12 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re # units, so a unit-less `hermes serve` keeps stale sys.modules. Runs AFTER # dashboard cleanup so a respawned manual dashboard isn't a survivor. Rows feed # reconciliation (survivor → exit 1); ``None`` = probe failed, stays fail-closed. + # Check if any pre-update serve/dashboard runtimes survived on pre-update code generations (#100479). + # This is the SUCCESS-path twin of the abort-recovery probe above: the restart phase only restarts + # units, so an sshd-spawned `serve --isolated` or a manual `hermes serve` (no unit) is left running its + # pre-update sys.modules graph — and its cron ticker keeps firing agent jobs that ImportError on every + # symbol added in the pulled range. The rows also feed the plan-vs-execution reconciliation below, so a + # survivor is escalated (exit 1) instead of merely printed. _stale_serve_rows: "list | None" = None with _best_effort('Failed to check for surviving serve runtimes: %s'): _stale_serve_rows = _surviving_pre_update_serve_runtimes(_pre_update_plan) @@ -1128,12 +1204,14 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re # Compare every live gateway's stamped code_sha against the fresh checkout # instead of assuming the restart phase worked. + # Phase 1 (#91277): post-update fleet version verification. _fleet_snapshot: list = [] with _best_effort('Fleet version verification failed: %s'): from hermes_cli.update_receipt import print_fleet_version_matrix # Cross-platform "rows expected" signal: (restarted_services or killed_pids) # never fires on Windows (pause/resume populates neither), so a healthy # resumed gateway yielded zero rows and exit 0. + # See #93406. _fleet_rows_expected = _m()._fleet_probe_expected_runtimes( _pre_update_plan, restart.pre_restart_gateway_pids, _windows_gateway_resume, restart.restarted_services, restart.killed_pids, @@ -1145,6 +1223,12 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re # collect_fleet_versions() swallows every failure, so zero rows with # expected runtimes is indistinguishable from health — fail (partial, exit 1). print( + # Fleet probe returned zero rows even though at least one gateway runtime was (or may have + # been) live pre-update — POSIX restart bookkeeping, the pre-restart PID snapshot, the + # pre-update plan inventory, or the Windows pause/resume token all count as that signal. + # Every failure path inside collect_fleet_versions() is swallowed via logger.debug(), so an + # empty list is indistinguishable from a healthy fleet in the current output. Treat it as + # verification failure so the receipt records "partial" and the exit code is 1 (#93406). "\n⚠ Fleet version check returned no rows even though" " gateway runtimes were expected — verification incomplete." ) @@ -1153,6 +1237,9 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re # Every runtime the PLAN saw must appear in restart bookkeeping; an # unaccounted one is a silent miss and escalates like a STALE/DOWN row. with _best_effort('Runtime-outcome reconciliation failed: %s'): + # An unaccounted runtime is the silent-miss class (a platform branch re-discovered its own targets + # and skipped one the inventory knew about) — escalate it exactly like a STALE/DOWN fleet row. See + # #91277. if _pre_update_plan is not None and _pre_update_plan.runtimes: from hermes_cli.update_inventory import (match_runtime_outcomes, report_unaccounted_runtimes) _runtime_outcomes = match_runtime_outcomes( @@ -1163,6 +1250,7 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re killed_pids=restart.killed_pids, failed_units=restart.failed_or_stale_units, # Serve/dashboard reconcile by incarnation liveness, not unit names. + # See #100479. stale_serve_pids=( {row.get("pid") for row in _stale_serve_rows} if _stale_serve_rows is not None @@ -1198,6 +1286,12 @@ def _restart_phase_failure_is_incomplete(surviving, pre_restart_pids) -> bool: Fail closed unless provably safe: ``surviving`` None (unprobeable) or non-empty → stale. ``[]`` proves safety ONLY if nothing ran beforehand; a pre-restart gateway (``pre_restart_pids`` non-empty or None) now gone was stopped unverified. + + * ``surviving is None`` — the survivor probe could not determine state (typically the freshly-pulled + ``hermes_cli.gateway`` no longer imports, one of the ways the phase aborts). That is proof-of-safety + ONLY when nothing was running before we touched anything. If a gateway was discovered pre-restart + (``pre_restart_pids`` non-empty, or ``None`` meaning the pre-state could not be read), it was stopped + without a verified replacement, so we still fail closed (#78574). """ if surviving is None or surviving: return True @@ -1219,8 +1313,13 @@ def _fleet_probe_expected_runtimes( profile resumes DETACHED). Counting it made every Windows update that paused a gateway exit 1 after a long silent wait; a live pre-update Windows gateway is already covered by ``pre_restart_pids`` and the plan. The same condition gates the settle sleep. + + See #93406. + See #78574. + See #93406. """ del windows_resume_token # excluded on purpose — see docstring + # See #93406. if restarted_services or killed_pids: return True if pre_restart_pids is None or pre_restart_pids: diff --git a/hermes_cli/update_cmd_git.py b/hermes_cli/update_cmd_git.py index fcd47cd989..b479cb8373 100644 --- a/hermes_cli/update_cmd_git.py +++ b/hermes_cli/update_cmd_git.py @@ -46,7 +46,12 @@ def _prune_orphan_rescue_refs( Each ref pins a possibly multi-GB snapshot against ``git gc``, so a repeatedly corrupted install would grow ``.git`` unbounded. Keep the ``keep`` newest AND drop any older than ``max_age_days`` by the ``YYYYMMDD-HHMMSS`` stamp (unparseable names left alone); names sort chronologically so - ``for-each-ref`` order is creation order. Best-effort, never blocks.""" + ``for-each-ref`` order is creation order. Best-effort, never blocks. + + A rescue ref pins every object reachable from that commit against ``git gc`` — and in the incident shape + those objects include a full working-tree snapshot (the autostash orphan commit), which can be multi-GB + when the tree holds large stray files. See #87694. + """ from hermes_cli.update_cmd import _git_run with suppress(OSError): prefix = f"refs/hermes-update-backups/orphan-{branch}-" @@ -264,7 +269,10 @@ def _sync_with_upstream_if_needed(git_cmd: list[str], cwd: Path, *, assume_yes: Returns True only when origin/main was actually verified against upstream/main; False when the check never happened, so the caller never reports "up to date" on an origin-only compare. Fetches only upstream/main: - a bare fetch drags in thousands of auto-generated branches.""" + a bare fetch drags in thousands of auto-generated branches. + + See #97052. + """ from hermes_cli.update_cmd import _count_commits_between, _has_upstream_remote, _no_prompt_git_kwargs, _should_skip_upstream_prompt if not _has_upstream_remote(git_cmd, cwd) and ( _should_skip_upstream_prompt() or not _offer_upstream_remote(git_cmd, cwd, assume_yes=assume_yes, input_fn=input_fn) @@ -359,13 +367,22 @@ def _git_is_trampoline(git_cmd: list) -> bool: The ~46KB ``bin\\git.exe``/``cmd\\git.exe`` shims re-exec git-core; when they can't find it every call dies with the launcher's guard message (a PATH problem, not network). Never raises; unknown states - report False so a probe failure can't block an update.""" + report False so a probe failure can't block an update. + + Git for Windows ships two ~46KB shims (``bin\\git.exe``, ``cmd\\git.exe``) that re-exec the real + ``mingw64\\libexec\\git-core\\git.exe``. See #87876. + """ return _probe_fork_bomb(git_cmd) is True def _portable_git_candidates() -> list: """PortableGit candidates: shared root first (where the managed tree actually lives, not the - profile-scoped HERMES_HOME), then profile home as a fallback for custom layouts.""" + profile-scoped HERMES_HOME), then profile home as a fallback for custom layouts. + + The Hermes-managed PortableGit tree lives under the SHARED root (``<root>/git/...``), not the + profile-scoped HERMES_HOME (``<root>/profiles/<name>``), so a profile-scoped ``hermes update`` must look + there (monerostar review, #87876). + """ from hermes_cli.update_cmd import get_default_hermes_root, get_hermes_home candidates = [] with suppress(Exception): @@ -376,7 +393,12 @@ def _portable_git_candidates() -> list: def _locate_real_git() -> Optional[Path]: """Find a real Git-for-Windows ``git-core/git.exe`` (standard locations + managed PortableGit) that runs without the trampoline guard. None when nothing suits — callers keep the broken command and let the - fetch-failure ZIP fallback handle it. A failed probe (None) disqualifies a candidate like a guard hit.""" + fetch-failure ZIP fallback handle it. A failed probe (None) disqualifies a candidate like a guard hit. + + The trampoline symptom is PATH-level: ``bin\\git.exe`` / ``cmd\\git.exe`` (both ~46KB shims) fail to + re-exec git-core, while the real binary at ``mingw64\\libexec\\git-core\\git.exe`` (≈4.4MB) works when + invoked directly (#87876). + """ candidates = [ Path(r"C:\Program Files\Git\mingw64\libexec\git-core\git.exe"), Path(r"C:\Program Files (x86)\Git\mingw64\libexec\git-core\git.exe"), @@ -424,7 +446,12 @@ def _normalize_managed_eol(git_cmd, repo_root): fix them. Pin and cleanup are one operation: under ``autocrlf=true`` a CRLF tree reads clean, so pinning alone would expose every file as modified (whole-tree autostash). Pin only after the tree verifies clean under it; a checkout we can't fully normalize is left as-is. Only ``true`` rewrites LF->CRLF - (unset/false/input leave the tree alone). Best-effort.""" + (unset/false/input leave the tree alone). Best-effort. + + Checkouts created before that landed never got the pin and cannot receive it — the bootstrap installer + reuses its build-pinned ``install.ps1`` forever — so ``hermes update``, which ships with the checkout + itself, is the only path left that can fix them. See #67730. + """ from hermes_cli.update_cmd import _git_run # -c, not config: evaluate the tree as it WOULD look pinned, persisting nothing. probe = git_cmd + ["-c", "core.autocrlf=false"] diff --git a/hermes_cli/update_cmd_maint.py b/hermes_cli/update_cmd_maint.py index b479268f11..3c3ada00b7 100644 --- a/hermes_cli/update_cmd_maint.py +++ b/hermes_cli/update_cmd_maint.py @@ -291,7 +291,14 @@ def _reload_process_scan_modules() -> None: """Reload the process-scan modules, dependency-first, so ``dashboard_procs`` binds against a fresh ``_subprocess_compat``: cleanup runs in the PRE-update process and a symbol the update added would otherwise ImportError after the code update succeeded. Called from the cleanup - entry point so every caller (git path, ZIP fallback) is covered.""" + entry point so every caller (git path, ZIP fallback) is covered. + + ``_finish_dashboard_update_cleanup`` runs in the PRE-update Python process, but + ``_scan_dashboard_processes`` does a function-level ``from hermes_cli._subprocess_compat import + bounded_probe_run``. If the update added a new symbol to ``_subprocess_compat`` (as #87134 did with + ``bounded_probe_run``), the cached OLD module object doesn't have it and the cleanup step crashes with + ImportError — after the code update itself already succeeded. + """ _reload_modules( ("hermes_cli._subprocess_compat", "hermes_cli.dashboard_procs"), modules=sys.modules, @@ -309,6 +316,8 @@ def _finish_dashboard_update_cleanup( *already_restarted_units*: systemd unit names (no ``.service``) the fleet-restart loop already restarted, so a Serve-only install isn't restarted a second time here. + + See #83595. """ from hermes_cli.update_cmd import _m, _reload_process_scan_modules if node_failures: @@ -333,7 +342,10 @@ def _finish_dashboard_update_cleanup( def _print_update_completion(message: str) -> None: """Print the outcome (with branch @ sha so drift is visible) plus, when launched by the - dashboard with an action id, a receipt line the Desktop matches after restart.""" + dashboard with an action id, a receipt line the Desktop matches after restart. + + See #47359, #58764. + """ from hermes_cli.update_cmd import _branch_head_suffix print(f"{message}{_branch_head_suffix()}") action_id = os.environ.get("HERMES_ACTION_ID", "") @@ -356,7 +368,12 @@ def _read_project_version() -> str | None: def _update_complete_message(pre_version: str | None) -> str: """Completion line with ``vA → vB`` when known; plain when either side is unknown or - the version did not change.""" + the version did not change. + + Ported from PrimeIntellect-ai/prime-agent#630: after a successful self-update, show both versions + (``v0.19.4 → v0.20.0``) so the user can see what they actually got. Falls back to the plain message when + either side is unknown or the version did not change (e.g. several commits landed within one release). + """ post_version = _read_project_version() if pre_version and post_version and pre_version != post_version: return f"✓ Update complete! (v{pre_version} → v{post_version})" @@ -410,7 +427,10 @@ def _clear_stale_sqlite_sidecars(db_path: Path) -> None: def _print_update_summary(*, node_failures: list, desktop_build_ok: bool, pre_update_version: str | None) -> bool: - """Final banner. A failed Desktop rebuild is non-fatal but must not print ``✓ Update complete!``.""" + """Final banner. A failed Desktop rebuild is non-fatal but must not print ``✓ Update complete!``. + + See #88251. + """ from hermes_cli.update_cmd import _post_update_sqlite_runtime_status, _update_complete_message sqlite_runtime_ok, sqlite_info = _post_update_sqlite_runtime_status() if sqlite_info is None: @@ -518,7 +538,10 @@ def _verify_and_restore_one_state_db(home: Path, *, label: str) -> None: def _verify_and_restore_state_dbs_post_update() -> None: """Integrity guard for the ROOT state.db AND every sibling profile's (the snapshot covers - siblings, so the guard must too or a corrupt profile DB goes undetected).""" + siblings, so the guard must too or a corrupt profile DB goes undetected). + + See #97994. + """ from hermes_cli.update_cmd import get_hermes_home home = get_hermes_home() _verify_and_restore_one_state_db(home, label="default home") @@ -579,6 +602,13 @@ def _ensure_fhs_path_guard() -> None: "-c", "command -v hermes", ], + # Fallback: blunt systemctl restart. This is what the old code always did; we get here only when + # the graceful path failed (unit missing SIGUSR1 wiring, drain exceeded the budget, + # restart-policy mismatch). Always `reset-failed` first. If systemd's own auto-restart attempts + # already parked the unit in a failed state (transient CHDIR / OOM / filesystem race after our + # drain + exit-75), a plain `systemctl restart` can wedge against the RestartSec backoff and + # leave the unit dead. Clearing the failed state first makes the restart idempotent. Mirrors the + # recovery path in `hermes gateway restart` (`systemd_restart()`) as of PR #20949. capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=10, @@ -625,6 +655,8 @@ def _ensure_acp_launcher() -> None: No-op on Windows (install.ps1 stages launchers into ``$HermesHome\bin``, never ``venv\Scripts`` which would shadow the user's python; launcher repair lives in _install_repair) and where it already exists. Unwritable dirs are skipped. Idempotent. + + ``/usr/local/bin`` as non-root) are skipped silently. See #83797. """ from hermes_cli.update_cmd import _m if _m().sys.platform == "win32": @@ -636,6 +668,9 @@ def _ensure_acp_launcher() -> None: if not (hermes_cmd.is_file() or hermes_cmd.is_symlink()): continue # is_symlink() catches broken symlinks exists() misses; never follow-and-overwrite. + # Already present — a console script (pip/pipx install), an earlier shim, or a symlink. + # is_symlink() catches broken symlinks that exists() would miss; never follow-and-overwrite (the + # #21454 failure mode). if acp_cmd.exists() or acp_cmd.is_symlink(): continue shim = ( @@ -796,6 +831,8 @@ def _run_pre_update_backup(args) -> Optional[str]: ``off`` — nothing. ``quick`` (default) — snapshot of critical small files under ``state-snapshots/``, files over 1 GiB skipped so a bloated state.db can't stall the update. ``full`` — quick snapshot PLUS a zip of HERMES_HOME under ``backups/`` (``hermes import``). + + Explicit user opt-out is honored fully. See #34600. """ mode = _resolve_pre_update_backup_mode(args) @@ -823,6 +860,10 @@ def _sweep_bytecode_after_update(branch: str) -> None: """Clear stale ``__pycache__`` (else gateway restart ImportErrors on names absent from old bytecode), re-stamp the fingerprint, refresh the bootstrap cache scripts.""" from hermes_cli.update_cmd import _m + # The update process is still the old Python interpreter process. Run one final cache/module refresh + # immediately before lazy backend refresh, which imports newly-pulled modules that may depend on fresh + # symbols in hermes_constants or lazy_deps. The dependency install above may also have regenerated + # bytecode from build-cache copies — this second sweep catches those stragglers (#60242, #65240). removed = _m()._clear_bytecode_cache(_m().PROJECT_ROOT) if removed: print(f" ✓ Cleared {removed} stale __pycache__ director{'y' if removed == 1 else 'ies'}") @@ -862,6 +903,7 @@ def _sync_profiles_after_update() -> None: # Backfill .env for profiles created before .env seeding (copy the default's) so they # keep the credentials they were effectively using. with suppress(Exception): + # See #44792. from hermes_cli.profiles import backfill_profile_envs backfilled = backfill_profile_envs(quiet=True) if backfilled: @@ -929,6 +971,10 @@ def _run_post_update_maintenance( from hermes_cli.update_cmd import _check_and_apply_config_migration, _m # macOS TCC: Desktop bundles are re-signed each update, so old grants can go stale # (toggle ON, yet macOS re-prompts with no Allow button). Tell users how to re-grant. + # With the post-#73681 identifier-pinned DR, new grants survive rebuilds — but a grant made to a pre-fix + # binary stays stale: the System Settings toggle shows ON while macOS re-prompts on every capture, and + # the modern prompt has no Allow button, so users loop. One line of guidance after update tells affected + # users how to complete the one-time re-grant. if sys.platform == "darwin" and had_desktop_app_before_update: print() print( @@ -941,6 +987,7 @@ def _run_post_update_maintenance( # macOS TCC interpreter anchor; boot-gated — a failed probe leaves the venv untouched. try: + # See #95596. from hermes_cli.macos_tcc_anchor import ensure_tcc_anchor ensure_tcc_anchor() except Exception: diff --git a/hermes_cli/update_cmd_stash.py b/hermes_cli/update_cmd_stash.py index 75e416429e..70be0cb7a0 100644 --- a/hermes_cli/update_cmd_stash.py +++ b/hermes_cli/update_cmd_stash.py @@ -19,6 +19,8 @@ _AUTOSTASH_NAME_PREFIX = "hermes-update-autostash-" #: Age past which a leftover autostash is called out. Younger entries are normal #: (recent --keep-stash park); older ones are almost always forgotten. +# Entries younger than this are normal (a parked stash from : the desktop updater's --keep-stash run minutes +# ago); older ones are almost : always forgotten (#63717 problem 6: an orphan persisted 9+ days unnoticed). _AUTOSTASH_WARN_AGE_DAYS = 7 _STASH_LEFT_IN_PLACE = " The stash was left in place. You can remove it manually after checking the result." @@ -106,6 +108,11 @@ def _warn_orphaned_update_autostashes(git_cmd: list[str], cwd: Path) -> int: Autostashes legitimately outlive a run (--keep-stash, failed restore) but nothing re-surfaces them. Deliberately NOT a GC: a stash may be the only copy of the user's work, so Hermes never drops one. + + Autostash entries legitimately outlive an update run (``--keep-stash`` parks them; a conflicted or + failed restore preserves them for safety), but nothing ever re-surfaces them afterwards — they sit in + ``git stash`` invisibly for weeks (#63717 problem 6). This prints a short notice naming the stale + entries with recovery/cleanup guidance. """ from hermes_cli.update_cmd import _git_run try: diff --git a/hermes_cli/update_cmd_windows.py b/hermes_cli/update_cmd_windows.py index acefb6a5d2..97deeed3c9 100644 --- a/hermes_cli/update_cmd_windows.py +++ b/hermes_cli/update_cmd_windows.py @@ -87,6 +87,10 @@ def _self_and_non_gateway_ancestor_pids(psutil) -> set[int]: gracefully; a detached child survives on Windows); interactive ancestry is never a blocker.""" _is_gw = None with suppress(Exception): + # Never return ourselves or our own ancestry: a CLI ``hermes update`` runs from the venv python and + # would otherwise nominate itself. Same #87594 carve-out as _detect_venv_python_processes: a GATEWAY + # ancestor is not "our own ancestry" in the interactive sense — it is the process the pause + # machinery must see (the /update-from-gateway topology makes the updater the gateway's child). from gateway.status import looks_like_gateway_command_line as _is_gw skip: set[int] = {os.getpid()} with suppress(Exception): @@ -199,7 +203,13 @@ def _holder_value_flags() -> frozenset: Derived so the holder classifier can't drift from argparse (a handwritten subset misparsed ``--reasoning high serve``). Pre-argparse profile selectors are added explicitly (stripped before argparse sees argv). Falls back - to a static snapshot when the parser can't import — the updater must classify holders even on a broken tree.""" + to a static snapshot when the parser can't import — the updater must classify holders even on a broken tree. + + Introspects ``build_top_level_parser()`` (every option with nargs != 0) so the holder classifier can + never drift from the argparse surface (#91869 review: a handwritten subset misparsed ``--reasoning high + serve`` as subcommand ``high`` and ``-m dashboard serve`` as ``dashboard`` — recreating the wrong-hint + class). + """ global _holder_value_flags_cache if _holder_value_flags_cache is not None: return _holder_value_flags_cache @@ -219,7 +229,11 @@ def _hermes_holder_subcommand(cmdline: str) -> str | None: """The actual Hermes SUBCOMMAND a venv-holder argv runs, or None (callers must NOT guess a label). Token-based, never substring (``kanban --preserve-cache`` contains "serve"): find the ``hermes_cli.main`` / - ``hermes(.exe)`` entry token, return the first following token that isn't a flag or a flag's value.""" + ``hermes(.exe)`` entry token, return the first following token that isn't a flag or a flag's value. + + Profile selectors (``--profile X``, ``-p X``) are skipped like the canonical gateway matcher does. See + #90778. + """ try: tokens = shlex.split(cmdline, posix=False) except Exception: @@ -250,7 +264,10 @@ def _format_venv_python_holders_message(matches: list[tuple[int, str, str]]) -> """Explain which venv processes block the update and how to clear them. Labels come from the parsed SUBCOMMAND, never substring: a standalone ``hermes dashboard`` must not be - called the Desktop backend, ``--preserve-cache`` must not match "serve". Unknown argv gets no hint.""" + called the Desktop backend, ``--preserve-cache`` must not match "serve". Unknown argv gets no hint. + + See #90778. + """ hint_by_subcommand = { "serve": " ← Hermes backend (if the Desktop app is open, close it)", "dashboard": " ← hermes dashboard (stop it: hermes dashboard stop, or close that terminal)", @@ -318,7 +335,11 @@ def _leftover_pausable_gateway_pids(matches: list[tuple[int, str, str]]) -> list def _refuse_gateway_ancestor_tree_kill(pids: list[int], *, gateway_mode: bool) -> bool: """Refuse a plain Windows update that would tree-kill its own ancestry (a chat agent's ``hermes update`` is a gateway child; ``taskkill /T /F`` kills the updater first). ``--gateway`` is exempt (detached delivery). - Refuse only when a nominated gateway is positively an ancestor; unknown ancestry keeps existing recovery.""" + Refuse only when a nominated gateway is positively an ancestor; unknown ancestry keeps existing recovery. + + The leftover holder recovery below uses ``taskkill /T /F`` on Windows, so force-stopping that gateway + also kills the updater before it can mutate the checkout (#98814). + """ if gateway_mode or not pids: return False def _ancestors(): @@ -426,7 +447,18 @@ def _orphaned_desktop_backend_pids(matches: list[tuple[int, str, str]]) -> list[ would dead-end the update with "Hermes is still running" and zero open windows. Qualifies only if cmdline is a Hermes backend AND the parent is demonstrably gone (PID missing or reused). Tree-aware: holders inside an accepted root's tree fold into it; only roots are returned (``taskkill /T`` reaps descendants). Any - live-parent backend, unjustified non-backend, unprovable case, or no psutil -> ``None``. Never raises.""" + live-parent backend, unjustified non-backend, unprovable case, or no psutil -> ``None``. Never raises. + + The venv-holder guard refuses on the Desktop app's ``serve`` backend by design: while the Desktop is + open, killing its backend is futile (the app supervises and respawns it within seconds), so the user + must close the app. But in the GUI-updater handoff path the Desktop has *already exited* — by contract + it tree-kills its backends and waits for the venv shim before spawning hermes-setup, and the + update-in-progress marker parks any relaunched Desktop from spawning a fresh backend (#50238). A + ``serve`` backend still holding the venv at that point is a straggler whose supervisor is gone: SIGTERM + raced its spawn, or it belongs to a crashed window. Nothing will respawn it, and refusing on it + dead-ends the update with "Hermes is still running" while the user stares at zero open windows (ryanc's + 2026-08-09 01:59/02:17 failures). + """ psutil = _psutil() if psutil is None: return None @@ -501,7 +533,11 @@ def _handoff_reapable_backend_pids(matches: list[tuple[int, str, str]]) -> list[ The orphan-only rung bails on ANY live parent (mid-teardown Electron, launcher->worker chain) and hung a hand-off. Inside the hand-off gate (marker + ``--gateway`` + no live ``hermes.exe`` shim) nothing legitimate supervises a ``serve`` from this venv, so survivors are leaks. Any non-backend holder or no psutil -> - ``None``. The CALLER must have confirmed the gate; outside it the stricter orphan-only path stands.""" + ``None``. The CALLER must have confirmed the gate; outside it the stricter orphan-only path stands. + + Any ``serve`` backend still holding the venv here is therefore a leak, live parent or not, and reaping + its tree is correct rather than a race. See #50238. + """ psutil = _psutil() if psutil is None: return None @@ -519,7 +555,10 @@ def _handoff_reapable_backend_pids(matches: list[tuple[int, str, str]]) -> list[ def _stop_process_trees(pids: list[int] | list[tuple[int, int]]) -> None: """Force-stop each PID with its full child tree (Windows); best effort, never raises. - ``taskkill /T /F``: stopping only the parent can leave a ``.hermes-runtime`` child holding the install open.""" + ``taskkill /T /F``: stopping only the parent can leave a ``.hermes-runtime`` child holding the install open. + + See #70026. + """ from gateway.status import get_process_start_time from hermes_cli._subprocess_compat import pid_is_hermes, windows_hide_flags for entry in pids: @@ -545,7 +584,12 @@ def _looks_like_desktop_control_plane(cmdline: str) -> bool: Not the messaging gateway — don't feed into ``looks_like_gateway_command_line``. Token-based via the parser-derived classifier, never substring (``kanban --preserve-cache``, ``-m dashboard chat``). - Undeterminable subcommand is NOT a control plane.""" + Undeterminable subcommand is NOT a control plane. + + See #92091. + A cmdline whose subcommand cannot be determined is NOT a control plane — callers must not guess + ownership. See #90778, #91869. + """ return "hermes_cli.main" in (cmdline or "").lower() and _hermes_holder_subcommand(cmdline) in _BACKEND_PURPOSES @@ -554,7 +598,10 @@ def _desktop_owns_gateway_lifecycle() -> bool: Not proof messaging is served: serve is the control plane, the gateway a detached sibling. Prefer the spawn ledger; fall back to the venv-holder scan. An orphaned control plane (supervisor gone) does not count; - without psutil orphanhood is unprovable and a live control plane suffices.""" + without psutil orphanhood is unprovable and a live control plane suffices. + + See #76129, #92091. + """ from hermes_cli.update_cmd import _m with _best_effort('Desktop-lifecycle ledger probe failed: %s'): from hermes_cli.process_identity import ledger_entries, spawner_is_dead @@ -778,6 +825,11 @@ def _request_socket_pauses(running_pids, profile_processes, service_gateway_pids mapped_pids.append(int(pid)) _write_update_planned_stop_marker(Path(proc.path), int(pid)) try: + # Socket-first pause (#92091 step 2): ask the gateway to drain and exit itself instead of + # relying on the marker poll + force-kill ladder. A positive ACK means the gateway is running + # its own graceful restart path (same drain as SIGUSR1/service restarts) and will release its + # venv handles on the way out. No answer (older gateway, no socket) → the marker watcher / + # force-kill fallback below behaves exactly as before this verb existed. from gateway.control_socket import pause_gateway_for_update ack = pause_gateway_for_update(Path(proc.path)) if ack and (ack.get("pausing") or ack.get("already_stopping")): @@ -856,7 +908,14 @@ def _cold_start_windows_gateway_after_update() -> bool: Idempotent: re-checks nothing is running so a concurrent autostart can't duplicate. A successful Popen doesn't prove survival (a job object denying breakaway kills it), so success is gated on the liveness poll. - Vouched PIDs are attested so a death AFTER updater exit is reported by the next CLI invocation.""" + Vouched PIDs are attested so a death AFTER updater exit is reported by the next CLI invocation. + + A successful ``Popen`` only proves the process was created, not that it survived (e.g. a Windows job + object denying breakaway kills it before it logs anything — #84185). So the success line is gated on the + same post-spawn liveness poll every other ``_spawn_detached`` caller uses + (``gateway_windows._report_gateway_start``), instead of being printed unconditionally from the returned + PID. + """ from hermes_cli.update_cmd import _desktop_owns_gateway_lifecycle, _m if not _m()._is_windows(): return True @@ -887,7 +946,14 @@ def _refresh_windows_gateway_launchers() -> None: """Regenerate installed Windows gateway launcher scripts after update; best-effort, never fails the update. Launchers are written once at install, so old installs kept launching via ``pythonw.exe`` (``sys.stderr is - None`` death). The task's /TR points at a stable path, so rewriting in place retargets it without UAC.""" + None`` death). The task's /TR points at a stable path, so rewriting in place retargets it without UAC. + + The Scheduled Task / Startup-folder launchers (``gateway.cmd`` + ``gateway.vbs``) are persistence + artifacts written once at install time — ``hermes update`` never touched them, so installs created + before the hidden-console rework (aa2ae36c3f) kept launching the gateway through ``pythonw.exe`` + forever: every descendant spawn flashed a conhost (#54220/#56747) and, since #70344, the console-less + gateway died at startup with ``RuntimeError: sys.stderr is None`` (#71671). + """ from hermes_cli.update_cmd import _m if not _m()._is_windows(): return @@ -903,7 +969,19 @@ def _refresh_bootstrap_cache_scripts(branch: str = "main") -> None: Old ``hermes-setup.exe`` builds NEVER re-download a cached branch-ref script, so a stale one runs months-old code forever. Guards mirror ``install_script.rs``: only the sanitized *branch* key is rewritten; - commit-SHA pins (7-40 hex) are immutable and skipped. Best-effort: never fails the update.""" + commit-SHA pins (7-40 hex) are immutable and skipped. Best-effort: never fails the update. + + Installer binaries built before the #67193 cache-refresh fix (June 2026 and earlier) NEVER re-download a + cached branch-ref script — ``install-main.ps1`` cached at install time is reused forever, executing + months-stale code with long-fixed bugs (the 2026-08-09 incident: a June 4 cached script's venv stage + lacked the 81327 process-tree sweep and died on ``Access denied``). The binary has no self-update path, + so the poisoned cache outlives every ``hermes update``. + Overwriting the cached script for *branch* with the freshly pulled ``scripts/install.ps1`` / + ``scripts/install.sh`` on every update turns the stale binary's unconditional reuse into a feature: it + "reuses" a file this function keeps permanently current. Post-#67193 installers re-download on each run + anyway, so for them this is a harmless pre-seed of the same bytes. + The .ps1 copy gets a UTF-8 BOM to match the installer's cache format (#67193 encoding fix). + """ from hermes_cli.update_cmd import _m with _best_effort('Could not refresh bootstrap-cache scripts after update: %s'): cache_dir = Path(_m().get_hermes_home()) / "bootstrap-cache" @@ -921,6 +999,7 @@ def _refresh_bootstrap_cache_scripts(branch: str = "main") -> None: data = src.read_bytes() if kind == "ps1" and not data.startswith(b"\xef\xbb\xbf"): data = b"\xef\xbb\xbf" + data # PowerShell needs the BOM or localized/em-dash text mis-decodes. + # See #67193. if cached.read_bytes() == data: continue tmp = cached.with_suffix(cached.suffix + ".tmp") @@ -974,6 +1053,14 @@ def _relaunch_paused_gateways(token: dict, profiles: dict, unmapped: list) -> tu relaunched.append(str(profile)) else: failed_profiles[str(profile)] = int(old_pid) + # Surface the outcome on the token (#91277 Phase 2 plan-vs-execution reconciliation): the git-based + # update path's fleet reconciliation cross-checks every planned runtime against restarted_services / + # relaunched_profiles / externally_supervised_profiles / killed_pids — bookkeeping this Windows-specific + # pause/resume never fed, so a correctly-paused-and-relaunched Windows gateway was reported + # "unaccounted" (loud warning + exit 1) even though the restart succeeded. The caller merges this into + # the shared relaunched_profiles list before reconciliation runs. A profile whose relaunch genuinely + # failed is deliberately left off this list — it must still surface as unaccounted so the user is told + # to restart it manually (Windows has no watcher to recover a failed relaunch). token["relaunched_profiles"] = relaunched unmapped_relaunched = 0 failed_unmapped = [] diff --git a/hermes_cli/update_cmd_zip.py b/hermes_cli/update_cmd_zip.py index b86c719f9f..01a98c6233 100644 --- a/hermes_cli/update_cmd_zip.py +++ b/hermes_cli/update_cmd_zip.py @@ -41,7 +41,15 @@ def _remove_path(path: str, *, ignore_errors: bool = False) -> None: def _atomic_replace_dir(src: str, dst: str) -> None: """Replace *dst* with *src* without a half-deleted window (naive ``rmtree; copytree`` loses the old tree when the copy fails partway — likely here, since the ZIP path only runs when file I/O is flaky). - Thin alias over the two-phase helpers; retained for the ``hermes_cli.main`` re-export surface.""" + Thin alias over the two-phase helpers; retained for the ``hermes_cli.main`` re-export surface. + + The naive ``rmtree(dst); copytree(src, dst)`` has a destructive window: if the copy fails partway + (common on the Windows ZIP-update path, which only runs because file I/O is already flaky on that + machine), the old directory is already gone and nothing replaced it — the install is left with a deleted + tree (issue #49145, where ``ui-tui/`` vanished and broke the TUI). + Now a thin single-entry alias over the two-phase helpers below, which generalise the same + stage-then-swap discipline across every entry the ZIP update touches (#76104). + """ _commit_staged_replacements([(_stage_replacement(src, dst), dst)]) @@ -78,6 +86,9 @@ def _commit_staged_replacements(staged) -> None: Each swap is an ``os.rename`` onto a just-moved-aside path — atomic on POSIX and NTFS, unlike ``copy2`` onto a live path. Stage-all-then-swap-all shrinks the failure window to N renames and a failed swap restores every entry already swapped, so the tree lands wholly new or wholly old. + + ``_atomic_replace_dir`` makes each *individual* directory swap safe, but the ZIP update replaces ~90 + top-level entries in a loop, and nothing made the loop atomic *as a whole*. See #63717, #76091, #76104. """ swapped: list[tuple[str, str]] = [] # (dst, backup) in swap order; "" = absent try: @@ -111,6 +122,8 @@ def _zip_overlay_block_reason(root: Path, *, ignore_staging_artifacts: bool = Fa edits and untracked files are gone. Fails closed when git status cannot run. ``ignore_staging_artifacts`` is for the pre-swap re-check: phase 1 leaves our own ``*.hermes-update-staging`` siblings that git reports as untracked; without the filter the re-check always refuses. + + Fail closed when git status cannot run: unknown dirtiness is not a license to clobber the tree (#87304). """ if not (root / ".git").exists(): return None @@ -119,6 +132,8 @@ def _zip_overlay_block_reason(root: Path, *, ignore_staging_artifacts: bool = Fa # -uall: a user-level ``status.showUntrackedFiles = no`` must not blind this guard. --ignored=matching: # gitignored files are still USER DATA the overlay would delete; ``matching`` reports an ignored dir # as one ``dir/`` line. ``--ignored=all`` is NOT a valid git mode (exits 128, would fail-close every update). + # ``matching`` reports an ignored directory as one ``dir/`` line instead of enumerating its contents + # (cheaper, same verdict for the top-level filter below). See #87392. git_cmd + ["status", "--porcelain", "--untracked-files=all", "--ignored=matching"], cwd=root, capture_output=True, text=True, encoding="utf-8", errors="replace", ) @@ -158,7 +173,10 @@ def _is_zip_staging_artifact_status_line(line: str) -> bool: def _abort_zip_update_if_dirty_tree() -> None: - """Refuse to overlay a ZIP onto a dirty git checkout.""" + """Refuse to overlay a ZIP onto a dirty git checkout. + + See #87304. + """ from hermes_cli.update_cmd import _m reason = _zip_overlay_block_reason(_m().PROJECT_ROOT) if reason is None: @@ -202,6 +220,13 @@ def _require_staging_space(extracted: str, entries: list[str], project_root: str """Staging costs one extra tree copy; swaps are renames, so require the copy plus 20% headroom — not 2x, which would block updates on exactly the space-constrained machines that hit this path.""" need = sum( + # Two-phase replace (#76104). Phase 1 copies every entry — directories AND top-level files — to a + # sibling staging path without touching anything live; phase 2 swaps them all in with + # same-filesystem renames and rolls back every swap if any one fails. Replacing entries + # one-at-a-time (the previous shape) meant an interruption partway left `agent/` new and `tools/` + # stale — all files valid, the tree unbootable. Files matter as much as directories here: the repo + # root holds 20 first-party modules (run_agent.py, cli.py, hermes_constants.py, ...). Check up front + # so we fail with a clear message instead of running out mid-copy. os.path.getsize(os.path.join(dirpath, f)) for entry in entries for dirpath, _dirs, files in os.walk(os.path.join(extracted, entry)) @@ -226,6 +251,9 @@ def _stage_entries(extracted: str, entries: list[str], project_root: str) -> lis staged.append((_stage_replacement(os.path.join(extracted, item), dst), dst)) # The source ZIP lacks apps/desktop/release/ (the BUILT desktop app); swapping `apps` without # it deletes the build and breaks the shortcut. Graft the live release dir in BEFORE the swap. + # #70337/#87331: the GitHub source ZIP contains only source — apps/desktop/release/ (the BUILT + # desktop app, win-unpacked/ Hermes.exe) exists only in the LIVE tree. Graft the live release + # dir into the staged copy BEFORE the swap so the commit preserves it atomically. if item == "apps": live_release = os.path.join(dst, "desktop", "release") staged_release = os.path.join(staged[-1][0], "desktop", "release") @@ -314,6 +342,7 @@ def _reinstall_python_deps_after_zip(active_tool_dependencies) -> None: except _shim_quarantine_error_type() as _sqe: # Runs inside the ZIP-fallback error handler, so cmd_update's boundary except cannot catch # it — refuse here with the same defer-via-marker contract. + # See #87331. _refuse_update_for_contended_shims(_sqe) install_prefix, install_env = [uv_bin, "pip"], uv_env else: @@ -354,6 +383,9 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo _sweep_bytecode_after_update(branch) # Self-lock deferral: the code swap is committed; defer only the dependency sync when this process # holds a native extension the sync must rewrite. + # Reinstall Python dependencies. Prefer .[all], but if one optional extra breaks on this machine, keep + # base deps and reinstall the remaining extras individually so update does not silently strip working + # capabilities. See #86735. _m()._abort_dependency_sync_if_self_locked() print("→ Updating Python dependencies...") _reinstall_python_deps_after_zip(active_tool_dependencies) @@ -384,6 +416,7 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo print(" ✓ Model catalog cache refreshed from checkout") # state.db integrity guard: root home AND every sibling profile, each auto-restored from its own snapshot. with _best_effort('Post-update state.db integrity check (zip path) failed: %s'): + # See #97994. _verify_and_restore_state_dbs_post_update() update_complete = _print_update_summary( node_failures=node_failures, desktop_build_ok=desktop_build_ok, pre_update_version=pre_update_version, @@ -393,6 +426,7 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo with _best_effort('Curator recent-run notice failed: %s'): _print_curator_recent_run_notice() # Don't stop a working dashboard when the Node refresh failed — see the git-update path for rationale. + # See #30271. _finish_dashboard_update_cleanup(node_failures) with _best_effort('Update receipt finalize (zip path) failed: %s'): from hermes_cli.update_receipt import finalize_update_receipt diff --git a/hermes_cli/update_inventory.py b/hermes_cli/update_inventory.py index 5a43cdd762..97dbc937f5 100644 --- a/hermes_cli/update_inventory.py +++ b/hermes_cli/update_inventory.py @@ -24,6 +24,7 @@ class RuntimeRecord: pid: Optional[int] = None supervisor: str = "manual" # systemd | launchd | desktop | windows-service | service | manual | manual-serve code_sha: Optional[str] = None # stamped running-code sha + # See #91283. code_version: Optional[str] = None restart_via: str = "" # mechanism id, see _RESTART_MECHANISMS detail: dict = field(default_factory=dict) @@ -50,6 +51,7 @@ def _detect_supervisor_for_pid(pid: int, service_pids: set, windows_service_pids if windows_service_pids and pid in windows_service_pids: # SCM-supervised Windows gateway: the update pause machinery stops the SERVICE via sc.exe # instead of killing the child, so reconciliation must plan it under its own mechanism id. + # See #91277. return "windows-service" if pid not in service_pids: return "manual" @@ -83,7 +85,13 @@ _SERVE_KINDS = ("serve", "dashboard") def _restart_mechanism(supervisor: str, profile: str) -> str: - """Machine-readable restart mechanism id for a runtime.""" + """Machine-readable restart mechanism id for a runtime. + + THE policy table (#91277 Phase 2): restart execution consumes these ids via + :func:`match_runtime_outcomes` / the update's restart phase, and the receipt records per-runtime + outcomes against them. Display strings are derived by :func:`describe_restart_mechanism` — never the + other way around. + """ return _RESTART_MECHANISMS.get(supervisor, "manual") @@ -127,6 +135,7 @@ def _collect_install_shape(plan: UpdatePlan) -> None: # container can look like `git` while the running filesystem is an immutable image. # Fail-closed: an invalid marker still flips the plan to not-updatable. with _probe("Image provenance probe"): + # See #91277. from hermes_cli.image_provenance import read_image_provenance provenance = read_image_provenance() @@ -146,6 +155,9 @@ def _supervisor_classifier() -> Callable[[int], str]: service_pids = _get_service_pids(all_profiles=True) or set() # Windows SCM services (no-op off Windows): the update's pause phase stops these via `sc.exe # stop` / restarts via `sc.exe start`, so the plan must carry the matching mechanism id. + # --- SCM-supervised gateway PIDs (Windows) ------------------------------ + # find_windows_gateway_services() maps validated gateway PIDs through process ancestry to running SCM + # service PIDs (no-op off Windows). See #91277. windows_service_pids: set = set() with _probe("Windows SCM service-ownership probe"): from hermes_cli.gateway import find_windows_gateway_services @@ -265,7 +277,12 @@ def print_update_plan(plan: UpdatePlan) -> None: def _serve_unit_matches_profile(profile: str, unit: object) -> bool: """Does *unit* name a ``hermes-serve*``/``hermes-dashboard*`` unit for *profile*? (OWN vocabulary; - the gateway's ``hermes-gateway*`` names never cover serve/dashboard runtimes.)""" + the gateway's ``hermes-gateway*`` names never cover serve/dashboard runtimes.) + + Exact names only — ``work`` must not claim ``hermes-serve-workbench`` — and a scope prefix + (``user/hermes-serve``) is tolerated because the restart phase records scope-qualified identities in + some lists. See #100479. + """ name = str(unit).removesuffix(".service").rsplit("/", 1)[-1] suffix = "" if profile == "default" else f"-{profile}" return name in {f"hermes-serve{suffix}", f"hermes-dashboard{suffix}"} @@ -294,6 +311,10 @@ def match_runtime_outcomes( reconciled in their OWN vocabulary and never borrow the gateway's outcome: with ``stale_serve_pids`` a pre-update serve whose incarnation is gone counts as ``restarted``, one still alive is ``unaccounted``; without the probe an untouched serve stays ``unaccounted``. + + See #91277. + They never borrow the gateway's outcome: ``relaunched_profiles`` and ``hermes-gateway*`` name a + different process that shares the profile, nothing more. See #100479. """ outcomes: list[dict[str, Any]] = [] try: @@ -353,6 +374,7 @@ def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool: print(" hermes -p <profile> gateway restart # named profile") if any(o.get("kind") in _SERVE_KINDS for o in missed): # A serve/dashboard is not reachable by any `gateway restart` command: name the process, not the wrong verb. + # See #100479. print(" systemctl --user restart hermes-serve.service # unit-managed serve") print(" relaunch `hermes serve` / `hermes dashboard` / the Desktop app") return True diff --git a/hermes_cli/update_receipt.py b/hermes_cli/update_receipt.py index 034128c921..1a4fecabab 100644 --- a/hermes_cli/update_receipt.py +++ b/hermes_cli/update_receipt.py @@ -193,6 +193,9 @@ def finalize_pending_update_receipt(exit_code: Optional[int] = None, stop_reason fetch failure) predating the inner finalize calls; finalizing here means refused/failed runs — where a receipt matters most — leave a record. Exit 0/None → ``success``, exit 2 → ``refused`` (preflight convention), else → ``failed``. + + No-op when no receipt is open (the inner paths already finalized — exactly-once via the popped + singleton) or when recording was never started. See #91283. """ if _current is None: return None @@ -247,6 +250,10 @@ def _socket_identity(home: Path) -> Optional[tuple[int, dict]]: didn't bind. """ try: + # Prefer the gateway-owned control socket (#92091): identity declared by the process itself, + # including its own supervisor provenance — no argv/PID inference. Scan fallback below. + # Prefer the gateway-owned control socket (#92091): a live `identify` answer is authoritative — no + # PID-reuse or stale-file heuristics. from gateway.control_socket import identify_gateway identity = identify_gateway(home) @@ -276,6 +283,14 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l Rollout safety: ``down`` requires membership in ``pre_restart_pids`` — a stale state file from a long-dead gateway (machine reboot, manual kill weeks ago) must NOT fail every future update. Without a pre-restart snapshot (``None``/empty) dead PIDs are skipped (historical behavior). + + ``stale`` — gateway stamped a code_sha that differs from the updated checkout's HEAD (it is still + serving pre-update modules). ``unknown`` — gateway predates the code-identity stamp (started before this + feature landed) or identity could not be resolved. ``down`` — the gateway was ALIVE when this update + started (``pre_restart_pids``), its runtime status still says running, but the PID is dead and no + successor rewrote the record: the restart phase stopped it and nothing came back. Without this row a + killed-and-never-replaced gateway produced NO entry at all and the matrix passed silently (Phase-1 + verification gap, #88848/#74973 class). """ _pre_restart = {int(p) for p in (pre_restart_pids or []) if isinstance(p, int)} results: list[dict[str, Any]] = [] @@ -309,6 +324,7 @@ def collect_fleet_versions(*, pre_restart_pids: Optional[list[int]] = None) -> l # behavior so the rollout can't false-positive. ``_pre_restart`` is a bare PID set, not # (pid, start_time) pairs, so a recycled PID from gateway A landing in B's stale record # could still mislabel B as down — inherent to the snapshot's data model. + # See #93258. gw_state = record.get("gateway_state") if pid in _pre_restart and isinstance(gw_state, str) and gw_state and gw_state not in _NOT_EXPECTED_STATES: results.append(_fleet_row(profile, pid, None, record.get("code_version"), None, state="down")) diff --git a/hermes_cli/voice.py b/hermes_cli/voice.py index 9c949abb42..872519482c 100644 --- a/hermes_cli/voice.py +++ b/hermes_cli/voice.py @@ -20,6 +20,8 @@ from typing import Any, Callable, Optional _VOICE_MOD_ALIASES = {"ctrl": "c-", "control": "c-", "alt": "a-", "option": "a-", "opt": "a-"} # Named keys prompt_toolkit accepts as ``c-<name>`` / ``a-<name>``; aliases collapse to canonical. +# Aliases collapse to prompt_toolkit's canonical spelling so the same config value binds identically in both +# runtimes (Copilot round-10 on #19835). _VOICE_NAMED_KEYS = { "space": "space", "spc": "space", "enter": "enter", "return": "enter", "ret": "enter", "tab": "tab", "escape": "escape", "esc": "escape", "backspace": "backspace", "bs": "backspace", "delete": "delete", "del": "delete", @@ -31,12 +33,23 @@ _VOICE_NAMED_KEYS = { # ``key.meta``), mirroring the TUI's darwin-only reservation — alt is reserved on darwin only. _VOICE_RESERVED_CHARS = frozenset({"c", "d", "l"}) +# On macOS the classic CLI's prompt_toolkit bindings for copy / exit / clear also claim ``a-c`` / ``a-d`` / +# ``a-l`` via the action-modifier lookup, and hermes-ink reports Alt as ``key.meta`` on many terminals. +# Mirror the TUI parser's darwin-only reservation so ``option+c`` etc. don't bind Alt+C in the CLI while the +# TUI silently falls back to Ctrl+B (Copilot round-14 on #19835). _DEFAULT_PT_KEY = "c-b" def voice_record_key_from_config(cfg: Any) -> Any: """Shape-safe ``cfg.voice.record_key``: a hand-edited ``voice: true`` / ``voice: cmd+b`` leaves - ``cfg["voice"]`` as a bool/str and the naive ``.get`` chain would raise before voice starts.""" + ``cfg["voice"]`` as a bool/str and the naive ``.get`` chain would raise before voice starts. + + ``load_config()`` deep-merges raw YAML and preserves scalar overrides, so a hand-edited ``voice: true`` + / ``voice: cmd+b`` leaves ``cfg["voice"]`` as a bool/str instead of a dict, and the naive + ``.get("voice", {}).get("record_key")`` chain raises AttributeError before voice can even start (Copilot + round-11 on 19835). Return ``None`` for malformed shapes so call sites can feed the result straight into + the normalizer/formatter and get the documented default. See #19835. + """ voice = cfg.get("voice") if isinstance(cfg, dict) else None return voice.get("record_key") if isinstance(voice, dict) else None @@ -49,6 +62,10 @@ def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str: ``c-b``; named keys collapse to canonical spelling (``ctrl+return`` → ``c-enter``). Exactly one modifier: multi-modifier chords bind different shortcuts in prompt_toolkit (a-c-r) and hermes-ink rejects them; a bare key is refused by the TUI parser. + + * ``super`` / ``win`` / ``windows`` → ``c-b`` (TUI-only modifiers — prompt_toolkit has no super mod; the + CLI binding site is expected to warn when this fallback fires so users see the cross-runtime split, + Copilot round-11 on #19835) """ if not isinstance(raw, str): return _DEFAULT_PT_KEY @@ -56,6 +73,10 @@ def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str: if len(parts) != 2: return _DEFAULT_PT_KEY modifier_token, key_token = parts + # ``super`` / ``win`` / ``windows`` are TUI-only (prompt_toolkit has no super modifier, so + # ``@kb.add(super+b)`` crashes the CLI at startup). Fall back to the documented default here; the CLI + # binding site is expected to log a warning when the configured value is one of these spellings so users + # know the TUI+CLI runtimes diverge on that shortcut (Copilot round-11 on #19835). normalized_mod = _VOICE_MOD_ALIASES.get(modifier_token) if not normalized_mod: return _DEFAULT_PT_KEY @@ -77,7 +98,11 @@ def pt_key_to_sequence(pt_key: str) -> tuple[str, ...]: def format_voice_record_key_for_status(raw: Any) -> str: """Render ``voice.record_key`` for ``/voice status`` as ``Ctrl+B`` / ``Alt+Space``; malformed - configs surface as the default so status never advertises a shortcut that won't bind.""" + configs surface as the default so status never advertises a shortcut that won't bind. + + Mirrors the TUI's ``formatVoiceRecordKey``: returns ``Ctrl+B`` / ``Alt+Space`` / ``Ctrl+Enter``. See + #19835. + """ normalized = normalize_voice_record_key_for_prompt_toolkit(raw) prefix = "Alt+" if normalized.startswith("a-") else "Ctrl+" key = normalized[2:] @@ -113,6 +138,8 @@ def _beeps_enabled() -> bool: voice_cfg = load_config().get("voice", {}) if isinstance(voice_cfg, dict): # is_truthy_value handles quoted YAML strings like "false" that bool() misreads. + # See #49883. + # See #49883. return is_truthy_value(voice_cfg.get("beep_enabled", True), default=True) except Exception: pass @@ -545,6 +572,10 @@ def _speak_streaming(text: str, stop_event: Optional[threading.Event]) -> bool: """ import queue + # One dispatcher, zero parallel streaming implementations (#58930): when the configured provider has a + # chunked streamer registered in tools.tts_streaming, route the whole reply through the same + # stream_tts_to_speaker pipeline the CLI voice mode uses — audio starts on sentence one instead of after + # full synthesis. Falls through to the legacy whole-file path when no streamer resolves. from tools.tts_streaming import resolve_streaming_provider from tools.tts_tool import _load_tts_config, stream_tts_to_speaker diff --git a/hermes_cli/web_git.py b/hermes_cli/web_git.py index 6f001a1e82..cef77ee936 100644 --- a/hermes_cli/web_git.py +++ b/hermes_cli/web_git.py @@ -586,6 +586,11 @@ def _worktree_for_existing(root: str, raw_name: str) -> dict: requested = _sanitize_branch(raw_name) if not requested: raise RuntimeError("Branch name is required.") + # "origin/feature" is a remote-tracking ref, not a branch git can check out — `git worktree add <dir> + # origin/feature` detaches HEAD. Create a local branch with the same short name that tracks the remote + # ref, like `git switch feature` does for a branch on exactly one remote. (Parity with the Electron op; + # a remote gateway serves this mirror, so the desktop's convert-a-branch flow must behave identically. + # #81724) remote = _remote_of_ref(root, requested) existing = requested.split("/", 1)[1] if remote else requested if not remote and existing == _default_branch(root): @@ -645,7 +650,10 @@ def _ref_names(cwd: str, *patterns: str, fmt: str = "%(refname:short)") -> list[ def branch_list(cwd: str) -> list[dict]: """Branches for the convert-a-branch picker: local heads first, then remote-tracking - refs with no local head yet (a teammate's branch without a manual checkout).""" + refs with no local head yet (a teammate's branch without a manual checkout). + + Parity with the Electron op — a remote gateway serves this mirror for the same desktop UI (#81724). + """ locals_ = _ref_names(cwd, "refs/heads") if not locals_: return [] diff --git a/hermes_cli/web_routers/actions.py b/hermes_cli/web_routers/actions.py index e6f8f0bb06..4251139ab2 100644 --- a/hermes_cli/web_routers/actions.py +++ b/hermes_cli/web_routers/actions.py @@ -360,6 +360,7 @@ async def get_action_status(name: str, lines: int = 200): # update (written by every run, incl. refused/failed; survives the # dashboard restarting itself mid-action). Surface it so clients READ # the outcome instead of inferring it from liveness probes. + # See #81193, #87359, #91277. update_receipt_summary = _latest_update_receipt_summary() proc = _ACTION_PROCS.get(name) @@ -396,7 +397,13 @@ def _read_latest_receipt() -> Optional[Dict[str, Any]]: def _latest_update_receipt_summary() -> Optional[Dict[str, Any]]: """Compact summary of the latest receipt (written by EVERY ``hermes update`` run, - incl. refused/failed), or None; never raises. Steps/skips stay in the full endpoint.""" + incl. refused/failed), or None; never raises. Steps/skips stay in the full endpoint. + + Phase-1 bullet 3 (#91277): the receipt (written by EVERY ``hermes update`` run since #91283, including + refused and failed ones, with a ``latest.json`` pointer) is the durable success signal the Desktop and + dashboard should read instead of inferring outcomes from liveness probes across the update's stop/start + gap (#81193, #87359). + """ receipt = _read_latest_receipt() if not receipt: return None @@ -417,7 +424,10 @@ async def get_update_receipt(): """The FULL latest update receipt (steps, skips, gateway restart outcome, fleet matrix) plus a compact ``summary``; 404 when no update has run since receipts landed. Clients read this instead of inferring success from backend liveness, which misread - the update's own restart gap as a failed update/boot.""" + the update's own restart gap as a failed update/boot. + + See #81193, #87359, #91277. + """ receipt = _read_latest_receipt() if not receipt: raise HTTPException(status_code=404, detail="No update receipt found (no `hermes update` run recorded).") diff --git a/hermes_cli/web_routers/analytics.py b/hermes_cli/web_routers/analytics.py index 95927dc044..46cd3eae9d 100644 --- a/hermes_cli/web_routers/analytics.py +++ b/hermes_cli/web_routers/analytics.py @@ -59,6 +59,7 @@ async def update_config_raw(body: RawConfigUpdate, profile: Optional[str] = None with _profile_scope(body.profile or profile): # Full-document replacement: the editor owns the whole file; never # merge omitted sections back from disk. + # See #62723. approvals_mode_changed = _approval_mode_of(parsed) != _approval_mode_of(read_raw_config()) save_config(parsed, merge_existing=False) # Same indicator refresh as the schema-driven save. @@ -109,6 +110,8 @@ def _get_usage_analytics(days: int = 30, profile: Optional[str] = None): # Fold in auxiliary usage (vision, compression, ...) from session_model_usage. # Aux calls never touch the sessions counters, so this is add-only — no double count. + # Without it the models list shows only the main agent model even when aux models are actively + # burning tokens (issue #23270). aux_rows = _aux_usage_rows(db, cutoff) by_model = _merge_aux_into_by_model(by_model, aux_rows) @@ -252,6 +255,7 @@ def _get_models_analytics(days: int = 30, profile: Optional[str] = None): # Aux-only models (dedicated vision/compression) as (model, provider) rows, # keyed like the GROUP BY above, so they appear on the Models page. + # See #23270. for aux in _aux_usage_rows(db, cutoff): raw_rows.append({ "model": aux.get("model") or "unknown", diff --git a/hermes_cli/web_routers/chat_ws.py b/hermes_cli/web_routers/chat_ws.py index 81825bcced..d83d6075a7 100644 --- a/hermes_cli/web_routers/chat_ws.py +++ b/hermes_cli/web_routers/chat_ws.py @@ -450,6 +450,7 @@ async def pty_ws(ws: WebSocket) -> None: # The client only pins the viewport to the bottom when it asked # for `?resume=`; announce the implicit active-session replay so # it gets the same follow-scroll treatment. + # See #93518. await ws.send_json({"type": "resume", "id": resume}) resolve_kwargs = {"resume": resume, "sidecar_url": sidecar_url, "profile": profile} diff --git a/hermes_cli/web_routers/config_env.py b/hermes_cli/web_routers/config_env.py index 3f37548c6a..1592cc63a3 100644 --- a/hermes_cli/web_routers/config_env.py +++ b/hermes_cli/web_routers/config_env.py @@ -324,6 +324,8 @@ def _api_key_display(entry: Dict[str, Any]) -> Tuple[bool, Optional[str]]: Keys live in ``.env`` behind ``key_env``; only older entries still carry a plaintext ``api_key``. Checking both keeps the panel honest either way. + + See #69449. """ plaintext = str(entry.get("api_key") or "").strip() if plaintext: @@ -411,6 +413,8 @@ def _detach_main_model_from_provider(cfg: Dict[str, Any], provider_key: str) -> construction, so deleting the endpoint without clearing it leaves the agent authenticating to the deleted host with the deleted key (and the key in config.yaml). Only touches ``model`` when it names the deleted provider. + + See #62269. """ model_cfg = cfg.get("model") if not isinstance(model_cfg, dict): @@ -438,6 +442,9 @@ def _write_custom_endpoint(cfg: Dict[str, Any], body: CustomEndpointUpdate) -> T if not model: raise HTTPException(status_code=400, detail="model required") + # Deliver the bearer token through a named provider entry. A bare ``provider: custom`` cannot carry a + # credential for this host: OPENAI_API_KEY is deliberately gated to openai.com (#28660), so the token + # was dropped and requests went out as "no-key-required". providers = cfg.get("providers") if not isinstance(providers, dict): providers = {} @@ -459,6 +466,7 @@ def _write_custom_endpoint(cfg: Dict[str, Any], body: CustomEndpointUpdate) -> T # ``body.models`` is the catalogue the panel's Test button discovered; # without it only the hand-typed model survived Save. A payload with no # ``models`` (older UI) still ensures the named default is present. + # See #69988. existing_models = entry.get("models") models_map: Dict[str, Any] = dict(existing_models) if isinstance(existing_models, dict) else {} for candidate in (*(body.models or ()), model): @@ -475,6 +483,7 @@ def _write_custom_endpoint(cfg: Dict[str, Any], body: CustomEndpointUpdate) -> T # API keys never belong in config.yaml: write to .env and reference it via # ``key_env`` — the indirection built-in providers use and that # runtime_provider.py resolves at load time. + # See #69449. env_var = custom_endpoint_key_env(endpoint_id) submitted_key = body.api_key.strip() if body.api_key is not None else None if submitted_key: diff --git a/hermes_cli/web_routers/dashboard_ui.py b/hermes_cli/web_routers/dashboard_ui.py index 8333eb1f5c..bdfcf0345c 100644 --- a/hermes_cli/web_routers/dashboard_ui.py +++ b/hermes_cli/web_routers/dashboard_ui.py @@ -289,7 +289,10 @@ async def serve_plugin_asset(plugin_name: str, file_path: str): allowlist — user plugins ship a ``plugin_api.py`` backend the browser never fetches, and without it anyone on the loopback port could curl a private plugin's source. Path traversal is blocked via ``resolve().is_relative_to()``; - user plugins must be enabled (bundled ones not disabled) (GHSA-mcfc-hp25-cjv7).""" + user plugins must be enabled (bundled ones not disabled) (GHSA-mcfc-hp25-cjv7). + + See #46435. + """ plugins = _get_dashboard_plugins() plugin = next((p for p in plugins if p["name"] == plugin_name), None) if not plugin or not _plugin_activated(plugin, *_plugin_enable_sets()): diff --git a/hermes_cli/web_routers/files.py b/hermes_cli/web_routers/files.py index 0daa58fb4e..de2a3b8659 100644 --- a/hermes_cli/web_routers/files.py +++ b/hermes_cli/web_routers/files.py @@ -64,6 +64,8 @@ _FS_READDIR_HIDDEN = { # points the managed root at HERMES_HOME. Mirrors the two canonical guards # (agent.file_safety.get_read_block_error, gateway.platforms.base # ._ROOT_CREDENTIAL_FILES) so the Files tab never lags behind them. +# These typically contain credentials (API keys, tokens) and exposing them through the dashboard file +# browser is a security leak — see issue #57505. _SENSITIVE_MANAGED_FILE_BASENAMES = frozenset({ "auth.json", "auth.lock", "credentials", "config.yaml", ".anthropic_oauth.json", "google_token.json", "google_oauth_pending.json", "google_oauth.json", @@ -93,7 +95,12 @@ def _is_sensitive_filename(name: str) -> bool: def _is_sensitive_path(path: Path) -> bool: """True when the basename is sensitive OR any path component (case- insensitive) is a credential directory. Read-side guard (list/read/ - download); the write endpoints are a separate threat class.""" + download); the write endpoints are a separate threat class. + + Read-side only: this guards list/read/download (the #57505 exfil surface). The write endpoints + (upload/mkdir/delete) are a separate threat class handled by the write-path checks; extending this guard + to them is out of scope for this fix. + """ if _is_sensitive_filename(path.name): return True return any(part.lower() in _SENSITIVE_MANAGED_DIR_NAMES for part in path.parts) diff --git a/hermes_cli/web_routers/messaging.py b/hermes_cli/web_routers/messaging.py index 1295253a1b..56e3c169f1 100644 --- a/hermes_cli/web_routers/messaging.py +++ b/hermes_cli/web_routers/messaging.py @@ -769,6 +769,10 @@ async def cancel_telegram_onboarding(pairing_id: str): async def get_messaging_platforms(profile: Optional[str] = None): # Profile-scoped so the global profile switcher shows the TARGET profile's channel state. def _run(): + # Profile-scoped so the dashboard's global profile switcher shows the TARGET profile's channel + # credentials/state, not the root install's. load_env() honors the HERMES_HOME contextvar override; + # the gateway status readers do NOT (they resolve process-level paths), so the profile directory is + # passed explicitly for those (#71211). with _profile_scope(profile) as scoped_dir: return { "env_path": str(get_env_path()), diff --git a/hermes_cli/web_routers/models.py b/hermes_cli/web_routers/models.py index f29772df10..1fa483eda9 100644 --- a/hermes_cli/web_routers/models.py +++ b/hermes_cli/web_routers/models.py @@ -266,11 +266,13 @@ def set_moa_models(body: MoaConfigPayload, profile: Optional[str] = None): # incomplete slots for the hardcoded defaults — correct tolerance at READ time, # silent data loss at WRITE time (desktop autosave of a half-filled slot replaced # the user's whole preset). Refuse loudly so no client can corrupt config here. + # See #64156. problems = validate_moa_payload(raw) if problems: raise HTTPException(status_code=422, detail="Invalid MoA config: " + "; ".join(problems)) normalized = normalize_moa_config(raw) # Merge, don't overwrite: hand-edited keys not in MoaConfigPayload (save_traces, trace_dir) survive. + # See issue #58819. cfg.setdefault("moa", {}).update(normalized) save_config(cfg) return {"ok": True, **normalized} diff --git a/hermes_cli/web_routers/ops.py b/hermes_cli/web_routers/ops.py index 668d302c9a..e39813d471 100644 --- a/hermes_cli/web_routers/ops.py +++ b/hermes_cli/web_routers/ops.py @@ -337,6 +337,13 @@ async def add_credential_pool_entry(body: CredentialPoolAdd): label = (body.label or "").strip() or f"key #{len(pool.entries()) + 1}" pool.add_entry(PooledCredential( provider=provider, + # Add a distinct, self-contained pool entry per account (matching the qwen-oauth / + # minimax-oauth multi-account patterns, and the xai-oauth path below) instead of routing + # through the singleton ``_save_codex_tokens`` save path. The singleton round-trip collapsed + # every added account into the latest login: a second ``hermes auth add openai-codex`` + # overwrote the first account's singleton-mirrored ``device_code`` entry rather than + # creating an independent one (#39236). ``manual:device_code`` entries refresh from their + # own token pair, so they need no singleton shadow. id=uuid.uuid4().hex[:6], label=label, auth_type=AUTH_TYPE_API_KEY, @@ -376,6 +383,8 @@ async def remove_credential_pool_entry(provider: str, index: int): the same RemovalStep registry as ``hermes auth remove``: each source cleans its external state and suppresses ``(provider, source)`` so seeders skip it. Manual entries have no step — nothing external, and they aren't re-seeded. + + See #55217. """ from agent.credential_pool import load_pool from agent.credential_sources import find_removal_step diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 960b0bdb8a..dcba4fe9ce 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -427,6 +427,8 @@ def get_profiles_sessions_sidebar( ``recents_profile`` scopes the WHOLE payload, not just recents — the sidebar has one scope, so a concrete profile must never show another profile's Telegram threads or cronjobs; ``all`` asks for everything. + + See #42651, #65710, #70629. """ targets = _profile_targets("GET /api/profiles/sessions/sidebar", lightweight=True) diff --git a/hermes_cli/web_routers/sessions.py b/hermes_cli/web_routers/sessions.py index 0b9eadb12a..4939201f64 100644 --- a/hermes_cli/web_routers/sessions.py +++ b/hermes_cli/web_routers/sessions.py @@ -444,6 +444,10 @@ async def delete_empty_sessions_endpoint(profile: Optional[str] = None): "Empty" means NO ``messages`` rows at all — a rewound/compacted chat reads ``message_count == 0`` while its soft-archived rows are the only transcript copy (see :meth:`SessionDB.delete_empty_sessions`). + + * Active sessions are skipped (``ended_at IS NULL``) so a live agent isn't yanked mid-handshake. * + Archived sessions are skipped — the user explicitly chose to keep those rows. * Children of deleted + parents are orphaned, not cascade-deleted. See #95868. """ deleted = await asyncio.to_thread( _with_db, profile, lambda db: db.delete_empty_sessions(), read_only=False) @@ -579,6 +583,13 @@ async def backfill_session_owner_profiles(body: SessionOwnerBackfill): A multi-connection Desktop fails closed on unowned rows. Each ``state.db`` belongs to exactly one profile, so this is a single-match, idempotent backfill (non-NULL owners are never overwritten). + + That was fine while one backend served everything, but a Desktop with registry topology (≥2 registered + connections) fails closed on unowned rows by design — leaving every pre-campaign session unresumable + with no migration path. Each profile's ``state.db`` belongs to exactly one profile, so stamping that + store's own name is a single-match backfill, never a guess; the value written is the SAME + serving-profile identity the list endpoints already stamp onto outgoing rows (``row_profile`` in + ``get_sessions``). See #95407. """ stamp = _serving_profile(body.profile) diff --git a/hermes_cli/web_routers/status.py b/hermes_cli/web_routers/status.py index 7656cd84a6..08b4394778 100644 --- a/hermes_cli/web_routers/status.py +++ b/hermes_cli/web_routers/status.py @@ -176,6 +176,15 @@ def _merge_profile_gateway_platforms(gateway_platforms: dict, profile_platforms: return merged +# --- Gateway liveness detection --- Delegated to the single shared ladder in gateway.status so this +# endpoint and /api/messaging/platforms can never disagree about whether the gateway is up (they used to: +# sidebar "running" while the Channels page rendered "The gateway is not running"). When ?profile=<name> was +# given, scope PID and state reads to that profile's directory — gateway identity files (PID, lock, runtime +# status) are written to the per-profile home, not the process-level HERMES_HOME (see issue #69143). Plain +# /api/status keeps the exact zero-arg call so its behavior (and cache signature) is unchanged. The +# module-level probe references are handed to the resolver so the long-standing +# `monkeypatch.setattr(web_server, "get_running_pid_cached", ...)` seam used across the test-suite still +# intercepts them. def _bounded_health_probe(): """Health probe with the route's blocking-call budget preserved. The resolver only reaches this rung when the local PID probe came up empty, so the timeout is paid at diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index faae2f3b89..f25e206e91 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -107,6 +107,11 @@ def _start_desktop_cron_ticker(stop_event: "threading.Event", interval: int = 60 profile's store like a multiplex gateway; external providers keep the single-store behavior (registries are not profile-scoped). Cross-process safe: the built-in tick takes the per-store ``cron/.tick.lock``. + + Every local profile's store is ticked, not just this backend's own (#69377's desktop sibling): the + desktop pools per-profile backends and reaps them after ~10 idle minutes, so a secondary profile's + ticker dies with its backend and that profile's jobs silently stop firing until the user next opens it + ("tasks on the sleeping profile could be idle" — community report, Aug 2026). """ from cron.scheduler_provider import InProcessCronScheduler, resolve_cron_scheduler @@ -610,6 +615,9 @@ async def _plugin_api_runtime_gate(request: Request, call_next): or _has_valid_query_token(request, path) ): try: + # Gate: only serve user plugins that are in plugins.enabled and not in plugins.disabled. This + # prevents the frontend from loading JS/CSS from plugins the user has not explicitly activated. + # (#46435) from hermes_cli.plugins_cmd import _get_enabled_set, _get_disabled_set enabled_set = _get_enabled_set() disabled_set = _get_disabled_set() diff --git a/hermes_cli/web_server_chat.py b/hermes_cli/web_server_chat.py index ae64dab3b5..438fec2c3c 100644 --- a/hermes_cli/web_server_chat.py +++ b/hermes_cli/web_server_chat.py @@ -47,6 +47,7 @@ _PTY_READ_CHUNK_TIMEOUT = 0.2 # Back-off between idle PTY reads so a quiet terminal does not spin the event # loop (keeps dashboard idle CPU low). +# A positive sleep lets other coroutines run and keeps dashboard idle CPU low (#42627). _PTY_IDLE_BACKOFF = 0.05 PTY_REGISTRY = PtySessionRegistry( ttl=30 * 60, max_sessions=16, buffer_cap=1 * 1024 * 1024, read_timeout=_PTY_READ_CHUNK_TIMEOUT) @@ -54,7 +55,12 @@ PTY_REGISTRY = PtySessionRegistry( async def _legacy_pump(ws: "WebSocket", bridge) -> None: """Original 1:1 socket<->PTY pump: stream until disconnect, then close the - bridge. Used when no ``?attach=`` token is supplied (keep-alive opt-in).""" + bridge. Used when no ``?attach=`` token is supplied (keep-alive opt-in). + + Behavior is identical to the pre-keep-alive ``pty_ws`` body, including the 54028 half-open-socket + protection (reader EOF → close the WS so the writer's ``ws.receive()`` unparks) and the #53227 + ``to_thread`` offloads for the blocking ``bridge.close()``. + """ loop = asyncio.get_running_loop() async def pump_pty_to_ws() -> None: @@ -77,6 +83,8 @@ async def _legacy_pump(ws: "WebSocket", bridge) -> None: # here too (idempotent): cancelling the handler the instant the WS # closes can skip the writer's ``finally``. with contextlib.suppress(Exception): + # The child has exited (EOF) or the send side broke. Closing from the EOF path makes the + # reap independent of that cancellation race (#54028). await asyncio.to_thread(bridge.close) with contextlib.suppress(Exception): await ws.close() diff --git a/hermes_cli/web_server_config.py b/hermes_cli/web_server_config.py index 11bd6493c8..a9d622b5f8 100644 --- a/hermes_cli/web_server_config.py +++ b/hermes_cli/web_server_config.py @@ -35,6 +35,8 @@ def _memory_provider_options() -> List[str]: (built-in only) is always first; discovery failures degrade to the bundled defaults. The literal ``builtin`` alias is deliberately NOT offered — built-in memory is not a provider plugin; ``_normalize_memory_provider_name`` maps legacy aliases back to ``""``. + + See #49513. """ options = [""] try: @@ -189,6 +191,8 @@ _CATEGORY_MERGE: Dict[str, str] = { "telemetry": "security", "plugins": "agent", "doctor": "general", + # `runtime.nofile_soft_limit` (#78873) is the only schema-surfaced runtime field — fold it into the + # agent tab rather than spawning a one-field orphan category. "runtime": "agent", "session": "general", "nous": "agent", @@ -512,6 +516,9 @@ def _dashboard_code_skew_guard() -> Optional[str]: resolve a fresh consumer module against a stale cached dependency -> ImportError. Mirrors the gateway's ``_model_switch_skew_guard``: refuse the risky call with an actionable message. Never a false positive (non-git installs return None). + + ``/api/model/options`` 500 after the update added ``agent.model_metadata.is_grok_46_family`` while the + running process kept serving the pre-update module (#86207). """ from gateway.code_skew import detect_code_skew @@ -529,7 +536,10 @@ def _dashboard_code_skew_guard() -> Optional[str]: def _dashboard_skew_restart_hint() -> str: """Restart advice matching how this process is owned — the same app backs the browser dashboard and Desktop-owned ``hermes serve``; naming a systemd unit would mislead - macOS/launchd hosts and Desktop SSH backends.""" + macOS/launchd hosts and Desktop SSH backends. + + See #97046. + """ if os.environ.get("HERMES_SERVE_HEADLESS") == "1": return ( "restart the Desktop-owned backend to load the new code " @@ -558,6 +568,7 @@ def _resolve_assignment_credentials(model_cfg: dict, provider: str, provider_ent key_env = str(raw_entry.get("key_env") or "").strip() if key_env: model_cfg["key_env"] = key_env + # #88990: carry the credential POINTER, never a resolved secret. model_cfg.pop("api_key", None) elif isinstance(provider_entry, dict) and provider_entry.get("api_key"): raw_key = str(raw_entry.get("api_key") or "").strip() @@ -701,6 +712,9 @@ def _apply_aux_assignment_sync(cfg: dict, provider: str, model: str, task: str, # endpoint must carry its own base_url/api_key (the auxiliary resolver reads # auxiliary.<task>.base_url/api_key), or it silently rebinds to model.base_url and # breaks once the main slot switches away. + # The auxiliary resolver already reads auxiliary.<task>.base_url/api_key + # (_resolve_task_provider_model), so persisting them here is what actually wires the endpoint + # in. See #65254. slot_cfg["base_url"] = base_url if api_key: slot_cfg["api_key"] = api_key diff --git a/hermes_cli/web_server_dashboard.py b/hermes_cli/web_server_dashboard.py index 6ededc9844..5898ec638b 100644 --- a/hermes_cli/web_server_dashboard.py +++ b/hermes_cli/web_server_dashboard.py @@ -119,6 +119,7 @@ def mount_spa(application: FastAPI): # path, a renderer whose spawn token no longer matched (e.g. after `hermes update`) # white-screened. Serve a token-only page at the exact root, but ONLY when the auth # gate is off: on a gated serve the token must never be readable without auth. + # See #94227, #95575. gated = bool(getattr(application.state, "auth_required", False)) if full_path == "" and not gated: return HTMLResponse( @@ -131,6 +132,14 @@ def mount_spa(application: FastAPI): return JSONResponse({"error": _HEADLESS_MSG}, status_code=404) return + # A missing WEB_DIST is deliberately NOT a mount-time terminal state (#82614): a long-lived `hermes + # dashboard --skip-build` process that survives a `git pull` (or starts before the first build) used to + # install a permanent no_frontend catch-all here and could never recover — every route answered 404 + # "Frontend not built" until the process was restarted, even after `npm run build` completed. The SPA + # routes below all cope with a missing dist per-request (`_serve_index` returns the same 404 JSON when + # index.html is unreadable; the asset mounts use check_dir=False and 404 on missing files), so mounting + # them unconditionally makes the dashboard recover the moment a build appears on disk — no restart + # needed. def _serve_index(prefix: str = ""): """index.html with the session token + base-path injected. @@ -427,6 +436,10 @@ def _safe_plugin_api_relpath(api_field: Any, *, dashboard_dir: Path) -> Optional ``/tmp/evil.py``) and ``../`` could climb out of it (GHSA-5qr3-c538-wm9j). Returns the original string when the resolved path stays under ``dashboard_dir``, else ``None`` so the plugin still loads its static JS/CSS but its backend ``api`` is rejected. + + The web server later imports this file as a Python module via ``importlib.util.spec_from_file_location`` + (arbitrary code execution by design — that's how plugins extend the backend). Pre-#29156 the field was + used as-is, which meant: """ if not isinstance(api_field, str) or not api_field.strip(): return None @@ -455,12 +468,29 @@ def _dashboard_plugin_search_dirs() -> List[tuple]: from hermes_constants import get_default_hermes_root bundled_root = get_bundled_plugins_dir() + # User dashboard plugins are a dashboard-owned asset (same category as theme YAML): resolve them from + # the process launch home so they don't vanish when a request is scoped to another profile via a + # context-local HERMES_HOME override (e.g. embedded /chat under --open-profile). #87197: when the + # process itself is profile-scoped (``--profile <name>`` sets ``HERMES_HOME=<root>/profiles/<name>``), + # the launch home is the profile directory, which has no ``plugins/`` — user plugins are installed in + # the hermes root (``~/.hermes/plugins``). Scan the default root as well (``get_default_hermes_root()`` + # unwraps ``<root>/profiles/<name>`` → ``<root>`` and returns a custom ``HERMES_HOME`` unchanged when it + # *is* the root), mirroring how ``hermes_cli.plugins`` resolves plugin install locations. The + # ``seen_names`` dedupe below keeps profile-local plugins (if any) authoritative over same-named root + # plugins. user_plugin_roots = [get_process_hermes_home() / "plugins"] root_plugins = get_default_hermes_root() / "plugins" if root_plugins.resolve(strict=False) != user_plugin_roots[0].resolve(strict=False): user_plugin_roots.append(root_plugins) search_dirs = [(d, "user") for d in user_plugin_roots] search_dirs += [(bundled_root / "memory", "bundled"), (bundled_root, "bundled")] + # GHSA-5qr3-c538-wm9j (#29156): the previous ``os.environ.get(...)`` check treated *any* non-empty + # string as truthy, so ``=0``, ``=false``, and ``=no`` — all of which the agent loader and operators + # correctly read as "disabled" — silently *enabled* the untrusted project source in the web server. + # Combined with the absolute-path RCE primitive on the manifest's ``api`` field (now patched below), + # this turned the opt-in into a sticky always-on switch. Use the shared truthy semantics (``1`` / + # ``true`` / ``yes`` / ``on``) so the gate matches ``hermes_cli/plugins.py`` and the documented user + # contract. if env_var_enabled("HERMES_ENABLE_PROJECT_PLUGINS"): search_dirs.append((Path.cwd() / ".hermes" / "plugins", "project")) return search_dirs @@ -738,6 +768,15 @@ def _mount_plugin_api_routes(): Each plugin's ``api`` file must expose a ``router`` (FastAPI APIRouter), mounted under ``/api/plugins/<name>/``. See ``_plugin_api_mount_skip_reason`` for the trust gates. + + Backend import is restricted to ``bundled`` and ``user`` sources. Project plugins + (``./.hermes/plugins/``) ship with the CWD and are therefore attacker-controlled in any threat model + where the user opens a malicious repo; they can extend the dashboard UI via static JS/CSS but their + Python ``api`` file is never auto-imported by the web server. See GHSA-5qr3-c538-wm9j (#29156). + Additionally, user plugins must be explicitly enabled via the ``plugins.enabled`` allow-list in + config.yaml before their backend code is imported. Without this gate, an installed-but-not-enabled + plugin's Python code would execute at dashboard startup — a code execution vector that bypasses the + user's intent. (#46435, GHSA-mcfc-hp25-cjv7) """ from hermes_cli.web_server import _get_dashboard_plugins, app try: diff --git a/hermes_cli/web_server_gateway.py b/hermes_cli/web_server_gateway.py index 713814428d..09203dccbf 100644 --- a/hermes_cli/web_server_gateway.py +++ b/hermes_cli/web_server_gateway.py @@ -309,6 +309,10 @@ def _dashboard_spawn_executable() -> str: case this fixes), and pyvenv.cfg discovery keys off argv0's unresolved location. On Windows the console python plus ``windows_detach_flags()`` keeps the action invisible without pythonw.exe (which makes every console descendant flash its own conhost). + + See #90026. + Falls back to ``sys.executable`` when no venv interpreter exists next to the install (in-process dev + runs, exotic layouts). See #54220, #56747. """ from hermes_cli.web_server import PROJECT_ROOT exe = Path(sys.executable) @@ -340,6 +344,7 @@ def _spawn_hermes_action( # The dashboard runs inside the gateway process, so os.environ carries _HERMES_GATEWAY=1; # inheriting it trips the child's in-process restart-loop guard (exit 1). Drop it, like # the gateway's own restart watcher does. + # The gateway's own restart watcher already drops it (gateway/run.py); mirror that here (#52470). action_env = {**os.environ, "HERMES_NONINTERACTIVE": "1"} action_env.pop("_HERMES_GATEWAY", None) detach = {"creationflags": windows_detach_flags()} if sys.platform == "win32" else {"start_new_session": True} diff --git a/hermes_cli/web_server_memory.py b/hermes_cli/web_server_memory.py index 6705e24d2a..0bf315b6af 100644 --- a/hermes_cli/web_server_memory.py +++ b/hermes_cli/web_server_memory.py @@ -129,6 +129,9 @@ def _run_setup_command( capture_output=True, text=True, # Lossy UTF-8 decode — setup tools emit UTF-8; a locale-mismatched byte must never raise. + # Force UTF-8 with lossy decoding so child output containing bytes that are invalid in the system + # locale (e.g. GBK on Chinese Windows) can't raise UnicodeDecodeError inside the drain threads and + # crash the gateway. See #53137. encoding="utf-8", errors="replace", timeout=timeout, diff --git a/hermes_cli/web_server_profiles.py b/hermes_cli/web_server_profiles.py index 7d3989edab..6b1cf02aab 100644 --- a/hermes_cli/web_server_profiles.py +++ b/hermes_cli/web_server_profiles.py @@ -211,6 +211,10 @@ def _profile_scope(profile: Optional[str]): ``get_hermes_home()`` so writes land in the live home even when the import-time binding is stale (test isolation, late HERMES_HOME override). Yields the profile dir for a named profile, None for the current one. + + ``tools.skills_sync`` (reset/diff/list-modified/opt-in/opt-out/ repair-official) needs NO retargeting: + since #65828 its directory lookups resolve at call time through the same contextvar override set in step + 1. """ from hermes_constants import get_hermes_home from tools import skills_tool as _skills_tool @@ -247,6 +251,12 @@ def _config_profile_scope(profile: Optional[str]): # Terminal backend picker rows — GUI counterpart of terminal.backend. Keep in sync with # tools/terminal_tool.py::_create_environment and the terminal.backend enum. +# --------------------------------------------------------------------------- Terminal execution backend +# picker — the GUI counterpart of terminal.backend in config.yaml. Each row carries a fast, defensive health +# probe (Docker daemon reachable, SSH host configured, Modal/Daytona credentials present) so the +# Capabilities panel can render Ready / Needs setup guidance instead of a bare enum (issues #57738 / +# #63783). Probes must never raise — a probe failure renders as a status, not a 500. +# --------------------------------------------------------------------------- _TERMINAL_BACKENDS: List[Dict[str, str]] = [ dict(zip(("name", "label", "description"), row)) for row in ( ("local", "Local", "Run commands directly on this machine. No isolation."), @@ -291,7 +301,10 @@ def _token_volume(row: Dict[str, Any]) -> Any: def _aux_usage_rows(db, cutoff: float) -> List[Dict[str, Any]]: """Per-(model, task) auxiliary usage within the window: the task-dimension rows (task != '') record_auxiliary_usage writes into session_model_usage. [] when the - table predates the task column (older DB opened read-only by newer code).""" + table predates the task column (older DB opened read-only by newer code). + + See #23270. + """ try: cur = db._conn.execute(""" SELECT u.model, diff --git a/hermes_cli/web_server_sessions.py b/hermes_cli/web_server_sessions.py index d34e3beb86..38e0e5cdf2 100644 --- a/hermes_cli/web_server_sessions.py +++ b/hermes_cli/web_server_sessions.py @@ -117,6 +117,9 @@ def _open_session_db_at_path(db_path: Path, *, read_only: bool): from hermes_state import SessionDB, is_malformed_schema_error + # Read-only file/sidecar preflight (port of kilocode#12508): repair-or-refuse BEFORE the first + # connection so users get an actionable message instead of an opaque "attempt to write a readonly + # database" from deep inside _init_schema. if not read_only: return SessionDB(db_path=db_path, read_only=False) diff --git a/hermes_cli/worktree_ops.py b/hermes_cli/worktree_ops.py index 9ca4711cc7..e9e6965033 100644 --- a/hermes_cli/worktree_ops.py +++ b/hermes_cli/worktree_ops.py @@ -318,6 +318,8 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True, *sync_base* branches from the fetched remote tip (``_resolve_worktree_base``), else local HEAD. *name* replaces the random ``hermes-<id>``; named trees lack the ``hermes-`` prefix so the pruner ages them on its slower schedule. + + Set ``worktree_sync: false`` in config to branch from local ``HEAD`` (the pre-#10760-followup behavior). """ repo_root = repo_root or _git_repo_root() if not repo_root: @@ -339,6 +341,9 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True, _ensure_worktrees_gitignored(repo_root) + # Resolve the base ref. By default branch from the freshly-fetched remote tip so the worktree starts + # current with the project, not from the (possibly stale) local HEAD of the standalone clone (#10760 + # follow-up). base_ref, base_label = (_resolve_worktree_base(repo_root) if sync_base else ("HEAD", "HEAD (local — worktree_sync disabled)")) diff --git a/hermes_constants.py b/hermes_constants.py index 067c1c695d..4dc3cd33e7 100644 --- a/hermes_constants.py +++ b/hermes_constants.py @@ -271,6 +271,11 @@ def get_hermes_dir(new_subpath: str, old_name: str, *, home: Path | None = None) """Resolve a Hermes subdirectory, honouring a populated legacy ``<old_name>/`` (no migration). An empty legacy dir does NOT count (install scaffolds, manual mkdir) so it cannot shadow the new path. + + A bare empty ``<old_name>/`` directory does **not** count as "the legacy install is in use" — install + scaffolds, manual ``mkdir`` work, and cleared-then-abandoned locations all create empty stubs that would + otherwise silently shadow real data populated at ``<new_subpath>/``. See #27602 for the pairing-store + regression where a dormant empty ``pairing/`` orphaned approved-user data in ``platforms/pairing/``. """ home = home or get_hermes_home() old_path = home / old_name @@ -340,6 +345,9 @@ _HERMES_NODE_TARGET_MAJOR = int(os.environ.get("HERMES_NODE_TARGET_MAJOR", "22") _managed_node_heal_attempted = False _NODE_BOOTSTRAP_SCRIPT = Path(__file__).resolve().parent / "scripts" / "lib" / "node-bootstrap.sh" +# Install tree root (this file lives at <install_root>/hermes_constants.py). Used by secure_parent_dir() to +# skip chmod on the install dir — chmodding it 0700 breaks hermes-user traversal in Docker (UID 10000). See +# #25821, #93050. _INSTALL_ROOT = Path(__file__).resolve().parent @@ -383,6 +391,8 @@ def managed_node_tree_in_use(home: Path | None = None) -> bool: Windows locks running executables against delete/overwrite, so the updater must not rewrite ``%HERMES_HOME%\\node`` while the desktop app holds it (``[WinError 5]`` on ``npm.cmd``). + + Always ``False`` on POSIX, which has no equivalent lock semantics. See #80926. """ if sys.platform != "win32": return False @@ -522,6 +532,14 @@ def _heal_managed_node_windows(home: Path | None = None) -> bool | None: tree is in use and the heal is deferred — callers must not record the once-per-process attempt for ``None``. Staging-first (extract to ``node.new-*``, rename live aside, rename staged in) so an interrupted heal cannot gut the install; a refused rename *is* the in-use signal. + + The replacement is staging-first: the new tree is fully downloaded and extracted to a sibling + ``node.new-*`` directory, then the live tree is renamed aside (``node.old-*``) and the staged tree + renamed into place. The live tree is never deleted before its replacement is ready, so an interrupted + heal cannot gut the running installation. Windows allows renaming a tree whose executables are running + (images are mapped with ``FILE_SHARE_DELETE`` — the same mechanism as the hermes.exe quarantine); when + the OS refuses the rename, that refusal *is* the in-use signal and the heal defers instead of forcing + the write and crashing with ``PermissionError: [WinError 5]`` on ``npm.cmd`` (#80926). """ import time @@ -587,6 +605,9 @@ def heal_hermes_managed_node() -> bool: """Redownload Hermes-managed Node when the tree exists but is broken; at most once per process. A Windows in-use deferral does NOT record the attempt so a later call can heal once free. + + POSIX installs shell out to ``heal_managed_node`` in ``scripts/lib/node-bootstrap.sh``; Windows + downloads the portable zip directly (same source as ``install.ps1``). See #80926. """ global _managed_node_heal_attempted if _managed_node_heal_attempted or not hermes_managed_node_tree_present(): @@ -675,6 +696,12 @@ def agent_browser_runnable(path: str | None) -> bool: """True when *path* is an agent-browser CLI that runs (``--version`` exits 0) or the npx fallback. Dead/wrong-arch/hung binaries are rejected so callers try the next candidate. + + A bare presence check (``shutil.which`` / ``Path.exists``) is not enough: agent-browser's npm + ``postinstall`` re-points a *global* install symlink (e.g. ``/opt/homebrew/bin/agent-browser``) at our + local ``node_modules/agent-browser/bin/...`` binary, which then disappears on the next ``hermes update`` + — leaving a **dangling symlink** that ``which`` still reports but exec fails on with exit 127 (issue + #48521). Callers that trust such a path silently break every browser tool. """ if not path: return False @@ -725,6 +752,7 @@ def secure_parent_dir(path: Path) -> None: return # Refuse the install tree: chmod 0700 breaks hermes-user traversal in Docker (UID 10000). # A credential file here means HERMES_HOME misresolved; surface it (caused production lockouts). + # See #25821, #93050. if parent == _INSTALL_ROOT or _INSTALL_ROOT in parent.parents: import logging @@ -991,7 +1019,10 @@ _container_detected: bool | None = None def is_container() -> bool: - """True inside a container (Docker/Podman/LXC/Kubernetes markers); cached per process.""" + """True inside a container (Docker/Podman/LXC/Kubernetes markers); cached per process. + + See: NousResearch/hermes-agent#47111 + """ global _container_detected if _container_detected is None: _container_detected = _detect_container() @@ -1071,13 +1102,26 @@ def venv_bin_dir(venv_dir, *, windows: bool | None = None) -> Path: """Venv executable dir (``Scripts``/``bin``); *windows* lets tests exercise Windows paths on Linux. Returned unconditionally — callers differ on whether a missing venv is an error. + + Canonical helper for venv layout. This was open-coded in seven places across four ``hermes_cli`` modules + using three different Windows predicates (``platform.system()``, ``is_windows()``, ``_is_windows()``); + each new call site had to re-derive it, and #76091 shipped an eighth copy because the correct behaviour + lived 2400 lines away in another function. A few sites outside ``hermes_cli`` + (``tools/code_execution_tool.py``, ``agent/lsp/install.py``, ``agent/lsp/servers.py``) still hand-roll + it — convert them as they are touched. """ windows = sys.platform == "win32" if windows is None else windows return Path(venv_dir) / ("Scripts" if windows else "bin") def project_venv_dir(project_root) -> Path | None: - """The project's ``venv`` or ``.venv`` dir when one exists (``uv venv`` defaults to ``.venv``).""" + """The project's ``venv`` or ``.venv`` dir when one exists (``uv venv`` defaults to ``.venv``). + + ``uv venv`` defaults to ``.venv`` while our installers create ``venv``, so both layouts are in the wild. + Call sites that only knew about ``venv`` silently no-oped on a ``.venv`` install — that is how the + Windows shim-lock preflight skipped itself entirely (#79542). ``venv`` wins when both exist, matching + what the installers write. + """ root = Path(project_root) return next((root / n for n in ("venv", ".venv") if (root / n).is_dir()), None) diff --git a/hermes_startup_watchdog.py b/hermes_startup_watchdog.py index 57ebd95621..3bfb34ebd5 100644 --- a/hermes_startup_watchdog.py +++ b/hermes_startup_watchdog.py @@ -86,6 +86,7 @@ _FIRE_EXIT_BOUND_S = 10.0 # Handle states. Transitions are guarded by the handle's state lock so a # disarm and a fire can never both "win": armed -> disarmed or armed -> firing. +# See #89750. _ARMED = "armed" _DISARMED = "disarmed" _FIRING = "firing" diff --git a/hermes_state.py b/hermes_state.py index 95453ba582..b9215f561b 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -26,6 +26,11 @@ from pathlib import Path from agent.message_sanitization import _sanitize_surrogates # Known-durable message marker (run_agent keeps a copy: circular import; a test pins them in sync). from agent.context_compressor import ( # noqa: F401 (re-exported; tests import it from here) + # Intrinsic persistence marker stamped on message dicts that are known-durable (#92231). One shared + # constant with agent.context_compressor (this module already imports agent.* at module level, and + # context_compressor is a transitive dependency via hermes_state_common). run_agent keeps its own + # predating copy — hermes_state cannot import run_agent (circular) — guarded by + # test_marker_constant_in_sync. _DB_PERSISTED_MARKER as _DB_PERSISTED_MARKER_KEY, ) from hermes_constants import get_hermes_home @@ -216,7 +221,11 @@ _STATE_DB_GUARD_EXTRA_DENY_ROOTS: Tuple[Path, ...] = () def _ensure_test_isolation(db_path: Path) -> None: """Raise before any connection/mkdir/pragma/byte probe when a pytest-context process - (env OR ancestry) resolves a production DB.""" + (env OR ancestry) resolves a production DB. + + Env alone is not enough: a child spawned with a rebuilt environment loses ``PYTEST_*`` and + ``HERMES_HOME`` together, which is precisely the state in which it writes to production (#82770). + """ if _STATE_DB_GUARD_BYPASS or os.environ.get(_STATE_DB_GUARD_BYPASS_ENV) or not _in_test_context(): return try: @@ -347,6 +356,14 @@ def _foreign_state_db_holders(db_path: Path) -> List[Tuple[int, str]]: return _state_holders.foreign_state_db_holders(db_path) +# ── Process-wide shared SessionDB registry (#90837) ── The registry itself lives in +# hermes_state_registry.py — a bounded module owning acquisition, generation identity, refcounting, +# retirement, and teardown. These re-exports keep the historical import path (``from hermes_state import +# get_shared_session_db``) working for every call site and test that imports from here. Routing rules (see +# hermes_state_registry for the full lifecycle): - Long-lived in-process callers (gateway, tui_gateway, +# cron, in-process tools) share ONE writer connection per resolved path via get_shared_session_db(). - CLI +# one-shots, recovery flows, and read-only cross-profile opens keep using SessionDB() directly with their +# own close(). from hermes_state_registry import ( # noqa: F401 (re-export) close_shared_session_dbs, get_shared_session_db, release_or_close, ) @@ -362,6 +379,7 @@ class SessionDB( # Only these state-owned producers join automatic stale-open reconciliation; messaging/UI # sources have their own lifecycle owners; unknown sources fail closed. + # See #60609. _AUTO_PRUNE_STALE_OPEN_SOURCES: Tuple[str, ...] = ( "cli", "cron", "kanban", "acp", "api_server", "subagent", "tool", ) @@ -376,6 +394,15 @@ class SessionDB( _WRITE_PATIENCE_S, _TRANSCRIPT_WRITE_PATIENCE_S, _ACTIVITY_WRITE_PATIENCE_S = 20.0, 60.0, 0.5 # A live compression lock gets a short wait (compression publishes in seconds), but the lease # is a correctness boundary: a writer still locked out afterwards is refused. + # Observation-only activity heartbeat/label writes (#76354 review S1): these run on (or adjacent to) the + # response-critical path and must never wait out the full routine patience under contention. Sub-second + # budget; a skipped write is retried naturally at the next heartbeat window. + # A live compression lock gets its own, much shorter budget than the write lock. Compression publishes + # in a couple of seconds, so a brief wait saves the overwhelming majority of concurrent turns (#75083). + # It deliberately stays short: the lease is a correctness boundary, not just a busy signal (see + # test_compression_lease_blocks_non_owner_but_allows_owner_flush), so a writer that is still locked out + # after this budget must still be refused rather than allowed to land a stale turn in a session whose + # compression is genuinely long-running or wedged. _COMPRESSION_BUSY_WAIT_S = 5.0 _WRITE_RETRY_MIN_S, _WRITE_RETRY_MAX_S = 0.020, 0.150 # fast jitter for the first _SLOW_AFTER_S _WRITE_RETRY_SLOW_AFTER_S = 2.0 @@ -446,6 +473,12 @@ class SessionDB( self._read_pool: "queue.LifoQueue[sqlite3.Connection]" = queue.LifoQueue(maxsize=_READ_POOL_MAX) # Permits bound PEAK descriptors (the pool bounds only the idle set), shared per # DATABASE PATH; acquired non-blocking so a permitless reader degrades to the writer lock. + # One permit per live read connection, held from before the open in _get_read_conn() until after the + # close in _close_read_conn(). See _READ_POOL_MAX. Acquired non-blocking on purpose: a reader that + # cannot get a permit must degrade to the writer lock, not queue here — blocking would convert fd + # exhaustion into a stall, which is the same outage with a different stack trace. Permits are shared + # per DATABASE PATH, not per instance: the descriptors they ration belong to the file, and one + # process holds several SessionDB objects on the same state.db (#98573). See _PathReadBudget. self._read_budget = _read_budget_for(self.db_path) self._read_budget.register(self) self._read_permits = self._read_budget.permits @@ -477,6 +510,8 @@ class SessionDB( self._token_writer_stop = self._token_writer_busy = False self._token_atexit_hook: Optional[Callable[[], None]] = None # Opened via get_shared_session_db(): close() releases a refcount instead. + # Set True when this instance is opened via get_shared_session_db(). Makes close() a no-op so the + # registry (not individual callers) controls the connection lifecycle (#90837). self._shared_registry_owned = False initialization_complete = False try: @@ -627,6 +662,12 @@ class SessionDB( _init_schema's DDL runs on a 1s-timeout connection, so a sibling's VACUUM or checkpoint used to fail the ENTIRE open and callers disabled persistence for the whole run. Non-lock errors propagate immediately.""" + # Lock contention during open: _init_schema's DDL/reconcile statements run on a 1s-timeout + # connection with no retry, so a sibling process holding the write lock (VACUUM, TRUNCATE checkpoint + # at close, a long FTS pass from an older still-running install) used to fail the ENTIRE open — + # callers then disable persistence for the whole run ("Failed to initialize SessionDB ... database + # is locked", #74478). The store is healthy; wait it out with the same jittered patience the write + # path uses. deadline = time.monotonic() + self._WRITE_PATIENCE_S while True: try: @@ -788,6 +829,10 @@ class SessionDB( compression_deadline: Optional[float] = None # set on the first compression-busy collision # One retry for SQLITE_IOERR raised by BEGIN IMMEDIATE itself (callback not run: nothing # replayed). Once fn has started, an IOERR leaves settlement unknown and must propagate. + # The callback has not run at that point, so there is no durable effect to replay and the retry is + # exactly-once safe (#99502's contract). Once the callback starts, an IOERR leaves the write's + # settlement unknown and must propagate — this helper owns non-idempotent transcript/counter + # mutations, not just idempotent UPSERTs. ioerr_begin_retried = False while True: self._raise_if_db_corrupt() @@ -817,6 +862,13 @@ class SessionDB( return result except SessionCompressionInProgressError: # Transient (see _COMPRESSION_BUSY_WAIT_S): a steer landing mid-compression must not abort. + # A live foreign compression lock is transient: the compressor publishes in a couple of + # seconds. Without any wait, a steer that lands mid-compression aborts the user's turn as + # session_persistence_failed and sends the operator hunting disk space that was never the + # problem (#75083). The budget is _COMPRESSION_BUSY_WAIT_S, not the write-lock patience: the + # lease is a correctness boundary, so a writer still locked out after a short wait must be + # refused rather than left to land a stale turn once a long-running or wedged compression + # finally lets go. if compression_deadline is None: compression_deadline = min(time.monotonic() + self._COMPRESSION_BUSY_WAIT_S, deadline) if self._sleep_before_write_retry( @@ -898,7 +950,10 @@ class SessionDB( def _ensure_db_file_generation(self) -> None: """Mint a once-per-file generation stamp (state_meta + application_id). First opener wins (INSERT - OR IGNORE); application_id is written only while 0 so racers converge. PASSIVE checkpoint only.""" + OR IGNORE); application_id is written only while 0 so racers converge. PASSIVE checkpoint only. + + See #45383. + """ if self.read_only or self._conn is None: return token = uuid.uuid4().hex @@ -984,6 +1039,9 @@ class SessionDB( """Stop writes (logging once) when the file was replaced or its WAL/SHM generation is gone: never run in-file repair on a new generation, never keep committing on a split WAL. Both flags are sticky.""" + # A reopen resolves the PATH again — if the file at that path is no longer the one this instance + # originally opened (out-of-band restore/cp/mv), reconnecting would write into the new generation + # through stale WAL/shm assumptions (#89332). Refuse instead. if self._db_replaced or self._db_file_was_replaced(): self._db_replaced = True logger.error(_STATE_DB_REPLACED_MSG) @@ -1077,7 +1135,11 @@ class SessionDB( def _try_wal_checkpoint(self) -> None: """Best-effort PASSIVE WAL checkpoint; never raises. PASSIVE never blocks writers; - TRUNCATE corrupted B-trees on 65K+ page databases under exclusive-lock I/O pressure.""" + TRUNCATE corrupted B-trees on 65K+ page databases under exclusive-lock I/O pressure. + + Previous TRUNCATE strategy caused B-tree corruption on large databases (65K+ pages) due to the + exclusive-lock I/O pressure from checkpointing thousands of frames at once (issue #45383). + """ if self._db_corrupt: return # quarantined: never checkpoint over a damaged image try: @@ -1089,7 +1151,16 @@ class SessionDB( logger.warning("WAL checkpoint (PASSIVE) failed: %s", exc) def __enter__(self) -> "SessionDB": - """``with SessionDB(path) as db:`` closes on exit; owners must release deterministically.""" + """``with SessionDB(path) as db:`` closes on exit; owners must release deterministically. + + Ownership of a SessionDB should be released explicitly. Historically an instance with a started + token writer pinned ITSELF (bound-method writer target plus a strong ``atexit`` drain hook), so + ``__del__`` never ran for exactly the instances that leaked descriptors (#88033). The writer now + retires after an idle window and the atexit hook holds only a weak reference, so abandoned handles + are eventually collectible — but "eventually, after the idle window and a GC cycle" is not a release + policy. Call sites owning a handle are still expected to close it deterministically (see the + ownership comments in ``run_agent.py`` and ``tui_gateway/methods_session.py``). + """ return self def __exit__(self, exc_type, exc, tb) -> bool: @@ -1099,7 +1170,16 @@ class SessionDB( def close(self): """Drain queued token deltas, then a PASSIVE checkpoint on writable handles (NOT TRUNCATE: a full WAL reset races the gateway's live writer, tearing - B-tree pages). A registry-shared instance RELEASES one refcount instead.""" + B-tree pages). A registry-shared instance RELEASES one refcount instead. + + Drains queued token deltas first (the background writer needs the connection). Read-only connections + never request a checkpoint. See #45383. + When this instance is shared (opened via ``get_shared_session_db``), ``close()`` RELEASES one + refcount instead of tearing down the connection: the registry owns the lifecycle and only closes on + the final release (#90837). This prevents one caller's close from tearing down the writer connection + that other callers in the same process are still using — while still letting legacy ``close()`` call + sites return their reference instead of leaking it. + """ if self._shared_registry_owned: from hermes_state_registry import release release(self) @@ -1125,6 +1205,11 @@ class SessionDB( ) elif not self.read_only: # PASSIVE, not TRUNCATE (see docstring) try: + # Every cron run_agent opens+closes a transient SessionDB, so a TRUNCATE here fires + # a full WAL reset many times/hour, racing the gateway's long-lived writer on large + # WAL databases and tearing hot B-tree pages -- the #45383 corruption this class's + # own periodic checkpoint was already made PASSIVE to avoid. TRUNCATE belongs only + # on a sole-opener/quiescent connection. self._conn.execute("PRAGMA wal_checkpoint(PASSIVE)") except Exception as exc: logger.debug("WAL checkpoint (PASSIVE) at close failed: %s", exc) @@ -1166,6 +1251,9 @@ class SessionDB( # Bot Mode's canonical chat is resolved by exact-title lookup: the title IS the identity, # so _set_session_title refuses renames of a hidden row holding it. + # Bot Mode's forever-chat registry: the session titled exactly this, on a bot's profile, IS the bot's + # canonical chat — resolved by exact-title lookup on every open (no session-id pointer exists). See + # #92473. CANONICAL_BOT_CHAT_TITLE = "Bot Chat" # ── Message storage constants (SessionMessagesMixin) ── diff --git a/hermes_state_common.py b/hermes_state_common.py index 6eb00fdf83..97f72bba68 100644 --- a/hermes_state_common.py +++ b/hermes_state_common.py @@ -126,17 +126,26 @@ _RESET_END_REASONS_SQL = ", ".join(f"'{reason}'" for reason in _RESET_END_REASON # by a fresh session.resume; startup_orphan_reap = dead-gateway sweep, same class as ws_orphan_reap but kept # distinct for forensics. _RECOVERABLE_END_REASONS = ("agent_close", "ws_orphan_reap", "superseded_by_resume", "startup_orphan_reap") +# Startup sweep of rows orphaned by a dead gateway process (#65194): the in-process ws-orphan grace timer +# died with the process, so the row was closed at the next boot instead. _RECOVERABLE_END_REASONS_SQL = ", ".join(f"'{reason}'" for reason in _RECOVERABLE_END_REASONS) # End reasons written by AUTOMATIC cleanup (shutdown, orphan reapers, idle/LRU eviction), not a deliberate # conversation boundary: "some runtime went away", so a writer that can prove liveness (e.g. a compression # rotation holding the lease) may clear it. Recoverable set plus the TUI gateway's automatic reasons. +# Superset of the recoverable set: those are already resumable accidents; the extra TUI reasons are the same +# accident class but were historically only known to tui_gateway's _AUTOMATIC_SESSION_END_REASONS. See +# #88197. _AUTOMATIC_END_REASONS = frozenset(_RECOVERABLE_END_REASONS) | { "tui_shutdown", "ws_disconnect", "idle_timeout", "lru_evict"} def is_automatic_end_reason(reason) -> bool: - """True when *reason* is an automatic-cleanup end stamp; compression-liveness sites must call this.""" + """True when *reason* is an automatic-cleanup end stamp; compression-liveness sites must call this. + + Single owner of the "accidental vs deliberate end" predicate — every compression-liveness site must call + this instead of re-implementing the reason taxonomy (#88197, never-patch-predicates). + """ return isinstance(reason, str) and reason in _AUTOMATIC_END_REASONS @@ -188,6 +197,10 @@ def _sql_session_last_active_by_id(session_id_expr: str) -> str: SCHEMA_VERSION = 30 # Auto-maintenance VACUUMs only above this freelist fraction; below it a rewrite costs more I/O than it returns. +# Auto-maintenance only VACUUMs when at least this fraction of the database file is reclaimable (``PRAGMA +# freelist_count / PRAGMA page_count``). Below it a full rewrite costs more I/O than it returns — pruning a +# handful of small sessions on a dense multi-GB state.db should never rewrite the whole file to reclaim a +# few MB (#54189). Composes with ``min_vacuum_interval_days``. AUTO_VACUUM_MIN_FREELIST_RATIO = 0.25 # FTS storage-layout version, tracked INDEPENDENTLY of SCHEMA_VERSION in the @@ -809,6 +822,28 @@ END; # only when provably dead, indeterminate liveness defers. `<db>.fts_rebuild.lock` is distinct from # `<db>.repair.lock` (offline schema surgery, minutes in VACUUM). Lives here: mixins cannot import hermes_state. +# ── Cross-process full-FTS-rebuild admission (single authority) ────────────── Several independent Hermes +# processes routinely share one state.db (gateway service, the Desktop app's `hermes serve` backend, +# interactive CLI sessions, the TUI slash worker). A full structural FTS rebuild — the FTS5 'rebuild' +# command or the drop/recreate script in `_recover_stale_fts` — must only ever run in ONE of them at a time: +# two concurrent rebuilds collide on write and have structurally corrupted state.db in production (PR +# #93200; the 2026-08-15 / 2026-08-23 incidents and issues #89293 / #90950). This is the single admission +# authority for every full structural rebuild entry point: `SessionSearchMixin.rebuild_fts()`, +# `SessionSchemaMixin._rebuild_fts_indexes()` (via `_init_schema`), and +# `SessionSchemaMixin._recover_stale_fts()`. The chunked deferred backfill (`fts_rebuild_step`) is +# deliberately NOT routed through it — it claims progress under `_execute_write`'s SQLite transaction +# authority and is intentionally multi-process. Semantics mirror `hermes_state._cross_process_repair_lock` +# (the schema- surgery authority): portable (msvcrt on Windows, flock elsewhere), bounded wait, and FAIL +# CLOSED — a caller that cannot acquire the lock must NOT rebuild. The kernel drops both lock types when the +# holder dies — UNLESS a forked child inherited the lock fd (flock rides the open file description, which +# fork() duplicates), in which case the orphaned descriptor holds the lock forever (issue #100108). +# `_acquire_db_flock` therefore records the holder's pid + start time under the lock and, when the recorded +# holder is provably dead, breaks the orphaned lock by unlinking and retaking it on a fresh inode; +# indeterminate liveness still defers. It lives here (not hermes_state) because the search/schema mixins +# cannot import hermes_state (cycle). The lock file is `<db>.fts_rebuild.lock`, distinct from +# `<db>.repair.lock`: schema surgery runs on an EXCLUSIVE offline connection and can legitimately take +# minutes in VACUUM, while runtime rebuilds run on live connections. The timeout is sized for a full +# 'rebuild' of both indexes on a large DB. logger = logging.getLogger("hermes_state") _FTS_REBUILD_LOCK_TIMEOUT_SECONDS = 120.0 @@ -861,7 +896,11 @@ def _rewrite_lock_file(handle, payload: bytes) -> None: def _write_lock_holder_record(handle) -> None: """Record this process as holder (best effort) so timed-out contenders can tell an orphaned-fd holder - from a live wedged one.""" + from a live wedged one. + + Written under the flock so contenders that time out can tell an orphaned-fd holder (recorded process + dead, flock inherited by a forked child — issue #100108) from a live wedged holder. + """ record = {"pid": os.getpid(), "start_ticks": _proc_start_ticks(os.getpid()), "acquired_at": time.time()} _rewrite_lock_file(handle, json.dumps(record, sort_keys=True).encode("utf-8")) @@ -900,7 +939,12 @@ def _acquire_db_flock(lock_path, handle, timeout_seconds, poll_seconds, descript without the held-by-another-process warning). ``flock`` rides the open file DESCRIPTION, which ``fork()`` duplicates, so a holder that forks then dies leaves the lock held forever; when the acquirer is provably dead the file is unlinked and retaken on a fresh inode (the orphan's flock excludes nobody). Every - acquire verifies its inode still names *lock_path*, so a racer on a dead inode retries.""" + acquire verifies its inode still names *lock_path*, so a racer on a dead inode retries. + + A holder that forks (multiprocessing worker, daemonized helper) and then dies leaves the flock held by a + child that will never release it — the kernel's holder-death release never triggers, and every contender + defers forever. Indeterminate liveness always defers (fail closed). See #100108. + """ import fcntl deadline = time.monotonic() + timeout_seconds broke_lock = False diff --git a/hermes_state_compression.py b/hermes_state_compression.py index 6e82ace9fd..dc696ca19e 100644 --- a/hermes_state_compression.py +++ b/hermes_state_compression.py @@ -137,6 +137,16 @@ class SessionCompressionMixin: if deleted.rowcount != 1: return False updated = conn.execute( + # A parent stamped ended by AUTOMATIC cleanup (tui_shutdown, ws_disconnect, orphan reap, + # idle/LRU evict) while a live agent is publishing its rotation is stale by construction — + # this writer holds the compression lease and is actively continuing the conversation the + # stamp claims is over. Left in place it wedges rotation forever: every attempt aborts here, + # nothing clears the stamp, and each attempt's pre-publish flush re-grows the parent until + # the provider rejects the request (#88197: 303 unique messages → 2,611 rows → HTTP 400). + # Clear it in this same transaction and proceed; the closure UPDATE below re-stamps the + # parent with its true boundary (end_reason='compression'). Deliberate boundaries + # (compression, session_reset, explicit close) still fail closed — those mean another path + # owns lineage. "UPDATE sessions SET ended_at = NULL, end_reason = NULL " "WHERE id = ? AND ended_at IS NOT NULL AND end_reason = 'compression'", (session_id,)) @@ -183,7 +193,11 @@ class SessionCompressionMixin: parent just before publishing and those rows are already in the handoff, so only ``(watermark, watermark_ceiling]`` is foreign tail (``None`` = unbounded). *require_lease_refresh* + *compression_lock_holder* refreshes the lease on the same ``conn`` before the expiry check (no - TOCTOU window), so a refresher that died on transient DB errors gets one last chance.""" + TOCTOU window), so a refresher that died on transient DB errors gets one last chance. + + See #75316. + ``None`` = unbounded (no internal flush happened). See #47202. + """ from hermes_state import CompressionSessionBusyError def _do(conn): if require_lease_refresh and compression_lock_holder: @@ -262,6 +276,8 @@ class SessionCompressionMixin: return self._write_sql_logged( "record_compression_failure_cooldown", session_id, + # Merge-max with any longer live deadline so a later shorter write cannot reopen the thrash + # window (#96775). The error column always takes the latest diagnostic. "UPDATE sessions SET compression_failure_cooldown_until = CASE " "WHEN compression_failure_cooldown_until IS NOT NULL AND compression_failure_cooldown_until > ? " "THEN compression_failure_cooldown_until ELSE ? END, compression_failure_error = ? WHERE id = ?", @@ -347,7 +363,12 @@ class SessionCompressionMixin: def get_compression_recovery_deadline(self, session_id: str) -> float: """Persisted anti-thrash recovery deadline (epoch; ``0.0`` = not armed). Durable - because the gateway rebuilds the compressor every turn / cache eviction.""" + because the gateway rebuilds the compressor every turn / cache eviction. + + The deadline is the durable half of the 14694 recovery clock: the gateway rebuilds the compressor on + every turn / cache eviction, so a process-local deadline restarted the wait on each rebuild and a + tripped session never earned its probe (#100185). + """ return self._read_session_number("compression_recovery_deadline", session_id, float, 0.0) def set_compression_recovery_deadline(self, session_id: str, deadline: float) -> None: @@ -545,7 +566,10 @@ class SessionCompressionMixin: def finalize_orphaned_compression_sessions(self) -> int: """Mark orphaned compression continuations (parent ended by compression; child has messages, no end_reason/ended_at, api_call_count=0, older than 7 days) as - ``orphaned_compression``. Non-destructive.""" + ``orphaned_compression``. Non-destructive. + + Fix for #20001. + """ cutoff = time.time() - 604800 # 7 days return self._write_rowcount( """ diff --git a/hermes_state_dbfile.py b/hermes_state_dbfile.py index b0f208959f..b2787d9872 100644 --- a/hermes_state_dbfile.py +++ b/hermes_state_dbfile.py @@ -171,7 +171,10 @@ def is_zeroed_state_db(path: Path, *, probe_bytes: int = 100, force: bool = Fals lock, even a running VACUUM's EXCLUSIVE); ``read_header_bytes_preopen`` refuses (-> False) once a connection is live. Pass ``force=True`` only for offline files (quarantined copies, snapshots). Prefers ``hermes_cli.backup.is_zeroed_sqlite_file``; this copy keeps SessionDB - openable without the CLI package in constrained embed paths.""" + openable without the CLI package in constrained embed paths. + + See #97568. + """ with contextlib.suppress(Exception): from hermes_cli.backup import is_zeroed_sqlite_file return is_zeroed_sqlite_file(path, probe_bytes=probe_bytes, force=force) diff --git a/hermes_state_errors.py b/hermes_state_errors.py index af735220da..17dd4b53fb 100644 --- a/hermes_state_errors.py +++ b/hermes_state_errors.py @@ -82,6 +82,8 @@ PERSISTENCE_ERROR_CAUSES = ( # "Database FILE structurally damaged" substrings. "database disk image is # malformed" contains "disk", so this check MUST run before the disk bucket in # classify_persistence_error or B-tree corruption reads as "free some disk space". +# Kept as plain substrings so sqlite3.DatabaseError, wrapped RPC strings, and logged message text all match +# the same helper. See #77386. _DB_CORRUPTION_MARKERS = ( "malformed", "file is not a database", "not a database", "database corruption", ) @@ -155,7 +157,15 @@ class StateDbCorruptError(sqlite3.DatabaseError): checkpointed 15 pages under wrong page numbers and turned a readable file into "file is not a database"; SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE on 3.12+ also stops SQLite's own). Subclasses sqlite3.DatabaseError so every degrade - path keeps working. Recovery boundary: restart on a repaired/restored file.""" + path keeps working. Recovery boundary: restart on a repaired/restored file. + + Stopping the writes is what prevents that; skipping the explicit checkpoint is the second line of + defence. SQLite still runs its own last-connection checkpoint inside ``close()`` (and deletes the + ``-wal`` sidecar) unless ``SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE`` is set — Python exposes it via + ``Connection.setconfig()`` on 3.12+, so quarantine disables the close-time checkpoint there and the WAL + survives on disk for forensics; on 3.11 the internal checkpoint is unavoidable (post-quarantine it can + only carry pre-corruption committed frames, since no further writes are accepted). See #90837. + """ _STATE_DB_CORRUPT_MSG = ( diff --git a/hermes_state_fts.py b/hermes_state_fts.py index 40b596d396..56b8ff60d7 100644 --- a/hermes_state_fts.py +++ b/hermes_state_fts.py @@ -26,6 +26,13 @@ logger = logging.getLogger("hermes_state") # while the index is complete-or-marker-gated. A stale index must keep its # triggers DROPPED — an external-content 'delete' for a rowid the index never # held is the canonical FTS5 corruption hazard. +# The trigram tokenizer needs >=3 chars per query term, so 1-2 char CJK terms (ubiquitous in Korean/Chinese: +# 일본, 구글, 项目, ...) fall through to a LIKE full-table scan — measured 3-6s CPU per query on multi-GB installs +# and the dominant base cost of session_search on CJK workloads. ``cjk_unicode61`` (native/fts5_cjk/, a +# ~250-line loadable FTS5 tokenizer with no dependencies) wraps unicode61: maximal CJK runs are re-emitted +# as overlapping character bigrams (Lucene CJKAnalyzer semantics), everything else passes through unchanged. +# FTS5 phrase semantics turn a query term's consecutive bigrams into exact substring matching down to 2 +# chars at index speed. Contributed by Soju06 (PR #65544). FTS_CJK_TABLE_SQL = """ CREATE VIEW IF NOT EXISTS messages_fts_cjk_src AS SELECT id, role, content, tool_name, tool_calls @@ -134,7 +141,14 @@ class SessionFtsSetupMixin: @staticmethod def _db_has_legacy_inline_fts(cursor: sqlite3.Cursor) -> bool: """messages_fts exists in ANY pre-v23 shape: every legacy shape lacks - tool_name, so "stored CREATE lacks tool_name" catches them all. False when absent.""" + tool_name, so "stored CREATE lacks tool_name" catches them all. False when absent. + + v23's messages_fts is external-content over THREE real columns (content, tool_name, tool_calls). + Every pre-v23 shape lacks the tool_name/tool_calls columns — whether the old inline single-column + form (v11..v22) or the even older external-content single-column form (v10-era, pre-#16751). We + therefore detect "needs optimize" as "the stored CREATE lacks the tool_name column", which is the + precise v23 marker and correctly catches BOTH legacy variants. + """ row = cursor.execute( "SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'messages_fts'" ).fetchone() diff --git a/hermes_state_gateway.py b/hermes_state_gateway.py index 2fce82e78c..1b381063a6 100644 --- a/hermes_state_gateway.py +++ b/hermes_state_gateway.py @@ -205,7 +205,10 @@ class SessionGatewayMixin: row. Self-healing: a missing target row (deferred ``create_session`` write, or crash between routing publication and row creation) is INSERTed with full identity rather than no-opped, so a gateway row is never first-created by the identity-less lazy writer (``update_token_counts``) and left - unroutable forever.""" + unroutable forever. + + See #9006. + """ if not session_id or not session_key: return identity = (session_key, source, user_id, chat_id, chat_type, thread_id, display_name, origin_json) @@ -283,7 +286,11 @@ class SessionGatewayMixin: """Keyed, still-open rows with no evidence of a single turn (no messages, tokens, tool/API calls, activity, or title): leaked fixtures or chats routed but never answered. Safe to drop — the gateway mints a fresh session on the next message. Needs its own selector because ``bulk prune``/``archive`` - are pinned to ``ended_at IS NOT NULL``. ``pinned``/``archived`` = explicit keep intent.""" + are pinned to ``ended_at IS NOT NULL``. ``pinned``/``archived`` = explicit keep intent. + + That is exactly the shape of a leaked test fixture (#82770) — and also of a chat that was routed but + never answered. + """ cutoff = time.time() - (float(older_than_days) * 86400.0) rows = self._read_all( """ @@ -383,7 +390,15 @@ class SessionGatewayMixin: exact-key fallback requires the complete peer tuple (never cross chats/threads/users) plus a profile fence: a Telegram DM's tuple is identical for every bot, so a sibling profile's legacy row would otherwise be adopted. Ours = profile_name is the owner or NULL; stores outside the profile tree - derive no owner and stay unfenced.""" + derive no owner and stay unfenced. + + See #60609. + Reset boundaries fence recovery (#68539): an intentional boundary such as ``session_reset`` (or any + explicit non-recoverable end_reason) must block fallback to an *older* row for the same peer. Each + candidate is therefore rejected when a boundary row for the peer ended *after* the candidate's last + activity — if the conversation's most recent event is an intentional reset, recovery returns nothing + rather than reaching behind it. + """ if not session_key: return None with self._read_ctx() as conn: @@ -392,6 +407,11 @@ class SessionGatewayMixin: return self._session_row_dict(row) if chat_id is None or chat_type is None: return None + # Profile fence (#74285): a Telegram DM's peer tuple is identical for every bot (chat_id == + # user_id, no thread), so a sibling profile's row written into this store before the per-profile + # partition (legacy data) would otherwise be adopted here. Every profile-tree store has one + # owner; a row is ours when its profile_name is the owner or NULL (legacy rows this store + # minted). Stores outside the tree derive no owner and keep the historical unfenced behavior. owner = self._own_profile_name() row = conn.execute( _PEER_BY_TUPLE_SQL, (source, user_id, chat_id, chat_type, thread_id, owner, owner, owner) @@ -466,6 +486,13 @@ class SessionGatewayMixin: if (donor is None or orphan is None or not donor["session_key"] or orphan["session_key"] or (donor["source"] or "") != (orphan["source"] or "")): return False + # Belt-and-suspenders for gateway routing metadata (#59527): the gateway re-records the peer on + # the child after rotation (d5b4879d4), but a hard crash between child creation and that write + # leaves the child row without origin columns, so ``find_latest_gateway_session_for_peer`` can't + # recover the mapping on restart. Inherit them from the parent at creation time — but ONLY for + # compression forks (parent already ended with end_reason='compression'). Delegate/subagent + # children are spawned while the parent is still live and must NOT inherit routing keys, or peer + # recovery could repoint gateway traffic into a subagent's session. conn.execute( """UPDATE sessions SET session_key = ?, diff --git a/hermes_state_guard.py b/hermes_state_guard.py index 2f4f528992..ec380a580b 100644 --- a/hermes_state_guard.py +++ b/hermes_state_guard.py @@ -41,6 +41,12 @@ def _real_platform_state_root() -> Optional[Path]: #: Exported by the hermetic conftest alongside the HERMES_HOME redirect. Unlike #: PYTEST_* it is OURS and inherits by default, so a child carrying it that #: resolves a production DB is by definition an isolation escape. +# : Env marker exported by the hermetic test conftest at the same moment it : redirects ``HERMES_HOME`` to +# the per-session tmp isolation root. Unlike ``PYTEST_*`` (owned by pytest, and : routinely scrubbed by +# tests that rebuild a child environment), this marker : is OURS: it declares "this process tree is running +# under Hermes test : isolation", and it inherits into subprocess children by default — so a : child that +# received the patched ``HERMES_HOME`` also received the marker, : and a child that resolves a production DB +# while carrying it is, by : definition, an isolation escape (#82770). _TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION" @@ -82,7 +88,13 @@ def _process_looks_like_pytest(proc: Any) -> bool: def _has_pytest_ancestor() -> bool: """True when an ancestor process is a pytest run: a child spawned with a rebuilt env loses PYTEST_* and the HERMES_HOME redirect together, ancestry - survives that. Fails open without psutil / on walk errors.""" + survives that. Fails open without psutil / on walk errors. + + ``_running_under_pytest`` reads ``PYTEST_*`` env vars, which a child spawned with a rebuilt environment + loses at the same moment it loses the ``HERMES_HOME`` redirect: that child aims at the production DB + *and* disarms the guard in one step (#82770). Ancestry is the one test-context signal that survives an + env rebuild, so it backs the env check up. + """ global _PYTEST_ANCESTOR if _PYTEST_ANCESTOR is not None: return _PYTEST_ANCESTOR diff --git a/hermes_state_maintenance.py b/hermes_state_maintenance.py index b1b71b571e..4cd202744d 100644 --- a/hermes_state_maintenance.py +++ b/hermes_state_maintenance.py @@ -135,7 +135,12 @@ class SessionMaintenanceMixin: whose sessions predate its first heartbeat, bounded so a PID-reuse respawn cannot protect rows forever. Disable the gate only for state.db-owned sources. SELECT, live-lease validation and UPDATE run in one ``BEGIN IMMEDIATE`` transaction; active leases/locks spare the row, expired guards are - removed so their owner is fenced.""" + removed so their owner is fenced. + + See #65194. + ``exclude_pinned`` is intended for broad automatic sweeps; pinned rows remain explicitly + recoverable. See #60609. + """ srcs = tuple(s for s in sources if s) if max_idle_seconds <= 0 or not srcs: return [] @@ -310,7 +315,13 @@ class SessionMaintenanceMixin: def _freelist_ratio(self) -> Optional[float]: """Reclaimable fraction (``freelist_count / page_count``) gating VACUUM in - :meth:`maybe_auto_prune_and_vacuum`; None = fall back to the time throttle.""" + :meth:`maybe_auto_prune_and_vacuum`; None = fall back to the time throttle. + + ``PRAGMA freelist_count / PRAGMA page_count`` read over the existing connection (never a byte-level + probe of the live file — see ``sqlite_safe_read``). This is what VACUUM would actually give back; it + is the gate :meth:`maybe_auto_prune_and_vacuum` uses to decide whether a full rewrite pays off + (#54189). + """ values = self._page_pragmas(("page_count", "freelist_count"), "Could not read freelist ratio: %s") return None if values is None else (values[1] / values[0] if values[0] > 0 else 0.0) @@ -356,7 +367,15 @@ class SessionMaintenanceMixin: :attr:`_AUTO_PRUNE_STALE_OPEN_SOURCES` older than ``retention_days`` are closed (``startup_orphan_reap``); they stay resumable and age from their close. Returns ``{"skipped", "pruned", "closed", "vacuumed"}`` plus ``"freelist_ratio"`` when a VACUUM was considered and - ``"error"`` on failure.""" + ``"error"`` on failure. + + Records the last run timestamp in state_meta so subsequent calls within ``min_interval_hours`` + no-op. Designed to be called once at startup from long-lived entrypoints (CLI, gateway, cron + scheduler). See #54189. + When *sessions_dir* is provided, on-disk transcript files (``.json`` / ``.jsonl`` / + ``request_dump_*``) for pruned sessions are removed as part of the same sweep (issue #3015). + Messaging and UI sources are never touched here. See #54189. + """ from hermes_state import _release_auto_maintenance_lock, _try_acquire_auto_maintenance_lock result: Dict[str, Any] = {"skipped": False, "pruned": 0, "closed": 0, "vacuumed": False} maintenance_lock = _try_acquire_auto_maintenance_lock(self.db_path) diff --git a/hermes_state_messages.py b/hermes_state_messages.py index 4deb712b9b..ef8f2f529a 100644 --- a/hermes_state_messages.py +++ b/hermes_state_messages.py @@ -182,8 +182,22 @@ class SessionMessagesMixin: NOT check compression_locks: the lock only stops two COMPRESSIONS colliding and archive_and_compact() commits against a watermark, so concurrent appends are safe (blocking them killed turns during slow summaries). Destructive user mutations opt in via ``reject_active_*`` so a compressor that captured - its watermark cannot resurrect the removed turn.""" + its watermark cannot resurrect the removed turn. + + Shared by :meth:`append_message` and :meth:`append_messages_batch` so the two writers can never + diverge on these correctness invariants (this guard has already needed targeted fixes — see the + #74478 patience note below). User-initiated transcript mutations may opt in to rejecting an active + unowned turn lease in that same transaction. + """ from hermes_state import CompressionSessionClosedError, SessionCompressionInProgressError, SessionTurnLeaseLostError + # NOTE (#75316 redesign): appends do NOT check compression_locks. The lock's job is to stop two + # COMPRESSIONS colliding, not to fence ordinary transcript writes. Concurrent appends during a + # compression are safe by construction: archive_and_compact() commits against a watermark captured + # at compression start and clones every row that arrived after it back into the live transcript, in + # the same write transaction. Blocking appends here was the root cause of a whole symptom family — + # turns dying as session_persistence_failed while a slow provider summary held the lease (#74568, + # #77386), including stale locks from dead PIDs blocking writes for the full TTL. Keep that narrow + # fence opt-in so ordinary appends retain the watermark behavior. if reject_active_compression_lock: active_lock = conn.execute(_COMPRESSION_LOCK_ROW_SQL, (session_id,)).fetchone() if active_lock is not None: @@ -435,7 +449,18 @@ class SessionMessagesMixin: """Atomically replace a session's messages (/retry, /undo, /compress). DESTRUCTIVE by default (rows DELETEd, leave FTS). ``active_only`` spares soft-archived rows (needed with in-place compaction). ``archive_dropped`` SOFT-archives live rows rewind-style: what rewind/edit/regenerate must use, since - DELETE leaves nothing to recover. ``reject_active_turn_lease``: in-txn lease check for user rewrites.""" + DELETE leaves nothing to recover. ``reject_active_turn_lease``: in-txn lease check for user rewrites. + + Pass ``archive_dropped=True`` to SOFT-archive the live rows instead of DELETEing them: the replaced + turns stay on disk with ``active = 0``, ``compacted = 0`` — the same "the user took it back" marking + :meth:`rewind_to_message` applies — and stay readable via :meth:`get_messages` with + ``include_inactive=True``. This is the mode a rewind/edit/regenerate must use: those flows overwrite + a transcript the user may not have meant to drop, and a plain DELETE also evicts the rows from the + FTS index, leaving nothing to recover from (#82756). It implies active-only handling — + already-archived rows are never touched — so ``active_only`` is redundant with it. The rewritten set + is inserted as fresh active rows exactly as in the destructive path, so the live view is identical + either way; only the durability of the dropped turns differs. + """ from hermes_state import CompressionSessionClosedError def _do(conn): if reject_active_turn_lease: @@ -454,7 +479,12 @@ class SessionMessagesMixin: self._execute_write(_do) def has_archived_messages(self, session_id: str) -> bool: - """True if the session has any soft-archived (``active = 0``) rows (tests/diagnostics).""" + """True if the session has any soft-archived (``active = 0``) rows (tests/diagnostics). + + Cheap existence probe — does not load rows. NOTE: production rewrite paths no longer branch on this + (they pass ``active_only=True`` unconditionally — a probe can fail open or race a concurrent + ``archive_and_compact``, #80216); kept for tests and diagnostics. + """ return self._read_one( "SELECT 1 FROM messages WHERE session_id = ? AND active = 0 LIMIT 1", (session_id,)) is not None @@ -494,7 +524,16 @@ class SessionMessagesMixin: in-txn so a reclaimed lease fails instead of clobbering the winner. *tail_count*: the LAST N compacted rows are the verbatim carried tail; their originals and the clones' originals are superseded duplicates and get rewind flags (``active=0, compacted=0``) so search doesn't return each carried - message once per compaction. ``model_config_patch`` merges in the same txn (``None`` removes a key).""" + message once per compaction. ``model_config_patch`` merges in the same txn (``None`` removes a key). + + Concurrent-append safety (#75316): when *watermark* is provided (the value of + :meth:`get_active_message_watermark` captured at compression START), rows that arrived during the + slow provider summary call (``id > watermark``) are NOT summarized away. They are re-sequenced after + the compacted set by a pure-SQL column clone (every column except ``id`` — content, api_content, + platform_message_id, token counts, reasoning sidecars all survive byte-exact, and the FTS triggers + index the clones naturally), and the originals are archived. NOTE: re-sequencing assigns the tail + rows fresh ids; consumers that reference durable row ids re-resolve by content (see 3e8ab0610). + """ from hermes_state import SessionCompressionInProgressError def _do(conn): if lock_holder is not None: @@ -657,7 +696,13 @@ class SessionMessagesMixin: """Redirect a resume target to the descendant holding the messages: follow the compression chain to the live tip (lineage-aware, so delegate/branch children never hijack it), then walk ``parent_session_id`` forward to the DEEPEST node with messages (a continuation may hold newer - turns), skipping branch/delegate/reset/tool children. Unchanged when nothing has messages; depth cap 32.""" + turns), skipping branch/delegate/reset/tool children. Unchanged when nothing has messages; depth cap 32. + + Context compression ends the current session and forks a new child session (linked via + ``parent_session_id``). The flush cursor is reset, so the child is where new messages actually land + — the parent ends up with ``message_count = 0`` rows unless messages had already been flushed to it + before compression. See #15000. + """ if not session_id: return session_id try: @@ -751,6 +796,11 @@ class SessionMessagesMixin: # Underscore-prefixed like ``_row_id``: transports strip it before the wire; compression's # assembly copies strip it so rotated child handoffs still flush (_fresh_compaction_message_copy). msg = {"role": row["role"], "content": content, _DB_PERSISTED_MARKER_KEY: True} + # Born durable (#92231): this dict is materialized FROM a durable row, so stamp the persistence + # marker at the source instead of relying on every restore caller to thread the loaded list back + # through a flush as ``conversation_history=`` — any identity-losing handoff (compression's + # durable-snapshot adoption, incremental persists with no history arg) would otherwise re-append + # the ENTIRE transcript on flush. if include_row_ids and row["id"] is not None: msg["_row_id"] = row["id"] msg.update((col, row[col]) for col in ("api_content", "display_kind") if row[col]) @@ -796,7 +846,14 @@ class SessionMessagesMixin: def get_resume_conversations(self, session_id: str) -> Tuple[List[Dict[str, Any]], List[Dict[str, Any]]]: """``(model_history, display_history)`` for a resume from ONE SELECT; byte-identical to the separate reads. model: the tip's active rows, alternation-repaired, summary marker kept for pre-compress - checkpointing. display: the full lineage (``/branch`` stands alone), compaction-archived rows deduped.""" + checkpointing. display: the full lineage (``/branch`` stands alone), compaction-archived rows deduped. + + The display projection also includes rows preserved by IN-PLACE compaction (``active=0, + compacted=1``), deduped by :meth:`_dedupe_display_generations`. Without them a compacted + conversation resumes showing only its summary plus the carried-forward tail — the user's own turns + read as deleted even though every row is still on disk, and the REST transcript read (which has + always included them) disagreed with this one about the same session (#92080). + """ rows = self._fetch_conversation_rows( self._resume_lineage_ids(session_id), _DISPLAY_ACTIVE_CLAUSE, with_session_id=True) # The model projection stays active-only: it is the compressed working context. @@ -910,6 +967,9 @@ class SessionMessagesMixin: return None return None + # ========================================================================= Rewind (soft-delete) — see + # /rewind slash command + issue #21910 + # ========================================================================= def get_active_message_ids(self, session_id: str) -> List[int]: """Ordered physical active ids for rewind CAS checks (includes legacy harness rows projections omit).""" return [int(row[0]) for row in self._read_all(_ACTIVE_IDS_SQL, (session_id,))] @@ -993,7 +1053,12 @@ class SessionMessagesMixin: return self._read_one(sql, (session_id,) if session_id else ())[0] def has_platform_message_id(self, session_id: str, platform_message_id: str) -> bool: - """True when *platform_message_id* exists (partial-index probe; the gateway's transient-failure dedupe).""" + """True when *platform_message_id* exists (partial-index probe; the gateway's transient-failure dedupe). + + Uses the idx_messages_platform_msg_id partial index for efficient lookup. Used by the gateway's + transient-failure dedupe guard (#47237) to skip re-persisting a user message that was already saved + on a prior retry of the same inbound platform message. + """ return self._read_one( "SELECT 1 FROM messages WHERE session_id = ? AND platform_message_id = ? LIMIT 1", (session_id, platform_message_id)) is not None diff --git a/hermes_state_portability.py b/hermes_state_portability.py index 54b4da8980..d2da283c50 100644 --- a/hermes_state_portability.py +++ b/hermes_state_portability.py @@ -220,7 +220,12 @@ class SessionPortabilityMixin: a complete adoption, donor rows are ARCHIVED (never deleted) with ``end_reason='adopted_by_profile'`` — deliberately NOT in the recoverable set, so resurrection cannot undo an adoption. Returns the ``import_sessions`` dict plus - ``adopted`` and ``donor_retired`` (True only when EVERY segment retired).""" + ``adopted`` and ``donor_retired`` (True only when EVERY segment retired). + + Once routing was fixed, the profile backend correctly received the RPCs but had no such session, so + the same chat 4001'd for the opposite reason. This method moves the conversation to where routing + now looks for it. See #93091, #93296. + """ payload = donor_db.export_session_lineage(session_id) if not payload: return {"ok": False, "adopted": False, "donor_retired": False, @@ -458,7 +463,14 @@ class SessionPortabilityMixin: Gateway routing, handoff, rewind and other live runtime state are reset: this restores history, not ownership of a live channel or process. Export INCLUDES ``last_activity_*`` but import RESETS them to NULL — resurrecting a stale - "working ..." label would fabricate activity the watchdog acts on (pinned).""" + "working ..." label would fabricate activity the watchdog acts on (pinned). + + Activity contract (#76354 review S4): export INCLUDES the live activity fields (``last_activity_at`` + / ``last_activity_description`` / ``last_activity_provenance``) because they are part of the durable + row, but import deliberately RESETS them to NULL. This asymmetry is intentional and covered by + regression + (tests/gateway/test_watchdog_review_76354.py::test_s4_export_includes_activity_import_resets_it). + """ if not isinstance(sessions, list): raise ValueError("sessions must be a list") if len(sessions) > self._IMPORT_MAX_SESSIONS: diff --git a/hermes_state_readpool.py b/hermes_state_readpool.py index f55087d55d..23f6f26c25 100644 --- a/hermes_state_readpool.py +++ b/hermes_state_readpool.py @@ -24,10 +24,23 @@ logger = logging.getLogger("hermes_state") # *returned*, and EMFILE is a peak-instant condition, so a connection holds a # permit for its whole lifetime; once permits are gone reads degrade to the # locked writer connection — slower, but not a wedge the supervisor can't see. +# Transient SQLITE_IOERR retry budget for READ-ONLY opens (#100436). A WAL database being actively written +# (checkpoint, WAL reset/truncate, frame flush) can surface "disk I/O error" to a concurrent ``mode=ro`` +# reader in a millisecond-wide transition window: the read-only connection cannot perform the WAL recovery a +# read through a stale or mid-update -shm file needs, because recovery requires writing the -shm index, +# which mode=ro refuses. The window closes on its own (the writer finishes the transition), so a bounded +# number of short retries makes the open succeed instead of 500-ing the whole /api/sessions poll (or any +# other read-only opener). Deliberately NOT attempted on writable opens: a writer owns the transition, so an +# IOERR there means a real storage/fd problem. _READ_POOL_MAX = 8 # Ceiling ALIVE in this PROCESS across every state.db (a multiplexed gateway # opens one per profile); three profiles' worth, then readers degrade likewise. +# _READ_POOL_MAX bounds one file. A multiplexed gateway serves N profiles from one process and each profile +# has its OWN state.db, so a per-file ceiling still lets the descriptor cost grow with the profile count — +# the same shape as the per-instance bug, one level out (#98573). Past it, readers on the (N+1)th file +# degrade to their writer connection instead of opening descriptors, which is the same trade _READ_POOL_MAX +# makes and for the same reason: a slow read path is recoverable, a process-wide EMFILE is not. _READ_POOL_PROCESS_MAX = 24 # Warn past this many SessionDB handles on one file in one process (diagnostic: @@ -36,6 +49,10 @@ _HANDLES_PER_PATH_WARN = 4 # Descriptors kept in reserve for everything that is NOT this module (httpx # sockets, terminal pipes, log files): the EMFILE SQLite pushes over surfaces elsewhere. +# The ceilings above bound Hermes's SQLite descriptors, which is only ever part of the fd table. The #98573 +# report is exactly that case: ~20 state.db descriptors were not the whole 256, they were the share that +# pushed httpx and terminal pipes over, and the EMFILE surfaced in tools/terminal_tool.py rather than here. +# So the read pool also yields when the PROCESS is close to its limit, whatever is consuming it. _FD_HEADROOM_RESERVE = 64 # The fd count is a directory listing; cache it briefly so a read burst isn't a @@ -123,7 +140,14 @@ class _PathReadBudget: """Read-connection permits for ONE database file, shared process-wide (per-instance semaphores let N SessionDBs peak at N x (1 + MAX)). An idle pooled connection keeps its permit, so a permit miss first reclaims an IDLE - connection from a peer on the same path.""" + connection from a peer on the same path. + + ``_READ_POOL_MAX`` used to be enforced by a ``BoundedSemaphore`` owned by each SessionDB, which bounded + the wrong noun: the descriptors are spent on a *file*, so N SessionDB objects on one state.db each got + their own allowance and peak scaled as ``N x (1 + _READ_POOL_MAX)``. A long-lived gateway holds at least + two (``SessionStore`` and ``GatewayRunner`` open independent handles per profile path) and the count + grows with the profile count, which is how a healthy process walked into EMFILE — #98573. + """ def __init__(self) -> None: self.permits = threading.BoundedSemaphore(_READ_POOL_MAX) @@ -143,6 +167,10 @@ class _PathReadBudget: # Writer connections cannot be capped; the only bound is not opening # redundant handles, so make the duplicate visible before it's an incident. logger.warning( + # The only real bound on writers is not opening redundant handles in the first place (which + # is what GatewayRunner borrowing SessionStore's handle does, #98573), so the next duplicate + # should be visible before it becomes an incident rather than inferred from an lsof after + # one. "%d live SessionDB handles on %s in this process; each holds " "its own writer connection (read connections are capped at %d " "for the file). A long-lived process should share one handle per path.", diff --git a/hermes_state_schema.py b/hermes_state_schema.py index f16862d03d..3b331711ee 100644 --- a/hermes_state_schema.py +++ b/hermes_state_schema.py @@ -492,7 +492,13 @@ class SessionSchemaMixin: fails closed at open, leaving search on LIKE; live write/search paths must never start a full rebuild, and a gateway opens state.db once for days, so "next open" never comes. Bounded doubling backoff, non-blocking admission, no new thread. True - only when the index was rebuilt and sync triggers restored. Never raises.""" + only when the index was rebuilt and sync triggers restored. Never raises. + + This is the in-process retry: bounded backoff from ``_FTS_STALE_RETRY_SECONDS`` doubling to + ``_FTS_STALE_RETRY_MAX_SECONDS``, non-blocking admission (``timeout=0``) so a live holder is skipped + and tried again later, no new thread — the caller is an existing periodic tick (gateway + housekeeping). See #100108, #97940. + """ if not self._fts_stale: return False if getattr(self, "_db_corrupt", False): @@ -691,7 +697,13 @@ class SessionSchemaMixin: """Rebuild ``gateway_routing`` when its PRIMARY KEY predates scoping (``session_key TEXT PRIMARY KEY``): the reconciler ADDs ``scope`` but SQLite cannot ALTER a PK, so every routing write fails (ON CONFLICT mismatch / cross-scope UNIQUE violation). Newest - row wins a cross-scope session_key collision (INSERT OR REPLACE in updated_at order).""" + row wins a cross-scope session_key collision (INSERT OR REPLACE in updated_at order). + + Early builds of the routing-index migration (#59203) created the table with ``session_key TEXT + PRIMARY KEY`` and no ``scope`` column. ``_reconcile_columns()`` ADDs the missing ``scope`` column on + those databases, but SQLite cannot ALTER a primary key, so the shipped composite ``PRIMARY KEY + (scope, session_key)`` never lands. On such tables every write path is broken: + """ pk_cols = self._live_pk_columns(cursor, "gateway_routing") if pk_cols is None or pk_cols == ["scope", "session_key"]: return @@ -720,7 +732,12 @@ class SessionSchemaMixin: window: INSERT OR IGNORE does NOT suppress FK violations, so an orphaned usage row would abort the rebuild (PRAGMA foreign_keys is a no-op inside a transaction; none is open here). OR IGNORE: COALESCE(task, '') on legacy NULL rows can collide with a - genuine ''-task row — keep the first.""" + genuine ''-task row — keep the first. + + Installs whose ``state.db`` reached ``schema_version >= 22`` before the ``task`` dimension was added + carry a 5-column PRIMARY KEY ``(session_id, model, billing_provider, billing_base_url, + billing_mode)``. See #73823. + """ pk_cols = self._live_pk_columns(cursor, "session_model_usage") if pk_cols is None or "task" in pk_cols: return @@ -732,6 +749,13 @@ class SessionSchemaMixin: try: self._rebuild_table( cursor, "session_model_usage", "session_model_usage_legacy_pk", _SESSION_MODEL_USAGE_HEAL_DDL, + # v20: per-model usage attribution (issue #51607). Going forward update_token_counts() + # records each API call into session_model_usage keyed by the live model, but existing + # sessions only have their aggregate totals on the sessions row. Seed one usage row per + # historical session from those aggregates so insights reads uniformly from the new table. + # INSERT OR IGNORE keeps it idempotent: if newer code already wrote a (session_id, model, + # provider) row for a session, the PK conflict skips the stale aggregate rather than + # doubling it. """INSERT OR IGNORE INTO session_model_usage ( session_id, model, billing_provider, billing_base_url, billing_mode, task, api_call_count, input_tokens, @@ -764,6 +788,14 @@ class SessionSchemaMixin: never skip a column; schema_version remains for data migrations only.""" # Startup-watchdog lease: on multi-GB files this is I/O-bound (near-zero CPU), which # the watchdog's CPU fallback would misread as a parked deadlock. + # Declare a startup-watchdog progress lease before potentially long synchronous work: on multi-GB + # state.db files the reconciliation + version-gated data migrations below are legitimately slow and + # can be I/O-bound (near-zero CPU), which the watchdog's CPU fallback would misread as a parked + # deadlock (OOF-298 / PR #89750). Single lease is deliberate: this is the one pre-loop phase that + # can legitimately exceed the 300s default deadline (multi-GB DBs), and the lease is clamped to + # _MAX_LEASE_S=900. Honest worst case: a genuinely wedged DB init delays supervisor respawn by up to + # the lease duration. Per-chunk renewal would shrink that, but adds complexity to the migration + # loops for a rare failure mode. report_startup_progress(600.0, phase="state_db_init_schema") cursor = self._conn.cursor() cursor.executescript(SCHEMA_SQL) @@ -771,10 +803,20 @@ class SessionSchemaMixin: # Column reconciliation, then the two table-shape repairs ADD COLUMN cannot express. self._reconcile_columns(cursor) self._heal_gateway_routing_pk(cursor) + # Rebuild session_model_usage if its PRIMARY KEY lacks the ``task`` column (5-column PK on installs + # already at v22+ when the column landed — the version-gated rebuild is unreachable there, #73823). + # Same PK-rebuild constraint as gateway_routing above. self._heal_session_model_usage_pk(cursor) # Indexes referencing reconciler-added columns must be created AFTER _reconcile_columns # (in SCHEMA_SQL the executescript would fail on legacy DBs). + # Heal NULL ``active`` rows unconditionally on every startup. On real-world DBs the reconciler-added + # ``active`` column can lack its NOT NULL DEFAULT 1 (older reconciler builds reconstructed the type + # without the default — see #51646: PRAGMA shows (17,'active','INTEGER',0,None,0) in the wild), so + # INSERTs that omitted the column wrote NULL and the ``WHERE active = 1`` transcript loaders hid the + # whole history. The INSERTs now set active=1 explicitly; this idempotent repair un-hides rows + # written before the fix. It was previously gated at ``current_version < 12`` which never re-ran for + # already-v12+ databases. try: cursor.execute( "CREATE INDEX IF NOT EXISTS idx_messages_platform_msg_id " @@ -805,6 +847,7 @@ class SessionSchemaMixin: if row is None: cursor.execute("INSERT INTO schema_version (version) VALUES (?)", (SCHEMA_VERSION,)) # Store provenance so fresh vs wiped stores are distinguishable. + # See #97568. now_iso = datetime.datetime.now(datetime.timezone.utc).isoformat() cursor.executemany( "INSERT OR IGNORE INTO state_meta (key, value) VALUES (?, ?)", @@ -824,6 +867,11 @@ class SessionSchemaMixin: # Renew the lease: the chain can rewrite whole tables on large DBs. report_startup_progress(600.0, phase="state_db_data_migrations") # (v10 trigram backfill and v11 inline FTS re-index were superseded by v23 and removed.) + # v11 (SUPERSEDED by v23): re-index FTS5 tables to cover tool_name + tool_calls in inline mode + # (#16751). v23 drops and rebuilds both FTS tables in external-content form, so running the v11 + # inline backfill first would only burn startup time and WAL space before v23 throws the work away — + # and its inline INSERT shape no longer matches the current external-content FTS_SQL anyway. Kept + # only for source archaeology; unreachable while SCHEMA_VERSION >= 23. if current_version < 16: # v16: tag delegate subagent rows so pickers stay clean after parent deletes orphan them. with contextlib.suppress(sqlite3.OperationalError): @@ -847,6 +895,8 @@ class SessionSchemaMixin: if current_version < 18: # v18: best-effort gateway metadata backfill from sessions.json. try: + # Backfill display_name / origin_json / expiry_finalized from sessions.json so pre-migration + # gateway sessions are discoverable from state.db without the JSON index. See #9006. self._backfill_gateway_metadata_from_sessions_json(cursor) except Exception as exc: logger.debug("v18 gateway metadata backfill skipped: %s", exc) @@ -877,6 +927,22 @@ class SessionSchemaMixin: # marker until optimize-storage runs. An INTERRUPTED optimize (markers, trash, or an # empty external index against non-empty messages) is NOT stamped: the marker is the # source of truth for "fully optimized" and keeps the resume offer alive. + # v23: FTS storage redesign (issues #22478, #43690, #55233). The v11 inline-mode FTS tables each + # store a full private copy of every message (content || tool_name || tool_calls), and the trigram + # index additionally covers role='tool' rows (~90% of message bytes: base64 payloads, file dumps) at + # ~2.6x amplification — together ~75% of state.db on heavy installs (observed: 18.9 GB of a 25 GB + # DB). OPT-IN, NOT AUTOMATIC. The transition (demote old vtables → new external-content schema → + # backfill → teardown → VACUUM) is disk-heavy (transient ~2x file size to fully reclaim via VACUUM) + # and long (~1-2h background on a 25 GB DB). Doing it silently on every big user's next open — with + # a completeness guarantee that depends on the process staying alive long enough — is the wrong + # default. So on an EXISTING install we touch nothing here: the v22 inline FTS keeps working exactly + # as before, and we only record a flag advertising that the optimization is available. `hermes + # sessions optimize-storage` performs the whole transition as one deliberate, disk-checked, + # progress-reported foreground operation. DECOUPLED VERSIONING. Crucially, this does NOT hold back + # the main schema_version. The FTS storage LAYOUT is tracked by an independent `fts_storage_version` + # marker (see _fts_storage_version / SETTLE below), so schema_version advances to SCHEMA_VERSION + # here like every other migration — future v24+ migrations land automatically for legacy-FTS users + # too. Only the FTS *layout* waits for opt-in. if ( fts5_available and not self._db_needs_fts_storage_upgrade(cursor) @@ -898,6 +964,10 @@ class SessionSchemaMixin: """v22: ``task`` joins the session_model_usage PRIMARY KEY ('' = main loop; aux calls named). SQLite cannot ALTER a PK, so rebuild; existing rows → task=''.""" try: + # v22: task-dimension usage attribution (issue #23270). session_model_usage gains a ``task`` + # column ('' = main agent loop; 'vision'/'compression'/'title_generation'/... = auxiliary calls) + # so aux model spend is visible in analytics. The reconciler will have already ADDed the plain + # column on legacy DBs (harmless); the rebuild bakes it into the PK properly. legacy_pk = cursor.execute( "SELECT COUNT(*) FROM pragma_table_info('session_model_usage') WHERE name = 'task' AND pk > 0" ).fetchone()[0] @@ -992,7 +1062,10 @@ class SessionSchemaMixin: simultaneously (the interleaving that corrupted state.db in production), so this FAILS CLOSED: on deferral the just-repaired triggers are dropped again and the stale breadcrumb persisted — triggers must never be live over an unrebuilt gap - (``_enter_fts_fail_open``'s ordering contract); a later recovery path restores both.""" + (``_enter_fts_fail_open``'s ordering contract); a later recovery path restores both. + + See #93200. + """ with fts_rebuild_admission(self.db_path) as admitted: if admitted: rebuild_fn() diff --git a/hermes_state_search.py b/hermes_state_search.py index e6e769205c..53d0c9fc92 100644 --- a/hermes_state_search.py +++ b/hermes_state_search.py @@ -191,6 +191,10 @@ class SessionSearchMixin: try: self._merge_fts_incrementally(max_pages=self._FTS_MERGE_MAX_PAGES_PER_INDEX) except Exception as exc: # noqa: BLE001 - post-commit maintenance + # The canonical write is already committed before this cadence runs. No maintenance failure — + # including the bare SystemError the CPython sqlite3 layer can raise under cross-thread errmsg + # scrambling — may escape and make the caller replay an ambiguous, possibly-durable write + # (#90734, #85079). logger.warning("FTS incremental merge failed after commit: %s", exc) # ── Deferred rebuild engine (base + CJK backfills) ───────────────────── @@ -341,7 +345,13 @@ class SessionSearchMixin: """Tear down one chunk of a demoted v22 FTS shadow table (a PLAIN table now); True while work remains. INTEGER single-column-key tables drain with a high-water marker so each chunk's scan is bounded (restarting the scan was O(n²)); compound-key tables - keep the chunked ``LIMIT`` delete — they are small by construction.""" + keep the chunked ``LIMIT`` delete — they are small by construction. + + Single-column-key trash tables (the common shape — FTS shadow tables carry a rowid/integer PK) are + drained with a high-water marker mirroring :meth:`fts_rebuild_step`: each chunk deletes only rows + after the previously-drained key, so the per-chunk scan is bounded instead of re-scanning from the + start of the table every chunk (O(n²) total on large trash tables, #79324). + """ with self._read_ctx() as conn: trash = [r[0] for r in conn.execute( "SELECT name FROM sqlite_master WHERE type = 'table' AND name LIKE ? ESCAPE '\\'", @@ -373,6 +383,9 @@ class SessionSearchMixin: if cur.rowcount > 0: self.set_meta(marker_key, str(upper), cursor=conn) return True + # Compound-key or rowid trash table: legacy chunked delete. These shadow tables are small, so + # the quadratic re-scan is not a concern (#79324 keeps the high-water path for the big + # single-key tables). cur = conn.execute( f"DELETE FROM {tbl} WHERE ({key}) IN (SELECT {key} FROM {tbl} LIMIT {self._FTS_REBUILD_CHUNK_ROWS})" ) @@ -474,7 +487,10 @@ class SessionSearchMixin: """Heal interrupted demote/backfill bookkeeping before optimize runs: orphan high_water gets progress re-seeded; an empty external index with messages and no markers gets a full backfill claim. Never invents markers on a still-legacy inline DB — optimize - would then skip demote and INSERT against the inline table forever.""" + would then skip demote and INSERT against the inline table forever. + + Covers two post-#65798 failure classes: + """ def _do(conn): if _meta_row(conn, "fts_rebuild_high_water") is not None: self._reseed_missing_progress(conn) @@ -568,6 +584,10 @@ class SessionSearchMixin: # a TRUNCATE reset from a transient CLI would race a live writer. try: with self._lock: + # Best-effort: fold the WAL back into the main file so the on-disk size settles now rather + # than at close(). Callers must therefore NOT size the result by stat()ing the file; use + # :meth:`logical_size_bytes`, which is truthful immediately regardless of readers. See + # #45383. self._conn.execute("PRAGMA wal_checkpoint(PASSIVE)") except Exception as exc: logger.debug("WAL checkpoint (PASSIVE) after optimize VACUUM failed: %s", exc) @@ -1046,6 +1066,10 @@ class SessionSearchMixin: except sqlite3.DatabaseError as exc: # Corruption parent class: detach the derived indexes and answer from # canonical rows; repair paths own the rebuild. + # A corrupt FTS index raises the malformed / "fts5: corrupt structure record" class on the + # MATCH read, the same class the write path handles (#66296). OperationalError (query + # syntax) is a subclass caught above; this arm is the corruption parent. The existing + # stale-open/repair paths retain rebuild ownership. if not self._enter_fts_fail_open(exc): raise matches = self._search_messages_like_fallback(query, limit=limit, offset=offset, sort=sort, **filters) @@ -1068,6 +1092,13 @@ class SessionSearchMixin: # "concatenate"). Skipped for role='tool' (both indexes exclude tool rows). if not matches and not is_cjk and not (bool(role_filter) and "tool" in role_filter): fb_query = _quote_fts_tokens(query.strip('"').strip()) + # ── CJK-bigram route (messages_fts_cjk, cjk_unicode61) ────── When the bigram index is + # available it serves EVERY CJK query shape the legacy code split between trigram (>=3 + # chars/token) and LIKE full scans (1-2 char tokens) — the whole point of the index (PR #65544). + # Exceptions stay on the legacy routes: - role_filter=['tool'] queries (tool rows aren't in the + # cjk index, same exclusion as trigram), - queries containing a LONE 1-char CJK run: the index + # stores bigrams for runs >=2, so a single-char term can only match isolated chars — LIKE + # substring semantics are broader. if self._fts_cjk_available: matches = self._match_rows("messages_fts_cjk", fb_query, **route) or matches if not matches and self._trigram_available and self._trigram_eligible_tokens(query): @@ -1175,7 +1206,19 @@ class SessionSearchMixin: that rejects writes while reads succeed. Two processes rebuilding one state.db concurrently corrupted production DBs, so this admits through ``fts_rebuild_admission`` and FAILS CLOSED, returning 0 on deferral (callers treat 0 - as "no progress" and use the stale-FTS breadcrumb path). Returns indexes rebuilt.""" + as "no progress" and use the stale-FTS breadcrumb path). Returns indexes rebuilt. + + Uses the FTS5 ``'rebuild'`` command, which rewrites the internal b-tree segments from the content + rows. Unlike ``optimize_fts`` (which merges existing segments), ``rebuild`` discards and recreates + the index data entirely. See #50502. + A full structural rebuild must never run concurrently in two processes sharing one state.db — that + interleaving has structurally corrupted the database in production (PR #93200) — so this admits + through the cross-process ``fts_rebuild_admission`` authority and FAILS CLOSED: if another process + holds the rebuild lock beyond the bounded wait, this call defers (returns 0) rather than racing it. + Callers already treat 0 as "rebuild made no progress" and fall back to the stale-FTS breadcrumb + path, which retries in-process from the gateway housekeeping tick (``retry_deferred_fts_recovery``) + and at next startup. + """ rebuilt = 0 with fts_rebuild_admission(self.db_path) as admitted: if not admitted: diff --git a/hermes_state_sessions.py b/hermes_state_sessions.py index ade7128b4f..ff27a0e244 100644 --- a/hermes_state_sessions.py +++ b/hermes_state_sessions.py @@ -100,6 +100,14 @@ def _session_filter_where( params: List[Any] = [] if exclude_children: where += [_LISTABLE_CHILD_SQL, f"{_delegate_from_json('s.model_config')} IS NULL"] + # Show roots and user-visible branch/reset sessions, while still hiding sub-agent runs and compression + # continuations. All four carry parent_session_id, so the shared predicate classifies the edge from + # stable markers plus legacy-compatible parent metadata. Branch sessions are identified two ways, OR'd + # for robustness: 1. A stable ``_branched_from`` marker in model_config, written by /branch at creation + # time. This survives the parent being reopened and re-ended with a different end_reason (e.g. + # tui_shutdown overwriting 'branched'), which otherwise hides the branch — see issue #20856. 2. The + # legacy heuristic (parent ended with 'branched' before the child started), covering branch sessions + # created before the marker existed. include_sources = [source] if source else list(sources or []) for clause, values in ( (f"s.source IN ({_session_ids_placeholders(include_sources)})", include_sources), @@ -125,6 +133,11 @@ def _collect_delegate_child_ids(conn, parent_ids: List[str]) -> List[str]: seeds = {sid for sid in parent_ids if sid} # Seed visited with the parents: a marker chain can loop back onto a parent, # which would then be collected as its own descendant. Never return parents. + # A delegation marker chain can loop back onto a parent — a cycle, or a parent that is also another + # parent's delegate child when several ids are deleted at once — and without this guard that parent + # would be collected as one of its own descendants and cascade-deleted along with all of its messages. + # Callers delete the parents separately, so parents must never appear in the returned child set. + # (#49148) found: set[str] = set(seeds) frontier = list(seeds) while frontier: @@ -173,6 +186,10 @@ def classify_session_status(role: Optional[str], has_tool_calls: bool, finish_re # Parent→child profile_name inheritance fence: keyless rows inherit freely; two # ``agent:<ns>:...`` keyed rows must agree on the namespace. +# ``agent:<ns>:...`` gateway keys encode the profile namespace; a keyless row (CLI / subagent lineage) +# carries none and inherits freely. Two keyed rows must agree on ``agent:<ns>:`` — a default child +# (``agent:main:``) forked from a sibling profile's row must not be durably mislabelled as that profile's. +# See #88381. _SAME_KEY_NAMESPACE_SQL = ( "p.session_key IS NULL OR sessions.session_key IS NULL" " OR substr(p.session_key, 1, instr(substr(p.session_key, 7), ':') + 6)" @@ -261,7 +278,29 @@ class SessionSessionsMixin: """Upsert a session row, never overwriting what an earlier writer set (the gateway creates a bare row before create_session carries the real model/prompt). chat_id/thread_id scope gateway /resume (IDOR). Children backfill from the parent; a missing profile_name is stamped with THIS - store's own (NULL reads as unowned).""" + store's own (NULL reads as unowned). + + When ``parent_session_id`` is set (compression fork, delegate/subagent spawn, branch continuation) + and this row's own ``cwd``/``git_repo_root``/ ``git_branch``/``profile_name`` are still NULL after + the insert, they are backfilled from the parent row. Callers of ``create_session`` for a child + session historically didn't propagate these fields themselves (e.g. the compression-fork path), so a + lineage could silently lose its working directory and drop out of the project sidebar every time it + forked (#64709), or lose its owning profile and be aggregated as "default" every time it rotated or + branched (the cross-profile session-jump bug). This only fills NULLs — an explicit value on the + child is never overwritten. For compression forks specifically (parent ended with + ``end_reason='compression'``), the gateway origin columns + (``user_id``/``session_key``/``chat_id``/``chat_type``/ + ``thread_id``/``display_name``/``origin_json``) are inherited too, so a crash before the gateway + re-records the peer can't strand the child without a recoverable routing mapping (#59527). + When the caller passes no ``profile_name`` at all, the row is stamped with THIS store's own profile + (:meth:`_own_profile_name`) instead of NULL. Every ``state.db`` belongs to exactly one profile — the + same single-match contract :meth:`backfill_null_session_profiles` relies on — so the stamp is + derivation, not a guess. Rows minted NULL after that one-shot #94724 backfill ran stayed NULL + forever, and profile-keyed consumers (desktop sidebar scope matching, ``@session:<profile>/<id>`` + deep links, the fail-closed owner ladder) treat NULL as unowned: the session vanishes from the + sidebar even though its transcript is intact (#99222). Stores outside the profile tree (explicit + ``db_path`` in tests, ad-hoc copies) derive nothing and keep NULL — never guess. + """ if not (profile_name or "").strip(): profile_name = self._own_profile_name() def _do(conn): @@ -331,7 +370,10 @@ class SessionSessionsMixin: return session_id def set_expiry_finalized(self, session_id: str, finalized: bool = True) -> None: - """Mirror ``SessionEntry.expiry_finalized`` so it survives a lost sessions.json.""" + """Mirror ``SessionEntry.expiry_finalized`` so it survives a lost sessions.json. + + See #9006. + """ if not session_id: return self._write_sql( @@ -421,7 +463,12 @@ class SessionSessionsMixin: def promote_to_session_reset(self, session_id: str, reason: str = "session_reset") -> bool: """Durably mark an intentional reset boundary on live rows or rows with a *recoverable* accidental end_reason (explicit boundaries are preserved): an ``agent_close`` row left recoverable would be - resurrected by stale-route recovery. Keep in sync with find_latest_gateway_session_for_peer.""" + resurrected by stale-route recovery. Keep in sync with find_latest_gateway_session_for_peer. + + Plain ``end_session()`` is NOT sufficient for reset boundaries: it no-ops on an already-ended row, + so a row that agent cleanup already closed as ``agent_close`` would stay recoverable and stale-route + recovery would resurrect the reset session with its full history (#61220, #61993, #63539). + """ if not session_id: return False now = time.time() @@ -504,7 +551,12 @@ class SessionSessionsMixin: provenance: Optional[ActivityProvenance] = None, ) -> None: """Stamp durable mid-turn activity (observation-only; rate-limited by the caller) so surfaces see - activity before any message row lands. Never moves ``last_activity_at`` backwards.""" + activity before any message row lands. Never moves ``last_activity_at`` backwards. + + Called (rate-limited) from ``AIAgent._touch_activity`` so gateway/CLI surfaces and stall consumers + observe API/tool/compaction activity even when no new message row has been written yet (#72016 / + #72039). + """ if not session_id: return when = float(ts if ts is not None else time.time()) @@ -519,9 +571,20 @@ class SessionSessionsMixin: patience_s=self._ACTIVITY_WRITE_PATIENCE_S, ) + # Observation-only write: never let it ride the full routine write-patience budget (#76354 review S1). + # Under contention a heartbeat that waits ~20s would delay the response-critical path it is merely + # observing; give up after a sub-second budget instead (the next due window retries naturally). def clear_session_activity_labels(self, session_id: str) -> None: """Clear activity labels after a turn (``last_activity_at`` is kept so idle / watchdog clocks stay - continuous). A no-op clear skips the write transaction.""" + continuous). A no-op clear skips the write transaction. + + Description and provenance are observation labels for *what was happening at* that timestamp during + an active turn; once the turn is idle they must not keep advertising "compressing" / "executing + tool" (#72039). + Response-critical-path contract (#76354 review S1): runs in the turn's ``finally``; a no-op clear + (labels already empty) skips the write transaction entirely, and a real clear uses the same short + sub-second busy budget as :meth:`touch_session_activity` instead of the full routine write patience. + """ if not session_id: return try: @@ -582,7 +645,13 @@ class SessionSessionsMixin: def update_session_model(self, session_id: str, model: str, provider: Optional[str] = None) -> None: """Set the model after a mid-session /model switch (unconditionally), null system_prompt so stale Model:/Provider: footers rebuild, and drop any Browser runtime lock (lineage markers - survive). *provider* is merged into model_config so resume recombines model and provider.""" + survive). *provider* is merged into model_config so resume recombines model and provider. + + When *provider* is given, it is merged into ``model_config`` alongside the model (``$.model`` / + ``$.provider``) so a later resume recombines the persisted model with the provider that actually + serves it instead of the config.yaml primary provider (#79536). Callers without provider knowledge + leave any stored provider untouched. + """ # Flush first: a still-queued pre-switch delta applied after this UPDATE would trip the # first_accounted_route overwrite and resurrect the old route. self.flush_token_counts() @@ -721,7 +790,13 @@ class SessionSessionsMixin: def backfill_null_session_profiles(self, profile_name: str) -> int: """Stamp this store's own profile onto legacy ``profile_name IS NULL`` rows, which the fail-closed - owner ladder cannot route. Never overwrites a non-NULL owner. Returns rows stamped.""" + owner ladder cannot route. Never overwrites a non-NULL owner. Returns rows stamped. + + Sessions created before the durable-ownership work (#95407 lineage) carry ``profile_name = NULL``. + On single-backend installs that was harmless, but once a Desktop registers a second connection the + fail-closed owner ladder (which is correct for new sessions) can no longer route those rows anywhere + — every pre-campaign session becomes unresumable after upgrade (#94724, field report). + """ stamp = (profile_name or "").strip() if not stamp: return 0 @@ -778,7 +853,16 @@ class SessionSessionsMixin: def unarchive_recoverable_session(self, session_id: str) -> bool: """Un-archive a session archived by a recoverable accident (ws_orphan_reap, agent_close); - deliberate archives are left alone. True when un-archived.""" + deliberate archives are left alone. True when un-archived. + + Registry-style lookups (Bot Mode's canonical "Bot Chat") use this to resurrect a row the ws-orphan + reaper (``ws_orphan_reap``) or older agent cleanup (``agent_close``) archived: those ends are + accidents, not user intent, so the identity-scoped canonical chat must survive them (#92687). + Sessions archived with no end_reason or an explicit boundary reason (user archived deliberately, + ``session_reset``, …) are left untouched — returns ``False`` for those, ``True`` only when the row + was archived for a recoverable reason and is now un-archived (whole compression lineage, via + :meth:`set_session_archived`). + """ if not session_id: return False try: @@ -1332,7 +1416,13 @@ class SessionSessionsMixin: def declared_scope_identity(self, session_id: str) -> Tuple[bool, str]: """(is_fork_child, source) in ONE read (prompt_cache_scope needs both from the same row). - Missing row → (False, ""); DB errors propagate (fail closed).""" + Missing row → (False, ""); DB errors propagate (fail closed). + + ``agent/prompt_cache_scope.py`` needs both to resolve a host-declared conversation scope, and both + live on the same ``sessions`` row; asking for them separately read that row twice per resolution + (@teknium1 on 98811). The marker rules stay here, beside :meth:`is_explicit_fork_child`, instead of + being re-implemented by the caller. See #98811. + """ session = self.get_session(session_id) if not session: return False, "" @@ -1361,6 +1451,9 @@ class SessionSessionsMixin: with self._read_ctx() as conn: if not conn.execute("SELECT 1 FROM sessions WHERE id = ? LIMIT 1", (session_id,)).fetchone(): return [] + # Use the borrowed read connection, never self._conn: handing the shared writer connection to a + # helper here executes on it without self._lock — the same unsynchronized-read class as + # #99349/#90734. delegate_ids = _collect_delegate_child_ids(conn, [session_id]) return [session_id, *sorted(delegate_ids)] @@ -1452,6 +1545,8 @@ class SessionSessionsMixin: # Shared by count_empty_sessions / delete_empty_sessions so badge and sweep agree. message_count # counts live rows only (rewind/compaction keep dropped turns as active = 0): NOT EXISTS is authority. + # The ``NOT EXISTS`` probe is the authority; : ``message_count = 0`` stays as a cheap prefilter. Same + # shape as every : other emptiness guard in this module. (#95868) _EMPTY_SESSION_WHERE = ( "message_count = 0 AND ended_at IS NOT NULL AND archived = 0 AND NOT EXISTS (" "SELECT 1 FROM messages WHERE messages.session_id = sessions.id)" diff --git a/hermes_state_telegram.py b/hermes_state_telegram.py index c8a37d97ae..995a68c1b3 100644 --- a/hermes_state_telegram.py +++ b/hermes_state_telegram.py @@ -112,7 +112,10 @@ class SessionTelegramTopicsMixin: part of startup reconciliation: operators can upgrade and keep the old bot behavior until a user runs /topic. Schema versions: v1 initial; v2 session_id FK ON DELETE CASCADE (pruning clears bindings); v3 ``profile_name`` on both tables so - multiplexed gateways sharing one state.db isolate topic state per profile.""" + multiplexed gateways sharing one state.db isolate topic state per profile. + + See #76423. + """ def _do(conn): for table, columns, ddl in _TOPIC_TABLES: conn.execute(f"CREATE TABLE IF NOT EXISTS {table} ({ddl})") @@ -149,7 +152,12 @@ class SessionTelegramTopicsMixin: allows_users_to_create_topics: Optional[bool]=None, ) -> None: """Enable Telegram DM topic mode for one private chat/user. Owns the explicit topic - migration; SessionDB startup must not create these tables.""" + migration; SessionDB startup must not create these tables. + + ``profile_name`` namespaces rows under a shared multiplex ``state.db`` (issue #76423). Callers + handling a multiplexed event must pass the routed profile from ``source.profile``, not the + process-global active profile. + """ self.apply_telegram_topic_migration() now = time.time() profile_name = _normalize_telegram_topic_profile_name(profile_name) @@ -245,7 +253,13 @@ class SessionTelegramTopicsMixin: binding, ``telegram_dm_topic_mode`` is flipped to ``enabled = 0`` in the same transaction, or a user who disabled topics in the Telegram client (not via ``/topic off``) stays stuck. Returns the number of rows deleted; absent binding or - unmigrated tables are silent no-ops (never raise from a cleanup hot path).""" + unmigrated tables are silent no-ops (never raise from a cleanup hot path). + + Without this prune, the stale row keeps living in ``telegram_dm_topic_bindings`` and the recovery + logic in ``gateway.run._recover_telegram_topic_thread_id`` cheerfully redirects future inbound + messages to the deleted topic, causing tool progress, approvals, and replies to land in the wrong + place. Issue #31501. + """ chat_id, thread_id = str(chat_id), str(thread_id) profile_name = _normalize_telegram_topic_profile_name(profile_name) @@ -327,7 +341,10 @@ class SessionTelegramTopicsMixin: ) -> List[Dict[str, Any]]: """This user's Telegram sessions not bound to a topic. Read-only: if the bindings table is absent, every session is unlinked and the profile-unscoped query is used. - Scoped by ``profile_name`` so multiplexed profiles do not surface each other.""" + Scoped by ``profile_name`` so multiplexed profiles do not surface each other. + + See #76423. + """ profile_name = _normalize_telegram_topic_profile_name(profile_name) with self._read_ctx() as conn: try: diff --git a/hermes_state_usage.py b/hermes_state_usage.py index e9ff699112..f272b76070 100644 --- a/hermes_state_usage.py +++ b/hermes_state_usage.py @@ -87,7 +87,10 @@ class SessionUsageMixin: ) -> None: """Unconditionally set the billing route (``update_token_counts`` only COALESCE-fills NULLs) so the dashboard reflects the latest /model switch; also nulls - ``system_prompt`` so the cached snapshot header is rebuilt.""" + ``system_prompt`` so the cached snapshot header is rebuilt. + + See #48173, #48248. + """ # Barrier against queued token deltas — see update_session_model. self.flush_token_counts() @@ -298,6 +301,11 @@ class SessionUsageMixin: # mid-session /model switch would attribute every token to the initial model. Only # the incremental path records here — absolute cumulative updates cannot be split # back into routes; Insights reconciles the residual instead. + # ``update_token_counts`` is the single chokepoint every per-API-call delta flows through (CLI, + # gateway, cron, delegated runs — see conversation_loop / codex_runtime), and each call carries the + # model/provider *active at the time of that call*. Recording the per-call delta into + # session_model_usage keyed by the live model preserves an accurate per-model breakdown regardless + # of how many times the user switches. See #51607. record_model_usage = (not absolute) and has_usage def _do(conn): @@ -334,7 +342,12 @@ class SessionUsageMixin: write txn after the ``sessions`` UPDATE. A missing model/provider falls back to the session row — except for aux rows (``task`` set), which must NOT inherit the main-loop route (vision on gemini while the main loop runs anthropic): missing - info stays 'unknown'/empty.""" + info stays 'unknown'/empty. + + ``task`` distinguishes what kind of work consumed the tokens: ``''`` (empty) is the main agent loop; + auxiliary calls record their task name (``vision``, ``compression``, ``title_generation``, ...) via + :meth:`record_auxiliary_usage` (issue #23270). + """ row = conn.execute( "SELECT model, billing_provider, billing_base_url, billing_mode FROM sessions WHERE id = ?", (session_id,), ).fetchone() @@ -357,7 +370,12 @@ class SessionUsageMixin: """Record an auxiliary LLM call's usage (vision, compression, title generation, ...) as a per-(model, provider, task) delta in ``session_model_usage`` WITHOUT touching the ``sessions`` summary row (the gateway overwrites those counters with absolute - main-loop totals). ``api_call_count`` may aggregate N calls. Best-effort.""" + main-loop totals). ``api_call_count`` may aggregate N calls. Best-effort. + + See #23270. + Background-review forks record an aggregate of N fork API calls in one write with + ``task='background_review'`` (issue #87250). + """ usage = {k: v for k, v in locals().items() if k in _MODEL_USAGE_FIELDS} if not session_id or not task: return diff --git a/hermes_state_wal.py b/hermes_state_wal.py index 1092ceffff..12fc2c55c2 100644 --- a/hermes_state_wal.py +++ b/hermes_state_wal.py @@ -94,7 +94,14 @@ def _darwin_pragma(conn: sqlite3.Connection, pragma: str) -> None: def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None: """Enable ``PRAGMA checkpoint_fullfsync`` on macOS. Apple's ``fsync(2)`` guarantees neither data-on-platter nor ordering, so without ``F_FULLFSYNC`` a launchd shutdown can turn a "durable" checkpoint into a malformed - ``state.db``. Checkpoint boundaries only (~+0.1 ms/commit vs ~+4 ms for ``fullfsync=1``).""" + ``state.db``. Checkpoint boundaries only (~+0.1 ms/commit vs ~+4 ms for ``fullfsync=1``). + + During a launchd *system* shutdown/reboot the OS page cache is dropped (effectively a power-loss event + for in-flight pages), so a WAL checkpoint whose ``fsync()`` "reported" durable may never have hit the + platter — corrupting ``state.db`` with a malformed image. This is the trigger in issue #30636 ("SIGTERM + during launchd shutdown under high load"), distinct from a plain in-session kill (which the page cache + survives and SQLite recovers from). + """ _darwin_pragma(conn, "PRAGMA checkpoint_fullfsync=1") @@ -176,7 +183,16 @@ def apply_wal_with_fallback(conn: sqlite3.Connection, *, db_label: str = "state. Invariant on every path: never downgrade to DELETE if the on-disk header reports WAL or cannot be read — other gateway/cron/worker connections may hold the DB open, and a live downgrade destroys their uncheckpointed - commits.""" + commits. + + An earlier revision of the lock-cancellation fix (#71724) reverted it on the theory that DELETE was "the + mode that corrupts", but that comparison was confounded: the clean WAL result came from SQLite 3.53.1, + which carries BOTH the WAL-reset fix AND 3.51.0's defenses against close()-broken POSIX locks, so it + says nothing about 3.50.4. Re-measured on the actually-bundled 3.50.4 with the lock fix in place, WAL + and DELETE are both clean (0/3 each) — i.e. there is no evidence that WAL is safer here, and upstream + still documents the WAL-reset bug as real through 3.51.2 with serious consequences. Until a fixed + runtime is delivered, keep new databases out of WAL. + """ from hermes_state import is_sqlite_wal_reset_vulnerable, resolve_journal_mode configured = resolve_journal_mode() @@ -194,6 +210,8 @@ def apply_wal_with_fallback(conn: sqlite3.Connection, *, db_label: str = "state. return "wal" if configured == "delete": + # #68545: honor the canonical database.journal_mode setting. Existing on-disk WAL databases were + # returned above and are never live-downgraded. if current_mode is None: # Probe failed (locked/busy): ownership not provably exclusive. Fail loudly. raise sqlite3.OperationalError(_CANNOT_VERIFY_DELETE_MSG) @@ -230,6 +248,13 @@ def _enable_wal(conn: sqlite3.Connection, db_label: str, require_wal: bool, curr msg = str(exc).lower() if not any(marker in msg for marker in _WAL_INCOMPAT_MARKERS): raise # unrelated OperationalError — don't silently swallow + # ``disk i/o error`` is ambiguous: on ZFS / APFS-CoW it is a deterministic WAL-incompatibility (SHM + # corruption under concurrent connection bursts — #55305, #71498), but it can also be a one-shot + # transient EIO (page-cache pressure, brief lock contention). Treating a transient EIO as a + # permanent downgrade signal produced the mixed-journal-mode corruption pattern fixed in 5c49cd0ed0 + # (process A downgrades to DELETE while sibling processes set WAL). Disambiguate by retrying the + # pragma a couple of times: transient EIO clears and we return "wal"; the deterministic filesystem + # cases keep failing and fall through to the guarded DELETE fallback. if "disk i/o error" in msg: # Retry twice: EIO is either deterministic WAL-incompatibility (ZFS / APFS-CoW) or a one-shot transient, # and treating a transient as a permanent downgrade produced mixed-mode corruption (A downgrades to @@ -316,7 +341,10 @@ def _apply_delete_for_wal_reset_bug(conn: sqlite3.Connection, *, db_label: str, def _wal_reset_repair_hint() -> str: - """Repair hint matching what ``hermes update`` can actually do for this install type.""" + """Repair hint matching what ``hermes update`` can actually do for this install type. + + See #75153. + """ try: from hermes_cli.config import detect_install_method, get_project_root, recommended_update_command_for_method method = detect_install_method(get_project_root()) diff --git a/mcp_serve.py b/mcp_serve.py index 6b0f42f2b3..6209fd7601 100644 --- a/mcp_serve.py +++ b/mcp_serve.py @@ -107,6 +107,10 @@ def _load_sessions_index() -> dict: state.db is primary (gateway session rows carry session_key/origin metadata); sessions.json is the fallback for pre-migration databases without session_keys. + + state.db is the primary source (#9006): gateway sessions persist their routing metadata (session_key, + chat/thread ids, display_name, origin) on the durable session row, so a single database read replaces + the old dual-file sessions.json dependency. """ return _load_sessions_index_from_db() or _load_sessions_index_from_json() @@ -281,6 +285,8 @@ class EventBridge: # Baseline existing history BEFORE polling so startup never replays old # messages as events; sessions appearing later default to last_seen=0.0 # in _poll_once, so new-conversation delivery is preserved. + # Unit tests that drive _poll_once directly bypass start() and still observe first-poll delivery. + # See #13414. self._establish_baseline() self._running = True self._thread = threading.Thread(target=self._poll_loop, daemon=True) @@ -393,6 +399,8 @@ class EventBridge: The routing index lives in the same file as the messages, so a new conversation and its first message land under a single mtime change (no dual-file race that could drop brand-new conversations). + + See #8925, #9006. """ db_mtime = _read_state_db_mtime() if db_mtime == self._state_db_mtime: diff --git a/model_tools.py b/model_tools.py index e1ec34bd6f..59916ae5d1 100644 --- a/model_tools.py +++ b/model_tools.py @@ -153,6 +153,12 @@ discover_builtin_tools() # gateway lazy-imports this module inside its event loop; each entry point # (gateway/run.py, cli.py, tui_gateway, acp_adapter) runs it at startup. try: # plugin tool discovery (user/project/pip plugins) + # MCP tool discovery (external MCP servers from config) used to run here as a module-level side effect. + # It was removed because discover_mcp_tools() internally uses a blocking future.result(timeout=120) + # wait, and the gateway lazy-imports this module from inside the asyncio event loop on the first user + # message — freezing Discord/Telegram heartbeats for up to 120s whenever any configured MCP server was + # slow or unreachable (#16856). - gateway/run.py -> start_gateway() uses run_in_executor - + # acp_adapter/server.py -> asyncio.to_thread on session init from hermes_cli.plugins import discover_plugins discover_plugins() except Exception as e: @@ -192,6 +198,11 @@ _LEGACY_TOOLSET_MAP = { _tool_defs_cache: Dict[tuple, List[Dict[str, Any]]] = {} _tool_defs_cache_lock = threading.Lock() # FIFO cap: 8 covers a long-lived gateway's warm set of platform/toolset combos. +# Hard cap on memoized get_tool_definitions() results. A long-lived Gateway process sees many distinct +# toolset/config fingerprints over its lifetime (per-session toolset sets, config edits, kanban-task +# toggles); without a bound the cache grows unboundedly. 8 comfortably covers the warm working set (the +# handful of distinct platform/toolset combos a gateway actually serves) while keeping the cap small. +# (#19251) _TOOL_DEFS_CACHE_MAX = 8 @@ -216,6 +227,13 @@ def get_tool_definitions(enabled_toolsets: Optional[List[str]] = None, disabled_ if not quiet_mode: return compute() cache_key = _tool_defs_cache_key(enabled_toolsets, disabled_toolsets, skip_tool_search_assembly) + # Cache the freshly-computed list, but hand callers a shallow copy so downstream mutations (e.g. + # run_agent appending memory/LCM tool schemas to self.tools) don't poison the cache. Without this, a + # long-lived Gateway process accumulates duplicate tool names across agent inits and providers that + # enforce unique tool names (DeepSeek, Xiaomi MiMo, Moonshot Kimi) reject the request with HTTP 400. + # Mirrors the cache-hit path above. (issue #17335) Bound the cache with LRU eviction so a long-lived + # Gateway process doesn't accumulate entries unboundedly across the many distinct toolset/config + # fingerprints it sees over its lifetime (#19251). with _tool_defs_cache_lock: cached = _tool_defs_cache.get(cache_key) if cache_key is not None else None if cached is None: @@ -312,6 +330,8 @@ def _select_tool_names(enabled_toolsets: Optional[List[str]], disabled_toolsets: tools.update(resolve_toolset(ts_name)) # Disabled toolsets are always subtracted LAST, so a tool in a disabled # toolset is stripped even when a composite (hermes-cli) re-enables it. + # This ensures that even if a composite toolset (like hermes-cli) is enabled, any tools belonging to a + # disabled toolset are strictly stripped out. See issue #17309. if disabled_toolsets: _apply_toolset_selection(tools, disabled_toolsets, quiet_mode, disable=True) return tools @@ -331,6 +351,8 @@ def _fn_def(schema: Dict[str, Any]) -> Dict[str, Any]: def _rewrite_execute_code(td: Dict[str, Any], available: set) -> Optional[Dict[str, Any]]: """List only sandbox tools that are actually available.""" + # Without this, the model sees "web_search is available in execute_code" even when the API key isn't + # configured or the toolset is disabled (#560-discord). from tools.code_execution_tool import SANDBOX_ALLOWED_TOOLS, build_execute_code_schema, _get_execution_mode return _fn_def(build_execute_code_schema(SANDBOX_ALLOWED_TOOLS & available, mode=_get_execution_mode())) @@ -488,6 +510,9 @@ def _resolve_active_context_length() -> int: if not model_id: return 0 from agent.model_metadata import get_cached_context_length, get_model_context_length + # Honor explicit `model.context_length` in config.yaml — short-circuits the OpenRouter /models probe + # at get_model_context_length step 0, so non-OpenRouter providers don't pay the ~2-3s OpenRouter + # fetch at every CLI startup. See issue #46620. raw_ctx = model_cfg.get("context_length") config_ctx = raw_ctx if isinstance(raw_ctx, int) and raw_ctx > 0 else None provider = str(model_cfg.get("provider") or "").strip() diff --git a/plugins/dashboard_auth/_shared.py b/plugins/dashboard_auth/_shared.py index 9d328eb970..8d17d0b4a0 100644 --- a/plugins/dashboard_auth/_shared.py +++ b/plugins/dashboard_auth/_shared.py @@ -204,6 +204,10 @@ def verify_jwt( try: signing_key = jwks_client.get_signing_key_from_jwt(token) except Exception as exc: + # Unreachable JWKS -> ProviderError (503); a bearer that is not one of our JWTs (opaque peer key, + # foreign kid) -> InvalidCodeError (None / next provider). Folding both into 503 produced #94558. + # Unreachable JWKS -> ProviderError (503); a bearer that is not one of our JWTs (opaque peer key, + # foreign kid) -> InvalidCodeError (None / next provider). Folding both into 503 produced #94558. raise classify_jwks_lookup_error(exc) from exc try: return jwt.decode( diff --git a/plugins/disk-cleanup/disk_cleanup.py b/plugins/disk-cleanup/disk_cleanup.py index 110c5a9bb7..f45e27f982 100755 --- a/plugins/disk-cleanup/disk_cleanup.py +++ b/plugins/disk-cleanup/disk_cleanup.py @@ -96,6 +96,9 @@ _NEVER_TRACK_TOP_LEVEL = frozenset({ "disk-cleanup", "logs", "memories", "sessions", "config.yaml", "skills", "plugins", ".env", "USER.md", "MEMORY.md", "SOUL.md", "auth.json", "hermes-agent", + # User-authored project trees — never sweep empty directories inside these (#75403). + # User-authored and project trees — never auto-delete files inside these just because they happen to be + # named test_* or tmp_* (#75403, also #32164, #37721). "patches", "projects", "skins", "themes", "contributors", "profiles", "backups", "optional-skills"}) @@ -109,6 +112,8 @@ def _protected_cron_paths() -> frozenset: for x in (base, base / "output", base / "jobs.json", base / ".tick.lock")) +# Paths under $HERMES_HOME that must NEVER be deleted by quick(), regardless of what the stored category +# says. This is a defense-in-depth guard against stale tracked.json entries from before #34840. def _is_protected_cron_path(p: Path) -> bool: return str(p.resolve()) in _protected_cron_paths() diff --git a/plugins/image_gen/krea/__init__.py b/plugins/image_gen/krea/__init__.py index ca3b873133..f64b219f14 100644 --- a/plugins/image_gen/krea/__init__.py +++ b/plugins/image_gen/krea/__init__.py @@ -494,6 +494,7 @@ class KreaImageGenProvider(StaticImageGenProvider): # Materialise locally — Krea result URLs may expire. try: + # See #26942. image_ref = str(save_url_image(result_image_url, prefix=f"krea_{model_id}")) except Exception as exc: # noqa: BLE001 logger.warning( diff --git a/plugins/image_gen/openai-codex/__init__.py b/plugins/image_gen/openai-codex/__init__.py index 57f67817bb..1bcee0ae3b 100644 --- a/plugins/image_gen/openai-codex/__init__.py +++ b/plugins/image_gen/openai-codex/__init__.py @@ -27,6 +27,15 @@ from plugins.image_gen._common import ( logger = logging.getLogger(__name__) +# NOTE: do NOT reintroduce an "account capability" classifier keyed on ``Tool choice 'image_generation' not +# found in 'tools' parameter``. That HTTP 400 is a *request-shape* rejection (the Codex backend resolves +# tool_choice as a function-tool name and never recognizes hosted-tool entries) — it is emitted for every +# account, including accounts where image generation works. A previous version of this file translated that +# 400 into "Image generation is not enabled for the current Codex account. Switch the image provider to +# OpenAI API key, FAL, or xAI.", which reported a universal bug in our own payload as the user's entitlement +# problem and sent people away from a provider that was never actually tried. The request-shape bug is fixed +# by omitting tool_choice (see ``_build_responses_payload``); any remaining HTTP error must surface verbatim +# so it stays diagnosable. See issues #19505, #49008 and #31335. _MAX_ERROR_BODY_CHARS = 500 # Hosts the ``image_generation`` tool call; ``API_MODEL`` does the image work. @@ -185,6 +194,13 @@ def _build_responses_payload( "background": "opaque", "partial_images": _PARTIAL_IMAGES_REQUESTED, }], + # No ``tool_choice`` is sent: the chatgpt.com/backend-api/codex backend rejects every shape we have + # for forcing the hosted ``image_generation`` tool. ``{"type": "allowed_tools", "mode": "required", + # "tools": [{"type": "image_generation"}]}`` (and the simpler ``{"type": "image_generation"}`` form) + # both 400 with ``Tool choice 'image_generation' not found in 'tools' parameter`` — the backend + # looks up tool_choice as a *function* name and never recognizes hosted-tool entries. Letting the + # host model decide is the only shape Codex currently accepts; the ``instructions`` above are what + # nudge it toward the tool. See issue #19505. "stream": True, } diff --git a/plugins/kanban/dashboard/plugin_api.py b/plugins/kanban/dashboard/plugin_api.py index d832647d94..6ab43fe1c1 100644 --- a/plugins/kanban/dashboard/plugin_api.py +++ b/plugins/kanban/dashboard/plugin_api.py @@ -648,7 +648,12 @@ def delete_task(task_id: str, board: Optional[str] = Query(None)): def _parents_blocking_ready(conn: sqlite3.Connection, task_id: str) -> list: - """Parent rows (id, title, status) not ``done`` that block promotion to ``ready``.""" + """Parent rows (id, title, status) not ``done`` that block promotion to ``ready``. + + Used to enrich the 409 response from :func:`update_task` so the dashboard can show an actionable toast + (#26744) instead of a silent no-op. Returns ``[]`` when nothing blocks the transition (e.g. no parents, + or all parents already done). + """ rows = conn.execute( "SELECT t.id, t.title, t.status FROM tasks t " "JOIN task_links l ON l.parent_id = t.id " @@ -902,7 +907,11 @@ class TerminateRunBody(BaseModel): @router.post("/runs/{run_id}/terminate") def terminate_run_endpoint(run_id: int, payload: TerminateRunBody, board: Optional[str] = _BOARD_Q): """Terminate an in-flight run via ``reclaim_task`` (same SIGTERM->SIGKILL flow, bookkeeping - and events as ``POST /tasks/{id}/reclaim``); 409 if already ended / not reclaimable.""" + and events as ``POST /tasks/{id}/reclaim``); 409 if already ended / not reclaimable. + + Closes the gap left by PR #28432, which shipped the read-only sibling endpoints (``/workers/active``, + ``/runs/{run_id}``, ``/runs/{run_id}/inspect``) but no termination control surface. + """ with _board_conn(board) as (board, conn): r = _require_run(conn, run_id) if r.ended_at is not None: diff --git a/plugins/memory/hindsight/__init__.py b/plugins/memory/hindsight/__init__.py index 639c436c74..3e33c850f3 100644 --- a/plugins/memory/hindsight/__init__.py +++ b/plugins/memory/hindsight/__init__.py @@ -117,7 +117,10 @@ def _fetch_hindsight_api_version(api_url: str, api_key: str | None = None, def _check_api_supports_update_mode_append(api_url: str, api_key: str | None = None) -> bool: """Cached ``update_mode='append'`` check for *api_url*. False on any probe failure - (safe default: per-process document_id, no update_mode = resume-overwrite fix intact).""" + (safe default: per-process document_id, no update_mode = resume-overwrite fix intact). + + Probes once per URL per process. See #6654. + """ if not api_url: return False with _append_capability_lock: @@ -360,7 +363,12 @@ class HindsightMemoryProvider(MemoryProvider): def unavailable_reason(self) -> str: """Install hint for an unavailable local_embedded runtime (is_available() gates - initialize() out, so the hint it would log never fires; agent_init shows this).""" + initialize() out, so the hint it would log never fires; agent_init shows this). + + ``is_available()`` returns False for local modes when the embedded runtime can't be imported, so + ``initialize()`` — and the hint it would log — is never reached (#7718). Surface the install + guidance here, where agent_init warns about an unavailable provider. + """ try: if _load_config().get("mode", "cloud") not in _LOCAL_MODES: return "" @@ -628,7 +636,14 @@ class HindsightMemoryProvider(MemoryProvider): stable session-scoped id with ``update_mode='append'``; older APIs get *fallback_document_id* (per-process unique) and no update_mode — the only way the resume-overwrite fix works there. The /version probe targets the - embedded client's dynamic per-profile port when running, else api_url.""" + embedded client's dynamic per-profile port when running, else api_url. + + On Hindsight ≥ 0.5.0 the API supports ``update_mode='append'``, which lets us reuse a stable + session-scoped ``document_id`` across process lifecycles without overwriting prior turns. On older + APIs we fall back to *fallback_document_id* (the per-process unique ``f"{session_id}-{start_ts}"`` + minted at initialize / switch time) and don't pass ``update_mode`` at all — that's the only way the + resume-overwrite fix (#6654) keeps working on legacy servers. + """ url = getattr(self._client, "url", None) if self._mode == "local_embedded" else None probe_url = str(url) if url else (self._api_url or "") if self._session_id and _check_api_supports_update_mode_append(probe_url, self._api_key): @@ -777,6 +792,8 @@ class HindsightMemoryProvider(MemoryProvider): logger.warning(msg) # Also print: otherwise the user would only see Hermes get sluggish. with contextlib.suppress(Exception): + # Surface to the terminal too — a daemon that never starts would otherwise fail silently and + # the user would only see Hermes get sluggish. (issue #13125) print(f" ⚠ {msg}", file=sys.stderr, flush=True) self._mode = "disabled" return @@ -882,6 +899,7 @@ class HindsightMemoryProvider(MemoryProvider): def prefetch(self, query: str, *, session_id: str = "") -> str: # Opt-in: recall synchronously against the *current* message so the # injected memories match this turn's query, not the previous turn's. + # See NousResearch/hermes-agent#5820. if self._recall_sync: return self._finish_prefetch(*(("", 0) if self._recall_disabled() else self._do_recall(query))) # Default: the background worker's result for the previous turn (capped join). @@ -941,6 +959,7 @@ class HindsightMemoryProvider(MemoryProvider): relative phrases in content) from the item timestamp: explicit occurred_at wins, else the configured event clock.""" item: Dict[str, Any] = { + # See #93568. "content": content, "metadata": metadata or self._build_metadata(message_count=1, turn_index=self._turn_index), "timestamp": (occurred_at or "").strip() or _event_timestamp(), @@ -1091,7 +1110,16 @@ class HindsightMemoryProvider(MemoryProvider): lose them), join the in-flight prefetch and drop its result (no stale recall for the new session), then set ``_session_id``, mint a fresh ``_document_id`` and clear the batch buffers. ``reset`` is accepted but unneeded: buffer - clearing is correct for every switch.""" + clearing is correct for every switch. + + Without this hook, initialize()-cached state (``_session_id``, ``_document_id``, ``_session_turns``, + ``_turn_counter``) would keep pointing at the previous session and writes would land in the wrong + document. See hermes-agent#6672. + Always update ``_session_id`` so metadata and tags on subsequent retains reflect the active session. + Always clear the accumulated batch buffers (``_session_turns``, ``_turn_counter``, ``_turn_index``) + — even for /resume and /branch, the new session's batching must start from zero so an in-flight + retain doesn't flush under the wrong ``_document_id``. See #1303. + """ new_id = str(new_session_id or "").strip() if not new_id: return @@ -1166,6 +1194,13 @@ class HindsightMemoryProvider(MemoryProvider): # thread, reclaimed at process exit. +# The module-global background event loop (_loop / _loop_thread) is intentionally NOT stopped here. It is +# shared across every HindsightMemoryProvider instance in the process — the plugin loader creates a new +# provider per AIAgent, and the gateway creates one AIAgent per concurrent chat session. Stopping the loop +# from one provider's shutdown() strands the aiohttp ClientSession + TCPConnector owned by every sibling +# provider on a dead loop, which surfaces as the "Unclosed client session" / "Unclosed connector" warnings +# reported in #11923. The loop runs on a daemon thread and is reclaimed on process exit; per-session cleanup +# happens via self._client.aclose() above. def register(ctx) -> None: """Register Hindsight as a memory provider plugin.""" ctx.register_memory_provider(HindsightMemoryProvider()) diff --git a/plugins/memory/hindsight/embedded.py b/plugins/memory/hindsight/embedded.py index 1af6925382..a155c67fc1 100644 --- a/plugins/memory/hindsight/embedded.py +++ b/plugins/memory/hindsight/embedded.py @@ -20,11 +20,26 @@ logger = logging.getLogger(__name__.rpartition(".")[0]) # Read by hindsight_embed.daemon_embed_manager AT IMPORT TIME: how long to wait # for a slow /health before killing the daemon as stale. Busy hosts exceed the # upstream 2s check and get needlessly restarted, so it's plugin config. +# Env var the embedded daemon manager reads (at import time, as a module-level constant) to size the grace +# window it waits for a slow /health before declaring a daemon stale and killing it. We surface it as plugin +# config so users can raise it without hand-setting an env var, consistent with "config.json, not raw env +# vars". See #13125. _PORT_HEALTH_GRACE_ENV = "HINDSIGHT_EMBED_PORT_HEALTH_GRACE_TIMEOUT" # Stale embedded-daemon connection markers (client recreated, operation retried once). _RETRIABLE_CONNECTION_MARKERS = ( "cannot connect to host", + # Connection-establishment / DNS failure message patterns. These surface when the exception TYPE is + # generic (RuntimeError/Exception from a local shim, MCP bridge, subprocess wrapper, or an SDK that + # re-raises without chaining) so the _TRANSPORT_ERROR_TYPES check never fires, and the error carries no + # HTTP status. Without message-level matching they fall through to FailoverReason.unknown, which misses + # the transport eager-fallback path in the retry loop (unknown retries the same dead endpoint for the + # full budget before fallback). Ported from anomalyco/opencode#40707, which hit the same bug shape: + # serialized midstream errors matched by type only. Deliberately EXCLUDES mid-stream disconnect strings + # ("connection reset by peer", "peer closed connection", "unexpected eof", "socket hang up") — those + # belong to _SERVER_DISCONNECT_PATTERNS, whose classification step runs later and routes large sessions + # to context-overflow compression. A connection that was never established cannot be a server-side + # overflow rejection, so these are safe to classify as plain retryable transport. "connection refused", "connect call failed", "clientconnectorerror", @@ -62,7 +77,12 @@ def _check_local_runtime() -> tuple[bool, str | None]: def _local_runtime_hint(reason: str | None) -> str: """Install guidance when the local_embedded runtime is missing: ``plugin.yaml`` declares only ``hindsight-client``, so a hand-written config, the legacy - ``"mode": "local"`` alias or a restored backup hits ``No module named 'hindsight'``.""" + ``"mode": "local"`` alias or a restored backup hits ``No module named 'hindsight'``. + + ``local_embedded`` imports ``from hindsight import HindsightEmbedded``, which is provided only by the + ``hindsight-all`` package (its wheel ships the top-level ``hindsight`` module). + NousResearch/hermes-agent#7718. + """ text = (reason or "").lower() if "no module named" in text and any(m in text for m in ("hindsight'", 'hindsight"', "hindsight_embed")): return ( diff --git a/plugins/memory/hindsight/settings.py b/plugins/memory/hindsight/settings.py index d677d421eb..6655b2a5d3 100644 --- a/plugins/memory/hindsight/settings.py +++ b/plugins/memory/hindsight/settings.py @@ -24,6 +24,8 @@ _DEFAULT_RETAIN_SOURCE = "" _HINDSIGHT_GLYPH = "👁️" # Hindsight 0.5.0 added ``update_mode='append'``; older APIs would silently # overwrite prior turns under a stable document_id, so they keep the per-process id. +# Mirrors hindsight-integrations/openclaw — Hindsight 0.5.0 added `update_mode='append'` semantics on retain +# (vectorize-io/hindsight#932). _MIN_VERSION_FOR_UPDATE_MODE_APPEND = "0.5.0" _VALID_BUDGETS = {"low", "mid", "high"} _PROVIDER_DEFAULT_MODELS = { diff --git a/plugins/memory/holographic/store.py b/plugins/memory/holographic/store.py index 8b8819b742..fa236e80dd 100644 --- a/plugins/memory/holographic/store.py +++ b/plugins/memory/holographic/store.py @@ -265,7 +265,13 @@ class MemoryStore: """Force-close every shared connection whose database lives under ``directory``; returns the count. close() is refcount-driven, so a live holder (e.g. an agent's provider) keeps a profile's SQLite handle open, which on Windows makes rmtree of the profile fail. The directory is going away, so later use by a - stale holder is expected to fail.""" + stale holder is expected to fail. + + That is exactly what a profile delete must break on Windows: the desktop's main ``serve`` process + opens ``memory_store.db`` for every known profile, and ``rmtree`` of the profile directory fails + with ``WinError 32`` while any of those handles is open (#88347). In a process that holds none (e.g. + the CLI deleting from outside serve) this is a harmless no-op returning 0. + """ root = os.path.normcase(str(Path(directory).expanduser().resolve())) + os.sep with cls._shared_guard: doomed = [cls._shared.pop(key) for key in list(cls._shared) if os.path.normcase(key).startswith(root)] @@ -290,6 +296,7 @@ class MemoryStore: finally: # Pop only OUR entry: after release_all_under() a same-path store may have # registered a FRESH entry under this key; a stale late close() must not evict it. + # See #88347. if MemoryStore._shared.get(self._key) is entry: MemoryStore._shared.pop(self._key, None) self._entry = None diff --git a/plugins/memory/honcho/client.py b/plugins/memory/honcho/client.py index 2e1707a39d..2149de742a 100644 --- a/plugins/memory/honcho/client.py +++ b/plugins/memory/honcho/client.py @@ -13,6 +13,10 @@ import ipaddress import json import logging import os +# --- per-identity client cache ------------------------------------------- One slot per client identity, +# replacing the single process-wide slot that pinned the first profile's workspace and bearer for every +# later profile in multi-profile processes (#69123 multiplexed gateway, #74065 dashboard). The legacy names +# above are retained only for reset bookkeeping. import threading as _threading from dataclasses import dataclass, field from pathlib import Path @@ -336,6 +340,9 @@ class HonchoClientConfig: ai_peer: str = "hermes" # True: peer_name wins over gateway runtime identity (Telegram UID, ...), so a # single-user deployment keeps one memory across platforms. + # This keeps memory unified across platforms for single-user deployments where Honcho's one peer-name is + # an unambiguous identity — otherwise each platform would fork memory into its own peer (#14984). + # Default ``False`` preserves existing multi-user behaviour. pin_peer_name: bool = False # Gateway runtime user id -> stable Honcho peer; host map replaces root map. user_peer_aliases: dict[str, str] = field(default_factory=dict) @@ -382,6 +389,11 @@ class HonchoClientConfig: explicitly_configured: bool = False # Provenance captured at resolution time; bound consumers use these instead of # re-resolving (the resolvers read a ContextVar background threads can't see). + # Provenance: WHERE this config was resolved from, captured at resolution time (inside the caller's + # profile scope). Bound consumers (session manager, OAuth refresh paths) use these instead of + # re-resolving resolve_config_path()/get_hermes_home() later — those resolvers read a ContextVar that + # background threads cannot see, so re-resolution from a daemon thread silently lands on the DEFAULT + # profile (#69123, #74065). config_path: Path | None = None hermes_home: Path | None = None @@ -503,7 +515,10 @@ def get_honcho_client(config: HonchoClientConfig | None = None) -> Honcho: IDENTITY (host, workspace, provenance paths, credential fingerprint, timeout) so multi-profile processes don't share a first-config-wins client. With no config the active honcho.json is resolved — correct only on threads that see the profile ContextVar; pass a - bound config elsewhere. Each identity's client is built once under concurrent first calls.""" + bound config elsewhere. Each identity's client is built once under concurrent first calls. + + See #69123, #74065. + """ key = _client_cache_key(config) slot = _slot_for(key) cached = slot.peek() diff --git a/plugins/memory/honcho/client_cache.py b/plugins/memory/honcho/client_cache.py index 223bc6f848..6af5549ec9 100644 --- a/plugins/memory/honcho/client_cache.py +++ b/plugins/memory/honcho/client_cache.py @@ -103,7 +103,11 @@ def _slot_identity(key: tuple) -> tuple: def _slot_for(key: tuple) -> SingletonSlot: """Slot for ``key``, evicting stale same-identity slots: a same (kind, host, paths) identity with a different credential/timeout drops the old slot so the replaced client stops being - served; otherwise credential churn leaks one pinned client per change.""" + served; otherwise credential churn leaks one pinned client per change. + + Without eviction, credential churn leaks one pinned client per change — the gap that made #81401's + retirement machinery inert. + """ identity = _slot_identity(key) with _client_slots_lock: slot = _client_slots.get(key) diff --git a/plugins/memory/honcho/session.py b/plugins/memory/honcho/session.py index 288291ddb3..44e80d8e70 100644 --- a/plugins/memory/honcho/session.py +++ b/plugins/memory/honcho/session.py @@ -101,7 +101,10 @@ class HonchoSessionManager(SessionAuthMixin, SessionPeersMixin, SessionContextMi """The Honcho client, refreshing a near-expiry OAuth token in place. Always goes through ``get_honcho_client`` WITH this manager's bound config: a long session can't outlive its 1h access token, and daemon threads can't see the ambient ContextVar profile, so a bare - ``get_honcho_client()`` would migrate them onto the first-built profile.""" + ``get_honcho_client()`` would migrate them onto the first-built profile. + + See #69123, #74065. + """ self._honcho = get_honcho_client(self._config) return self._honcho @@ -218,6 +221,7 @@ class HonchoSessionManager(SessionAuthMixin, SessionPeersMixin, SessionContextMi # Gateway sessions normally use the platform-native runtime identity so multi-user # bots scope memory per user; config can alias/prefix it, or pinPeerName pins all # identities to peerName for single-user deployments (see _resolve_user_peer_id). + # Determine peer IDs — no lock needed (read-only, no shared state mutation). See #14984. user_peer_id = self._resolve_user_peer_id(key) assistant_peer_id = self._sanitize_id(self._config.ai_peer if self._config else "hermes-assistant") diff --git a/plugins/memory/honcho/session_context.py b/plugins/memory/honcho/session_context.py index 4e2b0318ca..e5705691cb 100644 --- a/plugins/memory/honcho/session_context.py +++ b/plugins/memory/honcho/session_context.py @@ -331,7 +331,17 @@ class SessionContextMixin: ``reasoning_level`` is honored only when dialecticDynamic is true. ``apply_injection_cap`` clips to ``dialecticMaxChars`` (automatic injection only). ``raise_errors`` re-raises backend failures instead of returning "" so explicit tool calls can tell a timeout from an empty answer. - Raises HonchoAuthError when credentials are rejected after a forced refresh and one retry.""" + Raises HonchoAuthError when credentials are rejected after a forced refresh and one retry. + + Args: session_key: The session key to query against. query: Natural language question. + reasoning_level: Override the configured default (dialecticReasoningLevel). If None or + dialecticDynamic is false, uses the configured default. peer: Which peer to query — "user" (default) + or "ai". apply_injection_cap: Clip automatic injections to ``dialecticMaxChars``. Explicit + ``honcho_reasoning`` calls pass False because Honcho already bounds their output. raise_errors: + Re-raise backend failures instead of returning "". Explicit tool calls pass True so a timeout or + server error surfaces as an error, not as "no result" (#36098 issue 4: collapsing failures to "" + made auth errors, timeouts, and genuinely-empty answers indistinguishable). + """ session = self._cache.get(session_key) target_peer_id = self._resolve_peer_id(session, peer) if session else None if target_peer_id is None: diff --git a/plugins/memory/openviking/__init__.py b/plugins/memory/openviking/__init__.py index 227c55fbf6..b1f0caf3f3 100644 --- a/plugins/memory/openviking/__init__.py +++ b/plugins/memory/openviking/__init__.py @@ -334,7 +334,11 @@ class _VikingClient: def health_payload(self) -> dict: """``GET /health``, anonymous first so credentials never reach an unknown host. Hosted OpenViking requires auth on /health: when an API key is configured and the - anonymous call gets 401/403, retry once with the key (no tenant headers).""" + anonymous call gets 401/403, retry once with the key (no tenant headers). + + Prefer an anonymous probe so credentials are never sent to an unknown host during identity checks. + See #78410. + """ try: return self._anonymous_json("/health") except _OpenVikingHTTPError as exc: @@ -969,6 +973,10 @@ def _start_local_openviking_server(endpoint: str) -> tuple[str, str]: # Strip PYTHONPATH: the Desktop backend puts the Hermes venv on it, which # would shadow openviking-server's own site-packages (and on Windows lock # the Hermes venv's .pyd files, breaking `hermes update`). + # Do not let the server child inherit this process's PYTHONPATH. If inherited, openviking-server + # would import aiohttp and friends from the Hermes venv instead of its own (its venv's site-packages + # are shadowed because PYTHONPATH precedes them) — and on Windows the loaded DLLs then lock the + # Hermes venv, aborting `hermes update` with access-denied on .pyd files. (#78153) child_env = os.environ.copy() child_env.pop("PYTHONPATH", None) with log_path.open("ab") as log_file: @@ -1206,11 +1214,16 @@ class OpenVikingMemoryProvider(MemoryProvider): self._session_id, self._turn_count, self._hermes_home = "", 0, "" # (conn snapshot, user): keyed on the snapshot so every client built from it # shares the resolved user and a /reload invalidates it. + # Server-asserted user space for explicit-uid URIs (#91995). Key the cache on the connection + # snapshot so all clients built from the same snapshot share the resolved user. /reload can swap + # endpoint, credentials, and identity on this provider instance — a different snapshot invalidates + # the cache automatically. self._user_space_cache: Optional[tuple[Any, str]] = None self._run_id = uuid.uuid4().hex self._run_lock_file = self._run_lock_path = None # Until initialize() resolves the baseline, _ensure_client() must not # re-resolve from the environment (a hand-wired test client would be discarded). + # Set once initialize() has resolved the connection baseline. See #21130. self._env_refresh_enabled = False # _session_state_lock guards (_session_id, _turn_count): sync_turn increments on the # sync executor while on_session_end/_switch snapshot+reset on the caller thread. @@ -1221,6 +1234,10 @@ class OpenVikingMemoryProvider(MemoryProvider): (self._session_state_lock, self._inflight_lock, self._deferred_commit_lock, self._committed_session_lock, self._client_refresh_lock, self._runtime_start_lock, self._memory_write_lock) = (threading.Lock() for _ in range(7)) # Writers keyed by the sid they POST under so a commit can drain all of them. + # Guards the (_session_id, _turn_count) pair. sync_turn runs on the MemoryManager's background sync + # executor while on_session_end / on_session_switch run on the caller's thread, so the + # snapshot+reset of the turn counter and the session-id rotation must be atomic against a concurrent + # increment. See hermes-agent#28296 review. self._inflight_writers: Dict[str, Set[threading.Thread]] = {} self._deferred_commit_sids: Set[str] = set() self._deferred_commit_threads: Set[threading.Thread] = set() @@ -1388,6 +1405,7 @@ class OpenVikingMemoryProvider(MemoryProvider): # Baseline established — set here, not at the end, so an exception in the # connection attempt (swallowed by MemoryManager) can't leave the provider # stuck in never-refresh mode. + # See #21130. self._env_refresh_enabled = True self._session_id = session_id self._turn_count = 0 @@ -1425,6 +1443,10 @@ class OpenVikingMemoryProvider(MemoryProvider): ``/reload`` only refreshes ``os.environ``; the provider instance is not re-initialized, so re-resolve settings on every access and rebuild + health-check only when a value changed (hot path: one tuple compare). + + ``/reload`` only refreshes ``os.environ`` — the existing provider instance is not re-initialized — + so OPENVIKING_* values added to ``~/.hermes/.env`` after startup never reach the live client and + tools keep running against stale auth until the user restarts hermes (#21130). """ if not self._env_refresh_enabled: return self._client # no baseline yet: keep whatever the caller wired up @@ -2349,6 +2371,9 @@ class OpenVikingMemoryProvider(MemoryProvider): ``_session_id`` stays stuck at the initialize() value, later sync_turn writes land in the closed session and the new one never gets extracted. The old session's drain+commit is offloaded so command threads never block. + + The new session never accumulates messages, and memory extraction never fires for it. See + hermes-agent#28296. """ new_id = str(new_session_id or "").strip() if not new_id or not self._ensure_client(): @@ -2358,6 +2383,12 @@ class OpenVikingMemoryProvider(MemoryProvider): # Rotate under the lock so a concurrent sync_turn lands fully under old or new. with self._session_state_lock: + # Rotate cached session state synchronously (cheap, in-memory) and snapshot the old session + # under the lock so a concurrent sync_turn either lands fully before the rotation (counted under + # old) or fully after (counted under new) — never split. The OLD session's commit (drain + + # pending-token GET + commit POST, potentially many seconds) is then offloaded so /new, /branch, + # /resume, /undo never block the caller's command thread (cf. the end-of-turn-sync offload in + # #41945). old_session_id = self._session_id old_turn_count = self._turn_count rotate = not (rewound or new_id == old_session_id) @@ -2395,6 +2426,9 @@ class OpenVikingMemoryProvider(MemoryProvider): reload mid-write can't borrow a later peer; an empty peer there is intentional. getattr(): hand-wired providers (``__new__``) may lack ``_client`` / ``_agent``. """ + # Explicit-uid URIs are canonical under every auth mode; the uid-less `viking://user/peers/...` + # shorthand was removed upstream (#4196) and `viking://~/...` only expands for USER/ADMIN roles, not + # dev/ROOT. active_client = client if client is not None else getattr(self, "_client", None) agent = str(getattr(active_client, "_agent", getattr(self, "_agent", "")) or "").strip() peer_prefix = f"peers/{agent}/" if agent else "" diff --git a/plugins/model-providers/copilot/__init__.py b/plugins/model-providers/copilot/__init__.py index 37cec65951..98e9c54431 100644 --- a/plugins/model-providers/copilot/__init__.py +++ b/plugins/model-providers/copilot/__init__.py @@ -32,6 +32,7 @@ class CopilotProfile(ProviderProfile): # Honor a level the live catalog lists; otherwise clamp to the nearest WEAKER # supported level (never drop straight to medium, which inverted the ladder: # ultra < high). Bespoke levels the ladder can't place fall to medium (or [0]). + # See #74295. if effort not in supported: effort = clamp_reasoning_effort_to_supported(effort, list(supported)) if effort not in supported: diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index ea186482cc..bdc77ed333 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -45,6 +45,7 @@ class CustomProfile(ProviderProfile): if reasoning_config and isinstance(reasoning_config, dict): effort = (reasoning_config.get("effort") or "").strip().lower() if effort == "none" or reasoning_config.get("enabled", True) is False: + # See #14820. top_level["reasoning_effort"] = "none" if _looks_like_ollama_endpoint(ctx.get("base_url")): extra_body["think"] = False @@ -67,6 +68,9 @@ custom = CustomProfile( base_url="", # User-configured # Floor only (user model.max_tokens overrides); without it Ollama falls # back to num_predict=128 and truncates. + # Without this, no max_tokens is sent and Ollama falls back to its internal num_predict=128, truncating + # responses after a few tokens (#39281). This is only a floor used when the user hasn't set + # model.max_tokens — they can override per-model — so we set it generously rather than lowballing it. default_max_tokens=65536, ) diff --git a/plugins/model-providers/meta-ai/__init__.py b/plugins/model-providers/meta-ai/__init__.py index 439a9fa4bc..17ab0c5163 100644 --- a/plugins/model-providers/meta-ai/__init__.py +++ b/plugins/model-providers/meta-ai/__init__.py @@ -67,6 +67,7 @@ meta_ai = MetaAIProfile( # Natively multimodal, but only on user turns: an image envelope inside a role:tool # message 400s "content did not match any supported type". supports_vision=True, supports_vision_tool_messages=False, + # See #101668. default_aux_model="muse-spark-1.2-contributor", # Muse spends completion budget on hidden reasoning first; low caps can finish with empty content. default_max_tokens=16384, diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index 44bdd4b4f1..51bf812fc9 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -138,6 +138,20 @@ class OpenRouterProfile(ProviderProfile): # turn without a replayed thinking block) makes OpenRouter emit # ``thinking: {type: "disabled"}`` -> 400. Omit it; the user's effort # still reaches Anthropic's output_config.effort via top-level ``verbosity``. + # Reasoning-mandatory Anthropic models (Claude 4.6+ / fable / future named models) use + # *adaptive* thinking: the model decides how much to think, and OpenRouter ignores + # ``reasoning.effort`` for them entirely. Sending any ``reasoning`` field is therefore both + # pointless and actively harmful: - any enabled form, on a tool-continuation turn whose prior + # assistant tool_call carries no thinking block (chat_completions never replays signed thinking + # blocks), ALSO makes OpenRouter emit ``thinking: {type: "disabled"}`` → the same 400 on every + # turn after the first tool call. See hermes-agent#42991 (disable case) and the tool-replay + # follow-up. ``reasoning.effort`` being ignored does NOT mean these models have no effort lever + # — OpenRouter honors the requested effort on the top-level ``verbosity`` field instead (it maps + # to Anthropic's ``output_config.effort``; ``reasoning.effort`` is accepted but ignored — + # confirmed by OpenRouter's Claude migration docs and a live token-spend probe in + # hermes-agent#43432). Route the existing ``reasoning_config["effort"]`` (sourced from + # ``agent.reasoning_effort``) onto ``verbosity`` so the knob the user already sets keeps working + # for these models. if _anthropic_reasoning_is_mandatory(model): cfg = reasoning_config or {} effort = cfg.get("effort") diff --git a/plugins/model-providers/upstage/__init__.py b/plugins/model-providers/upstage/__init__.py index f054e79b9e..cedd34c267 100644 --- a/plugins/model-providers/upstage/__init__.py +++ b/plugins/model-providers/upstage/__init__.py @@ -29,6 +29,9 @@ class UpstageProfile(ProviderProfile): return {}, {"reasoning_effort": "medium"} # unset -> reasoning ON for agents if reasoning_config.get("enabled") is False: return {}, {} # explicitly disabled -> Solar's own default (minimal = off) + # Map Hermes' effort vocabulary onto Solar's accepted set via the shared clamp + # (agent.reasoning_effort). minimal → omit (Solar's minimal means off); unknown-but-enabled bespoke + # levels collapse to high rather than silently downgrading (#62650 precedent). effort = (reasoning_config.get("effort") or "").strip().lower() if not effort: return {}, {"reasoning_effort": "medium"} diff --git a/plugins/platforms/a2a/adapter.py b/plugins/platforms/a2a/adapter.py index 5e85344b42..a9e8b1100c 100644 --- a/plugins/platforms/a2a/adapter.py +++ b/plugins/platforms/a2a/adapter.py @@ -168,7 +168,10 @@ class A2ARequestHandler(BaseHTTPRequestHandler): return self.client_address[0] if self.client_address else "" def _request_public_url(self) -> str: - """A2A_PUBLIC_URL > X-Forwarded-Host / Host (scheme from X-Forwarded-Proto) > "" (bind host).""" + """A2A_PUBLIC_URL > X-Forwarded-Host / Host (scheme from X-Forwarded-Proto) > "" (bind host). + + Empty means "caller has no info, fall back to bind host". See #41711. + """ explicit = os.getenv("A2A_PUBLIC_URL", "").strip() if explicit: return explicit @@ -249,6 +252,9 @@ class A2AAdapter(BasePlatformAdapter): extra = getattr(config, "extra", {}) or {} # Scope-aware: a secondary multiplex profile must not borrow the default profile's bridged # A2A_PORT (falls closed to the module default). advertised_toolsets is deliberately unscoped. + # (advertised_toolsets has the same env-leak shape but is left unscoped here — see the "Scope note" + # in this fix's PR description: open PR #98937 is actively rewriting this field's None-vs-empty-list + # semantics.) self._security_context = security.A2ASecurityContext.capture() _port_env = None if _profile_scoped() else os.getenv("A2A_PORT") self.port = int(_port_env or extra.get("port", _DEFAULT_PORT)) @@ -281,7 +287,11 @@ class A2AAdapter(BasePlatformAdapter): @property def authorization_is_upstream(self) -> bool: """Requests are authenticated in ``do_POST``; the gateway's A2A_ALLOWED_USERS list would - otherwise reject peers (identity is a token-derived name or IP). Wrong credentials still 401.""" + otherwise reject peers (identity is a token-derived name or IP). Wrong credentials still 401. + + This is authorization delegated to the A2A bearer-token transport, not a fail-open: every request is + 401'd if the credential is wrong. Reported by kuangmi-bit (PR #41711 comment, Jun 27). + """ return True async def connect(self, **_kwargs) -> bool: diff --git a/plugins/platforms/a2a/tools.py b/plugins/platforms/a2a/tools.py index b0753595fe..1190b8f1dc 100644 --- a/plugins/platforms/a2a/tools.py +++ b/plugins/platforms/a2a/tools.py @@ -331,7 +331,13 @@ _TOOLS: dict[str, tuple[Any, str, dict, list[str]]] = { def _a2a_tools_available() -> bool: """check_fn: serve the client tools ONLY when the operator opted into A2A (peers under - ``a2a_agents``, inbound platform enabled, or A2A_PORT set). Fail closed.""" + ``a2a_agents``, inbound platform enabled, or A2A_PORT set). Fail closed. + + Maintainer-directed (#95681): these registered unconditionally, so every session on every install paid + ~561 tok/call for tools whose only possible output without config is 'no peers configured'. A2A is + unrelated to Bot Mode (bots talk over gateway RPCs) — for most installs this toolset is foreign-agent + plumbing they never enabled. Config adds mid-session surface at the next compaction (#97073). + """ cfg = {} with contextlib.suppress(Exception): cfg = _load_config() diff --git a/plugins/platforms/buzz/adapter.py b/plugins/platforms/buzz/adapter.py index 0f151330ff..3a8328d0af 100644 --- a/plugins/platforms/buzz/adapter.py +++ b/plugins/platforms/buzz/adapter.py @@ -24,6 +24,9 @@ from pathlib import Path from typing import Any, Dict, List, Optional, Tuple from urllib.parse import urlsplit, urlunsplit +# Profile-scoped read (adapter startup, Slack pattern #59739): a scoped read honors the profile's own +# secret; only an UNSCOPED read under multiplex (default-profile startup loop) falls back to the process +# env, which is that profile's own value. from agent.secret_scope import ( UnscopedSecretError as _UnscopedSecretError, current_secret_scope as _current_secret_scope, get_secret as _scoped_get_secret, is_multiplex_active as _is_multiplex_active, @@ -34,10 +37,112 @@ from gateway.platforms._shared import profile_scoped as _profile_scoped def _get_scoped_secret(name, default=None): """Scope-aware credential read: an active scope is authoritative (a miss is ``default``, never an env borrow). Unscoped adds one rung over ``_shared``: the startup gate runs before any scope exists, so - externally managed secrets are consulted via a one-shot profile-scope build.""" + externally managed secrets are consulted via a one-shot profile-scope build. + + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and connects *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash + startup/reconnect (#70652 class); there ``os.environ`` is that profile's own value, so fall back to it. + Same pattern as ``whatsapp_common._get_wsecret`` and the WeCom/IRC/ntfy plugin adapters. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + Secondary profiles construct their adapters under a profile secret scope -- the scope is authoritative + and a scoped miss returns ``default`` (no cross-profile borrow from ``os.environ``, which may hold + another profile's value). The DEFAULT profile's adapter constructs and sends *unscoped* under + multiplexing, where a bare ``get_secret`` would raise ``UnscopedSecretError`` and crash this path; there + ``os.environ`` is that profile's own value, so fall back to it. Same pattern as the Slack + ``SLACK_APP_TOKEN`` read (#59739) and ``gateway/platforms/whatsapp_common.py::_get_wsecret``. + """ try: val = _scoped_get_secret(name, None) except _UnscopedSecretError: + # DEFAULT profile's adapter constructs/connects outside any _profile_runtime_scope under + # multiplexing; os.environ is that profile's own value there. Same pattern as Slack SLACK_APP_TOKEN + # (#59739) and the Matrix recovery key. A *scoped* miss still returns the default (no cross-profile + # borrow). val = os.getenv(name) if val is None and _current_secret_scope() is None: val = _unscoped_profile_secrets().get(name) @@ -67,7 +172,15 @@ def _unscoped_profile_secrets() -> Dict[str, str]: def _scoped_platform_setting(env_name, extra, key): - """Raw non-secret setting; in a secondary profile scope ``os.environ`` is the DEFAULT profile's, so ``extra`` wins.""" + """Raw non-secret setting; in a secondary profile scope ``os.environ`` is the DEFAULT profile's, so ``extra`` wins. + + Inside a secondary profile scope ``os.environ`` holds the DEFAULT profile's YAML-to-env bridge output + (#98738), so the profile's ``PlatformConfig.extra`` is authoritative and env is not consulted: a missing + key yields ``None`` and callers fail closed to their default instead of silently borrowing the default + profile's relay, channels, or allowlist. Everywhere else — single-profile gateways, the default profile + under multiplexing — the legacy ``os.getenv`` read is returned unchanged, so env-over-config precedence + is preserved. + """ return (extra or {}).get(key) if _profile_scoped() else os.getenv(env_name) @@ -82,6 +195,10 @@ from gateway.config import Platform _CHAT_KIND = 9 # ``messages get`` also returns housekeeping kinds, never dispatched # Chat + forum post/comment; stream kinds wait for confirmed semantics. ``_is_direct_message_event`` # stays kind-9-only so a p-tagged forum post can't be reclassified as a DM and bypass mention gating. +# Kinds that carry agent-relevant conversation content and are dispatched (#90309): chat messages (9) plus +# the Buzz forum kinds — 45001 is a forum post (thread root) and 45003 a comment reply on it. Block's own +# ACP harness documents this set (``buzz-acp --kinds 9,46010,40007,45001, 45002,45003``); the stream kinds +# (46010/40007/45002) are left out until their dispatch semantics are confirmed. _DISPATCH_KINDS = frozenset({_CHAT_KIND, 45001, 45003}) _UNRESOLVED_MENTION_ERROR_RE = re.compile(r"mention '@(?P<name>[^']+)' does not match a current channel member") _BUZZ_PRESENTATION_MENTION_SEPARATOR = "\u200b" @@ -149,6 +266,9 @@ def _attachment_origin(value: str) -> Optional[tuple[str, int]]: # WebSocket transport (NIP-42 authenticated Nostr subscription). _WS_AUTH_TIMEOUT = 20.0 # Last-resort read bound: an unsurfaced relay-side close (CLOSE_WAIT) would leave us "connected" with inbound stopped. +# The library keepalive (ping_interval/ping_timeout below) should catch a dead relay first, but a relay-side +# close the transport never surfaces (observed as a CLOSE_WAIT socket with the loop parked on recv, #98097) +# leaves the gateway "connected" while inbound stops; this timeout forces the normal reconnect path instead. _WS_READ_IDLE_TIMEOUT = 300.0 _WS_MAX_MESSAGE_BYTES = 2_000_000 _WS_MEMBERSHIP_KIND = 44100 # Buzz channel-membership event — live DM discovery @@ -547,6 +667,10 @@ class BuzzAdapter(BasePlatformAdapter): self._membership_since = self._poll_count = 0 self._lock_key: Optional[str] = None # Channels the relay permanently rejected ("restricted"); persists across reconnects so we never re-subscribe. + # channel_id -> { "chat_type", "last_ts", "seen": OrderedDict[event_id, None], "event_meta": + # OrderedDict[event_id, (author_pubkey, content_snippet)], } event_meta backs NIP-10 reply-parent + # resolution for require_mention (thread replies to our own messages count as addressed — #75826). + # "restricted: not a channel member"). self._restricted_channels: set = set() # channel_id -> {"chat_type", "last_ts", "seen": OrderedDict[event_id, None], "event_meta": # OrderedDict[event_id, (author_pubkey, snippet)]}; event_meta backs NIP-10 reply-parent resolution. @@ -568,7 +692,12 @@ class BuzzAdapter(BasePlatformAdapter): @staticmethod def normalize_user_id(user_id: str) -> Optional[str]: - """Normalize a user reference (hex or npub) to hex — authz_mixin allowlist hook.""" + """Normalize a user reference (hex or npub) to hex — authz_mixin allowlist hook. + + Optional hook consumed by ``gateway/authz_mixin`` when matching the profile allowlist carried in + ``config.extra.allowed_users`` (#98738): entries may be npubs while inbound ``user_id`` is always + the hex pubkey, so a plain string compare would deny listed users. + """ return _normalize_user_ref(user_id) # ── buzz-cli plumbing ───────────────────────────────────────────────── @@ -658,6 +787,8 @@ class BuzzAdapter(BasePlatformAdapter): ) # Seed high-water marks so a (re)start never replays history — except where a restored cursor lets # events that landed while down still dispatch. + # Skip any channel the relay has permanently rejected in a previous session (e.g. "restricted: not a + # channel member") so we don't reconnect-loop on them. See #90464. self._load_cursors() for channel_id in watch: if channel_id in self._restricted_channels: @@ -789,7 +920,17 @@ class BuzzAdapter(BasePlatformAdapter): async def _run_message_send(self, args: List[str], content: str, mention_pubkeys: Optional[List[str]] = None): """Send with bounded recovery (each rung once): explicit ``--mention``s; on "not channel members" retry - without; escape an unresolvable ``@token`` and retry; finally ``--mention <self>`` (downgrades @names to text).""" + without; escape an unresolvable ``@token`` and retry; finally ``--mention <self>`` (downgrades @names to text). + + 1. publish with explicit ``--mention`` pubkeys resolved from the content (#83414) so genuine member + mentions carry p-tags and mention-subscribed agents actually wake; 2. if the CLI rejects because a + resolved pubkey is no longer a member (membership drift), retry without the explicit mentions — + deliver the message rather than lose it; 3. if the CLI's preflight rejects an unresolvable + presentation ``@token`` in prose, escape exactly that token with an invisible separator and retry + (#82646 / #78797); 4. if the error persists and we know our own pubkey, retry once with ``--mention + <self>`` — supplying any explicit identity downgrades unresolvable @names to presentation-only text + (#83414); the echo de-dupe already suppresses self-notification. + """ mention_args: List[str] = [] for pk in mention_pubkeys or []: mention_args += ["--mention", pk] @@ -838,6 +979,9 @@ class BuzzAdapter(BasePlatformAdapter): if receipt_error: return SendResult(success=False, error=receipt_error) assert event_id is not None + # Belt-and-braces echo suppression: the poll loop already skips our own pubkey, but marking the + # verified id seen makes de-dupe explicit. Also record event_meta so a thread reply to this send + # matches even if the WS/poll echo never arrives (#75826). self._mark_seen(str(chat_id), event_id) return SendResult(success=True, message_id=event_id) @@ -901,7 +1045,10 @@ class BuzzAdapter(BasePlatformAdapter): self, chat_id: str, file_path: Path, *, caption: Optional[str] = None, reply_to: Optional[str] = None, metadata: Optional[Dict[str, Any]] = None, probe: bool = True, ) -> SendResult: - """Upload a local file as a native attachment; ``probe=False`` when the caller already verified it (a re-probe could race).""" + """Upload a local file as a native attachment; ``probe=False`` when the caller already verified it (a re-probe could race). + + See #74999. + """ local = Path(file_path).expanduser() if probe and not local.is_file(): # Never leak host filesystem paths into chat-visible errors. @@ -914,7 +1061,10 @@ class BuzzAdapter(BasePlatformAdapter): async def send_image_file( self, chat_id: str, image_path: str, caption: Optional[str] = None, reply_to: Optional[str] = None, metadata: Optional[Dict[str, Any]] = None, **kwargs) -> SendResult: - """Upload a local image via ``--file``; missing paths keep the Base fallback so host paths never reach chat.""" + """Upload a local image via ``--file``; missing paths keep the Base fallback so host paths never reach chat. + + See #74999. + """ local = Path(image_path).expanduser() if local.is_file(): return await self._send_file_attachment(chat_id, local, caption=caption, reply_to=reply_to, metadata=metadata, probe=False) @@ -949,6 +1099,10 @@ class BuzzAdapter(BasePlatformAdapter): # ── Inbound: WebSocket transport (NIP-42) — same _handle_event() as the poll loop ────── + # ── Inbound: WebSocket transport (NIP-42 authenticated) ────────────── Push transport contributed in PR + # #73636 by @ScaleLeanChris, adapted to dispatch through the same _handle_event() machinery as the poll + # loop so de-dupe, mention gating, DM latching, and the allow-list behave identically on both + # transports. def _websocket_url(self) -> str: parsed = urlsplit(self.relay_url.strip()) scheme = {"http": "ws", "https": "wss"}.get(parsed.scheme, parsed.scheme) @@ -983,6 +1137,12 @@ class BuzzAdapter(BasePlatformAdapter): raise ConnectionError("Buzz relay did not send a NIP-42 AUTH challenge") # BUZZ_AUTH_TAG is per-identity: a scoped profile without one fails closed to "" rather than borrowing # the default profile's env tag. Resolved lazily so a re-auth on a bare adapter stays scope-correct. + # BUZZ_AUTH_TAG is per-identity NIP-OA owner attestation, so it must resolve through the profile + # secret scope (#98738): inside a scoped multiplex profile a missing tag fails closed to "" instead + # of attaching the default profile's tag from os.environ, while single-profile and unscoped + # default-profile reads keep the legacy env behavior. connect() populates ``self._auth_tag`` via + # ``_resolve_auth_tag`` (scope-aware read + credentials-file fallback, #79514); resolve lazily here + # as well so a re-auth on a bare adapter stays scope-correct. auth_tag = getattr(self, "_auth_tag", "") or "" if not auth_tag: try: @@ -1009,6 +1169,11 @@ class BuzzAdapter(BasePlatformAdapter): async def _send_channel_subscription(self, websocket, subscription_id: str, channel_id: str) -> None: state = self._channel_state.get(channel_id) or {} last_ts = int(state.get("last_ts") or 0) + # A conversation adopted mid-run with no high-water mark is fresh: its history IS the conversation, + # so subscribe from the beginning instead of `since ≈ now` — otherwise the message that *created* + # the conversation (created_at fractionally before this subscription) is silently dropped (#78429). + # `limit` bounds the replay to the same window the poll transport fetches; the seed path gives real + # channels a non-zero last_ts, so they never take this branch. request_filter = {"kinds": sorted(_DISPATCH_KINDS), "#h": [channel_id]} if last_ts: # Resume from the high-water mark (same-second overlap de-duped by id). @@ -1047,7 +1212,14 @@ class BuzzAdapter(BasePlatformAdapter): async def _ws_discovery_loop(self, websocket, subscriptions: Dict[str, Optional[str]]) -> None: """Periodic discovery on the poll cadence: relays don't guarantee a kind-44100 event for every new - conversation. Failures retry next tick; the read loop alone owns connection health.""" + conversation. Failures retry next tick; the read loop alone owns connection health. + + The kind-44100 membership subscription is the fast path, but relays do not guarantee a membership + event for every conversation that materializes mid-session (#93557) — some emit none at all for new + DM-shaped conversations. The poll transport papers over this by re-running discovery every + ``_DM_DISCOVERY_EVERY`` sweeps; this loop gives the WS transport the same guarantee on the same + cadence. + """ interval = max(self.poll_interval * _DM_DISCOVERY_EVERY, _MIN_POLL_INTERVAL) while True: await asyncio.sleep(interval) @@ -1212,7 +1384,12 @@ class BuzzAdapter(BasePlatformAdapter): return int(state.get("last_ts") or 0), len(seen), (next(reversed(seen), None) if seen else None) def _restore_channel_state(self, channel_id: str, chat_type: str) -> bool: - """Install a persisted cursor (True when one existed): seeding would mark downtime arrivals as seen.""" + """Install a persisted cursor (True when one existed): seeding would mark downtime arrivals as seen. + + Restoring is what closes the restart gap: seeding from current history instead would mark everything + that arrived while the gateway was down as already seen, so the relay's durable copy is never + dispatched (#90464). + """ restored = self._restored_cursors.pop(channel_id, None) if restored is None: return False @@ -1238,6 +1415,7 @@ class BuzzAdapter(BasePlatformAdapter): state["seen"][str(event_id)] = None state["last_ts"] = max(state["last_ts"], int(event.get("created_at") or 0)) # History is never dispatched but feeds event_meta (post-restart replies to us must match) and latches DMs. + # See #75826. self._remember_event(state, event) self._maybe_latch_dm(channel_id, state, event) self._trim_seen(state) @@ -1245,7 +1423,11 @@ class BuzzAdapter(BasePlatformAdapter): async def _discover_dms(self, *, seed: bool) -> None: """Watch DMs: startup ones are seeded, mid-run ones dispatch from their start. ``dms list`` is best-effort (some relays return ``[]``); the fallback shape is a ``channels list`` entry named "DM" with empty - description. Named rooms and missing metadata fail closed as groups.""" + description. Named rooms and missing metadata fail closed as groups. + + ``dms list`` is only a best-effort source: on some hosted relays it returns ``[]`` even when DM + conversations exist (#68871). + """ code, out, _err = await self._run_cli(["dms", "list"]) for dm in _parse_json_list(out) if code == 0 else []: dm_id = str(dm.get("dm_id") or "") @@ -1264,12 +1446,17 @@ class BuzzAdapter(BasePlatformAdapter): continue if self._may_reclassify_as_dm(ch_id): # DM-shaped entries promote to DM — including ones already watched. + # See #77987, #87899, #99431. if ch_id in self._channel_state: self._channel_state[ch_id]["chat_type"] = "dm" else: await self._adopt_conversation(ch_id, seed) elif ch_id not in self._channel_state and not seed and not self.channels: # Watch-all mode adopts channels joined mid-run (seeded: history predates us); explicit lists stay authoritative. + # Live adoption of real community channels joined mid-run (#75107): in watch-all mode (no + # explicit channels list) a channel the agent is added to after connect() must start + # dispatching without a gateway restart. Unlike a fresh DM its history predates us, so it is + # always seeded from its newest events — only messages sent after adoption dispatch. await self._seed_channel(ch_id, chat_type="group") logger.info("Buzz: adopted newly joined channel %s (%s)", ch_id, self._channel_names.get(ch_id, ch_id)) @@ -1421,6 +1608,7 @@ class BuzzAdapter(BasePlatformAdapter): return # Cache before any early return so self-echo and concurrent-author traffic can still be reply parents. self._remember_event(state, event) + # See #75826. if pubkey == self._self_pubkey: return # Reclassify a leaked DM before gating so its first un-mentioned message both latches and dispatches. @@ -1469,6 +1657,12 @@ class BuzzAdapter(BasePlatformAdapter): # ── DM classification: DMs leak in via ``channels list`` as "group"; a real channel's p-tag is only addressing ── + # ── DM classification (issue #68871) ────────────────────────────────── ``buzz dms list`` returns [] on + # some hosted relays even when DM conversations exist, so DMs can leak in through ``channels list`` as + # chat_type="group". Relay-materialized DMs are named "DM" with an empty description, which periodic + # discovery promotes to DM even when messages omit recipient p-tags. Named channels and missing metadata + # fail closed. In normal channels a p-tag is only an addressing signal and must wake the agent without + # changing the conversation type. def _may_reclassify_as_dm(self, channel_id: str) -> bool: """True when metadata does not rule out a DM (name "DM", empty description); missing metadata fails closed.""" meta = self._channel_meta.get(channel_id) @@ -1737,6 +1931,8 @@ def check_requirements() -> bool: """Check if Buzz is configured: a relay URL plus a resolvable key.""" if _profile_scoped(): # Secondary profile: os.environ's BUZZ_* are the default profile's and must not satisfy the gate. + # Consult the profile's own config.yaml (via the scoped home override) and its secret scope instead; + # an unconfigured profile fails closed. See #98738. extra = _profile_buzz_extra() return bool(str(extra.get("relay_url") or "").strip() and _resolve_private_key(extra)) # The gate runs before per-profile scopes install; the relay can be externally managed too. @@ -1748,6 +1944,7 @@ def validate_config(config) -> bool: extra = getattr(config, "extra", {}) or {} # Scoped: extra is authoritative; unscoped: env read gains the external-secret rung. if _profile_scoped(): + # See #98738. relay = _scoped_platform_setting("BUZZ_RELAY_URL", extra, "relay_url") relay = relay if relay is not None else extra.get("relay_url", "") else: @@ -1779,6 +1976,10 @@ def _apply_yaml_config(yaml_cfg: dict, buzz_cfg: dict) -> Optional[dict]: if not isinstance(extra, dict): return None # A secondary profile must NOT write the process-global env (first-writer-wins would pin it for every profile). + # Under multiplex, a secondary profile's config loads inside its runtime scope; its values must NOT be + # written to the process-global env, where first-writer-wins would pin them for every other profile + # (issue #72348 Telegram/Discord mirror, Buzz side of #98738). Its adapter reads the profile's + # PlatformConfig.extra directly instead. skip_env_bridge = _profile_scoped() interval = extra.get("poll_interval") if interval is not None and not skip_env_bridge and not os.getenv("BUZZ_POLL_INTERVAL"): @@ -1799,6 +2000,9 @@ def _env_enablement() -> Optional[dict]: if _profile_scoped(): # Process env holds the default profile's BUZZ_*; never fabricate Buzz for a secondary profile. return None + # Secondary profile scope (#98738): the process env's BUZZ_* values are the default profile's + # configuration, not this profile's — env enablement must not fabricate a Buzz platform for a profile + # that did not configure one. relay = os.getenv("BUZZ_RELAY_URL", "").strip() if not relay or not _resolve_private_key(): return None diff --git a/plugins/platforms/dingtalk/adapter.py b/plugins/platforms/dingtalk/adapter.py index 95c0463efb..96813ef689 100644 --- a/plugins/platforms/dingtalk/adapter.py +++ b/plugins/platforms/dingtalk/adapter.py @@ -82,13 +82,26 @@ def _csv_set(raw: Any) -> Set[str]: def dingtalk_deps_present() -> bool: - """PASSIVE registry ``check_fn`` — must never install; credentials are gated separately.""" + """PASSIVE registry ``check_fn`` — must never install; credentials are gated separately. + + Registry ``check_fn`` — called from status displays and config loading, so it must never install + anything. The ACTIVE lazy-installer (``check_dingtalk_requirements``) is registered as + ``ensure_deps_fn`` and runs from ``create_adapter()`` when this returns False (#79812). + """ return DINGTALK_STREAM_AVAILABLE and HTTPX_AVAILABLE def ensure_dingtalk_deps() -> bool: """ACTIVE deps-only installer (registry ``ensure_deps_fn``); rebinds module globals. Deliberately does NOT - check credentials: an ``extra``-configured platform would otherwise be vetoed before ever installing (deadlock).""" + check credentials: an ``extra``-configured platform would otherwise be vetoed before ever installing (deadlock). + + Lazy-installs dingtalk-stream/httpx and rebinds module globals. Deliberately does NOT check credentials + — ``ensure_deps_fn``'s contract is deps-only ("Returns True once deps are importable"); credentials are + gated by ``is_connected``/``validate_config``. Otherwise a platform configured via + ``PlatformConfig.extra`` (which ``_is_connected`` accepts) would pass enablement, reach + ``create_adapter()``, and have the installer veto on env-var grounds before ever installing — + re-creating the #79812 deadlock for extra-configured setups. + """ global DINGTALK_STREAM_AVAILABLE, dingtalk_stream, ChatbotMessage, CallbackMessage, AckMessage, HTTPX_AVAILABLE, httpx if DINGTALK_STREAM_AVAILABLE and HTTPX_AVAILABLE: return True @@ -694,7 +707,12 @@ def _nested_allowed_users(yaml_cfg: dict, dingtalk_cfg: dict): def _apply_yaml_config(yaml_cfg: dict, dingtalk_cfg: dict) -> dict | None: """Translate config.yaml dingtalk: keys into DINGTALK_* env vars (apply_yaml_config_fn); env wins, returns None. The docs put the allowlist at - ``gateway.platforms.dingtalk.extra.allowed_users`` but gateway authz only consults DINGTALK_ALLOWED_USERS, so nested-only allowlists are bridged too.""" + ``gateway.platforms.dingtalk.extra.allowed_users`` but gateway authz only consults DINGTALK_ALLOWED_USERS, so nested-only allowlists are bridged too. + + Implements the apply_yaml_config_fn contract (#24849). Mirrors the legacy dingtalk_cfg block from + gateway/config.py::load_gateway_config(). Env vars take precedence over YAML (each assignment guarded by + not os.getenv(...)). Returns None — everything flows through env. + """ for key, env, encode in (("require_mention", "DINGTALK_REQUIRE_MENTION", lambda v: str(v).lower()), ("mention_patterns", "DINGTALK_MENTION_PATTERNS", json.dumps)): if key in dingtalk_cfg and not os.getenv(env): os.environ[env] = encode(dingtalk_cfg[key]) diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index 4a98793c5a..6101704e57 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -216,7 +216,13 @@ sys.path.insert(0, str(_Path(__file__).resolve().parents[3])) def _is_discord_transport_error(exc: BaseException) -> bool: """True for connection-shaped send failures (dead/dropping WS) that never reached Discord, so - the delivery ledger can replay them; timeouts excluded (a timed-out send may have landed).""" + the delivery ledger can replay them; timeouts excluded (a timed-out send may have landed). + + These are the failures where the message demonstrably did NOT reach Discord because the transport itself + was down — the delivery-obligation ledger can safely replay them after reconnect (#95382). HTTP-level + rejections (permissions, formatting, 4xx) are NOT transport errors and must keep their original error + string. + """ if isinstance(exc, asyncio.TimeoutError): return False if isinstance(exc, (ConnectionError, OSError)): @@ -526,6 +532,12 @@ def _clean_discord_id(entry: str) -> str: # secret scope (contextvar propagates into connect()) and falls back to os.getenv outside multiplex. # Authorization/gate env vars snapshotted per-adapter at connect() time. +# ── per-profile gate env reads (issue #72348) ──────────────────────────── Under +# gateway.multiplex_profiles, os.environ is process-global and the YAML→env bridge in _apply_yaml_config is +# first-writer-wins, so a raw os.getenv() on an allow/deny gate can return ANOTHER profile's value. +# _scoped_gate_env reads the active profile's secret scope when one is installed (secondary adapters connect +# — and their discord.py event tasks are created — inside _profile_runtime_scope, so the contextvar +# propagates) and falls back to os.getenv only outside multiplex. _GATE_ENV_KEYS = ( "DISCORD_ALLOWED_USERS", "DISCORD_ALLOWED_ROLES", "DISCORD_ALLOWED_CHANNELS", "DISCORD_IGNORED_CHANNELS", "DISCORD_NO_THREAD_CHANNELS", "DISCORD_FREE_RESPONSE_CHANNELS", @@ -554,7 +566,12 @@ def _multiplex_active() -> bool: def discord_deps_present() -> bool: """PASSIVE probe: is discord.py importable? Registry ``check_fn`` — must never install - (the ACTIVE installer ``check_discord_requirements`` runs as ``ensure_deps_fn``).""" + (the ACTIVE installer ``check_discord_requirements`` runs as ``ensure_deps_fn``). + + Registry ``check_fn`` — called from status displays and config loading, so it must never install + anything. The ACTIVE lazy-installer (``check_discord_requirements``) is registered as ``ensure_deps_fn`` + and runs from ``create_adapter()`` when this returns False (#79812). + """ return DISCORD_AVAILABLE @@ -973,6 +990,9 @@ class DiscordAdapter(BasePlatformAdapter): supports_code_blocks = True # Discord markdown renders fenced code blocks natively splits_long_messages = True # send() chunks via truncate_message(MAX_MESSAGE_LENGTH) # Safety ceiling on split deliveries: chunks beyond the cap become a notice (degenerate turns). + # Safety ceiling on split deliveries (#86581): a degenerate turn can produce tens of thousands of + # characters — without a cap the adapter posts every 2000-char chunk back-to-back and floods the channel + # (the incident delivered 60,698 chars as 31 messages). MAX_SPLIT_MESSAGES = 8 # Voice auto-disconnect after N idle seconds (discord.voice_channel_inactivity_timeout_seconds; 0 off). @@ -994,6 +1014,7 @@ class DiscordAdapter(BasePlatformAdapter): self._allowed_user_ids: set = set() # For button approval authorization self._allowed_role_ids: set = set() # For DISCORD_ALLOWED_ROLES filtering # Gate env snapshot captured in connect() inside the owning profile's scope; None until then. + # None until then; accessors fall back to live scope-aware reads (issue #72348). self._gate_env_snapshot: Optional[Dict[str, str]] = None self.gateway_runner = None # Set by gateway/run.py for cross-platform delivery self._voice_clients: Dict[int, Any] = {} # guild_id -> VoiceClient @@ -1024,6 +1045,9 @@ class DiscordAdapter(BasePlatformAdapter): # Persistent typing loops per channel (DMs don't reliably show bot typing events). self._typing_tasks: Dict[str, asyncio.Task] = {} self._bot_task: Optional[asyncio.Task] = None + # Background task that runs post-connect housekeeping (command-menu registration + DM-topic setup) + # off the connect path so a slow Bot API call (e.g. a set_my_commands stall for certain tokens) + # cannot blow the gateway's connect timeout (#46298). self._post_connect_task: Optional[asyncio.Task] = None # WS liveness probe: REST 200 can't prove Gateway events still arrive, so sample WS # ready/open/ACK + heartbeat latency; consecutive failures -> retryable-fatal. 0 disables. @@ -1060,6 +1084,10 @@ class DiscordAdapter(BasePlatformAdapter): self._nonconversational_messages = _DiscordNonConversationalMessageTracker() # Last truncated mid-stream preview per (chat_id, message_id): past the 2000 cap every edit # truncates to the SAME text, and re-sending only burns edit rate limit. Dropped on finalize. + # Once an oversized streaming edit saturates at the 2000-char preview cap, every subsequent + # progressive edit truncates to the SAME text; re-sending it is a no-op that still counts against + # Discord's edit rate limit (~1 edit per stream tick for the rest of a long reply). Mirrors the + # Telegram #58563 fix. self._last_overflow_preview: Dict[tuple, str] = {} self._warned_fail_closed_default = False @@ -1148,6 +1176,9 @@ class DiscordAdapter(BasePlatformAdapter): if not self._acquire_platform_lock('discord-bot-token', self.config.token, 'Discord bot token'): return False # Snapshot gate env inside the owning profile's scope (immune to the first-writer-wins bridge). + # Snapshot this profile's gate env vars (issue #72348): connect() runs inside the owning + # profile's runtime scope under multiplex, so the snapshot holds THIS adapter's values, immune + # to the first-writer-wins process-global env bridge. self._snapshot_gate_env() self._allowed_user_ids = self._get_allowed_users() # DISCORD_ALLOWED_ROLES: comma-separated role IDs; ANY match grants access. @@ -1169,6 +1200,8 @@ class DiscordAdapter(BasePlatformAdapter): logger.info("[%s] Using proxy for Discord: %s", self.name, proxy_url) # proxy= for HTTP, connector= for SOCKS; allowed_mentions per _build_allowed_mentions. # Close any existing client first: a zombie client also fires on_message -> double responses. + # Without this, the old client remains connected to Discord gateway and both fire on_message, + # causing double responses. See #18187. if self._client is not None: try: if not self._client.is_closed(): @@ -1308,6 +1341,7 @@ class DiscordAdapter(BasePlatformAdapter): ) if _is("PrivilegedIntentsRequired"): # Name the exact intents requested (Server Members only when allowlists need lookups). + # See #79430. guidance = _format_privileged_intents_guidance( needs_members=_needs_server_members_intent( getattr(self, "_allowed_user_ids", None), @@ -2752,7 +2786,13 @@ class DiscordAdapter(BasePlatformAdapter): def _cap_split_chunks(self, chunks: List[str]) -> List[str]: """Cap chunks at ``MAX_SPLIT_MESSAGES``: keep the first N-1 and replace the rest with a - notice so a degenerate turn can't flood the channel (full text stays in session history).""" + notice so a degenerate turn can't flood the channel (full text stays in session history). + + Cap the number of chunks sent for one logical response (#86581). + A degenerate turn can produce tens of thousands of characters; the 86581 incident delivered 60,698 + chars as 31 back-to-back Discord messages. The full response remains available in the gateway + session history / logs. See #86581. + """ if len(chunks) <= self.MAX_SPLIT_MESSAGES: return chunks kept = chunks[: self.MAX_SPLIT_MESSAGES - 1] @@ -2836,6 +2876,9 @@ class DiscordAdapter(BasePlatformAdapter): await self._nonconversational_messages.mark_many(message_ids) elif not _looks_like_nonconversational_history_message(content): self._last_self_message_id[_target_id] = message_ids[-1] + # Connection-shaped failure (WS drop / closed session): use the ledger's runtime-retryable + # marker so the reconnect sweep can replay this final response instead of stranding it until a + # process restart (#95382 silent partial loss). result = SendResult( success=True, message_id=message_ids[0] if message_ids else None, @@ -2949,7 +2992,12 @@ class DiscordAdapter(BasePlatformAdapter): ) -> SendResult: """Edit a sent Discord message. Oversized text (>2,000) must neither truncate silently nor fail (consumer re-sends -> dupe): mid-stream keep a truncated preview (splitting would move - the edit target every tick); ``finalize=True`` delivers all via ``_edit_overflow_split``.""" + the edit target every tick); ``finalize=True`` delivers all via ``_edit_overflow_split``. + + Mid-stream (``finalize=False``) we keep editing the original message with a truncated preview — + splitting mid-stream would move the edit target to a continuation and the next accumulated-token + tick would re-split, looping forever (the Telegram #48648 lesson). + """ if not self._client: return SendResult(success=False, error="Not connected") try: @@ -2968,6 +3016,8 @@ class DiscordAdapter(BasePlatformAdapter): formatted = self.truncate_message(formatted, self.MAX_MESSAGE_LENGTH)[0] _saturated_preview = True # Saturated-preview dedup: past the cap every edit is the same text; skip until finalize. + # Re-sending it is a visual no-op that still counts against Discord's edit rate limit — skip + # silently until finalize (mirrors the Telegram #58563 fix). if self._last_overflow_preview.get(_preview_key) == formatted: return SendResult(success=True, message_id=message_id) elif not finalize: @@ -3095,7 +3145,10 @@ class DiscordAdapter(BasePlatformAdapter): ) -> SendResult: """Send a local file as a Discord attachment (forum channels get a new thread). Path-based ``discord.File`` only: the open-handle form can race the multipart encoder after an image - batch and yield zero attachments — a silent drop for video/document MEDIA tags.""" + batch and yield zero attachments — a silent drop for video/document MEDIA tags. + + See #66797. + """ if not self._client: return SendResult(success=False, error="Not connected") if not os.path.isfile(file_path): @@ -3119,6 +3172,8 @@ class DiscordAdapter(BasePlatformAdapter): attachments = getattr(msg, "attachments", None) or [] if not attachments: # Discord accepted the message but attached nothing: fail loud instead of a silent drop. + # Discord accepted the message but attached nothing — the failure mode reported in #66797 (MEDIA + # video stripped from text, no attachment, no prior log line). logger.warning( "[%s] Discord returned message %s with no attachments for %s", self.name, getattr(msg, "id", "?"), filename, @@ -3878,6 +3933,11 @@ class DiscordAdapter(BasePlatformAdapter): if self._gateway_allow_all_users(): return True # Channel-scoped access needs validated channel context; not a user-wide bypass. + # In shared channels, respond only when addressed — unless require_mention is disabled, in which + # case respond to every message. A NIP-10 thread reply whose direct parent is one of our + # messages is treated as addressed (parity with Signal/WhatsApp; fixes #75826 — e.g. Desktop + # "/approve session" replies that never type @name). Explicit addressing is a text @mention OR a + # signed recipient p-tag (#92781). DMs always dispatch. if ( not is_dm and channel_ids is not None @@ -4003,6 +4063,7 @@ class DiscordAdapter(BasePlatformAdapter): return (False, "missing interaction.user") user_id = str(user.id) # guild + is_dm scope the role check so the cross-guild DM bypass can't land via slash. + # See #12136. interaction_guild = getattr(interaction, "guild", None) if not self._is_allowed_user( user_id, author=user, guild=interaction_guild, is_dm=in_dm, @@ -4319,6 +4380,10 @@ class DiscordAdapter(BasePlatformAdapter): if to_resolve: print(f"[{self.name}] Could not resolve usernames: {', '.join(to_resolve)}") # Adapter-local: under multiplex_profiles os.environ writes would clobber other profiles. + # Update the internal set. Keep the resolved IDs adapter-local first: under multiplex_profiles, + # writing os.environ here would clobber every OTHER profile's DISCORD_ALLOWED_USERS after this + # adapter's on_ready — an unguarded runtime mutation of process-global state (issue #72348). Refresh + # this adapter's own snapshot instead. self._allowed_user_ids = numeric_ids snap = getattr(self, "_gate_env_snapshot", None) if snap is not None: @@ -4525,7 +4590,13 @@ class DiscordAdapter(BasePlatformAdapter): def _register_skill_group(self, tree) -> None: """Register one flat ``/skill`` command with autocomplete on ``name``. A nested ``/skill <category> <name>`` layout blew Discord's ~8000-byte payload cap and broke - ``tree.sync()``; autocomplete options are fetched dynamically. Entries live on ``self``.""" + ``tree.sync()``; autocomplete options are fetched dynamically. Entries live on ``self``. + + The older nested layout (``/skill <category> <name>``) registered one giant command whose serialized + payload grew linearly with the skill catalog — with the default ~75 skills the payload was ~14 KB + and ``tree.sync()`` rejected the entire slash-command batch (issues 11321, #10259, #11385, #10261, + #10214). + """ try: existing_names = set() try: @@ -4658,6 +4729,9 @@ class DiscordAdapter(BasePlatformAdapter): # Forum threads inherit the parent forum's topic. chat_topic = self._get_effective_topic(interaction.channel, is_thread=is_thread) # guild_id/parent_chat_id feed profile_routes matching, as on_message does. + # guild_id/parent_chat_id feed profile_routes matching in build_source, exactly as on_message passes + # them — without them a guild- or channel-routed profile never matches a native slash command + # (#69178). parent_id = (self._get_parent_channel_id(interaction.channel) if is_thread else None) or "" source = self.build_source( chat_id=str(interaction.channel_id), chat_name=chat_name, chat_type=chat_type, @@ -4803,6 +4877,13 @@ class DiscordAdapter(BasePlatformAdapter): # Under multiplex_profiles os.environ is process-global (first-writer-wins), so raw os.getenv # would leak profile A into B. Order: connect()-time env snapshot, config.extra, scoped env read. + # ── per-adapter authorization gates (issue #72348) ─────────────────── Under gateway.multiplex_profiles + # every Discord adapter must enforce ITS OWN profile's allow/deny lists. os.environ is process-global + # and the YAML→env bridge is first-writer-wins, so raw os.getenv reads here would leak profile A's gates + # into profile B. Each accessor reads, in order: the per-adapter env snapshot taken inside the owning + # profile's runtime scope at connect() (authoritative under multiplex), then this adapter's + # PlatformConfig.extra (per-profile YAML), with the live scope-aware env read as the pre-connect + # fallback. Single-profile deployments resolve to plain os.getenv, unchanged. def _snapshot_gate_env(self) -> None: """Snapshot gate env vars; must run inside the owning profile's runtime scope (connect() does under multiplex) to capture that profile's values.""" @@ -5210,7 +5291,13 @@ class DiscordAdapter(BasePlatformAdapter): def _derive_auto_thread_name(self, content: str) -> str: """Fast placeholder thread name with mentions stripped (raw <@id> tokens mean nothing to humans). - Semantic renaming happens after the first agent turn, once an LLM session title exists.""" + Semantic renaming happens after the first agent turn, once an LLM session title exists. + + Strip Discord mention syntax (users / roles / channels) so thread titles don't show raw <@id>, + <@&id>, or <#id> markers — the ID isn't meaningful to humans glancing at the thread list (#6336). + Real semantic naming is done after the first agent turn, when Hermes has an LLM-generated session + title and can safely rename only this newly-created thread. + """ content = (content or "").strip() # <@123>, <@!123>, <@&123>, <#123> — collapse to empty; normalize spaces. content = re.sub(r"<@[!&]?\d+>", "", content) @@ -5232,7 +5319,11 @@ class DiscordAdapter(BasePlatformAdapter): async def _auto_create_thread(self, message: 'DiscordMessage') -> Optional[Any]: """Create an auto-thread from a user message; returns the thread or ``None``. - Primary path and seed-message fallback each retry once after a short backoff (transient errors).""" + Primary path and seed-message fallback each retry once after a short backoff (transient errors). + + ``Cannot connect to host discord.com:443``) don't immediately burn through to the caller's failure + path (#20243). + """ thread_name = self._derive_auto_thread_name(message.content or "") display_name = getattr(getattr(message, "author", None), "display_name", None) or "unknown user" reason = f"Auto-threaded from mention by {display_name}" @@ -5703,7 +5794,12 @@ class DiscordAdapter(BasePlatformAdapter): async def _cache_discord_document(self, att, ext: str) -> bytes: """Download a document attachment: ``att.read()`` first, SSRF-gated aiohttp fallback. - Caller passes the bytes to ``cache_document_from_bytes`` (and injects text if applicable).""" + Caller passes the bytes to ``cache_document_from_bytes`` (and injects text if applicable). + + This closes the gap where the old document path made raw ``aiohttp.ClientSession`` requests with no + safety check (#11345). The caller is responsible for passing the returned bytes to + ``cache_document_from_bytes`` (and, where applicable, for injecting text content). + """ raw_bytes = await self._read_attachment_bytes(att, media_type="document") if raw_bytes is not None: return raw_bytes @@ -5919,6 +6015,9 @@ class DiscordAdapter(BasePlatformAdapter): # Auto-threading is the routing target; do NOT fall back to an inline parent-channel # reply (dumps the task into a shared channel). Surface an error and skip the run. try: + # That breaks thread-first Discord workflows by dumping a new task into a shared + # channel. Surface a short visible error so the user can retry once Discord + # recovers, and skip agent invocation for this message. See #20243. await message.channel.send( "⚠️ Hermes could not create a Discord thread for " "this message, so the request was not processed. Please retry." @@ -6084,6 +6183,9 @@ def _component_check_auth( return False # Scope-aware reads: interaction tasks inherit the owning profile's secret-scope contextvar; # under multiplex a raw os.getenv could return ANOTHER profile's allow-all flag. + # Scope-aware reads (issue #72348): component interactions are dispatched from discord.py tasks + # descended from the task created inside the owning profile's runtime scope, so the profile's + # secret-scope contextvar is inherited here. if _scoped_gate_env("DISCORD_ALLOW_ALL_USERS").strip().lower() in {"true", "1", "yes"}: return True if _scoped_gate_env("GATEWAY_ALLOW_ALL_USERS").strip().lower() in {"true", "1", "yes"}: @@ -7119,7 +7221,11 @@ _YAML_WEBSOCKET_LIVENESS_KEYS = ( def _apply_yaml_config(yaml_cfg: dict, discord_cfg: dict) -> dict | None: """Translate ``config.yaml`` ``discord:`` keys into env vars (``apply_yaml_config_fn``). The adapter reads ``DISCORD_*`` via ``os.getenv()`` at ~50 sites, so this hook owns YAML→env; - ``extra`` stays the per-adapter truth for liveness (multiplex isolation). Returns liveness settings.""" + ``extra`` stays the per-adapter truth for liveness (multiplex isolation). Returns liveness settings. + + Implements the ``apply_yaml_config_fn`` contract (#24836). Mirrors the legacy ``discord_cfg`` block that + used to live in ``gateway/config.py::load_gateway_config()`` before this migration. + """ def _env_default(env_key: str, value) -> None: # First-writer-wins: an explicit env var always beats the YAML value. if not os.getenv(env_key): @@ -7142,6 +7248,9 @@ def _apply_yaml_config(yaml_cfg: dict, discord_cfg: dict) -> dict | None: seeded_extra = {} # Gate keys are ALWAYS seeded into PlatformConfig.extra (per-profile lists); the os.environ writes # below are first-writer-wins for legacy consumers and skipped for profile-scoped multiplex loads. + # The os.environ writes below remain first-writer-wins for legacy env-only consumers, but are skipped + # for profile-scoped loads under multiplex — a secondary profile's gates must never land in + # process-global env where they'd become another profile's policy. See #72348. _skip_env_bridge = _profile_scoped_config_load() def _gate(key: str, env_key: str, *, from_platform_extra: bool, lower: bool = False) -> None: @@ -7228,6 +7337,11 @@ def register(ctx) -> None: install_hint="Run `hermes setup` to install Discord support.", setup_fn=interactive_setup, # YAML→env bridge: ``discord:`` config keys → ``DISCORD_*`` env vars read via os.getenv(). + # YAML→env config bridge — owns the translation of ``config.yaml`` ``discord:`` keys + # (require_mention, free_response_channels, auto_thread, reactions, ignored_channels, + # allowed_channels, no_thread_channels, allow_mentions.*, reply_to_mode, thread_require_mention) + # into ``DISCORD_*`` env vars that the adapter reads via ``os.getenv()``. Replaces the hardcoded + # block that used to live in ``gateway/config.py``. Hook contract: #24836. apply_yaml_config_fn=_apply_yaml_config, allowed_users_env="DISCORD_ALLOWED_USERS", allow_all_env="DISCORD_ALLOW_ALL_USERS", diff --git a/plugins/platforms/email/adapter.py b/plugins/platforms/email/adapter.py index 9bbd709cf7..82271b56cf 100644 --- a/plugins/platforms/email/adapter.py +++ b/plugins/platforms/email/adapter.py @@ -54,6 +54,7 @@ _AUTH_METHOD_RE = re.compile(r"\b(dmarc|dkim|spf)\s*=\s*([a-z]+)", re.IGNORECASE _AUTH_PROP_RE = re.compile(r"\b(header\.from|header\.d|smtp\.mailfrom|smtp\.from|envelope-from)\s*=\s*([^\s;]+)", re.IGNORECASE) +# Backwards-compatible alias for the name used by the original #59076 hunks. def _esecret_int(name: str, default: int) -> int: """Scope-aware integer read.""" return coerce_port(str(_get_secret(name, "")).strip() or default, default) @@ -83,7 +84,14 @@ def _tls_context(verify: bool, host: str) -> ssl.SSLContext: def _close_imap(imap: "imaplib.IMAP4") -> None: """Teardown that guarantees the socket closes: ``logout()`` only guards ``OSError``, so ``IMAP4.abort`` on a - broken connection skipped ``shutdown()`` and leaked one fd per failed poll (fatal on macOS's 256 soft limit).""" + broken connection skipped ``shutdown()`` and leaked one fd per failed poll (fatal on macOS's 256 soft limit). + + ``IMAP4.logout()`` only guards against ``OSError`` internally: a broken connection makes + ``_simple_command('LOGOUT')`` raise ``IMAP4.abort`` (which is *not* an ``OSError``), so ``logout()`` + propagates before its own ``shutdown()`` call and the TCP socket stays open. On macOS, where the default + soft fd limit is 256 and pollers may run through a local proxy, these abandoned sockets accumulate one + per failed poll until the gateway hits ``[Errno 24] Too many open files`` (#79889). + """ try: imap.logout() except Exception: @@ -154,12 +162,23 @@ def _is_automated_sender(address: str, headers: dict) -> bool: def check_email_requirements() -> bool: - """True when all email settings are present and non-blank (blank keys left by an abandoned setup must not enable the platform).""" + """True when all email settings are present and non-blank (blank keys left by an abandoned setup must not enable the platform). + + Treats blank/whitespace-only values as missing so an abandoned setup that left empty ``EMAIL_*`` keys in + ``.env`` does not enable the platform (#40715). + """ return all(_get_secret(name, "").strip() for name in ("EMAIL_ADDRESS", "EMAIL_PASSWORD", "EMAIL_IMAP_HOST", "EMAIL_SMTP_HOST")) def _safe_decode(payload: bytes, charset: "Optional[str]") -> str: - """Decode without ever raising: ``errors="replace"`` does not guard a missing codec (``LookupError``), so fall back alias → UTF-8 → latin-1.""" + """Decode without ever raising: ``errors="replace"`` does not guard a missing codec (``LookupError``), so fall back alias → UTF-8 → latin-1. + + Unknown or malformed charset labels (``unknown-8bit``, misspelled names, attacker-controlled garbage) + previously raised ``LookupError`` from ``bytes.decode`` — ``errors="replace"`` only guards decode + errors, not a missing codec — which aborted the whole IMAP fetch and dropped every message in the batch + (#35901, #55381, #55383). Fall back through a small alias table, then UTF-8, then latin-1 (which never + fails). + """ label = (charset or "utf-8").strip().strip("\"'").lower() or "utf-8" for candidate in (_CHARSET_ALIASES.get(label, label), "utf-8"): try: @@ -170,7 +189,11 @@ def _safe_decode(payload: bytes, charset: "Optional[str]") -> str: def _decode_header_value(raw: str) -> str: - """Decode an RFC 2047 header into a plain string; never raises.""" + """Decode an RFC 2047 header into a plain string; never raises. + + Never raises: malformed encoded-words or unknown charsets degrade to replacement characters instead of + crashing the fetch loop (#55381). + """ try: parts = decode_header(raw) except Exception: # malformed RFC 2047 structure @@ -329,6 +352,8 @@ class EmailAdapter(BasePlatformAdapter): self._poll_task: Optional[asyncio.Task] = None self._last_fetch_failed, self._last_fetch_error = False, "" # "checked, nothing new" vs "the check itself failed" # chat_id (sender email) -> last subject + message-id for threading + # Track the last IMAP fetch attempt so the poll loop can distinguish "checked, nothing new" from + # "the check itself failed" (#80016). self._thread_context: Dict[str, Dict[str, str]] = {} logger.info("[Email] Adapter initialized for %s", self._address) @@ -359,6 +384,11 @@ class EmailAdapter(BasePlatformAdapter): @contextmanager def _inbox(self): """Logged-in IMAP handle on INBOX; always ``_close_imap``-ed on exit (a login/select failure used to leak one fd per reconnect).""" + # Test IMAP connection. The handle is closed in ``finally`` — before this, a failure in + # login/select/search left the TCP socket open with no owner, leaking one fd per connect attempt. + # Under the gateway's reconnect watcher (fresh adapter instance per retry) against an + # unreachable/proxied host this grew monotonically until fd exhaustion on macOS's 256 soft limit + # (#79889). imap = self._connect_imap() try: imap.login(self._address, self._password) @@ -476,6 +506,8 @@ class EmailAdapter(BasePlatformAdapter): if self._last_fetch_failed: # The IMAP check itself failed (not an empty inbox): route through the fatal-error hook so the gateway's # reconnect/backoff re-establishes the mailbox. The handler runs detached (gateway/run.py), so awaiting it is safe. + # The handler runs in a detached task (gateway/run.py), so awaiting it from our own poll task is + # safe even though teardown cancels this task. See #80016. self._last_fetch_failed = False self._set_fatal_error("email_imap_fetch_failed", self._last_fetch_error or "IMAP fetch failed", retryable=True) await self._notify_fatal_error() @@ -494,6 +526,8 @@ class EmailAdapter(BasePlatformAdapter): continue # transient per-UID refusal: leave unseen so the next poll retries # Mark seen once a response arrived (even malformed) so garbage is skipped once, not retried forever — # but NOT before the fetch: a connection failure must leave the rest of the batch eligible for the next poll. + # IMAP fetch can return unexpected structures (e.g. a single bytes item instead of a + # list of tuples). See #80032. self._seen_uids.add(uid) self._trim_seen_uids() try: @@ -506,6 +540,7 @@ class EmailAdapter(BasePlatformAdapter): continue # One poison message (unparseable headers, pathological attachment, DNS hiccup) must not abort the batch or force a reconnect. try: + # See #80032. parsed = self._parse_fetched_message(uid, raw_email) except Exception as parse_exc: logger.error("[Email] Failed to process message UID %s, skipping: %s", uid, parse_exc) @@ -513,6 +548,8 @@ class EmailAdapter(BasePlatformAdapter): if parsed is not None: results.append(parsed) except Exception as e: + # _close_imap guarantees the socket dies even when logout() raises IMAP4.abort on a broken + # connection (#79889). logger.error("[Email] IMAP fetch error: %s", e) self._last_fetch_failed, self._last_fetch_error = True, str(e) # Keep the reconnect snapshot current so a mid-outage adapter recreation does not re-dispatch messages already processed. diff --git a/plugins/platforms/feishu/adapter.py b/plugins/platforms/feishu/adapter.py index 6160996418..7614366cb7 100644 --- a/plugins/platforms/feishu/adapter.py +++ b/plugins/platforms/feishu/adapter.py @@ -978,6 +978,23 @@ def _strip_edge_self_mentions(text: str, mentions: Sequence[FeishuMentionRef]) - # * ``websockets.connect`` becomes one dispatcher that merges the calling thread's # registered ping overrides, so profiles stop racing over the global patch. +# --------------------------------------------------------------------------- Multiplex isolation for the +# lark_oapi WebSocket client (#73779) +# --------------------------------------------------------------------------- ``lark_oapi.ws.client`` keeps +# the asyncio loop used by ``Client.start()`` and every coroutine it spawns in a *module-level global* +# (``loop``), and Hermes also monkey-patches ``websockets.connect`` on the shared ``websockets`` module to +# inject per-adapter ping settings. In multiplex mode every profile runs its own WS client on a dedicated +# thread, so the N threads overwrite each other's module globals (last-write-wins): a client ends up +# scheduling tasks on a sibling profile's loop ("Future attached to a different loop" crashes) or binds to +# the wrong loop at construction time and goes deaf from the start. The fix installs process-wide, +# thread-dispatching shims exactly once: * ``ws_client_module.loop`` becomes a proxy that forwards every +# attribute access to the loop registered by the *current thread*. All SDK reads of the global happen on the +# thread that owns the loop (``start()`` blocks in ``run_until_complete`` and every ``create_task`` callback +# runs on the loop's own thread), so each profile transparently sees its own loop. Threads that never +# registered one (single-profile installs, CLI) fall back to the SDK's original module loop. * +# ``websockets.connect`` becomes a single dispatcher that merges the per-thread ping overrides registered by +# the calling profile, so profiles no longer race over the global patch or restore each other's hooks while +# a sibling is still connected. _WS_ISOLATION_LOCK = threading.Lock() _WS_ISOLATION_INSTALLED = False _ws_isolation_state = threading.local() # per WS thread: .loop and .connect_kwargs @@ -1106,6 +1123,10 @@ def feishu_deps_present() -> bool: Uses cheap importlib.metadata lookups; the real import is deferred to ``_load_lark_oapi`` and the ACTIVE installer is ``check_feishu_requirements`` (``ensure_deps_fn``). + + Registry ``check_fn`` — called from status displays and config loading, so it must never install + anything. The ACTIVE lazy-installer (``check_feishu_requirements``) is registered as ``ensure_deps_fn`` + and runs from ``create_adapter()`` when this returns False (#79812). """ if FEISHU_AVAILABLE: return True @@ -1188,6 +1209,7 @@ class FeishuAdapter(BasePlatformAdapter): self._client: Optional[Any] = None # Adapter-owned pool for blocking SDK calls, recreated on demand: a torn-down default # executor can no longer wedge sends with "Executor shutdown has been called". + # See issue #10849. self._sdk_executor_lock = threading.Lock() self._sdk_executor: Optional[concurrent.futures.ThreadPoolExecutor] = None self._sdk_executor_closing = False # set on disconnect so a real teardown isn't resurrected @@ -1262,6 +1284,7 @@ class FeishuAdapter(BasePlatformAdapter): # Env-only so adapter and gateway auth bypass share one source (yaml feishu.allow_bots # is bridged to the env var at config load). Scoped read: under multiplex a secondary # profile's .env must govern its own adapter. + # See #86905. allow_bots = _get_scoped_secret("FEISHU_ALLOW_BOTS", "none").strip().lower() if allow_bots not in {"none", "mentions", "all"}: logger.warning( @@ -1332,7 +1355,12 @@ class FeishuAdapter(BasePlatformAdapter): ) def _get_sdk_executor(self) -> concurrent.futures.ThreadPoolExecutor: - """Adapter-owned executor; recreated after an *external* shutdown, never after our own close.""" + """Adapter-owned executor; recreated after an *external* shutdown, never after our own close. + + Recreates the pool if it was never built or was shut down by an *external* teardown of the loop's + default executor, so that can no longer permanently wedge sends (#10849). Refuses to resurrect once + the adapter itself is closing — a real disconnect/shutdown stays shut. + """ lock = getattr(self, "_sdk_executor_lock", None) # bare adapters (tests) may lack __init__ state if lock is None: lock = self._sdk_executor_lock = threading.Lock() @@ -1428,6 +1456,10 @@ class FeishuAdapter(BasePlatformAdapter): await self._cancel_pending_tasks(self._pending_media_batch_tasks) self._reset_batch_buffers() # ``_disable_websocket_auto_reconnect()`` nils ``_ws_client`` — capture first. + # Send a WebSocket CLOSE frame to Feishu BEFORE tearing down the thread loop. Without this, Feishu's + # server never learns the connection is dead and continues routing messages to the stale endpoint — + # the channel goes silent until the server-side CLOSE-WAIT expires (minutes to hours). See issue + # #10202. ws_client = self._ws_client ws_thread_loop = self._ws_thread_loop self._disable_websocket_auto_reconnect() @@ -1533,6 +1565,8 @@ class FeishuAdapter(BasePlatformAdapter): # Decide markdown-vs-text once for the whole message: a chunk of a long # markdown reply may be plain prose that fails the per-chunk regex and would # otherwise render as literal ``**bold`` / fences while other chunks render. + # Lock the markdown decision at the whole-message level so every chunk consistently uses ``post``. + # See #26841. prefer_post = bool(_MARKDOWN_HINT_RE.search(formatted)) last_response = None @@ -2593,6 +2627,7 @@ class FeishuAdapter(BasePlatformAdapter): ) response.raise_for_status() # Snapshot headers + body inside the context so pooled connections fully release. + # See #18451. content_type_hdr = str(response.headers.get("Content-Type", "")) body = response.content filename = self._derive_remote_filename( @@ -2900,6 +2935,10 @@ class FeishuAdapter(BasePlatformAdapter): # Lark's native "audio" is an in-app voice recording (uploaded audio arrives as # file/media → "document"). VOICE makes the gateway auto-transcribe it like # Discord/DingTalk/Telegram; as AUDIO it would be silently ignored. + # Classify it as VOICE so the gateway auto-transcribes it (Opus → STT) the same way + # Discord/DingTalk/Telegram/etc. do — otherwise a Feishu voice note reaches the agent as an + # untranscribable AUDIO attachment and is silently ignored. Follow-up to #28993, which added + # native voice-note transcription for Discord + DingTalk. return MessageType.VOICE if preferred in ("photo", "document"): default = MessageType.PHOTO if preferred == "photo" else MessageType.DOCUMENT @@ -3428,6 +3467,11 @@ class FeishuAdapter(BasePlatformAdapter): # Feishu clients render markdown tables inside ``post`` ``md`` elements natively, so tables # take the common markdown path (no text downgrade). ``prefer_post`` lets ``send`` keep every # chunk of a split markdown reply as ``post`` even when a chunk alone looks like prose. + # The previous table-downgrade branch forced any table-containing message to ``text``, which left + # Feishu readers seeing the raw pipe-and-dash source instead of a rendered table. ``prefer_post`` + # lets ``send`` treat the chunk as part of a larger markdown document: when a long markdown reply is + # split at MAX_MESSAGE_LENGTH, the per-chunk regex would otherwise mis-classify a plain-prose chunk + # as ``text``. See #26841. if prefer_post or _MARKDOWN_HINT_RE.search(content): return "post", _build_markdown_post_payload(content) return "text", json.dumps({"text": content}, ensure_ascii=False) @@ -3623,6 +3667,8 @@ class FeishuAdapter(BasePlatformAdapter): ``lark_oapi.start()`` only returns on fatal errors; without this watcher a dead thread left the profile silently deaf until a gateway restart. Rebuild with capped backoff. + + See #73779. """ backoff = initial_backoff = float(self._ws_restart_backoff) last_dead: Optional[asyncio.Future] = None @@ -3677,6 +3723,7 @@ class FeishuAdapter(BasePlatformAdapter): self._prepare_client() await self._hydrate_bot_identity() # client_max_size backstops the bounded reader in _handle_webhook_request on every read path. + # See #58536, #58902, #59180. app = web.Application(client_max_size=_FEISHU_WEBHOOK_MAX_BODY_BYTES) app.router.add_post(self._webhook_path, self._handle_webhook_request) self._webhook_runner = web.AppRunner(app) @@ -4065,6 +4112,14 @@ def _qr_register_inner(*, initial_domain: str, timeout_seconds: int) -> Optional # --- Plugin glue: register(ctx) + the hook fns that replaced the per-platform core touchpoints --- +# ────────────────────────────────────────────────────────────────────────── Plugin migration glue (#41112 / +# #3823) Added when the Feishu adapter (+ its feishu_comment / feishu_comment_rules / feishu_meeting_invite +# satellites) moved from gateway/platforms/ into this bundled plugin. Mirrors the Discord (#24356) / Slack +# migrations: a register(ctx) entry point plus hook implementations that replace the per-platform core +# touchpoints (the Platform.FEISHU elif in gateway/run.py, the feishu_cfg YAML→env block + +# _PLATFORM_CONNECTED_CHECKERS entry in gateway/config.py, the _setup_feishu wizard + _PLATFORMS["feishu"] +# static dict in hermes_cli/gateway.py, and the _send_feishu dispatch in tools/send_message_tool.py). +# ────────────────────────────────────────────────────────────────────────── _MIGRATION_IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".webp", ".gif"} _MIGRATION_VIDEO_EXTS = {".mp4", ".mov", ".avi", ".mkv", ".webm", ".3gp"} _MIGRATION_AUDIO_EXTS = {".ogg", ".opus", ".mp3", ".wav", ".m4a", ".flac"} @@ -4228,7 +4283,11 @@ def interactive_setup() -> None: def _apply_yaml_config(yaml_cfg: dict, feishu_cfg: dict) -> dict | None: - """apply_yaml_config_fn: bridge config.yaml feishu.allow_bots to FEISHU_ALLOW_BOTS (env wins); returns None.""" + """apply_yaml_config_fn: bridge config.yaml feishu.allow_bots to FEISHU_ALLOW_BOTS (env wins); returns None. + + Implements the apply_yaml_config_fn contract (#24849). Mirrors the legacy feishu_cfg block from + gateway/config.py::load_gateway_config() (allow_bots). Env vars take precedence over YAML. + """ if "allow_bots" in feishu_cfg and not os.getenv("FEISHU_ALLOW_BOTS"): os.environ["FEISHU_ALLOW_BOTS"] = str(feishu_cfg["allow_bots"]).lower() return None diff --git a/plugins/platforms/google_chat/adapter.py b/plugins/platforms/google_chat/adapter.py index 4008c0b97b..2e8422c625 100644 --- a/plugins/platforms/google_chat/adapter.py +++ b/plugins/platforms/google_chat/adapter.py @@ -415,7 +415,13 @@ class GoogleChatAdapter(BasePlatformAdapter): # -- configuration ------------------------------------------------------- def _load_sa_credentials(self) -> Any: - """SA credentials: ``extra['service_account_json']`` → GOOGLE_APPLICATION_CREDENTIALS → ADC.""" + """SA credentials: ``extra['service_account_json']`` → GOOGLE_APPLICATION_CREDENTIALS → ADC. + + Priority: 1. Explicit ``extra['service_account_json']`` (path or inline JSON) 2. 3. Application + Default Credentials via ``google.auth.default()`` — works on Cloud Run / GCE / GKE with a workload + identity attached, or locally via ``gcloud auth application-default login``. Lets operators run the + gateway in GCP without managing SA key files. Pattern lifted from PR #14965. + """ sa_path = self.config.extra.get("service_account_json") or _get_scoped_secret("GOOGLE_APPLICATION_CREDENTIALS") try: credentials = _load_sa_credentials_from(sa_path) @@ -1145,7 +1151,10 @@ class GoogleChatAdapter(BasePlatformAdapter): @classmethod def format_message(cls, content: str) -> str: - """Convert standard Markdown to Google Chat's dialect (see ``cards.format_message``).""" + """Convert standard Markdown to Google Chat's dialect (see ``cards.format_message``). + + Pattern lifted from PR #14965. + """ return _format_message(content) def _resolve_thread_id(self, reply_to: Optional[str], metadata: Optional[Dict[str, Any]], @@ -1171,7 +1180,10 @@ class GoogleChatAdapter(BasePlatformAdapter): async def _call_with_retry(self, sync_fn: Callable[[], Any], *, op_name: str = "chat-api-call") -> Any: """Run ``sync_fn`` in a thread with bounded retry + jittered backoff; only - transient failures are retried, permanent ones bubble up on the first attempt.""" + transient failures are retried, permanent ones bubble up on the first attempt. + + Pattern lifted from PR #14965. + """ delay = _RETRY_BASE_DELAY for attempt in range(1, _RETRY_MAX_ATTEMPTS + 1): try: diff --git a/plugins/platforms/google_chat/cards.py b/plugins/platforms/google_chat/cards.py index b92b1cf58a..eaf5b8da2c 100644 --- a/plugins/platforms/google_chat/cards.py +++ b/plugins/platforms/google_chat/cards.py @@ -12,6 +12,7 @@ from typing import Any, Callable, Dict # Invisible Unicode codepoints that render as tofu (□) in Google Chat's # restricted font stack: ZWS/ZWNJ/ZWJ, bidi marks, word joiner, BOM and # Variation Selectors (Chat ignores them and often shows a blank box). +# Pattern lifted from PR #14965. _INVISIBLE_RE = re.compile( "[" "\u200b" # Zero-Width Space diff --git a/plugins/platforms/homeassistant/adapter.py b/plugins/platforms/homeassistant/adapter.py index 80ba8ab735..47ceb99316 100644 --- a/plugins/platforms/homeassistant/adapter.py +++ b/plugins/platforms/homeassistant/adapter.py @@ -297,6 +297,13 @@ class HomeAssistantAdapter(BasePlatformAdapter): # -- Standalone (out-of-process) sender — cron deliver=homeassistant --------- +# ────────────────────────────────────────────────────────────────────────── Plugin migration glue (#41112 / +# #3823) Added when the Email adapter moved from gateway/platforms/email.py into this bundled plugin. +# register() exposes the platform via the registry, replacing the Platform.EMAIL elif in gateway/run.py, the +# _PLATFORM_CONNECTED_CHECKERS entry in gateway/config.py, the _PLATFORMS["email"] static dict in +# hermes_cli/gateway.py, and the _send_email dispatch in tools/send_message_tool.py. EMAIL_* +# env→PlatformConfig seeding stays in core. +# ────────────────────────────────────────────────────────────────────────── async def _standalone_send( pconfig, chat_id: str, message: str, *, thread_id: Optional[str] = None, media_files: Optional[list] = None, force_document: bool = False, diff --git a/plugins/platforms/irc/adapter.py b/plugins/platforms/irc/adapter.py index c18f9d1384..6435c301b0 100644 --- a/plugins/platforms/irc/adapter.py +++ b/plugins/platforms/irc/adapter.py @@ -602,6 +602,8 @@ def register(ctx): label="IRC", adapter_factory=IRCAdapter, check_fn=check_requirements, + # ACTIVE lazy-installer — create_adapter() calls this when check_fn is False, right before the + # gateway connects Teams (#79812). validate_config=validate_config, is_connected=is_connected, required_env=["IRC_SERVER", "IRC_CHANNEL", "IRC_NICKNAME"], diff --git a/plugins/platforms/line/adapter.py b/plugins/platforms/line/adapter.py index fdcbcf1aa7..462e6bab5d 100644 --- a/plugins/platforms/line/adapter.py +++ b/plugins/platforms/line/adapter.py @@ -91,7 +91,11 @@ _MD_STRIP_RULES: Tuple[Tuple[re.Pattern, Any], ...] = ( def strip_markdown_preserving_urls(text: str) -> str: """Strip Markdown LINE can't render; ``[label](url)`` → ``label (url)`` keeps URLs - tappable (LINE auto-links bare URLs only). Code-block content is kept.""" + tappable (LINE auto-links bare URLs only). Code-block content is kept. + + Source: PR #18153 (leepoweii) — adapted to keep code-block content visible (LINE users frequently want + command snippets to land as plain text, not be eaten by the fence). + """ if not text: return text for pattern, repl in _MD_STRIP_RULES: @@ -149,7 +153,10 @@ class _CacheEntry: class RequestCache: - """In-memory cache for slow-LLM postback retrieval (PENDING → READY|ERROR → DELIVERED).""" + """In-memory cache for slow-LLM postback retrieval (PENDING → READY|ERROR → DELIVERED). + + We keep the same model here. See #18153. + """ def __init__(self) -> None: self._entries: Dict[str, _CacheEntry] = {} @@ -202,14 +209,20 @@ _SOURCE_KINDS = {"group": ("groupId", "group"), "room": ("roomId", "room"), "use def _resolve_chat(source: Dict[str, Any]) -> Tuple[str, str]: - """Return ``(chat_id, chat_type)`` from a LINE event ``source`` block (user/group/room).""" + """Return ``(chat_id, chat_type)`` from a LINE event ``source`` block (user/group/room). + + Source: PR #21023 (perng), unchanged. + """ kind = _SOURCE_KINDS.get((source or {}).get("type", "")) return ("", "dm") if kind is None else (source.get(kind[0], ""), kind[1]) def _allowed_for_source( source: Dict[str, Any], *, allow_all: bool, user_ids: Set[str], group_ids: Set[str], room_ids: Set[str]) -> bool: - """Three-list gate: users, groups, rooms.""" + """Three-list gate: users, groups, rooms. + + See #18153. + """ if allow_all: return True sid, chat_type = _resolve_chat(source) @@ -286,7 +299,10 @@ def _text_messages(content: str) -> List[Dict[str, Any]]: def build_postback_button_message(text: str, button_label: str, request_id: str) -> Dict[str, Any]: """Slow-LLM postback bubble. Template Buttons stay tappable from history (Quick - Reply chips vanish on the next message). LINE limits: text ≤160, altText ≤400.""" + Reply chips vanish on the next message). LINE limits: text ≤160, altText ≤400. + + See #18153. + """ truncated = text if len(text) <= 160 else text[:157] + "..." alt = text if len(text) <= 400 else text[:397] + "..." action = { @@ -618,6 +634,8 @@ class LineAdapter(BasePlatformAdapter): if pending_rid and not _is_system_bypass(content): self._cache.set_ready(pending_rid, content) return SendResult(success=True, message_id=pending_rid) + # System busy-acks (interrupting / queued / steered) bypass the postback cache and route directly to + # LINE so they reach the user as visible bubbles. Source: PR #18153. return await self._send_text_chunks(chat_id, content, force_push=False) async def _send_text_chunks(self, chat_id: str, content: str, *, force_push: bool) -> SendResult: @@ -733,7 +751,13 @@ class LineAdapter(BasePlatformAdapter): async def _handle_media(self, request) -> Any: """Serve a registered local file for LINE's media URLs. Defence-in-depth: the resolved - path is rechecked against allowed roots (tempdir, ``/tmp``→``/private/tmp`` on macOS, HERMES_HOME).""" + path is rechecked against allowed roots (tempdir, ``/tmp``→``/private/tmp`` on macOS, HERMES_HOME). + + Defence-in-depth: even though ``_register_media`` is only called from trusted internal code, we + recheck the resolved path against an allowed-roots set before serving. Sources allowed: + ``tempfile.gettempdir()``, ``/tmp`` (which resolves to ``/private/tmp`` on macOS), and + ``HERMES_HOME``. PR #8398. + """ from aiohttp import web token = request.match_info["token"] file_path, expires_at = self._media_tokens.get(token) or ("", 0.0) @@ -785,6 +809,7 @@ class LineAdapter(BasePlatformAdapter): if err: return err # LINE requires previewImageUrl: use the supplied preview, else a stdlib 1×1 PNG. + # Use one if supplied, otherwise write a stdlib 1×1 PNG to /tmp and serve it. PR #8398. if preview_path and Path(preview_path).is_file(): preview_url = self._serve_file(Path(preview_path)) else: diff --git a/plugins/platforms/matrix/adapter.py b/plugins/platforms/matrix/adapter.py index 41bdd26374..6f435bd039 100644 --- a/plugins/platforms/matrix/adapter.py +++ b/plugins/platforms/matrix/adapter.py @@ -327,6 +327,9 @@ _MatrixModelPickerPrompt = _MatrixChoicePickerPrompt = _MatrixPickerPrompt # Spec allows ~65 KB events; 4000 was too small (split Markdown tables mid-row). +# Matrix message size limit. The spec allows large events (~65 KB), but very large bodies can render poorly +# in some clients. The previous 4,000-char default was overly conservative and split Markdown tables mid-row +# (#53026). DEFAULT_MAX_MESSAGE_LENGTH = 16000 MATRIX_MAX_MESSAGE_LENGTH_CEILING = 65535 @@ -354,6 +357,7 @@ MAX_MESSAGE_LENGTH = DEFAULT_MAX_MESSAGE_LENGTH # back-compat alias for importe # E2EE store dir is resolved per adapter in connect() (``_resolve_store_dir``), NOT at module scope: # the multiplex gateway imports this once and a module constant would collide every profile's Olm # identity in one crypto.db. +# Store directory for E2EE keys and sync state. Mirrors the pairing-store fix (a6397c379). See #89168. from hermes_constants import get_hermes_dir as _get_hermes_dir _STARTUP_GRACE_SECONDS = 5 # ignore messages older than this many seconds before startup @@ -434,7 +438,14 @@ def _create_matrix_session(proxy_url: str | None): def _check_e2ee_deps() -> bool: """True if all four E2EE deps import: olm, PgCryptoStore (also drives sqlite), asyncpg, aiosqlite. - Without all four, encrypted rooms fail at connect with ``No module named 'asyncpg'``.""" + Without all four, encrypted rooms fail at connect with ``No module named 'asyncpg'``. + + Verifies python-olm (via mautrix.crypto.OlmMachine), the SQLite crypto store backend + (mautrix.crypto.store.asyncpg.PgCryptoStore — yes, the PgCryptoStore class also drives the sqlite + backend in mautrix 0.21), and the database drivers actually used at connect time (``asyncpg`` for the + underlying upgrade_table machinery, ``aiosqlite`` for the ``sqlite:///`` URL we pass to + ``Database.create``). See #31116. + """ try: from mautrix.crypto import OlmMachine # noqa: F401 from mautrix.crypto.store.asyncpg import PgCryptoStore # noqa: F401 @@ -561,7 +572,12 @@ def _handle_generated_matrix_recovery_key(mxid: str, recovery_key: str) -> None: def _scoped_recovery_key() -> str: """MATRIX_RECOVERY_KEY via the profile-scoped secret store (see _startup_env_secret): a bare os.getenv under multiplex resolves the default profile's key and verification fails with - "Key MAC does not match".""" + "Key MAC does not match". + + We read through :func:`get_secret`, which is scope-aware. An *unscoped* read under multiplex (e.g. the + default-profile startup loop) raises ``UnscopedSecretError``; in that context ``os.environ`` is that + profile's own value, so we fall back to it — mirroring the established Slack app-token pattern (#59739). + """ return _startup_env_secret("MATRIX_RECOVERY_KEY") @@ -597,7 +613,10 @@ def _pre_sanitize_matrix_markdown(text: str) -> str: def _startup_env_secret(name: str) -> str: """Scope-aware credential read: a scoped miss is empty (never borrow the process env); - only an UNSCOPED read (default-profile startup loop) falls back to os.environ.""" + only an UNSCOPED read (default-profile startup loop) falls back to os.environ. + + See #59739. + """ try: return (get_secret(name) or "").strip() except UnscopedSecretError: @@ -605,7 +624,12 @@ def _startup_env_secret(name: str) -> str: def matrix_deps_present() -> bool: - """PASSIVE registry ``check_fn`` — must never install; ``ensure_matrix_deps`` is the installer.""" + """PASSIVE registry ``check_fn`` — must never install; ``ensure_matrix_deps`` is the installer. + + Registry ``check_fn`` — called from status displays and config loading, so it must never install + anything. The ACTIVE lazy-installer (``check_matrix_requirements``) is registered as ``ensure_deps_fn`` + and runs from ``create_adapter()`` when this returns False (#79812). + """ try: from tools.lazy_deps import is_available return is_available("platform.matrix") @@ -630,7 +654,13 @@ def check_matrix_requirements() -> bool: def ensure_matrix_deps() -> bool: """ACTIVE deps-only installer (registry ``ensure_deps_fn``); rebinds the type globals. Installs the whole ``platform.matrix`` group when ANY declared package is missing — short-circuiting on - ``import mautrix`` left asyncpg/aiosqlite uninstalled forever.""" + ``import mautrix`` left asyncpg/aiosqlite uninstalled forever. + + Lazy-installs the full ``platform.matrix`` feature group via ``tools.lazy_deps.ensure_and_bind`` + whenever any of the declared packages (mautrix, Markdown, aiosqlite, asyncpg, aiohttp-socks) is missing + — not just mautrix itself. Previously this short-circuited on ``import mautrix``, which left the other + four packages uninstalled forever and broke E2EE connect with ``No module named 'asyncpg'`` (#31116). + """ try: from tools.lazy_deps import feature_missing, ensure_and_bind missing = feature_missing("platform.matrix") @@ -1154,6 +1184,9 @@ class MatrixAdapter(BasePlatformAdapter): async def _verify_or_bootstrap_cross_signing(self, olm: Any, client: Any) -> None: """Verify cross-signing via MATRIX_RECOVERY_KEY, or bootstrap a new key (non-fatal).""" + # Honor the active profile's secret scope so a secondary profile under gateway.multiplex_profiles + # resolves its own recovery key instead of the default profile's (which fails E2EE verification with + # "Key MAC does not match", #69090). recovery_key = _scoped_recovery_key() if recovery_key: try: @@ -1771,7 +1804,13 @@ class MatrixAdapter(BasePlatformAdapter): def _is_self_sender(self, sender: str) -> bool: """True if *sender* is the bot itself (case-insensitive: homeservers vary localpart case). With no resolved user_id we can't prove a sender is NOT us, so return True — dropping our own - events beats an echo loop ("hall of mirrors").""" + events beats an echo loop ("hall of mirrors"). + + Matrix user IDs are byte-compared after trimming whitespace and lowercasing — some homeservers + normalize the localpart case differently at different API surfaces, and the reply-loop tail of the + "hall of mirrors" bug (#15763) has been observed with the bot's own account bypassing a + case-sensitive equality check. + """ own = (self._user_id or "").strip().lower() return not own or sender.strip().lower() == own @@ -1779,7 +1818,13 @@ class MatrixAdapter(BasePlatformAdapter): def _is_system_or_bridge_sender(sender: str) -> bool: """True for appservice/bridge/system identities (``@_telegram_123:server``) or malformed IDs. Never offer these a pairing code: an approved bridge would relay every outbound message - back as an "authorized user message" (echo loop).""" + back as an "authorized user message" (echo loop). + + We treat these as system identities for pairing purposes: they should never be offered a pairing + code, because an operator approving the code would hand the bridge itself permanent authorization — + and every outbound message relayed by the bridge would then loop back into the agent as an + "authorized user message", which is the root of issue #15763. + """ localpart = (sender or "").strip().lstrip("@").partition(":")[0] return not localpart or localpart.startswith("_") @@ -1795,6 +1840,11 @@ class MatrixAdapter(BasePlatformAdapter): def _reset_clock_skew_detector(self) -> None: """State for _note_late_grace_drop: consecutive-drop count, their skew, and the once-only warning.""" + # Clock-skew detection: count grace-check drops that happen well after startup (i.e. not + # initial-sync backfill). If the host's system clock is set ahead of real time, the startup grace + # check `event_ts < startup_ts - 5` silently drops every live message. See #12614 — the symptom is + # "bot joins rooms but never replies". Drops only count when their skew matches the first sampled + # drop (within 60s), so varied-age backfill from freshly-invited rooms doesn't trip the heuristic. self._late_grace_drops: int = 0 self._late_grace_skew: float = 0.0 self._clock_skew_warned: bool = False @@ -1831,6 +1881,10 @@ class MatrixAdapter(BasePlatformAdapter): if self._is_self_sender(sender): return # Bridge/system identities must never reach the pairing flow (echo loop once paired). + # Ignore own messages (case-insensitive; also drops when our own user_id hasn't been resolved yet — + # see _is_self_sender docstring and issue #15763). + # Once a bridge user is paired, every outbound message it relays would loop back as an authorized + # user message (the "hall of mirrors" in #15763). if self._is_system_or_bridge_sender(sender): logger.debug("Matrix: ignoring system/bridge sender %s in %s", sender, room_id) return @@ -2913,7 +2967,12 @@ _YAML_LIST_KEYS = ( def _apply_yaml_config(yaml_cfg: dict, matrix_cfg: dict) -> dict | None: """apply_yaml_config_fn: config.yaml matrix: keys → MATRIX_* env (env wins). Returns None. Lowercased - flags apply whenever the key is present (None still writes "none"); list-valued keys skip None.""" + flags apply whenever the key is present (None still writes "none"); list-valued keys skip None. + + Implements the apply_yaml_config_fn contract (#24849). Mirrors the legacy matrix_cfg block from + gateway/config.py::load_gateway_config(). Env vars take precedence over YAML. Returns None — everything + flows through env. + """ for key, env_name in _YAML_LOWER_KEYS: if key in matrix_cfg and not os.getenv(env_name): os.environ[env_name] = str(matrix_cfg[key]).lower() diff --git a/plugins/platforms/mattermost/adapter.py b/plugins/platforms/mattermost/adapter.py index 53c6d17aba..95d48719c6 100644 --- a/plugins/platforms/mattermost/adapter.py +++ b/plugins/platforms/mattermost/adapter.py @@ -447,6 +447,11 @@ class MattermostAdapter(BasePlatformAdapter): # healthy with a dead listener). Type-based: substring "401" matching misclassified transient errors. if isinstance(exc, aiohttp.WSServerHandshakeError) and exc.status in {401, 403}: logger.error("Mattermost WS auth failed (HTTP %d) — stopping reconnect", exc.status) + # Escalate through the fatal-error hook instead of a bare return: the old silent exit + # left _running True, so is_connected() kept reporting healthy while the listener was + # dead and the gateway was never told (OOF-156 class). Type-based only — the substring + # fallback that used to sit below this branch misclassified transient errors whose + # message merely contained "401" (#80489). self._set_fatal_error( "mattermost_auth_error", f"Mattermost WebSocket authentication rejected (HTTP {exc.status}). The bot token is " @@ -706,6 +711,10 @@ def _apply_yaml_config(yaml_cfg: dict, mattermost_cfg: dict) -> dict | None: Env vars win over YAML (writes guarded by ``not os.getenv``). Under a multiplexed secondary profile the env write is skipped (it would leak into every profile via ``os.environ``); the values are returned so the caller seeds this profile's ``extra``, which read sites check first. + + Implements the ``apply_yaml_config_fn`` contract (#24836 / #25443). Mirrors the legacy + ``mattermost_cfg`` block that used to live in ``gateway/config.py::load_gateway_config()`` before this + migration. """ skip_env_bridge = _profile_scoped_config_load() seeded: dict = {} diff --git a/plugins/platforms/photon/adapter.py b/plugins/platforms/photon/adapter.py index fcbec01b13..3c90c6f97a 100644 --- a/plugins/platforms/photon/adapter.py +++ b/plugins/platforms/photon/adapter.py @@ -55,6 +55,9 @@ _DEFAULT_SIDECAR_BIND = "127.0.0.1" _MAX_MESSAGE_LENGTH = 8000 # iMessage caps practical size at ~16 KB; conservative, matches BlueBubbles # Out-of-process senders (cron, `hermes send`) need the live sidecar's port + spawn-time # token; persisted once /healthz passes, removed on every stop / failed-start path. +# --------------------------------------------------------------------------- Sidecar runtime record The +# gateway persists this record once the sidecar passes its /healthz readiness check, and removes it on every +# stop / failed-start path so a stale record never outlives a dead sidecar. See #69960. _RUNTIME_RECORD_NAME = "photon-sidecar.json" _DEDUP_MAX_SIZE = 4000 # the gRPC stream is at-least-once and a reconnect can replay _DEDUP_WINDOW_SECONDS = 48 * 3600 @@ -1024,6 +1027,20 @@ class PhotonAdapter(BasePlatformAdapter): # CancelledError into the fatal handler before the reconnect is queued, leaving # Photon permanently dead — so let it finish exiting on its own. if self._sidecar_supervisor_task is not asyncio.current_task(): + # _stop_sidecar() is called both from external cleanup (Gateway shutdown, explicit + # disconnect) AND, indirectly, from WITHIN the supervisor task's own crash-handling + # chain: _supervise_sidecar() detects the sidecar exit, calls _set_fatal_error() + + # self._notify_fatal_error(), which the Gateway's fatal-error handler answers by calling + # adapter.disconnect() -> this same _stop_sidecar(). In that second case, + # self._sidecar_supervisor_task IS the currently-running task. Cancelling it raises + # CancelledError into its own call stack (at the next await point in + # _notify_fatal_error() or here), which aborts the fatal-error handler before the + # Gateway ever reaches the "queue for background reconnection" step -- Photon then stays + # permanently dead until a manual restart, since asyncio.CancelledError inherits from + # BaseException (not Exception) and isn't caught by the handler's `except Exception` + # guards (issue #73159). A task cannot legally cancel itself anyway (the cancellation + # would only take effect at its own next await, which is exactly the corruption + # described above), so skip it here and let the task finish exiting on its own instead. self._sidecar_supervisor_task.cancel() self._sidecar_supervisor_task = None @@ -1301,7 +1318,10 @@ class PhotonAdapter(BasePlatformAdapter): @staticmethod def _is_permanent_sidecar_failure(result: SendResult) -> bool: """``auth_or_config`` / ``target_not_allowed`` can't be fixed by retrying or by the - plain-text resend — either would just double-send a doomed request.""" + plain-text resend — either would just double-send a doomed request. + + See #50971. + """ raw = result.raw_response return (isinstance(raw, dict) and raw.get("retryable") is False and raw.get("error_class") in ("auth_or_config", "target_not_allowed")) @@ -1493,6 +1513,7 @@ def _standalone_error(resp: Any) -> Dict[str, Any]: def _standalone_token_from_record(port: int) -> Tuple[Optional[str], int, str]: """``(token, port, error)`` from the runtime record the gateway persists once the sidecar passes /healthz — the token otherwise exists only in the gateway env.""" + # See #69960. record = _read_runtime_record() stale_hint = "" if record and record.get("token"): diff --git a/plugins/platforms/photon/cli.py b/plugins/platforms/photon/cli.py index f2c5232042..8f704d6053 100644 --- a/plugins/platforms/photon/cli.py +++ b/plugins/platforms/photon/cli.py @@ -121,6 +121,11 @@ def _setup_credentials(token: str, dashboard_id: str, name: str) -> Optional[str id *is* the Spectrum id. A valid existing secret is reused: regenerating breaks a running sidecar's sends until restart. Returns the secret or None.""" try: + # 3. Spectrum is always enabled and provisioned at create-time, and the dashboard project id *is* + # the Spectrum project id (ids unified), so there's nothing to enable — the id we already have is + # the Spectrum id. Regenerating invalidates the credential that a running sidecar holds in its + # process env, causing all outbound sends to fail with AuthenticationError until the gateway is + # restarted (GH #50755). print("[3/5] Provisioning Spectrum credentials...") existing_id, existing_secret = photon_auth.load_project_credentials() secret: str = "" diff --git a/plugins/platforms/raft/adapter.py b/plugins/platforms/raft/adapter.py index 7f833f539c..617a4e07fc 100644 --- a/plugins/platforms/raft/adapter.py +++ b/plugins/platforms/raft/adapter.py @@ -509,7 +509,13 @@ def _is_connected(config: PlatformConfig) -> bool: def _env_enablement() -> Optional[dict]: - """Auto-enable during gateway config load when the scope-aware RAFT_PROFILE is set.""" + """Auto-enable during gateway config load when the scope-aware RAFT_PROFILE is set. + + Auto-enables when RAFT_PROFILE is set (the adapter needs it anyway). Scope-aware: consults the active + profile's own RAFT_PROFILE (env, or a secondary profile's own .env via the secret scope) instead of the + default profile's bridged env value (mirrors the Buzz/SimpleX fix for 98738) — see + ``_resolve_raft_profile``. See #98738. + """ return {"enabled": True} if _resolve_raft_profile() else None diff --git a/plugins/platforms/simplex/adapter.py b/plugins/platforms/simplex/adapter.py index f21183b45e..4106683b21 100644 --- a/plugins/platforms/simplex/adapter.py +++ b/plugins/platforms/simplex/adapter.py @@ -107,6 +107,9 @@ class SimplexAdapter(BasePlatformAdapter): self.auto_accept = bool(extra.get("auto_accept", True)) # Without SIMPLEX_GROUP_ALLOWED group messages are ignored (safer default); ``*`` = any group. group_allowed_str = _get_scoped_secret("SIMPLEX_GROUP_ALLOWED", "") or extra.get("group_allowed", "") + # Parse allowlists — group policy is derived from presence of group allowlist Scoped reads (#93522): + # allowlists are per-profile authorization config; raw os.getenv misses secondary profiles' .env + # values and leaks the default profile's list into them. self.group_allow_from = set(_parse_comma_list(group_allowed_str)) self._ws = None # websockets connection self._ws_task: Optional[asyncio.Task] = None diff --git a/plugins/platforms/slack/adapter.py b/plugins/platforms/slack/adapter.py index 8457a2a1a8..0ad49182de 100644 --- a/plugins/platforms/slack/adapter.py +++ b/plugins/platforms/slack/adapter.py @@ -220,15 +220,24 @@ class _ThreadContextCache: message_count: int = 0 parent_text: str = "" # root text, for mention wake checks # Root author ("" unknown): lets _bot_authored_thread_root spot roots posted outside send(). + # The Slack user_id of the thread parent message author. Used by _bot_authored_thread_root (#63530) to + # detect threads whose root was posted by the bot via direct chat.postMessage (outside the gateway's + # send() path). Empty string when the parent could not be fetched or did not have a user_id field. parent_user_id: str = "" # Raw conversations.replies payloads so a watermark (``after_ts``) re-format needs no API call. + # Kept so context can be re-formatted with a different watermark (``after_ts``) without an extra API + # call (#23918). messages: List[Dict[str, Any]] = field(default_factory=list) def slack_deps_present() -> bool: """PASSIVE probe: are slack-bolt/slack-sdk importable right now? Registry ``check_fn`` (status displays, config loading) — must never install. The active - installer is ``check_slack_requirements`` (``ensure_deps_fn``).""" + installer is ``check_slack_requirements`` (``ensure_deps_fn``). + + The ACTIVE lazy-installer (``check_slack_requirements``) is registered as ``ensure_deps_fn`` and runs + from ``create_adapter()`` when this returns False (#79812). + """ return SLACK_AVAILABLE @@ -264,7 +273,12 @@ def check_slack_requirements() -> bool: def _collect_slack_block_mentions(blocks: list) -> list: """``<@UID>`` mentions authored in non-quoted Block Kit text (flat ``text`` omits block-only - mentions); ``rich_text_quote`` is ignored so quoted/forwarded text can't summon the bot.""" + mentions); ``rich_text_quote`` is ignored so quoted/forwarded text can't summon the bot. + + Slack's flat top-level ``text`` field does NOT contain mentions that were authored only inside Block Kit + ``blocks`` (e.g. a ``rich_text_section`` with a ``user`` element). This walker recovers those mentions + so the gates can see Block-Kit-only mentions instead of silently dropping them (#52387). + """ mentions: list = [] def _walk(node, in_quote: bool) -> None: @@ -291,7 +305,12 @@ def _collect_slack_block_mentions(blocks: list) -> list: def _slack_mention_detection_text(event: dict) -> str: - """Text for @mention detection: flat ``text`` plus non-quoted Block-Kit-only mentions.""" + """Text for @mention detection: flat ``text`` plus non-quoted Block-Kit-only mentions. + + Combines the flat top-level ``text`` with any ``<@UID>`` mentions recovered from non-quoted Block Kit + blocks (#52387), so a genuine Block-Kit-only mention reaches the gates while quoted/forwarded mentions + stay ignored. + """ flat = event.get("text", "") or "" blocks = event.get("blocks") extra = [m for m in _collect_slack_block_mentions(blocks) if m not in flat] if blocks else [] @@ -683,7 +702,10 @@ def _resolve_slack_proxy_url() -> Optional[str]: def _slack_dedup_ttl_seconds() -> float: """Dedup window for Socket Mode replays (override: ``SLACK_DEDUP_TTL_SECONDS``). Slack replays un-acked events on reconnect, sometimes minutes later, so the window must span the - worst-case gap; memory is bounded by ``MessageDeduplicator(max_size=...)``, not the TTL.""" + worst-case gap; memory is bounded by ``MessageDeduplicator(max_size=...)``, not the TTL. + + See #4777. + """ raw = os.getenv("SLACK_DEDUP_TTL_SECONDS", "") if raw: try: @@ -868,6 +890,8 @@ class SlackAdapter(BasePlatformAdapter): self._dm_conversation_cache: Dict[str, str] = {} # Dedup for Socket Mode reconnect replays; TTL must outlast the worst-case # redelivery gap (max_size bounds memory, so a long window is safe). + # Dedup cache: prevents duplicate bot responses when Socket Mode reconnects redeliver events + # (#4777). self._dedup = MessageDeduplicator(ttl_seconds=_slack_dedup_ttl_seconds()) # ts of messages already routed to the agent, so later edits don't re-trigger a reply. self._processed_message_ts: Dict[str, float] = {} @@ -886,11 +910,17 @@ class SlackAdapter(BasePlatformAdapter): self._agent_view_contexts: Dict[Tuple[str, str], Dict[str, str]] = {} # (channel, thread, status key) → last status bubble ts, so repeated # progress callbacks edit ONE message instead of spamming the thread. + # Status-bubble dedup (issue #30045, extended to Slack): remember the message ts of the last status + # bubble per (channel, thread, status key) so repeated progress callbacks (compression retries, + # fallback switches, ...) edit ONE message in place instead of appending a new bubble per event — + # long retry loops used to spam threads with dozens of out-of-order status messages. self._status_message_ids: Dict[Tuple[str, str, str], str] = {} self._thread_context_cache: Dict[str, _ThreadContextCache] = {} # Threads already rehydration-checked this process (first reply after a restart injects # missed messages exactly once); message IDs with reaction lifecycle (bounded: an exception # between add and finalize would leak entries). + # Persistent sessions survive gateway restarts, but messages that arrived while the gateway was DOWN + # never reached the session. Keys follow the thread session-key scoping. See #63530. self._thread_rehydration_checked: set = set() self._reacting_message_ids: set = set() # Active Assistant statuses by (team_id, channel_id, thread_ts) so cleanup @@ -950,7 +980,11 @@ class SlackAdapter(BasePlatformAdapter): cls, entries: Any, count: int, ts_getter: Callable[[Any], Any] = lambda e: e) -> None: """Discard the *count* entries (set or dict keys) with the oldest embedded Slack ts. Sets iterate in arbitrary order, so ``list(entries)[:count]`` could evict the most ACTIVE - entry; sort chronologically by the embedded ts instead.""" + entry; sort chronologically by the embedded ts instead. + + For bounded tracking sets whose members are keys CONTAINING a Slack timestamp (tuples or + colon-joined strings) rather than bare ts values. See #51019. + """ if count <= 0: return oldest = sorted(entries, key=lambda e: cls._slack_timestamp_sort_key(ts_getter(e)))[:count] @@ -969,6 +1003,9 @@ class SlackAdapter(BasePlatformAdapter): def _trim_mentioned_threads(self) -> None: if len(self._mentioned_threads) > self._MENTIONED_THREADS_MAX: + # Keys are "team:channel:thread_ts[:user]" — evict the oldest threads first. Evicting an ACTIVE + # thread's key would re-run its rehydration check and re-inject the missed delta (#51019-style + # arbitrary eviction), so never pop in set order. self._discard_oldest_by_thread_ts( self._mentioned_threads, self._MENTIONED_THREADS_MAX // 2) @@ -976,7 +1013,11 @@ class SlackAdapter(BasePlatformAdapter): def _trim_oldest_dict_entries(mapping: Dict[Any, Any], max_size: int) -> None: """Evict oldest-inserted entries down to half the cap once *mapping* exceeds *max_size*. Dict insertion order makes ``list(mapping)[:excess]`` truly oldest-first (sets would not - be); halving amortizes eviction like the sibling caches.""" + be); halving amortizes eviction like the sibling caches. + + Evicts down to half the cap so eviction runs amortized-once per max_size//2 writes, matching the + sibling tracking structures. See #51019. + """ if len(mapping) <= max_size: return excess = len(mapping) - max_size // 2 @@ -1026,7 +1067,14 @@ class SlackAdapter(BasePlatformAdapter): ``while True`` retry loop that never checks ``closed``, so anything inside it when ``close_async()`` drops the session retries forever. Cancel every task that can reach ``connect()`` BEFORE closing (it rebinds task attrs on success, so a mid-close snapshot - races a moving target).""" + races a moving target). + + Everything that can reach ``connect()`` therefore has to be stopped first. + ``monitor_current_session()`` and ``receive_messages()`` each get there on their own, and + ``connect()`` rebinds the client's task attributes on success, so the set of live tasks changes + across the awaits inside ``close()``. Cancelling from a snapshot taken partway through that would + race a moving target. See slackapi/python-slack-sdk#1913. + """ handler, task = self._handler, self._socket_mode_task self._handler = self._socket_mode_task = None client = getattr(handler, "client", None) @@ -1252,7 +1300,15 @@ class SlackAdapter(BasePlatformAdapter): async def _send_slash_ephemeral(self, ctx: Dict[str, Any], content: str) -> "SendResult": """Replace the ephemeral ack via ``response_url`` (``replace_original`` valid 30 min). First chunk replaces the ack, the rest post as new ephemerals; Slack caps a response_url at 5 - POSTs so overflow gets a truncation notice. ``success=False`` lets ``send()`` fall back.""" + POSTs so overflow gets a truncation notice. ``success=False`` lets ``send()`` fall back. + + Long replies are chunked: the first chunk replaces the ack, the rest are posted as additional + ephemeral messages. Slack allows at most 5 POSTs to a response_url, so anything beyond that is + closed with an explicit truncation notice instead of being silently dropped (#19688). + Returns ``success=False`` on delivery failure so the caller (``send()``) can fall back to normal + channel delivery — the reply must never be silently dropped just because the ephemeral swap failed + (#19688). + """ # Slack's response_url has the same ~40k char limit as chat_postMessage. chunks = self._format_chunks(content) # 5-POST cap per response_url: 1 replace + 4 follow-ups; announce the rest. @@ -1286,7 +1342,10 @@ class SlackAdapter(BasePlatformAdapter): self, chat_id: str, ctx: Dict[str, Any], content: str) -> "SendResult": """Deliver a slash reply via ``chat.postEphemeral`` when ``response_url`` fails. Keeps the reply private (a public channel post must never happen for an ephemeral reply). - Cannot ``replace_original``, so the ack stays; no 5-POST cap applies here.""" + Cannot ``replace_original``, so the ack stays; no 5-POST cap applies here. + + See #19688. + """ user_id = ctx.get("user_id", "") if not user_id: return SendResult(success=False, error="no user_id in slash context for postEphemeral") @@ -1395,6 +1454,20 @@ class SlackAdapter(BasePlatformAdapter): self._app.event(event_type)(_listener_for(handler)) # Catch-all ack: unacked envelopes count as failures and past 95%/60-min Slack disables # Event Subscriptions (ALL inbound). Registered AFTER all named handlers (first match wins). + # Catch-all no-op ack for any other subscribed event type that Hermes has no listener for (e.g. + # user_change, user_huddle_changed, member_joined_channel, channel_archive, pin_added, etc.). Two + # reasons this must exist (issues #6572 and the Event Subscriptions auto-disable failure mode): 1. + # Correctness at scale: without a matching listener, slack-bolt returns HTTP 404 for every unhandled + # event envelope and never sends the Socket Mode ack. When the app is subscribed to high-volume + # events (user_change fires on every presence/status change for the whole org), the flood of + # un-acked 404s pushes Slack's failure rate past its 95%/60-min threshold and Slack auto-disables + # the app's Event Subscriptions — silently killing ALL inbound delivery until manually re-enabled. + # 2. Noise: each unhandled envelope also logs a slack_bolt "Unhandled request" WARNING, flooding + # gateway logs in busy channels. Registered AFTER every named handler: bolt dispatches to the first + # matching listener, so the named handlers above always win and this only fires for truly unhandled + # types. The envelope is acked with 200, keeping the failure rate near 0% regardless of which events + # the Slack app manifest subscribes to. A debug line preserves visibility into unknown event types + # without per-message WARNING noise. @self._app.event(re.compile(r".*")) async def handle_unhandled_event(event, body, logger): logger.debug( @@ -1507,6 +1580,11 @@ class SlackAdapter(BasePlatformAdapter): # Scoped secret is authoritative; only an UNSCOPED read falls back to # process env, else a secondary profile inherits the default's app. try: + # Multiplex: profile secrets live in the secret scope, not process os.environ. When a scope is + # installed (secondary-profile connect), it is AUTHORITATIVE — do not fall through to os.getenv, + # or a secondary profile missing SLACK_APP_TOKEN silently inherits the default profile's Socket + # Mode app (#59739). Only an UNSCOPED read under multiplex (default-profile startup loop, + # background reconnect rebuild) falls back to process env, which is that profile's own. app_token = get_secret("SLACK_APP_TOKEN") except UnscopedSecretError: app_token = os.getenv("SLACK_APP_TOKEN") @@ -1530,6 +1608,11 @@ class SlackAdapter(BasePlatformAdapter): # A zombie Socket Mode handler would double-respond to every event. await self._stop_socket_mode_handler() await self._close_workspace_clients() + # Close any previous handler before creating a new one so that calling connect() a second time + # (e.g. during a gateway restart or in-process reconnect attempt) does not leave a zombie Socket + # Mode connection alive. Both the old and new connections would otherwise receive every Slack + # event and dispatch it twice, producing double responses — the same bug that affected + # DiscordAdapter (#18187). self._app = None self._app_token = app_token self._proxy_url = proxy_url @@ -1580,6 +1663,16 @@ class SlackAdapter(BasePlatformAdapter): def _hint_allow_bots(self) -> None: """INFO hint: bot events can be swallowed upstream of allow_bots (manifest, allowlist).""" + # Bot-event interop diagnostic. When the user has opted into bot messages via ``slack.allow_bots`` / + # ``SLACK_ALLOW_BOTS``, surface the additional plumbing they almost certainly also need so + # bot-to-bot interop doesn't silently fail. See #30091: a user reported that with ``allow_bots: + # all`` configured, bot messages in shared threads were still dropped. Two things upstream of this + # code can swallow them: 1. The Slack app's event subscriptions in the manifest — Socket Mode does + # not deliver events the app hasn't subscribed to (``message.channels`` for public channels, + # ``message.groups`` for private channels, ``message.im`` for DMs). 2. The SLACK_ALLOWED_USERS / + # GATEWAY_ALLOWED_USERS per-user allowlists — the other bot's user id must be present (or + # GATEWAY_ALLOW_ALL_USERS=true). Logging once at INFO keeps the startup line discoverable without + # requiring DEBUG to enable. _allow_bots_cfg = self._slack_allow_bots() if _allow_bots_cfg != "none": logger.info( @@ -1687,7 +1780,12 @@ class SlackAdapter(BasePlatformAdapter): async def _ensure_dm_conversation(self, chat_id: str, team_id: Optional[str] = None) -> str: """Resolve a bare user ID (U/W...) to a DM conversation ID via ``conversations.open`` (``chat.postMessage``/``files_upload_v2`` reject user IDs); cached per (team, user). Returns - ``chat_id`` unchanged when not applicable or on failure (downstream surfaces the error).""" + ``chat_id`` unchanged when not applicable or on failure (downstream surfaces the error). + + Resolution goes through the workspace-scoped client so multi-workspace installs open the DM with the + right bot token, and results are cached per (team, user) so repeated sends don't re-open. See + #17261, #19236. + """ cid = str(chat_id or "") if not cid or cid[0] not in ("U", "W"): return chat_id @@ -1715,7 +1813,14 @@ class SlackAdapter(BasePlatformAdapter): self, chat_id: str, metadata: Optional[Dict[str, Any]] = None) -> None: """Best-effort status clear for send() paths that skip the normal clear (empty responses, ephemeral slash replies, exceptions before ``thread_ts`` resolved) so the thread doesn't - stay "is thinking...". Errors must not mask the SendResult.""" + stay "is thinking...". Errors must not mask the SendResult. + + Issue #24117: the assistant thread can stay stuck "is thinking..." when a turn ends through a path + that never reaches the regular ``if thread_ts: stop_typing`` clear — an empty final response, a + slash-command ephemeral reply, or an exception raised before ``thread_ts`` was resolved. + ``stop_typing`` is already idempotent (clearing an unset status is a no-op on Slack's side), so this + just guarantees it runs without letting a cleanup error mask the caller's SendResult. + """ try: await self.stop_typing(chat_id, metadata=metadata) except Exception as e: # pragma: no cover - defensive cleanup @@ -1902,6 +2007,9 @@ class SlackAdapter(BasePlatformAdapter): # Slack returns ``no_text`` for blank posts; still the end of a # delivery attempt, so the "is thinking..." status must clear. await self._clear_thread_status_quietly(chat_id, metadata) + # This is still the end of a delivery attempt: if the turn produced no visible text (e.g. + # "(empty)" final responses are filtered upstream), the assistant thread status must not + # stay stuck on "is thinking..." (#24117). return SendResult(success=True) thread_ts = self._resolve_thread_ts(reply_to, metadata) last_result = await self._post_chunks(chat_id, team_id, content, formatted, thread_ts) @@ -1921,6 +2029,10 @@ class SlackAdapter(BasePlatformAdapter): # Clear the status even when the failure preceded thread_ts resolution: # stop_typing falls back to metadata / the uniquely tracked status. await self._clear_thread_status_quietly(chat_id, metadata) + # Clear the assistant status even when the failure happened BEFORE thread_ts was resolved + # (formatting, slash-context, DM resolution): stop_typing falls back to metadata / the uniquely + # tracked status for this channel, so a failed turn cannot leave "is thinking..." visible + # (#24117). logger.error("[Slack] Send error: %s", e, exc_info=True) _retryable = self._is_retryable_upload_error(e) return SendResult( @@ -2016,7 +2128,13 @@ class SlackAdapter(BasePlatformAdapter): metadata: Optional[Dict[str, Any]] = None) -> SendResult: """Send a status message or edit the previous one with the same (channel, thread, key) so progress callbacks edit one bubble. If the edit fails (deleted, too old) the cached ts is - dropped and a fresh message is sent.""" + dropped and a fresh message is sent. + + Issue #30045 (Telegram) extended to Slack: progress/status callbacks (context-pressure, compression + retries, model fallback, lifecycle) used to append a fresh bubble on every call, spamming threads + during long retry loops. The first call posts and the message ts is remembered; subsequent calls + with the same (channel, thread, status_key) edit that message in place via ``chat.update``. + """ thread_ts = self._resolve_thread_ts(None, metadata) or "" key = (str(chat_id), str(thread_ts), str(status_key)) cached_id = self._status_message_ids.get(key) @@ -2273,6 +2391,11 @@ class SlackAdapter(BasePlatformAdapter): if status_key: # Keep the first start time across _keep_typing refreshes so long turns show elapsed # time; stored in the status entry so it shares eviction/stop_typing cleanup. + # Heartbeat (#45702): preserve the first refresh's start time across _keep_typing refreshes so a + # long turn surfaces elapsed time ("still working… (2m03s)") instead of a static "is + # thinking..." that reads as stuck — which is what provokes mid-turn "you there?" pings. Stored + # inside the tracked status entry so it shares the existing bounds/eviction and is dropped by + # stop_typing with the rest of the status state. _prev_entry = self._active_status_threads.get(status_key) if isinstance(_prev_entry, dict): _status_started = _prev_entry.get("started") @@ -2390,7 +2513,14 @@ class SlackAdapter(BasePlatformAdapter): def _slack_api_human_users(self) -> frozenset: """User IDs whose Web-API posts count as human (``extra.api_human_users`` / ``SLACK_API_HUMAN_USERS``): ``xoxp-`` posts carry ``app_id`` and no ``client_msg_id`` so - look like bots. Users only — an app-id allowlist would admit the app's own posts.""" + look like bots. Users only — an app-id allowlist would admit the app's own posts. + + A message posted with a *user* token (``xoxp-``) is authored by a real person, but Slack still + stamps it with the posting ``app_id`` and it carries no ``client_msg_id`` — exactly the #35777 + app/bot signature in ``_event_declares_bot_sender``. Operators running their own front-end + (dashboard, mobile shell) allowlist those *users* via ``platforms.slack.extra.api_human_users`` + (``SLACK_API_HUMAN_USERS`` fallback) instead of ``allow_bots: all``. + """ cached = getattr(self, "_api_human_users_cache", None) if cached is None: raw = self.config.extra.get("api_human_users") @@ -2411,6 +2541,8 @@ class SlackAdapter(BasePlatformAdapter): # App-originated events may lack bot_id/subtype but carry app_id and no client_msg_id # (humans have one) → bot-authored unless the user is in _slack_api_human_users # (classic bot posts have no ``user`` so never match). + # Real human-authored messages normally carry client_msg_id, so treat the combination as + # app/bot-authored (#35777). if event.get("app_id") and not event.get("client_msg_id"): return event.get("user") not in self._slack_api_human_users() return False @@ -2651,7 +2783,11 @@ class SlackAdapter(BasePlatformAdapter): def _maybe_blocks(self, content: str) -> Optional[list]: """Block Kit for ``content``: ``markdown_blocks`` (native block, "platform AI" apps only, 12k cap) over ``rich_blocks`` (local renderer). ``None`` when disabled or declined — a - ``text`` fallback always accompanies blocks, so ``None`` is safe at any point.""" + ``text`` fallback always accompanies blocks, so ``None`` is safe at any point. + + 1. ``markdown_blocks`` — Slack's native ``markdown`` block renders the *raw* standard markdown + (tables, headers, code fences with syntax highlighting) with Slack doing the translation (#8552). 2. + """ if self._extra_flag("markdown_blocks"): md_blocks = self._markdown_block_payload(content) if md_blocks: @@ -3420,6 +3556,9 @@ class SlackAdapter(BasePlatformAdapter): "event_ts": event.get("event_ts")}} if team_id: synthetic["team"] = team_id + # Optional handoff target (#45265): route the reaction-triggered turn into a configured channel (and + # optionally thread) instead of the source thread. A channel-only target is a handoff, not a reply — + # respond top-level there. target_channel, target_thread = self._slack_reaction_trigger_target() if target_channel: synthetic["channel"] = target_channel @@ -3569,7 +3708,14 @@ class SlackAdapter(BasePlatformAdapter): self, channel_id: str, thread_ts: str, team_id: str = "") -> bool: """True when this bot authored the thread root — catches roots posted via direct chat.postMessage (not in _bot_message_ts) and survives restarts. Cache first, then a - TTL-bounded fetch on a miss.""" + TTL-bounded fetch on a miss. + + Used by the wake-decision to detect threads where the bot posted the root via direct + chat.postMessage (outside the gateway's send() path) — see #63530. Without this, human replies in + bot-initiated threads were silently dropped when there was no active session and no @mention. + Root-authorship is derived from the Slack API, so unlike the in-memory _bot_message_ts set it also + survives gateway restarts. + """ if not thread_ts: return False bot_uid = self._team_bot_user_ids.get(team_id, self._bot_user_id) or "" @@ -3593,7 +3739,13 @@ class SlackAdapter(BasePlatformAdapter): team_id: str = "", chat_type: str = "group") -> bool: """Return True if the bot should wake on an un-mentioned message. Checks, in order: root sent via send() (_bot_message_ts); thread previously @-mentioned; active session; - bot-authored root via raw chat.postMessage; thread parent @-mentioned the bot.""" + bot-authored root via raw chat.postMessage; thread parent @-mentioned the bot. + + 1. 2. _mentioned_threads (someone @-mentioned us earlier) 3. _has_active_session... (there's + already an agent session) 4. _bot_authored_thread_root (#63530: the bot posted the thread root via + direct chat.postMessage, outside the gateway send() path — derived from the Slack API, so it also + survives restarts). + """ if not event_thread_ts: return False thread_marker = self._workspace_message_marker(team_id, event_thread_ts) @@ -3612,6 +3764,9 @@ class SlackAdapter(BasePlatformAdapter): channel_id=channel_id, thread_ts=event_thread_ts, team_id=team_id): return True # Thread PARENT @-mentioned the bot before this process (restart): a bare "run" is for us. + # 5th check (#24848): the thread PARENT @-mentioned the bot, but the mention event predates this + # process (restart) or the parent asked the bot to wait for a follow-up (e.g. A plain reply like + # "run" in that thread is addressed to the bot even though the reply itself carries no mention. if is_thread_reply: bot_uid = self._team_bot_user_ids.get(team_id, self._bot_user_id) if bot_uid: @@ -3783,7 +3938,19 @@ class SlackAdapter(BasePlatformAdapter): return thread_ts if event.get("_hermes_no_thread_response"): return event.get("thread_ts") or None + # Reaction handoff into a configured target channel (#45265): the response should be a new top-level + # message in the target channel, never a thread under the synthetic ts (which is the reaction's + # event_ts — not a real message there). + # Channel message session scoping. Three cases: (a) genuine thread reply → scope session per + # thread (b) top-level, reply_in_thread=true (the default) → legacy behaviour: each top-level + # message becomes its own thread, so the UX still "replies in a thread" and sessions are keyed per + # thread root (c) top-level, reply_in_thread=false → scope one session across the whole channel so + # context accumulates across messages (#15421 bug 1) event_thread_ts_raw = event.get("thread_ts") + # Align with ``is_thread_reply`` below — a ``thread_ts == ts`` payload (some thread-root shapes) is + # not a real reply and must not prevent the shared-session path from taking effect. Matching the + # same invariant here keeps the two branches in sync even if Slack introduces new payload variants + # (Copilot on #15464). if event_thread_ts_raw and event_thread_ts_raw != ts: return event_thread_ts_raw if self.config.extra.get("reply_in_thread", True): @@ -3798,7 +3965,18 @@ class SlackAdapter(BasePlatformAdapter): full thread + root images once, set watermark. Session + @mention: delta past watermark (cache bypassed). Session, first plain reply this process: restart rehydration; later replies only advance the watermark. Context goes into the NEW turn only (prompt caching).""" + # - Active thread + explicit @mention: refresh with only the delta since the last hydrate/refresh + # (#23918), bypassing the TTL cache. The delta is injected as part of the NEW turn (via + # ``channel_context``) — prior conversation history is never rewritten, so prompt caching is + # preserved. Keep recovered history separate from ``text``. Prepending it here moves a recognized + # command away from character zero, so downstream command routing can misclassify it as + # conversational text. ``channel_context`` is prepended only after command dispatch. channel_context = None + # Thread-root images recovered on the cold-start hydrate: when the bot is mentioned mid-thread for + # the first time, the thread root is very often the artifact the mention is about ("@bot what's in + # this chart?" replying under an image post) — deliver its images with this first turn. One-time by + # construction: the cold-start path is guarded by _has_active_session_for_thread, so subsequent + # turns in the same session never re-deliver (adapted from #69185). thread_root_media_urls: List[str] = [] thread_root_media_types: List[str] = [] if not is_thread_reply: @@ -3826,6 +4004,12 @@ class SlackAdapter(BasePlatformAdapter): elif is_mentioned: await _fetch(after_ts=self._get_thread_watermark(**watermark_args), force_refresh=True) else: + # Restart rehydration (#63530 restart gap / #33215): persistent sessions survive gateway + # restarts, but thread replies posted while the gateway was down never reached the session. On + # the FIRST ordinary reply per thread in this process, fetch the delta past the persisted + # watermark and inject anything missed as part of this new turn. Checked at most once per thread + # per process; a non-empty watermark plus an empty delta costs one cached conversations.replies + # call. rehydration_key = self._thread_rehydration_key( channel_id, event_thread_ts, user_id, team_id) if rehydration_key in self._thread_rehydration_checked: @@ -3904,6 +4088,7 @@ class SlackAdapter(BasePlatformAdapter): return True if allow_bots == "mentions": # Mentions may live only in Block Kit, not the flat text. + # See #52387. text_check = _slack_mention_detection_text(event) if self._bot_user_id and f"<@{self._bot_user_id}>" not in text_check: logger.debug( @@ -3918,6 +4103,10 @@ class SlackAdapter(BasePlatformAdapter): Returns ``(event, team_id, channel_id)`` for messages the handler should consider.""" # Entry log BEFORE any filtering so operators can tell "dropped here" # from "never subscribed in the manifest". Metadata only, never text. + # DEBUG entry log — fires BEFORE any filtering so users debugging bot-to-bot interop, allow_bots + # config, or SLACK_ALLOWED_USERS drops can confirm whether the event actually arrived from Slack + # (vs. being silently filtered upstream by the app's event subscriptions — Socket Mode will not + # deliver events the app manifest hasn't subscribed to). See #30091. if logger.isEnabledFor(logging.DEBUG): _bot_profile = event.get("bot_profile") or {} logger.debug( @@ -3931,6 +4120,9 @@ class SlackAdapter(BasePlatformAdapter): if event is None: return None # Socket Mode redelivers after reconnects. Scope by workspace: ts is only unique per team. + # Dedup: Slack Socket Mode can redeliver events after reconnects (#4777) Scope the dedup id by + # workspace: Slack event ts values are only unique within one workspace, so two teams' events with + # the same ts must not suppress each other. event_ts = event.get("_slack_changed_event_ts") or event.get("ts", "") dedup_team_id = self._event_team_id(event, payload) if event_ts and self._dedup.is_duplicate(self._workspace_event_id(dedup_team_id, event_ts)): @@ -4042,6 +4234,7 @@ class SlackAdapter(BasePlatformAdapter): thread_ts = self._session_thread_ts(event, ts, is_dm, assistant_meta) bot_uid = self._team_bot_user_ids.get(team_id, self._bot_user_id) # Mentions may live only in Block Kit blocks. + # See #52387. routing_text = _slack_mention_detection_text(event) or original_text or "" is_mentioned = bool( (bot_uid and f"<@{bot_uid}>" in routing_text) @@ -4469,6 +4662,10 @@ class SlackAdapter(BasePlatformAdapter): # Preferred: the injected profile-bound check (``set_authorization_check``); unlike the # ``__self__`` introspection below it works under multiplex (handler is a closure). # getattr: object.__new__ test doubles never ran BasePlatformAdapter.__init__. + # Preferred path: the auth callback GatewayRunner injects at connect time + # (``set_authorization_check``) runs the full, profile-bound ``_is_user_authorized`` chain. Unlike + # the ``__self__`` introspection below it also resolves on a multiplexed adapter, whose message + # handler is a profile closure with no ``__self__`` (#72657, same class as Telegram's #86296). if getattr(self, "_authorization_check", None) is not None: injected = self._is_sender_authorized( normalized_user_id, chat_type, str(channel_id or "")) @@ -4768,7 +4965,11 @@ class SlackAdapter(BasePlatformAdapter): after_ts: str = "", force_refresh: bool = False) -> str: """Prior thread messages as formatted context ("" on failure/empty). Cold-start only (session history holds them afterwards); ``after_ts`` = session watermark returns only - unseen messages; ``force_refresh`` bypasses the _THREAD_CACHE_TTL cache (Tier 3 API).""" + unseen messages; ``force_refresh`` bypasses the _THREAD_CACHE_TTL cache (Tier 3 API). + + mentioned mid-thread for the first time, or when an explicit @mention on an active thread requests a + context refresh (#23918). + """ cache_key = self._thread_cache_key(channel_id, thread_ts, team_id) now = time.monotonic() cached = None if force_refresh else self._thread_context_cache.get(cache_key) @@ -4852,7 +5053,10 @@ class SlackAdapter(BasePlatformAdapter): channel_id: str, after_ts: str = "") -> Tuple[str, str]: """Format Slack replies into an injected thread-context block. With ``after_ts``, only messages strictly newer than the watermark are included (delta - refresh); parent text is still captured. Returns ``(content, parent_text)``.""" + refresh); parent text is still captured. Returns ``(content, parent_text)``. + + See #23918. + """ bot_uid = self._team_bot_user_ids.get(team_id, self._bot_user_id) context_parts = [] parent_text = "" @@ -4932,7 +5136,11 @@ class SlackAdapter(BasePlatformAdapter): ) -> str: """Return the thread parent's text ("" on any failure). Shares the per-thread cache with :meth:`_fetch_thread_context`; on a cold cache does a - single-message ``conversations.replies`` fetch.""" + single-message ``conversations.replies`` fetch. + + Used to check whether the root mentions the bot (#24848). Set ``strip_bot_mention=False`` to + preserve the mention. + """ cache_key = self._thread_cache_key(channel_id, thread_ts, team_id) now = time.monotonic() cached = self._thread_context_cache.get(cache_key) @@ -5198,6 +5406,7 @@ class SlackAdapter(BasePlatformAdapter): return False # A key the reset policy (daily/idle/suspended) would roll is NOT active: # treating it as such would suppress the first-turn thread-history reseed. + # See #55239. should_reset = getattr(type(session_store), "_should_reset", None) return not (callable(should_reset) and should_reset(session_store, entry, source)) except Exception: @@ -5399,6 +5608,15 @@ class SlackAdapter(BasePlatformAdapter): # Standalone-send cache: user ID -> DM conversation ID, keyed "{token}:{user_id}" (multi-workspace). +# ────────────────────────────────────────────────────────────────────────── Plugin migration glue (#41112 / +# #3823) Everything below this line was added when the Slack adapter moved from +# ``gateway/platforms/slack.py`` into this bundled plugin. It mirrors the Discord migration (PR #24356) +# exactly: a ``register(ctx)`` entry point plus the hook implementations (``_standalone_send``, +# ``interactive_setup``, ``_apply_yaml_config``, ``_is_connected``, ``_build_adapter``) that replace the +# per-platform core touchpoints (the ``Platform.SLACK`` elif in ``gateway/run.py``, the ``slack_cfg`` +# YAML→env block in ``gateway/config.py``, the ``_setup_slack`` wizard + ``_PLATFORMS["slack"]`` static dict +# in ``hermes_cli/{setup,gateway}.py``, and the ``_send_slack`` dispatch in ``tools/send_message_tool.py``). +# ────────────────────────────────────────────────────────────────────────── _slack_dm_cache: Dict[str, str] = {} _SLACK_DM_CACHE_MAX = 5000 @@ -5641,6 +5859,9 @@ async def _standalone_send( return {"error": "Slack send failed: SLACK_BOT_TOKEN not configured"} token = tokens[0] # Slack rejects bare user IDs (U.../W...) with channel_not_found; open the DM first. + # User-targeted delivery: chat.postMessage / files_upload_v2 reject bare user IDs (U.../W...) — resolve + # to a DM conversation ID (D...) first via conversations.open so `deliver=slack:U…` cron jobs reach the + # user's DM instead of failing with channel_not_found (#17444). chat_id = str(chat_id or "") if chat_id[:1] in ("U", "W"): resolved = None @@ -5804,7 +6025,11 @@ _YAML_LIST_KEYS = ( def _apply_yaml_config(yaml_cfg: dict, slack_cfg: dict) -> dict | None: """``apply_yaml_config_fn`` hook: ``slack:`` YAML keys → ``SLACK_*`` env vars (the adapter reads - ``os.getenv()``; explicit env wins). Returns None: nothing is seeded into ``extra``.""" + ``os.getenv()``; explicit env wins). Returns None: nothing is seeded into ``extra``. + + Implements the ``apply_yaml_config_fn`` contract (#24849). Mirrors the legacy ``slack_cfg`` block that + used to live in ``gateway/config.py::load_gateway_config()`` before this migration. + """ for key, env in _YAML_BOOL_KEYS: if key in slack_cfg and not os.getenv(env): os.environ[env] = str(slack_cfg[key]).lower() @@ -5842,6 +6067,11 @@ def register(ctx) -> None: install_hint="Run `hermes setup` to install Slack support.", setup_fn=interactive_setup, # YAML→env bridge: config.yaml slack: keys → SLACK_* env vars read via os.getenv(). + # YAML→env config bridge — owns the translation of config.yaml slack: keys (require_mention, + # strict_mention, ignore_other_user_mentions, thread_require_mention, allow_bots, + # free_response_channels, reactions, disable_dms, allowed_channels, ignored_channels) into SLACK_* + # env vars that the adapter reads via os.getenv(). Replaces the hardcoded block in + # gateway/config.py. Hook contract: #24849. apply_yaml_config_fn=_apply_yaml_config, allowed_users_env="SLACK_ALLOWED_USERS", allow_all_env="SLACK_ALLOW_ALL_USERS", diff --git a/plugins/platforms/sms/adapter.py b/plugins/platforms/sms/adapter.py index 083d032e9e..93fb0b5ea2 100644 --- a/plugins/platforms/sms/adapter.py +++ b/plugins/platforms/sms/adapter.py @@ -127,6 +127,7 @@ class SmsAdapter(BasePlatformAdapter): self._webhook_port) # client_max_size bounds every read path (incl. chunked bodies with no # Content-Length) before the handler's own 413 checks run. + # See #58536, #58902, #59180. app = web.Application(client_max_size=_TWILIO_WEBHOOK_MAX_BODY_BYTES) app.router.add_post("/webhooks/twilio", self._handle_webhook) app.router.add_get("/health", lambda _: web.Response(text="ok")) @@ -280,6 +281,13 @@ _SMS_MARKDOWN_SUBS = ( (re.compile(r"\n{3,}"), "\n\n")) +# ────────────────────────────────────────────────────────────────────────── Plugin migration glue (#41112 / +# #3823) Added when the SMS (Twilio) adapter moved from gateway/platforms/sms.py into this bundled plugin. +# register() exposes the platform via the registry, replacing the Platform.SMS elif in gateway/run.py, the +# _PLATFORM_CONNECTED_CHECKERS entry in gateway/config.py, the _PLATFORMS["sms"] static dict in +# hermes_cli/gateway.py, and the _send_sms dispatch in tools/send_message_tool.py. TWILIO_* +# env→PlatformConfig seeding stays in core. +# ────────────────────────────────────────────────────────────────────────── def _strip_markdown_for_sms(message: str) -> str: """Strip markdown — SMS renders it as literal characters.""" for pattern, repl in _SMS_MARKDOWN_SUBS: diff --git a/plugins/platforms/teams/adapter.py b/plugins/platforms/teams/adapter.py index 1a9cf21123..990918a598 100644 --- a/plugins/platforms/teams/adapter.py +++ b/plugins/platforms/teams/adapter.py @@ -10,6 +10,10 @@ Requires the ``teams`` extra (auto-installed by the gateway on first start, or from __future__ import annotations import asyncio +# microsoft-teams-apps calls ``load_dotenv(find_dotenv(usecwd=True))`` at ``microsoft_teams.apps.app`` +# import time. Importing it during plugin discovery / ``TeamsSummaryWriter`` imports would pollute process +# ``os.environ`` from a cwd-discovered ``.env`` (#62935). Detect presence via find_spec only; bind symbols +# in ``check_teams_requirements()`` behind a dotenv no-op. import importlib.util import json import logging @@ -247,10 +251,18 @@ _SDK_IMPORTS = { "microsoft_teams.cards": ("AdaptiveCard", "ExecuteAction", "TextBlock")} +# Keep the old name as an alias so existing test imports don't break. NOTE: ``check_requirements`` is the +# PASSIVE probe (registry ``check_fn``, status / unit tests) — it must never trigger a pip install. +# ``check_teams_requirements`` is the ACTIVE lazy-installer, registered as ``ensure_deps_fn``: the +# registry's ``create_adapter()`` runs it when the passive probe fails, right before the gateway connects +# Teams (#79812). ``connect()`` re-checks defensively. @contextmanager def _suppress_third_party_dotenv() -> Iterator[None]: """No-op ``dotenv.load_dotenv`` while importing the Teams SDK: ``microsoft_teams.apps.app`` loads a - cwd-discovered ``.env`` at import, mutating process-global ``os.environ``. Hermes owns dotenv loading.""" + cwd-discovered ``.env`` at import, mutating process-global ``os.environ``. Hermes owns dotenv loading. + + See #62935. + """ try: import dotenv as _dotenv except ImportError: @@ -358,6 +370,10 @@ class TeamsAdapter(BasePlatformAdapter): return False try: # aiohttp app first — the bridge adapter wires SDK routes into it. + # Set up aiohttp app first — the bridge adapter wires SDK routes into it. client_max_size: Bot + # Framework activities are JSON (caps out well under 1 MiB); an explicit cap keeps + # oversized/chunked bodies from being buffered unbounded on a 0.0.0.0 bind (same pattern as + # webhook.py / raft, #58536/#58902). aiohttp_app = web.Application(client_max_size=_MAX_BODY_BYTES) aiohttp_app.router.add_get("/health", lambda _: web.Response(text="ok")) self._app = App( diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 3a56697339..38a48cfa56 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -32,7 +32,13 @@ def _redact_telegram_error_text(error: object) -> str: def _scoped_gate_env(name: str, default: str = "") -> str: - """Per-profile TELEGRAM_*/GATEWAY_* gate env read (multiplex env is first-writer-wins).""" + """Per-profile TELEGRAM_*/GATEWAY_* gate env read (multiplex env is first-writer-wins). + + Under gateway.multiplex_profiles the process env is first-writer-wins (the YAML→env bridge in + ``_apply_yaml_config``), so a raw ``os.getenv`` can return ANOTHER profile's allowlist (issue #72348, + Telegram mirror). Reads the active profile's secret scope when installed; falls back to ``os.getenv`` + outside multiplex — identical single-profile behavior. + """ try: from gateway.authz_mixin import _platform_gate_env return _platform_gate_env(name, default) @@ -54,7 +60,15 @@ async def _await_with_thread_deadline(awaitable, timeout: float, *, on_abandon=N """Wall-clock deadline that survives a blocked loop / cancellation-shielded PTB+httpcore init. ``on_abandon`` runs detached so an abandoned initialize() can't leak an httpx pool. Raises - ``asyncio.TimeoutError`` on expiry (feeds the PTB retry ladder).""" + ``asyncio.TimeoutError`` on expiry (feeds the PTB retry ladder). + + Thin wrapper over :func:`agent.deadline.run_bounded_async` (#85125 Phase 2f) — this adapter's private + implementation was the ancestor of that primitive and is now consolidated onto it. The unified layer + keeps every property the 9 call sites here rely on: thread-timer deadline that survives a blocked event + loop (#63309), abandonment of cancellation-shielded tasks (PTB/httpcore init inside anyio scopes), + detached best-effort ``on_abandon`` cleanup so an abandoned initialize() can't leak an httpx pool per + retry attempt, and off-loop stack-dump diagnostics when the loop never processes the expiry. + """ result = await run_bounded_async(awaitable, timeout, label="telegram-init", on_abandon=on_abandon) if result.timed_out: raise asyncio.TimeoutError() @@ -138,6 +152,9 @@ from utils import env_float, env_int _TELEGRAM_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".webp", ".gif"} # Max seconds a send/edit may sleep inline on a flood-control RetryAfter; longer penalties fail # closed with ``flood_control:{wait}`` so the caller's retry machinery owns the wait. +# Longer server penalties fail closed with a ``flood_control:{wait}`` SendResult so the caller's retry +# machinery (delivery ledger, streaming fallback) owns the wait instead of the coroutine pinning its worker +# — a 97-minute penalty on the boot path froze inbound on every platform (#91969). _FLOOD_INLINE_WAIT_CAP_SECS = 5.0 @@ -196,7 +213,12 @@ def _probe_voice_duration_seconds(path: str) -> Optional[int]: def telegram_deps_present() -> bool: """PASSIVE registry ``check_fn``: is python-telegram-bot importable? Never installs - (``check_telegram_requirements`` is the active ``ensure_deps_fn``).""" + (``check_telegram_requirements`` is the active ``ensure_deps_fn``). + + Registry ``check_fn`` — called from status displays and config loading, so it must never install + anything. The ACTIVE lazy-installer (``check_telegram_requirements``) is registered as + ``ensure_deps_fn`` and runs from ``create_adapter()`` when this returns False (#79812). + """ return TELEGRAM_AVAILABLE @@ -296,16 +318,41 @@ _DISCONNECT_STEP_TIMEOUT = 2.0 # other disconnect() steps: short, so a swallowe _UPDATER_START_TIMEOUT = 30.0 # start_polling() can hang on a degraded pool after a drain # Initial connect is unhealthy until getUpdates completes one round trip; bootstrap fails closed so # GatewayRunner disposes the adapter and retries fresh. +# Per-step bound for disconnect() awaits that are not updater.stop() itself. Kept short so a +# cancellation-swallowing lifecycle/PTB close cannot burn the gateway's whole fatal-handler budget before +# the reconnect queue is useful (#80598). updater.stop() keeps the longer _UPDATER_STOP_TIMEOUT. +# start_polling() can also hang when the connection pool is in a degraded state after +# _drain_polling_connections(), particularly when both primary and fallback Telegram endpoints are +# unreachable. Bounding start_polling() prevents the reconnect ladder from stalling indefinitely and allows +# the heartbeat loop to trigger its own recovery path. Refs: NousResearch/hermes-agent#59614 _INITIAL_POLLING_PROGRESS_TIMEOUT = 60.0 # Bounded drain (shutdown()/initialize() of the getUpdates request) so a wedged socket can't freeze # _polling_error_task and gate every escalation path behind its in-flight guard. +# shutdown()/initialize() on the getUpdates httpx request close and rebuild the connection pool. When a +# connection is wedged on a stale CLOSE-WAIT socket that close can block forever, hanging +# _drain_polling_connections() and freezing the whole reconnect ladder (the tracked _polling_error_task +# never completes, so every escalation path stays gated behind its in-flight guard). Bound the drain so the +# ladder always advances toward the fatal-restart escalation. Matches _UPDATER_STOP_TIMEOUT. Refs: +# NousResearch/hermes-agent#66377 _DRAIN_TIMEOUT = 15.0 # Wedged-recovery watchdog: healthy worst case is stop + 2x drain + start + 60s backoff ≈ 135s, so # 300s in flight is unambiguously stuck and the heartbeat force-escalates. +# Every recovery path (the reconnect ladder's re-entry, the pending-update probe, PTB's error callback) +# gates new recovery on ``_polling_error_task.done()``; if that task ever wedges on a hung await that no +# local bound covers, the whole gateway goes silently deaf with nothing retrying. The heartbeat loop +# force-escalates a recovery task that stays in-flight far longer than any healthy ladder attempt could take +# — stop (_UPDATER_STOP_TIMEOUT) + drain (2x_DRAIN_TIMEOUT) + start (_UPDATER_START_TIMEOUT) + max backoff +# (60s) is ~135s, so 300s is unambiguously stuck. See #66377. _POLLING_ERROR_TASK_STUCK_TIMEOUT = 300.0 _POLLING_PROGRESS_TIMEOUT = 60.0 # generation unhealthy until getUpdates returns; exceeds one idle long-poll # Telegram answers a long-poll within ~50s; no round-trip for ~3x that while get_me() is healthy and # nothing is queued means a consumer wedged on a socket that never raises (CLOSE-WAIT behind a route flip). +# Telegram holds a long-poll open for at most ~50s before answering (empty or not), so a healthy idle poller +# completes a getUpdates round-trip well inside this window. If no round-trip has completed for longer than +# this — while get_me() on the general request path stays healthy and no updates are queued server-side — +# the long-poll consumer is wedged on a socket that never raises (CLOSE-WAIT behind a TUN/proxy route flip, +# #92991) and no other probe can see it. ~3x the worst-case poll window leaves ample margin against false +# positives while still recovering within a few heartbeat intervals. _POLLING_STALL_TIMEOUT = 150.0 # sendVideo transcodes before answering, outlasting the 20s read timeout; also how long a user waits # to hear the attachment failed, so kept modest. @@ -335,6 +382,7 @@ class TelegramAdapter(BasePlatformAdapter): # edit_message applies MarkdownV2 only on finalize=True; without this flag stream_consumer skips # the final edit when raw text is unchanged. + # Fixes #25710. REQUIRES_EDIT_FINALIZE: bool = True FALLBACK_ON_FINAL_EDIT_FLOOD: bool = True # retrying a final edit burns the same flood budget RESEND_FINAL_ON_EMPTY_STREAM_FALLBACK: bool = True # a failed final edit may leave a partial preview @@ -423,6 +471,8 @@ class TelegramAdapter(BasePlatformAdapter): self._polling_progress_accepting = self._polling_teardown_started = False self._polling_error_callback_ref = None # Stall watchdog: generation start and last successful getUpdates (None = unknown). + # Monotonic timestamps for the polling stall watchdog (#92991): when the current polling generation + # began, and when the last successful getUpdates round-trip completed. self._polling_generation_started_monotonic: Optional[float] = None self._polling_last_progress_monotonic: Optional[float] = None # Live @username: PTB caches getMe() at initialize() and only rewrites it inside get_me(), so a @@ -436,6 +486,14 @@ class TelegramAdapter(BasePlatformAdapter): self._polling_pending_stuck_count = self._polling_not_running_count = 0 # Degraded until getUpdates makes progress; while True, send() short-circuits to failure so callers # (cron live-adapter branch) fall through to standalone delivery. + # Consecutive heartbeat probes that saw queued updates the running poller is not consuming. get_me() + # can't see this — the send path is healthy while the getUpdates consumer is wedged — so the + # heartbeat also probes get_webhook_info().pending_update_count and escalates to recovery after two + # consecutive stuck probes (#42909). + # Consecutive heartbeat probes that found the updater stopped entirely (running=False) while we are + # in polling mode with no reconnect in flight. Distinct from the wedged-but-running case above: the + # long-poll task is simply gone, so neither the connectivity probe nor PTB's error_callback ever + # fires and the gateway silently stops receiving messages with the process still alive (#55769). self._send_path_degraded: bool = False self._general_request_drain_lock = asyncio.Lock() self._dm_topics: Dict[str, int] = {} # topic_name -> message_thread_id @@ -460,6 +518,9 @@ class TelegramAdapter(BasePlatformAdapter): # "all": every message notifies (display.platforms.telegram.notifications). self._notifications_mode: str = "important" # send_or_update_status(): {(chat_id, status_key) -> message_id} so repeat calls edit in place. + # send_or_update_status() bookkeeping: {(chat_id, status_key) -> bot message_id} Tracks status + # bubbles owned by this adapter so subsequent calls with the same key edit the same message instead + # of appending new ones (#30045). self._status_message_ids: Dict[tuple, str] = {} # Last truncated mid-stream preview per (chat_id, message_id): past the 4096 cap every edit # truncates to the SAME text, and resending burns flood budget. Dropped on finalize. @@ -485,6 +546,8 @@ class TelegramAdapter(BasePlatformAdapter): self._drop_delayed_deliveries = True super()._set_fatal_error(code, message, retryable=retryable) # Permanent fatal: no reconnect will drain, so discard the hold queue (later holds are refused). + # Discard the hold queue now and refuse further holds (teardown salvage / late enqueue must not + # re-populate a queue that can never drain — review #83878). if not retryable: held = getattr(self, "_held_inbound_events", None) n = len(held) if held else 0 @@ -557,7 +620,12 @@ class TelegramAdapter(BasePlatformAdapter): def _hold_inbound_event(self, event: "MessageEvent", *, where: str, schedule: bool = True) -> None: """Preserve an inbound event that cannot be dispatched now (PTB already acked the update, so dropping is silent loss). - Capped, identity-deduped; permanent fatal discards. ``schedule=False`` inside a drain avoids poison-event loops.""" + Capped, identity-deduped; permanent fatal discards. ``schedule=False`` inside a drain avoids poison-event loops. + + The disconnect drop-guard (#55971) correctly prevents dispatch into a torn-down session. Destroying + the event is wrong: by the time we reach enqueue/flush, python-telegram-bot has already acked the + update and advanced the offset — silent permanent loss, no log, no error. + """ if self._is_permanent_fatal(): logger.warning( "[Telegram] Discarding inbound under non-retryable fatal (%s, %d chars)", where, len(getattr(event, "text", None) or "")) @@ -652,6 +720,13 @@ class TelegramAdapter(BasePlatformAdapter): def _legacy_runner_auth_fn(self): """``runner._is_user_authorized`` resolved off the bound handler (bare-adapter tests, direct embedding); None under multiplex where the handler is a profile closure.""" + # Resolve through the runner's full auth chain (platform + group allowlists, pairing store, + # allow-all flags). Prefer the platform-bound callback registered via set_authorization_check: it + # routes to GatewayRunner._is_user_authorized AND survives multiplex handler wrapping, whereas the + # bound-handler __self__ lookup is None when the primary handler is a profile closure — which + # silently dropped the chat allowlist and default-denied allowlisted group members under + # multiplex_profiles (#87132). Fall back to the bound handler for setups without a registered + # callback. runner = getattr(getattr(self, "_message_handler", None), "__self__", None) auth_fn = getattr(runner, "_is_user_authorized", None) return auth_fn if callable(auth_fn) else None @@ -697,6 +772,8 @@ class TelegramAdapter(BasePlatformAdapter): decision = self._env_allowlist_decision(normalized_user_id) if decision is None: # Fail-closed: no allowlist means deny unless GATEWAY_ALLOW_ALL_USERS is set. + # The runner auth path in _is_user_authorized() handles GATEWAY_ALLOW_ALL_USERS; this fallback + # must not silently allow everyone (fixes #24457). return _scoped_gate_env("GATEWAY_ALLOW_ALL_USERS").lower() in {"true", "1", "yes"} return decision @@ -878,13 +955,25 @@ class TelegramAdapter(BasePlatformAdapter): Forum topics use ``message_thread_id``; native Bot API DM topics opt in via explicit ``direct_messages_topic_id`` metadata; Hermes private-chat topic lanes are marked ``telegram_dm_topic_reply_fallback``. Anchor-less synthetic sends prefer the Hermes topic's ``message_thread_id`` (the native DM-topic id renders in a different chat lane). - ``reply_to_mode="off"`` suppresses the anchor but keeps ``message_thread_id``.""" + ``reply_to_mode="off"`` suppresses the anchor but keeps ``message_thread_id``. + + Live replies send the private topic thread id together with a reply anchor. Synthetic/resumed sends + without an anchor (loop wakeups, background-process notifications, queued follow-ups after a gateway + restart) prefer the Hermes topic's ``message_thread_id`` so they stay in the active topic lane + (#87051); ``direct_messages_topic_id`` is only used when no topic thread resolves, since the native + DM-topic id does not match the Hermes topic lane and can render the message in a different chat + lane. + """ fallback = cls._dm_topic_fallback(metadata) if fallback and reply_to_mode != "off": if reply_to_message_id is None: reply_to_message_id = cls._metadata_reply_to_message_id(metadata) if reply_to_message_id is None: # Anchor-less synthetic send: prefer the Hermes topic thread id (see docstring). + # Anchor-less synthetic sends (loop wakeups, watch notifications, restart-resumed + # follow-ups) must stay in the active topic lane: prefer the Hermes topic thread id when it + # resolves (#87051). Routing via direct_messages_topic_id here sent these to a different + # lane than the topic the session runs in. thread_message_id = cls._message_thread_id_for_send(thread_id) if thread_message_id is not None: return {"message_thread_id": thread_message_id} @@ -930,7 +1019,17 @@ class TelegramAdapter(BasePlatformAdapter): def _prune_stale_dm_topic_binding(self, chat_id: Any, thread_id: Any, *, metadata: Optional[Dict[str, Any]] = None) -> None: """Drop the stale ``telegram_dm_topic_bindings`` row for a topic Telegram confirmed deleted, else ``_recover_telegram_topic_thread_id`` keeps steering inbound to the dead thread. Best-effort. - Rows are namespaced by profile: the send's ``hermes_profile`` wins over the adapter's stamp.""" + Rows are namespaced by profile: the send's ``hermes_profile`` wins over the adapter's stamp. + + Without this prune the recovery logic in ``gateway.run._recover_telegram_topic_thread_id`` keeps + steering future inbound messages to the dead thread (the bug behind #31501 — tool progress, + approvals, replies all end up in the wrong place even though the user has moved on to a fresh + topic). Best-effort: we never raise from a send-fallback path — a failed cleanup must not turn into + a failed user-facing send. + Under ``gateway.profile_routes`` the transport adapter may not be the profile that wrote the + binding, so the send's ``hermes_profile`` metadata wins over the adapter's own profile stamp; + single-profile bots fall back to ``"default"``. See #76423. + """ if chat_id is None or thread_id is None: return db = getattr(getattr(self, "_session_store", None), "_db", None) @@ -963,7 +1062,12 @@ class TelegramAdapter(BasePlatformAdapter): def _should_retry_without_dm_topic_reply_anchor( cls, error: Exception, metadata: Optional[Dict[str, Any]], reply_to_message_id: Optional[int]) -> bool: """True when a DM-topic send should be retried with routing stripped: (1) stale anchor — reply - target deleted; (2) anchor-less synthetic send whose ``direct_messages_topic_id`` Bot API rejects.""" + target deleted; (2) anchor-less synthetic send whose ``direct_messages_topic_id`` Bot API rejects. + + 2. The synthetic-event case (added when #27937 introduced ``direct_messages_topic_id`` fallback for + sends without an anchor): if Bot API rejects the topic id itself with any BadRequest that mentions + topic/thread routing, we retry without routing rather than dropping the message. + """ if not cls._dm_topic_fallback(metadata) or not cls._is_bad_request_error(error): return False err_lower = str(error).lower() @@ -1130,12 +1234,22 @@ class TelegramAdapter(BasePlatformAdapter): return any(self._RICH_MATH_IN_DETAILS_RE.search(block) for block in self._RICH_DETAILS_RE.findall(content)) def _has_telegram_desktop_cjk_rich_garble_shape(self, content: str) -> bool: - """True for CJK content: Telegram Mac/Desktop rich rendering leaves overlapping glyphs.""" + """True for CJK content: Telegram Mac/Desktop rich rendering leaves overlapping glyphs. + + Telegram Mac/Desktop Bot API 10.1 rich-message rendering currently leaves overlapping draft/overlay + glyph artifacts for CJK text (#47653). The legacy MarkdownV2 path renders the same text cleanly, so + skip rich delivery up front until affected clients age out. + """ return bool(content and self._RICH_CJK_RE.search(content)) def _needs_rich_rendering(self, content: str) -> bool: """True for constructs MarkdownV2 degrades: pipe tables, task lists, <details>, block math. - Ordinary replies stay on MarkdownV2 so clients render consistent font weight/spacing.""" + Ordinary replies stay on MarkdownV2 so clients render consistent font weight/spacing. + + The rich endpoint is reserved for constructs where raw markdown materially improves output: pipe + tables (MarkdownV2 has no table syntax and rewrites them into bullet lists), GFM task lists, + collapsible ``<details>`` blocks, and block math. Adapted from #45995 (@YonganZhang). + """ if not content: return False if any(_TABLE_SEPARATOR_RE.match(line) for line in content.splitlines()): @@ -1175,7 +1289,14 @@ class TelegramAdapter(BasePlatformAdapter): def prefers_fresh_final_streaming(self, content: str, metadata: Optional[Dict[str, Any]] = None) -> bool: """Replace a streamed preview with a fresh rich final — DM topics only. Root DMs stay off (a live draft has no preview id); DM *topics* degrade to edit-in-place whose MarkdownV2 preview Telegram - refuses to rich-edit, so a fresh sendRichMessage + delete is the only way to keep native tables.""" + refuses to rich-edit, so a fresh sendRichMessage + delete is the only way to keep native tables. + + Root DMs keep this off (#46206 / #47048): successful draft streaming has no preview ``message_id``, + so the hook is not consulted, and in-place ``editMessageText.rich_message`` would duplicate a live + draft turn. Private DM *topics* often reject ``sendMessageDraft``; the consumer then degrades to + edit-in-place. Telegram rejects a rich edit of that plain MarkdownV2 preview, and the fallback + formatter permanently turns pipe tables into bullet lists. + """ metadata = metadata or {} if not (metadata.get("telegram_dm_topic_reply_fallback") or self._metadata_direct_messages_topic_id(metadata)): return False @@ -1396,6 +1517,7 @@ class TelegramAdapter(BasePlatformAdapter): if not await self._bounded_request_step(polling_req.shutdown(), "Polling request shutdown failed/timed out (non-fatal)"): # initialize() only rebuilds the client when ``client.is_closed``; an abandoned aclose() # leaves it false, so start_polling would reuse the CLOSE-WAIT socket (alive but deaf). + # Swap in a fresh client before initialize(). See #87057. self._orphan_and_rebuild_polling_client(polling_req) if await self._bounded_request_step(polling_req.initialize(), "Polling request re-initialize failed/timed out (non-fatal)"): logger.debug("[%s] Polling request pool drained before reconnect", self.name) @@ -1413,7 +1535,12 @@ class TelegramAdapter(BasePlatformAdapter): def _orphan_and_rebuild_polling_client(self, polling_req) -> None: """Replace a wedged HTTPXRequest client after a hung aclose(): swap in a fresh client and close - the old one in a detached, bounded task so it can't block the reconnect ladder.""" + the old one in a detached, bounded task so it can't block the reconnect ladder. + + PTB's ``HTTPXRequest.initialize()`` only calls ``_build_client()`` when the current client reports + ``is_closed``. If ``shutdown()`` was abandoned on a CLOSE-WAIT socket, that flag stays false and the + next ``start_polling()`` reuses the dead getUpdates connection (#87057). + """ old = getattr(polling_req, "_client", None) build = getattr(polling_req, "_build_client", None) if old is None or not callable(build) or getattr(old, "is_closed", True): @@ -1465,6 +1592,7 @@ class TelegramAdapter(BasePlatformAdapter): self._polling_progress_accepting = True self._send_path_degraded = True # Reset stall-watchdog timestamps: no proven progress yet, age measured from here. + # See #92991. self._polling_generation_started_monotonic = time.monotonic() self._polling_last_progress_monotonic = None return self._polling_generation, self._polling_progress_event @@ -1510,7 +1638,13 @@ class TelegramAdapter(BasePlatformAdapter): """Instrument one dedicated PTB getUpdates request with progress tracking. PTB request classes use ``__slots__`` (no ``__dict__`` on 3.13), so re-tag the instance to a thin ``__slots__ = ()`` - subclass overriding ``do_request`` — identical layout makes the ``__class__`` swap legal; works for test doubles too.""" + subclass overriding ``do_request`` — identical layout makes the ``__class__`` swap legal; works for test doubles too. + + On Python 3.13 their instances no longer carry a ``__dict__`` (the ``AbstractAsyncContextManager`` + MRO stopped yielding one), so ``request.do_request = wrapper`` raises ``AttributeError: + 'HTTPXRequest' object attribute 'do_request' is read-only`` and the whole Telegram connect fails + (#64482). It only appeared to work on Python 3.12, where those instances still had a ``__dict__``. + """ adapter = self class _InstrumentedPollingRequest(type(request)): @@ -1718,6 +1852,8 @@ class TelegramAdapter(BasePlatformAdapter): try: # Same watchdog bound as the reconnect ladders; the TimeoutError is an OSError subclass, so # the except below classifies it as a network error → background recovery. + # Same watchdog bound as the reconnect ladders: a wedged httpx connection pool can hang + # start_polling() forever at bootstrap too (#59614). generation, progress = await self._start_polling_once( self._app, drop_pending_updates=drop_pending_updates, error_callback=effective_callback, abandon_app_on_timeout=require_progress, @@ -1848,6 +1984,11 @@ class TelegramAdapter(BasePlatformAdapter): PROBE_TIMEOUT = 15 # seconds before declaring the path dead # Wedged-recovery watchdog: note when a recovery task is first seen in-flight and force-escalate # if the *same* task object still runs past the stuck timeout. + # Tracked locally so no _polling_error_task assignment site needs to stamp a timestamp: the + # heartbeat notes when it first observes a given recovery task still in-flight, and force-escalates + # if the *same* task object is still running after _POLLING_ERROR_TASK_STUCK_TIMEOUT. A healthy + # ladder attempt completes (task done) or chains to a new task well before then, so a single + # long-lived task is unambiguously wedged. See #66377. stuck_task_ref: Optional[asyncio.Task] = None stuck_task_since = 0.0 while True: @@ -1857,6 +1998,9 @@ class TelegramAdapter(BasePlatformAdapter): return # A recovery task hung on an unbounded await gates every other recovery path forever # (alive but deaf): force retryable-fatal so the reconnector rebuilds the adapter. + # Independent wedged-recovery watchdog (#66377): if the tracked recovery task has hung (any + # await no local bound covers), every other recovery path is gated behind it and returns + # early forever — the gateway stays alive but deaf. recovery_task = self._polling_error_task if recovery_task is not None and not recovery_task.done(): now = time.monotonic() @@ -1890,8 +2034,17 @@ class TelegramAdapter(BasePlatformAdapter): self._bot_identity_checked_at = time.monotonic() self._note_bot_username(getattr(bot, "username", None)) # get_me() OK proves only the send path; a wedged long-poll shows as server-side queue. + # get_me() succeeded — the general/send request path is healthy. That does NOT prove the + # getUpdates consumer is alive: PTB can report updater.running=True while the long-poll task + # is wedged, so DMs queue in the Bot API and never reach handlers (#42909). get_me() is + # blind to this; get_webhook_info() exposes it via pending_update_count. Escalate only after + # two consecutive probes see a non-zero queue while we believe we're polling, so a single + # in-flight update (consumed before the next probe) never trips recovery. await self._probe_pending_updates(bot, PROBE_TIMEOUT) # An empty queue can't hide a wedge forever: no round-trip past the stall threshold ⇒ dead. + # Even an empty queue cannot hide a wedged long-poll forever: Telegram answers within ~50s, + # so a consumer with no successful round-trip past the stall threshold is dead (#92991). + # Pure local-state check — no Bot API call needed. await self._check_polling_stall() except asyncio.CancelledError: return @@ -1907,7 +2060,14 @@ class TelegramAdapter(BasePlatformAdapter): PTB can report ``updater.running`` while the long-poll is stuck; get_me() stays healthy yet DMs queue in the Bot API. A stuck queue over two consecutive probes ⇒ dead consumer. Also covers the updater having stopped - entirely (``running=False``, no reconnect in flight).""" + entirely (``running=False``, no reconnect in flight). + + PTB can report ``updater.running == True`` while its long-poll task is silently stuck (e.g. a socket + that epoll keeps reporting readable on WSL2). ``get_me()`` stays healthy because it uses the general + request path, so the CLOSE-WAIT heartbeat never fires — yet DMs queue in the Bot API and never reach + handlers (#42909). + We detect the stopped updater directly and feed the same ladder (#55769). + """ # Polling mode only: in webhook mode Telegram pushes and holds no server-side queue. if self._teardown_started or self._webhook_mode: return @@ -1924,6 +2084,11 @@ class TelegramAdapter(BasePlatformAdapter): # Long-poll task gone, general-path calls still succeed, so no error_callback/probe ever # fires. Debounced over two probes so a just-starting updater never trips it. self._polling_pending_stuck_count = 0 + # We are in polling mode with no reconnect in flight, yet PTB's updater has stopped entirely. + # This is distinct from the wedged-but-running consumer handled below: the long-poll task is + # gone, get_me()/get_webhook_info() on the general request path still succeed, so no + # error_callback or connectivity probe ever fires and the gateway silently stops receiving + # messages while the process stays alive (#55769). self._polling_not_running_count += 1 logger.warning( "[%s] Telegram polling heartbeat: updater stopped while in polling mode (stuck probe %d/2)", self.name, @@ -1966,7 +2131,10 @@ class TelegramAdapter(BasePlatformAdapter): async def _check_polling_stall(self) -> None: """Watchdog the last successful getUpdates round-trip: a long-poll can wedge without raising (CLOSE-WAIT after a route flip) while every other probe stays blind; no round-trip for - ``_POLLING_STALL_TIMEOUT`` ⇒ escalate through the bounded reconnect ladder.""" + ``_POLLING_STALL_TIMEOUT`` ⇒ escalate through the bounded reconnect ladder. + + See #92991. + """ if self._webhook_mode or self._teardown_started or self.has_fatal_error or self._recovery_in_flight(): return now = time.monotonic() @@ -2089,11 +2257,20 @@ class TelegramAdapter(BasePlatformAdapter): return # Stable local ref: a concurrent disconnect() may null self._app across the awaits above. app = self._app + # Capture a stable local reference: self._app can be reassigned to None by a concurrent + # disconnect() while we're suspended across the awaits above (same race #55992 fixed on the + # network path). Re-reading self._app after that point would raise AttributeError deep inside + # start_polling instead of failing fast here, where the except below reschedules or escalates to + # fatal. expected_generation = self._polling_generation + 1 if not app: raise RuntimeError("Telegram application was torn down during conflict reconnect") # drop_pending_updates=True makes Telegram terminate any other getUpdates session for this # token (zombie or our own prior retry); without it each retry is immediately 409'd. + # The competing session is either a zombie from the previous gateway process (whose long-poll + # hasn't expired server-side yet) or our own previous retry's still-expiring session. Without + # this, each retry starts a new getUpdates session that immediately gets 409'd by the previous + # one, creating the very conflict we are trying to recover from (#75017). self._polling_conflict_recovery_generation = expected_generation try: await self._start_polling_once(app, drop_pending_updates=True, error_callback=self._polling_error_callback_ref) @@ -2364,7 +2541,11 @@ class TelegramAdapter(BasePlatformAdapter): self.name, len(menu_commands), hidden_count, max_commands) async def _run_post_connect_housekeeping(self) -> None: - """Command menu, status indicator and DM topics off the connect path; every step is non-fatal.""" + """Command menu, status indicator and DM topics off the connect path; every step is non-fatal. + + DM topics — all off the connect path so a slow Bot API call cannot blow the gateway connect timeout + (#46298). + """ try: try: await self._register_command_menu() @@ -2528,6 +2709,11 @@ class TelegramAdapter(BasePlatformAdapter): } # CLOSE_WAIT fd leak: PTB's httpx.AsyncClient has no keepalive tuning; inject platform_httpx_limits() # while preserving PTB's max_connections (httpx_kwargs is spread last, so `limits` here wins). + # CLOSE_WAIT fd leak (#31599, same class as #18451): PTB's HTTPXRequest builds the underlying + # httpx.AsyncClient with `limits = httpx.Limits(max_connections=connection_pool_size)` and *no* + # keepalive tuning, so httpx's default keepalive_expiry=5.0 applies. Behind an HTTP proxy + # (Cloudflare Warp etc.) a peer-initiated FIN can sit in CLOSE_WAIT longer than that, leaking fds in + # the general request pool (_request[1]) which _drain_polling_connections never resets. from gateway.platforms._http_client_limits import platform_httpx_limits _base_limits = platform_httpx_limits() if _base_limits is not None: @@ -2576,6 +2762,12 @@ class TelegramAdapter(BasePlatformAdapter): logger.info("[%s] Telegram fallback IPs active: %s", self.name, ", ".join(fallback_ips)) # Separate request/update pools reduce contention during polling reconnect + bootstrap calls. _transport_kwargs: dict = {"socket_options": tcp_keepalive_socket_options()} + # Keep request/update pools separate to reduce contention during polling reconnect + bot API + # bootstrap/delete_webhook calls. httpx ignores the client-level `limits` kwarg when a custom + # `transport` is supplied (#58790). Unlike the proxy/direct branches (which inject limits at the + # client level via `_with_limits`), this branch MUST pass the tuned limits directly into + # TelegramFallbackTransport so its inner AsyncHTTPTransport instances honour keepalive_expiry — + # do not route this through `_with_limits`, httpx would discard it. if _pool_limits is not None: _transport_kwargs["limits"] = _pool_limits _updates_transport_kwargs = dict(_transport_kwargs) @@ -2764,6 +2956,12 @@ class TelegramAdapter(BasePlatformAdapter): # line made healthy startups look stalled at "attempt 1/8". logger.warning("[%s] Connected to Telegram (%s mode)", self.name, "webhook" if self._webhook_mode else "polling") # Heartbeat only in polling mode: webhook mode has no long-poll socket to wedge in CLOSE-WAIT. + # WARNING, not INFO: the "Connecting to Telegram (attempt N/8)…" line above is emitted at + # WARNING and reaches the terminal (the gateway's default stderr handler is WARNING-only), but + # this success line was INFO and went to the log file only. A healthy startup therefore looked + # permanently stalled at "attempt 1/8" on the console — the logging illusion in #90835. Both + # sides of the connect transition must share a terminal-visible level so a real hang is the + # *absence* of this line, not ambiguity. if not self._webhook_mode: self._restart_task_attr("_polling_heartbeat_task", self._polling_heartbeat_loop()) # Seed the live identity from PTB's initialize() cache; polling rides the heartbeat's get_me(), @@ -2774,6 +2972,10 @@ class TelegramAdapter(BasePlatformAdapter): self._restart_task_attr("_bot_identity_refresh_task", self._bot_identity_refresh_loop()) # Command menu / DM topics / status indicator can stall for some tokens: defer to a cancellable # task so one slow call can't sink the (gateway-timed) connect while transport is live. + # Command-menu registration, DM-topic setup, and the status indicator each make Bot API calls + # that can stall for certain tokens. Running them here — inside the connect() coroutine that the + # gateway wraps in a connect timeout — means one slow call blows the whole connect and the + # adapter never comes up, even though polling/webhook is already live (#46298). self._start_post_connect_housekeeping() return True except Exception as e: @@ -2837,6 +3039,9 @@ class TelegramAdapter(BasePlatformAdapter): ], current_task) awaitable_tasks = [t for t in pending_tasks if asyncio.isfuture(t) or asyncio.iscoroutine(t)] + # Hold-queue redispatch must be cancellable+awaitable on teardown so it cannot dispatch + # handle_message into a torn-down session (same lifecycle rule teknium called out on #72037 for + # shielded flush dispatch). for task in pending_tasks: task.cancel() if awaitable_tasks: @@ -2862,13 +3067,18 @@ class TelegramAdapter(BasePlatformAdapter): async def _await_disconnect_step(self, awaitable, timeout: float, step: str) -> bool: """Await one disconnect step; detach on timeout so teardown advances (``wait_for`` would wait for a - PTB close that swallows ``CancelledError`` on a half-dead socket). Abandoned tasks are observed.""" + PTB close that swallows ``CancelledError`` on a half-dead socket). Abandoned tasks are observed. + + ``asyncio.wait_for`` cancels an overdue child but then waits for it to exit. Detach at the deadline + and continue — the abandoned task is observed via ``_consume_abandoned_task``. See #80598. + """ task = asyncio.ensure_future(awaitable) try: done, _pending = await asyncio.wait({task}, timeout=timeout if timeout > 0 else None) except asyncio.CancelledError: # asyncio.wait does NOT cancel its futures when itself cancelled; don't orphan the inner task. task.cancel() + # Mirror the pattern used by GatewayRunner._await_adapter_cleanup_with_timeout. See #80598. task.add_done_callback(_consume_abandoned_task) raise if task in done: @@ -2905,6 +3115,7 @@ class TelegramAdapter(BasePlatformAdapter): self._polling_progress_event = asyncio.Event() self._send_path_degraded = True # Release the bot-token lock immediately so a wedged close cannot block the reconnect watcher. + # The rest of teardown is best-effort against a half-dead transport. See #80598. self._release_platform_lock() # Cancel and await both polling lifecycle owners right after the fence, before any other teardown # await lets them start a new generation. @@ -2922,6 +3133,9 @@ class TelegramAdapter(BasePlatformAdapter): self._polling_progress_accepting = False self._send_path_degraded = True # Cancel deferred post-connect housekeeping so it cannot fire into a half-torn-down bot client. + # Cancel deferred post-connect housekeeping (command-menu / DM-topic / status-indicator Bot API + # calls) so it cannot fire into a half-torn-down bot client (#46298). getattr guards the + # object.__new__ test pattern where __init__ (which sets this attr) is never called. post_connect_task = getattr(self, "_post_connect_task", None) if post_connect_task and not post_connect_task.done(): post_connect_task.cancel() @@ -2945,6 +3159,7 @@ class TelegramAdapter(BasePlatformAdapter): logger.warning( "[%s] updater.stop() failed during disconnect: %s", self.name, _redact_telegram_error_text(stop_error)) # app.stop()/shutdown() can also block on a half-dead httpx pool. + # Detach-on-timeout so disconnect always returns (#80598). if self._app.running: await self._await_disconnect_step(self._app.stop(), _DISCONNECT_STEP_TIMEOUT, "app.stop()") await self._await_disconnect_step(self._app.shutdown(), _DISCONNECT_STEP_TIMEOUT, "app.shutdown()") @@ -3071,6 +3286,10 @@ class TelegramAdapter(BasePlatformAdapter): wait = float(retry_after) if retry_after is not None else 1.0 safe_send_error = _redact_telegram_error_text(send_err) # Never sleep a long server RetryAfter verbatim — it once pinned send() for 97 minutes. + # Mirror the edit path: a RetryAfter past a few seconds is not something to hold this + # coroutine open for. Sleeping the server value verbatim pinned send() for 97 minutes in + # production and froze inbound on every platform when it ran on the gateway boot path + # (#91969). if wait > _FLOOD_INLINE_WAIT_CAP_SECS: logger.warning( "[%s] Telegram flood control on send (retry_after=%.1fs > %.0fs); failing closed instead of sleeping: %s", @@ -3166,7 +3385,12 @@ class TelegramAdapter(BasePlatformAdapter): async def send_or_update_status( self, chat_id: str, status_key: str, content: str, *, metadata: Optional[Dict[str, Any]] = None) -> SendResult: """Send a status message, or edit the previous one with the same ``(chat_id, status_key)``; if the - edit fails (deleted, too old, …) the cached id is dropped and a fresh message is sent.""" + edit fails (deleted, too old, …) the cached id is dropped and a fresh message is sent. + + Issue #30045: progress/status callbacks (context-pressure, lifecycle, compression, etc.) used to + append a fresh bubble on every call. With this method, the first call sends and the message id is + remembered; subsequent calls with the same (chat_id, status_key) edit that same message in place. + """ key = (str(chat_id), str(status_key)) cached_id = self._status_message_ids.get(key) if cached_id is not None: @@ -3212,12 +3436,24 @@ class TelegramAdapter(BasePlatformAdapter): return SendResult(success=False, error="Not connected") # Rich finalize (Bot API 10.1): edit the preview IN PLACE via rich_message — no fresh send + delete. # Before the 4,096 pre-flight because the rich cap is 32,768; falls back to legacy on rejection. + # Rich finalize (Bot API 10.1): when the completed content has constructs the legacy MarkdownV2 edit + # degrades (tables → bullet lists, task lists, <details>, block math) and rich is available, edit + # the preview IN PLACE via editMessageText's rich_message param. No fresh send + delete → no + # duplicate preview (the problem #46206 reverted the fresh-final path for). Attempted before the + # 4,096 overflow pre-flight because the rich text cap is 32,768 — a rich table that exceeds the + # MarkdownV2 limit must not be split into legacy chunks. Falls back to the legacy edit path + # (overflow split included) on capability/permanent rejection. if finalize and self._rich_eligible(content): rich_result = await self._try_edit_rich(chat_id, message_id, content, metadata=metadata) if rich_result is not None: return rich_result # Pre-flight: over-limit content is split-and-delivered on finalize; mid-stream we truncate instead # (splitting moves the edit target to a continuation → infinite duplication loop). + # Pre-flight: if content already exceeds the limit, split-and-deliver without round-tripping a + # doomed edit. During streaming (finalize=False) we truncate instead of splitting — splitting + # creates continuation messages whose IDs become the new edit target, and on the next token chunk + # the full accumulated text is re-edited into the continuation, triggering another split → infinite + # duplication loop (#48648). _preview_key = (str(chat_id), str(message_id)) _saturated_preview = False if finalize: @@ -3255,6 +3491,7 @@ class TelegramAdapter(BasePlatformAdapter): if finalize: return await self._edit_overflow_split(chat_id, message_id, content, finalize=finalize, metadata=metadata) # Mid-stream: truncate and retry instead of splitting (saturated-preview dedup as above). + # See #48648. truncated = self._truncate_stream_overflow_preview(content) if self._last_overflow_preview.get(_preview_key) == truncated: return SendResult(success=True, message_id=message_id) @@ -3290,7 +3527,12 @@ class TelegramAdapter(BasePlatformAdapter): def _truncate_stream_overflow_preview(self, content: str) -> str: """One-message preview for oversized streaming edits (edits must keep targeting the original id; - final edits use ``_edit_overflow_split``).""" + final edits use ``_edit_overflow_split``). + + Splitting a mid-stream preview creates continuation messages and moves the active message id, so the + next accumulated-token edit repeats the overflow cycle (#48648). Final edits still use + ``_edit_overflow_split`` to deliver the complete response. + """ return self.truncate_message(content, self.MAX_MESSAGE_LENGTH, len_fn=utf16_len)[0] async def _send_overflow_continuation( @@ -3378,7 +3620,13 @@ class TelegramAdapter(BasePlatformAdapter): return SendResult(success=True, message_id=last_id, continuation_message_ids=tuple(continuation_ids)) async def delete_message(self, chat_id: str, message_id: str) -> bool: - """Delete a bot-posted message (Bot API allows it within 48h); failures are non-fatal.""" + """Delete a bot-posted message (Bot API allows it within 48h); failures are non-fatal. + + Used by the stream consumer's fresh-final cleanup path (ported from openclaw/openclaw#72038) to + remove long-lived preview messages after sending the completed reply as a fresh message. Telegram's + Bot API ``deleteMessage`` works for bot-posted messages in the last 48 hours. Failures are non-fatal + — the caller leaves the preview in place and logs at debug level. + """ if not self._bot: return False try: @@ -3439,7 +3687,12 @@ class TelegramAdapter(BasePlatformAdapter): async def _send_message_with_thread_fallback(self, **kwargs): """Send a control-style message (approval prompts, pickers), retrying once without - message_thread_id on 'Message thread not found' (stale thread_id); ``send`` has its own.""" + message_thread_id on 'Message thread not found' (stale thread_id); ``send`` has its own. + + Used for control-style sends (approval prompts, model picker, update prompts) that can carry a stale + thread_id from a DM reply chain. The streaming send loop has its own equivalent (PR #3390) at the + body of ``send``; this helper applies the same retry pattern to the non-streaming control paths. + """ if not self._bot: raise RuntimeError("Not connected") message_thread_id = kwargs.get("message_thread_id") @@ -4017,6 +4270,9 @@ class TelegramAdapter(BasePlatformAdapter): # Resolve FIRST (unblocks the agent thread), render after: a tap landing after the wait timed out # (count == 0) must NOT claim "Approved" — the command was already denied. try: + # Rendering happens after so the message reflects what actually occurred: a tap that lands after + # the approval wait timed out (count == 0) must NOT claim "Approved" — the command was already + # denied and will not run (#63501 regression follow-up: 60s waits made stale taps common). from tools.approval import resolve_gateway_approval count = resolve_gateway_approval(session_key, choice) logger.info( @@ -4288,6 +4544,10 @@ class TelegramAdapter(BasePlatformAdapter): async def _send_voice_bubble(self, audio_file, chat_id, reply_to, metadata, caption, duration_secs): """sendVoice with caption variants: MarkdownV2 when it fits 1024 chars, plain fallback when the Bot API rejects the entities; anything else is a real error.""" + # Render caption markdown (#32029): auto-TTS captions carry the agent's markdown reply, which showed + # literal *asterisks* and [links](...) without a parse_mode. Format to MarkdownV2 when it fits the + # 1024-char caption cap; fall back to the raw text (previous behaviour) when formatting would + # overflow or the Bot API rejects the entities. _caption_variants: List[tuple] = [] if caption: try: @@ -4952,6 +5212,12 @@ class TelegramAdapter(BasePlatformAdapter): @classmethod def _entity_span(cls, source_text: str, entity) -> Optional[str]: """The entity's text, or None when its offsets are unusable.""" + # Telegram's official group-disambiguation form for slash commands (``/cmd@botname``) is emitted as + # a single ``bot_command`` entity covering the whole span — there is no accompanying ``mention`` + # entity. Treat it as a direct address to this bot when the ``@botname`` suffix matches. This is the + # form Telegram's own command menu autocomplete produces in groups, so dropping it at the mention + # gate would break /new, /reset, /help, ... for every group that has ``require_mention`` enabled + # (#15415). offset = int(getattr(entity, "offset", -1)) length = int(getattr(entity, "length", 0)) if offset < 0 or length <= 0: @@ -5265,7 +5531,13 @@ class TelegramAdapter(BasePlatformAdapter): async def _surface_media_cache_failure( self, msg: Message, event: MessageEvent, kind: str, exc: Exception, display_name: Optional[str] = None) -> None: """Surface a failed media download to BOTH the user (reply asking to retry) and the agent (observed - note) — otherwise the turn dispatches silently with empty media_urls.""" + note) — otherwise the turn dispatches silently with empty media_urls. + + This (1) replies to the user in Telegram so they know to retry, and (2) appends an agent-visible + notice to event.text via the existing observed-note channel so the agent knows an attachment was + attempted and failed — never a silent empty turn. No new event fields (the structured-event refactor + is out of scope per #23045). + """ named = f" ({display_name})" if display_name else "" try: await msg.reply_text( @@ -5315,6 +5587,11 @@ class TelegramAdapter(BasePlatformAdapter): the ``guest_mode`` @mention bypass crosses it) and then any of free_response chat/topic, ``require_mention`` off, reply to the bot, @mention (incl. ``/cmd@botname``), or a wake-word match.""" # Learn the live handle BEFORE any mention gate routes on it, then drop our own echoed messages. + # Filter out the bot's own messages (returned by getUpdates in some environments like + # groups/supergroups where the bot can see its own messages). Without this, outbound messages are + # counted as incoming unread in the Hermes inbox (#52363). Otherwise a BotFather rename leaves the + # stale handle in place and the exclusive-mention gate reads a message addressed to us as one + # addressed to some other bot. self._observe_bot_identity_from_message(message) if self._is_own_message(message): return False @@ -5964,6 +6241,12 @@ class TelegramAdapter(BasePlatformAdapter): from gateway import rich_sent_store reply_to_text = rich_sent_store.lookup(str(message.chat.id), reply_to_id) except Exception: + # Extract reply context if this message is a reply. Prefer Telegram's native partial quote + # (message.quote, TextQuote) so a user replying to a single selected substring of a prior + # multi-section message doesn't get the whole replied-to message injected into the agent's + # context — which can cause the agent to act on unrelated actionable-looking text the user + # didn't quote (#22619). Fall back to the full replied-to message text / caption when no + # native quote is present. reply_to_text = None return reply_to_id, reply_to_text @@ -5975,6 +6258,12 @@ class TelegramAdapter(BasePlatformAdapter): telegram_chat_type = self._chat_type_str(chat) # str() so PTB enums and plain-string mocks both work chat_type = "group" if telegram_chat_type in {"group", "supergroup"} else ("channel" if telegram_chat_type == "channel" else "dm") # Shared normalizer so gating and session routing agree (reply-UI anchors dropped, General → "1"). + # Resolve routable thread id for DM topics and forum group topics via the shared normalizer, so + # gating and session routing agree on one value. Only real topic/forum messages keep a thread id; + # ordinary reply-UI anchors are dropped (they are not durable session threads and sends against them + # hit 'Message thread not found', #3206), while forum General-topic messages + # (message_thread_id=None) normalize to the General-topic id so replies route back to General + # (#22423). thread_id_str = self._effective_message_thread_id(message) chat_topic, topic_skill = self._resolve_topic_binding(message, chat_type, thread_id_str) has_full_name = hasattr(chat, "full_name") @@ -6051,6 +6340,14 @@ class TelegramAdapter(BasePlatformAdapter): # config, setup wizard, standalone sender). +# ────────────────────────────────────────────────────────────────────────── Plugin migration glue (#41112 / +# #3823) Added when the Telegram adapter (+ its telegram_network satellite) moved from gateway/platforms/ +# into this bundled plugin. Mirrors the Discord (#24356) / Slack migrations: a register(ctx) entry point +# plus hook implementations that replace the per-platform core touchpoints (the Platform.TELEGRAM branch in +# gateway/run.py, the telegram_cfg YAML→env/extra block in gateway/config.py, the _setup_telegram wizard + +# _PLATFORMS["telegram"] static dict in hermes_cli/{setup,gateway}.py, and the _send_telegram dispatch in +# tools/send_message_tool.py). Telegram uses the generic token connected check, so no is_connected override +# is needed. ────────────────────────────────────────────────────────────────────────── def _resolve_notifications_mode() -> str: """Notification mode (all/important) from env, else config.yaml display.platforms.telegram.notifications.""" mode = os.getenv("HERMES_TELEGRAM_NOTIFICATIONS", "") @@ -6112,12 +6409,17 @@ def interactive_setup() -> None: def _apply_yaml_config(yaml_cfg: dict, telegram_cfg: dict) -> dict | None: """Translate config.yaml telegram: keys into TELEGRAM_* env vars and PlatformConfig.extra. Env vars - take precedence over YAML. Returns extras to merge into PlatformConfig.extra, or None.""" + take precedence over YAML. Returns extras to merge into PlatformConfig.extra, or None. + + Implements the apply_yaml_config_fn contract (#24849). Mirrors the legacy telegram_cfg block from + gateway/config.py::load_gateway_config(). + """ import json as _json extras: dict = {} # Under multiplex a secondary profile's authorization gates must NOT hit the process-global env # (first-writer-wins would pin them for every profile); they flow via extra/secret scope. try: + # See #72348. from agent.secret_scope import current_secret_scope, is_multiplex_active _skip_env_bridge = bool(is_multiplex_active() and current_secret_scope() is not None) except Exception: diff --git a/plugins/platforms/telegram/telegram_network.py b/plugins/platforms/telegram/telegram_network.py index 91a104de6d..fd784d5a3b 100644 --- a/plugins/platforms/telegram/telegram_network.py +++ b/plugins/platforms/telegram/telegram_network.py @@ -17,6 +17,8 @@ _TELEGRAM_API_HOST = "api.telegram.org" # TCP keepalive so a half-open/CLOSE-WAIT long-poll errors out instead of blocking getUpdates forever # (Windows leaves SO_KEEPALIVE off). Idle/interval knobs are best-effort per Python/OS combo. +# Windows does not enable SO_KEEPALIVE on new sockets by default, so a dead api.telegram.org peer can hang +# forever (#87057). _TCP_KEEPALIVE_IDLE_S = 30 _TCP_KEEPALIVE_INTERVAL_S = 10 _TCP_KEEPALIVE_COUNT = 3 @@ -58,6 +60,7 @@ class TelegramFallbackTransport(httpx.AsyncBaseTransport): api.telegram.org (like ``curl --resolve``) so a blackholed IPv6 AAAA can't pin initialize().""" # Bound every pool: httpx's 100-connection default × (wedged endpoint + seed IPs) can outgrow the fd limit. + # See #63311. _POOL_LIMITS = httpx.Limits(max_connections=8, max_keepalive_connections=4) def __init__(self, fallback_ips: Iterable[str], **transport_kwargs): @@ -99,7 +102,11 @@ class TelegramFallbackTransport(httpx.AsyncBaseTransport): async def _reset_fallback(self, ip: str) -> None: """Discard a failed fallback pool: a peer-closed connect leaves a CLOSE_WAIT socket in it, and the - poisoned pool would leak one fd per retry.""" + poisoned pool would leak one fd per retry. + + Retaining the poisoned pool leaks one descriptor per retry until the process hits its file limit and + can no longer accept connections or resolve DNS (#63311). + """ async with self._fallback_lock: transport = self._fallbacks.pop(ip, None) if transport is None: @@ -229,11 +236,20 @@ async def _query_doh_provider(client: httpx.AsyncClient, provider: dict) -> list async def discover_fallback_ips() -> list[str]: """Resolve api.telegram.org via Google + Cloudflare DoH; unique A records, in order. IPs matching the system resolver are deliberately KEPT (often the most reliable path). Falls back to - ``SEED_FALLBACK_IPS`` only when DoH yields nothing usable.""" + ``SEED_FALLBACK_IPS`` only when DoH yields nothing usable. + + IPs that match the local system resolver are kept rather than excluded: in many networks the system-DNS + IP is the most reliable path to api.telegram.org and a transient primary-path failure should be retried + against the same address via the IP-rewrite path before the seed list is consulted (#14520). + """ async with httpx.AsyncClient(timeout=httpx.Timeout(_DOH_TIMEOUT)) as client: system_dns_task = asyncio.ensure_future(asyncio.to_thread(_resolve_system_dns)) results = await asyncio.gather(*[_query_doh_provider(client, p) for p in _DOH_PROVIDERS], return_exceptions=True) # The getaddrinfo leg has no timeout of its own and only feeds the log line below — bound it. + # The system-resolver leg runs socket.getaddrinfo in a worker thread with no timeout of its own — a + # wedged OS resolver (broken VPN/DNS) can sit for minutes. Its result only feeds the no-usable-answers + # log line below, so it must never gate discovery: bound it and move on (#63309). The DoH legs are + # already bounded by the client timeout above. system_ips: set[str] = set() try: system_result = await asyncio.wait_for(system_dns_task, timeout=_DOH_TIMEOUT) diff --git a/plugins/platforms/wecom/adapter.py b/plugins/platforms/wecom/adapter.py index 8e527a2fe3..89b591a16f 100644 --- a/plugins/platforms/wecom/adapter.py +++ b/plugins/platforms/wecom/adapter.py @@ -175,6 +175,7 @@ class WeComAdapter(WeComStreamMixin, WeComMediaMixin, ChatSendQueueMixin, BasePl return self._startup_failure("wecom_missing_credentials", "WeCom startup failed: WECOM_BOT_ID and WECOM_SECRET are required", "[%s] %s") try: # Tighter keepalive so idle CLOSE_WAIT drains promptly. + # See #18451. from gateway.platforms._http_client_limits import platform_httpx_limits from gateway.platforms.base import _ssrf_redirect_guard from tools.url_safety import create_ssrf_safe_async_client diff --git a/plugins/platforms/wecom/callback_adapter.py b/plugins/platforms/wecom/callback_adapter.py index e3dbecde9b..505b75a313 100644 --- a/plugins/platforms/wecom/callback_adapter.py +++ b/plugins/platforms/wecom/callback_adapter.py @@ -53,7 +53,14 @@ def check_wecom_callback_requirements() -> bool: def ensure_wecom_callback_requirements() -> bool: - """ACTIVE lazy-installer (``ensure_deps_fn``): installs ``defusedxml`` and rebinds globals.""" + """ACTIVE lazy-installer (``ensure_deps_fn``): installs ``defusedxml`` and rebinds globals. + + Registered as ``ensure_deps_fn``: the registry's ``create_adapter()`` runs it when the passive probe + fails, right before the gateway connects the platform (#79812). Installs ``defusedxml`` (the only + non-core dep; aiohttp/httpx ship with every messaging install) and rebinds the module globals. Before + this hook existed, the passive ``check_fn`` returned False forever on installs without the ``wecom`` + extra and the ``platform.wecom_callback`` LAZY_DEPS entry was never exercised. + """ if check_wecom_callback_requirements(): return True @@ -118,6 +125,7 @@ class WecomCallbackAdapter(BasePlatformAdapter): except (ConnectionRefusedError, OSError): pass try: + # Tighter keepalive so idle CLOSE_WAIT drains promptly (#18451). from gateway.platforms._http_client_limits import platform_httpx_limits self._http_client = httpx.AsyncClient(timeout=20.0, limits=platform_httpx_limits()) # client_max_size → 413 before our handler / any signature work runs. @@ -240,6 +248,7 @@ class WecomCallbackAdapter(BasePlatformAdapter): return tuple(request.query.get(k, "") for k in ("msg_signature", "timestamp", "nonce")) def _is_duplicate(self, message_id: str) -> bool: + # Deduplicate: WeCom retries callbacks on timeout, producing duplicate inbound messages (#10305). now = time.time() if now - self._seen_messages.get(message_id, float("-inf")) < MESSAGE_DEDUP_TTL_SECONDS: return True diff --git a/plugins/video_gen/fal/__init__.py b/plugins/video_gen/fal/__init__.py index 2bf7b48b2b..bc9e2a7f74 100644 --- a/plugins/video_gen/fal/__init__.py +++ b/plugins/video_gen/fal/__init__.py @@ -295,6 +295,7 @@ class FALVideoGenProvider(VideoGenProvider): def capabilities(self) -> Dict[str, Any]: # RESOLVED family's surface so the dynamic tool schema gates params on what the selected model honors; union fallback (never raises). try: + # Falls back to the cross-family union if resolution fails (never raises). See #97057. _family_id, family = _resolve_family(None) except Exception: # noqa: BLE001 family = None diff --git a/plugins/web/ddgs/provider.py b/plugins/web/ddgs/provider.py index 0a4053114f..48d7050854 100644 --- a/plugins/web/ddgs/provider.py +++ b/plugins/web/ddgs/provider.py @@ -23,6 +23,7 @@ logger = logging.getLogger(__name__) # Hard wall-clock cap per search: ``DDGS(timeout=…)`` only bounds individual HTTP requests; # ddgs's multi-engine retry loop has no overall cap, so a rate-limited response could # otherwise hang the shared agent loop indefinitely. +# Enforce a hard cap here by killing a disposable worker process (#68096). _SEARCH_TIMEOUT_SECS = 30 _POLL_INTERVAL_SECS = 0.1 # parent stdout / interrupt-flag poll cadence _TERMINATE_GRACE_SECS = 1.0 # wait after terminate() before escalating to kill() @@ -36,7 +37,11 @@ class _SearchInterrupted(Exception): def _run_ddgs_search(query: str, safe_limit: int) -> list[dict[str, Any]]: """Blocking ddgs query → normalized hits (module-level: the child worker imports it, - tests patch it for in-process runs).""" + tests patch it for in-process runs). + + ``DDGS(timeout=…)`` bounds each individual HTTP request; the overall wall-clock cap is enforced by the + parent via process timeout (#68096). + """ from ddgs import DDGS # type: ignore results: list[dict[str, Any]] = [] with DDGS(timeout=10) as client: @@ -182,7 +187,10 @@ class DDGSWebSearchProvider(BaseWebSearchProvider): def search(self, query: str, limit: int = 5) -> Dict[str, Any]: """Run the search in a disposable child with a hard wall-clock timeout so a - hung native ``primp`` call cannot freeze the Hermes process.""" + hung native ``primp`` call cannot freeze the Hermes process. + + See #36776, #68096. + """ if not self.is_available(): return search_fail("ddgs package is not installed — run `pip install ddgs`") try: diff --git a/run_agent.py b/run_agent.py index 999dcfe1ff..0bcb45cf74 100644 --- a/run_agent.py +++ b/run_agent.py @@ -210,6 +210,8 @@ def _pool_may_recover_from_rate_limit(pool) -> bool: """Wait for credential-pool rotation (True) or fall back to ``fallback_model`` (False) after a 429. Rotation only helps when the pool has somewhere to go; a single-credential pool would retry the same quota. + + See issues #11314 and #13636. """ return pool is not None and pool.has_available() and len(pool.entries()) > 1 @@ -346,6 +348,11 @@ class AIAgent( from hermes_cli.profiles import get_active_profile_name profile_for_session = get_active_profile_name() except Exception: + # Persist the profile name EXPLICITLY, including "default". NULL used to stand in for the + # default profile, but the #94724 legacy-owner backfill already stamps literal "default" + # onto old rows, and profile-keyed consumers (sidebar scope matching, + # @session:<profile>/<id> deep links) treat NULL as unowned — rows minted NULL after the + # one-shot backfill vanished from the sidebar (#99222). profile_for_session = None # Carry the gateway routing identity: when the gateway SessionStore degraded to JSONL (corrupt # state.db) this lazy create is the ONLY durable write, and an identity-less row is unrecoverable. @@ -414,6 +421,7 @@ class AIAgent( self._usage_anchor = None self._turn_base_usage_anchor = None + # Turn counter (added after reset_session_state was first written — #2635) self._user_turn_count = 0 # Copilot x-initiator: True for the first API call of a user turn, False for tool-loop follow-ups. self._is_user_initiated_turn = False @@ -701,7 +709,16 @@ class AIAgent( def _is_ollama_glm_backend(self) -> bool: """Ollama-hosted GLM models misreport finish_reason='stop'. Matches only explicit Ollama signatures (port 11434, "ollama" in URL, provider ollama), never arbitrary local proxies; excludes Ollama Cloud - (``ollama.com`` / ``:cloud``), which reports faithfully — rewriting it would manufacture truncations.""" + (``ollama.com`` / ``:cloud``), which reports faithfully — rewriting it would manufacture truncations. + + Crucially it does NOT match arbitrary local/private endpoints (LiteLLM/sglang/vLLM/LM Studio + proxies, Tailscale boxes), which report finish_reason correctly and were the source of #13971's + false-positive truncation continuations. + Two signatures identify it: the ``ollama.com`` host (provider ``ollama-cloud``) and the ``:cloud`` + model suffix (cloud generation proxied through a local 11434 endpoint, #98406). Applying the + stop→length rewrite to them manufactures false truncations and causes the continuation nudge to + consume the model's output budget on the next retry, making further false-positives more likely. + """ model_lower = (self.model or "").lower() provider_lower = (self.provider or "").lower() if "glm" not in model_lower and provider_lower != "zai": @@ -758,6 +775,8 @@ class AIAgent( # Structural clone at the single chokepoint: the fork sanitizes in place, and a shallow copy would # alias the live history's nested tool_calls/content. + # Structural clone at the single chokepoint every review path (automatic, /refine, idle-queue + # deferral) goes through. See #100795. from agent.turn_finalizer import _clone_background_review_messages kwargs = dict(messages_snapshot=_clone_background_review_messages(messages_snapshot), review_memory=review_memory, review_skills=review_skills, focus=focus, task_cfg=task_cfg) @@ -872,6 +891,12 @@ class AIAgent( Uses ``original_user_message`` (``user_message`` may carry injected skill content). Interrupted turns are skipped: partial output is not durable truth. Best-effort — an offline backend never blocks. + + A partial assistant output, an aborted tool chain, or a mid-stream reset is not durable + conversational truth — mirroring it into an external memory backend pollutes future recall with + state the user never saw completed. The prefetch is gated on the same flag: the user's next message + is almost certainly a retry of the same intent, and a prefetch keyed on the interrupted turn would + fire against stale context. See #15218. """ if interrupted or not (self._memory_manager and final_response and original_user_message): return @@ -955,6 +980,11 @@ class AIAgent( def _drop_shared_client(self, close_fn: Callable[[Any], None]) -> None: """Hand the shared OpenAI/httpx client to ``close_fn`` and clear the attribute.""" + # Retire the OpenAI/httpx client to release sockets immediately. #70773: eviction runs on the + # gateway's memory-manager thread — a cross-thread hard close of the shared client can release TLS + # FDs under a still-unwinding worker (FD-recycle → SQLite corruption). Retirement shuts the pooled + # sockets down (the memory/socket win we want here) and lets GC release the FDs once no thread holds + # them. client = getattr(self, "client", None) if client is not None: close_fn(client) @@ -991,6 +1021,7 @@ class AIAgent( if getattr(self, "_owns_session_db", False) and session_db is not None: self._owns_session_db = False # Shared instances no-op on close(); release the refcount so the registry closes on the last caller. + # See #90837. from hermes_state import release_or_close release_or_close(session_db) @@ -1098,6 +1129,10 @@ class AIAgent( return False # A native compaction checkpoint makes a carrier never thinking-only, regardless of api_mode or # reasoning field. Checked above every reasoning branch so no carrier shape is dropped. + # The checkpoint is the server-side stand-in for already-pruned history and exists in exactly one + # place; the codex_responses adapter also surfaces commentary text via msg["reasoning"], so the + # string branch below would otherwise drop a carrier before the sidecar is ever inspected. See + # #82108. from agent.native_compaction import has_compaction_checkpoint if has_compaction_checkpoint(msg.get("codex_reasoning_items")): @@ -1194,7 +1229,10 @@ class AIAgent( def _has_pending_fallback(self) -> bool: """Whether a fallback provider remains (mirrors ``try_activate_fallback``'s guard) — gates the - "trying fallback..." status so we never announce one that won't be attempted.""" + "trying fallback..." status so we never announce one that won't be attempted. + + See #17446. + """ return getattr(self, "_fallback_index", 0) < len(getattr(self, "_fallback_chain", None) or []) _restore_primary_runtime = _forward("agent.agent_runtime_helpers", "restore_primary_runtime") diff --git a/scripts/release.py b/scripts/release.py index b906fdd4fb..10eb846a63 100755 --- a/scripts/release.py +++ b/scripts/release.py @@ -2206,6 +2206,11 @@ def update_version_files(semver: str, calver_date: str): r'^version\s*=\s*"[^"]+"', f'version = "{semver}"', pyproject, + # Turn-end file-mutation verifier footer appended by run_agent.py + # (``_format_file_mutation_failure_footer``). It's a UI affordance — reading "warning file mutation + # verifier, 2 files were NOT modified..." aloud is noise (#40772). The footer is a ``⚠️ + # File-mutation verifier:`` header line followed by indented ``•`` bullet lines; strip the whole + # block. flags=re.MULTILINE, ) PYPROJECT_FILE.write_text(pyproject, encoding="utf-8") diff --git a/skills/productivity/google-workspace/scripts/gws_bridge.py b/skills/productivity/google-workspace/scripts/gws_bridge.py index 5fa7a9daca..9cd33a7d37 100755 --- a/skills/productivity/google-workspace/scripts/gws_bridge.py +++ b/skills/productivity/google-workspace/scripts/gws_bridge.py @@ -46,6 +46,13 @@ def refresh_token(token_data: dict) -> dict: "client_id": token_data["client_id"], "client_secret": token_data["client_secret"], "refresh_token": token_data["refresh_token"], + # The refresh token goes in BOTH the body and the ``x-nous-refresh-token`` header. Portal's token + # endpoint requires ``refresh_token`` in the body (its request schema rejects a header-only request + # as ``invalid_request``), and additionally reconciles the header against the body — sending both + # lets Portal keep the value out of body-access-logs while still satisfying the schema. The header + # name must match Portal's ``REFRESH_TOKEN_HEADER`` exactly (``x-nous-refresh- token``); any other + # name is silently ignored. (Verified against the NAS #293 preview deploy: header-only → 400 + # invalid_request; body → accepted.) "grant_type": "refresh_token", }).encode() diff --git a/tools/ansi_strip.py b/tools/ansi_strip.py index 6a81fb2288..f389962416 100644 --- a/tools/ansi_strip.py +++ b/tools/ansi_strip.py @@ -30,6 +30,13 @@ _CONTROL_CHARS_RE = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]") # Unicode TAG chars (U+E0000–U+E007F) render as nothing but LLM tokenizers see them: # the "ASCII smuggling" injection channel. Emoji tag sequences (TR51: U+1F3F4 base + # tag spec + U+E007F CANCEL TAG, e.g. Scotland/Wales flags) are the only legit use. +# Deprecated as language tags, these render as nothing in every terminal and chat UI but are perfectly +# visible to an LLM tokenizer — the classic "ASCII smuggling" prompt-injection channel (hide +# `\u{E0069}\u{E0067}\u{E006E}...` = invisible instructions inside otherwise benign tool output). Ported +# from block/goose#10746. The ONLY legitimate modern use is emoji tag sequences (Unicode TR51): a U+1F3F4 +# black-flag base followed by tag spec characters and the U+E007F CANCEL TAG terminator (e.g. the flags of +# Scotland/Wales/England). goose strips those too; we preserve them — same rationale as keeping ZWJ inside +# emoji sequences. _UNICODE_TAG_SUB_RE = re.compile( r"(\U0001F3F4[\U000E0020-\U000E007E]+\U000E007F)" # valid emoji tag seq (kept) r"|[\U000E0000-\U000E007F]" # any other tag char (stripped) @@ -48,7 +55,14 @@ def sanitize_display_text(text: str) -> str: sequences AND bare control chars, keeping only newlines/tabs (CRs become newlines so ``\\r``-overwrite spoofing can't hide content). Rich's ``Text()`` does NOT neutralize raw escape bytes, so a replayed ``/resume`` message must not be able - to clear the screen, retitle the window, or restyle UI.""" + to clear the screen, retitle the window, or restyle UI. + + Use this when re-rendering conversation history or other persisted text in a terminal UI (e.g. the + ``/resume`` recap): a message that arrived with embedded escapes — pasted content, gateway-origin text, + or model output echoing injected tool results — must not be able to clear the screen, retitle the + window, move the cursor, or restyle adjacent UI when replayed. Mirrors openai/codex#31494 + (``sanitize_user_text``). + """ if not text or not _HAS_CONTROL.search(text): return text text = strip_ansi(text) @@ -59,7 +73,11 @@ def sanitize_display_text(text: str) -> str: def strip_unicode_tags(text: str) -> str: """Remove invisible Unicode TAG chars (a prompt-injection smuggling channel in - untrusted tool output); valid emoji tag sequences are preserved.""" + untrusted tool output); valid emoji tag sequences are preserved. + + Returns the input unchanged (fast path) when no plane-14 tag characters are present. Ported from + block/goose#10746. + """ if not text or not _HAS_UNICODE_TAG.search(text): return text return _UNICODE_TAG_SUB_RE.sub(lambda m: m.group(1) or "", text) diff --git a/tools/approval.py b/tools/approval.py index c6101aa7a7..2755ebf708 100644 --- a/tools/approval.py +++ b/tools/approval.py @@ -113,6 +113,8 @@ def _denial_breaker_addendum(session_key: str) -> str: threshold = _get_denial_breaker_threshold() if threshold <= 0 or count < threshold: return "" + # WARNING (was DEBUG): a failed/blocked guardian call is a real event the operator needs to see — the + # whole point of #82846 is that the hang was invisible. Log the elapsed time and error class too. logger.warning( "Smart-approval circuit breaker tripped for session %s: %d consecutive denials (threshold %d)", session_key, count, threshold, @@ -126,6 +128,8 @@ def _denial_breaker_addendum(session_key: str) -> str: # --- Gateway approval queue (the blocking wait loop lives in approval_gateway_wait) --------------------------------- +# Optional free-text reason supplied with an explicit deny (``/deny <reason>``) so the agent can adapt +# instead of only hearing "denied". Ported from qwibitai/nanoclaw#2832. _gateway_queues: dict[str, list] = {} # session_key → [_ApprovalEntry, …] _gateway_notify_cbs: dict[str, object] = {} # session_key → callable(approval_data) @@ -644,6 +648,8 @@ def _smart_gate(spec: _GateSpec, command: str, description: str, pattern_key: st if human_present: return None, True return { + # Unattended programmatic platforms (webhook/msgraph_webhook/ api_server): respect unattended_mode + # config. Resolves instantly — never a pending approval nobody can answer (#37284, #87509). "approved": False, "message": (f"BLOCKED by smart approval: {description}. The command was assessed as genuinely " f"dangerous. Do NOT retry.{_denial_breaker_addendum(session_key)}"), @@ -1057,6 +1063,11 @@ def check_execute_code_guard(code: str, env_type: str, has_host_access: bool = F Documented limitation: a purely local non-interactive non-gateway session returns approved (the terminal auto-approve contract); the hardline floor still blocks catastrophic ``terminal()`` commands the script issues. + + See #30882. + The hardline floor still blocks catastrophic ``terminal()`` commands the script issues; running + arbitrary code headlessly without any approval surface is trusted-by-config (set a gateway/ask surface + or ``approvals.cron_mode`` to require approval). See #30882. """ pattern_key = "execute_code" description = _EXECUTE_CODE_DESCRIPTION diff --git a/tools/approval_context.py b/tools/approval_context.py index 7dd0da0364..150522f747 100644 --- a/tools/approval_context.py +++ b/tools/approval_context.py @@ -135,7 +135,13 @@ _UNATTENDED_APPROVAL_PLATFORMS = frozenset({"webhook", "msgraph_webhook", "api_s def _is_unattended_platform_approval_context() -> bool: - """True when the session platform is a programmatic/unattended surface.""" + """True when the session platform is a programmatic/unattended surface. + + Webhook, msgraph_webhook, and api_server sessions bind ``HERMES_SESSION_PLATFORM`` like chat gateways + do, but there is no human who can resolve a pending approval. Treating them as gateway approval contexts + blocks the session for the full approval timeout (60-300s) and then fails closed anyway — the deadlock + in #37284/#87509. + """ return _get_session_platform() in _UNATTENDED_APPROVAL_PLATFORMS @@ -156,6 +162,12 @@ def _is_gateway_approval_context() -> bool: context even when it originated from a platform (cron binds the platform for delivery routing): falling through would submit a pending approval with no listener and block the job indefinitely; unattended platforms likewise. + + Unattended programmatic platforms (webhook, msgraph_webhook, api_server) are excluded for the same + reason: those adapters have no ``send_exec_approval`` and no way to receive ``/approve`` replies. + Submitting a pending approval there blocks the session for the full approval timeout (60-300 s) with no + human who can resolve it (#37284, 87509). Their dangerous-command handling is governed by + ``approvals.unattended_mode`` config (default deny), mirroring cron. """ from tools import approval as _a if _a._is_cron_approval_context() or _is_unattended_platform_approval_context(): diff --git a/tools/approval_floors.py b/tools/approval_floors.py index f288ffeca7..5a648fc910 100644 --- a/tools/approval_floors.py +++ b/tools/approval_floors.py @@ -74,6 +74,12 @@ def _save_blocked_payload(command: str) -> str | None: "# Auto-saved by Hermes: this command exceeded the inline command\n" "# parser limit and was blocked from direct execution. Review it,\n" f"# then run it via: bash {path}\n" + command + ("" if command.endswith("\n") else "\n"), + # Force UTF-8 + lossy decode so non-UTF-8 child output can't crash the gateway thread on + # locale-mismatched Windows (#53137). + # Force UTF-8 + lossy decode so non-UTF-8 child output can't crash the gateway thread on + # locale-mismatched Windows (#53137). + # Force UTF-8 + lossy decode so non-UTF-8 child output can't crash the gateway thread on + # locale-mismatched Windows (#53137). encoding="utf-8", errors="replace", ) return str(path) @@ -129,6 +135,7 @@ def _sudo_stdin_block_result(description: str) -> dict: # literal to the outer shell — but they become executable again if an option like `-c`/`-e`/`--eval` (or a git `-c # alias.x=!...`) hands the quoted argument to another interpreter, so quoted control chars only disqualify a command # when such an option is present. +# Port of can1357/oh-my-pi#7553. _SHELL_CONTROL_CHARS = frozenset("\n\r;&|<>`$()") _REINTERPRETED_ARGUMENT_RE = re.compile(r"(?:^|[ \t])(?:-[^-\s]*[ce]|--(?:command|eval))(?:[= \t]|$)") diff --git a/tools/approval_gateway_wait.py b/tools/approval_gateway_wait.py index d999a9a081..f75044661b 100644 --- a/tools/approval_gateway_wait.py +++ b/tools/approval_gateway_wait.py @@ -54,6 +54,9 @@ def _poll_event(event: threading.Event, session_key: str, *, interrupt_log: str) heartbeat = activity_heartbeat("waiting for user approval") with human_wait_window(session_key): while True: + # The poll loop below is verifiably blocked on a human answer (the user tapping approve/deny on + # the gateway surface), bounded by the approval timeout. Record it as human-wait time so the + # concurrent batch deadline excludes it (#79719). if is_interrupted(): logger.info(interrupt_log, session_key) return "interrupted" diff --git a/tools/approval_human_wait.py b/tools/approval_human_wait.py index d4573370e4..4020d3aae0 100644 --- a/tools/approval_human_wait.py +++ b/tools/approval_human_wait.py @@ -17,6 +17,16 @@ import threading import time +# ========================================================================= Human-wait accounting (per +# session) ========================================================================= Tracks the wall-clock +# time the agent spends verifiably blocked on a HUMAN prompt (CLI approval prompt, gateway approval +# round-trip). The concurrent tool batch deadline in agent/tool_executor.py excludes this time so a slow +# human answer never times a batch out — but ONLY this time. Measuring human waits at the source (rather +# than residency in the authorization gate, which is arbitrary code) is what keeps a wedged pre_tool_call +# plugin or a dead approval client from growing the exclusion 1:1 with wall clock and defeating the deadline +# entirely (#79719). Keyed by session so one gateway session's pending approval cannot extend a different +# session's batch deadline. State is process-global like the rest of this module's approval state; entries +# are bounded by _HUMAN_WAIT_MAX_SESSIONS. class _HumanWaitState: __slots__ = ("pending", "window_started", "completed_seconds") @@ -102,7 +112,10 @@ def human_wait_window(session_key: str | None = None): excludes this time; wrapping anything else re-creates the hang where arbitrary wedged code pushes the deadline out forever. Overlapping windows for the same session coalesce (pending counter), so two serialized approval - prompts don't double-count the same wall clock.""" + prompts don't double-count the same wall clock. + + See #79719. + """ key = _resolve_key(session_key) now = time.monotonic() with _human_wait_lock: @@ -135,7 +148,13 @@ def human_wait_seconds(session_key: str | None = None) -> float: safe direction: the deadline fires sooner). Deadline consumers snapshot a baseline at batch start and use the delta. Each window's contribution is clamped to :func:`human_wait_ceiling` (belt-and-braces against the - wedged-window hang).""" + wedged-window hang). + + Each window's contribution is clamped to :func:`human_wait_ceiling`: every legitimate human wait + self-terminates at ``approvals.timeout`` (both the CLI prompt join and the gateway poll loop enforce + it), so a window that overstays that bound is itself wedged and must not keep extending a batch deadline + (belt-and-braces for #79719). + """ key = _resolve_key(session_key) now = time.monotonic() # Resolve the clamp outside the lock: it reads the config cache, which must never nest under _human_wait_lock. diff --git a/tools/approval_prompt.py b/tools/approval_prompt.py index 632949becd..d0fcf5719f 100644 --- a/tools/approval_prompt.py +++ b/tools/approval_prompt.py @@ -34,12 +34,15 @@ def prompt_dangerous_approval(command: str, description: str, timeout_seconds: i Returns 'once', 'session', 'always', 'deny', or 'timeout'. 'timeout' means no user response — still blocked (fail-closed), but callers report "no response" rather than an explicit denial. + + See #81887. """ from tools import approval as _a if timeout_seconds is None: timeout_seconds = _a._get_approval_timeout() # Everything below is a human prompt (callback panel or input() fallback, both bounded by the approval deadline): # record it as human-wait time so the concurrent batch deadline excludes it. + # See #79719. with human_wait_window(): return _ask_human(command, description, timeout_seconds, allow_permanent, approval_callback, allow_session, smart_denied) @@ -101,6 +104,11 @@ def _ask_human(command: str, description: str, timeout_seconds: int, allow_perma # invisible deadlock. Deny loudly instead; threads needing interactive approval must install a callback via # tools.terminal_tool.set_approval_callback() first. try: + # Deny fast and log loudly instead so the caller can surface a real error to the agent. Any thread + # that needs interactive approval must install a callback via + # tools.terminal_tool.set_approval_callback() before reaching this point (see delegate_tool.py, + # run_agent.py _execute_tool_calls_concurrent / _spawn_background_review for the established + # pattern). See #15216. from prompt_toolkit.application.current import get_app_or_none if get_app_or_none() is not None: logger.warning("Dangerous-command approval requested on a thread with no " diff --git a/tools/approval_smart.py b/tools/approval_smart.py index eefa4bf532..6c11c377b7 100644 --- a/tools/approval_smart.py +++ b/tools/approval_smart.py @@ -72,13 +72,18 @@ def _get_smart_policy() -> str: def _smart_approve(command: str, description: str) -> str: - """Ask the auxiliary LLM; return 'approve', 'deny', or 'escalate' (uncertain/failed).""" + """Ask the auxiliary LLM; return 'approve', 'deny', or 'escalate' (uncertain/failed). + + Inspired by OpenAI Codex's Smart Approvals guardian subagent (openai/codex#13860). + """ _smart_t0 = time.monotonic() try: from agent.auxiliary_client import _get_task_timeout, call_llm # Pass the timeout explicitly AND log call + duration: this synchronous call gates EVERY flagged command, and # a stalled provider once froze turns for tens of minutes with zero log output. + # Pass the same configured value explicitly (belt) and log the call + duration (suspenders) so a + # hang is visible in the logs instead of silent. See #72500, #82846. smart_timeout = _get_task_timeout("approval") logger.debug("Smart approvals: assessing risk for command (timeout=%ss)", smart_timeout) system_prompt = _SYSTEM_PROMPT diff --git a/tools/arg_coercion.py b/tools/arg_coercion.py index 3287565def..5410ace9ac 100644 --- a/tools/arg_coercion.py +++ b/tools/arg_coercion.py @@ -97,6 +97,8 @@ def _normalize_json_strings_for_schema(value: Any, schema: Any) -> Any: Schema-guided: a string is only parsed when its schema position expects a container, so legitimate JSON-looking ``type: string`` fields survive. Returns the same object when nothing changed (identity = cheap no-op check). + + Ported from cline/cline#11803, adapted to hermes-agent's coercion layer. """ if not isinstance(schema, dict): return value diff --git a/tools/async_delegation.py b/tools/async_delegation.py index 7902293d2b..5ce576f59a 100644 --- a/tools/async_delegation.py +++ b/tools/async_delegation.py @@ -132,7 +132,14 @@ def _initialize_schema(conn: sqlite3.Connection) -> None: @contextmanager def _transaction() -> Iterator[sqlite3.Connection]: """Open a connection, commit/rollback on exit, and ALWAYS close it (``with conn:`` - alone leaks the connection and WAL/SHM fds until GC).""" + alone leaks the connection and WAL/SHM fds until GC). + + ``sqlite3.Connection.__enter__``/``__exit__`` only commit or roll back the transaction; they do not + close the connection. Using ``with _connect()`` alone therefore leaks a connection — and its WAL/SHM + file descriptors — on every durable dispatch, completion, and delivery-claim, deferring the close to the + garbage collector. On a long-running gateway that exhausts ``RLIMIT_NOFILE`` (the cron-ledger sibling of + this bug was #69567 / PR #69594). + """ conn = _connect() try: with conn: @@ -251,7 +258,15 @@ def restore_undelivered_completions(target_queue) -> int: Restored events are stamped ``restored=True`` in memory only: they came from a PREVIOUS process, so drains without an ownership filter must leave them for a consumer that can prove ownership. Rows older than ``_MAX_COMPLETION_REPLAY_AGE_S`` are terminally dropped - instead of replaying a turn nobody is waiting on.""" + instead of replaying a turn nobody is waiting on. + + Every restored event is stamped ``restored=True`` (in-memory only — the stamp is added after the durable + payload is deserialized and is never persisted). Restored events originate from a *previous* process, so + no consumer in THIS process implicitly owns them: drain paths that run without an ownership filter (the + legacy single-session behavior) must leave them queued for a consumer that can positively prove + ownership, otherwise a brand-new session adopts a dead session's delegation results seconds after boot + (#64484). + """ recover_abandoned_delegations() now, restored = time.time(), 0 with _DB_LOCK, _transaction() as conn: @@ -795,7 +810,10 @@ def _children_activity_from_token(token: Any, now: float) -> Optional[List]: def list_async_delegations() -> List[Dict[str, Any]]: """Snapshot of async delegations (running + recently completed) without callables or private monitor bookkeeping; adds computed live fields for UIs (``seconds_since_progress``, - ``children_activity``/``in_tool`` sampled from ``progress_fn``) and stall context once tripped.""" + ``children_activity``/``in_tool`` sampled from ``progress_fn``) and stall context once tripped. + + Safe to call from any thread. See #51690. + """ now = time.time() samplers: Dict[str, Callable] = {} with _records_lock: diff --git a/tools/bot_failure_reasons.py b/tools/bot_failure_reasons.py index 7e9359d4df..75aeb3163b 100644 --- a/tools/bot_failure_reasons.py +++ b/tools/bot_failure_reasons.py @@ -48,6 +48,7 @@ def is_auto_retryable(reason: str) -> bool: # sanctioned context mutation) on the same session first; everything else # (auth/quota/config/model/unknown) is never auto-retried — it can't be fixed by # a retry and only burns quota. +# See #93091. RETRY_RESUME = "resume" RETRY_COMPRESS_THEN_RESUME = "compress_then_resume" RETRY_NONE = "none" diff --git a/tools/bot_mode_dm.py b/tools/bot_mode_dm.py index ebad41e988..cd56958e64 100644 --- a/tools/bot_mode_dm.py +++ b/tools/bot_mode_dm.py @@ -257,6 +257,8 @@ def _try_relay_delivery(root: Path, raw_target: str, content: str, me: str, *, envelope = enqueue_envelope(root, target=match, message=content, sender_profile=me, sender_handle=_handle(me)) except EnvelopeRefusedError as exc: # Fail fast: target definitively offline — nothing was queued. + # Structured refusal so the agent can distinguish it from a resolution error ('runtime_offline' + # per the #93091 reason enum). return json.dumps({"error": str(exc), "reason": exc.reason}) label = f"@{match['handle']} on {match['connection_label'] or match['connection_id']}" return _spawn_delivery(waiter_command(root, envelope), label, task_id=task_id, agent=agent) @@ -321,9 +323,13 @@ def _delivery_lock(argv: list[str], *, stdin_file: bool): """Per-profile turn lock for a LOCAL teammate delivery: local and relay deliveries into one profile both run a Bot Chat turn here, so the turn window is serialized on ``tools.bot_relay``'s cross-process lock. Peer transports (stdin mode) are locked - on the remote gateway by its own deliver path.""" + on the remote gateway by its own deliver path. + + See #93091. + """ # Match the CLI element by basename: argv[0] may be an absolute venv path # (service contexts lack PATH) and carries .exe on Windows; split on both separators. + # Split on both separators so the shape matches regardless of which platform built the argv. See #93590. cli = (argv[0] if argv else "").rsplit("\\", 1)[-1].rsplit("/", 1)[-1] if stdin_file or len(argv) < 3 or cli not in ("hermes", "hermes.exe") or argv[1] != "-p": return contextlib.nullcontext() @@ -352,6 +358,7 @@ def _run_local_turn(argv: list[str], dm_file: str) -> int: if proc.returncode != 0 and "already has a live owner" in (proc.stderr or ""): # The target's Bot Chat is held live by another surface (Desktop); the turn # never ran — tell the sender plainly instead of leaking a raw lease error. + # See #100523. who = argv[argv.index("-p") + 1] if "-p" in argv[:-1] else "the teammate" print(json.dumps({ "error": f"Delivery failed: @{who}'s Bot Chat is open on another " @@ -371,7 +378,14 @@ def _run_local_turn(argv: list[str], dm_file: str) -> int: def _run_delivery(argv: list[str], dm_file: str, *, stdin_file: bool) -> int: """Run one DM transport and remove its plaintext file after consumption. The turn window (not the enqueue) holds the target profile's cross-process lock, so two - deliveries into one profile queue; a bounded wait ends in a 'target_busy' refusal.""" + deliveries into one profile queue; a bounded wait ends in a 'target_busy' refusal. + + Local (query-file) turns get one policy-gated retry (#93091 item 5): transient failures re-run the same + session; a context_overflow re-run lets the retried turn's pre-API compaction pass compact the Bot Chat + transcript first (agent/conversation_loop.py) — the sanctioned compression lever; no fresh session is + ever minted. Auth/quota/config failures never retry. Peer transports (stdin mode) retry on their own + gateway's deliver path, not here. + """ try: with _delivery_lock(argv, stdin_file=stdin_file): if not stdin_file: @@ -455,6 +469,7 @@ def _delivery_main(args: list[str]) -> int: # 'target_busy': the queued delivery gave up after its bounded wait — surface the # structured payload on stdout so the completion notification carries it back. if getattr(exc, "reason", "") == "target_busy": + # See #93091. print(json.dumps({"error": str(exc), "reason": "target_busy"})) else: print(f"message_agent delivery failed: {type(exc).__name__}: {exc}", file=sys.stderr) diff --git a/tools/bot_relay.py b/tools/bot_relay.py index a714b8cce8..08b45fef25 100644 --- a/tools/bot_relay.py +++ b/tools/bot_relay.py @@ -51,7 +51,11 @@ ROSTER_FRESH_SECONDS = 600 class EnvelopeRefusedError(RuntimeError): - """``enqueue_envelope`` refused to queue (nothing written); ``reason`` is a stable machine code.""" + """``enqueue_envelope`` refused to queue (nothing written); ``reason`` is a stable machine code. + + ``reason`` is a stable machine code; ``str(exc)`` is the human text. 'runtime_offline' matches the + #93091 item-1 failure-reason enum (plain literal here so the branches merge cleanly). + """ def __init__(self, reason: str, message: str): super().__init__(message) @@ -241,7 +245,12 @@ def _expire_if_stale(root: Path | str, path: Path, ttl: float, now: float) -> bo def claim_pending_envelopes(root: Path | str) -> list[dict]: """Drain the outbox (rename → claimed/ so a second drain can't double-deliver). - TTL-expired envelopes get a 'queued_expired' reply and are removed instead.""" + TTL-expired envelopes get a 'queued_expired' reply and are removed instead. + + Envelopes older than ``bot_mode.envelope_ttl_seconds`` are NOT delivered: each gets an error reply + (reason ``'queued_expired'``) so the sender's waiter resolves, and its outbox file is removed (#93091 + item 2). + """ base = _ensure_dirs(root) _sweep_stale(base) ttl = _envelope_ttl_seconds() @@ -318,6 +327,8 @@ def waiter_command(root: Path | str, envelope: dict) -> str: # raw literal parses the folded backslash literally. No-op on POSIX, and \' # still cannot terminate a raw literal, so the injection defense holds. code = ( + # Encode label with !r so roster fields cannot break out of the generated python -c source (quotes, + # parens, or extra statements in connection_id). See #93590. "import json,os,sys,time\n" f"p = r{reply_path!r}\n" f"label = r{label!r}\n" @@ -328,6 +339,7 @@ def waiter_command(root: Path | str, envelope: dict) -> str: " if d.get('error'):\n" # Typed reason code rides ahead of the free text so the sender can # branch on it without parsing provider prose. + # See #93091. " code = str(d.get('reason') or '').strip()\n" " tag = ' [reason: ' + code + ']' if code else ''\n" " print('Delivery to ' + label + ' failed' + tag + ': ' + d['error'])\n" @@ -346,7 +358,15 @@ def waiter_command(root: Path | str, envelope: dict) -> str: def _hermes_cli() -> str: """hermes CLI beside this interpreter, then ``shutil.which``, then the bare name - (service contexts lack PATH, so a bare "hermes" died with ENOENT).""" + (service contexts lack PATH, so a bare "hermes" died with ENOENT). + + The deliver RPC runs on the target gateway, whose process is the venv python — its bin/Scripts directory + holds the matching ``hermes`` entrypoint. A bare ``"hermes"`` relies on PATH, which is exactly what + service contexts (systemd units, desktop launchers, non-login SSH shells) do not provide, so delivery + died with ENOENT there (#93590). When no sibling exists (e.g. running from a source tree without an + installed script), a ``shutil.which`` lookup runs next — it honors whatever PATH the process does have — + before falling back to the bare name, preserving today's behavior for interactive shells. + """ sibling = Path(sys.executable or "").parent / ("hermes.exe" if sys.platform == "win32" else "hermes") return str(sibling) if sibling.is_file() else shutil.which("hermes") or "hermes" @@ -363,8 +383,19 @@ def local_delivery_command(profile: str, query_file: str) -> list[str]: # crashed turn can never wedge the profile. +# ── per-profile turn lock (#93091) ─────────────────────────────────────────── Two deliveries into the SAME +# target profile must never run their Bot Chat turns concurrently: deliveries spawn separate ``hermes`` +# subprocesses, so an in-memory mutex is useless — the lock is a per-profile lockfile under +# ``<root>/bot_relay/locks/`` held with ``fcntl.flock`` for exactly the turn execution window. flock is +# released by the kernel when the holder's fd closes (including process death), so a crashed turn can never +# wedge the profile. A queued delivery waits up to ``bot_mode.turn_wait_seconds`` and then fails with a +# structured 'target_busy' refusal instead of blocking forever. class TurnBusyError(RuntimeError): - """A delivery turn is already running for the target profile (``waited_seconds`` ≈ time queued).""" + """A delivery turn is already running for the target profile (``waited_seconds`` ≈ time queued). + + ``reason`` is 'target_busy' — extends the #93091 item-1 structured refusal enum. ``waited_seconds`` is + roughly how long the caller queued behind the current turn before giving up. + """ reason = "target_busy" diff --git a/tools/browser_cdp_tool.py b/tools/browser_cdp_tool.py index e31d675f7a..d034d861d4 100644 --- a/tools/browser_cdp_tool.py +++ b/tools/browser_cdp_tool.py @@ -52,6 +52,8 @@ def _redact_cdp_output(value: Any, *, always_paths: tuple = (), flagged_paths: t suffixes propagate only into the matching subtree, so ``base64Encoded`` is honored solely as a sibling on the trusted carrier object — never as ambient trust a ``Runtime.evaluate`` by-value object could spoof. + + See #94138, #94142. """ from agent.redact import redact_sensitive_text if isinstance(value, str): diff --git a/tools/browser_tool.py b/tools/browser_tool.py index a61f7b329a..dc139c61f6 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -49,6 +49,7 @@ def __getattr__(name: str): # Env keys re-added to the agent-browser subprocess AFTER credential stripping. # agent-browser is a Node process loading npm deps: a compromised transitive # dependency could read every Hermes secret from process.env. +# Strip by default, then re-add only the browser-backend keys the worker legitimately needs. See #29157. _BROWSER_PASSTHROUGH_KEYS: tuple[str, ...] = ( "BROWSERBASE_API_KEY", "BROWSERBASE_PROJECT_ID", "BROWSER_USE_API_KEY", "FIRECRAWL_API_KEY", "FIRECRAWL_API_URL", "FIRECRAWL_BROWSER_TTL", @@ -84,6 +85,8 @@ except Exception: _sensitive_query_param_name = lambda url: None # noqa: E731 — best-effort fallback # Browser-provider ABC + registry; per-vendor providers live under # ``plugins/browser/<vendor>/``. Legacy class names are re-exported as shims. +# The dispatcher consults the registry; the legacy class names are re-exported below as backward-compat +# shims for callers that import them from this module. See #25214. from agent.browser_provider import BrowserProvider as CloudBrowserProvider # noqa: F401 (legacy alias) from agent.browser_registry import get_provider as _registry_get_browser_provider # noqa: F401 (test-patchable) try: @@ -161,6 +164,8 @@ AGENT_BROWSER_NPX_SPEC = "agent-browser@^0.26.0" # Process caches (``_cached_X`` + ``_X_resolved`` pairs) for config-derived lookups; # reset by ``cleanup_all_browsers``. Written/read by the sibling modules via the origin. _cached_command_timeout: Optional[int] = None +# Flip the resolved flag BEFORE nulling the cache so a concurrent reader never sees ``resolved=True`` with +# ``cache=None`` (#14331). _command_timeout_resolved = False _cached_snapshot_threshold: Optional[int] = None _snapshot_threshold_resolved = False @@ -440,13 +445,16 @@ _session_last_activity: Dict[str, float] = {} # Owner Hermes home per session: the janitor is one process-global thread, so each # teardown must re-enter the OWNING profile's scope (copy_context at spawn would # pin the first profile's secrets onto every other profile's teardown). +# See #86402. _session_owner_homes: Dict[str, str] = {} # Consecutive janitor failures per session; force-reaped after MAX_INACTIVITY_CLEANUP_FAILURES. +# See #100738. _cleanup_failures: Dict[str, int] = {} MAX_INACTIVITY_CLEANUP_FAILURES = 3 # Session keys flagged suspect after a command timeout (written lock-free by # mark_suspect; consumed by ensure_healthy() at next use, which recycles). +# See #72205. _suspect_browser_sessions: Dict[str, str] = {} @@ -689,6 +697,12 @@ def _url_policy_error(url: str, *, auto_local: bool = False) -> Optional[dict]: f"({sensitive_query_key}). Cloud browser backends are third-party " "readers; use a local browser/CDP session or remove the sensitive " "query parameter before navigating.") + # Always-blocked floor: cloud metadata / IMDS endpoints are denied regardless of backend, hybrid + # routing, or allow_private_urls. There's no legitimate agent use case for navigating to 169.254.169.254 + # / metadata.google.internal / ECS task metadata via a browser, and routing those to a local Chromium + # sidecar on an EC2/GCP/Azure host exfiltrates IAM credentials (#16234). The floor is UNCONDITIONAL — it + # must fire for every backend, including the pure-local headless Chromium and off-host CDP cases (a + # local Chromium on a cloud VM still reaches the host IMDS). if _is_always_blocked_url(url): return _err("Blocked: URL targets a cloud metadata endpoint") if not local and not auto_local and not _allow_private_urls() and not _is_safe_url(url): diff --git a/tools/browser_tool_cloud.py b/tools/browser_tool_cloud.py index 3364811f0a..c5946a4288 100644 --- a/tools/browser_tool_cloud.py +++ b/tools/browser_tool_cloud.py @@ -181,6 +181,8 @@ def _is_local_backend() -> bool: if _bt._get_cloud_provider() is not None: return False # Scope-aware: under gateway multiplexing the routed profile's terminal backend lives in the per-turn scope. + # When terminal runs in a container, browser on host can access internal networks the terminal can't → + # treat as non-local. See #68559. from tools.terminal_scope import terminal_env return terminal_env("TERMINAL_ENV", "local").strip().lower() in ("local", "") diff --git a/tools/browser_tool_install.py b/tools/browser_tool_install.py index ff0043819d..6661f45466 100644 --- a/tools/browser_tool_install.py +++ b/tools/browser_tool_install.py @@ -161,6 +161,14 @@ def warm_agent_browser_npx_cache(timeout: float = 60.0) -> bool: Runs with the credential-scrubbed env every other agent-browser spawn uses (registry-fetched npm code must never see the operator keyring), in its own process group, and tree-kills on timeout so a surviving descendant cannot hold the capture pipe open. Never raises; True only when npx exited 0. + + agent-browser is no longer a root package.json dependency (#43564) — it resolves lazily via ``npx + agent-browser`` instead, which keeps it out of the npm workspace install graph entirely (nothing to + prune it anymore) but means the first real invocation in a session would otherwise pay npx's + registry-lookup/fetch cost. Calling this during ``hermes update`` (or ``hermes doctor --fix``) warms + npx's own cache ahead of time, restoring the "available before any session starts" property + agent-browser had while it was an eager root dependency — without re-entangling it with the workspace + graph. """ _bt = _origin() npx_bin = _bt._resolve_npx_bin() @@ -316,7 +324,12 @@ def check_browser_requirements() -> bool: def check_browser_vision_requirements() -> bool: - """Advertise ``browser_vision`` only with BOTH a working browser AND a vision backend.""" + """Advertise ``browser_vision`` only with BOTH a working browser AND a vision backend. + + Without the vision check, the tool stays in the model's tool list even when no vision provider is + configured, then fails at call time with a cryptic provider-side error like ``unknown variant + `image_url`, expected `text``` (issue #31179). + """ if not _origin().check_browser_requirements(): return False try: diff --git a/tools/browser_tool_lifecycle.py b/tools/browser_tool_lifecycle.py index 396e951982..f907c9636a 100644 --- a/tools/browser_tool_lifecycle.py +++ b/tools/browser_tool_lifecycle.py @@ -141,6 +141,8 @@ def _cleanup_inactive_browser_sessions(): Each teardown runs under its owner profile's scope. A session whose cleanup keeps failing is force-reaped after MAX_INACTIVITY_CLEANUP_FAILURES attempts; only a successful cleanup clears its failure count. + + See #100738, #86402. """ current_time = time.time() @@ -417,7 +419,10 @@ def _stop_browser_cleanup_thread(): def _update_session_activity(task_id: str): """Touch the activity timestamp and record the owning Hermes home on first sight (the - janitor tears down under the owner's scope). Does NOT reset ``_cleanup_failures``.""" + janitor tears down under the owner's scope). Does NOT reset ``_cleanup_failures``. + + See #86402. + """ with _bt._cleanup_lock: _bt._session_last_activity[task_id] = time.time() _bt._session_owner_homes.setdefault(task_id, str(get_hermes_home())) @@ -430,6 +435,17 @@ def _kill_process_tree(proc: "subprocess.Popen") -> None: daemon grandchild keep a capture pipe open so ``communicate()`` never sees EOF, so the whole tree must go (no grace: the caller already burned its timeout). Delegates to :func:`agent.deadline.kill_process_tree`, falling back to the legacy kill. + + ``Popen.kill()`` only signals the direct child PID. npm/npx routinely fork further processes + (registry-fetch helpers, npm's own lifecycle runner, agent-browser's own detached daemon grandchild) + that can survive a plain ``kill()`` of the top-level PID and keep a ``capture_output``-style pipe open, + hanging the caller's ``communicate()`` past the nominal timeout — the same orphaned-pipe hazard already + hit in production on POSIX (see ``tools/process_registry.py``'s ``_reader_loop``, issue 68915: a + backgrounded grandchild inheriting a pipe's write end kept it from ever reaching EOF). That hazard is + cross-platform, not Windows-specific; what *is* Windows-specific is the lack of a remedy other than + killing the tree — anonymous pipes there don't support overlapped I/O, so there's no ``select()``-style + non-blocking read to poll around a stuck grandchild the way POSIX can. Killing the whole process + group/tree the child was launched into reaches those descendants on both platforms. See #68915. """ try: from agent.deadline import kill_process_tree as _deadline_kill_tree @@ -560,7 +576,12 @@ def _kill_verified_daemon(socket_dir: str, session_name: str) -> bool: def _release_session_resources(task_id: str, session_info: Dict[str, Any]) -> None: """Untrack ``task_id``, close its cloud provider session, kill its daemon — the - unconditional tail of a teardown, and the whole of the janitor's force-reap path.""" + unconditional tail of a teardown, and the whole of the janitor's force-reap path. + + The unconditional tail of ``_cleanup_single_browser_session``; also the whole of the janitor's + force-reap path (#100738), which skips the polite agent-browser/Camofox ``close`` that kept failing but + must still release the cloud session and the local Chromium. + """ bb_session_id = session_info.get("bb_session_id", "unknown") _forget_session_tracking(task_id, session=True) @@ -581,7 +602,10 @@ def _release_session_resources(task_id: str, session_info: Dict[str, Any]) -> No def _force_reap_browser_session(task_id: str) -> None: - """Janitor last resort: skip the failing ``close`` round-trips, release resources directly.""" + """Janitor last resort: skip the failing ``close`` round-trips, release resources directly. + + Janitor last resort after repeated cleanup failures (#100738). + """ _bt._stop_cdp_supervisor(task_id) with _bt._cleanup_lock: session_info = _bt._active_sessions.get(task_id) diff --git a/tools/browser_tool_session.py b/tools/browser_tool_session.py index c999de236f..76e55f9275 100644 --- a/tools/browser_tool_session.py +++ b/tools/browser_tool_session.py @@ -336,6 +336,7 @@ def _discard_timed_out_browser_session(task_id: str, session_info: Dict[str, Any return try: # Tree-kill: terminating only the daemon PID leaks the Chromium tree. + # See #68139. from agent import deadline as _deadline _deadline.kill_process_tree(daemon_pid) @@ -388,6 +389,14 @@ def _handle_browser_command_timeout(task_id: str, session_info: Dict[str, Any], Local daemon wedged/dead: tree-kill and evict now (Chromium children would leak). Both local branches ``mark_suspect`` first so the poisoned-cache invariant holds even if eviction races another thread's replacement. + + See #68139, #72205. + * **Local daemon alive** (PID readable, process alive, identity-verified as ours, control socket accepts + a connection): the *command* wedged — page hang, stuck navigation — but the daemon itself is fine. + Killing it would be overkill and slow. Mark the session suspect only; the next use recycles it through + ``ensure_healthy`` → clean agent-browser ``close`` → fresh session. Tree-kill the daemon's process tree + via ``agent.deadline.kill_process_tree`` and evict the cache entry now; the next browser call respawns + from scratch. See #72206. """ if session_info.get("bb_session_id") or session_info.get("cdp_url"): _bt._discard_timed_out_browser_session(task_id, session_info, task_socket_dir) diff --git a/tools/browser_use_cli.py b/tools/browser_use_cli.py index 69ac8fdf72..e28596154c 100644 --- a/tools/browser_use_cli.py +++ b/tools/browser_use_cli.py @@ -140,6 +140,11 @@ def _base_subprocess_env() -> dict: env = _build_browser_env() # The CLI runs under its own Python (uv tool / uvx); an inherited PYTHONPATH/PYTHONHOME # (Hermes's venv) wins over its site-packages → wrong-ABI C-extensions and a crash. + # PYTHONPATH/PYTHONHOME inherited from the agent process point at Hermes's venv site-packages, and a + # child interpreter honors them ahead of its own site-packages — so the CLI imports compiled + # C-extensions (e.g. pydantic_core) built for the wrong interpreter and crashes on ABI mismatch (#83427, + # #84841, #86006, #86104). Strip both — the CLI manages its own environment and never needs Hermes's + # import path. env.pop("PYTHONPATH", None) env.pop("PYTHONHOME", None) env["PATH"] = _floor_subprocess_path(env.get("PATH", "")) diff --git a/tools/budget_config.py b/tools/budget_config.py index 37697f53ca..bf19c0b9df 100644 --- a/tools/budget_config.py +++ b/tools/budget_config.py @@ -23,7 +23,11 @@ MCP_TOOL_PREFIX: str = "mcp_" def _configured_mcp_result_size() -> int: """Read ``tool_budget.mcp_result_size_chars`` via ``load_config_readonly`` (the sanctioned path; raw config.yaml parsing outside owner modules is test-guarded). - Any error, missing key or non-positive value returns the built-in default.""" + Any error, missing key or non-positive value returns the built-in default. + + The ``tool_budget:`` block name is shared with the wider configurable-caps proposal (#80508) so the two + can merge without a key rename. + """ try: from hermes_cli.config import load_config_readonly data = load_config_readonly() @@ -51,7 +55,11 @@ class BudgetConfig: """Priority: pinned -> tool_overrides -> mcp_ prefix -> registry per-tool -> default. MCP tools get ``mcp_result_size`` (no registry entry). MCP and registry values are capped at ``default_result_size`` so a context-scaled budget for a small - model still constrains tools registering a fixed 100K ``max_result_size_chars``.""" + model still constrains tools registering a fixed 100K ``max_result_size_chars``. + + For the default budget this is a no-op because both equal 100K; for a scaled-down budget it prevents + a per-tool registry value from re-inflating the cap past the model's window (#23767). + """ if tool_name in PINNED_THRESHOLDS: return PINNED_THRESHOLDS[tool_name] if tool_name in self.tool_overrides: @@ -84,7 +92,13 @@ def budget_for_context_window(context_length: int | None) -> BudgetConfig: """Return a BudgetConfig scaled to the model's context window: the fixed defaults suit 200K+ models but on 65K one result/turn can fill the window. The proportional value is clamped to the defaults as a CAP (large models - stay byte-identical) and floored so a usable preview always survives.""" + stay byte-identical) and floored so a usable preview always survives. + + The fixed defaults (100K result / 200K turn chars) are correct for large (200K+ token) models but blind + to small ones: on a 65K-token model a single tool result persisted at the 100K-char threshold, or a + 200K-char turn budget (~50K tokens), can by itself approach or exceed the whole window and force an + oversized request (#23767). + """ mcp_result_size = _configured_mcp_result_size() if not context_length or context_length <= 0: if mcp_result_size == DEFAULT_MCP_RESULT_SIZE_CHARS: diff --git a/tools/checkpoint_manager.py b/tools/checkpoint_manager.py index b5fddae36c..f820aca01f 100644 --- a/tools/checkpoint_manager.py +++ b/tools/checkpoint_manager.py @@ -659,7 +659,13 @@ class CheckpointManager: return results def list_all_checkpoints(self) -> List[Dict]: - """Checkpoints across every registered project (most recent first), each tagged ``workdir``.""" + """Checkpoints across every registered project (most recent first), each tagged ``workdir``. + + Surgical reapply of PR #10633 by @nightq (#10505) onto the v2 single-store layout: iterate + ``projects/<hash>.json`` metadata via ``_list_projects`` instead of the pre-v2 per-shadow-dir scan. + Each entry carries the extra ``workdir`` key so callers can label which project a checkpoint belongs + to. + """ store = _store_path(CHECKPOINT_BASE) if not _store_has_head(store): return [] diff --git a/tools/clarify_gateway.py b/tools/clarify_gateway.py index 1c6a72b332..dd03b6e185 100644 --- a/tools/clarify_gateway.py +++ b/tools/clarify_gateway.py @@ -73,6 +73,8 @@ def wait_for_response(clarify_id: str, timeout: float) -> Optional[str]: remaining = 1.0 if deadline is None else deadline - time.monotonic() if remaining <= 0 or entry.event.wait(timeout=min(1.0, remaining)): break + # Periodic activity touch so the gateway's inactivity timeout doesn't kill the agent during long + # code execution (#10807). if touch_activity_if_due is not None: touch_activity_if_due(activity_state, "waiting for user clarify response") with _lock: @@ -270,7 +272,11 @@ def resolve_clarify_timeout(config: dict) -> int: def get_clarify_timeout() -> int: """Clarify timeout from config.yaml; 0/negative = unlimited. Default 3600: long enough that a user who stepped away still finds a live entry when they tap, short enough that - an abandoned prompt eventually unblocks the agent thread instead of pinning the guard.""" + an abandoned prompt eventually unblocks the agent thread instead of pinning the guard. + + The old 600s default evicted the entry mid-think, so a late tap landed on a dead entry and the agent + hung on ``running: clarify`` (#32762). + """ try: from hermes_cli.config import load_config return resolve_clarify_timeout(load_config() or {}) diff --git a/tools/clarify_tool.py b/tools/clarify_tool.py index e3032ce3d2..7cf852221f 100644 --- a/tools/clarify_tool.py +++ b/tools/clarify_tool.py @@ -101,6 +101,8 @@ def _is_timeout(raw) -> bool: return raw is None or (isinstance(raw, str) and raw.strip() == TIMEOUT_RESPONSE) +# ============================================================================= Batch (multi-question) +# support — issue #18450 ============================================================================= def _normalize_questions(questions) -> tuple: """Validate the ``questions`` batch param -> ``(normalized, error)``; an empty list gives ``(None, None)`` (fall back to the single-question path). Entries carry ``qid`` (stable @@ -181,7 +183,19 @@ def clarify_tool(question: str, choices: Optional[List[str]] = None, multi_selec questions: Optional[List[dict]] = None, callback: Optional[Callable] = None) -> str: """Ask one question (``question``/``choices``/``multi_select``) or a batch (``questions`` wins when non-empty). ``callback(question, choices, multi_select=False) -> str`` is - platform injected (batch-capable ones also take ``questions=``). Returns result JSON.""" + platform injected (batch-capable ones also take ``questions=``). Returns result JSON. + + Args: question: The question text to present. choices: Up to 4 predefined answer choices. When + omitted the question is purely open-ended. multi_select: When True, the user can select multiple choices + (checkboxes). The ``user_response`` in the output JSON will be a list of strings instead of a single + string. Has no effect when ``choices`` is omitted. questions: Up to 5 independent questions asked as + one batch (issue #18450). When present (non-empty), the single ``question``/``choices``/``multi_select`` + parameters are ignored and the result JSON is ``{"responses": [...]}`` (plus ``"timed_out": true`` when + the user stopped answering partway). callback: Platform-provided function that handles the actual UI + interaction. Batch-capable platforms additionally accept a ``questions`` keyword and receive the + normalized list in one call; platforms without it are looped one question at a time. Injected by the + agent runner (cli.py / gateway). + """ if questions is not None: normalized, error = _normalize_questions(questions) if error: diff --git a/tools/code_execution_env.py b/tools/code_execution_env.py index e8039c03da..daaee3ae64 100644 --- a/tools/code_execution_env.py +++ b/tools/code_execution_env.py @@ -127,6 +127,10 @@ def _build_child_env(*, rpc_endpoint: str, rpc_token: str, tmpdir: str, # mode changes CWD. Hermes's root is added ONLY when the child runs in Hermes's Python env — # exposing Hermes's site-packages to an external interpreter can mix incompatible compiled # extensions (3.12 NumPy under a 3.9 venv). Inherited Hermes-owned entries are stripped first. + # Before re-injecting PYTHONPATH, strip Hermes-owned entries that leaked through _scrub_child_env + # (PYTHONPATH is in _SAFE_ENV_PREFIXES so it passes the scrub). They are redundant for same-Hermes- + # environment children and may be incompatible with external interpreters (project mode can select a + # different venv), so they must not shadow or poison the child's sys.path (#74817). from tools.environments.local import _strip_hermes_owned_pythonpath _strip_hermes_owned_pythonpath(child_env) _existing_pp = child_env.get("PYTHONPATH", "") @@ -232,7 +236,10 @@ def _resolve_child_python(mode: str) -> str: def _resolve_child_cwd(mode: str, staging_dir: str, task_id: str = "") -> str: """Child cwd. Strict: the staging dir. Project mirrors the terminal/file-tool ladder so every file-writing path agrees: session cwd record (`cd` state) → registered ``session.cwd.set`` - override → TERMINAL_CWD → os.getcwd() → staging dir (never Popen on a missing cwd).""" + override → TERMINAL_CWD → os.getcwd() → staging dir (never Popen on a missing cwd). + + (#56047) + """ if mode != "project": return staging_dir if task_id: diff --git a/tools/code_execution_tool.py b/tools/code_execution_tool.py index 0066eca007..6a306dc1ac 100644 --- a/tools/code_execution_tool.py +++ b/tools/code_execution_tool.py @@ -578,6 +578,7 @@ def _run_remote_per_call(env, env_type: str, code: str, effective_task_id: str, _ship_file_to_remote(env, f"{sandbox_dir}/script.py", code) # Wrapped so the thread inherits the turn's approval context + callbacks # (tools.thread_context) — else sandbox RPC tool calls lose approval routing. + # See #30882. rpc_thread = threading.Thread( target=propagate_context_to_thread(_rpc_poll_loop), daemon=True, args=(env, f"{sandbox_dir}/rpc", effective_task_id, [], tool_call_counter, @@ -638,6 +639,9 @@ def _execute_remote(code: str, task_id: Optional[str], enabled_tools: Optional[L # run-to-completion transport. Spawn failure falls OPEN to the per-call # path below so a degraded remote host never blocks execution. try: + # --- Session-kernel path (hermes-agent#96873) ------------------- Same always-on model as + # local: one persistent kernel per owner, rebuilt on the run-to-completion transport (detached + # runner + file cell protocol). from tools.code_kernel_remote import execute_in_remote_kernel kernel_result = execute_in_remote_kernel( code, env=env, env_type=env_type, task_env_id=effective_task_id, @@ -677,6 +681,7 @@ def execute_code( # Fail closed under a terminal-policy refusal scope: the routed profile's terminal # policy is unresolved, so refuse rather than inherit the launch process's ambient policy. try: + # See #68559. from tools.terminal_scope import enforce_no_refusal enforce_no_refusal() except Exception as refusal: @@ -688,6 +693,10 @@ def execute_code( # Hard-block gateway-lifecycle commands (mirrors the terminal_tool guard — otherwise # `os.system("launchctl bootout ...")` here bypasses it and SIGTERMs the gateway mid-task). # Gated on PID-file ownership, not the inherited env marker. + # Hard-block gateway-lifecycle commands, mirroring the terminal_tool guard (#68289): without this, + # execute_code is a straight bypass — the terminal() path refuses `launchctl bootout ai.hermes.gateway`, + # but the identical command inside `os.system(...)` / `subprocess.run([...])` here sailed through and + # SIGTERM'd the gateway mid-task. from tools.process_registry import _is_supervised_gateway_process if _is_supervised_gateway_process(): from cron.lifecycle_guard import contains_gateway_lifecycle_command @@ -704,6 +713,7 @@ def execute_code( # Arbitrary Python never passes through terminal()/DANGEROUS_PATTERNS, so guard the whole # script before either dispatch path spawns it — in this (tool-executor) thread, which holds # the session context. A Docker sandbox with host bind mounts gets no container fast-path. + # See #30882. from tools.approval import check_execute_code_guard _guard = check_execute_code_guard(code, env_type, has_host_access=_docker_has_host_access(_env_config)) if not _guard.get("approved", False): @@ -834,6 +844,8 @@ def build_execute_code_schema(enabled_sandbox_tools: set = None, ) # Remote hosts that fail open to per-call are not worth schema words; the result's # `kernel` field tells the truth per call. + # Session kernels are always on (kernel_mode retired in #96787): persistence is part of the tool's one + # description, not a bolt-on paragraph behind a dead conditional. description = ( "Run Python that calls Hermes tools programmatically. Use when you " "need 3+ tool calls with logic between them: filtering/reducing " diff --git a/tools/code_kernel.py b/tools/code_kernel.py index e981a28d4e..cdf44ef10c 100644 --- a/tools/code_kernel.py +++ b/tools/code_kernel.py @@ -388,6 +388,7 @@ _KERNELS: Dict[Tuple, SessionKernel] = _REGISTRY.kernels # Bounded lifecycle defaults (config: code_execution.max_session_kernels / kernel_idle_timeout). # A long-lived gateway must never accumulate one live child per finished conversation: # stable owner id, owner-teardown disposal, idle reaping, max-live bound. +# See #88637. DEFAULT_MAX_SESSION_KERNELS = 4 DEFAULT_KERNEL_IDLE_TIMEOUT = 1800 @@ -437,7 +438,10 @@ def shutdown_all_kernels() -> None: def shutdown_kernels_for_owner(owner: str) -> None: """Dispose every kernel a session owns — wired into ``tools.approval.clear_session`` - so kernels die at the same boundary that clears approval/yolo state (/new, session close).""" + so kernels die at the same boundary that clears approval/yolo state (/new, session close). + + See #88637. + """ if owner: _REGISTRY.shutdown(owner) diff --git a/tools/computer_use/backend.py b/tools/computer_use/backend.py index 842ee737c1..50be4d9c2a 100644 --- a/tools/computer_use/backend.py +++ b/tools/computer_use/backend.py @@ -55,6 +55,7 @@ class UIElement: attributes: Dict[str, Any] = field(default_factory=dict) # Opaque per-snapshot handle from cua-driver, passed alongside `index` for explicit stale-detection: a # stale token errors instead of silently re-resolving to a different element. None on older drivers. + # None for pre-#1961 drivers that didn't carry the field. element_token: Optional[str] = None @@ -74,6 +75,7 @@ class CaptureResult: png_bytes_len: int = 0 # raw bytes sent to Anthropic, for token estimation # MIME type of `png_b64` when the backend supplied it (cua-driver-rs emits `mimeType` on every image # part). None → consumers fall back to base64-prefix sniffing (older drivers). + # See #1961, #47072. image_mime_type: Optional[str] = None # Guidance appended to the summary by capture lanes that intentionally return no elements (e.g. # full-screen composited grabs) to point the model at an interactive lane. @@ -86,7 +88,11 @@ class ActionResult: tool/transport success only — NOT the semantic verdict; read ``effect`` / ``escalation`` (cua-driver's structured verdict) to pick the next rung of the verify → escalate ladder. Structured fields are optional and additive: an older driver that omits - ``structuredContent`` leaves them ``None``, behavior unchanged.""" + ``structuredContent`` leaves them ``None``, behavior unchanged. + + Beyond the transport-level ``ok`` flag, this carries cua-driver's structured action verdict so the model + can follow the documented verify → escalate ladder (NousResearch/hermes-agent#67052). + """ ok: bool action: str diff --git a/tools/computer_use/cua_backend.py b/tools/computer_use/cua_backend.py index 75ad3034c5..356d8284fe 100644 --- a/tools/computer_use/cua_backend.py +++ b/tools/computer_use/cua_backend.py @@ -52,7 +52,10 @@ def _cua_no_overlay() -> bool: """Pass ``--no-overlay``? ``computer_use.no_overlay`` overrides; else off on macOS (cursor-overlay redraw loop can peg a core after a session), headless Linux / WSL2 / containers, and Linux X11 (the overlay is a fullscreen always-on-top all-workspaces window with no compositor-owned lifecycle, so an unclean session - end can leave it wedged over every app); on for Windows and Linux Wayland (compositor owns the surface).""" + end can leave it wedged over every app); on for Windows and Linux Wayland (compositor owns the surface). + + Explicit ``True`` / ``False`` overrides auto-detection. See #28152, #47032. + """ val = _computer_use_cfg().get("no_overlay") if val is not None or sys.platform != "linux": return bool(val) if val is not None else sys.platform == "darwin" @@ -60,6 +63,13 @@ def _cua_no_overlay() -> bool: with contextlib.suppress(Exception), open("/proc/version", encoding="utf-8") as f: wsl = "microsoft" in f.read().lower() return wsl or not os.environ.get("DISPLAY") or ( + # Linux/X11: the cursor overlay is a fullscreen, always-on-top, all-workspaces X11 window + # (save-unders path). An unclean session end (agent interrupted mid-capture, stale target window) + # can leave it stuck above every app on every workspace, wedging desktop input until the app + # restarts — the same failure class as the HUD window on Mutter/X11 (#83473). There is no + # compositor-owned surface to tear down with the client connection, so default the overlay off on + # X11 too; set computer_use.no_overlay: false to keep the cursor. Wayland keeps it: the compositor + # owns the overlay surface lifecycle there. os.environ.get("XDG_SESSION_TYPE") != "wayland" and not os.environ.get("WAYLAND_DISPLAY")) def _cua_telemetry_disabled() -> bool: @@ -113,6 +123,8 @@ def sanitized_cua_driver_env() -> Dict[str, str]: never inherit API keys. Falls back to the unsanitized telemetry env if the sanitizer can't import.""" env = cua_driver_child_env() with contextlib.suppress(Exception): + # cua-driver is a third-party binary — never hand it provider API keys via inherited env (same + # policy as the manifest probe and MCP spawn; #53503/#55709/#58889 lineage). from tools.environments.local import _sanitize_subprocess_env return _sanitize_subprocess_env(env) return env @@ -140,6 +152,8 @@ def _linux_session_locked() -> Optional[bool]: """Is the graphical session locked? (Linux; best-effort.) A locked KDE/GNOME session freezes renderers and half-disables the AX tree, so discovery legitimately returns nothing — which otherwise reads as a driver bug. True/False when loginctl answers, None when unavailable (non-Linux, no systemd-logind, probe failure).""" + # Auto-detect: macOS overlay can peg a core indefinitely after a computer_use session (#47032). Prefer + # off until the driver teardown is solid; set computer_use.no_overlay: false to keep the cursor. if sys.platform != "linux": return None try: @@ -295,6 +309,11 @@ class CuaDriverBackend(_CaptureMixin, _InputMixin, ComputerUseBackend): def _clear_active_target(self) -> None: """Forget a capture/focus target so a failed lookup cannot misroute input.""" self._active_pid = self._active_window_id = self._last_app = self._last_target = None + # Surface 6 of NousResearch/hermes-agent#47072: per-snapshot `element_index -> element_token` map + # populated on capture(). Action tools (click/scroll/set_value/...) attach the matching token + # alongside `element_index` so cua-driver detects "stale" explicitly instead of silently + # re-resolving to a different element. Cleared whenever a fresh capture overwrites the snapshot + # context. self._snapshot_tokens: Dict[int, str] = {} def _set_active_target(self, target: Dict[str, Any]) -> None: diff --git a/tools/computer_use/cua_backend_capture.py b/tools/computer_use/cua_backend_capture.py index 23b2e9e564..b8877a8691 100644 --- a/tools/computer_use/cua_backend_capture.py +++ b/tools/computer_use/cua_backend_capture.py @@ -60,7 +60,11 @@ def _select_capture_target(windows: List[Dict[str, Any]], *, app_requested: bool """Best window from z-sorted (frontmost-first) list_windows output. Unqualified default captures on Linux (no app filter, no exact target) skip desktop/shell helper windows first — targetable but capture as empty — and when every remaining candidate shares one ``z_index`` (the common X11 case) - ``_NET_ACTIVE_WINDOW`` beats list order. Exact-target captures never pay for the ``xprop`` probe.""" + ``_NET_ACTIVE_WINDOW`` beats list order. Exact-target captures never pay for the ``xprop`` probe. + + Callers pass windows already sorted by ``z_index`` descending (higher = frontmost). When ordering is + informative, keep that frontmost contract. See #58026. + """ pool = [w for w in windows if not w["off_screen"]] if not exact_target and not app_requested and sys.platform == "linux": pool = [w for w in pool if _is_real_app_window(w)] or pool @@ -260,6 +264,10 @@ class _CaptureMixin: tree, window_title = _tree_and_title(gws_out) # Prefer the canonical structuredContent.elements (real frames); the markdown regex fallback yields # (0,0,0,0) bounds. + # Surface 2 of NousResearch/hermes-agent#47072: prefer the canonical structuredContent.elements + # array (trycua/cua#1961). Falls back to markdown regex parsing for cua-driver builds that didn't + # carry the structured shape — those bounds come back (0,0,0,0); the structured path preserves real + # frames. sc_elements = (gws_out.get("structuredContent") or {}).get("elements") elements = (_parse_elements_from_structured(sc_elements) if isinstance(sc_elements, list) and sc_elements else _parse_elements_from_tree(tree) if tree else []) @@ -297,7 +305,11 @@ class _CaptureMixin: def _capture_full_screen(self, mode: str) -> CaptureResult: """Composited PrtScn-style grab via `get_desktop_state` (the shell window would only show wallpaper + icons). Never enumerates, so it also works when Windows UIA hangs. Pixels only — `elements` is empty and `note` points - the model at the interactive lanes. ``capture_scope`` is switched to desktop for the call and restored afterwards.""" + the model at the interactive lanes. ``capture_scope`` is switched to desktop for the call and restored afterwards. + + Bonus resilience (2ndNatureAI, #60081): this lane works even when Windows UIA enumeration + (`list_windows` / `list_apps`) hangs (trycua/cua#2110/#2113), because it never enumerates. + """ self._clear_active_target() previous_scope: Optional[str] = None try: diff --git a/tools/computer_use/cua_backend_driver.py b/tools/computer_use/cua_backend_driver.py index 9e0442c6aa..2215a0ad7b 100644 --- a/tools/computer_use/cua_backend_driver.py +++ b/tools/computer_use/cua_backend_driver.py @@ -128,7 +128,17 @@ def _resolve_mcp_invocation(driver_cmd: str, *, timeout: float = 6.0) -> Tuple[s """``(command, args)`` that spawn cua-driver's stdio MCP server, asked of the driver itself via ``cua-driver manifest`` (``mcp_invocation``) so a subcommand rename keeps working. Falls back to ``(driver_cmd, ["mcp"])`` on older drivers or any discovery failure — the wrapper must not refuse to start over a failed discovery hop. - ``--no-overlay`` appended when allowed.""" + ``--no-overlay`` appended when allowed. + + Surface 8 of NousResearch/hermes-agent#47072: instead of hardcoding ``["mcp"]`` we ask the driver itself + via ``cua-driver manifest`` (trycua/cua#1961). The manifest carries a stable ``mcp_invocation`` pointer + with both ``command`` and ``args``, so a future cua-driver that renames or relocates the subcommand + keeps working without a Hermes patch. + When ``computer_use.no_overlay`` is enabled (or auto-detected — macOS, headless/WSL2/X11 Linux), + ``--no-overlay`` is appended to suppress the cursor overlay rendering loop that can consume CPU + indefinitely when idle (#28152, #47032). Older drivers that don't recognise the flag will reject it; + callers should fall back to the no-overlay invocation on spawn failure. + """ manifest = _driver_json(driver_cmd, "manifest", timeout=timeout, require_ok=True) or {} invocation = manifest.get("mcp_invocation") args = _valid_mcp_args(invocation) @@ -188,7 +198,10 @@ def cua_driver_update_check(*, timeout: Optional[float] = None) -> Optional[Dict or ``None`` when the binary is missing, the driver predates the verb, the GitHub check failed (``error`` set) or the output didn't parse. Never raises. ``timeout`` defaults to 8s on POSIX / 25s on Windows: first spawn of the exe routinely eats seconds in Defender scanning, and callers treat ``None`` as indeterminate (the upgrade - path used to fall through to a full reinstall on a false timeout).""" + path used to fall through to a full reinstall on a false timeout). + + See #1734. + """ timeout = (25.0 if sys.platform == "win32" else 8.0) if timeout is None else timeout driver_cmd = _cb().resolve_cua_driver_cmd() data = _driver_json(driver_cmd, "check-update", "--json", timeout=timeout, require_ok=False) if driver_cmd else None diff --git a/tools/computer_use/cua_backend_input.py b/tools/computer_use/cua_backend_input.py index a8b5b036f4..ba4107d37b 100644 --- a/tools/computer_use/cua_backend_input.py +++ b/tools/computer_use/cua_backend_input.py @@ -46,7 +46,11 @@ class _InputMixin: """Attach delivery_mode to an input-action args dict. Background is the default and needs no flag. Foreground is only sent when the live action schema accepts it; on an older driver we refuse with ``foreground_unsupported`` instead of silently downgrading to background (which would land input - where the model didn't expect).""" + where the model didn't expect). + + Returns an ActionResult to short-circuit on refusal, or None to proceed. See + NousResearch/hermes-agent#67052 phase B. + """ if not delivery_mode or delivery_mode == "background": return None if delivery_mode != "foreground": @@ -90,6 +94,10 @@ class _InputMixin: return refusal # Tool is chosen by click_count only; `button` goes through click's enum (the driver rejects unknown # buttons). `right_click` / `middle_click` MCP tools are deprecated aliases and never invoked here. + # Choose tool by click_count only — single-vs-double — and pass the button through to `click`'s + # `button` enum (Surface 5 of NousResearch/hermes-agent#47072). cua-driver-rs gained an explicit + # `button: "left"|"right"|"middle"` arg on `click` in trycua/cua#1961 which rejects unknown buttons; + # before that, `middle` was silently mapped to a left-click via name-routing through `right_click`. button_norm = (button or "left").lower() if button_norm not in {"left", "right", "middle"}: return _refuse("click", f"unknown button {button!r} — expected left, right, middle.") diff --git a/tools/computer_use/cua_backend_parse.py b/tools/computer_use/cua_backend_parse.py index f337472666..0800df9043 100644 --- a/tools/computer_use/cua_backend_parse.py +++ b/tools/computer_use/cua_backend_parse.py @@ -40,7 +40,10 @@ def _action_result_from(name: str, ok: bool, message: str, meta: Dict[str, Any], structured: Dict[str, Any], *, requested_delivery: Optional[str] = None) -> ActionResult: """Build an ActionResult, lifting cua-driver's structured verdict. structuredContent is canonical, the flattened ``meta`` copy the fallback. Every structured field is additive: a driver that omits one leaves the attribute - ``None`` so old drivers see unchanged behavior.""" + ``None`` so old drivers see unchanged behavior. + + See the action response shape in cua-driver's mcp-tool-notes and NousResearch/hermes-agent#67052. + """ sc = structured if isinstance(structured, dict) else {} def _raw(key: str) -> Any: @@ -79,7 +82,11 @@ def _is_real_app_window(w: Dict[str, Any]) -> bool: def _parse_elements_from_tree(markdown: str) -> List[UIElement]: """Parse UIElements from get_window_state AX-tree markdown — last-resort fallback for drivers without ``structuredContent.elements``. Bounds are always ``(0, 0, 0, 0)`` (the markdown carries none), fine for - element-index clicks since the driver resolves the frame.""" + element-index clicks since the driver resolves the frame. + + Last-resort fallback for cua-driver builds that don't carry the canonical ``structuredContent.elements`` + array (see ``_parse_elements_from_structured`` — Surface 2 of #47072 prefers that path). + """ return [ # groups 3-6: value / quoted / paren / id= label (first non-None wins) UIElement(index=int(m.group(1)), role=m.group(2), @@ -90,7 +97,11 @@ def _parse_elements_from_tree(markdown: str) -> List[UIElement]: def _parse_elements_from_structured(raw_elements: List[Dict[str, Any]]) -> List[UIElement]: """Read the canonical ``structuredContent.elements`` array: ``element_index``, ``role``, ``label`` and, when AT-SPI / AXFrame returned usable bounds, ``frame`` ``{x, y, w, h}`` — so real pixel bounds survive (the - markdown path loses them). Malformed entries are skipped.""" + markdown path loses them). Malformed entries are skipped. + + Surface 2 of NousResearch/hermes-agent#47072: read the canonical ``structuredContent.elements`` array + cua-driver-rs emits on every ``get_window_state`` response (trycua/cua#1961). + """ elements: List[UIElement] = [] for raw in raw_elements: idx = raw.get("element_index") if isinstance(raw, dict) else None @@ -147,7 +158,13 @@ def _tool_envelope(data: Any, images: List[str], structured: Any, is_error: bool def _extract_tool_result(mcp_result: Any) -> Dict[str, Any]: """Flatten an mcp CallToolResult into ``{data, images, image_mime_types, structuredContent, isError}``. ``data`` is the joined text parts (parsed as JSON when it looks like JSON); ``image_mime_types`` is parallel to - ``images`` with ``""`` where the part carried no mimeType (older drivers — callers then sniff the base64 prefix).""" + ``images`` with ``""`` where the part carried no mimeType (older drivers — callers then sniff the base64 prefix). + + `image_mime_types` is the explicit `mimeType` cua-driver emits on every image part as of trycua/cua#1961 + (Surface 7 of NousResearch/hermes-agent#47072). Each entry corresponds index-for-index with `images`; an + empty string entry signals the part carried no mimeType (older cua-driver build), and the caller should + fall back to base64-prefix sniffing. + """ data: Any = None images: List[str] = [] image_mime_types: List[str] = [] diff --git a/tools/computer_use/cua_backend_session.py b/tools/computer_use/cua_backend_session.py index e88139c9a5..43a64b8158 100644 --- a/tools/computer_use/cua_backend_session.py +++ b/tools/computer_use/cua_backend_session.py @@ -182,6 +182,7 @@ class _CuaDriverSession: "get_window_state", "list_apps", "list_windows"}) # A timed-out MCP session is wedged for later calls, so it is recreated before the next non-lifecycle # call_tool. Class-level default: tests that bypass __init__ see healthy. + # See #74799. _timeout_suspect = False def __init__(self, bridge: _AsyncBridge, embedded_daemon: Optional[Any] = None) -> None: @@ -190,6 +191,10 @@ class _CuaDriverSession: # Per-tool capability-token sets from `tools/list` (read via supports_capability). Raw input schemas are # the source of truth for action properties: 0.9-era drivers advertise delivery_mode in inputSchema # without the ``input.delivery_mode`` token. + # Keys are tool names (e.g. "click", "get_window_state"); values are sets of capability strings + # (e.g. "accessibility.element_tokens", "input.keyboard.type.terminal_safe"). Empty until the + # session starts; consumers should call `supports_capability` rather than reading directly. See + # #47072. self._capabilities: Dict[str, set] = {} self._tool_schemas: Dict[str, Dict[str, Any]] = {} self._capability_version, self._ready_event = "", threading.Event() @@ -197,6 +202,8 @@ class _CuaDriverSession: self._lifecycle_future = None # concurrent.futures.Future self._setup_error: Optional[BaseException] = None # Declared via start_session; revives an ended-session rejection non-re-entrantly. + # Stable driver-side identity declared through start_session. Used to revive a logical ended-session + # rejection without recursive call_tool re-entry or backend-owned state (#71166). self._declared_session_id: Optional[str] = None self._transport_generation, self._transport_reset_callback = 0, None @@ -211,6 +218,8 @@ class _CuaDriverSession: self._shutdown_event = asyncio.Event() # built on the loop's own thread _t0 = _time.monotonic() # Phase marker: the ready-timeout error reports HOW FAR a wedged startup got. + # Phase marker surfaced by the ready-timeout error (issue #57025): when startup wedges, the caller + # reports HOW FAR it got instead of an opaque "never reached ready". self._startup_phase = "binary-check" try: driver_cmd = _cb.resolve_cua_driver_cmd() @@ -247,6 +256,12 @@ class _CuaDriverSession: # rebuilds. Atomic bool write — stop() may hold _lock. self._session, self._started = None, False + # Reset _started so a session that dies for ANY reason (MCP connection drop, driver crash, unexpected + # coro exit) is re-enterable: the next start()/call sees _started False and rebuilds the session instead + # of hanging forever on a dead one via _require_started(). On the normal stop() path this is a harmless + # idempotent no-op (stop() already set it False). A plain bool write is atomic in CPython, so this is + # safe from the bridge-loop thread without taking self._lock (which stop() may hold while awaiting this + # coro's future). See #55048 Bug 1. async def _populate_capabilities(self, session: Any) -> None: """Cache per-tool capability sets, input schemas and capability_version from tools/list. Soft prerequisite: on failure the map stays empty (capability False).""" @@ -286,6 +301,8 @@ class _CuaDriverSession: self._lifecycle_future = asyncio.run_coroutine_threadsafe(self._lifecycle_coro(), loop) if not self._ready_event.wait(timeout=30.0): self._signal_shutdown_locked() + # Surface which startup phase wedged (issue #57025) — "doctor passes but the wrapper times out" + # reports are undiagnosable from a bare "never reached ready". from hermes_constants import display_hermes_home raise RuntimeError( f"cua-driver session never reached ready (timeout 30s; stuck in phase: " @@ -336,8 +353,12 @@ class _CuaDriverSession: return _extract_tool_result(await self._session.call_tool(name, args)) # ── Capability detection ───────────────────────────────────────── + # See #47072. def supports_capability(self, capability: str, tool: Optional[str] = None) -> bool: - """Driver advertises *capability* for *tool* (or ANY tool). False before start.""" + """Driver advertises *capability* for *tool* (or ANY tool). False before start. + + capability token (trycua/cua#1961 capability vocabulary). + """ caps = [self._capabilities.get(tool, set())] if tool is not None else self._capabilities.values() return any(capability in c for c in caps) @@ -462,6 +483,9 @@ class _CuaDriverSession: result = self._bridge.run(self._call_tool_async(name, args), timeout=timeout) except concurrent.futures.TimeoutError as e: # Fail closed: the action may have landed, so never replay it. + # MCP deadline hit (#74799): the session is suspect and must be recreated before the next call. + # Fail closed — the action may have taken effect on the remote screen, so never replay it here; + # surface the uncertainty instead (#74799). self._timeout_suspect = True logger.warning("cua-driver MCP timed out on %s; marking session suspect " "for recreation before the next call", name) diff --git a/tools/computer_use/doctor.py b/tools/computer_use/doctor.py index 0f8e167d91..f30e58f716 100644 --- a/tools/computer_use/doctor.py +++ b/tools/computer_use/doctor.py @@ -268,7 +268,10 @@ def _compose_fallback_report(binary: str, *, reason: str = "", timeout: float = def _apply_display_count_guard(report: Report) -> Report: """Fail an 'ok' screen_capture_capability with ``display_count=0``: macOS ScreenCaptureKit reports 0 on headless / asleep panels — TCC fine, health_report ok, yet every capture is 0x0. Turns a silent failure actionable; applied - at the report seam so the real and the fallback path both get it.""" + at the report seam so the real and the fallback path both get it. + + Composed from PR #52949 (sujeet111) and PR #67259 (webtecnica). + """ checks = report.get("checks") for check in (c for c in (checks if isinstance(checks, list) else ()) if isinstance(c, dict) and c.get("name") == "screen_capture_capability"): data = check.get("data") diff --git a/tools/computer_use/permissions.py b/tools/computer_use/permissions.py index 9059c19c98..60b99363d8 100644 --- a/tools/computer_use/permissions.py +++ b/tools/computer_use/permissions.py @@ -20,7 +20,11 @@ _RUNTIME_PLATFORMS = frozenset({"darwin", "win32", "linux"}) # mirrors the tool _BOOLS = ("accessibility", "screen_recording", "screen_recording_capturable") def _child_env() -> Dict[str, str]: - """cua-driver child env (telemetry policy + provider secrets stripped); ``os.environ`` on import error.""" + """cua-driver child env (telemetry policy + provider secrets stripped); ``os.environ`` on import error. + + cua-driver is a third-party binary — it must never inherit provider API keys (#53503/#55709/#58889 + lineage). Each layer degrades gracefully so permission probes never break on a helper import error. + """ try: from tools.computer_use.cua_backend import sanitized_cua_driver_env return sanitized_cua_driver_env() diff --git a/tools/computer_use/tool.py b/tools/computer_use/tool.py index ec5b622312..1405e548f8 100644 --- a/tools/computer_use/tool.py +++ b/tools/computer_use/tool.py @@ -35,6 +35,7 @@ def set_approval_callback(cb) -> None: # Hard-blocked regardless of approval level (e.g. logout kills the session Hermes runs in). Alt is # canonicalized to option, so the Windows variants are blocked before any backend sees them. +# See #4562. _BLOCKED_KEY_COMBOS = { frozenset({"cmd", "shift", "backspace"}), frozenset({"cmd", "option", "backspace"}), # empty trash / force delete frozenset({"cmd", "ctrl", "q"}), frozenset({"cmd", "shift", "q"}), # lock screen / log out @@ -81,6 +82,9 @@ _backend_permission_modes: Dict[str, str] = {} _AUX_VISION_ROUTE_CACHE: Dict[Tuple[str, str], bool] = {} # process-scoped: (provider, model) → bool # Approval state keyed by session_id so a gateway serving concurrent sessions can't leak one run's # "always approve" into another; callers without a session_id share "". +# Falls back to a shared "" bucket for callers that don't pass a session_id (e.g. the classic single-run +# CLI). Values: _session_auto_approve[sid] -> bool ("always_approve everything") _always_allow[sid] +# -> set of (action, delivery_mode) scope keys See NousResearch/hermes-agent#67052 gap 4. _approval_lock = threading.Lock() _session_auto_approve: Dict[str, bool] = {} # sid -> "always_approve everything" _always_allow: Dict[str, set] = {} # sid -> set of (action, delivery_mode) scope keys @@ -188,7 +192,12 @@ def release_computer_use_session(session_id: str) -> bool: def _shutdown_backend_atexit() -> None: """Stop all cached backends so cua-driver subprocesses don't outlive us. atexit only, no signal handlers: a ``SystemExit`` from a prompt_toolkit key binding corrupts its coroutine state and makes the process unkillable. - Never raises. Drops the global lock before stop(): teardown budgets 5s and must not block spawns.""" + Never raises. Drops the global lock before stop(): teardown budgets 5s and must not block spawns. + + Each session backend holds a long-lived ``cua-driver`` subprocess, so without this a driver can survive + the Hermes process that spawned it (#28152 item 3). #69903 kept the orphan from burning a core by + disabling the cursor overlay; the process itself still lingered. + """ global _backend with _backend_lock: unique = {id(b): (b, _backend_call_locks.get(sid)) for sid, b in _backends.items()} @@ -261,7 +270,12 @@ def handle_computer_use(args: Dict[str, Any], **kwargs) -> Any: def _request_approval(action: str, args: Dict[str, Any], session_id: str = "") -> Optional[str]: """None if approved, else a JSON error string. Scoped by (action, delivery_mode) AND session_id: foreground delivery is a visible focus change, so a background ``approve_session`` must NOT cover it; the blanket - ``always_approve`` does. No CLI approval wired -> default allow (gateway approval runs one layer out).""" + ``always_approve`` does. No CLI approval wired -> default allow (gateway approval runs one layer out). + + ``always_approve`` (the blanket "auto-approve everything" unlock) still covers foreground, since the + user explicitly opted into unattended operation. State is keyed on session_id so concurrent runs don't + leak unlocks into one another. See #67052. + """ scope_key = (action, "foreground" if args.get("delivery_mode") == "foreground" else "background") with _approval_lock: if _session_auto_approve.get(session_id) or scope_key in _always_allow.get(session_id, set()): @@ -545,6 +559,10 @@ def _capture_response(cap: CaptureResult, max_elements: int = _DEFAULT_MAX_ELEME "meta": {"mode": cap.mode, "width": v.width, "height": v.height, "elements": v.total, "png_bytes": cap.png_bytes_len, **_present(screenshot_path=v.screenshot_path, elements_file=v.elements_file, bounds_scale=v.bounds_scale)}, } + # Decide whether to hand the screenshot to the auxiliary.vision pipeline (text-only result) or keep + # the multimodal envelope (main model handles vision natively). Issue #24015: previously the + # multimodal envelope was returned unconditionally, so non-vision main models tripped HTTP 404 / 400 + # at the provider boundary even when auxiliary.vision was explicitly configured to handle this. routed = _route_capture_through_aux_vision( cap, summary, visible_elements=v.visible, truncated_elements=v.truncated, elements_file=v.elements_file, screenshot_path=v.screenshot_path) diff --git a/tools/computer_use/vision_routing.py b/tools/computer_use/vision_routing.py index ace2ecedb4..52c6632290 100644 --- a/tools/computer_use/vision_routing.py +++ b/tools/computer_use/vision_routing.py @@ -78,6 +78,11 @@ def should_route_capture_to_aux_vision(provider: str, model: str, cfg: Optional[ """True iff the screenshot should be pre-analysed via aux vision; False keeps the multimodal envelope. *provider* is the lower-case canonical id, *model* the slug sent to the provider, *cfg* the loaded ``config.yaml`` dict (or None). Steps follow the module docstring's decision order.""" + # auto: an explicitly configured auxiliary.vision backend is the DE-FACTO choice — the user named a + # dedicated vision model, so that's what they want images to go through, even when the main model has + # native vision (maintainer decision, 2026-08-28, reversing #29135's fallback-only posture: config that + # only takes effect when the main model gets worse is a trap, not a setting). Native vision remains the + # default for unconfigured installs, and the fallback when the aux backend is unset. if _explicit_aux_vision_override(cfg): return True user_declared = _lookup_user_declared_supports_vision(provider, model, cfg) diff --git a/tools/credential_files.py b/tools/credential_files.py index 550274d123..d39e5c3250 100644 --- a/tools/credential_files.py +++ b/tools/credential_files.py @@ -77,6 +77,11 @@ def register_credential_file(relative_path: str, container_base: str = "/root/.h if not resolved.is_file(): logger.debug("credential_files: skipping %s (not found)", resolved) return False + # Master credential stores are never mountable, even though they sit inside HERMES_HOME and therefore + # pass the containment check above. Fails CLOSED: if the canonical guard can't be consulted we refuse + # the mount rather than risk bind-mounting auth.json into a sandbox. The import lives at module top (no + # circular-import concern — file_safety is stdlib-only); the sentinel + logger.exception keep guard + # failures debuggable instead of silently swallowed (#67665). if get_read_block_error is None: logger.error("credential_files: refusing %r — agent.file_safety could not be " "imported, so the master-store deny-list cannot be consulted", relative_path) @@ -243,7 +248,11 @@ _CACHE_DIRS: list[tuple[str, str]] = [ ("cache/spillover", "cache/spillover"), # oversized tool results; host side is canonical # Flat top-level desktop staging dirs (tui_gateway attach RPCs; no legacy alias), # mounted so vision/file tools in sandboxes reach uploads and dropped files. + # Mount it so vision can reach uploads inside sandbox containers (#69575). No legacy alias exists, so + # both tuple slots are ``images``. ("images", "images"), + # Mount it so the agent's file tools can read dropped binaries (zip/pdf/...) from inside sandbox + # containers instead of dangling host paths (#76577). ("attachments", "attachments"), ] @@ -260,6 +269,12 @@ def _cache_dir_roots(container_base: str, *, create_missing: bool) -> Iterator[T # would dangle for the container's life: create it now (empty bind mount is free). # get_hermes_dir already picked new-vs-legacy, so this can't shadow a legacy dir. try: + # Create missing staging dirs instead of skipping them: Docker snapshots this mount list at + # container CREATION, so a dir that appears later (first desktop attachment, first clipboard + # image) would dangle for the whole life of a persistent container (#76577). An empty + # bind-mounted dir costs nothing; a missing mount costs the feature. get_hermes_dir() + # already resolved new-vs-legacy layout, so creating its answer cannot shadow a populated + # legacy dir. host_dir.mkdir(parents=True, exist_ok=True) except OSError: continue # unwritable home (tests, RO mounts) — skip as before @@ -304,6 +319,12 @@ def to_agent_visible_cache_path(host_path: str, container_base: str = "/root/.he ``/root/.hermes``; ssh/daytona/vercel_sandbox under ``~/.hermes``; plugin backends declare ``cache_path_base`` (None = host paths stay correct); local/singularity/unknown unchanged (Apptainer auto-binds the host home, so translation would dangle). + + * docker / modal — bind-mounted (docker) or per-file-synced (modal) at ``/root/.hermes`` (the + *container_base* default). * ssh / daytona / vercel_sandbox — file-synced under the remote user's home; + ``~/.hermes`` is shell-expanded by the remote shell, so tool commands resolve it regardless of the + actual remote home. Previously these backends synced the bytes but still rendered the dangling host path + (#76577 gap). """ backend = (os.environ.get("TERMINAL_ENV") or "local").strip().lower() if backend in _HOME_RELATIVE_BACKENDS: diff --git a/tools/cronjob_job_args.py b/tools/cronjob_job_args.py index ee7bdfa217..181685f80e 100644 --- a/tools/cronjob_job_args.py +++ b/tools/cronjob_job_args.py @@ -45,7 +45,15 @@ def _origin_from_env() -> Optional[Dict[str, str]]: def _local_delivery_notice(job: Dict[str, Any], user_deliver: Optional[str]) -> Optional[str]: """Notice when a created job won't deliver anywhere: CLI/TUI sessions have no capturable origin, so deliver='origin' (or omitted) saves output but never delivers it. None when the - user explicitly asked for ``local`` or the job resolves to a real target.""" + user explicitly asked for ``local`` or the job resolves to a real target. + + TUI/CLI sessions cannot be captured as a cron ``origin`` (no ``HERMES_SESSION_PLATFORM``/``CHAT_ID`` is + set for them), so a ``deliver="origin"`` request — or an omitted ``deliver`` that defaults to + origin-or-local — produces a job that runs and saves output to ``last_output`` but is never delivered + back into the session. This is by design (there is no live-delivery channel for local sessions), but + silently dropping the user's "tell me when it runs" intent is the trap reported in 51568. Surface it at + create time so the agent can relay it instead of promising a delivery that never happens. See #51568. + """ if (user_deliver or "").strip().lower() == "local": return None try: @@ -383,7 +391,12 @@ def _format_job(job: Dict[str, Any]) -> Dict[str, Any]: def _gateway_liveness_notice(plural: bool = False) -> dict: """``gateway_running``/``warning`` payload via the shared CLI helper so CLI and tool agree - on "scheduler active". False -> warning (no gateway process), None -> probe failed.""" + on "scheduler active". False -> warning (no gateway process), None -> probe failed. + + Thin adapter over the shared CLI helper ``hermes_cli.cron._builtin_gateway_liveness`` (#87033) so the + CLI and this tool can never disagree about what "scheduler active" means. ``plural`` rewords the warning + for multi-job results (the ``list`` action). + """ try: from hermes_cli.cron import _builtin_gateway_liveness _gw = _builtin_gateway_liveness() diff --git a/tools/cronjob_prompt_scan.py b/tools/cronjob_prompt_scan.py index 6f8522d8ee..90077d9832 100644 --- a/tools/cronjob_prompt_scan.py +++ b/tools/cronjob_prompt_scan.py @@ -13,6 +13,17 @@ logger = logging.getLogger("tools.cronjob_tools") # Strict patterns — user prompt only. A directive-shaped cron prompt has no business # containing `cat ~/.hermes/.env` or `rm -rf /`; there it is a smoking gun, not prose. +# Two threat surfaces, two scanners: 1. `_scan_cron_prompt()` runs against this at create/update time and as +# a runtime defense-in-depth. 2. Assembled prompt that includes loaded skill content (large markdown bodies, +# often security docs, postmortems, runbooks discussing attack patterns in PROSE). Reusing the strict +# patterns here false-positives every time a skill *describes* a command — see #3968 follow-up: the +# `hermes-agent-dev` skill contains a security postmortem mentioning `cat ~/.hermes/.env`, which tripped +# `read_secrets` and silently killed all PR-scout jobs. Skill bodies are user-curated and scanned at install +# time by `skills_guard.py`. The runtime cron scan only needs to catch the patterns whose phrasing does NOT +# survive normal English prose: classic prompt-injection directives ("ignore previous instructions", +# "disregard your rules"), deception directives, and invisible unicode. `_scan_cron_skill_assembled()` runs +# against the assembled prompt with this tighter pattern set. Both scanners share the invisible-unicode +# check and the GitHub Authorization header exemption. _CRON_THREAT_PATTERNS = [ (r'ignore\s+(?:\w+\s+)*(?:previous|all|above|prior)\s+(?:\w+\s+)*instructions', "prompt_injection"), (r'do\s+not\s+tell\s+the\s+user', "deception_hide"), diff --git a/tools/cronjob_tools.py b/tools/cronjob_tools.py index fccad6090f..3350e56799 100644 --- a/tools/cronjob_tools.py +++ b/tools/cronjob_tools.py @@ -16,9 +16,14 @@ logger = logging.getLogger(__name__) # Heartbeat cadence keeping the caller's inactivity watchdog at bay while a manual # `cronjob(action="run")` executes in-process (comfortably below HERMES_AGENT_TIMEOUT). +# Mirrors the 10s cadence of tools/environments/base.py::touch_activity_if_due (delegate_task's heartbeat +# uses 30s) — comfortably below the 1800s default HERMES_AGENT_TIMEOUT. See #76502. _CRON_RUN_HEARTBEAT_INTERVAL = 10.0 # Hard ceiling: with HERMES_CRON_TIMEOUT=0 a truly hung run would otherwise mask the # gateway watchdog forever; past this the heartbeat stops and the watchdog regains authority. +# The child cron run has its own inactivity watchdog (HERMES_CRON_TIMEOUT, default 600s) that bounds a +# wedged job, but with HERMES_CRON_TIMEOUT=0 (explicit "unlimited") a truly hung run_one_job would otherwise +# mask the gateway watchdog forever — pre-#76502 the parent was at least reaped at ~1800s. _CRON_RUN_HEARTBEAT_CEILING = 6 * 3600.0 sys.path.insert(0, str(Path(__file__).parent.parent)) @@ -153,7 +158,13 @@ def _forward_relay_fronted_run(job: Dict[str, Any], extra_prompt: Optional[str] def _manual_run_delivery_note(deliver: str, refreshed: Dict[str, Any]) -> str: """Parenthetical delivery note for a manual run's summary; follows the refreshed record's - ``last_delivery_error`` so the summary never claims success over a failed delivery.""" + ``last_delivery_error`` so the summary never claims success over a failed delivery. + + Follows the refreshed job record (#83993): ``run_one_job`` writes ``last_delivery_error`` via + ``mark_job_run`` when the post-run delivery (telegram/discord/…) failed, and the summary must not claim + success over that record — the calling agent relays this line to the user. Local jobs never deliver; an + empty/missing error keeps the legacy wording byte-for-byte. + """ # Falsy deliver ("", stored JSON null) is normalized to "local" at fire time -> saved # locally. Whitespace-only values fall through so the fire-time "no target" error surfaces. if not deliver or deliver == "local": @@ -208,6 +219,16 @@ def _run_heartbeat(job_name: str): stop = threading.Event() thread = None try: + # run_one_job records last_run_at/last_status via mark_job_run (which also clears the fire claim) + # and returns True iff it processed the job. ``job`` here is the exact claimed snapshot + # (owner-bearing), so the shared body fences every terminal write by that owner. A manual `run` + # executes the job synchronously on the caller's thread, and a cron job is itself a full agent run + # that routinely takes minutes. The calling turn emits no tool activity for that entire window, so + # the gateway inactivity watchdog concludes the agent is hung and kills the parent turn (#76502). + # Fire a heartbeat into the caller's activity tracker (the same signal tool progress uses) while the + # job runs, so the watchdog sees a working tool instead of a silent one — mirrors the delegate_task + # heartbeat pattern. Best-effort: if no activity callback is registered (direct Python callers, + # tests), behavior is unchanged. from tools.environments.base import get_activity_callback # Capture on THIS thread: the callback is thread-local (installed by the tool # executor), so a freshly spawned thread cannot read it. @@ -255,6 +276,9 @@ def _run_claimed_job(job: Dict[str, Any], extra_prompt: Optional[str] = None) -> # In-flight dedupe: the fire claim's TTL is routinely outlived by real jobs, so # register in the scheduler's shared running set (same guard the ticker uses; # also visible to the gateway shutdown drain). + # In-flight dedupe (idea from #53395 by @izumi0uu): the fire claim's TTL (300s) is routinely + # outlived by real jobs, so it alone cannot stop a manual run from double-firing a job the ticker + # (or another manual run) is still executing. if not try_register_running_job(job_id): return {"claimed": True, "success": False, "error": _ALREADY_RUNNING_ERROR} _registered = True @@ -265,6 +289,10 @@ def _run_claimed_job(job: Dict[str, Any], extra_prompt: Optional[str] = None) -> # Inside the gateway process deliver on the loop that owns clients such as # Matrix/aiohttp (a standalone asyncio.run() loop breaks them). runner_ref = getattr(sys.modules.get("gateway.run"), "_gateway_runner_ref", None) + # Manual runs invoked from a gateway agent execute outside the scheduler ticker, but they still + # share the process with the live platform adapters. Calling those clients from run_one_job's + # standalone asyncio.run() loop raises errors like "Timeout context manager should be used inside a + # task" and can break encrypted Matrix delivery (#61495 — salvaged from #63586 by @Fly-onlyone). runner = runner_ref() if callable(runner_ref) else None adapters = getattr(runner, "adapters", None) if runner is not None else None gateway_loop = getattr(runner, "_gateway_loop", None) if runner is not None else None @@ -289,6 +317,9 @@ def _run_claimed_job(job: Dict[str, Any], extra_prompt: Optional[str] = None) -> run_error = refreshed.get("last_error") if last_status == "delivery_failed" and not run_error: run_error = refreshed.get("last_delivery_error") + # That is NOT a success for the caller — the calling agent relays this result — so report it as + # failed and surface the delivery error, which lives in last_delivery_error (last_error is None for + # these runs, and a bare success=False with error=None reads as an unexplained failure). See #83993. ok = last_status == "ok" if execution is not None and execution.get("status") != "completed": ok = False @@ -329,6 +360,11 @@ def _reap_stale_executions(job_name: str) -> None: invocations have no such moment, so a stale claim would block every later manual run. Best-effort self-heal: must not block dispatch.""" try: + # Reap any execution row this job (or any job) left stranded 'claimed'/ 'running' by a dead owner + # process -- e.g. a PRIOR one-shot `hermes cron run` invocation whose dispatched runner died with + # the exiting process before writing a terminal status (issue #86721). Safe and cheap: only + # provably-dead owners (PID gone, or PID reused by a different process per its start time) are + # reaped; a genuinely live owner's row is left untouched. from cron.executions import recover_interrupted_executions _reclaimed = recover_interrupted_executions() if _reclaimed: @@ -400,6 +436,9 @@ def _try_dispatch_background_run( # Routing capture BEFORE the claim: no routable session = no durable consumer for a detached # completion, so don't claim-and-dispatch (direct callers like `hermes cron run` exit right after). session_key = _background_session_key(session_id) + # CLI path: the approval contextvar is only bound during gateway/TUI turns. The CLI drain filters + # completions by the durable agent session id (#64240), so stamp it as the key — an empty key would fail + # closed and the completion could never be claimed. if not session_key: return None @@ -557,6 +596,8 @@ def _action_list(a: Dict[str, Any]) -> str: _result = {"success": True, "count": len(jobs), "jobs": jobs} # Same inert-job class as create; an empty list has nothing inert. if jobs: + # Same silent-inert-job class as create (#87033): an agent inspecting existing jobs in a + # gateway-less environment must learn they are not firing, not just see a clean list. _result.update(_gateway_liveness_notice(plural=True)) return _dumps(_result) @@ -588,6 +629,7 @@ def _action_run(job: Dict[str, Any], a: Dict[str, Any]) -> str: # `prompt` on run is transient per-fire context appended to the stored prompt, never # persisted; same strict scan as stored prompts. extra_prompt = a["prompt"] or None + # See #57331, #57342, #57360. if extra_prompt: scan_error = _scan_cron_prompt(extra_prompt) if scan_error: diff --git a/tools/daemon_pool.py b/tools/daemon_pool.py index 6b79615f8d..5682a8f6ae 100644 --- a/tools/daemon_pool.py +++ b/tools/daemon_pool.py @@ -46,6 +46,8 @@ class DaemonThreadPoolExecutor(ThreadPoolExecutor): num_threads = len(self._threads) if num_threads < self._max_workers: thread_name = "%s_%d" % (self._thread_name_prefix or self, num_threads) + # Carry the active profile into the review thread so MEMORY.md / skill review writes land in the + # right profile (#54937). t = threading.Thread( name=thread_name, target=_worker, daemon=True, args=(weakref.ref(self, weakref_cb), self._work_queue, self._initializer, self._initargs), diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 6fc870113e..4f73e911f7 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -84,6 +84,15 @@ def _open_child_session_db(parent_agent) -> Any: a background child still flushes (transcript silently dropped). It MUST open the same db FILE as the parent's handle (non-launch profiles), else lineage / session_search break; released by the child's close() via _owns_session_db.""" + # Each child gets a DEDICATED SessionDB connection instead of the parent's live object. The parent's + # handle is owned by the parent's lifecycle (cron run_job's finally block, gateway session end, /new) + # and can be closed while a fire-and-forget background child is still flushing on a daemon thread — + # every subsequent flush then hits the closed handle and the child's transcript is silently dropped + # (#81267). It MUST point at the same database FILE as the parent's handle: parents can hold non-default + # per-profile handles (tui_gateway opens SessionDB(db_path=<profile>/ state.db) for non-launch + # profiles), and a bare SessionDB() would write the child's transcript into the launch profile's db, + # breaking parent_session_id lineage and session_search. AsyncSessionDB wrappers (gateway) forward + # .db_path via __getattr__, so this works through them. parent_session_db = getattr(parent_agent, "_session_db", None) if parent_session_db is None: return None @@ -188,6 +197,8 @@ def _build_child_agent( child._print_fn = getattr(parent_agent, "_print_fn", None) if child_session_db is not None: child._owns_session_db = True # released by the child's close(), never by the parent + # Ownership transfer for the dedicated handle: the child's close() must release it (nothing else holds a + # reference), and no parent teardown can close it out from under a background child (#81267). child_session_ref["session_id"] = getattr(child, "session_id", "") or "" child._progress_identity_ref = child_session_ref child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role @@ -235,6 +246,8 @@ def _run_single_child( "max_iterations" only for genuine budget exhaustion (completed=False with no failure fields), never for errors. truncated == (exit_reason == "max_iterations"). + + * ``"completed"`` — normal finish. See #97655. """ child_progress_cb = getattr(child, "tool_progress_callback", None) child_pool, leased_cred_id = _lease_child_credential(child) @@ -386,6 +399,8 @@ def delegate_task( try: creds = _resolve_delegation_credentials(credentials_cfg if credentials_cfg else cfg, parent_agent) except ValueError as exc: + # Explicit-pin preflight failures (e.g. pinned delegation.command missing from PATH) refuse the + # spawn loudly (#80450). return tool_error(str(exc)) max_children = _get_max_concurrent_children() task_list, err = _normalize_task_list(goal, context, tasks, output_schema, top_role, max_children) diff --git a/tools/delegate_tool_child_run.py b/tools/delegate_tool_child_run.py index 7434a394dc..2b1b473339 100644 --- a/tools/delegate_tool_child_run.py +++ b/tools/delegate_tool_child_run.py @@ -341,6 +341,12 @@ def _defer_close_after_timeout(child: Any, child_future: Any) -> None: resources until process exit. """ child_future.add_done_callback(lambda _done: _close_child(child, "Failed to close timed-out child after worker exit")) + # Bounded drain (#94248 native half): the deferred close above only fires once the abandoned worker + # unwinds, but that worker is typically parked inside an in-flight OpenSSL read (Codex / httpx). Never + # hard-close that transport from this thread — releasing FDs under a live SSL read is the #29507/#70773 + # native-corruption family. Instead shutdown() the child's pooled sockets, which is FD-safe from any + # thread and settles the blocked read with EOF/EPIPE so the worker can unwind and trigger the deferred + # close. _drain = getattr(child, "_drain_transports_after_abandonment", None) if not callable(_drain): return diff --git a/tools/delegate_tool_config.py b/tools/delegate_tool_config.py index de0b2c99d6..b152acff5e 100644 --- a/tools/delegate_tool_config.py +++ b/tools/delegate_tool_config.py @@ -166,7 +166,10 @@ def _normalized_runtime_url(value: Any) -> str: def _inherit_parent_capabilities(parent_agent, override_provider, override_base_url) -> Optional[dict]: """Parent's endpoint-trust capability map for a child, or None. ``agent.capabilities`` is a trust decision scoped to one provider+endpoint: inherited ONLY when the child runs the parent's exact route; any provider or base_url - override stays DEFAULT-DENY (matches the /model switch posture).""" + override stays DEFAULT-DENY (matches the /model switch posture). + + See #94036, #97292. + """ if override_provider or override_base_url: return None parent_caps = getattr(parent_agent, "capabilities", None) @@ -204,7 +207,14 @@ def _resolve_child_credential_pool( its fixed credential). Custom endpoints all collapse to ``provider="custom"``, so they are matched by endpoint identity (the ``custom:<name>`` pool key) — sharing the parent's pool across different custom endpoints would overwrite the child's delegated base_url on lease; an unregistered custom endpoint (no custom_providers entry) - keeps the child's fixed credential rather than inherit the parent's.""" + keeps the child's fixed credential rather than inherit the parent's. + + Custom endpoints are a special case: every direct ``delegation.base_url`` runtime collapses to + ``provider="custom"``, so bare provider equality would treat two *different* custom endpoints as + interchangeable and let the child inherit the parent's pool. We therefore resolve custom runtimes by + endpoint identity (the ``custom:<name>`` pool key derived from the base_url) and only share the parent's + pool when both resolve to the *same* custom endpoint. See #7833. + """ parent_pool = getattr(parent_agent, "_credential_pool", None) if not effective_provider: return parent_pool @@ -274,6 +284,8 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict: """``delegation.base_url`` branch: provider/api_mode from URL heuristics.""" # Shared URL-based api_mode detector so Anthropic-compatible direct endpoints (/anthropic suffix: Azure AI # Foundry, MiniMax, Zhipu, LiteLLM) get the Messages transport instead of 404ing on chat_completions. + # Without this, subagents would default to chat_completions and hit 404s on endpoints that only speak + # the Anthropic Messages protocol. Fixes #10213. from hermes_cli.runtime_provider import _detect_api_mode_for_url base_lower = v["base_url"].lower() host = base_url_hostname(v["base_url"]) @@ -329,6 +341,8 @@ def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict: f"Delegation provider '{configured_provider}' resolved but has no API key. " f"Set the appropriate environment variable or run 'hermes auth'." ) + # A pinned ACP transport command must exist — refuse the spawn loudly rather than letting the child + # silently fall back to another transport (#80450). pinned_command = runtime.get("command") _require_pinned_command( pinned_command, f"Delegation provider '{configured_provider}' is pinned to the " @@ -410,6 +424,12 @@ def _resolve_child_runtime( # api_mode: each provider has its own wire, so a different provider re-derives (None) instead of inheriting (404s # otherwise). Nous Portal is dual-wire within one provider (anthropic/* → Messages, else chat_completions), so # same-provider inheritance would pin the child on the wrong wire — re-derive. + # Bug #20558 / PR #20563: api_mode must NOT be inherited when the child uses a different provider than + # the parent — each provider has its own API surface (e.g. MiniMax uses anthropic_messages, DeepSeek + # uses chat_completions). Inheriting the parent's mode causes 404 errors when the child routes to the + # wrong endpoint. Derive the mode from the target provider when it differs. Same-provider inheritance + # would pin a child Hermes/Qwen subagent onto the parent's Claude Messages wire (or the reverse). + # agent_init honors an explicit api_mode above its nous branch, so re-derive here before construction. _parent_provider = getattr(parent_agent, "provider", None) or "" if override_api_mode is not None: effective_api_mode = override_api_mode @@ -432,8 +452,14 @@ def _resolve_child_runtime( ) # A pinned provider must use direct API calls; inheriting the parent's ACP # transport would bypass the override credentials entirely. + # Inheriting acp_command unconditionally causes run_agent.py to initialize CopilotACPClient, bypassing + # override credentials entirely (issue #16816). if override_provider and not override_acp_command: effective_acp_command, effective_acp_args = None, [] + # Defensive: validate trusted delegation.command exists on PATH before honoring it. An explicitly pinned + # transport that cannot run must fail the spawn loudly (#80450) — silently falling back to the default + # transport would run the child somewhere the user explicitly routed it away from. Normally unreachable + # via delegate_task, which pre-validates the command in _resolve_delegation_credentials. if override_acp_command: # Forced ACP transport requires provider copilot-acp for run_agent to init the client. effective_provider, effective_api_mode = "copilot-acp", "chat_completions" diff --git a/tools/delegate_tool_dispatch.py b/tools/delegate_tool_dispatch.py index 687acf8226..e40d9c163a 100644 --- a/tools/delegate_tool_dispatch.py +++ b/tools/delegate_tool_dispatch.py @@ -187,6 +187,10 @@ def _resolve_async_wake_sid(origin_wake_sid: str) -> Optional[str]: id. """ try: + # Finite sessions cannot route a detached subagent result back to the agent after their turn/process + # ends. This includes stateless HTTP requests (#10760) and one-shot Kanban workers (#63169). Fall + # back to SYNCHRONOUS execution so the result returns in this same turn instead of handing out a + # handle with no durable consumer. Mirrors the pool-at-capacity inline fallback below. from gateway.session_context import async_delivery_supported if async_delivery_supported(): return "" @@ -229,6 +233,11 @@ def _batch_progress_token(child_agents: List[Any]) -> tuple: so a child streaming a long response counts as alive; a fully frozen token past the threshold means the batch is wedged. ``in_tool`` is True while ANY child is inside a tool so slow tools get the higher ceiling (mirrors the sync heartbeat).""" + # Progress token for the async registry's stale monitor: the combined (api_call_count, current_tool, + # last_activity_ts) of every child. last_activity_ts is ticked by _touch_activity on every streamed + # chunk ("receiving stream response"), every tool transition, and every API-call start/completion — so a + # child streaming a long response is alive even though api_call_count only advances when the call + # completes (same liveness signal as the compaction inactivity budget, PR #71508). parts = [] in_tool = False for c in child_agents: diff --git a/tools/delegate_tool_progress.py b/tools/delegate_tool_progress.py index cf4433626c..f76f572522 100644 --- a/tools/delegate_tool_progress.py +++ b/tools/delegate_tool_progress.py @@ -169,6 +169,7 @@ def _build_child_system_prompt( # install-tree-fallback leak doesn't apply. Best-effort. _ctx_files = "" with _quiet("subagent: workspace context-files load failed", exc_info=True): + # See #64590. from agent.prompt_builder import build_context_files_prompt _ctx_files = build_context_files_prompt(cwd=str(workspace_path), skip_soul=True) if _ctx_files.strip(): diff --git a/tools/delegate_tool_tasks.py b/tools/delegate_tool_tasks.py index 4c393a3219..282ceec2cd 100644 --- a/tools/delegate_tool_tasks.py +++ b/tools/delegate_tool_tasks.py @@ -11,6 +11,7 @@ from typing import Any, Dict, List, Optional # `{file path}`, `<FEATURE-NAME>`), the shape LLM templates leave behind. Bare single-word brackets must never be # rejected: legitimate goals are full of generics (`Vec<T>`), HTML tags (`<div>`), dict snippets (`{"key": 1}`), glob # braces (`{a,b}`) and f-string style (`{i}`). +# See #81141. _PLACEHOLDER_GOAL_RE = re.compile(r"^(todo|task\s*\d+)$", re.IGNORECASE) _TEMPLATE_MARKER_RE = re.compile( r"<[A-Za-z][A-Za-z0-9]*(?:[ _-][A-Za-z0-9]+)+>|\{[A-Za-z][A-Za-z0-9]*(?:[ _-][A-Za-z0-9]+)+\}" @@ -37,7 +38,10 @@ def _validate_batch_tasks(task_list: List[Dict[str, Any]]) -> Optional[str]: array is the canonical single-task shape (legacy top-level `goal` is wrapped into one). Duplicate goals are deliberately NOT rejected — identical-goal fan-outs (best-of-N / ensemble sampling) are legitimate and blocking them broke real workflows. The too-short check applies only to multi-task fan-outs (terse goals there are - usually unexpanded templates); a SINGLE task legitimately uses short goals ("Fix the tests").""" + usually unexpanded templates); a SINGLE task legitimately uses short goals ("Fix the tests"). + + See #81141. + """ for i, task in enumerate(task_list): goal = str(task.get("goal", "")).strip() if _PLACEHOLDER_GOAL_RE.match(" ".join(goal.lower().split())): diff --git a/tools/drive_preview_tool.py b/tools/drive_preview_tool.py index a05af03343..66cef662fe 100644 --- a/tools/drive_preview_tool.py +++ b/tools/drive_preview_tool.py @@ -62,6 +62,7 @@ ACT_PREVIEW_SCHEMA = { # Response-shape teaching kept only where skipping it wastes calls (delta # semantics, rebound refs, strobe's burst): a model that doesn't know them # re-reads pages or loops strobe. + # See #95681. "description": ( "Use the web page open in the desktop preview pane (the one " "`desktop_preview` opens): log in, fill forms, click through flows. ALWAYS " diff --git a/tools/env_probe.py b/tools/env_probe.py index 49405141a6..75f7205833 100644 --- a/tools/env_probe.py +++ b/tools/env_probe.py @@ -21,6 +21,8 @@ logger = logging.getLogger(__name__) # signals completion. Callers block at most ``_PROBE_WAIT_TIMEOUT`` s then fail open # with "" — a stuck probe (e.g. a Windows pipe wedged by an orphaned pip descendant) # can degrade only the probe line, never system-prompt construction. +# Module-level cache. The probe result is deterministic for the lifetime of the process — Python install +# state doesn't change mid-session in any way that would matter for the system prompt. See #67964. _CACHE_LOCK = threading.Lock() _CACHED_LINE: Optional[str] = None # None = not probed yet; "" = probed, nothing to say. _PROBE_DONE = threading.Event() @@ -169,7 +171,11 @@ def get_environment_probe_line(*, force_refresh: bool = False) -> str: """Return the cached probe line (building it on first call); "" when the environment is clean, so the prompt assembler drops the section. Waits at most ``_PROBE_WAIT_TIMEOUT`` on the single worker, then fails open with "". - ``force_refresh`` is for tests.""" + ``force_refresh`` is for tests. + + A wedged probe subprocess (#67964) therefore can never block system-prompt construction — at worst the + toolchain line is absent from prompts built while the probe is stuck. + """ global _WAIT_ALREADY_TIMED_OUT if force_refresh: _reset_cache_for_tests() @@ -177,6 +183,7 @@ def get_environment_probe_line(*, force_refresh: bool = False) -> str: # routed profile's backend lives in the per-turn terminal scope, which the worker # thread does not inherit. Remote backends answer "" without consulting the cache # — the cached line describes the HOST toolchain. + # See #68559. backend = _resolve_terminal_backend() if backend in _REMOTE_BACKENDS or _plugin_backend_is_remote(backend): return "" diff --git a/tools/environments/base.py b/tools/environments/base.py index ad0577afb9..3dc508409f 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -44,6 +44,8 @@ _DEBUG_INTERRUPT = bool(os.getenv("HERMES_DEBUG_INTERRUPT")) # Extra seconds the ``run_bounded_sync`` backstop waits past the inner ``_wait_for_process`` # deadline: the inner loop returns partial output + 124; the outer bound only fires when that # loop never returns. Keep small so a healthy timeout still comes from the inner path. +# The inner poll loop is what returns partial output + returncode 124; this outer bound only exists for when +# that loop itself never returns (family A of #94285: a blocked wait that silently disables asyncio timers). _EXECUTE_WAIT_BOUND_GRACE_S = 2.0 if _DEBUG_INTERRUPT: @@ -79,7 +81,12 @@ def set_activity_callback(cb: Callable[[str], None] | None) -> None: def get_activity_callback() -> Callable[[str], None] | None: - """Thread-local activity callback; capture it before handing work to another thread.""" + """Thread-local activity callback; capture it before handing work to another thread. + + Public accessor for callers outside this module that need to capture the calling thread's callback + before handing work to another thread (the callback is thread-local, so a freshly spawned thread cannot + read it back) — e.g. the manual cron-run heartbeat (#76502). + """ return getattr(_activity_callback_local, "callback", None) @@ -300,7 +307,12 @@ class BaseEnvironment(ABC): may move the wait onto a ``run_bounded_sync`` worker while ``/stop`` still interrupts the original tid, so both bits are honored. ``KeyboardInterrupt``/``SystemExit`` mid-poll kills the process first — the local backend spawns into its own process group, so an - unkilled child would be orphaned.""" + unkilled child would be orphaned. + + The default (False) preserves full-fidelity capture for internal consumers — file-operation ``cat`` + reads feeding the patch engine, code-execution RPC reads, log reads — where truncation would corrupt + data. See #64435. + """ output = _new_output_collector(proc, bounded_capture) drain_thread = _start_drain_thread(proc, output) _now = time.monotonic() @@ -416,7 +428,12 @@ class BaseEnvironment(ABC): tool may set it — internal full-fidelity consumers (file-op ``cat`` reads feeding the patch engine, RPC reads, log reads) MUST leave it False or data is corrupted. The wait is bounded by ``agent.deadline.run_bounded_sync`` so a wedged poll loop cannot hang past - ``timeout`` and silently disable every asyncio timer in the process.""" + ``timeout`` and silently disable every asyncio timer in the process. + + ``bounded_capture=True`` caps stdout/stderr retention at ``tool_output.max_bytes`` WHILE the stream + is drained (head/tail window) instead of holding the full output in memory (#64435). + See #94285. + """ self._before_execute() exec_command, sudo_stdin = self._prepare_command(command) @@ -463,6 +480,7 @@ class BaseEnvironment(ABC): # pipe/poll hang), every asyncio timer is silently disabled. ``run_bounded_sync`` drives # expiry from a daemon worker + ``Event.wait`` so a blocked loop cannot disable it; the # grace lets the inner loop return the partial-output 124 path. + # See #94285. from agent.deadline import run_bounded_sync try: diff --git a/tools/environments/base_output.py b/tools/environments/base_output.py index 460568cd25..48350450a1 100644 --- a/tools/environments/base_output.py +++ b/tools/environments/base_output.py @@ -356,6 +356,20 @@ def _drain_stdout(proc: ProcessHandle, output: _BoundedOutputCollector) -> None: stream = proc.stdout if stream is None: return + # Non-blocking drain via select(). The old pattern — ``for line in proc.stdout`` — blocks on + # ``readline()`` until the pipe reaches EOF. When the user's command backgrounds a process (``cmd &``, + # ``setsid cmd & disown``, etc.), that backgrounded grandchild inherits the write-end of our stdout pipe + # via ``fork()``. Even after ``bash`` itself exits, the pipe stays open because the grandchild still + # holds it — so the drain thread never returns and the tool hangs for the full lifetime of the + # grandchild (issue #8340: users reported indefinite hangs when restarting uvicorn with ``setsid ... & + # disown``). The fix: select() with a short poll interval, and stop draining shortly after ``bash`` + # exits even if the pipe hasn't EOF'd yet. Any output the grandchild writes after that point goes to an + # orphaned pipe (harmless — the kernel reaps it when our end closes). Decoding: we ``os.read()`` raw + # bytes in fixed-size chunks (4096) so a single multibyte UTF-8 character can split across reads. An + # incremental decoder buffers partial sequences across chunks, and ``errors="replace"`` mirrors the + # baseline ``TextIOWrapper`` (which was constructed with ``encoding="utf-8", errors="replace"`` on + # ``Popen``) so binary or mis-encoded output is preserved with U+FFFD substitution rather than + # clobbering the whole buffer. decoder = codecs.getincrementaldecoder("utf-8")(errors="replace") try: fd = stream.fileno() diff --git a/tools/environments/base_session_env.py b/tools/environments/base_session_env.py index 12a8f85b47..927b7de966 100644 --- a/tools/environments/base_session_env.py +++ b/tools/environments/base_session_env.py @@ -15,6 +15,17 @@ from typing import Iterable # would make every LATER session source a foreign identity. Every bridged name starts with # one of these prefixes (or is HERMES_UI_SESSION_ID); unit tests use this regex as the # Python-side contract for the exclusion set. +# Per-session variables that the gateway bridges freshly onto every command's process environment (via +# tools/environments/local._inject_session_context_env, reading gateway.session_context._VAR_MAP). They must +# NEVER be persisted into the shared bash session snapshot: a single long-lived backend serves many +# concurrent sessions (the messaging gateway, TUI, desktop/web dashboard all collapse the terminal to one +# "default" environment), so ``export -p`` dumping the FIRST session's HERMES_SESSION_ID into the snapshot +# makes every LATER session ``source`` that stale value and see a FOREIGN session's identity — overriding +# the correct per-command Popen env (issue: cross-session HERMES_SESSION_ID leak via the shared snapshot). +# Stripping them from the snapshot is safe because they are re-injected on every command; a snapshot should +# only carry the user's own shell state (PATH, functions, exports they set), not Hermes' per-turn session +# identity. Used by unit tests as the Python-side contract for the exclusion set; the dump path unsets by +# name/prefix instead of grepping declare lines (see below / issue #71296). _SNAPSHOT_EXCLUDED_ENV_REGEX = ( "^declare -x (HERMES_SESSION_|HERMES_UI_SESSION_ID|HERMES_CRON_AUTO_DELIVER_|" "HERMES_CRON_SESSION|HERMES_BROWSER_CONTROL_)") @@ -43,7 +54,12 @@ def _export_dump_excluding_session_vars(tmp_path: str, excluded_names: Iterable[ survive into the snapshot and execute on the next ``source``. ``|| true`` keeps the success contract. The dump is a brace group with the redirection on the group: *tmp_path* is usually a shell-variable expansion, and a redirect on a pipeline segment would expand it inside that - segment's subshell, inconsistently with the parent that expands the follow-up ``mv``.""" + segment's subshell, inconsistently with the parent that expands the follow-up ``mv``. + + ``curl … | bash #`` smuggled into a Matrix room/display name via ``HERMES_SESSION_CHAT_NAME``) land in + the snapshot and execute on the next ``source`` (issue #71296). Unsetting first means ``export -p`` + never emits those vars — including any continuation lines. + """ # ${!PREFIX*} is bash 3.2+ name-prefix expansion; empty matches are ignored # under 2>/dev/null. Caller names are quoted so malformed config can never # become shell syntax (valid names stay unquoted by shlex.quote()). diff --git a/tools/environments/docker.py b/tools/environments/docker.py index ce2bc690d3..f07ae8a153 100644 --- a/tools/environments/docker.py +++ b/tools/environments/docker.py @@ -252,6 +252,12 @@ _DEFAULT_PIDS_LIMIT = "256" # applied only when the pids cgroup controller is a # Docker's 64 MB /dev/shm default crashes Chromium/Playwright tabs and PyTorch # DataLoader workers. tmpfs is lazily allocated so a 1g ceiling costs nothing # until used (and still counts against --memory). Empty/"0" config omits the flag. +# Docker's built-in default is a tiny 64 MB, which silently breaks shared-memory-hungry workloads inside the +# sandbox: Chromium / Playwright renderers crash tabs, and PyTorch DataLoader workers die with "bus error" / +# "insufficient shared memory" once they exceed it. tmpfs is lazily allocated, so a 1g ceiling costs nothing +# until actually used (and usage still counts against the container's --memory cgroup limit). Configurable +# via ``terminal.docker_shm_size`` in config.yaml; an empty value (or "0") omits the flag and falls back to +# Docker's 64 MB default. Ported from nanocoai/nanoclaw#2748. _DEFAULT_SHM_SIZE = "1g" @@ -516,6 +522,11 @@ class DockerEnvironment(BaseEnvironment): # Resolved once so it works when /usr/local/bin is not in PATH (macOS services). self._docker_exe = find_docker() or "docker" + # s6-overlay images (e.g. hermes-agent:latest) already use /init as PID 1 and exec + # /run/s6/basedir/bin/init during startup. For those images we must (a) skip Docker's --init (two + # competing PID-1 inits) and (b) mount /run with exec instead of noexec, or s6 stage0 dies with exit + # 126 "Permission denied". Detected once here; defaults are kept on any inspection failure. See + # issue #34628. image_uses_s6_init = _image_uses_init_entrypoint(self._docker_exe, image) if image_uses_s6_init: logger.info( @@ -711,6 +722,9 @@ class DockerEnvironment(BaseEnvironment): lifetime). s6-overlay images already provide PID 1, so ``--init`` is skipped for them.""" label_args = [arg for k, v in self._labels.items() for arg in ("--label", f"{k}={v}")] return [ + # tini/catatonit as PID 1 reaps zombie children — but s6-overlay images already provide their + # own /init PID 1, so adding --init there creates two competing inits and breaks startup + # (#34628). self._docker_exe, "run", "-d", *([] if self._image_uses_s6_init else ["--init"]), "--name", name, @@ -744,12 +758,22 @@ class DockerEnvironment(BaseEnvironment): # --- Env forwarding --- def _docker_client_env(self, values: dict[str, str]) -> dict[str, str] | None: """Env for the docker-client subprocess carrying forwarded values (pairs with name-only - ``-e KEY`` flags to keep secrets out of cmdline); ``None`` = inherit when empty.""" + ``-e KEY`` flags to keep secrets out of cmdline); ``None`` = inherit when empty. + + Name-only ``-e KEY`` flags make the docker CLI read each value from its own process environment, + keeping secrets out of the client's world-readable ``/proc/<pid>/cmdline`` (issue #96268). Values + live in ``/proc/<pid>/environ`` instead, which is owner/root-only. Returns ``None`` (inherit as + before) when there is nothing to add. + """ return {**os.environ, **values} if values else None def _build_init_env_args(self) -> list[str]: """Name-only ``-e`` args for init_session so ``export -p`` captures docker_env - plus the current profile's forwarded values (values travel via the client env).""" + plus the current profile's forwarded values (values travel via the client env). + + The VALUES intentionally do not appear in the argv — they are passed via the docker client + subprocess env (see _docker_client_env and issue #96268); the flags here are name-only ``-e KEY``. + """ passthrough_env, unset_names = self._resolve_passthrough_env() exec_env = {**self._env, **passthrough_env} for name in unset_names: @@ -790,7 +814,10 @@ class DockerEnvironment(BaseEnvironment): def _build_runtime_env_args_with_unsets(self) -> tuple[list[str], tuple[str, ...], dict[str, str]]: """Runtime name-only forwarding args, names absent from scope, and the values - to inject into the docker client subprocess env.""" + to inject into the docker client subprocess env. + + See #96268. + """ passthrough_env, unset_names = self._resolve_passthrough_env() return _name_only_env_args(passthrough_env), tuple(sorted(unset_names)), dict(passthrough_env) @@ -805,6 +832,10 @@ class DockerEnvironment(BaseEnvironment): if stdin_data is not None: cmd.append("-i") + # Init seeds the snapshot. Profile-scoped passthrough values are also injected on every later + # command because this container can be shared by multiple routed profiles in one gateway process. + # Env flags are name-only; values travel via the client subprocess env so they never hit + # world-readable /proc/*/cmdline (#96268). unset_names: tuple[str, ...] = () env_values: dict[str, str] = {} if login: @@ -971,7 +1002,13 @@ class DockerEnvironment(BaseEnvironment): delay per session; reclamation is ``reap_orphan_containers()`` at next startup. ``persist_across_processes=False`` or ``force_remove=True`` (explicit-teardown hook, unused so far) does ``docker stop`` + ``docker rm -f`` on a daemon thread that the - atexit hook joins via ``wait_for_cleanup`` so the work completes before exit.""" + atexit hook joins via ``wait_for_cleanup`` so the work completes before exit. + + Cleanup runs on a daemon thread with bounded ``subprocess.run`` calls (not the racy ``Popen(... &)`` + pattern from before PR #33645). The atexit hook in ``tools/terminal_tool.py`` waits up to 15s for + the thread to finish before the interpreter exits, so ``docker stop`` / ``docker rm`` actually + completes when we do trigger it. + """ container_id = self._container_id if not container_id: # Bind-mount dirs are still dropped in non-persistent mode. diff --git a/tools/environments/local.py b/tools/environments/local.py index 0bd4b9fe87..73d74bef27 100644 --- a/tools/environments/local.py +++ b/tools/environments/local.py @@ -179,7 +179,14 @@ def _resolve_safe_cwd(cwd: str) -> str: """``cwd`` if enterable, else the nearest usable ancestor, else ``tempfile.gettempdir()``. MSYS paths are normalized first on Windows so a valid ``pwd -P`` result is not rejected. Lets ``_run_bash`` recover from a deleted or - inaccessible cwd instead of ``Popen`` raising and wedging every later call.""" + inaccessible cwd instead of ``Popen`` raising and wedging every later call. + + Used by ``_run_bash`` to recover when the configured cwd is gone — most commonly because a previous tool + call deleted its own working directory (issue #17558) — or inaccessible to this user, e.g. ``/root`` + leaking from a root-launched CLI session into a non-root gateway's cron jobs (issue #65583). Without + this guard, ``subprocess.Popen(..., cwd=...)`` raises ``FileNotFoundError``/``PermissionError`` before + bash starts, wedging every subsequent terminal call until the gateway restarts. + """ cwd = _msys_to_windows_path(cwd) if cwd and _cwd_usable(cwd): return cwd @@ -288,6 +295,9 @@ def _scrubbed_env(parts, plugin_strip: frozenset, fix_path) -> dict: for items, unwrap_force in parts: _filter_secret_env(items, out, unwrap_force=unwrap_force, plugin_strip=plugin_strip) path_key = _path_env_key(out) + # Keep bare ``hermes`` invocations available to child jobs even when the gateway was launched by a + # service manager or cron without the console script's directory on PATH. The terminal environment + # already applies this invariant; Cron scripts use this sanitizer directly (#92998). if path_key is not None: out[path_key] = _prepend_hermes_bin_dir(fix_path(out.get(path_key, ""))) return _finalize_child_env(out) @@ -433,6 +443,7 @@ def _prepend_git_bash_dirs(existing_path: str) -> str: # POSIX-sh-family shells that understand spawn_local's ``[shell, "-lic", "set +m; …"]`` # invocation; fish, csh/tcsh, nushell, elvish, xonsh would error, so _find_shell # falls back to bash for them. +# (#42203) _SPAWN_COMPATIBLE_SHELLS = frozenset({"bash", "zsh", "sh", "dash", "ksh", "mksh"}) @@ -517,7 +528,16 @@ def _append_missing_sane_path_entries(existing_path: str) -> str: def _apply_windows_msys_bash_env_defaults(env: dict) -> None: """Disable MSYS argument path conversion (``/FO`` -> ``C:/.../git/FO`` breaks tasklist/schtasks/wmic/``cmd /c``). Git for Windows honors ``MSYS_NO_PATHCONV``; - MSYS2/Cygwin bash honor ``MSYS2_ARG_CONV_EXCL`` — set both; users can override.""" + MSYS2/Cygwin bash honor ``MSYS2_ARG_CONV_EXCL`` — set both; users can override. + + Git Bash rewrites arguments that look like Unix paths (``/FO``, ``/TN``, ``/Create``) into + ``C:/.../git/FO``-style paths, which breaks native Windows commands such as ``tasklist``, ``schtasks``, + and ``wmic``. Hermes runs terminal commands through bash on Windows, so set the standard MSYS opt-out by + default. Refs #56700. + MSYS2-proper and Cygwin bash (which ``_find_bash`` can still return via the final ``shutil.which`` + fallback) ignore it and honor ``MSYS2_ARG_CONV_EXCL`` instead, so set both. ``*`` disables all argv + conversion — the semantic equivalent of ``MSYS_NO_PATHCONV=1``. Also fixes ``cmd /c`` mangling (#56147). + """ if _IS_WINDOWS: env.setdefault("MSYS_NO_PATHCONV", "1") env.setdefault("MSYS2_ARG_CONV_EXCL", "*") @@ -745,6 +765,11 @@ class LocalEnvironment(BaseEnvironment): (e.g. a command ``rm -rf``'d its own cwd) — otherwise Popen raises before bash starts and every subsequent call fails. A benign MSYS→Windows normalization is not warned about.""" + # Recover when the cwd has been deleted out from under us — usually by a previous tool call that ran + # ``rm -rf`` on its own working dir (issue #17558). On Windows, ``_resolve_safe_cwd`` also + # normalises Git Bash-style POSIX paths (``/c/Users/...``) to native form so a perfectly valid ``pwd + # -P`` result from bash isn't mistakenly treated as "missing" and spammed as a warning on every + # command. safe_cwd = _resolve_safe_cwd(self.cwd) if safe_cwd == self.cwd: return @@ -805,6 +830,7 @@ class LocalEnvironment(BaseEnvironment): def cleanup(self): """Clean up temp files, including orphaned atomic-write snapshots (``snap.tmp.<bashpid>``) a failed/interrupted mv could leave behind.""" + # See #38249. import glob try: stale = glob.glob(f"{self._snapshot_path}.tmp.*") diff --git a/tools/environments/local_env_policy.py b/tools/environments/local_env_policy.py index d6ead6fbda..d11d2f45d4 100644 --- a/tools/environments/local_env_policy.py +++ b/tools/environments/local_env_policy.py @@ -70,7 +70,12 @@ def _build_provider_env_blocklist() -> frozenset: # every scrub surface, so an import-time discard would leak BUZZ_PRIVATE_KEY into # non-terminal children; the Buzz carve-out is terminal-only and context-gated # (``_is_terminal_first_party_env``). + # It is set and owned by the user's Claude Code install (subscription OAuth), not a Hermes-managed + # inference credential — Claude subscription auth is not a working Hermes provider path. It arrives via + # the registry loop above (anthropic api_key_env_vars), so remove it explicitly. See #55878. blocked.discard("CLAUDE_CODE_OAUTH_TOKEN") + # BUZZ_* is deliberately NOT discarded here, even for Buzz-managed agents (BUZZ_MANAGED_AGENT set by the + # buzz-acp harness). See #76243, #78026, #78065, #78511. return frozenset(blocked) @@ -84,6 +89,34 @@ _HERMES_PROVIDER_ENV_BLOCKLIST = _build_provider_env_blocklist() # runs a Buzz gateway must not get the signing key. Values are used directly, never # scope-resolved (UnscopedSecretError under multiplex); the snapshot treats them as # profile-scoped. Prefix-based so future BUZZ_* names need no code change. +# First-party platform credentials the agent's own platform adapters need in terminal children (e.g. the +# ``BUZZ_*`` vars for the Buzz messaging platform, which drive the platform-mandated ``buzz`` CLI: +# BUZZ_PRIVATE_KEY, BUZZ_AUTH_TAG, BUZZ_RELAY_URL, and the other BUZZ_* names). These are the agent's OWN +# credentials — a Buzz community agent is expected to operate the ``buzz`` CLI — so they are carved out of +# the terminal scrub. CONTEXT-GATED: the carve-out applies ONLY when this process/session is actually +# operating as a Buzz agent — either the process is a Buzz-ACP managed agent (``BUZZ_MANAGED_AGENT`` is set, +# only by Buzz Desktop's buzz-acp harness; see #76243 / #78511) or the current session's platform is +# ``buzz`` (the gateway's ``HERMES_SESSION_PLATFORM`` ContextVar; concurrency safe under a multi-session +# host). A Telegram/CLI/cron session on a host that also runs a Buzz gateway does NOT get BUZZ_PRIVATE_KEY +# in its terminal children — blanket passthrough of a signing key to every terminal child on the host would +# be wrong (maintainer triage note on #76243: don't expose the key to unrelated shell commands). +# ``_sanitize_subprocess_env`` is also consumed by search workers (e.g. the ddgs web-search subprocess), the +# computer-use driver binary, and user-script runners (bang ``!`` commands, quick commands, cron scripts, +# webhook-filter scripts), so those children receive the vars too — matching the approved background/PTY +# scope. Every other surface stays sealed — execute_code scrubbing, :func:`hermes_subprocess_env` (browser / +# TUI host / copilot-executor spawns), docker children, and ``env_passthrough`` registration (skills/config +# still cannot register these names). The GHSA-rhgp-j443-p4rf seal is preserved because no registration path +# is opened; this is a scrub-path exemption, not an allowlist addition. First-party matches use the merged +# env value directly — they are the process's own env values and are never scope-resolved (a profile secret +# scope under multiplex would otherwise raise UnscopedSecretError at passthrough-resolution call sites); +# only skill/config passthrough names resolve through the profile secret scope. The snapshot mechanism +# treats these names like profile-scoped passthrough names (see +# ``LocalEnvironment._additional_profile_scoped_passthrough_names``) so they never persist in the shared +# terminal snapshot across profiles. Contrast with CLAUDE_CODE_OAUTH_TOKEN above, which is discarded from +# the blocklist entirely because it is NOT a Hermes credential; these ARE Hermes-managed first-party +# platform credentials, so they stay IN the blocklist for every non-terminal surface. See issue #78026 (Buzz +# agents could not use ``buzz`` from the terminal tool) and #76243 (Buzz Desktop managed agent wakes but +# cannot reply). _TERMINAL_FIRST_PARTY_ENV_PREFIXES = ("BUZZ_",) @@ -97,7 +130,10 @@ def _buzz_terminal_context_active() -> bool: """True when this process/session operates as a Buzz agent: ``BUZZ_MANAGED_AGENT`` in the process env (set only by Buzz Desktop's buzz-acp harness), or the live session's platform is ``buzz`` via the gateway ContextVar — authoritative under a concurrent - multi-session host, so a sibling Telegram session resolves its OWN platform.""" + multi-session host, so a sibling Telegram session resolves its OWN platform. + + Gateway / CLI / cron / kanban processes never carry it. See #76243. + """ if os.environ.get("BUZZ_MANAGED_AGENT"): return True try: @@ -118,6 +154,18 @@ def _is_terminal_first_party_env(name: str) -> bool: # ANOTHER project's deps into the Hermes venv (still reachable via PATH, so stripping # is safe); PYTHONHOME redirects a child interpreter's stdlib to the Hermes venv # (version-mismatch crashes). PYTHONPATH is handled separately (Hermes-owned entries only). +# The gateway runs inside its own venv, so its process environment carries VIRTUAL_ENV (and possibly +# CONDA_PREFIX). If those leak into commands the agent runs against OTHER Python projects, tools like +# ``uv``/``poetry`` treat the inherited value as the active environment and build/sync that other project's +# dependencies into the Hermes venv path instead of the project's own ``.venv`` — silently clobbering the +# Hermes environment (e.g. a project pinned to a different Python version overwrites it and breaks the +# gateway). PYTHONHOME is included because a gateway-inherited value redirects the standard-library search +# of ANY child interpreter — including unrelated system/venv Pythons — to the Hermes venv's stdlib, which +# crashes with version-mismatch errors before a child script even imports a package (#75018). Hermes itself +# treats PYTHONHOME as contamination in its own child processes (managed_uv.py, sqlite_runtime.py), so +# stripping it from subprocess envs is consistent. Users who need PYTHONHOME for a specific child can set it +# explicitly in the command. PYTHONPATH is NOT included here — it's handled by +# _strip_hermes_owned_pythonpath() which removes only Hermes-owned entries, preserving user-set paths. _ACTIVE_VENV_MARKER_VARS = ("VIRTUAL_ENV", "CONDA_PREFIX", "PYTHONHOME") diff --git a/tools/environments/local_pythonpath.py b/tools/environments/local_pythonpath.py index 54a46b7bd2..a7fcbeba08 100644 --- a/tools/environments/local_pythonpath.py +++ b/tools/environments/local_pythonpath.py @@ -124,7 +124,12 @@ def _strip_hermes_owned_pythonpath_and_runtime_markers(env: dict) -> None: def _strip_hermes_owned_pythonpath(env: dict) -> None: """Remove Hermes-owned PYTHONPATH entries: only exact matches of the repo root (any launcher spelling) and runtime site-packages — never descendants, which are - user paths. Empty components (= cwd) and everything else are preserved.""" + user paths. Empty components (= cwd) and everything else are preserved. + + Everything else -- user libs, Nix plugin paths, a pythonX.Y/site-packages entry meant for a DIFFERENT + child version -- is preserved byte-for-byte: ownership is decided by path provenance, never by a + cross-version heuristic (#74817 follow-up). + """ pp = env.get("PYTHONPATH") if not pp: return diff --git a/tools/environments/ssh.py b/tools/environments/ssh.py index af654ddc20..ae5565e34e 100644 --- a/tools/environments/ssh.py +++ b/tools/environments/ssh.py @@ -19,6 +19,7 @@ logger = logging.getLogger(__name__) # Windows OpenSSH has no Unix-socket ControlMaster: ControlPath/ControlMaster options # fail the connection outright ('getsockname failed: Not a socket'). Skip multiplexing there. +# Skip multiplexing there; each command pays a fresh connection but the backend works. See #73927. _SSH_MULTIPLEX = os.name != "nt" diff --git a/tools/file_operations.py b/tools/file_operations.py index 0bcfa7eabc..490e31835f 100644 --- a/tools/file_operations.py +++ b/tools/file_operations.py @@ -243,7 +243,11 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): transport (which decodes stdout with ``errors="replace"`` and manufactures U+FFFD for every undecodable byte, including a multibyte char cut in half by ``head -c``). None when no clean base64 came back (no ``base64`` binary); - callers then fall back to the text heuristic.""" + callers then fall back to the text heuristic. + + Wrapping the sample in base64 lets the original bytes survive the transport, so binary detection can + happen at the byte layer where it is well-defined (#80308 and friends). + """ result = self._exec(f"head -c {length} {self._escape_shell_arg(path)} 2>/dev/null | base64") if result.exit_code != 0: return None @@ -270,7 +274,10 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): multibyte sequence at the very end (artifact of the byte-boundary cut). NUL bytes or mid-stream invalid UTF-8 stay read-only so a read→edit→write round-trip never rewrites undecodable bytes as U+FFFD; a file that - legitimately CONTAINS U+FFFD is valid UTF-8 and reads as text.""" + legitimately CONTAINS U+FFFD is valid UTF-8 and reads as text. + + See #80308. + """ if not sample: return False if b"\x00" in sample: @@ -374,6 +381,21 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): tmpl = self._escape_shell_arg(".hermes-tmp.XXXXXX") script = ( "set -e; " + # One shell script, fully quoted. Notes: - `mkdir -p "$d"` is folded in here so the parent + # directory is created in the same subprocess that writes the temp file — saves one entire + # subprocess spawn vs. a separate mkdir call. - `mktemp` lands the temp in the target's own dir + # (-p) so `mv` is same-FS atomic; we fall back to a PID-stamped name if the backend lacks mktemp + # (rare; busybox/macOS/Linux all ship it). - `chmod --reference` is GNU-only, so we read the + # octal mode with `stat` (GNU `-c%a` or BSD `-f%Lp`) and `chmod` it explicitly; silent + # best-effort — a perms-copy failure must not abort the write (the file then lands at mktemp's + # 0600, same as pre-fix). - brand-new targets get `chmod "=rw"` — the POSIX who-less symbolic + # form, which sets rw minus the process umask (e.g. 0644 under umask 022) instead of mktemp's + # hardcoded 0600 (#70856). Deliberately NOT shell arithmetic on `$(umask)`: zsh (reachable via + # _find_bash's $SHELL fallback) parses leading-zero constants as decimal and silently computes a + # garbage mode, while `chmod "=rw"` is spec-identical in bash/dash/ash/zsh and degrades to 0600 + # (pre-fix behavior) if an exotic chmod rejects it. - `trap ... EXIT` guarantees the temp is + # removed on every error path (cat failure, mv failure, signal) but NOT after a successful mv + # (the temp no longer exists by then). - we `cat >` the temp, then `mv -f` it over the target. f"d={q_parent}; t={q_path}; " 'if [ -L "$t" ]; then ' 'rt="$(readlink -f "$t" 2>/dev/null || realpath "$t" 2>/dev/null || true)"; ' @@ -390,6 +412,8 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): '[ -n "$m" ] && chmod "$m" "$tmp" 2>/dev/null || true; ' "fi; " 'cat > "$tmp"; ' + # new file: umask-default perms instead of mktemp's 0600 (#70856). Runs AFTER cat so a + # write-masking umask can't EACCES the stream; quoted "=rw" so zsh doesn't =word-expand it. 'if [ ! -e "$t" ]; then chmod "=rw" "$tmp" 2>/dev/null || true; fi; ' 'mv -f "$tmp" "$t"; ' "trap - EXIT") @@ -452,6 +476,9 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): # → BE; both parities or a single zero → real binary. Legacy 8-bit # encodings (GBK, Big5) are never guessed — a wrong silent guess is worse # than a clear refusal. + # UTF-16 rescue constants (ported from MoonshotAI/kimi-code#2647, detection derived from VS Code's + # encoding sniffer): sample the leading bytes; trust a BOM first, then a zero-byte parity heuristic — + # zeros clustering at odd indices mean UTF-16 LE (`0xAA 0x00`), at even indices UTF-16 BE (`0x00 0xAA`). _UTF16_MAX_BYTES = 10 * 1024 * 1024 _UTF16_SAMPLE_BYTES = 512 @@ -768,7 +795,13 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): def _read_binary_file(self, path: str, offset: int, limit: int, file_size: int, sample_bytes: Optional[bytes]) -> ReadResult: """Binary branch shared by every read path: UTF-16 text (Notepad, PowerShell - ``>``) trips the binary guard; transcode it, else refuse with the type name.""" + ``>``) trips the binary guard; transcode it, else refuse with the type name. + + UTF-16 rescue (ported from MoonshotAI/kimi-code#2647): the terminal env decodes stdout as UTF-8 with + errors="replace", so a UTF-16 text file (Windows Notepad .txt, PowerShell `>` redirects) arrives + mangled with U+FFFD and trips the binary guard. Probe the raw bytes via the backend's Python and + transcode to UTF-8 when a BOM or the zero-byte parity heuristic identifies UTF-16. + """ utf16_result = self._try_read_utf16(path, offset, limit, file_size) if utf16_result is not None: return utf16_result diff --git a/tools/file_tools.py b/tools/file_tools.py index 9053745425..4674d3ddec 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -78,6 +78,12 @@ def _truncate_to_char_budget(content: str, max_chars: int) -> tuple[str, int, bo Returns ``(kept_text, lines_kept, truncated)`` so the caller can offer a ``next_offset`` instead of rejecting the read. If not even the first line fits it is clamped mid-line so the read is never empty and the cursor advances. + + Ported in spirit from nearai/ironclaw#5029 (dual line/byte cap on ``read_file``). Where hermes + previously hard-rejected an oversized read (forcing the model to guess a smaller ``limit`` and burn a + round-trip returning nothing), this trims the content to the last *complete line* that fits within + ``max_chars`` and reports how many lines were kept so the caller can offer a ``next_offset`` + continuation. """ if len(content) <= max_chars: return content, (content.count("\n") + 1 if content else 0), False @@ -274,6 +280,12 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str): # Re-apply the container cwd guard: a gateway/TUI/ACP override is a raw HOST # path and ``docker run -w <host-path>`` makes search_files & co silently # return nothing. Valid in-container overrides (/workspace, /root) pass. + # Re-apply the container cwd guard that _get_env_config() already ran on config["cwd"] (see #50636). A + # per-task cwd override registered by the gateway/TUI/ACP for workspace tracking is a raw host path + # (e.g. a Desktop session's /Users/<me>/workspace or C:\\Users\\<me>). On a container backend that + # reaches ``docker run -w <host-path>`` and the container starts in a directory that doesn't exist + # inside the sandbox, so search_files and friends silently return empty results (#54447). Sanitize it + # back to the already-validated config["cwd"] so the override can't bypass the guard. if env_type in _CONTAINER_BACKENDS and _is_unusable_container_cwd(cwd): if cwd != config["cwd"]: logger.info( @@ -317,6 +329,12 @@ def _get_file_ops(task_id: str = "default") -> ShellFileOperations: return cached # Env was cleaned up: rescue its cwd into the session record FILL-ONLY # (``cached.cwd`` is the SHARED env's cwd, not this session's own). + # Environment was cleaned up -- preserve the old cwd in the session record before invalidating + # the stale cache entry (fixes #26211: silent file-creation failures in long-running + # conversations). Usually a no-op: every completed command already recorded its cwd. Fill-only: + # ``cached.cwd`` is a snapshot of the SHARED env's cwd at cache-build time, so it is not + # attributable to this session (same class as the interrupted-command bug, #85658). Rescue a + # session that has no record, but never overwrite a record the session wrote for itself. old_cwd = getattr(cached, "cwd", None) if old_cwd: try: @@ -515,6 +533,12 @@ def _record_successful_read(task_data: dict, task_id: str, path: str, resolved_s if not partial: try: + # Background-review read-before-write guard integration (#61521): when the self-improvement + # review fork reads a skill file with read_file (now whitelisted dispatch-side), register the + # read the same way skill_view does, so a follow-up skill_manage(action='patch') on the loaded + # file is accepted. A partial read doesn't count — the guard requires the CURRENT full content + # to have been seen. No-op outside review forks (mark_background_review_skill_read gates on + # is_background_review). from tools.skill_manager_tool import mark_background_review_skill_read mark_background_review_skill_read(Path(resolved_str)) except Exception: diff --git a/tools/file_tools_paths.py b/tools/file_tools_paths.py index a4246049be..d40325a308 100644 --- a/tools/file_tools_paths.py +++ b/tools/file_tools_paths.py @@ -21,7 +21,11 @@ _ENV_CLASS_NAME_HINTS = ("local", "ssh", "docker", "singularity", "modal", "dayt def _expand_tilde(path: str) -> str: """Expand ``~`` using the effective profile home (``get_subprocess_home``) so - gateway/cron runs, whose process HOME may differ, agree with interactive CLI sessions.""" + gateway/cron runs, whose process HOME may differ, agree with interactive CLI sessions. + + This mirrors ``hermes_constants.get_subprocess_home()`` so that ``~`` resolves consistently regardless + of whether the tool runs interactively or inside a gateway-driven cron job (#48552). + """ if not path or "~" not in path: return path try: @@ -87,6 +91,7 @@ def _sentinel_free_abs_cwd(raw: str | None) -> str | None: def _configured_terminal_cwd() -> str | None: """Return ``$TERMINAL_CWD`` only when it names a real (absolute, non-sentinel) anchor. Scope-aware: under gateway multiplexing the routed profile's cwd lives in the per-turn scope.""" + # See #68559. from agent.runtime_cwd import scope_terminal_cwd return _sentinel_free_abs_cwd(scope_terminal_cwd() or None) diff --git a/tools/file_tools_read_tracking.py b/tools/file_tools_read_tracking.py index 6c2d0fd304..92063efe7b 100644 --- a/tools/file_tools_read_tracking.py +++ b/tools/file_tools_read_tracking.py @@ -199,7 +199,11 @@ def _invalidate_dedup_for_path(filepath: str, task_id: str) -> None: def _update_read_timestamp(filepath: str, task_id: str) -> None: """After a successful write: invalidate dedup and refresh the stored mtime so - consecutive edits by the same task don't trigger false staleness warnings.""" + consecutive edits by the same task don't trigger false staleness warnings. + + Also invalidates the dedup cache for the written path so that subsequent reads return fresh content + (fixes #13144). + """ _invalidate_dedup_for_path(filepath, task_id) resolved = _resolved_or_none(filepath, task_id) if resolved is None: diff --git a/tools/file_tools_write_guards.py b/tools/file_tools_write_guards.py index 147a1a2f35..39b61e7632 100644 --- a/tools/file_tools_write_guards.py +++ b/tools/file_tools_write_guards.py @@ -99,6 +99,13 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None # vector (AGENTS.md / CLAUDE.md / SOUL.md / .cursorrules / project .hermes tree). # Writes ALWAYS require human approval — even under --yolo — and fail closed # without a human channel. Basenames match in ANY directory, case-insensitively. +# Ported from: RooCodeInc/Roo-Code RooProtectedController (Apache-2.0). Companion: the terminal-tool vector +# is covered separately (#58631); this gate covers the write_file/patch vector. Symlink lesson from #41351: +# always realpath before matching. Scope decision (documented): basenames match in ANY directory, because +# project-context instruction files are loaded from cwd trees — an AGENTS.md anywhere the agent might later +# run from is a live target. Basenames match case-insensitively so case-variant spellings on +# case-insensitive filesystems (macOS/Windows) cannot slip past; on case-sensitive filesystems most loaders +# probe common case variants too, so the stricter behavior is kept uniform. _PROTECTED_INSTRUCTION_BASENAMES = frozenset({ "agents.md", "claude.md", "soul.md", ".cursorrules"}) @@ -125,7 +132,12 @@ def _protected_instruction_reason(filepath: str, task_id: str = "default", *, enabled: bool | None = None, extra_patterns: list[str] | None = None) -> str | None: """Return a short label when ``filepath`` targets a protected instruction file, else ``None``. - Matches BOTH the normalized input and its realpath so no symlink direction escapes.""" + Matches BOTH the normalized input and its realpath so no symlink direction escapes. + + Matching runs on BOTH the normalized input path and its realpath so neither a symlink pointing AT a + protected file (#41351) nor a protected name that is itself a symlink escapes the gate. ``..`` traversal + is neutralized by normpath/realpath before the basename compare. + """ if enabled is None or extra_patterns is None: enabled, extra_patterns = _protected_instruction_config() if not enabled: @@ -327,7 +339,13 @@ def _check_cross_profile_path(filepath: str, task_id: str = "default") -> str | def _check_binary_document_write(filepath: str, task_id: str = "default") -> str | None: """Reject text-tool writes that would corrupt a binary document (read_file showed EXTRACTED text, so the model may write it back). Opaque formats are always rejected; - .pdf only when OVERWRITING an existing file (raw PDF syntax is text-authorable).""" + .pdf only when OVERWRITING an existing file (raw PDF syntax is text-authorable). + + ``read_file`` auto-extracts .docx/.xlsx/.pptx (and PDF, via anydoc) to readable text, so the model + plausibly believes it holds the file's contents and tries to write the edited text back with + write_file/patch. A plain-text write can never produce a valid OOXML/OLE/ODF container, so that write + silently destroys the document (port of nearai/ironclaw#7109). + """ if has_opaque_document_extension(filepath): ext = filepath[filepath.rfind("."):].lower() return ( diff --git a/tools/image_generation_tool.py b/tools/image_generation_tool.py index a87ac04e81..bc9d3fc5f5 100644 --- a/tools/image_generation_tool.py +++ b/tools/image_generation_tool.py @@ -167,7 +167,14 @@ def _read_configured_image_model(): def _read_configured_image_provider(): """``image_gen.provider`` from config.yaml, or None (unset keeps the in-tree FAL fallback even - when other providers are registered; ``"fal"`` routes via ``plugins/image_gen/fal/``).""" + when other providers are registered; ``"fal"`` routes via ``plugins/image_gen/fal/``). + + We only consult the plugin registry when this is explicitly set — an unset value keeps users on the + in-tree FAL fallback even when other providers happen to be registered (e.g. a user has OPENAI_API_KEY + set for other features but never asked for OpenAI image gen). ``"fal"`` explicitly routes through + ``plugins/image_gen/fal/`` (which delegates back into this module's pipeline via call-time indirection — + see issue #26241). + """ return _read_image_gen_key("provider") @@ -565,6 +572,7 @@ IMAGE_GENERATE_SCHEMA = { }, # image_url / reference_image_urls / upscale are added per-capability; never statically. }, + # See #95681. "required": ["prompt"], }, } diff --git a/tools/interrupt.py b/tools/interrupt.py index 45b5ad0306..04f0a0ab47 100644 --- a/tools/interrupt.py +++ b/tools/interrupt.py @@ -48,7 +48,10 @@ def is_interrupted() -> bool: def is_thread_interrupted(thread_id: int | None) -> bool: """Whether *thread_id* has an interrupt bit set (``None`` never is). Used when a wait moves onto a deadline worker (``run_bounded_sync``) so ``/stop`` - targeting the original tool-worker tid still kills the subprocess.""" + targeting the original tool-worker tid still kills the subprocess. + + See #94285. + """ if thread_id is None: return False with _lock: diff --git a/tools/kanban_tools.py b/tools/kanban_tools.py index 8b69793c30..35e1cb74c6 100644 --- a/tools/kanban_tools.py +++ b/tools/kanban_tools.py @@ -164,7 +164,12 @@ def _stamp_worker_session_metadata(task_id: str, metadata: Optional[dict]) -> Op def _enforce_worker_task_ownership(tid: str) -> None: """A dispatcher-spawned worker may only mutate its own HERMES_KANBAN_TASK; a prompt-injected ``task_id`` must not corrupt sibling/cross-tenant runs. - Orchestrators (toolset enabled, no env task) legitimately route child tasks.""" + Orchestrators (toolset enabled, no env task) legitimately route child tasks. + + Tools like ``kanban_complete`` / ``kanban_block`` / ``kanban_heartbeat`` mutate run-lifecycle state, so + a buggy or prompt-injected worker that passed an explicit ``task_id`` for some other task could corrupt + sibling or cross-tenant runs (see #19534). + """ env_tid = os.environ.get("HERMES_KANBAN_TASK") if env_tid and tid != env_tid: raise _Reject( @@ -392,6 +397,19 @@ def _goal_gate(tool_name: str, task, tid: str, evidence: str) -> None: # rate-limited per process (a race costs one harmless extra write); no-op outside a # dispatcher-spawned worker. +# --------------------------------------------------------------------------- Runtime-activity → +# board-heartbeat bridge (#31752) +# --------------------------------------------------------------------------- When the agent ticks +# ``_touch_activity`` during normal work (between tool calls, mid-stream chunks, etc.), we want the kanban +# board's ``last_heartbeat_at`` columns to reflect that liveness so the dispatcher watchdog (which reads +# ``tasks.last_heartbeat_at``, not the agent's in-process timestamp) doesn't reclaim an actively-running +# worker as stale. The model is not required to call the explicit ``kanban_heartbeat`` tool for this to work +# — that tool stays available for workers that want to attach a note or pre-emptively extend a claim across +# a known-long op. Constraints: - Best-effort: never raise. The agent loop must not care if the bridge fails +# (board missing, DB locked, etc.). - Rate-limited to one DB write per 60s per-process; runtime activity can +# tick on every chunk/tool result and we don't need that resolution. - No-op outside dispatcher-spawned +# worker context (no ``HERMES_KANBAN_TASK``). - No durable note on these auto-heartbeats; that's reserved +# for the explicit tool which carries a model-supplied note. _AUTO_HEARTBEAT_MIN_INTERVAL_SECONDS = 60.0 _auto_heartbeat_last_attempt: float = 0.0 @@ -536,6 +554,9 @@ def _handle_complete(args: dict, **kw) -> str: _require_dict_metadata(metadata) metadata = _stamp_worker_session_metadata(tid, metadata) with _board(args.get("board")) as (kb, conn): + # Goal-mode pre-completion judge gate (Issue #38367). Prevent workers from bypassing the auxiliary + # judge by calling kanban_complete before acceptance criteria are met. Only enforce when a judge is + # actually reachable — see _goal_judge_available for why an unavailable judge fails open. task = kb.get_task(conn, tid) _goal_gate("kanban_complete", task, tid, (summary or result or "").strip()) try: @@ -543,6 +564,11 @@ def _handle_complete(args: dict, **kw) -> str: conn, tid, result=result, summary=summary, metadata=metadata, created_cards=created_cards, expected_run_id=_worker_run_id(tid)) except kb.ArtifactPreservationError as artifact_err: + # Structured rejection — surface the phantom ids so the worker can retry with a corrected list + # or drop the field. Audit event already landed in the DB. The task itself was NOT mutated (the + # gate runs before the write txn), so the worker can simply call kanban_complete again. Spell + # that out — without it the model often interprets a tool_error as a terminal failure and either + # blocks or crashes the run instead of retrying. See #22923. return tool_error( f"kanban_complete could not preserve the declared artifacts: {artifact_err}. " f"Your task is still in-flight and its scratch workspace was kept. Fix the " @@ -576,6 +602,13 @@ def _handle_block(args: dict, **kw) -> str: # The goal loop treats ANY blocked status as terminal, so kanban_block # would be an escape hatch around the completion judge: goal_mode tasks # may only block on genuine external blockers. + # Goal-mode block gate (Issue #38696, sibling of the kanban_complete judge gate in #38367). + # kanban_block is a second exit path out of the goal loop — run_kanban_goal_loop() treats ANY + # `blocked` status as terminal, identically to `done`, regardless of kind. Without this, a worker + # that learns kanban_complete is gated can just call kanban_block(reason="anything") to escape the + # loop instead. Restrict goal_mode tasks to the kinds that represent a genuine external blocker the + # worker cannot resolve itself; `capability` and `transient` (or an unset kind) route back through + # kanban_complete, which the judge now gates. task = kb.get_task(conn, tid) _check(not (task and task.goal_mode and kind not in _GOAL_MODE_BLOCK_ALLOWED_KINDS), f"goal_mode tasks can only block with kind in " @@ -653,6 +686,10 @@ def _handle_comment(args: dict, **kw) -> str: # injected into future workers' system prompts, so an args["author"] override could # forge a directive from ``hermes-system``. Cross-task commenting stays unrestricted — # it is the handoff channel between tasks. + # Comments are injected into the next worker's system prompt by ``build_worker_context`` as + # ``**{author}** (timestamp): {body}`` — accepting an ``args["author"]`` override let a worker forge a + # comment from an authoritative-looking name like ``hermes-system`` and poison the future-worker context + # with what reads as a system directive. See #19713. author = os.environ.get("HERMES_PROFILE") or "worker" with _board(args.get("board")) as (kb, conn): cid = kb.add_comment(conn, tid, author=author, body=str(body)) @@ -778,6 +815,7 @@ def _handle_create(args: dict, **kw) -> str: # mutate review evidence or race its checkout). Project identity is the one safe thing # to inherit implicitly (the DB turns it into a fresh per-task worktree). workspace_kind, workspace_path = args.get("workspace_kind"), args.get("workspace_path") + # See #67567. project_id = args.get("project") or args.get("project_id") project_source_task_id = None triage, skills, goal_mode = ( diff --git a/tools/lazy_deps.py b/tools/lazy_deps.py index ad03f43f09..234d61c79f 100644 --- a/tools/lazy_deps.py +++ b/tools/lazy_deps.py @@ -117,6 +117,9 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { # brotlicffi: aiohttp needs its 2-arg Decompressor for Discord CDN Brotli attachments # (google's 1-arg `Brotli` fails "Can not decode br"). aiohttp is only capped transitively # by these adapters, so pin the patched floor explicitly. + # Without it, aiohttp falls back to google's `Brotli` package (1-arg API), and any .txt/.md/.doc + # uploaded to the Discord gateway fails to decode at att.read() with "Can not decode content-encoding: + # br" — see #12511 / #15744. "platform.discord": ( "discord.py[voice]==2.7.1", "brotlicffi==1.2.0.1", @@ -176,6 +179,9 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { ), # Pillow and firecrawl-anydoc are CORE deps; these entries self-heal lean/partial installs. # Call sites use prompt=False so read_file / vision never block on input() mid-session. + # Vision image-resize recovery (Pillow). Pillow is now a CORE dependency (pyproject `dependencies`), so + # this entry is a belt-and-suspenders fallback for stripped/source-build installs that somehow dropped + # it. See #40490. "tool.vision": ("Pillow==12.3.0",), "tool.doc_extract": ("firecrawl-anydoc==0.2.4",), # imports as `anydoc`; lockstep with pyproject # MCP client SDK for the cua-driver, so computer_use never dead-ends on `No module named 'mcp'`. @@ -187,6 +193,15 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { # huggingface-hub is SHARED with transformers (>=1.5.0,<2 via Hindsight) and marked active # on mere presence, so `hermes update` re-asserts this pin everywhere hub exists. MUST stay # inside transformers' window and match uv.lock (tests/test_project_metadata.py enforces). + # HF Agent Trace Viewer upload (hermes trace upload / /upload-trace). huggingface-hub is a SHARED + # dependency: transformers (pulled by sentence-transformers for local Hindsight embeddings) requires + # >=1.5.0,<2, and faster-whisper/tokenizers depend on it transitively. Because active_features() marks a + # feature active from mere package presence, the `hermes update` lazy-refresh pass re-asserts THIS pin + # on every install where hub is present — so an exact pin below 1.5.0 force-downgrades the shared + # package and breaks Hindsight startup (#60783). Policy: keep the exact pin (no ranges — security + # posture), but it MUST stay inside transformers' accepted window and MUST match uv.lock so the whole + # tree converges on ONE hub version (tests/test_project_metadata.py enforces both). When bumping: update + # here AND `uv lock --upgrade-package huggingface-hub` in lockstep. "tool.trace_upload": ("huggingface-hub==1.24.0",), } @@ -421,7 +436,16 @@ def _installed_dist_roots(spec: str, target: Optional[Path]) -> set[Path]: def _warm_installed_bytecode(specs: tuple[str, ...], target: Optional[Path]) -> None: """Byte-compile what was just installed: a fresh install writes no ``__pycache__``, so the next import (often a user request, ~2-10s for a big SDK, reading as a hang) would pay the compile. Pay - it here while the caller already waits. Best-effort; never fails the install.""" + it here while the caller already waits. Best-effort; never fails the install. + + A pip/uv install writes ``.py`` sources and no ``__pycache__`` — and an install of the *same* version + still deletes the cache the old copy had. Whoever imports the package next pays the whole compile: for + ``anthropic==0.87.0`` (541 modules) on cpython-3.12.13 that measured 2.2-2.7s cold against 0.7-1.0s + warm, and 10.5s cold under concurrent load. That bill lands wherever the first import happens, and for a + lazily-installed backend that is the foreground of a user request (#100461) — with nothing printed while + it runs, so it reads as a hang. Worse, N per-profile daemons cold-starting together each pay it in full + before any of them has written the cache. + """ if sys.dont_write_bytecode: return try: diff --git a/tools/mcp_oauth.py b/tools/mcp_oauth.py index 5f280905b6..47cd867670 100644 --- a/tools/mcp_oauth.py +++ b/tools/mcp_oauth.py @@ -108,6 +108,9 @@ def _safe_filename(name: str) -> str: # Callback-port reservation: bound-but-not-listening sockets keyed by port, held from selection # until the waiter adopts them (closes the select→bind TOCTOU window). Bounded so reconnect loops cannot leak fds. +# Holding the socket from port-selection time until _wait_for_callback adopts it closes the TOCTOU window +# where another process could grab the port between _find_free_port() closing its probe socket and +# HTTPServer binding minutes later (#22161). _reserved_sockets: "dict[int, socket.socket]" = {} _MAX_RESERVED_SOCKETS = 8 @@ -177,7 +180,12 @@ def _is_interactive() -> bool: def _raise_if_non_interactive(lead: str) -> None: - """Raise ``OAuthNonInteractiveError`` unless interactive; *lead* is the boundary-specific first sentence.""" + """Raise ``OAuthNonInteractiveError`` unless interactive; *lead* is the boundary-specific first sentence. + + ``lead`` is the boundary-specific first sentence; this helper appends the shared, actionable ``hermes + mcp login`` next-step so the guidance wording lives in one place across every non-interactive OAuth + boundary (#57836). + """ if not _is_interactive(): raise OAuthNonInteractiveError( f"{lead} Run `hermes mcp login <server>` interactively to (re)authorize, then restart or reload the gateway." @@ -186,13 +194,22 @@ def _raise_if_non_interactive(lead: str) -> None: def force_interactive_oauth(): """Treat the context as interactive despite no TTY (GUI-driven auth: the user IS present, just not - on stdin). Crosses the MCP event-loop thread like ``suppress_interactive_oauth``.""" + on stdin). Crosses the MCP event-loop thread like ``suppress_interactive_oauth``. + + Opens the browser + localhost callback flow that the TTY heuristic would otherwise refuse. Same + ContextVar propagation story as suppress_interactive_oauth() (#35927). + """ return _contextvar_set(_oauth_interactive_forced, True) def suppress_interactive_oauth(): """Disable stdin-based OAuth prompts for the current context; ContextVar-based so a - background-discovery thread's suppression reaches the coroutine on the MCP event-loop thread.""" + background-discovery thread's suppression reaches the coroutine on the MCP event-loop thread. + + Uses a ContextVar so the suppression propagates from a background-discovery thread onto the coroutine + scheduled (via run_coroutine_threadsafe) on the dedicated MCP event-loop thread — where the OAuth + callback actually runs (#35927). A threading.local would not cross that thread boundary. + """ return _contextvar_set(_oauth_interactive_enabled, False) @@ -219,8 +236,18 @@ def _read_json(path: Path) -> dict | None: def _write_json(path: Path, data: dict) -> None: """Atomically write *data* as JSON created at 0o600 (``O_EXCL`` + mode avoids the write-then-chmod window where the file inherits a world-readable umask); parent dir tightened to 0o700. The random - per-process tmp suffix avoids clashes with concurrent writers/crash leftovers.""" + per-process tmp suffix avoids clashes with concurrent writers/crash leftovers. + + The previous ``write_text`` + post-write ``chmod`` opened a TOCTOU window where the temp file briefly + inherited the process umask (commonly 0o644 = world-readable), exposing OAuth tokens to other local + users between create and chmod. Mirrors the fix in ``agent/google_oauth.py`` (#19673). + """ path.parent.mkdir(parents=True, exist_ok=True) + # secure_parent_dir refuses to chmod /, top-level dirs, or the hermes-agent install tree (#25821, + # #93050). + # Tighten parent dir to 0o700 so siblings can't traverse to the creds. No-op on Windows (POSIX mode bits + # aren't enforced); ignore failures. secure_parent_dir refuses to chmod /, top-level dirs, or the + # hermes-agent install tree (#25821, #93050). secure_parent_dir(path) tmp = path.with_suffix(f".tmp.{os.getpid()}.{secrets.token_hex(4)}") try: @@ -527,7 +554,11 @@ def _announce_authorization_url(authorization_url: str, port: int, redirect_uri: def _make_redirect_handler(port: int, redirect_uri: str | None = None): """Redirect handler closing over this flow's port (a closure, not ``_oauth_port``, keeps concurrent - flows isolated). ``redirect_uri`` is a configured proxy callback (None for loopback) and only tailors the hint.""" + flows isolated). ``redirect_uri`` is a configured proxy callback (None for loopback) and only tailors the hint. + + Using a closure instead of reading the module-level ``_oauth_port`` avoids cross-server state pollution + when multiple MCP servers run OAuth concurrently (fixes #44588). + """ async def _redirect_handler(authorization_url: str) -> None: dashboard_flow = get_dashboard_oauth_flow() if dashboard_flow is not None: @@ -535,6 +566,12 @@ def _make_redirect_handler(port: int, redirect_uri: str | None = None): return # Fail fast when non-interactive: a cached-but-unusable token makes the SDK fall through to the # authorization-code flow past the token-file guard, and the waiter would block for the full timeout. + # Fail fast at the authorization boundary in non-interactive contexts (systemd gateway, cron, + # background MCP discovery). Without this check we would print a URL and launch a browser flow no + # operator can complete, then block in _wait_for_callback for the full timeout. Raise before + # launching so gateway adapters start promptly and the caller can skip this server with an + # actionable warning. This intentionally re-checks interactivity here rather than trusting the + # token-file existence guard alone. See #57836. _raise_if_non_interactive( "MCP OAuth requires browser authorization but no interactive session is available (non-interactive/background context)." ) @@ -587,7 +624,13 @@ def _make_callback_waiter(port: int, cimd_url: str | None = None, timeout: float """Callback waiter bound to one flow's port. ``timeout`` is where ``oauth.timeout`` applies (mcp 2.0 dropped the provider's own). ``cimd_url`` only tailors the timeout message: a server refusing the document aborts at the authorization endpoint, so no redirect arrives and a bare "timed out" would - hide the cause. Raises ``OAuthNonInteractiveError`` on timeout or when non-interactive.""" + hide the cause. Raises ``OAuthNonInteractiveError`` on timeout or when non-interactive. + + Closing over the port (instead of reading the module-level ``_oauth_port``) keeps concurrent OAuth flows + isolated: flow A's waiter listens on flow A's port even when flow B's ``_configure_callback_port`` + overwrites the legacy global afterwards (#34260, the callback-side sibling of the #44588 + redirect-handler fix). + """ async def _wait(): dashboard_flow = get_dashboard_oauth_flow() if dashboard_flow is not None: @@ -595,6 +638,14 @@ def _make_callback_waiter(port: int, cimd_url: str | None = None, timeout: float return _authorization_code_result(*await dashboard_flow.wait_for_callback()) # The SDK entered the authorization-code flow, so any cached token is unusable. Reject BEFORE # binding: binding would block for the full timeout and collide with the TIME_WAIT port on retry. + # Reject before binding the callback listener in non-interactive contexts. Reaching here means the + # SDK entered the authorization-code flow (a valid or refreshable token would never call the + # callback handler), so a cached token file is present but unusable. Binding the listener here would + # block for the full 300s timeout and — on the next connection retry — collide with the + # still-bound/TIME_WAIT port, surfacing as ``OSError: [Errno 98] Address already in use``. Failing + # fast keeps gateway startup independent of an unusable optional MCP server. This guard holds + # "regardless of whether a token file exists" — the point the build_oauth_auth token-file guard + # cannot cover. See #57836. _raise_if_non_interactive( "OAuth callback requires an interactive session but none is available (non-interactive/background " "context); skipping browser authorization without binding a callback listener.") @@ -658,6 +709,8 @@ def _is_valid_cimd_url(url: str) -> bool: # Pinned ports this process committed to (never released: a provider keeps its port for the process # lifetime), including ports restored from a cached registration. +# Includes a port restored from a cached client registration, so a sibling server is never handed a port +# another one is already registered on (#34260). _assigned_cimd_ports: "list[int]" = [] @@ -665,7 +718,12 @@ def _pick_cimd_port() -> int | None: """Reserve a pinned CIMD callback port, or None when none is usable. Holding the bound socket makes contention cooperative: a sibling finds the bind refused and moves down the range. Once every pinned port belongs to this process the range wraps rather than falling back to DCR — a reused port only - bites if both servers authorize at the same moment (reported by the waiter); DCR may be unsupported entirely.""" + bites if both servers authorize at the same moment (reported by the waiter); DCR may be unsupported entirely. + + Holding the bound socket until ``_wait_for_callback`` adopts it does the same job here as + ``_reserve_callback_port`` does for ephemeral ports (#22161): a fixed port is just as stealable in the + minutes between selection and the browser redirect arriving. + """ for port in _CIMD_PORTS: if port not in _assigned_cimd_ports and _bind_reserved(port) is not None: _assigned_cimd_ports.append(port) @@ -727,7 +785,12 @@ def _configure_callback_port(cfg: dict, storage: "HermesTokenStorage | None" = N """Resolve the callback port into ``cfg['_resolved_port']`` (0 = non-loopback URI). Precedence: dashboard flow / cached https redirect URI → CIMD pinned port (sets ``cfg['_cimd_url']``) → ``oauth.redirect_port`` → cached registration port → fresh ephemeral port (the only parked one). - Also sets the legacy ``_oauth_port``.""" + Also sets the legacy ``_oauth_port``. + + NOTE: also sets the legacy module-level ``_oauth_port`` so existing calls to ``_wait_for_callback`` keep + working. The legacy global is the root cause of issue #5344 (port collision on concurrent OAuth flows); + replacing it with a ContextVar is out of scope for this consolidation PR. + """ global _oauth_port dashboard_flow = get_dashboard_oauth_flow() if dashboard_flow is not None: @@ -747,6 +810,12 @@ def _configure_callback_port(cfg: dict, storage: "HermesTokenStorage | None" = N # A cached port may be a pinned CIMD port from an earlier login; claim it from siblings. if port in _CIMD_PORTS and port not in _assigned_cimd_ports: _assigned_cimd_ports.append(port) + # Precedence: explicit config port → cached client-registration port → fresh ephemeral port. The cached + # port keeps re-auth consistent with the redirect URI pinned at dynamic client registration (providers + # reject a mismatched URI). Only a truly fresh ephemeral pick goes through _reserve_callback_port(), + # which keeps the socket bound until _wait_for_callback adopts it — closing the select→bind TOCTOU race + # (#22161). Explicit and cached ports are fixed, known values and bind via the reuse_address path + # instead. cfg["_resolved_port"] = port _oauth_port = port return port @@ -823,7 +892,11 @@ def _invalidate_tokens_on_client_change( """Drop cached tokens when the configured client identity changes: tokens minted under the old ``client_id`` fail refresh with ``invalid_client``, and pre-registered clients are exempt from auto-poison, so stale tokens would wedge every request until a manual wipe. Compares on-disk - ``client.json`` BEFORE it is overwritten; a matching identity is a no-op.""" + ``client.json`` BEFORE it is overwritten; a matching identity is a no-op. + + Matching identity is a no-op so live sessions and valid tokens are preserved. Port of + cline/cline#12983's "invalidate tokens when OAuth client changes" invariant. + """ existing = _read_json(storage._client_info_path()) old_client_id = existing.get("client_id") if isinstance(existing, dict) else None if not old_client_id or (old_client_id == new_client_id and (existing.get("client_secret") or None) == (new_client_secret or None)): diff --git a/tools/mcp_oauth_manager.py b/tools/mcp_oauth_manager.py index 9f749de74e..e443d3b193 100644 --- a/tools/mcp_oauth_manager.py +++ b/tools/mcp_oauth_manager.py @@ -44,7 +44,11 @@ class HermesMCPOAuthProvider(HermesProviderMixin, *_SDK_BASES): """OAuthClientProvider with pre-flow disk-mtime reload (external refreshes become visible to a running session), expiry seeding on cold load, pre-flight metadata discovery, dead-client registration detection and the bidirectional ``async_auth_flow`` bridge. Token-endpoint - fixes come from ``HermesProviderMixin``. Only usable when the SDK's OAuth module imported.""" + fixes come from ``HermesProviderMixin``. Only usable when the SDK's OAuth module imported. + + Reference: Claude Code's ``invalidateOAuthCacheIfDiskChanged`` (``src/utils/auth.ts:1320``, CC-1096 / + GH#24317). + """ _hermes_logger = logger @@ -170,7 +174,10 @@ class HermesMCPOAuthProvider(HermesProviderMixin, *_SDK_BASES): is dead server-side: delete ``client.json`` (+ stale metadata) so the SDK re-runs DCR next flow. Conservative: acts ONLY on 400/401 at the discovered ``token_endpoint`` (the only request carrying our ``client_id``) with ``invalid_client`` in the body; pre-registered clients are never poisoned; any failure - is swallowed. The browser-side "Redirect URI Mismatch" case has no HTTP signal (``hermes mcp reauth``).""" + is swallowed. The browser-side "Redirect URI Mismatch" case has no HTTP signal (``hermes mcp reauth``). + + See #36767. + """ try: if (self._hermes_preregistered or getattr(response, "status_code", None) not in (400, 401) or not await self._is_invalid_client_at_token_endpoint(response)): @@ -202,6 +209,14 @@ class HermesMCPOAuthProvider(HermesProviderMixin, *_SDK_BASES): self._log_nonfatal("pre-flow disk-watch", exc) # Bridge the bidirectional generator by hand: a naive ``async for item in inner: yield # item`` DISCARDS the responses httpx sends back via ``asend``, and the SDK crashes on None. + # Manually bridge the bidirectional generator protocol. httpx's auth_flow driver + # (httpx._client._send_handling_auth) calls ``auth_flow.asend(response)`` to feed HTTP responses + # back into the generator. A naive wrapper using ``async for item in inner: yield item`` DISCARDS + # those .asend(response) values and resumes the inner generator with None, so the SDK's ``response = + # yield request`` branch in mcp/client/auth/oauth2.py sees response=None and crashes at ``if + # response.status_code == 401`` with AttributeError. The bridge below forwards each .asend() value + # into the inner generator via inner.asend(incoming), preserving the bidirectional contract. + # Regression from PR #11383 caught by tests/tools/test_mcp_oauth_bidirectional.py. inner = super().async_auth_flow(request) resource_lock_released = retry_after_concurrent_auth = False sent_access_token = None @@ -228,6 +243,8 @@ class HermesMCPOAuthProvider(HermesProviderMixin, *_SDK_BASES): await inner.aclose() retry_after_concurrent_auth = True break + # Sniff the response for a dead-client-registration signal before handing it back to the SDK + # (best-effort, GH#36767). await self._maybe_flag_poisoned_client(incoming) outgoing = await inner.asend(incoming) except StopAsyncIteration: diff --git a/tools/mcp_oauth_provider.py b/tools/mcp_oauth_provider.py index cfe6247f3d..38757caa9a 100644 --- a/tools/mcp_oauth_provider.py +++ b/tools/mcp_oauth_provider.py @@ -31,6 +31,8 @@ class HermesProviderMixin: def __init__(self, *args: Any, token_user_agent: str | None = None, **kwargs: Any): super().__init__(*args, **kwargs) + # oauth.user_agent — stamped onto token-endpoint requests only; some authorization servers/WAFs + # reject httpx's default (#75576). self._hermes_token_user_agent = token_user_agent def _prepare_token_request(self, request): diff --git a/tools/mcp_tool.py b/tools/mcp_tool.py index 31a68debc2..f02ae3b156 100644 --- a/tools/mcp_tool.py +++ b/tools/mcp_tool.py @@ -269,6 +269,7 @@ def _client_session_accepts(kwarg: str) -> bool: # MCP logging levels (RFC 5424 syslog severities) -> Python logging levels. +# Port of anomalyco/opencode#34529's serverLog mapping. _MCP_LOG_LEVEL_MAP = { "debug": logging.DEBUG, "info": logging.INFO, "notice": logging.INFO, "warning": logging.WARNING, "error": logging.ERROR, "critical": logging.ERROR, @@ -368,6 +369,11 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM # Rapid-drop budget: a session is UNPROVEN until it survives a keepalive interval or a # successful call; only a proven session clears the budget, so a post-handshake flapper # still parks. + # Rapid-drop budget (#62212): a freshly (re)established session is UNPROVEN until it demonstrates + # real health — it survived at least one full keepalive interval (keepalive success path) or served + # at least one successful tool call. Only a proven session clears the reconnect budget; a transport + # that flaps right after the handshake keeps getting charged and still reaches the park instead of + # hot-cycling respawns forever. self._session_proven: bool = False # Never cleared (unlike _ready): separates first-connect from reconnect failures. self._ever_connected: bool = False @@ -375,10 +381,13 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM self._was_parked: bool = False # In-flight RPC tasks so a deliberate teardown fails them fast; _reconnecting is True # during that teardown so _track_inflight_rpc turns the cancel into a retryable error. + # In-flight RPC bookkeeping (#48069 salvage): user-visible requests registered while running so a + # reconnect/shutdown teardown can fail them fast instead of orphaning them on a dying transport. self._inflight_tasks: set = set() self._reconnecting: bool = False # Latched by races (teardown-vs-keepalive, auth-lock corruption); ensure_healthy() # verifies before the next call. + # See #77765, #81051, #84132. self._suspect_reason: Optional[str] = None # Teardown that failed in-flight calls => next reconnect is RACE RECOVERY, not a # budget charge. @@ -387,6 +396,9 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM # cycle before parking. self._permanent_grace_used: bool = False # Children of the current stdio transport: in-flight calls fail FAST when one dies. + # PIDs of the stdio subprocess spawned for the current transport (captured in _run_stdio). Used to + # fail in-flight calls FAST when the child dies instead of waiting out the full tool timeout + # (#81995). self._stdio_child_pids: Set[int] = set() self._auth_type: str = "" self._refresh_lock = asyncio.Lock() @@ -400,6 +412,10 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM self._lifecycle_started_at = self._last_tool_call_at = time.monotonic() self._idle_timeout_seconds = self._max_lifetime_seconds = self._recycled_reason = None # Handshake InitializeResult: the server's REAL advertised capabilities. + # Captures the ``InitializeResult`` returned by ``await session.initialize()`` so downstream code + # can inspect the server's real advertised capabilities (``.capabilities.resources``, + # ``.capabilities.prompts``) instead of assuming every ``ClientSession`` method attribute + # corresponds to a supported server method. See #18051. self.initialize_result: Optional[Any] = None # SEP-2549 cache hints from the last tools/list (ttl_ms, cache_scope). self._list_cache_meta: dict = {} @@ -421,6 +437,7 @@ _server_connecting: set[str] = set() _server_connect_errors: Dict[str, str] = {} # Lazy startup: servers registered from the schema cache without connecting; popped on # first real connection. +# Keyed by server name; entries are popped once a real connection is established on first use. See #56832. _lazy_server_configs: Dict[str, dict] = {} _lazy_server_fingerprints: Dict[str, str] = {} _lazy_server_tool_names: Dict[str, List[str]] = {} @@ -433,12 +450,31 @@ _connect_server_claim: contextvars.ContextVar[Optional[Callable[[MCPServerTask], # without it every ``discover_mcp_tools()`` (one per worker session) would respawn it — a # restart storm whose unreaped children destabilise healthy servers. Exponential-backoff # deadline honoured by ``register_mcp_servers``; cleared on success. +# Connection-retry cooldown (per-server isolation against restart storms). A single stdio MCP server that +# fails to spawn (bad PATH, ``exec: not found``, crash-on-start) is never recorded in ``_servers`` -- +# ``start()`` raises and ``_discover_and_register_server`` aborts before the ``_servers[name] = server`` +# line. Without a cooldown, EVERY subsequent ``discover_mcp_tools()`` (one per agent worker session, i.e. +# every few seconds) sees the server as "not connected" and re-spawns it from scratch. That is the restart +# storm in #50394: the failing server is re-attempted on the shared MCP event loop on every worker session, +# the subprocesses pile up unreaped, and the churn destabilises the healthy co-located servers (their tools +# intermittently surface as "Unknown tool"). Fix: after a failed connection attempt, stamp a monotonic +# ``retry_after`` deadline with exponential backoff. ``register_mcp_servers`` skips a server whose cooldown +# has not elapsed, so a chronically failing server is retried on a backoff schedule instead of on every +# worker session -- isolating it from the rest of the bridge. A successful connection clears the state. _server_connect_retry_after: Dict[str, float] = {} # name -> monotonic deadline _server_connect_failures: Dict[str, int] = {} # name -> consecutive failures _CONNECT_RETRY_BASE_BACKOFF_SEC, _CONNECT_RETRY_MAX_BACKOFF_SEC = 30.0, 600.0 # Per-server circuit breaker: closed -> open (calls short-circuit until the cooldown) -> # half-open (next call probes). Mutate only via _bump_server_error / _reset_server_error. +# After _CIRCUIT_BREAKER_THRESHOLD consecutive failures, the handler returns a "server unreachable" message +# that tells the model to stop retrying, preventing the 90-iteration burn loop described in #10447. State +# machine: closed — error count below threshold; all calls go through. open — threshold reached; +# calls short-circuit until the cooldown elapses. half-open — cooldown elapsed; the next call is a probe +# that actually hits the session. Probe success → closed. Probe failure → reopens (cooldown re-armed). +# ``_server_breaker_opened_at`` records the monotonic timestamp when the breaker most recently transitioned +# into the open state. Use the ``_bump_server_error`` / ``_reset_server_error`` helpers to mutate this state +# — they keep the count and timestamp in sync. _server_error_counts: Dict[str, int] = {} _server_breaker_opened_at: Dict[str, float] = {} _CIRCUIT_BREAKER_THRESHOLD, _CIRCUIT_BREAKER_COOLDOWN_SEC = 3, 60.0 @@ -569,6 +605,7 @@ def _update_death_supervisor(verb: str, pgids) -> None: # groups still registered, an unregister must still rebuild coverage for the # survivors. return + # See #93517. proc = _spawn_death_supervisor() _death_supervisor = proc if proc is None: @@ -623,6 +660,7 @@ def _server_registry_scope(name: str) -> Optional[str]: # Cross-process discovery guard: advisory file lock so gateway + CLI + TUI don't all discover. +# See issue #62771. _LOCK_UNAVAILABLE: Any = object() # sentinel: locking broken/unavailable _MCP_DISCOVERY_LOCK_PATH: Optional[str] = None # resolved lazily # Bounded wait when another process holds the lock. diff --git a/tools/mcp_tool_agent.py b/tools/mcp_tool_agent.py index 83d809db52..2795d647e9 100644 --- a/tools/mcp_tool_agent.py +++ b/tools/mcp_tool_agent.py @@ -224,6 +224,7 @@ def _reinject_post_build_tools(agent, tools_list: list, name_set: set) -> set: # The `context_engine` toolset is intentionally empty, so lcm_* tools exist only via this # append. Honor the enabled_toolsets gate agent_init uses, or a restricted-toolset platform # would re-leak tools the build excluded. + # See #5544. staged_engine_names: set = set() try: get_schemas = _schema_getter("context_compressor", "get_tool_schemas") diff --git a/tools/mcp_tool_content.py b/tools/mcp_tool_content.py index 92467bf40f..4b7c303400 100644 --- a/tools/mcp_tool_content.py +++ b/tools/mcp_tool_content.py @@ -15,6 +15,11 @@ logger = logging.getLogger("tools.mcp_tool") # Hard ceiling for one MCP text payload (chars), deliberately far ABOVE the budget layer's 50K # spillover threshold so ordinary large results reach spillover intact; only floods are lossy. +# This is the FIRST line of defense against a buggy or malicious MCP server returning multi-megabyte text: +# without it the full payload is allocated, JSON-encoded and handed downstream before the budget/spillover +# layer ever sees it (#56059). Distilled from #56060 (Stoltemberg), #56072 (AlexFucuson9) and #56511 +# (Tranquil-Flow), which capped at get_max_bytes() (50K) — correct protection, but at that level it would +# truncate before spillover could preserve the data. The 40% head / 60% tail split is #56511's shape. _MCP_HARD_RESULT_CAP_CHARS = 2_000_000 # Cap on decoded resource bytes per block (a misbehaving server can't fill the cache disk). # Base64 expands ~4/3; oversized payloads are rejected BEFORE decoding (never doubled in memory). @@ -24,7 +29,10 @@ _MCP_RESOURCE_MAX_B64_CHARS = _MCP_RESOURCE_MAX_BYTES * 4 // 3 + 4 def _truncate_mcp_text_result(text: str, max_chars: int = _MCP_HARD_RESULT_CAP_CHARS) -> str: """Pass text at or under ``max_chars`` unchanged; otherwise keep a 40% head / 60% tail - split with an omission notice between.""" + split with an omission notice between. + + Bound pathological MCP text before it propagates (#56059). + """ if len(text) <= max_chars: return text head_chars = int(max_chars * 0.4) @@ -37,7 +45,10 @@ def _truncate_mcp_text_result(text: str, max_chars: int = _MCP_HARD_RESULT_CAP_C def _is_reserved_mcp_meta_key(key: str) -> bool: """True if an MCP ``_meta`` key uses a protocol-reserved prefix: a ``modelcontextprotocol`` or ``mcp`` label followed by at least one more label. A trailing one - (``com.example.mcp/...``) is a vendor namespace.""" + (``com.example.mcp/...``) is a vendor namespace. + + Ported from MoonshotAI/kimi-code#2600. + """ slash = key.find("/") if slash <= 0: return False diff --git a/tools/mcp_tool_discovery.py b/tools/mcp_tool_discovery.py index 2cf6d09b75..caf352f102 100644 --- a/tools/mcp_tool_discovery.py +++ b/tools/mcp_tool_discovery.py @@ -93,7 +93,11 @@ def _request_lazy_reconnect(server_name: str, server: _core.MCPServerTask) -> bo def _resolve_server_lazy(name: str, config: dict) -> bool: - """True when ``mcp_servers.<name>.lazy`` defers connect to first tool use (default off).""" + """True when ``mcp_servers.<name>.lazy`` defers connect to first tool use (default off). + + Gated per-server by ``mcp_servers.<name>.lazy`` in config (default OFF), following the same per-server + key pattern as ``idle_timeout_seconds``. Design from #56832 (Vansh5632). + """ return _core._parse_boolish(config.get("lazy", False), default=False) @@ -125,7 +129,10 @@ def _adopt_server(name: str, server: _core.MCPServerTask) -> None: def _ensure_lazy_server_connected(server_name: str) -> bool: """Connect a lazily-registered server on demand (sync; blocks). Honours the cooldown and the ``_server_connecting`` dedup set; routes through ``_discover_and_register_server`` so - park/recycle/cooldown bookkeeping stays in one place. True when a live session exists.""" + park/recycle/cooldown bookkeeping stays in one place. True when a live session exists. + + See #50394. + """ with _core._lock: server = _core._servers.get(server_name) if server is not None and server.session is not None: @@ -166,7 +173,11 @@ def _ensure_lazy_server_connected(server_name: str) -> bool: def _get_connected_server_for_call(server_name: str) -> Optional[_core.MCPServerTask]: """Return a connected server; the single first-use connect point for lazy servers and - the wake-up point for recycled stdio ones.""" + the wake-up point for recycled stdio ones. + + Also the single first-use connect point for lazy (schema-cache registered) servers, so raw tool calls + AND the resource/prompt utility handlers all trigger the deferred spawn (#56832). + """ with _core._lock: server = _core._servers.get(server_name) is_lazy = server_name in _core._lazy_server_configs @@ -218,6 +229,9 @@ def _select_new_servers(servers: Dict[str, dict]) -> Dict[str, dict]: mid-reconnect with tools deregistered, so nothing else can nudge them: signal a reconnect.""" with _core._lock: connecting = set(_core._server_connecting) + # Only attempt servers that aren't already connected (or currently connecting) and are enabled. + # Checking ``_server_connecting`` prevents duplicate subprocess spawns when ``discover_mcp_tools()`` + # is called from multiple entry-points before the first batch finishes (#58862). new_servers = { k: v for k, v in servers.items() if k not in _core._servers and k not in connecting and k not in _core._lazy_server_configs @@ -242,6 +256,8 @@ def _register_lazy_from_cache(new_servers: Dict[str, dict]) -> Tuple[Dict[str, d """Register ``lazy: true`` servers from a valid schema-cache entry without connecting (missing/stale entry or failed registration -> eager). Returns (eager servers, lazy tool count, lazy server count).""" + # A missing or stale cache entry falls back to the normal eager connect below (which write-through + # refreshes the cache for next time). See #56832. eager_servers: Dict[str, dict] = dict(new_servers) lazy_registered = 0 lazy_server_count = 0 @@ -368,6 +384,9 @@ def _acquire_discovery_lock_with_retry(): cookie = _core._try_acquire_mcp_discovery_lock() if cookie is not None: break + # Cross-process discovery guard (#62771). A lock loser waits for the holder, then performs its own + # process-local discovery. If locking is unavailable or the bounded wait expires, preserve the previous + # fail-soft behavior by running discovery unguarded. if cookie is None: logger.warning("MCP discovery lock still held after %d retries -- running discovery unguarded", _core._MCP_DISCOVERY_LOCK_MAX_RETRIES) diff --git a/tools/mcp_tool_errors.py b/tools/mcp_tool_errors.py index eb0aa5f3c2..e01327c078 100644 --- a/tools/mcp_tool_errors.py +++ b/tools/mcp_tool_errors.py @@ -36,14 +36,24 @@ def _handshake_rejected_as_modern(exc: BaseException) -> bool: def _is_method_not_found_error(exc: BaseException) -> bool: """True if *exc* is a JSON-RPC ``method not found`` (-32601; ``ping`` is optional in MCP). The substring fallback includes "Unknown method: <name>" — without it the ping→list_tools keepalive - fallback never latches and reconnect-loops.""" + fallback never latches and reconnect-loops. + + The substring fallback matters when a server reports method-not-found without a structural ``-32601`` + code (e.g. surfaced as a plain exception string). Besides the canonical "method not found", many + JSON-RPC implementations phrase it as "Unknown method: <name>" — agentmemory's MCP server is one such + case (#50028). + """ return _jsonrpc_matches( exc, (_core._JSONRPC_METHOD_NOT_FOUND,), (str(_core._JSONRPC_METHOD_NOT_FOUND), "method not found", "unknown method", "not found: ping")) class InvalidMcpUrlError(ValueError): - """A remote MCP server's ``url`` is not parseable http(s):// — validated once at startup to fail fast.""" + """A remote MCP server's ``url`` is not parseable http(s):// — validated once at startup to fail fast. + + Validated once at startup so we fail fast with a clear message instead of burning through the + reconnect-backoff loop on every attempt. (Ported from anomalyco/opencode#25019.) + """ class NonMcpEndpointError(ConnectionError): @@ -274,6 +284,8 @@ def _is_auth_error(exc: BaseException) -> bool: # Lower-cased substrings meaning the transport session expired / was GC'd (OAuth token still valid). +# Substrings (lower-cased match) that indicate the MCP server rejected the request because its server-side +# transport session expired / was garbage-collected. See #13383. _SESSION_EXPIRED_MARKERS: tuple = ( "invalid or expired session", "expired session", "session expired", "session not found", "unknown session", "session terminated", "closedresourceerror", "closed resource", diff --git a/tools/mcp_tool_handlers.py b/tools/mcp_tool_handlers.py index a799aa8b8f..24c2eb3fc6 100644 --- a/tools/mcp_tool_handlers.py +++ b/tools/mcp_tool_handlers.py @@ -162,7 +162,14 @@ def _handle_auth_error_and_retry(server_name: str, exc: BaseException, retry_cal def _handle_session_expired_and_retry(server_name: str, exc: BaseException, retry_call, op_description: str): """Transport reconnect + one retry on session expiry; None to fall through. Skips - ``handle_401``: the token is valid, only the server-side session is stale.""" + ``handle_401``: the token is valid, only the server-side session is stale. + + Unlike :func:`_handle_auth_error_and_retry`, this does **not** call the OAuth manager's ``handle_401`` — + the access token is still valid, only the server-side session state is stale. Setting + ``_reconnect_event`` causes the server task's lifecycle loop to tear down the current + ``streamablehttp_client`` + ``ClientSession`` and rebuild them, reusing the existing OAuth provider + instance. See #13383. + """ srv = _lookup_reconnectable_server(server_name, require_loop=True) if _is_session_expired_error(exc) else None if srv is None: return None @@ -182,7 +189,13 @@ class _StdioChildExited(RuntimeError): def _handle_stdio_child_exited_and_retry(server_name: str, exc: Exception, retry_call, op_description: str): """Respawn a dead stdio child and retry once; None if not our error. Never spawns itself: it sets ``_reconnect_event`` and waits, so spawn frequency stays governed by ``run()``'s - rapid-drop budget. Single-shot: a child that dies again reports and stops.""" + rapid-drop budget. Single-shot: a child that dies again reports and stops. + + Why retrying here cannot hot-cycle respawns: this function never spawns anything. It sets + ``_reconnect_event`` (one signal, same as before) and waits for the server task to publish a fresh + session. Spawn frequency stays governed entirely by ``run()``'s rapid-drop budget, which parks a + transport that keeps dropping without proving healthy (#62212). + """ if not isinstance(exc, _StdioChildExited): return None reconnected = False @@ -241,7 +254,13 @@ def _dispatch(server_name: str, server: Any, op: str, call, tool_timeout: float, async def _track_inflight_rpc(server: Any, server_name: str, op: str): """Register the running RPC so teardown can fail it fast. A deliberate teardown (``_reconnecting`` set first) turns the cancel into a retryable RuntimeError; external - cancels propagate unchanged. Doubles without ``_inflight_tasks`` skip tracking.""" + cancels propagate unchanged. Doubles without ``_inflight_tasks`` skip tracking. + + Every user-visible request family wraps its RPC in this context (#48069 salvage). If a deliberate + reconnect/shutdown teardown cancels the task (``_fail_inflight_calls`` sets ``_reconnecting`` first), + the cancel is converted into a clean retryable RuntimeError instead of a raw CancelledError; external + cancels (caller timeout, user interrupt) propagate unchanged. + """ inflight, task = getattr(server, "_inflight_tasks", None), asyncio.current_task() tracked = task is not None and inflight is not None if tracked: @@ -263,6 +282,8 @@ async def _call_tool_racing_stdio_death(server, server_name: str, tool_name: str child must not hold the slot for the full timeout) and mid-call (race against ``_watch_stdio_children``). Both raise :class:`_StdioChildExited` for the respawn path, which owns the reconnect signal. callable()/``is True`` because MagicMock attributes are truthy.""" + # Fast-fail (#81995): a stdio subprocess that is already dead must not own this call slot — fail + # immediately instead of waiting out the full tool timeout on a transport nobody will ever answer. _stdio_dead = getattr(server, "_stdio_children_dead", None) if callable(_stdio_dead) and _stdio_dead() is True: raise _StdioChildExited(f"MCP stdio subprocess for '{server_name}' had already exited when the call was dispatched") @@ -271,6 +292,8 @@ async def _call_tool_racing_stdio_death(server, server_name: str, tool_name: str if not (inspect.iscoroutinefunction(_watch_children) and asyncio.iscoroutine(_call_coro)): # Stubbed sessions return a non-awaitable, or there is no child-watcher to race: plain await. return await _call_coro if asyncio.iscoroutine(_call_coro) else _call_coro + # Fast-fail machinery (#81995): the RPC races a stdio-children watcher so a dead subprocess fails the + # call immediately instead of riding out the full tool timeout. rpc_task = asyncio.ensure_future(_call_coro) watch_task = asyncio.ensure_future(_watch_children()) try: @@ -302,6 +325,12 @@ def _render_content_blocks(result, server_name: str) -> Tuple[str, int]: (whitespace-only text and drop notices excluded) that the structuredContent arbitration uses.""" parts: List[str] = [] usable_parts = 0 + # MCP tool results can also include ImageContent blocks (screenshot / Blockbench / Playwright etc.); + # cache those via the gateway's image-cache helper so they flow through Hermes' MEDIA: tag convention + # and out to messaging adapters that render images natively. Without this, image blocks were silently + # dropped and the agent got an empty response. Distilled from #17915 (c3115644151) and #10848 + # (gnanirahulnutakki), both too stale to cherry-pick. #10848's approach (integrate with Hermes' MEDIA + # tag + cache_image_from_bytes) was the cleaner of the two — plugs into existing infrastructure. for block in (result.content or []): if getattr(block, "text", None): parts.append(strip_unicode_tags(block.text)) @@ -328,6 +357,23 @@ def _render_content_blocks(result, server_name: str) -> Tuple[str, int]: def _capped_structured_content(result): """``structuredContent`` (or None); over the hard cap it degrades to the head+tail truncated JSON string (multi-MB JSON flood guard).""" + # Hard-cap pathological payloads before they propagate (#56059); ordinary large results pass untouched + # to the spillover layer. + # content and structuredContent are ALTERNATIVES — never both forwarded (ported from + # MoonshotAI/kimi-code#3234). Spec-following servers already render their data into content (the + # verbatim dual-emit SHOULD, or a faithful human reorganisation), so forwarding both sent the same + # information to the model twice. content wins whenever it rendered anything usable; there is no + # reliable signal that the structured payload is richer than what the server put in content (semantic + # equality misses faithful reorganisations, size ratios misjudge both directions), so no heuristic is + # attempted. structuredContent fills in only when the content blocks rendered effectively empty, which + # keeps structuredContent-only servers working. Server-level `_meta` is also surfaced (ported from + # MoonshotAI/kimi-code#2596): servers return namespaced metadata there (validated contracts, + # browser-handoff payloads, ...) that was previously invisible to the agent. Protocol-reserved keys are + # dropped first (kimi-code#2600) — per the MCP spec's key-name rules a prefix is reserved when a + # `modelcontextprotocol` or `mcp` label is followed by at least one more label (e.g. + # `modelcontextprotocol.io/...`, `tools.mcp.com/...`); those carry host/protocol plumbing, not + # model-facing data. Unprefixed and vendor-namespaced keys (`com.example.mcp/...`) pass through — their + # semantics belong to the server. structured = mcp_field(result, "structured_content", "structuredContent") try: as_json = json.dumps(structured, ensure_ascii=False, default=str) if structured is not None else "" @@ -354,6 +400,9 @@ def _render_call_tool_result(result, server_name: str) -> str: return json.dumps({"result": text_result}, ensure_ascii=False) # Key order is part of the output: "result" leads when there is text, otherwise "_meta" precedes it. payload: Dict[str, Any] = {"result": text_result} if text_result else {} + # Cap structuredContent too — a malicious server could flood context via a multi-MB JSON payload + # (#56059). When the serialized form exceeds the hard cap, replace it with the truncated string (head + + # tail preserved) so it degrades gracefully instead of flooding downstream. if structured is not None: payload["structuredContent" if text_result else "result"] = structured if meta is not None: diff --git a/tools/mcp_tool_health.py b/tools/mcp_tool_health.py index 4a8af5753a..69f42c352b 100644 --- a/tools/mcp_tool_health.py +++ b/tools/mcp_tool_health.py @@ -71,7 +71,13 @@ class MCPServerHealthMixin: return task def _make_logging_callback(self): - """``logging_callback`` forwarding server ``notifications/message`` into Hermes logging (SDK default drops them).""" + """``logging_callback`` forwarding server ``notifications/message`` into Hermes logging (SDK default drops them). + + Routes MCP ``notifications/message`` log notifications from the server into Hermes' logging + (agent.log via hermes_logging), tagged with the server name. Without this, the SDK's default + callback silently discards them, so server-side warnings/errors during a tool call were invisible. + Port of anomalyco/opencode#34529. + """ async def _on_log(params): try: level = _core._MCP_LOG_LEVEL_MAP.get(str(getattr(params, "level", "info")).lower(), logging.INFO) @@ -187,7 +193,11 @@ class MCPServerHealthMixin: def _mark_session_proven(self) -> None: """Record that the session demonstrated real health (keepalive or tool-call success). Only then is the reconnect budget cleared: a handshake that drops moments later must keep - consuming ``_reconnect_retries`` so a flapping transport still reaches the park.""" + consuming ``_reconnect_retries`` so a flapping transport still reaches the park. + + Called from the keepalive success path (session survived at least one full keepalive interval) and + the tool-call success path. See #62212. + """ if self._session_proven: return self._session_proven = True @@ -200,7 +210,11 @@ class MCPServerHealthMixin: self._permanent_grace_used = self._teardown_race = False def mark_suspect(self, reason: str) -> None: - """Latch a suspicion (no I/O); the NEXT call verifies via :meth:`ensure_healthy` and recycles on failure.""" + """Latch a suspicion (no I/O); the NEXT call verifies via :meth:`ensure_healthy` and recycles on failure. + + The NEXT call verifies via :meth:`ensure_healthy` and recycles the transport if the probe fails, + instead of the connection silently staying poisoned until process restart (#81051/#77765/#84132). + """ if self._suspect_reason is None and reason: logger.warning("MCP server '%s': connection marked suspect (%s); next call will health-check it", self.name, reason) @@ -262,6 +276,10 @@ class MCPServerHealthMixin: return False async def _watch_stdio_children(self) -> None: - """Poll child liveness during a stdio RPC; resolves when a tracked child dies so the caller cancels the RPC.""" + """Poll child liveness during a stdio RPC; resolves when a tracked child dies so the caller cancels the RPC. + + See #81995. + """ while not self._stdio_children_dead(): + # Async context — never block the loop (#36163). await asyncio.sleep(0.25) diff --git a/tools/mcp_tool_lifecycle.py b/tools/mcp_tool_lifecycle.py index fc9ccbf482..7cd0defbd1 100644 --- a/tools/mcp_tool_lifecycle.py +++ b/tools/mcp_tool_lifecycle.py @@ -30,6 +30,11 @@ def _snapshot_child_pids() -> set: # loop thread, so union every task's children — reading only the main thread's file # returns an empty set on every Linux install. try: + # ``/proc/<pid>/task/<tid>/children`` is per-THREAD — a child forked from thread T is listed only + # under T's task dir. stdio_client() spawns from the background MCP loop thread, so reading only the + # main thread's file (``task/<pid>/children``) returned an empty set on every Linux install and left + # ``_stdio_child_pids`` / ``_stdio_pids`` empty: the #81995 dead-child fast-fail, the #96452 respawn + # signal, and the killpg shutdown sweep never saw the subprocess. task_dir = f"/proc/{my_pid}/task" found: set = set() for tid in os.listdir(task_dir): @@ -95,6 +100,10 @@ def shutdown_mcp_servers(*, scope: Optional[str] = None): selected = [name for name in _core._servers if scope is None or _core._server_scope_keys.get(name) == scope] servers_snapshot = [_core._servers[name] for name in selected] + # Fast path: nothing to shut down. The connect-cooldown maps can still be populated here — a server that + # failed to connect is never recorded in ``_servers`` (that is the very premise of the #50394 cooldown), + # so "no live servers" is the MOST likely state in which stale backoff entries exist. Clear them so a + # post-shutdown restart re-attempts every configured server immediately. if servers_snapshot: async def _shutdown(): results = await asyncio.gather(*(server.shutdown() for server in servers_snapshot), return_exceptions=True) @@ -155,6 +164,10 @@ def _signal_mcp_process(pid: int, sig: int, server_name: str, pgid: Optional[int # Child shares the gateway's pgroup: killpg would kill the gateway too, so use # per-pid kill. Warn because per-pid kill can't reach grandchildren in this group. logger.warning("MCP server '%s' pgid %d matches gateway pgid; skipping " + # Fall through to the per-pid kill() path instead. Warn because per-pid kill + # cannot reach grandchildren in this shared group — if the direct child has + # already exited, they may leak (inherent: group-killing them would also kill the + # gateway). See #47134. "killpg to avoid self-kill and using per-pid kill — any " "grandchildren in this group may not be reaped", server_name, pgid) else: diff --git a/tools/mcp_tool_loop.py b/tools/mcp_tool_loop.py index 11533e87e0..677e95caff 100644 --- a/tools/mcp_tool_loop.py +++ b/tools/mcp_tool_loop.py @@ -264,6 +264,9 @@ def _stop_mcp_loop(*, only_if_idle: bool = False) -> bool: # Drain before stopping: tasks still suspended when the loop closes get resumed by the GC # against a closed loop. shutdown_mcp_servers only reaps _servers; everything else ends here. future = None + # Drain before stopping: closing the loop with tasks still suspended leaves their coroutines for the GC, + # whose finalizer then resumes them to run cleanup against a loop that is already closed -> "Event loop + # is closed" (#60197). if loop.is_running(): from agent.async_utils import safe_schedule_threadsafe diff --git a/tools/mcp_tool_registration.py b/tools/mcp_tool_registration.py index 35af9bb289..a9f4a9af67 100644 --- a/tools/mcp_tool_registration.py +++ b/tools/mcp_tool_registration.py @@ -111,6 +111,12 @@ def _make_tool_filter(name: str, config: dict) -> Callable[[str], bool]: """Include/exclude predicate for a server's tool names: ``tools.include`` is a whitelist (``[]`` = register nothing), ``tools.exclude`` a blacklist; entries are exact names or fnmatch globs; include wins over exclude.""" tools_filter = config.get("tools") or {} + # Selective tool loading: honour include/exclude lists from config. Rules (matching issue #690 spec, + # extended with glob support): tools.include — whitelist: only matching tool names are registered + # tools.exclude — blacklist: all tools EXCEPT matching ones are registered entries may be exact names or + # fnmatch globs (e.g. "*_radar_*") include takes precedence over exclude include: [] → register nothing + # (an explicit empty whitelist, as written by the install checklist's "uncheck everything" path) Neither + # set → register all tools (backward-compatible default) include_raw = tools_filter.get("include") include_set = _normalize_name_filter(include_raw, f"mcp_servers.{name}.tools.include") exclude_set = _normalize_name_filter(tools_filter.get("exclude"), f"mcp_servers.{name}.tools.exclude") @@ -185,6 +191,14 @@ def _resolve_name_collisions(name: str, candidates: List[_Candidate]) -> List[_C unique.append(c) ambiguous: Dict[str, List[str]] = {} shadowed: set[tuple[str, str]] = set() + # A generated resource/prompt utility that normalizes onto a server-native tool's name must not knock + # that native tool out of the registry: the native tool is the capability the user connected the server + # for, while the generated utility (read_resource/list_resources/list_prompts/get_prompt) is optional + # sugar that only matters when the server exposes no such tool of its own (#87112). Resolve that + # specific collision in favour of the native tool — keep it, drop the shadowed utility — and fall back + # to the conservative skip-everything only for genuinely ambiguous collisions (two or more native tools + # normalizing to one name, which we cannot disambiguate). The four utility keys are distinct, so a + # colliding set holds at most one utility origin. for registry_name, origins in origins_by_name.items(): if len(origins) <= 1: continue @@ -243,6 +257,8 @@ def _register_candidates(name: str, candidates: List[_Candidate], *, check_fn: C def _write_schema_cache(name: str, server: "MCPServerTask", config: dict, should_register) -> None: """Write-through: persist the manifest so the next startup registers this server lazily (no spawn). Never raises.""" try: + # Write-through (#56832): refresh the on-disk schema cache after a live connect so the next startup + # can lazily register this server without spawning it. Cache failures never break registration. from tools.mcp_schema_cache import config_fingerprint, write_cache_entry tools_payload = [] for t in server._tools: @@ -287,7 +303,11 @@ def _register_server_tools(name: str, server: "MCPServerTask", config: dict) -> def _register_from_cache_sync(name: str, config: dict, entry: dict) -> List[str]: """Lazy startup: register from a cached manifest with no child process (first real call goes through ``_ensure_lazy_server_connected``). Trust metadata is recorded first so the - call-time gate is identical for live and cached registrations.""" + call-time gate is identical for live and cached registrations. + + Lazy startup (#56832, design by Vansh5632): tools appear in the registry immediately; the first real + call routes through ``_get_connected_server_for_call`` → ``_ensure_lazy_server_connected``. + """ from tools.mcp_schema_cache import config_fingerprint, tools_from_cache_entry, utility_tools_from_cache_entry tool_timeout = _resolve_tool_timeout(config) cached_tools = _cached_tools(tools_from_cache_entry(entry)) diff --git a/tools/mcp_tool_schema.py b/tools/mcp_tool_schema.py index be7b6040cf..dba5de056d 100644 --- a/tools/mcp_tool_schema.py +++ b/tools/mcp_tool_schema.py @@ -91,13 +91,21 @@ def _repair_object_shape(node): return repaired +# Lazy (schema-cache registered) servers are available: the first real call spawns/connects them (#56832). def _normalize_mcp_input_schema(schema: dict | None) -> dict: """Normalize MCP input schemas so one form is valid on OpenAI, Anthropic, Gemini and Moonshot. Order matters: ``definitions`` -> ``$defs``; nullable ``anyOf`` unions collapsed to the non-null branch (Anthropic rejects nullable branches; optionality lives in the parent's ``required``; the ``nullable: true`` hint is kept so runtime coercion can map a model-emitted ``"null"`` string to ``None``); same-typed const unions -> enum (AFTER the - nullable strip); then object-shape repair.""" + nullable strip); then object-shape repair. + + * Missing or ``null`` ``type`` on an object-shaped node is coerced to ``"object"`` (some servers omit + it). See PR #4897. * When an ``object`` node lacks ``properties``, an empty ``properties`` dict is added + so ``required`` entries don't dangle. * ``required`` arrays are pruned to only names that exist in + ``properties``; otherwise Google AI Studio / Gemini 400s with ``property is not defined``. See PR #4651. + * MCP/Pydantic optional fields commonly arrive as ``anyOf: [{...}, {"type": "null"}], default: null``. + """ if not schema: return dict(_EMPTY_OBJECT_SCHEMA) from tools.schema_sanitizer import collapse_const_unions, strip_nullable_unions @@ -121,6 +129,9 @@ def sanitize_mcp_name_component(value: str) -> str: # ``mcp__<server>__<tool>``: the convention shared by Claude Code, Codex and OpenCode. The # double underscore disambiguates the server/tool boundary even when either contains # underscores, and matches the Anthropic-OAuth wire form. +# Native MCP tool-name prefix. It also aligns native registration with the Anthropic-OAuth wire form +# (``_MCP_TOOL_PREFIX`` in anthropic_adapter.py), removing the single->double rewrite that path previously +# had to perform. See #33533. MCP_TOOL_NAME_PREFIX = "mcp__" @@ -201,6 +212,11 @@ def matches_name_filter(tool_name: str, patterns: set[str]) -> bool: # response for the handler to be registered. Without this gate a tools-only server got all # four stubs and every call returned JSON-RPC -32601, making the model conclude the server # was broken. +# Source of truth: MCP spec — capabilities.resources / capabilities.prompts are present on the response only +# when the server actually implements those request families. Context7 @upstash/context7-mcp, which +# advertises only ``tools``) had all four utility stubs registered and every model call to them came back +# with JSON-RPC ``-32601 Method not found``, which made the model conclude the server was broken even when +# the real tools worked. See #18051. _UTILITY_CAPABILITY_ATTRS = { "list_resources": "resources", "read_resource": "resources", "list_prompts": "prompts", "get_prompt": "prompts"} diff --git a/tools/mcp_tool_server_run.py b/tools/mcp_tool_server_run.py index 6559be2ffe..9c2f0fd3cf 100644 --- a/tools/mcp_tool_server_run.py +++ b/tools/mcp_tool_server_run.py @@ -52,7 +52,12 @@ class MCPServerRunMixin: down, transport re-entered; event cleared first) or ``"recycle"`` (stdio idle/lifetime limit; restarts lazily on next call). Shutdown wins a tie. A keepalive (``ping``, list_tools fallback) runs every ``keepalive_interval`` (must stay below the server's - session TTL); a failure triggers a reconnect.""" + session TTL); a failure triggers a reconnect. + + Periodically sends a lightweight keepalive (``ping``, with a ``list_tools`` fallback for servers + that don't implement the optional ping utility — see :meth:`_keepalive_probe`) to prevent + TCP/session state from going stale during idle periods (#17003). + """ keepalive_interval = max( _core._MIN_KEEPALIVE_INTERVAL, float(self._config.get("keepalive_interval", _core._DEFAULT_KEEPALIVE_INTERVAL))) @@ -73,6 +78,7 @@ class MCPServerRunMixin: return "recycle" # Timeout: probe for a stale session — NEVER while an RPC is in flight (a # concurrent ping can wedge the stdio stream; a busy server is alive anyway). + # Timeout — no lifecycle event fired. See #48069. if self.session: if self._rpc_lock.locked() or any(not t.done() for t in self._inflight_tasks): continue @@ -87,6 +93,7 @@ class MCPServerRunMixin: self._reconnect_event.set() break # Survived a full keepalive interval: real proof of health. + # Clear the rapid-drop budget (#62212). self._mark_session_proven() finally: await self._cancel_waiters(shutdown_task, reconnect_task) @@ -94,6 +101,7 @@ class MCPServerRunMixin: self._fail_inflight_calls("shutdown") return "shutdown" # Deliberate teardown: fail in-flight RPCs NOW instead of riding out the tool timeout. + # See #48069, #81995. self._fail_inflight_calls("reconnect") self._reconnect_event.clear() return "reconnect" @@ -117,6 +125,13 @@ class MCPServerRunMixin: leaves the server unrevivable). With tools deregistered no call can reach the breaker probe, so the wait is TIMED (one self-probe per ``_PARKED_RETRY_INTERVAL``); an explicit ``_reconnect_event.set()`` wakes it immediately.""" + # Do NOT return — exiting the task orphans the server: nothing would ever listen for + # _reconnect_event again and the server would be permanently wedged for the life of the process + # (#16788). Instead, drop the phantom tools from the registry and park. Because parking deregisters + # the tools, no tool call can reach the circuit-breaker half-open probe or _signal_reconnect — so + # the park is a TIMED wait: every _PARKED_RETRY_INTERVAL we wake and attempt one reconnect ourselves + # (#57129). An explicit _reconnect_event.set() (OAuth recovery, manual /mcp refresh) still wakes us + # immediately. self._was_parked = True self._deregister_tools() self._reconnect_event.clear() @@ -188,6 +203,11 @@ class MCPServerRunMixin: break except asyncio.CancelledError: # Not a connection failure: re-raise so shutdown()'s ``await self._task`` completes. + # Task was cancelled (shutdown, gateway restart, explicit task.cancel()). Don't treat this + # as a connection failure — CancelledError inherits from BaseException (not Exception) in + # Python 3.11+, so the broad ``except Exception`` below would NOT catch it; we'd silently + # exit the reconnect loop and the MCP server would stay dead until Hermes is fully + # restarted. See #9930. self.session = None raise except Exception as exc: @@ -214,6 +234,12 @@ class MCPServerRunMixin: logger.debug("MCP server '%s': reconnecting (OAuth recovery or manual refresh)", self.name) # A clean return is NOT proof of health (a flapper handshakes fine, then drops). Only a # PROVEN session clears the budget; a teardown race is recovery, never a park charge. + # A clean transport return means a session was established and then asked to rebuild (auth recovery + # / manual refresh / keepalive failure / transport TaskGroup drop). That alone is NOT proof of + # health: a flapping transport handshakes fine and drops moments later, and resetting the budget + # here let such servers respawn forever (#62212 — 6212 spawns in 63h). Only clear the + # consecutive-failure budget once the session PROVED healthy — survived >=1 full keepalive interval + # or served >=1 successful tool call (_mark_session_proven). if self._teardown_race and not self._session_proven: logger.info("MCP server '%s': reconnect after teardown race (in-flight calls were failed); " "not charging the rapid-drop budget", self.name) @@ -268,6 +294,12 @@ class MCPServerRunMixin: self._recycled_reason = None # Initial-connect ladder (a startup blip must not kill the server); gated on # _ever_connected, not _ready (which clears every reconnect cycle). + # If this is the first connection attempt, retry with backoff before giving up. Gated on + # ``_ever_connected`` rather than ``_ready`` — ``_ready`` is cleared on every reconnect cycle (see + # below), so a server that already registered tools once and then dropped would otherwise be + # misclassified as never having connected and re-enter this initial-connect ladder (#94654). + # ``_ever_connected`` itself is set once and never cleared. (Ported from Kilo Code's MCP resilience + # fix.) if not self._ever_connected: return await self._on_initial_connect_error(exc, root, failure_class, budget) if self._shutdown_event.is_set(): diff --git a/tools/mcp_tool_transport.py b/tools/mcp_tool_transport.py index 0816e70063..cd856527a8 100644 --- a/tools/mcp_tool_transport.py +++ b/tools/mcp_tool_transport.py @@ -48,7 +48,13 @@ class MCPServerTransportMixin: def _advertises_tools(self) -> bool: """False only when captured capabilities omit ``tools`` (prompt-/resource-only servers, - where ``tools/list`` raises -32601); True without capability info (legacy fallback).""" + where ``tools/list`` raises -32601); True without capability info (legacy fallback). + + Per the MCP spec, ``InitializeResult.capabilities.tools`` is non-None iff the server implements the + ``tools/*`` request family. Prompt-only or resource-only servers omit it, and calling ``tools/list`` + against them raises ``MCPError(-32601 Method not found)`` — which previously killed the connection + during discovery and made every keepalive fail. (Ported from anomalyco/opencode#31271.) + """ caps = getattr(self.initialize_result, "capabilities", None) return caps is None or getattr(caps, "tools", None) is not None @@ -109,6 +115,20 @@ class MCPServerTransportMixin: self._ready.set() self._ever_connected = True _core._reset_server_error(self.name) + # Session is live again: clear any breaker state from a prior outage so the first call after + # recovery isn't gated on a stale consecutive-failure count (#16788). + # A completed handshake alone is NOT proof of health: a flapping transport can handshake fine and + # drop moments later, forever (#62212). The session must prove itself (keepalive success or a + # successful tool call) before the reconnect budget is cleared — see _mark_session_proven. + # Session is live again: clear any breaker state from a prior outage so the first call after + # recovery isn't gated on a stale consecutive-failure count (#16788). + # Unproven until keepalive/tool-call success (#62212). + # Session is live again: clear any breaker state from a prior outage so the first call after + # recovery isn't gated on a stale failure count (#16788). + # Unproven until keepalive/tool-call success (#62212). + # Session is live again: clear any breaker state from a prior outage so the first call after + # recovery isn't gated on a stale consecutive-failure count (#16788). + # Unproven until keepalive/tool-call success (#62212). self._session_proven = False reason = await self._wait_for_lifecycle_event() if label and reason == "reconnect": @@ -201,6 +221,13 @@ class MCPServerTransportMixin: # off-loop because the reaper blocks up to 2s. await asyncio.to_thread(_core._kill_orphaned_mcp_children) pids_before = _core._snapshot_child_pids() # so the new child can be identified after spawn + # Reap any orphaned subprocesses from prior failed connection attempts before spawning a new one. + # Without this, each retry in the run() reconnect loop spawns a fresh process pair while the + # previous failed pair lingers — leading to rapid zombie accumulation (see #57355, #57228). The + # unscoped sweep also opportunistically reaps orphans left by *other* servers that never reconnect; + # per-server filtering via ``server_name`` remains available for scoped call sites. Run in a worker + # thread: the reaper blocks up to 2s (SIGTERM → wait → SIGKILL) when orphans exist, which would + # otherwise stall the shared MCP event loop. new_pids: set = set() # Subprocess stderr goes to ~/.hermes/logs/mcp-stderr.log so banners can't corrupt the TUI. _core._write_stderr_log_header(self.name) @@ -271,7 +298,20 @@ class MCPServerTransportMixin: TaskGroup, so a transient drop escapes as a ``BaseExceptionGroup`` that would otherwise park the server for 300s over a sub-second glitch. Re-raise when it is not one: shutdown in progress (``_shutdown_event`` is set before cancel), KeyboardInterrupt/SystemExit or a real CancelledError in the group, or no live session - this attempt (``_ready`` unset — connect failures must back off, not hot-loop).""" + this attempt (``_ready`` unset — connect failures must back off, not hot-loop). + + Streamable-HTTP / SSE transports run their stream pump inside an anyio TaskGroup. A transient stream + drop (idle timeout, brief backend blip, server-side TCP close) surfaces as a ``BaseExceptionGroup`` + escaping the transport context manager. Left unwrapped it reaches ``run()``'s error path, which + applies exponential backoff and eventually *parks* the server for 300s and deregisters its tools — a + multi-minute tool outage for what is usually a sub-second glitch while the POST path stays healthy + (issue #66092). + - the group carries a ``KeyboardInterrupt`` / ``SystemExit`` — fatal signals must propagate to the + interpreter, never be converted into a reconnect; - the group carries a real ``CancelledError`` + (task cancellation must propagate to asyncio, mirroring the ``run()`` guard for #9930); - we never + reached a live session this attempt (``_ready`` unset) — a connect/handshake failure SHOULD fall + through to ``run()``'s backoff rather than hot-loop reconnects against a broken endpoint. + """ if (self._shutdown_event.is_set() or eg.split((KeyboardInterrupt, SystemExit))[0] is not None or eg.split(asyncio.CancelledError)[0] is not None @@ -375,9 +415,15 @@ class MCPServerTransportMixin: # -------------------------------------------------------------- discovery + # Legacy Streamable-HTTP transport TaskGroup dropped: reconnect immediately instead of backoff/park + # (#66092). async def _discover_tools(self): """Discover tools from the connected session. Capability-gated: prompt-/resource-only - servers raise ``MCPError(-32601)`` on ``tools/list``, which would abort the connection.""" + servers raise ``MCPError(-32601)`` on ``tools/list``, which would abort the connection. + + Skip the call when the server doesn't advertise the ``tools`` capability. (Ported from + anomalyco/opencode#31271.) + """ self._ping_unsupported = False # fresh transport: re-probe ``ping`` across the reconnect if self.session is None: return diff --git a/tools/memory_tool_store.py b/tools/memory_tool_store.py index 6abd008dd0..dcb04ef364 100644 --- a/tools/memory_tool_store.py +++ b/tools/memory_tool_store.py @@ -71,6 +71,7 @@ class MemoryStore: # Failed consolidation attempts (overflow / zero-match) allowed per turn before # a TERMINAL "save skipped" result, so a fragile replace/add can't loop the turn # to budget exhaustion and suppress the user's reply. + # See #42405. _MAX_CONSOLIDATION_FAILURES_PER_TURN = 3 def __init__(self, memory_char_limit: int = 2200, user_char_limit: int = 1375, *, @@ -82,6 +83,8 @@ class MemoryStore: self._system_prompt_snapshot: Dict[str, str] = {"memory": "", "user": ""} self._consolidation_failures = 0 # per turn; reset by reset_consolidation_failures() + # Per-turn counter of failed at-capacity consolidation attempts; reset at each turn boundary by + # reset_consolidation_failures() (#42405). def target_enabled(self, target: str) -> bool: return self.user_profile_enabled if target == "user" else self.memory_enabled @@ -91,7 +94,12 @@ class MemoryStore: def _consolidation_failure(self, response: Dict[str, Any]) -> Dict[str, Any]: """Count a consolidation failure: under the per-turn cap return ``response`` - (it says how to retry); past it a TERMINAL result so the model stops looping.""" + (it says how to retry); past it a TERMINAL result so the model stops looping. + + Once the cap is exceeded, drop the retry instruction and return a TERMINAL result so the model stops + looping memory calls and proceeds to answer the user — a failed memory side effect must never block + the turn's reply (#42405). + """ self._consolidation_failures += 1 if self._consolidation_failures <= self._MAX_CONSOLIDATION_FAILURES_PER_TURN: return response @@ -327,6 +335,8 @@ class MemoryStore: """TERMINAL and WITHOUT the entries list: echoing entries invites the model to "find more to fix" and re-issue the same ops. A successful write resets the per-turn failure budget.""" + # A successful write means the consolidation loop made progress, so the per-turn failure budget + # resets (the cap counts consecutive failures, not lifetime ones within a turn) (#42405). self._consolidation_failures = 0 return {"success": True, "done": True, "target": target, "usage": self._usage_pct(target, self._char_count(target)), @@ -349,6 +359,12 @@ class MemoryStore: if not path.exists(): return "", True try: + # utf-8-sig strips a leading UTF-8 BOM (Notepad-edited memory files on Windows) and is + # byte-identical to utf-8 otherwise. Plain utf-8 kept U+FEFF glued to the first entry, + # corrupting matching/dedup for that entry forever (#10878 / PR #10888). Decode errors stay + # STRICT on purpose: errors="replace" would hand read-modify-write callers a lossy view that a + # subsequent save persists over the real bytes — the wipe class documented above. Undecodable + # bytes must surface as read_ok=False. return path.read_text(encoding="utf-8-sig"), True except (OSError, UnicodeDecodeError): return "", False diff --git a/tools/osv_check.py b/tools/osv_check.py index 5c87fadaab..e1e827ca3d 100644 --- a/tools/osv_check.py +++ b/tools/osv_check.py @@ -28,6 +28,10 @@ _TIMEOUT = 10 # seconds # restarts reuse warm verdicts; expiry is absolute wall-clock time so it survives restarts. # Trade-off: a MAL advisory published right after a clean verdict is noticed at TTL expiry # (<= 1h by default) rather than at next process start — lower OSV_CHECK_CACHE_TTL to tighten. +# Without a cache, a flapping server turns into a sustained OSV query/DNS stream — the #75485 incident +# logged 779K api.osv.dev DNS queries in 16h from revival loops. Malware advisories don't appear or vanish +# on second-to-second timescales, so a successful verdict (clean OR blocked) is reusable. The window is the +# same one the in-process cache already accepted; it just now spans restarts. _CACHE_TTL_S = float(os.getenv("OSV_CHECK_CACHE_TTL", "3600")) _CACHE_MAX_ENTRIES = 256 _cache: dict = {} diff --git a/tools/process_registry.py b/tools/process_registry.py index a55b131d8d..b690fbb05d 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -21,6 +21,7 @@ from pathlib import Path _IS_WINDOWS = platform.system() == "Windows" # systemd transient scopes exist only on Linux; gate every scope-path branch on this # (not merely "not Windows") so macOS and other POSIX platforms never touch systemd. +# See #70716. _IS_LINUX = platform.system() == "Linux" from tools.environments.local import _find_shell, _resolve_safe_cwd, _sanitize_subprocess_env from hermes_cli._subprocess_compat import windows_hide_flags @@ -49,6 +50,13 @@ WATCH_STRIKE_LIMIT = 3 # Lifetime cap, independent of strikes: a pattern recurring just above the cooldown never # strikes yet forces a full-context agent turn each time; watch_patterns is "ONLY for # rare one-shot signals", so after this many deliveries fall back to notify_on_complete. +# MAX_ACTIVE_PROCESS_AGE = 86400 # 24h default — see session_reset.bg_process_max_age_hours (#29177) +# A process whose pattern recurs at a cadence just above WATCH_MIN_INTERVAL_SECONDS (e.g. a service +# restarted repeatedly over a day) never trips the consecutive-strike limit, since each match lands in its +# own clean cooldown window, yet still forces a full-context agent turn every single time (#93513). +# watch_patterns is documented as "ONLY for rare one-shot mid-process signals", so once a session has +# delivered this many matches over its whole life we disable it and fall back to notify_on_complete, same as +# the strike-limit path. WATCH_LIFETIME_MAX_HITS = 8 # Global circuit breaker across all sessions so concurrent siblings can't collectively # flood the user even when each is under its own cap. @@ -62,6 +70,11 @@ WATCH_GLOBAL_COOLDOWN_SECONDS = 30 # cgroup, so a memory-heavy executor can get the ENTIRE gateway killed by systemd-oomd; # ``systemd-run --user --scope`` gives the worker its own transient cgroup. Usability is # probed once (binary present but user D-Bus absent in system services/containers). +# A memory-heavy executor (Codex, tests, Node) can push the whole cgroup past MemoryMax and trigger +# systemd-oomd to kill the ENTIRE gateway — taking down the messaging control plane and silently losing the +# active turn. We probe *once* whether ``systemd-run --user --scope`` is actually usable (the binary can +# exist on the PATH while the user D-Bus session is unavailable — common for system services and +# containers), and cache the result for the process lifetime. See #70716. _SYSTEMD_SCOPE_AVAILABLE: Optional[bool] = None _SYSTEMD_SCOPE_PROBE_LOCK = threading.Lock() _SYSTEMD_SCOPE_PROBED_AT = 0.0 @@ -75,7 +88,11 @@ def _worker_memory_max_bytes() -> int: """Finite per-worker cgroup limit that can never widen host risk. ``TERMINAL_LOCAL_MEMORY_MAX_MB`` is honored only when it *tightens* the safe bound (min of the gateway's cgroup-v2 ``memory.max`` and half of physical RAM, - capped at 4 GiB), so an oversized override cannot exceed the enclosing slice.""" + capped at 4 GiB), so an oversized override cannot exceed the enclosing slice. + + The proposed local-memory-guard environment override is honored when it tightens the safe bound, so this + isolation composes with PR #57121 instead of inventing a second knob. + """ override_bound: Optional[int] = None override = os.getenv("TERMINAL_LOCAL_MEMORY_MAX_MB", "").strip() if override: @@ -186,7 +203,11 @@ def _is_supervised_gateway_process() -> bool: def _build_systemd_scope_argv(shell_argv: List[str], unit_suffix: str) -> List[str]: """Wrap *shell_argv* in a ``systemd-run --user --scope`` invocation with its own - memory accounting, so an OOM in the worker cannot kill the gateway cgroup.""" + memory accounting, so an OOM in the worker cannot kill the gateway cgroup. + + ``--collect`` makes the transient scope self-clean after exit; ``--unit`` gives it a recognisable name + for ``systemctl --user status`` / journalctl. See #70716. + """ import shutil binary = shutil.which("systemd-run") @@ -230,7 +251,10 @@ def _stop_systemd_unit(unit_name: str) -> bool: Reaps the *entire* cgroup — catching double-forked descendants reparented to init inside the scope that survive a plain PID signal (SIGTERM all, SIGKILL after ``TimeoutStopSec``). True if stopped or already gone; False if ``systemctl`` is - unavailable or the stop failed.""" + unavailable or the stop failed. + + See #70716. + """ import shutil binary = shutil.which("systemctl") @@ -299,6 +323,8 @@ class ProcessSession: pid_scope: str = "host" # "host" for local/PTY PIDs, "sandbox" for env-local PIDs systemd_unit: str = "" # transient scope unit name when spawned under systemd-run # Watcher/notification routing (persisted for crash recovery) + # systemd_unit: str = "" # transient scope unit name when spawned under systemd-run + # (#70716) watcher_platform: str = "" watcher_chat_id: str = "" watcher_user_id: str = "" @@ -390,6 +416,7 @@ class ProcessRegistry: # turn), but the CLI has the poll result inline in the same turn, so # drain_notifications() skips these to avoid a duplicate [SYSTEM: ...]; # gateway/tui watchers deliberately ignore this set. + # See #8228. self._poll_observed: set = set() # Global watch-match circuit breaker across all sessions. self._global_watch_lock = threading.Lock() @@ -734,6 +761,7 @@ class ProcessRegistry: the supervised gateway (own cgroup: an OOM kills only the worker, not the gateway and its messaging control plane).""" argv = [_find_shell(), "-lic", f"set +m; {safe_command}"] + # This applies to both pipe mode and the PTY path above. See #70716. in_supervised_gateway = _IS_LINUX and _is_supervised_gateway_process() if in_supervised_gateway and _systemd_run_user_scope_available(): session.systemd_unit = f"hermes-worker-{unit_suffix}.scope" @@ -794,6 +822,7 @@ class ProcessRegistry: # Bash parses ``A && B &`` as ``(A && B) &`` — a subshell that holds our stdout # pipe open forever when B is a long-running server. The rewriter turns it into # ``A && { B & }``. Lazy import: terminal_tool imports this module. + # Guard against the `A && B &` subshell-wait trap (issue #68915). from tools.terminal_tool import _rewrite_compound_background as _rewrite_bg safe_command = _rewrite_bg(command) @@ -845,6 +874,9 @@ class ProcessRegistry: # Scope teardown is the authoritative cleanup for the worker cgroup # (never killpg here); the wrapper PID is terminated as fallback. _stop_systemd_unit(session.systemd_unit) + # The worker runs in its own systemd scope and, since the #70716 session-isolation fix, its + # own session. Stop the scope (kills every process in the worker cgroup), then terminate the + # systemd-run wrapper PID as fallback. self._terminate_host_pid(proc.pid, session.host_start_time) elif not _IS_WINDOWS: try: @@ -903,12 +935,21 @@ class ProcessRegistry: end so EOF never arrives while it lives, which would park this thread and never fire ``notify_on_complete``; on POSIX we ``select()`` and stop draining shortly after the direct child exits (mirrors ``environments/base.py::_wait_for_process``). - Windows pipes lack select(), so the lazy ``_reconcile_local_exit`` is the net.""" + Windows pipes lack select(), so the lazy ``_reconcile_local_exit`` is the net. + + Windows pipes don't support select(); the blocking path is kept there and the lazy reconcile in + poll()/wait() remains the safety net. See #68915, #8340. + """ first_chunk = True # A split multibyte UTF-8 char would become U+FFFD with stateless decoding; the # incremental decoder holds the partial sequence until the rest arrives. decoder = codecs.getincrementaldecoder("utf-8")(errors="replace") + # Incremental decoder: raw pipe reads can split a multibyte UTF-8 character across two read1() + # chunks. A stateless per-chunk ``bytes.decode(errors="replace")`` turns both halves into U+FFFD + # mojibake. The incremental decoder holds the partial sequence until the continuation bytes arrive — + # same treatment the foreground path already has in + # ``tools/environments/base.py::_wait_for_process``. (Ported from openclaw/openclaw#112325.) def _append_chunk(chunk: str): nonlocal first_chunk if first_chunk: @@ -950,6 +991,7 @@ class ProcessRegistry: # buffered tail, then stop rather than wait forever on an orphaned # grandchild's pipe. if proc.poll() is not None: + # See #68915. idle_after_exit += 1 if idle_after_exit >= 3: break @@ -1073,6 +1115,8 @@ class ProcessRegistry: """Background thread: read output from a PTY process.""" pty = session._pty # Same split-multibyte handling as _reader_loop. + # PTY reads can split a multibyte UTF-8 character across chunks just like pipe reads — hold partial + # sequences until the rest arrives. (Ported from openclaw/openclaw#112325.) decoder = codecs.getincrementaldecoder("utf-8")(errors="replace") try: while pty.isalive(): @@ -1167,7 +1211,14 @@ class ProcessRegistry: watchers aren't the parent's to wait for. ``task_id=None`` waits on every tracked process; ``timeout=None`` reads ``terminal.oneshot_completion_wait_seconds`` (``<= 0`` disables). Each pass re-reconciles child state so an orphaned-pipe exit can't wedge - the linger. Returns ``{"waited", "completed", "timed_out"}`` id lists.""" + the linger. Returns ``{"waited", "completed", "timed_out"}`` id lists. + + Bot Mode handoff REPLIES are the visible casualty (#90879): a recipient invoked as ``hermes -p <bot> + chat -Q --query-file ...`` dispatches its reply via ``message_agent`` / ``bot_relay`` exactly this + way, then exits, and the reply process is destroyed ~3s later. The sender waits forever for a reply + that was already killed. + See #17327. + """ if timeout is None: timeout = self._oneshot_completion_wait_seconds() result: dict = {"waited": [], "completed": [], "timed_out": []} @@ -1201,6 +1252,14 @@ class ProcessRegistry: break # Reconcile first so orphaned-pipe and detached exits fire the event. with suppress(Exception): + # Reconcile first: catches direct-child exits whose reader is blocked on a pipe held + # open by a descendant (#17327) and detached/env sessions, so the event actually + # fires. + # Reconcile against real child state before reading session.exited. Guards against + # orphaned-pipe reader hangs (issue #17327). + # Reconcile against real child state — guards against orphaned- pipe reader hangs + # where the reader is blocked but the direct child has already exited (issue + # #17327). self._reconcile_local_exit(session) self._refresh_detached_session(session) if session.exited: @@ -1227,7 +1286,12 @@ class ProcessRegistry: def _drain_should_skip(self, session_id: str, *, skip_poll_observed: bool = True) -> bool: """Skip a completion the CLI agent already has this turn — consumed via wait/log or observed inline via poll(). Gateway/tui watchers check only - ``is_completion_consumed`` so a read-only poll never suppresses their turn.""" + ``is_completion_consumed`` so a read-only poll never suppresses their turn. + + Skips when the agent has either truly consumed the output (wait/log → ``_completion_consumed``) or + observed the exit inline via poll() (``_poll_observed``). In both cases the CLI agent already has + the result this turn, so injecting a [SYSTEM: ...] completion would be a duplicate (#8228). + """ return session_id in self._completion_consumed or (skip_poll_observed and session_id in self._poll_observed) @staticmethod @@ -1342,7 +1406,14 @@ class ProcessRegistry: descendant (e.g. a daemon from ``hermes update``) holds the pipe open, poll() would report "running" forever. If ``Popen.poll()`` has an exit code, drain readable bytes non-blocking and flip ``exited``. No-op for env/PTY, exited and - detached sessions.""" + detached sessions. + + The reader thread (`_reader_loop`) sets `session.exited = True` only in its `finally` block, which + runs when `stdout.read()` returns EOF. If the direct `Popen` child has exited but a descendant + process (e.g. a daemon spawned by `hermes update` restarting the gateway) is still holding the + stdout pipe open, the reader blocks forever and poll() keeps returning "running" indefinitely (issue + #17327 — 74 polls over 7 minutes on Feishu). + """ if session is None or session.exited: return proc = getattr(session, "process", None) @@ -1418,6 +1489,9 @@ class ProcessRegistry: total_lines = len(lines) # offset=None -> last N lines; an explicit offset=0 means the HEAD (don't # conflate the two via falsiness). + # An explicit offset=0 means "start from the first line" — previously it was conflated with the + # default and silently returned the TAIL instead of the head (same falsy-coercion class as the + # wait() timeout guard; salvaged from PR #60004, credit @isheng-eqi). if offset is None and limit > 0: selected = lines[-limit:] observed_completion_output = bool(selected) or total_lines == 0 @@ -1514,6 +1588,12 @@ class ProcessRegistry: if session.exited: # A double-forked descendant may still be alive in the systemd scope even # though the main process exited — stop the scope to reap survivors. + # See #70716. + # If the worker was spawned in its own systemd scope (#70716), stop the entire unit to reap any + # double-forked descendants that were reparented inside the scope and survived the PID signal + # above (reviewer gap #2). ``systemctl --user stop`` sends SIGTERM to every process in the + # cgroup and escalates to SIGKILL after TimeoutStopSec. This is additive — the PID-based kill + # above already handled the main process; this catches stragglers. if session.systemd_unit: _stop_systemd_unit(session.systemd_unit) with session._lock: @@ -1569,6 +1649,9 @@ class ProcessRegistry: # Identity check, not bare liveness: a gone/recycled PID means our # process exited — never tree-kill the stranger. Still stop an owned # scope: a daemonized descendant may survive the wrapper PID. + # If this recovered session also carries an owned systemd scope, stop that scope before + # returning: a daemonized descendant may still be alive there even though the wrapper PID exited + # or was recycled across the gateway restart (#70716, teknium1 review). if not self._host_pid_is_ours(session.pid, session.host_start_time): if session.systemd_unit: _stop_systemd_unit(session.systemd_unit) @@ -1583,6 +1666,10 @@ class ProcessRegistry: self._terminate_host_pid(session.pid, session.host_start_time) else: return { + # Reject non-positive timeouts — the schema declares minimum=1, but not every caller + # enforces schemas before dispatch. timeout=0 is falsy, so without this guard it silently + # fell through (`0 or max_timeout`) to the DEFAULT wait instead of erroring. Salvaged from + # PR #60004 (credit @isheng-eqi). "status": "error", "error": "Recovered process cannot be killed after restart because " "its original runtime handle is no longer available", @@ -1666,7 +1753,13 @@ class ProcessRegistry: def list_sessions(self, task_id: str = None, session_key: str = None) -> list: """Running and recently-finished processes for ``task_id`` and/or ``session_key``; cross-task entries sharing the gateway session (a forgotten preview server - blocking session reset) are flagged ``"session_scoped": true``.""" + blocking session reset) are flagged ``"session_scoped": true``. + + When ``task_id`` is given, processes for that task are included. When ``session_key`` is also given, + session-scoped background processes (``background: true``) registered under that gateway session are + surfaced too, even if they belong to a different task — so the agent can discover a forgotten + preview server that is blocking session reset (#29177). + """ with self._lock: all_sessions = list(self._running.values()) + list(self._finished.values()) all_sessions = [self._refresh_detached_session(s) for s in all_sessions] @@ -1687,6 +1780,8 @@ class ProcessRegistry: "status": "exited" if s.exited else "running", "output_preview": s.output_buffer[-200:] if s.output_buffer else "", } + # Flag processes surfaced only because they share the gateway session (not the current task) — + # these are the long-lived background processes a user may have forgotten about (#29177). if task_id and session_key and s.task_id != task_id and s.session_key == session_key: entry["session_scoped"] = True # Trigger metadata for goal-loop judges (a watcher may never exit). @@ -1795,6 +1890,7 @@ class ProcessRegistry: # Redact inline credentials before persisting (~/.hermes/processes.json). # Recovery uses command only for display (adoption re-validates the # PID, never re-runs it), so masking is lossless. + # See #77484. entry["command"] = redact_sensitive_text(s.command, code_file=True) entry["owner_task_id"] = s.owner_task_id or s.task_id entries.append(entry) @@ -1889,6 +1985,7 @@ PROCESS_SCHEMA = { "name": "process_manage", # The enum names the verbs; the description keeps only non-obvious semantics # (write-vs-submit is the one real trap: a lone \n on a Windows PTY is not Enter). + # See #95681. "description": ( "Poll, wait on, or kill background terminal processes (from " "terminal(background=true)). " @@ -1936,7 +2033,10 @@ def _redact_process_result(result: dict) -> dict: """Redact secrets from background-process output before it reaches the model, session.db and CLI, mirroring the foreground ``terminal`` redaction so the two surfaces can't diverge. Respects ``security.redact_secrets``; ``redact_terminal_output`` - picks ``code_file`` from the recorded command. The command itself is redacted too.""" + picks ``code_file`` from the recorded command. The command itself is redacted too. + + The command string itself is also redacted in case it carried an inline credential. See #43025. + """ if not isinstance(result, dict): return result from agent.redact import redact_sensitive_text, redact_terminal_output @@ -1955,6 +2055,7 @@ def _list_processes(task_id) -> dict: # server): they share the gateway session_key and can block session reset. session_key = "" with suppress(Exception): + # See #29177. from tools.approval import get_current_session_key session_key = get_current_session_key(default="") or "" return {"processes": [ diff --git a/tools/process_registry_notifications.py b/tools/process_registry_notifications.py index 0f70f371fd..15c9355867 100644 --- a/tools/process_registry_notifications.py +++ b/tools/process_registry_notifications.py @@ -26,7 +26,12 @@ def _format_age(seconds: float) -> str: def _model_not_found_patterns() -> "list[str]": """Model-not-found phrases from ``agent.error_classifier`` (the failover path's - own list, so nothing drifts); a minimal built-in set if the import fails.""" + own list, so nothing drifts); a minimal built-in set if the import fails. + + Imported from ``agent.error_classifier`` so the batch renderer applies the SAME classification the + failover path consumes — no hand-copied pattern list to drift. Fails open to a minimal built-in set so a + classifier import problem never hides the per-task blocks. See #97667. + """ try: from agent.error_classifier import _MODEL_NOT_FOUND_PATTERNS return list(_MODEL_NOT_FOUND_PATTERNS) diff --git a/tools/project_tools.py b/tools/project_tools.py index 52e64680b5..c5925b9206 100644 --- a/tools/project_tools.py +++ b/tools/project_tools.py @@ -91,6 +91,8 @@ def project_create(name: str, path: Optional[str] = None, task_id: Optional[str] existing = pdb.find_by_primary_path(conn, folder) if folder else None if existing is not None: # Idempotent create: duplicates would render N identical sidebar subtrees. + # Idempotent create: the folder already belongs to a project. Re-activating it beats minting + # a duplicate — duplicated projects render N identical sidebar subtrees (#75820). pdb.set_active(conn, existing.id) proj = existing else: @@ -129,6 +131,8 @@ def _handle_project(args, **kw): # One action enum instead of three tools: each re-taught "desktop Projects" (244 -> ~145 tok). +# Consolidated (#95681, maintainer-directed): project_list/create/switch each re-taught "desktop Projects +# (named workspaces)"; one action enum says it once (244 -> ~145 tok). registry.register( name="desktop_project", toolset="project", diff --git a/tools/read_extract.py b/tools/read_extract.py index a96e5431b3..b6ba1ab5fa 100644 --- a/tools/read_extract.py +++ b/tools/read_extract.py @@ -127,7 +127,12 @@ def extract_document_bytes(data: bytes, path: str) -> str: def _anydoc_missing_error(path: str) -> str: - """Teaching text for anydoc-gated formats (not in the schema: only sessions hitting one pay).""" + """Teaching text for anydoc-gated formats (not in the schema: only sessions hitting one pay). + + Response-time hint (#95681 pattern): the schema no longer lists the anydoc-gated formats or the + availability caveat — a session that never touches a .doc/.odt/.epub never pays for the explanation, and + one that does gets the full story here, with the fix. + """ return ( f"Cannot convert {path!r}: this format needs the optional anydoc " "converter, which is not installed (install blocked or first " diff --git a/tools/registry.py b/tools/registry.py index 0d402d308e..b82ae4aeea 100644 --- a/tools/registry.py +++ b/tools/registry.py @@ -278,6 +278,8 @@ def _run_check_fn_uncached(fn: Callable, *, unresolved_scope: bool = False) -> b # any profile secret scope exists, so get_secret raises by design. No traceback, # so it isn't mistaken for a crashed check_fn. logger.debug( + # The tool re-probes on the first scoped turn — log without a traceback so this cannot be + # mistaken for a crashed check_fn (#100697). "check_fn %s hit the multiplex fail-closed path with no " "profile secret scope active; dependent tools re-probe on the first scoped turn", _fn_label(fn)) @@ -689,6 +691,10 @@ class ToolRegistry: owner = self._plugin_owner_of(entry.handler) # Ownership binds to the plugin package root (``hermes_plugins.{name}``), not # the exact module: a submodule's handler is still the package's to remove. + # A handler defined in ``hermes_plugins.pkg.handlers`` is still owned by the + # ``hermes_plugins.pkg`` package — exact string equality would wrongly block root-module + # cleanup code from removing tools registered by a submodule of the same plugin (egilewski + # review on #55840). same_plugin = bool(owner and caller_owner == owner) if ( caller_owner is not None diff --git a/tools/schema_sanitizer.py b/tools/schema_sanitizer.py index c673998f28..45fe07aa66 100644 --- a/tools/schema_sanitizer.py +++ b/tools/schema_sanitizer.py @@ -237,7 +237,13 @@ def _sanitize_node(node: Any, path: str) -> Any: """Recursively sanitize a JSON-Schema fragment: bare-string schemas → ``{"type": <value>}`` (unknown strings → permissive object); object nodes gain ``properties: {}``; ``type`` arrays are normalized; property keys are renamed to the provider-safe pattern and ``required`` - follows, with entries missing from ``properties`` pruned.""" + follows, with entries missing from ``properties`` pruned. + + - Normalizes ``type: [X, "null"]`` arrays to single ``type: X`` (keeping ``nullable: true`` as a hint), + and multi-type arrays like ``["number", "string"]`` to an ``anyOf`` of single-type schemas so no branch + is dropped (ported from anomalyco/opencode#31877). - Recurses into ``properties``, ``items``, + ``additionalProperties``, ``anyOf``, ``oneOf``, ``allOf``, and ``$defs`` / ``definitions``. + """ if isinstance(node, str): if node in _BARE_TYPE_NAMES: logger.debug("schema_sanitizer[%s]: replacing bare-string schema %r with {'type': %r}", @@ -256,6 +262,16 @@ def _sanitize_node(node: Any, path: str) -> Any: if isinstance(props_in, dict) else {}) out: dict = {} for key, value in node.items(): + # JSON Schema ``type`` arrays (e.g. ``["number", "string"]``, common in MCP tool schemas) are + # rejected by several tool-call backends: * llama.cpp's grammar generator only accepts a singular + # string type. * Gemini (including OpenAI-compatible transports such as GitHub Copilot proxying to + # Gemini) rejects the array form outright — plain @ai-sdk/google rewrites it, but the + # OpenAI-compatible path forwards it verbatim and the backend 400s. Normalize per the SDK's + # behavior: * single non-null type → ``type: X`` (+ ``nullable: true`` if the array also contained + # "null"). No data lost. * multiple non-null types → ``anyOf`` of single-type schemas, so EVERY + # branch survives instead of silently dropping all but the first. ``null`` is lifted into + # ``nullable: true``. * all-null / empty → ``type: "null"`` (or object fallback). Ported from + # anomalyco/opencode#31877. if key == "type" and isinstance(value, list): _normalize_type_array(value, out) elif key in {"properties", "$defs", "definitions"} and isinstance(value, dict): diff --git a/tools/send_message_senders.py b/tools/send_message_senders.py index d287f03359..79290d6acd 100644 --- a/tools/send_message_senders.py +++ b/tools/send_message_senders.py @@ -96,7 +96,11 @@ async def _send_telegram_message_with_retry(bot, *, attempts: int = 3, **kwargs) def _is_telegram_thread_not_found(error: Exception) -> bool: - """Mirror of the gateway adapter's ``_is_thread_not_found_error``.""" + """Mirror of the gateway adapter's ``_is_thread_not_found_error``. + + Matches the gateway adapter's ``_is_thread_not_found_error`` for the standalone ``_send_telegram`` path + (issue #27012). + """ return "thread not found" in str(error).lower() @@ -169,6 +173,8 @@ async def _telegram_send_text_chunk(bot, chat_id, chunk, parse_mode, has_html, t try: return await send(chunk, parse_mode) except Exception as md_error: + # Thread not found — retry without message_thread_id so the message still delivers (matching the + # gateway adapter's fallback behaviour, issue #27012). if _is_telegram_thread_not_found(md_error) and text_kwargs.get("message_thread_id") is not None: logger.warning("Thread %s not found in _send_telegram, retrying without message_thread_id", text_kwargs.pop("message_thread_id")) @@ -238,6 +244,7 @@ async def _send_telegram(token, chat_id, message, media_files=None, thread_id=No from plugins.platforms.telegram.telegram_ids import normalize_telegram_chat_id from gateway.platforms.base import BasePlatformAdapter, utf16_len # Telegram accepts a numeric chat_id OR an @username string; never force-int. + # See #13206. int_chat_id = normalize_telegram_chat_id(chat_id) media_files = media_files or [] thread_kwargs = _telegram_thread_kwargs(thread_id) @@ -479,10 +486,16 @@ async def _send_signal(extra, chat_id, message, media_files=None): return _error(f"Signal send failed: {e}") +# "ephemeral connect (may re-init E2EE per send, see #46310)", async def _send_matrix_via_adapter(pconfig, chat_id, message, media_files=None, thread_id=None): """Matrix adapter send (native media preserved). Prefer the live gateway adapter's persistent olm/megolm session: ephemeral per-send connects re-init E2EE and claim one-time keys, which - under bursts exhausts recipient OTKs and silently drops messages — ephemeral is cron-only.""" + under bursts exhausts recipient OTKs and silently drops messages — ephemeral is cron-only. + + When a live gateway adapter is available (i.e. the tool runs inside a running gateway), the persistent + connection is reused — one olm/megolm session for all sends. This avoids per-message E2EE re-init storms + that exhaust recipient OTKs and silently drop messages (issue #46310). + """ media_files = media_files or [] metadata = {"thread_id": thread_id} if thread_id else None from gateway.config import Platform diff --git a/tools/send_message_tool.py b/tools/send_message_tool.py index 53c87729f8..8a4ae6c8c0 100644 --- a/tools/send_message_tool.py +++ b/tools/send_message_tool.py @@ -132,6 +132,9 @@ def _handle_send(args): return tool_error(err) if duplicate_skip := _maybe_skip_cron_duplicate_send(platform_name, chat_id, thread_id): return json.dumps(duplicate_skip) + # Slack: resolve user targets to DM channel IDs before sending. _parse_target_ref emits internal + # ``user:U...`` / ``user_name:@handle`` targets; a bare U... id can also arrive from session metadata or + # the home-channel config. All are opened via conversations.open (fixes #19236). if platform_name == "slack" and chat_id: chat_id, resolve_err = _slack_dm_chat_id(pconfig, chat_id) if resolve_err: @@ -381,6 +384,10 @@ async def _send_via_adapter(platform, pconfig, chat_id, chunk, *, thread_id=None async def _send_chunks(chunks, send_one): """``send_one(chunk, is_last)`` in order; stop at the first error dict, else last result.""" result = None + # --- Matrix: route ALL sends through the native adapter so text is encrypted in E2EE rooms too (issue: + # text-only sends arrived with a red padlock because they took the raw-HTTP standalone path). The + # adapter reuses the live gateway's E2EE session when available (#46310) and falls back to an + # encryption-aware ephemeral adapter for standalone/cron. --- for i, chunk in enumerate(chunks): result = await send_one(chunk, i == len(chunks) - 1) if isinstance(result, dict) and result.get("error"): diff --git a/tools/session_search_tool.py b/tools/session_search_tool.py index 91313695ce..99c9a26e4b 100644 --- a/tools/session_search_tool.py +++ b/tools/session_search_tool.py @@ -20,6 +20,12 @@ from hermes_state_common import _RESET_END_REASONS _HIDDEN_SESSION_SOURCES = ("kanban", "subagent", "tool") # Searchable but DEMOTED below interactive sessions: cron vocabulary dominates bare # BM25 and starves out the user's own sessions ("recall blindness"). +# Automation sources that are kept searchable but DEMOTED below interactive sessions in discover ranking. +# Cron jobs run on a schedule and accumulate large volumes of repetitive vocabulary (recurring project +# names, dates, "session", summaries); under bare BM25 they dominate the top-N FTS rows and starve out the +# user's own interactive sessions, producing "recall blindness" where only cron sessions surface (#19434). +# Demoting — not excluding — keeps cron content reachable when it's the only match, while interactive +# sessions always win when both match. _DEMOTED_SESSION_SOURCES = ("cron",) # FTS rows scanned before dedup-by-lineage — well above the distinct sessions a query # returns, so interactive matches buried under cron hits survive the demotion pass. @@ -267,6 +273,7 @@ def _discover(db, query: str, role_filter: Optional[List[str]], limit: int, sort # can't starve the user's own sessions out of the top `limit`; stable sort keeps BM25 # order within each class. raw_results = sorted(raw_results, key=lambda r: (r.get("source") or "") in _DEMOTED_SESSION_SOURCES) + # See #19434. if not raw_results and not title_result: return _discover_payload(db, query, detail, [], message=( "No matching sessions found. FTS5 ANDs all terms by default — " @@ -285,6 +292,16 @@ def _discover(db, query: str, role_filter: Optional[List[str]], limit: int, sort if len(seen_sessions) >= limit: break raw_sid, resolved_sid = r["session_id"], _resolve_lineage(db, r["session_id"]) + # Skip the current session lineage — UNLESS the hit's transcript has left live context. Three + # sub-cases: Legacy compression rotation: the FTS hit lives in a session that itself ended with + # end_reason='compression'. That session's content has been replaced by a summary in the + # continuation child, so it must stay discoverable. /new-reset (and idle/daily/CLI new_session): the + # predecessor was ended without carrying any transcript into the child. Same lineage root, but the + # prior conversation is NOT in the active context — hiding it made gateway recall go blind after + # every /new (#85756). A live delegation child has end_reason=None, so it stays excluded. In-place + # compaction: the FTS hit lives on the SAME session_id as the current session, but the matched + # message row is an archived (active=0, compacted=1) row. The live-context load filters active=1, so + # that content is no longer in context — let it through. is_compacted_hit = _is_compacted_message(db, r.get("id")) if current_lineage_root and resolved_sid == current_lineage_root and not ( _session_left_live_context(db, raw_sid) or is_compacted_hit): diff --git a/tools/setup_mcp_tool.py b/tools/setup_mcp_tool.py index 95539cc16c..d36ca7fc9d 100644 --- a/tools/setup_mcp_tool.py +++ b/tools/setup_mcp_tool.py @@ -20,6 +20,12 @@ _ACTIONS = ("install", "enable", "authorize") def setup_mcp_tool(server: str = "", action: str = "install", reason: str = "", callback: Optional[Callable] = None) -> str: """Ask the desktop GUI to run an MCP setup flow; return its JSON outcome.""" if callback is None: + # Still down — the server task is reconnecting, or it has exhausted its retry budget and parked + # (e.g. a dead stdio subprocess). Probing here would write into a dead/absent transport and re-arm + # the breaker forever (#16788). Instead, ask the (always-present) server task to rebuild the + # transport — which respawns a dead stdio subprocess — and return a clean "reconnecting" error so + # the model backs off without burning iterations. The breaker resets once the fresh session + # initializes (_run_stdio/_run_http call _reset_server_error). return tool_error( "setup_mcp is only available in the Hermes desktop app. Use the " "terminal instead: `hermes mcp install <name>` for catalog entries, " diff --git a/tools/skill_ledger.py b/tools/skill_ledger.py index baf281ace6..1c852abed4 100644 --- a/tools/skill_ledger.py +++ b/tools/skill_ledger.py @@ -338,7 +338,10 @@ def _validate_entry_paths(entry: Dict[str, Any]) -> Optional[str]: def rollback_entry(entry_id: str) -> Tuple[bool, str]: """Restore the before-state of mutation *entry_id*. Fail-closed (mirrors agent/curator_backup.rollback): every before-blob must exist BEFORE any change, and a - pre-rollback safety entry of every touched path's CURRENT state is appended first.""" + pre-rollback safety entry of every touched path's CURRENT state is appended first. + + 1. 2. See #63366. + """ entry = get_entry(entry_id) if entry is None: return False, f"no ledger entry with id '{entry_id}'" diff --git a/tools/skill_manager_batch.py b/tools/skill_manager_batch.py index 6eae86588d..9f38adff28 100644 --- a/tools/skill_manager_batch.py +++ b/tools/skill_manager_batch.py @@ -183,6 +183,10 @@ def _skill_manage_batch(operations, default_name: str = None, task_id: str = Non logger.warning("skill_manage batch rollback failed, snapshots kept at %s", snap_root) else: shutil.rmtree(snap_root, ignore_errors=True) + # utf-8-sig + errors="replace": SKILL.md files are user-authored and sometimes carry a Notepad BOM or + # stray non-UTF-8 bytes. Pinning UTF-8 with replacement keeps skill_view deterministic across platforms + # — falling back to the machine locale (cp1252/GBK) would make the same skill render differently per + # host (see PR #51701). return json.dumps( {"success": True, "operations_applied": len(results), "results": results}, ensure_ascii=False) diff --git a/tools/skill_manager_guards.py b/tools/skill_manager_guards.py index 12e321669e..c2427cdd21 100644 --- a/tools/skill_manager_guards.py +++ b/tools/skill_manager_guards.py @@ -104,7 +104,15 @@ def _is_path_redirect(path: Path) -> bool: def _validate_delete_target(skill_dir: Path) -> Optional[str]: """Last-line guard before rmtree: even a poisoned tree must never delete (1) a path outside - every known skills root, (2) a skills root itself, (3) a symlink/junction (rmtree follows it).""" + every known skills root, (2) a skills root itself, (3) a symlink/junction (rmtree follows it). + + ``_find_skill`` already restricts ``skill_dir`` to a real ``SKILL.md`` parent discovered by walking the + skills roots, so the agent cannot inject an arbitrary path the way Kilo Code's HTTP endpoint could + (their issue 11227: a built-in-skill sentinel resolved to the server cwd and a recursive delete wiped + the user's entire working directory). This is the matching defense-in-depth for our agent-facing + ``skill_manage`` delete path: even if discovery or a poisoned tree hands us a bad directory, never + recursively delete See #11227. + """ if _is_path_redirect(skill_dir): return (f"Refusing to delete '{skill_dir}': the skill directory is a " f"symlink/junction. Remove the link target manually if intended.") @@ -186,6 +194,14 @@ def _background_review_write_guard( # on presence made the policy depend on the guard's own side effect: the # first write created a null record, the next identical write was refused). usage_rec = skill_usage.load_usage().get(name) + # Skills that are not curator-managed are off-limits to autonomous curation. This prevents the LLM + # consolidation pass from mutating skills the user owns (manually authored, URL-installed, or + # created by a foreground `skill_manage(create)` at the user's request), which lack the `created_by: + # "agent"` marker. Keying on `isinstance(usage_rec, dict)` made the policy depend on the guard's own + # side effect: a local skill with no telemetry record passed, the successful write called + # bump_patch() which created a `created_by: null` record, and the very same write was refused from + # then on. "Allowed exactly once" is not a policy — it is a race with our own bookkeeping. Fail + # closed for both shapes; `hermes curator adopt <name>` is the supported way in. See #67140. if not skill_usage._is_curator_managed_record(usage_rec): _detail = (f"created_by={usage_rec.get('created_by')!r}" if isinstance(usage_rec, dict) else "no usage record") @@ -227,7 +243,13 @@ def _curator_consolidation_delete_guard( """Fail closed on unverified deletes during the curator consolidation pass. The fork's only legitimate delete is a consolidation declared via ``absorbed_into=<umbrella>`` (existence validated in ``_delete_skill``); the deterministic inactivity prune never calls skill_manage, - so a bare delete here can only be the LLM pass pruning without evidence.""" + so a bare delete here can only be the LLM pass pruning without evidence. + + A delete with no forwarding target — ``absorbed_into`` omitted (``None``) or empty (``""``) — is the + fail-open behavior reported in #29912: the consolidation pass archived whole clusters of active skills + with zero verified consolidations (``consolidated_this_run == 0``), leaving active automations pointing + at names that no longer resolve. Refuse it; keep the skill active. + """ if not _is_background_review() or (isinstance(absorbed_into, str) and absorbed_into.strip()): return None return _refusal( diff --git a/tools/skills_guard.py b/tools/skills_guard.py index 19dbfaec8b..1155736864 100644 --- a/tools/skills_guard.py +++ b/tools/skills_guard.py @@ -292,6 +292,10 @@ THREAT_PATTERNS = [ # critical for AGENT config files (exactly how persistence attacks instruct the agent; project-skill # quarantine only acts on "dangerous") but high for Hermes/other config (setup docs routinely say # "edit config.yaml"); bare references = low. + # Flagging any mention as critical produced permanent false-positive blocks for popular community skills + # (#92021). * Mechanical persistence (shell redirection, sed -i, tee, cp/mv into the file) is critical — + # an unambiguous write path. * Prose modification intent — an imperative-position verb or an explicit + # directive ("you must edit ...") aimed at the file. (_prose_modify_re(_AGENT_CONFIG_FILES), "agent_config_mod", "critical", "persistence", "instructs modification of agent config files (could persist instructions across sessions)"), (_shell_write_re(_AGENT_CONFIG_FILES), @@ -436,7 +440,12 @@ def scan_skill(skill_path: Path, source: str = "community") -> ScanResult: def _content_digest(skill_path: Path) -> str: """Canonical SHA-256 over (POSIX relative path, file bytes) ORDERED by the rel-path STRING — Path sorting is case-insensitive on Windows and diverged from ``skills_hub.bundle_content_hash`` (every installed skill then - reported ``update_available`` forever). String order keeps both sides byte-symmetric.""" + reported ``update_available`` forever). String order keeps both sides byte-symmetric. + + Ordering by ``sorted(rglob(...))`` diverged from the bundle side on Windows: Path comparison is + case-insensitive there (normcase), while ``bundle_content_hash`` sorts plain strings — the same skill + hashed to different digests and every installed skill reported ``update_available`` forever (#62310). + """ if not skill_path.is_dir(): return hashlib.sha256(skill_path.read_bytes()).hexdigest() h = hashlib.sha256() diff --git a/tools/skills_hub_github.py b/tools/skills_hub_github.py index 8283a8be9f..3f4b3cb9c3 100644 --- a/tools/skills_hub_github.py +++ b/tools/skills_hub_github.py @@ -282,6 +282,12 @@ class GitHubSource(SkillSource): return False self._add_support_file(repo, item_path, rel_path, files, item_path, ref=ref) for rel_path in sorted(referenced): + # A SKILL.md-linked support path that isn't in the tree is a dangling link — a repo-only dev + # tool, prose over-match, or a file the author forgot to push. Warn and install without it + # rather than aborting the whole install (#66760/#90081): the skill body still works, and the + # gap is visible in the log. A referenced path that IS in the tree but as a symlink (or any + # non-regular entry) stays a hard rejection — that shape is an escape attempt, not a forgotten + # file. if rel_path in symlinked: logger.warning("Rejected non-regular referenced file in skill bundle: %s%s", prefix, rel_path) return False diff --git a/tools/skills_hub_install.py b/tools/skills_hub_install.py index 70b62fbb03..3d55c16cbd 100644 --- a/tools/skills_hub_install.py +++ b/tools/skills_hub_install.py @@ -107,6 +107,10 @@ def _check_install_target(install_dir: Path) -> None: install and stays overwritable (hub installs are lock-guarded in do_install). """ from tools.skills_hub import _skills_dir + # Refuse to nest a skill inside an existing skill directory. Installing with ``--category + # <name-of-an-existing-skill>`` would create a hybrid skill-plus-category directory; a later update or + # uninstall of the outer skill would then rmtree the inner one — the sibling case of the category-bucket + # wipe reported in issue #75983. skills_root = _skills_dir().resolve() ancestor = install_dir.parent while ancestor != skills_root and ancestor.is_relative_to(skills_root): @@ -119,6 +123,13 @@ def _check_install_target(install_dir: Path) -> None: if not install_dir.is_dir(): raise ValueError(f"Refusing to install: '{install_dir.name}' already exists " f"and is not a directory. Remove it or choose a different skill name.") + # Guard against silent data loss when the install target collides with an existing category bucket (a + # directory that holds other skills). This was reported as GitHub issue #75983: installing a skill with + # --name matching an existing category directory caused rmtree to wipe all sibling skills. A directory + # that directly contains SKILL.md is an existing skill installation and stays overwritable + # (hub-installed skills are additionally guarded by the lock-file check in do_install()). But a + # directory that contains *other* skill directories is a category bucket and must NOT be silently + # deleted. if not (install_dir / "SKILL.md").exists(): skill_dirs_in = _category_skill_dirs(install_dir) if skill_dirs_in: @@ -217,6 +228,8 @@ def bundle_content_hash(bundle: SkillBundle) -> str: carry backslashes, which changed both bytes and sort order and made every skill report ``update_available`` forever — normalize before hashing. The path is hashed too so swapping contents between two files changes the hash. + + That function keys files by ``relative_to(...).as_posix()`` — forward slashes on every OS. See #62310. """ h = hashlib.sha256() normalized = {rel_path.replace("\\", "/"): content for rel_path, content in bundle.files.items()} diff --git a/tools/skills_hub_models.py b/tools/skills_hub_models.py index 24eb07bed2..bcefb0224e 100644 --- a/tools/skills_hub_models.py +++ b/tools/skills_hub_models.py @@ -253,6 +253,9 @@ _VALUELESS_QUERY_FLAG_RE = re.compile(r"(?:[A-Za-z0-9_~-]|%[0-9A-Fa-f]{2})+\Z") # Same-directory links (``](./FILE.ext)`` / ``](FILE.ext)``): siblings of SKILL.md the document links # explicitly (e.g. ./CONTEXT-FORMAT.md). Dropping them made the install "succeed" with unresolved links. # The extension requirement keeps prose words out; support-dir links stay on _LOCAL_LINK_RE. +# Skills legitimately ship supporting docs next to SKILL.md instead of under a support directory (e.g. +# mattpocock/skills' domain-modeling links ./CONTEXT-FORMAT.md); dropping them made the install "succeed" +# while the bundle came out with unresolved links (#96310). _SAMEDIR_LINK_RE = re.compile(r"\]\(([^)\s\"'<>]+)") _SAMEDIR_NAME_RE = re.compile(r"^(?:\./)?[A-Za-z0-9][A-Za-z0-9._-]*$") diff --git a/tools/skills_sync.py b/tools/skills_sync.py index be429cbd87..c6f842a547 100644 --- a/tools/skills_sync.py +++ b/tools/skills_sync.py @@ -36,6 +36,10 @@ MANIFEST_FILE = SKILLS_DIR / ".bundled_manifest" # retarget HERMES_HOME after import, and frozen constants would resolve (and for # reset_bundled_skill() DELETE) against the wrong profile. Accessors honor an explicitly # patched module global and otherwise re-resolve on every call. +# Same bug class and same fix as skills_tool (f8723c478) and skill_manager_tool (c6a3d412d): long-lived +# multi-profile runtimes (Dashboard console, TUI/Desktop backend, cron, kanban workers) import this module +# once under the launch HERMES_HOME and later scope requests to a different profile via +# set_hermes_home_override(). See #65828. _HERMES_HOME_AT_IMPORT = HERMES_HOME _SKILLS_DIR_AT_IMPORT = SKILLS_DIR _MANIFEST_FILE_AT_IMPORT = MANIFEST_FILE @@ -394,7 +398,13 @@ def sync_skills(quiet: bool = False) -> dict: def _rmtree_writable(path: Path) -> None: """rmtree that first makes read-only entries writable (Nix/deb/rpm keep r-x dirs; unlinking a child needs a writable parent, so chmod both). Scope guard: refuses anything not a STRICT - child of the active skills root (bad join / missing HERMES_HOME / malicious manifest entry).""" + child of the active skills root (bad join / missing HERMES_HOME / malicious manifest entry). + + Handles immutable package sources (Nix store, deb/rpm installs) that preserve read-only permissions on + copied files *and* directories (``r-xr-xr-x``). Removing a child requires write permission on its parent + directory, so the retry handler makes the failing path **and its parent** writable before re-attempting. + See #34860, #34972. + """ target = Path(path).resolve() skills_root = _skills_dir().resolve() if skills_root not in target.parents: diff --git a/tools/skills_sync_bundled_ops.py b/tools/skills_sync_bundled_ops.py index f3d0e922be..fe63ec96ec 100644 --- a/tools/skills_sync_bundled_ops.py +++ b/tools/skills_sync_bundled_ops.py @@ -36,6 +36,8 @@ def reset_bundled_skill(name: str, restore: bool = False) -> dict: if not in_manifest and not is_bundled: return _fail("not_in_manifest", f"'{name}' is not a tracked bundled skill. Nothing to reset. " f"(Hub-installed skills use `hermes skills uninstall`.)") + # Step 1 (optional): delete the user's copy so next sync re-copies bundled. Must happen BEFORE manifest + # deletion so that a failed rmtree does not leave the skill in a manifest-less limbo state (see #34972). deleted_user_copy = False if restore: # delete the user's copy BEFORE the manifest so a failed rmtree can't strand it if not is_bundled: diff --git a/tools/subagent_worktree.py b/tools/subagent_worktree.py index be9ce84de4..9b1e4add6a 100644 --- a/tools/subagent_worktree.py +++ b/tools/subagent_worktree.py @@ -110,6 +110,8 @@ def mark_worktree_payload_unproven(payload: Dict[str, Any], reason: str, *, The parent only sees this dict, so the uncertainty must travel in it or "0 commits, clean" reads as "the child produced nothing". *unmeasured* names only the fields actually left unproven (one probe can succeed while the other fails). + + See #88113. """ path, branch = payload.get("path", ""), payload.get("branch", "") payload["inspection_failed"] = True @@ -140,6 +142,7 @@ def finalize_subagent_worktree(info: Dict[str, str], *, prune: bool = True) -> D return payload # Without a base commit the count is an unproven default, and the prune # condition reads payload["commits"] — so it must not prune either. + # See #88113. if not base_commit: return mark_worktree_payload_unproven( payload, "no base_commit recorded — commit count unmeasurable", unmeasured="commits") diff --git a/tools/terminal_scope.py b/tools/terminal_scope.py index 1d63081893..6bef7819a9 100644 --- a/tools/terminal_scope.py +++ b/tools/terminal_scope.py @@ -59,7 +59,13 @@ def get_terminal_scope() -> Optional[Dict[str, str]]: def enforce_no_refusal() -> None: - """Raise when the active scope is a refusal scope (fail closed).""" + """Raise when the active scope is a refusal scope (fail closed). + + Execution paths (terminal tool, execute_code) call this before spawning anything: under a refusal scope + the profile's terminal policy could not be resolved, and running with the launch process's ambient + policy is exactly the authority leak this module closes (#68559 requires refusal, not fallback). + Non-scoped and policy-scoped contexts pass silently. + """ scope = _terminal_scope_var.get() if isinstance(scope, TerminalPolicyRefusal): raise TerminalPolicyUnavailable( diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 8ba8a06890..3aee5a2909 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -441,9 +441,20 @@ def _resolve_container_task_id(task_id: Optional[str]) -> str: scope = _session_scope() if task_id and scope.session_isolated: return _resolve_container_alias(task_id) + # Per-session isolation: when a session key is present (the WebUI streaming layer sets it per-session, + # the gateway per-message via contextvars), scope the container to it so switching profiles can't reuse + # a previous profile's SSHEnvironment and silently run commands on the wrong remote host. Subagents + # inherit the same session key, so they still collapse onto the parent's container (the #16177 + # shared-container intent). CLI mode has no session key and falls through to "default", behaviour + # unchanged. See commit e00f940a9. This runs *after* the isolation-override and + # docker/container_persistent branches above: those paths already key containers per task_id, so they + # stay authoritative where they apply and this only covers the cases that would otherwise collapse to + # the shared "default" key (notably SSH). session_key = _current_session_key() shared = _tenv("TERMINAL_DOCKER_SHARED_CONTAINER_KEY", "").strip() if scope.docker_profile_scoped else "" if shared: + # Explicit opt-in: trusted profiles configuring the same terminal.docker_shared_container_key share + # ONE container/cache slot (and sandbox dir) regardless of profile name (#84671). return f"shared:{shared}" if not session_key: return "default" @@ -549,6 +560,8 @@ def _ensure_terminal_env_bridged() -> None: defaults are backfilled only when none is set. A per-turn terminal scope suppresses the bridge entirely: writing scope values into the process-global env would re-create the first-writer-wins cross-profile leak the scope fixes. + + terminal_tool reads ALL terminal settings from os.environ (TERMINAL_*). See #61115, #65696. """ from tools.terminal_scope import get_terminal_scope @@ -773,6 +786,8 @@ def _resolve_command_cwd( container backends a recorded HOST path (a desktop/TUI surface registering its workspace) is unusable in the sandbox — ``cd <host path>`` fails with exit 126 — so it is discarded in favor of ``default_cwd``. + + Same guard class as the env-creation sanitizers (#50636, #54447); this is the per-command sibling site. """ if workdir: return workdir @@ -906,6 +921,7 @@ def _plan_execution( # Fail closed under a refusal scope: the routed profile's terminal # policy could not be resolved, so running with the launch process's # ambient policy is forbidden. + # See #68559. if not _host_local: from tools.terminal_scope import enforce_no_refusal diff --git a/tools/terminal_tool_lifecycle.py b/tools/terminal_tool_lifecycle.py index 91647b523a..78c48c636d 100644 --- a/tools/terminal_tool_lifecycle.py +++ b/tools/terminal_tool_lifecycle.py @@ -187,6 +187,12 @@ def ensure_task_env(task_id: Optional[str] = None): paths) bring the sandbox up on demand with the same machinery as the terminal tool. No-op on local. Returns the env, or ``None`` when local or when creation fails (best-effort; the caller's fail-closed path stays intact). + + :func:`terminal_tool` creates the environment on the first terminal command, but nothing else did — so + under a non-local backend (ssh, docker, …) a session whose first action is ``vision_analyze`` on a + container-only path hit "no active sandbox session" because the SSH/Docker handshake never ran (issue + #62825). vision reads such paths inside the sandbox (see ``tools.image_source``), so it calls this to + bring the env up on demand, reusing the same creation machinery as the terminal tool. """ from tools.terminal_tool import ( _active_environments, _creation_locks, _creation_locks_lock, _env_lock, diff --git a/tools/terminal_tool_result.py b/tools/terminal_tool_result.py index 6285bcecfe..a049d99658 100644 --- a/tools/terminal_tool_result.py +++ b/tools/terminal_tool_result.py @@ -84,6 +84,13 @@ def _interpret_exit_code(command: str, exit_code: int) -> str | None: if exit_code == 0: return None if (signal_note := _interpret_signal_exit(exit_code)) is not None: + # Signal terminations (ported from Kilo-Org/kilocode#12698, adapted to Python semantics). Two shapes + # reach the model: * negative codes — subprocess.Popen reports a signal-killed process as + # ``-signum`` (definite signal death), and * 128+signum — the conventional shell encoding when bash + # reports a signal-killed child (heuristic: a program *can* ``exit 139``, so these notes say + # "usually"). Without a note the model sees a bare ``exit_code=-9`` or ``137`` and burns turns + # re-running or mis-diagnosing (137 = OOM kill is the big one). 130/SIGINT is deliberately absent: + # the executor has bespoke interrupt-marker handling for rc=130. return signal_note # The last command of a pipeline/chain determines the exit code; base # command = its first word that isn't a VAR=val assignment, basename'd. @@ -213,6 +220,11 @@ def finalize_foreground_result( from agent.redact import redact_terminal_output from tools.ansi_strip import strip_ansi output = strip_ansi(output) + # For source/config dumps (MAX_TOKENS=100, "apiKey": "x" fixtures, postgresql:// f-string templates) the + # ENV/JSON/template passes are skipped to avoid false positives (code_file=True). But for env-dump + # commands (env/printenv/set/export/declare) the output IS a KEY=value credential dump, so + # redact_terminal_output runs the ENV pass (code_file=False) to mask opaque tokens with no vendor + # prefix. Real prefixes, auth headers, JWTs, private keys are masked in both modes. See issue #43025. output = redact_terminal_output(output.strip(), command) if output else "" exit_note = _interpret_exit_code(command, returncode) diff --git a/tools/tirith_security.py b/tools/tirith_security.py index 368dea1de0..79cee1db48 100644 --- a/tools/tirith_security.py +++ b/tools/tirith_security.py @@ -68,6 +68,12 @@ _install_failure_reason: str = "" # reason tag when _resolved_path is _INSTALL_ # retry loop. Reset on success. Lock-free on purpose: a racing double-increment only opens the # breaker one call early; no corruption or security bypass is possible. _CRASH_LIMIT = 3 +# Reset on successful execution (see _record_tirith_crash / check_command_security). Thread safety: +# _crash_count and _circuit_open are module-level globals mutated without a lock. check_command_security can +# be called from concurrent agent threads (gateway multi-session). The race is benign — at worst two threads +# both increment past _CRASH_LIMIT and both set _circuit_open = True, opening the breaker one call early. +# This intentionally matches the lock-free style of error counters in mcp_tool.py rather than the locked +# _warn_once pattern, because the worst case is harmless. See #41400. _crash_count: int = 0 _circuit_open: bool = False @@ -480,6 +486,9 @@ def check_command_security(command: str) -> dict: cfg = _load_security_config() if not cfg["tirith_enabled"]: return _verdict("allow") + # Circuit breaker: if tirith has crashed _CRASH_LIMIT times in a row, stop trying for the rest of the + # process. Without this, a corrupted or missing binary causes every tool call to hit the same spawn + # failure → fail-open → agent retry loop, hanging the user for 20+ minutes (issue #41400). if _circuit_open: return _verdict("allow", "tirith disabled (circuit breaker)") # No binary for this platform, ever: skip the resolver so we never spawn. diff --git a/tools/todo_tool.py b/tools/todo_tool.py index 33b2138038..1f5db2f85a 100644 --- a/tools/todo_tool.py +++ b/tools/todo_tool.py @@ -221,6 +221,7 @@ def check_todo_requirements() -> bool: TODO_SCHEMA = { "name": "todo_list", "description": ( + # See #95681. "Track a task list for multi-step work (3+ steps). Use for complex tasks " "with 3+ steps or when the user provides multiple tasks. " "For 'all N items' tasks, enumerate every instance as its own checklist " diff --git a/tools/tool_backend_helpers.py b/tools/tool_backend_helpers.py index 1087852ffd..f5d9bf8abd 100644 --- a/tools/tool_backend_helpers.py +++ b/tools/tool_backend_helpers.py @@ -113,7 +113,11 @@ def resolve_provider_secret(env_var: str, provider_id: str, config_value: str = ``config_value`` -> profile secret scope / env -> ``.env`` via ``env_getter`` (or ``hermes_cli.config.get_env_value``) -> credential pool for ``provider_id``. Under an active multiplex turn the profile scope is authoritative: a miss returns ``""`` rather - than borrowing another profile's env or pool. Never raises.""" + than borrowing another profile's env or pool. Never raises. + + Resolution order (fixes #68003 — keys added via ``hermes auth add <provider>`` were invisible to the + voice tools, which only consulted env/.env): + """ key = str(config_value or "").strip() or _scoped_credential(env_var) if key: return key @@ -144,7 +148,12 @@ def resolve_provider_secret(env_var: str, provider_id: str, config_value: str = def resolve_openai_audio_api_key() -> str: """Prefer VOICE_TOOLS_OPENAI_KEY, else OPENAI_API_KEY (scope-aware, pool fallback for the latter). Must go through the secret scope: a raw ``os.environ`` read could bill another - profile's account under multiplex.""" + profile's account under multiplex. + + Outside a multiplexed turn, ``OPENAI_API_KEY`` additionally falls back to the credential pool (``hermes + auth add openai-api``) via ``resolve_provider_secret`` — same #68003 fix as the other voice providers. + The dedicated voice-tools override remains env/scope-only. + """ return (resolve_provider_secret("VOICE_TOOLS_OPENAI_KEY", "") or resolve_provider_secret("OPENAI_API_KEY", "openai-api")) @@ -219,6 +228,15 @@ def selection_exists(section: str) -> bool: REMOVED_BACKENDS: Dict[str, Dict[str, str]] = {} +# Backends that once shipped in-tree but were removed. A config that still points at one otherwise fails +# silently at the FIRST tool call with a generic "no registered provider has that name" — no migration, no +# startup notice (reported after the Tavily removal in #99199). Both the startup config check +# (hermes_cli.config.validate_config_structure) and selection_error() consult this map so the user learns +# what actually happened and what to do. Declared data, one policy — add future removals here, never as +# one-off string checks at call sites. +# Currently empty: the Tavily removal (#99199) that introduced this registry was reverted by the #99731 +# restore. Future backend removals add an entry here, e.g. "web": {"<name>": "the <Name> backend was removed +# in vX.Y.Z (...)"}, def removed_backend_note(section: str, name: str) -> Optional[str]: """Explanation for a backend that used to ship in-tree, or None. ``name`` tolerates the quoted form callers pass to selection_error().""" diff --git a/tools/tool_search_validation.py b/tools/tool_search_validation.py index 511e8d5446..68be500e93 100644 --- a/tools/tool_search_validation.py +++ b/tools/tool_search_validation.py @@ -76,7 +76,11 @@ def validate_deferred_call_args(name: str, args: Dict[str, Any]) -> Optional[str downstream failure makes cheap models loop. Required-field probe first, then the same schema-guided coercion normal dispatch applies, then jsonschema on the repaired copy. Missing/malformed schemas, no validator, and external refs all fail OPEN. Returns a JSON - error string when invalid, ``None`` when the call should dispatch.""" + error string when invalid, ``None`` when the call should dispatch. + + This restores the concrete-schema checks that the provider cannot perform through the generic + ``arguments: object`` bridge. See #5149. + """ try: from tools.registry import registry as _registry schema = _registry.get_schema(name) diff --git a/tools/tour_tool.py b/tools/tour_tool.py index f0a780e19c..256118d70e 100644 --- a/tools/tour_tool.py +++ b/tools/tour_tool.py @@ -79,6 +79,7 @@ TOUR_SCHEMA = { "name": "gui_tour", # Description keeps the targets-first flow + stable-selector preference: # without them the model guesses selectors on re-rendering UI. + # See #95681. "description": ( "Guided tour in the desktop GUI: dim the screen, highlight an " "element, attach a titled popover. Surfaces: 'app' (Hermes itself) " diff --git a/tools/transcription_cloud.py b/tools/transcription_cloud.py index aca2f9a2de..e6e2ee1a47 100644 --- a/tools/transcription_cloud.py +++ b/tools/transcription_cloud.py @@ -367,6 +367,8 @@ def _direct_openai_credentials(cfg_api_key: str, cfg_base_url: str) -> Optional[ from tools.transcription_tools import resolve_openai_audio_api_key if cfg_api_key: return cfg_api_key, (cfg_base_url or OPENAI_BASE_URL) + # A local OpenAI-compatible server needs no key — send a placeholder so the SDK doesn't refuse to + # construct a client (#25193, credit @nnnet). if cfg_base_url and _is_local_or_private_url(cfg_base_url): return "not-needed", cfg_base_url direct_api_key = resolve_openai_audio_api_key() diff --git a/tools/transcription_command.py b/tools/transcription_command.py index c0e63a6fe2..07a2f55263 100644 --- a/tools/transcription_command.py +++ b/tools/transcription_command.py @@ -35,6 +35,10 @@ logger = logging.getLogger("tools.transcription_tools") # (always wins) > stt.providers.<name> command > plugin TranscriptionProvider > # "No STT provider available". The single-env-var HERMES_LOCAL_STT_COMMAND escape # hatch stays untouched via the built-in ``local_command`` path. +# Lets any whisper CLI / ASR CLI / curl pipeline become an STT backend with zero Python. 1. Built-in +# (``local``, ``local_command``, ``groq``, ``openai``, ``mistral``, ``xai``) → native handler. +# **Always wins.** 2. 3. 4. Use the command-provider registry when you want MULTIPLE shell-driven STT +# engines, or you want a named provider you can pick via ``stt.provider`` in config.yaml. See #17843. DEFAULT_COMMAND_STT_TIMEOUT_SECONDS = 300 DEFAULT_COMMAND_STT_LANGUAGE = "en" DEFAULT_COMMAND_STT_OUTPUT_FORMAT = "txt" @@ -121,6 +125,9 @@ def _unregistered_stt_provider_error(provider: str) -> Dict[str, Any]: error_type="provider_not_registered") +# --------------------------------------------------------------------------- Plugin provider dispatch +# (issue follow-up to #30398 — STT pluggability) +# --------------------------------------------------------------------------- def _dispatch_to_plugin_provider( file_path: str, provider: str, stt_config: Optional[Dict[str, Any]] = None, *, model: Optional[str] = None, language: Optional[str] = None, prompt: Optional[str] = None, @@ -180,6 +187,9 @@ def _dispatch_to_plugin_provider( # Fields a pre_transcription hook may mutate; ``file_path`` is read-only (logged and dropped). +# --------------------------------------------------------------------------- pre_transcription plugin hook +# (issue #64168 — STT prompt/vocab threading) +# --------------------------------------------------------------------------- _PRE_TRANSCRIPTION_MUTABLE_FIELDS = ("prompt", "language", "model") # Whisper-family models only use the final ~224 tokens of the prompt; longer values diff --git a/tools/transcription_common.py b/tools/transcription_common.py index 94db230c20..1d6ba9ec2d 100644 --- a/tools/transcription_common.py +++ b/tools/transcription_common.py @@ -44,6 +44,8 @@ GROQ_MODELS = {"whisper-large-v3", "whisper-large-v3-turbo", "distil-whisper-lar # Providers with native handlers. Kept in sync with ``agent.transcription_registry._BUILTIN_NAMES`` # (a regression test fails on drift); plugins may not register under these names and the # dispatcher short-circuits them before command/plugin lookup. +# The plugin hook from issue #30398-style follow-up rejects plugins registering under any of these names; +# the dispatcher in ``transcribe_audio`` short-circuits them defensively as well. BUILTIN_STT_PROVIDERS = frozenset({ "local", "local_command", "groq", "openai", "mistral", "xai", "elevenlabs", "deepinfra"}) # Built-in providers that upload audio to a remote API. diff --git a/tools/transcription_local.py b/tools/transcription_local.py index 49ec0054b8..bfee78216c 100644 --- a/tools/transcription_local.py +++ b/tools/transcription_local.py @@ -63,6 +63,7 @@ def _try_lazy_install_stt() -> bool: from tools.lazy_deps import ensure # prompt=False: a bare input() deadlocks under the interactive CLI where prompt_toolkit # owns stdin; the install is already gated by security.allow_lazy_installs. + # prompt=False: never raise a blocking input() prompt mid-session. See #40490. ensure("stt.faster_whisper", prompt=False) if _ilu.find_spec("faster_whisper"): return True @@ -125,7 +126,11 @@ def _load_local_whisper_model(model_name: str, device: str = "auto", compute_typ """Load faster-whisper with graceful CUDA → CPU fallback. ``device="auto"`` picks CUDA whenever the ctranslate2 wheel ships CUDA libs, even on hosts without the NVIDIA runtime (WSL2, headless servers): try the requested config first; on a CUDA library load failure fall back to - CPU + int8. Pass ``stt.local.device`` / ``compute_type`` to pin.""" + CPU + int8. Pass ``stt.local.device`` / ``compute_type`` to pin. + + ``device`` / ``compute_type`` default to ``"auto"`` so the historical behaviour is unchanged; pass + explicit values from ``stt.local.device`` / ``stt.local.compute_type`` to pin a configuration (#9088). + """ force_cpu = _should_force_faster_whisper_cpu() if force_cpu: # Importing ctranslate2 can itself abort on Apple Silicon/Rosetta when @@ -238,6 +243,8 @@ def _transcribe_local_command( input_path=shlex.quote(prepared_input), output_dir=shlex.quote(output_dir), language=shlex.quote(language), model=shlex.quote(normalized_model)) # Scrub Hermes secrets from the child env (same policy as _run_command_stt). + # Scrub Hermes secrets from the child env (sibling path to #56332 / _run_command_stt — this + # local-whisper path previously inherited the full process environment). from tools.environments.local import hermes_subprocess_env _run_quiet(shlex.split(command), timeout=300, env=hermes_subprocess_env(inherit_credentials=False)) txt_files = sorted(Path(output_dir).glob("*.txt")) diff --git a/tools/transcription_tools.py b/tools/transcription_tools.py index 082fe8389c..d3ce8e0d95 100644 --- a/tools/transcription_tools.py +++ b/tools/transcription_tools.py @@ -91,6 +91,7 @@ _HAS_FASTER_WHISPER, _HAS_OPENAI, _HAS_MISTRAL, _HAS_PILK = map( # Local model singleton; the lock guards check-then-load against concurrent voice messages. _local_model: Optional[object] = None _local_model_name: Optional[str] = None +# See #24767. _local_model_lock = threading.Lock() # Idle unload: one daemon thread releases the model (hundreds of MB of RAM/VRAM) after a @@ -137,6 +138,9 @@ def _resolve_stt_language( def _openai_audio_unavailable_reason() -> Optional[str]: """None when OpenAI audio has usable credentials (config, env, or managed gateway); else the reason.""" try: + # Resolve directly instead of via the boolean probe: the probe flattens + # _resolve_openai_audio_client_config's selection-specific ValueError into False, so a managed + # openai-audio gateway outage would be logged as a generic "no API key" hint (#93045). _resolve_openai_audio_client_config() return None except ValueError as exc: @@ -336,6 +340,10 @@ def _get_or_load_local_model(model_name: str, local_cfg: Dict[str, Any]): strong reference stays valid even if the idle watcher nulls the global mid-transcription.""" global _local_model, _local_model_name model = _local_model + # Lazy-load the model (downloads on first use, ~150 MB for 'base'). Double-checked lock: concurrent + # voice messages must not both download/load the model (#24767). ``model`` is a strong local reference + # bound under the lock: the idle watcher may null the module global at any time, but this transcription + # keeps using the instance it grabbed. if model is None or _local_model_name != model_name: with _local_model_lock: if _local_model is None or _local_model_name != model_name: @@ -491,6 +499,7 @@ def _dispatch_stt_provider( return handler(file_path, model_name, language=language, prompt=prompt) # Command providers: after built-ins (``stt.providers.openai.command`` can't override the # real handler) and BEFORE plugins, since config is more local than a plugin install. + # User-declared command-type provider (``stt.providers.<name>: type: command``). See #17843. command_provider_config = _resolve_command_stt_provider_config(provider, stt_config) if command_provider_config is not None: return _transcribe_command_stt(file_path, provider, command_provider_config, stt_config, @@ -508,6 +517,7 @@ def _no_provider_error(provider: str, stt_config: Dict[str, Any]) -> Dict[str, A if "provider" in stt_config and provider_key and provider_key not in BUILTIN_STT_PROVIDERS and provider_key != "none": return _unregistered_stt_provider_error(provider_key) # An explicit openai selection flattened to "none" has a specific reason (e.g. managed gateway down). + # Surface it — with its `hermes tools` remediation — instead of the all-provider setup hint (#93045). if provider_key == "none" and str(stt_config.get("provider") or "") == "openai" and _HAS_OPENAI: reason = _openai_audio_unavailable_reason() if reason is not None: diff --git a/tools/tts_streaming.py b/tools/tts_streaming.py index 5447cf272d..70eab8e62f 100644 --- a/tools/tts_streaming.py +++ b/tools/tts_streaming.py @@ -222,7 +222,11 @@ class OpenAIStreamer(StreamingTTSProvider): @register("gemini") class GeminiStreamer(StreamingTTSProvider): - """Gemini ``streamGenerateContent?alt=sse`` → SSE feed of base64 PCM chunks (24 kHz), bounded streamed body.""" + """Gemini ``streamGenerateContent?alt=sse`` → SSE feed of base64 PCM chunks (24 kHz), bounded streamed body. + + Salvaged from PR #47588 (@Cdddo) and rebased onto the post-campaign infrastructure: credentials via the + provider-secret resolver, requests (not httpx) with a bounded streamed body, and main's provider ABC. + """ @staticmethod def available() -> bool: @@ -276,7 +280,10 @@ class XAIStreamer(StreamingTTSProvider): """xAI WebSocket TTS (``wss://api.x.ai/v1/tts``) → binary PCM frames (24 kHz mono int16). Credentials route through ``resolve_xai_http_credentials`` (OAuth or XAI_API_KEY), same as the sync path. ``_collect_async`` bridges the async WS loop to the sync iterator contract (test - seam).""" + seam). + + Salvaged from PR #47588 (@Cdddo): xAI's chunked TTS API is WebSocket-only (``wss://api.x.ai/v1/tts``). + """ @staticmethod def available() -> bool: diff --git a/tools/tts_text_normalize.py b/tools/tts_text_normalize.py index dccb01f088..6aa5ca5374 100644 --- a/tools/tts_text_normalize.py +++ b/tools/tts_text_normalize.py @@ -167,6 +167,8 @@ def smooth_whitespace_for_tts(text: str) -> str: # ``/reasoning show`` emits ``<think>...</think>`` in the final message: users want to # SEE reasoning, not hear it. An unterminated block (streaming cut-off) is also silenced. +# Reasoning blocks: models with ``/reasoning show`` enabled emit ``<think>...</think>`` blocks in the final +# assistant message. See #34213. _THINK_BLOCK_RE = re.compile(r"<think[\s>].*?</think>", flags=re.DOTALL | re.IGNORECASE) _THINK_BLOCK_OPEN_RE = re.compile(r"<think[\s>].*\Z", flags=re.DOTALL | re.IGNORECASE) @@ -187,7 +189,10 @@ def strip_nonspoken_blocks(text: str) -> str: def flatten_newlines_for_payload(text: str) -> str: """Collapse newlines into sentence breaks for single-line TTS payloads: some OpenAI-compatible backends (e.g. Kokoro) truncate at the first newline; smoothing already ends each line with - punctuation, so this is safe.""" + punctuation, so this is safe. + + See #9004. + """ if not text: return "" for pattern, repl in ((r"\n{2,}", ". "), (r"(?<=[.!?;:,])\n", " "), (r"\n", ". "), (r"\.\s*\.", "."), diff --git a/tools/tts_tool.py b/tools/tts_tool.py index d5ab28f54f..440878eef2 100644 --- a/tools/tts_tool.py +++ b/tools/tts_tool.py @@ -148,7 +148,16 @@ DEFAULT_OUTPUT_DIR = _DEFAULT_OUTPUT_DIR_AT_IMPORT = _get_default_output_dir() def _default_output_dir() -> str: """The active profile's audio output dir at call time (long-lived runtimes switch profiles - after import); a monkeypatched ``DEFAULT_OUTPUT_DIR`` wins.""" + after import); a monkeypatched ``DEFAULT_OUTPUT_DIR`` wins. + + Same bug class as skills_tool (f8723c478) and skills_sync (#65828): long-lived multi-profile runtimes + (dashboard console, TUI/Desktop backend, cron, kanban workers) import this module once under the launch + HERMES_HOME and later scope requests to a different profile via + ``hermes_constants.set_hermes_home_override()`` — a frozen module constant keeps writing synthesized + audio into the launch profile's cache instead of the active profile's (#98749). Keep the legacy + ``DEFAULT_OUTPUT_DIR`` module attribute for tests and external patchers; when it has not been patched, + re-resolve from the live profile-scoped HERMES_HOME on every call. + """ if DEFAULT_OUTPUT_DIR != _DEFAULT_OUTPUT_DIR_AT_IMPORT: return DEFAULT_OUTPUT_DIR return _get_default_output_dir() @@ -273,6 +282,9 @@ def _finalize_voice_delivery( return file_str, native and want_opus and file_str.endswith(".ogg") if not opted_in: return file_str, False + # Plugin-registered provider (issue #30398). Voice-bubble delivery opts in via + # ``TTSProvider.voice_compatible`` (mirrors the command-provider opt-in). Plugins that already write + # Opus skip the ffmpeg conversion. if not file_str.endswith(".ogg"): file_str = _convert_to_opus(file_str) or file_str return file_str, file_str.endswith(".ogg") @@ -354,6 +366,12 @@ def _text_to_speech_single( logger.info("Generating speech with command TTS provider '%s'...", provider) file_str = _generate_command_tts( text, file_str, provider, command_provider_config, tts_config) + # Plugin-registered TTS backend (issue #30398). Fires when the configured provider is neither a + # built-in nor a command-type entry, AND a plugin is registered under that name. The walrus binds + # `_plugin_path` only when the dispatcher returns a path (i.e. a plugin was actually found); a None + # return falls through to the built-in elif chain so unknown names hit the Edge TTS default at the + # bottom. The dispatcher itself enforces built-ins-always-win + command-wins-over-plugin + # defensively. elif provider not in BUILTIN_TTS_PROVIDERS and ( _plugin_path := _dispatch_to_plugin_provider(text, file_str, provider, tts_config) ) is not None: diff --git a/tools/tts_tool_plugins.py b/tools/tts_tool_plugins.py index 19960ab67e..12eeda9be1 100644 --- a/tools/tts_tool_plugins.py +++ b/tools/tts_tool_plugins.py @@ -36,7 +36,14 @@ def _dispatch_to_plugin_provider(text: str, output_path: str, provider: str, tts Invariants re-checked here so a caller refactor can't break them: built-in names never reach the registry; a same-named ``type: command`` provider wins; only an exact registered name - dispatches. Plugin exceptions propagate to ``text_to_speech_tool``'s error envelope.""" + dispatches. Plugin exceptions propagate to ``text_to_speech_tool``'s error envelope. + + Resolution invariants enforced here (matches issue #30398): + 1. The caller is responsible for the elif chain that handles ``edge``/``openai``/etc.; this function + explicitly rejects those names defensively. 2. 3. Plugin dispatch fires only when ``provider`` matches a + registered :class:`TTSProvider` whose ``name`` equals the configured value. Unknown names return None + (caller falls through to Edge default). See #17843. + """ key = (provider or "").lower().strip() if not key or key in BUILTIN_TTS_PROVIDERS: return None diff --git a/tools/tts_tool_providers.py b/tools/tts_tool_providers.py index 2e1dc2a8ec..09b12a02ea 100644 --- a/tools/tts_tool_providers.py +++ b/tools/tts_tool_providers.py @@ -289,6 +289,7 @@ def _generate_xai_tts(text: str, output_path: str, tts_config: Dict[str, Any]) - # TTS is API-billed: a subscription OAuth bearer can authorize chat while # returning 403 for /v1/tts, so prefer an explicit XAI_API_KEY over OAuth. + # See #87045, #88040. creds = resolve_xai_http_credentials(prefer_api_key=True) api_key = str(creds.get("api_key") or "").strip() if not api_key: diff --git a/tools/tts_tool_speaker.py b/tools/tts_tool_speaker.py index f095b46aec..51afdaadd6 100644 --- a/tools/tts_tool_speaker.py +++ b/tools/tts_tool_speaker.py @@ -156,6 +156,7 @@ class _StreamerPlayback: # macOS skips sounddevice entirely: PortAudio/CoreAudio init triggers a # kTCCServiceMediaLibrary prompt though output needs no media-library access. # None routes every sentence through tempfile -> afplay. + # See PR #62601 / #13291. if platform.system() == "Darwin": return None try: diff --git a/tools/video_generation_tool.py b/tools/video_generation_tool.py index 1397c2ac57..3f10d846e4 100644 --- a/tools/video_generation_tool.py +++ b/tools/video_generation_tool.py @@ -68,6 +68,8 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = { }, # Capability-gated args are added by _build_dynamic_video_schema; never statically. }, + # NOTE (schema diet, #95681): image_url / reference_image_urls / negative_prompt / audio / seed / + # upscale are added per-capability by _build_dynamic_video_schema. "required": ["prompt"], }, } diff --git a/tools/vision_tools.py b/tools/vision_tools.py index 007d851486..addb59a700 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -267,6 +267,11 @@ _MAX_BASE64_BYTES = 20 * 1024 * 1024 # Proactive embed caps for history reuse: the native path bakes the data URL into the tool # result, re-sent every later turn (a 4 MB embed cost ~100-260K billed tokens). Anthropic # downsamples to a 1568px long edge anyway, so pixels past that cost wire bytes for no fidelity. +# The 20 MB hard ceiling / Anthropic 5 MB reject-cap still apply as safety nets; those are one-shot viewing +# limits, not history-reuse sizes. A 4 MB / 7900px embed was observed at ~400K chars and ~100–260K billed +# tokens per image (#92699), so we size for model reading instead: 256 KB keeps a 1568px screenshot cheap +# enough to ride the session (PNGs that exceed it are downscaled further by the byte-budget ladder), well +# under every provider's per-image limit. _EMBED_TARGET_BYTES = 256 * 1024 _EMBED_MAX_DIMENSION = 1568 @@ -314,6 +319,9 @@ def _import_pillow_for_resize(): except ImportError: try: from tools.lazy_deps import ensure as _ensure_dep + # prompt=False: never raise a blocking input() prompt mid-session. Under the interactive CLI + # prompt_toolkit owns stdin, so a bare input() deadlocks the terminal (#40490). The install is + # already gated by security.allow_lazy_installs, so reaching here is opt-in. _ensure_dep("tool.vision", prompt=False) from PIL import Image except Exception: @@ -334,6 +342,12 @@ def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, ``max_dimension``: force a downscale above this long edge even when bytes fit (Anthropic's 8000px cap is independent of bytes). ``force_jpeg``: re-encode PNG as JPEG when resizing — halving PNG dimensions destroys text legibility on dense screenshots. + + Args: max_dimension: If set, images whose longest side exceeds this pixel count are forcibly downscaled + even if they're under the byte budget. Anthropic enforces an 8000 px per-side cap independently of the 5 + MB byte cap. force_jpeg: Re-encode as JPEG even for PNG input when a resize is needed. History-reuse + embeds (#92699) opt in so a text-heavy screenshot keeps its readable resolution and shrinks via JPEG + quality instead. Images already under both caps are returned unchanged (still PNG). """ file_size = image_path.stat().st_size estimated_b64 = (file_size * 4) // 3 + 100 # base64 ~4/3 + data URL header @@ -602,6 +616,9 @@ async def _vision_analyze_native( # Proactive embed cap: this image is re-sent on every later turn, so resize DOWN to the # history-reuse target whenever the byte or long-edge cap is exceeded, not just at 20 MB. _scale_info: dict = {} + # Anthropic still rejects >5 MB / >8000px with a non-retryable 400, but those are one-shot viewing + # limits — history embeds are sized smaller so repeated vision_analyze turns don't blow the context + # (#92699). _over_dims = await _run_encode_on_cpu_executor( _image_exceeds_dimension, prepared.path, _EMBED_MAX_DIMENSION) if len(image_data_url) > _EMBED_TARGET_BYTES or _over_dims: @@ -803,6 +820,8 @@ def check_vision_requirements() -> bool: Mirrors its fallback chain: explicit ``auxiliary.vision.provider``, then auto (main provider → openrouter → nous) — without the auto step the tool would vanish whenever the explicit name was unresolvable. Probe mode skips real SDK client construction. + + See #31179. """ try: from agent.auxiliary_client import aux_probe_mode, resolve_vision_provider_client @@ -822,6 +841,9 @@ VISION_ANALYZE_SCHEMA = { # native result says so itself); region keeps its pre-effect guidance — a # model that doesn't know crops keep full resolution never zooms. "description": ( + # Dieted (#95681): routing mechanics (native attach vs aux-model text fallback) removed — the route + # is automatic and the native path's own tool result says "you can see it natively now"; the schema + # doesn't need to predict plumbing. "Load an image into the conversation so you can see it. Call it " "any time the user references an image — then answer from what " "you see." diff --git a/tools/voice_mode.py b/tools/voice_mode.py index a7293da8a0..2215588490 100644 --- a/tools/voice_mode.py +++ b/tools/voice_mode.py @@ -54,7 +54,13 @@ def _import_audio(): def _sounddevice_output_allowed() -> bool: """False on macOS: PortAudio/CoreAudio OUTPUT init triggers a kTCCServiceMediaLibrary - prompt, so output goes through ``afplay`` there. Input (recording) is unaffected.""" + prompt, so output goes through ``afplay`` there. Input (recording) is unaffected. + + Returns False on macOS: importing/initializing sounddevice (PortAudio/CoreAudio) for output triggers a + kTCCServiceMediaLibrary permission prompt, even though playback needs no media-library access. This does + NOT affect audio *input* (recording), which legitimately needs microphone permission. See PR #62601 / + #13291. + """ return platform.system() != "Darwin" @@ -112,6 +118,8 @@ def _default_input_samplerate(sd) -> int: # ── Environment detection ── def _voice_capture_install_hint() -> str: + # sounddevice imports but PortAudio's shared library is missing — a pip install can't fix that; point at + # the system package instead of misreporting missing Python packages (#18432). if _is_termux_environment(): return "pkg install python-numpy portaudio && python -m pip install sounddevice" # Inside a venv a bare `pip install` may hit whichever Python the shell @@ -205,7 +213,13 @@ def _pulse_socket_candidates() -> List[str]: def _pulse_socket_reachable() -> bool: """True if a PulseAudio/PipeWire socket on disk accepts a connection (a stale socket of - a dead server does not count). Covers a local sound server without PULSE_SERVER set.""" + a dead server does not count). Covers a local sound server without PULSE_SERVER set. + + Covers the common case where a sound server runs locally (e.g. on a remote SSH host) without + ``PULSE_SERVER``/``PIPEWIRE_REMOTE`` being set -- the client just connects to the default socket under + the runtime dir. We look at ``PULSE_SERVER`` unix paths, ``PULSE_RUNTIME_PATH``, and ``XDG_RUNTIME_DIR`` + for a ``pulse/native`` or ``pipewire-0`` socket (issue #35622). + """ import socket import stat for path in _pulse_socket_candidates(): @@ -275,6 +289,8 @@ def detect_audio_environment() -> dict: def report(notice: str, warning: str) -> None: (notices if has_forwarded_audio else warnings).append(notice if has_forwarded_audio else warning) + # SSH detection -- normally no audio devices, but honor a reachable sound server (PulseAudio/PipeWire + # socket or forwarding env vars), which works fine over SSH (issue #35622). if any(os.environ.get(v) for v in ('SSH_CLIENT', 'SSH_TTY', 'SSH_CONNECTION')): report("Running over SSH with a reachable PulseAudio/PipeWire sound server", "Running over SSH -- no audio devices available.\n" @@ -283,6 +299,9 @@ def detect_audio_environment() -> dict: " export XDG_RUNTIME_DIR=/run/user/$(id -u)\n" " # or: export PULSE_SERVER=unix:$XDG_RUNTIME_DIR/pulse/native") + # Docker/Podman container detection — honor host audio forwarding. When the user mounts a + # PulseAudio/PipeWire socket into the container and points PULSE_SERVER / PIPEWIRE_REMOTE at it, audio + # works fine (issue #21203). Only block when no forwarding is configured. from hermes_constants import is_container if is_container(): report("Running inside container (Docker/Podman/LXC) with host audio forwarding", @@ -1044,6 +1063,7 @@ def _run_system_player(cmd: List[str]) -> bool: proc = None try: # Sibling of the TTS/STT credential scrub: players must not inherit tokens/keys. + # See #56332, #70342. from tools.environments.local import hermes_subprocess_env proc = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, stdin=subprocess.DEVNULL, env=hermes_subprocess_env(inherit_credentials=False)) diff --git a/tools/voice_mode_transcript.py b/tools/voice_mode_transcript.py index fd1992ee1a..f9d5319054 100644 --- a/tools/voice_mode_transcript.py +++ b/tools/voice_mode_transcript.py @@ -79,6 +79,7 @@ DEFAULT_TTS_ECHO_SIMILARITY_THRESHOLD = 0.6 # Minimum normalized-transcript length before the sliding-window fallback runs. Below # this a genuine one-word barge-in ("yes") landing verbatim inside a longer reply would # score a trivial 1.0; a real self-capture spans pre-roll plus time-to-silence, so it is longer. +# See #75792. MIN_FRAGMENT_LENGTH_FOR_ECHO = 10 diff --git a/tools/web_result_cache.py b/tools/web_result_cache.py index 0243bd2a8b..a544ebd9c9 100644 --- a/tools/web_result_cache.py +++ b/tools/web_result_cache.py @@ -123,6 +123,7 @@ class SearchMemo: # Bound the lock table, but never evict a HELD lock: dropping one lets a concurrent # identical request mint a fresh lock and issue a duplicate paid call. locked() is a # safe snapshot under _store_lock because holders already have their reference. + # See #94618. if len(self._key_locks) > 256: self._key_locks = {k: v for k, v in self._key_locks.items() if v.locked()} lock = self._key_locks[key] = threading.Lock() @@ -199,7 +200,11 @@ def _url_digest(url: str, format: Optional[str], provider: str = "") -> str: def _entry_file_path(url: str, format: Optional[str], provider: str) -> Optional[Path]: """Dedicated cache file per (url, format, provider) — deliberately NOT the truncate-store file - (keyed on URL alone), which html/markdown or two providers' copies of one URL would overwrite.""" + (keyed on URL alone), which html/markdown or two providers' copies of one URL would overwrite. + + The truncate-store file keeps its role for read_file paging; these files exist only for cache reuse and + carry the full key in their name. See #94618. + """ if (d := _cache_dir()) is None: return None slug = "page" diff --git a/tools/web_tools.py b/tools/web_tools.py index 5d73034f9c..d6f6b182fa 100644 --- a/tools/web_tools.py +++ b/tools/web_tools.py @@ -59,7 +59,13 @@ logger = logging.getLogger(__name__) # ─── Backend Selection ──────────────────────────────────────────────────────── def _env_value(name: str) -> str: - """Resolve ``name`` via the config-aware env layer (``hermes config set`` values), then process env.""" + """Resolve ``name`` via the config-aware env layer (``hermes config set`` values), then process env. + + Mirrors the SearXNG provider's ``_searxng_url()`` so that values set through Hermes' config/.env layer + (``hermes config set``, ``hermes tools``) are honored here too — not just raw process-env exports. + Without this, a config-only ``SEARXNG_URL`` (or any provider key) leaves the backend auto-detect cascade + and ``check_web_api_key()`` blind to it. See #34290. + """ try: from hermes_cli.config import get_env_value val = get_env_value(name) @@ -213,7 +219,14 @@ _LEGACY_WEB_BACKENDS = frozenset(_BUILTIN_AVAILABILITY) def _is_backend_available(backend: str) -> bool: """True when *backend* is usable — the single availability chokepoint. Non-legacy names delegate to the - registered provider's ``is_available()`` (unregistered names fall through); built-ins use cheap probes.""" + registered provider's ``is_available()`` (unregistered names fall through); built-ins use cheap probes. + + For plugin-registered backends (any name outside :data:`_LEGACY_WEB_BACKENDS`), availability is + delegated to the provider's ``is_available()`` via the web_search_registry. This is the single + chokepoint through which ``_get_backend``, ``_get_capability_backend``, and ``check_web_api_key`` all + resolve availability — fixing custom-provider discovery for every caller at once (issues #28651, #31873, + #32698). Built-in backends keep their cheap hardcoded probes below. + """ backend = (backend or "").lower().strip() provider = None if backend in _LEGACY_WEB_BACKENDS else _registered_web_provider(backend) if provider is not None: @@ -222,6 +235,11 @@ def _is_backend_available(backend: str) -> bool: return probe() if probe else False +# ─── Firecrawl Client ──────────────────────────────────────────────────────── After PR #25182, the +# firecrawl client, lazy SDK proxy, dual-auth config resolution, response normalizers, and +# check_firecrawl_api_key() all live in plugins.web.firecrawl.provider and are re-exported at the top of +# this module so external callers (integration tests, tool-registry gating) and unit tests that patch +# tools.web_tools.<name> continue to work. def _web_requires_env() -> list[str]: """Tool-registry metadata env vars for the web backends. Gateway vars are always listed: gating them on ``managed_nous_tools_enabled()`` cost a synchronous portal HTTP refresh at every CLI startup. @@ -237,10 +255,22 @@ _debug = DebugSession("web_tools", env_var="WEB_TOOLS_DEBUG") # ─── Dispatch ───────────────────────────────────────────────────────────────── +# ─── Exa / Parallel inline helpers — moved into plugins ────────────────────── After PR #25182, the exa +# client + search/extract and parallel client + search/extract helpers all live in their respective plugins: +# - plugins/web/exa/provider.py - plugins/web/parallel/provider.py Both plugins register through +# agent.web_search_registry and the dispatchers in this file resolve them via get_active_*_provider(). def _ensure_web_plugins_loaded() -> None: """Idempotently run plugin discovery so the web registry is populated. Dispatch is reachable from contexts that never triggered discovery (subprocess agent runs, delegate children, scripts); without it a - configured backend yields a misleading "No web ... provider" error.""" + configured backend yields a misleading "No web ... provider" error. + + Every bundled web provider (brave-free, ddgs, searxng, exa, parallel, tavily, firecrawl, keenable) + registers itself via ``plugins/web/<vendor>/__init__.py`` during plugin discovery. Tool dispatch can be + reached from contexts that haven't already triggered discovery — subprocess agent runs, delegate + children, standalone scripts, certain test paths — and without it the registry is empty and + ``get_provider('firecrawl')`` returns ``None`` even when the user has ``web.extract_backend: firecrawl`` + configured and ``FIRECRAWL_API_KEY`` set. See #27580. + """ try: from hermes_cli.plugins import _ensure_plugins_discovered _ensure_plugins_discovered() @@ -407,6 +437,8 @@ def _provider_is_ready(provider) -> bool: ``get_active_*_provider()`` returns an explicitly configured backend even when ``is_available()`` is False (so dispatch can emit a precise error), so readiness gates (tool check_fn, ``hermes doctor``) must probe for real. Keyless mode (Exa/Parallel free tier) is a working state, not a misconfig. + + See #78412. """ if provider is None: return False @@ -421,6 +453,8 @@ def check_web_api_key() -> bool: A plugin-registered provider reporting ``is_available()`` must light the tools up even with no built-in credentials; resolution funnels through :func:`_is_backend_available`. + + See #28651, #31873. """ # Boolean OR over configured + built-ins — probe order is irrelevant here. candidates = [c for c in (_configured_backend(),) if c] + list(_LEGACY_WEB_BACKENDS) diff --git a/tools/web_tools_truncate.py b/tools/web_tools_truncate.py index b12bb45937..a2b69b4f5c 100644 --- a/tools/web_tools_truncate.py +++ b/tools/web_tools_truncate.py @@ -14,6 +14,13 @@ logger = logging.getLogger("tools.web_tools") # Per-page char budget sent to the model (override: web.extract_char_limit); larger pages are head+tail # truncated, full text stored on disk. +# ─── Parallel / Tavily / Firecrawl helpers — moved into plugins ────────────── After PR #25182, the +# per-vendor client construction, request helpers, and response normalizers all live in +# plugins.web.<vendor>.provider: - parallel: plugins/web/parallel/provider.py - tavily: +# plugins/web/tavily/provider.py - firecrawl: plugins/web/firecrawl/provider.py The names from the firecrawl +# plugin (Firecrawl proxy, _get_firecrawl_client, _to_plain_object, _normalize_result_list, +# _extract_web_search_results, _extract_scrape_payload, _is_tool_gateway_ready, etc.) are re-exported at the +# top of this module for backward-compat with integration tests and unit-test patches. DEFAULT_EXTRACT_CHAR_LIMIT = 15000 # Ceiling on the full-text file written to cache/web so a multi-MB page can't write unbounded bytes on # every extract; the model only ever sees char_limit. diff --git a/tools/write_approval.py b/tools/write_approval.py index 2a9c5423cf..1c13433421 100644 --- a/tools/write_approval.py +++ b/tools/write_approval.py @@ -185,7 +185,10 @@ def _prompt_inline_memory_approval(summary: str, detail: str) -> Optional[bool]: CLI approval callback (``tools.terminal_tool.set_approval_callback``) directly, not ``prompt_dangerous_approval``: that wrapper falls back to ``input()`` (deadlock-prone under prompt_toolkit; silent deny in gateway sessions) and turns callback errors into a deny, whereas - here a missing channel or failed prompt must stage instead.""" + here a missing channel or failed prompt must stage instead. + + See #15216. + """ try: from tools.terminal_tool import _get_approval_callback except Exception: diff --git a/tools/x_search_tool.py b/tools/x_search_tool.py index 2c5aff1996..abc574eeb9 100644 --- a/tools/x_search_tool.py +++ b/tools/x_search_tool.py @@ -57,7 +57,14 @@ def _get_x_search_int(key: str, default: int, floor: int) -> int: def _resolve_xai_bearer() -> Tuple[str, str, str]: """Return ``(api_key, base_url, source)``; ``source`` is ``"xai-oauth"`` or ``"xai"``. Raises RuntimeError - when no credential is usable (expiry between registration and call -> clean tool error, not a 401).""" + when no credential is usable (expiry between registration and call -> clean tool error, not a 401). + + x_search is API-index access: when a subscription OAuth credential is configured alongside a paid + ``XAI_API_KEY``, the OAuth path authorizes but answers ``/v1/responses`` in a degraded Grok explanatory + mode with no citations, while the API key returns real posts (#88040). Pass ``prefer_api_key=True`` so + the shared resolver checks the explicit API key first — same root cause as the TTS fix for #87045 + (#87081) — keeping OAuth as the fallback when no API key is configured. + """ creds = resolve_xai_http_credentials(prefer_api_key=True) api_key = str(creds.get("api_key") or "").strip() if not api_key: diff --git a/tools/xai_http.py b/tools/xai_http.py index 1c6a5de0b6..3041521625 100644 --- a/tools/xai_http.py +++ b/tools/xai_http.py @@ -198,6 +198,15 @@ def resolve_xai_http_credentials( behind the same origin-pinning validation. ``force_refresh=True`` forces an OAuth refresh; pass the rejected bearer as ``api_key_hint`` so a multi-account pool refreshes the issuing entry, not whichever its strategy selects first. + + Prefers Hermes-managed xAI OAuth credentials when available, then falls back to ``XAI_API_KEY`` resolved + via ``hermes_cli.config.get_env_value`` so keys stored in ``~/.hermes/.env`` (the standard Hermes + location) are honored — not just ones already exported into ``os.environ``. This keeps direct xAI + endpoints (images, TTS, STT, etc.) aligned with the main runtime auth model and preserves the regression + contract from PR #17140 / #17163. + The key is read through :func:`tools.tool_backend_helpers.resolve_provider_secret` so profile secret + scoping is identical to the fallback branch, and the base URL honors ``HERMES_XAI_BASE_URL`` / + ``XAI_BASE_URL`` behind the same origin-pinning validation as the OAuth branch. See #87045, #88040. """ import hermes_cli.auth as auth_mod if prefer_api_key and (explicit_key := str(_resolve_explicit_xai_api_key() or "").strip()): diff --git a/toolsets.py b/toolsets.py index 6df4907967..64bc8625a7 100644 --- a/toolsets.py +++ b/toolsets.py @@ -262,6 +262,11 @@ def get_toolset(name: str, *, include_registry: bool = True) -> Optional[Dict[st and resolves registry-only (plugin/MCP) toolsets and aliases; False returns a copy of the static TOOLSETS entry only, so platform reverse-mapping is unaffected by registry additions. + + Args: name (str): Name of the toolset include_registry (bool): When True (default), merge in tools that + plugins/overlays registered into this toolset via the registry. Platform reverse-mapping in + ``_get_platform_tools`` uses False so that a tool registered into a toolset but absent from a platform's + static composite does not drop the whole toolset from inference. See issue #49622. """ toolset = TOOLSETS.get(name) if not include_registry: @@ -333,7 +338,13 @@ def _plugin_platform_bundle(name: str) -> List[str]: def resolve_toolset(name: str, visited: Set[str] = None, *, include_registry: bool = True) -> List[str]: """Recursively resolve a toolset (and its includes) to a sorted tool-name list. - include_registry=False resolves the static TOOLSETS view only.""" + include_registry=False resolves the static TOOLSETS view only. + + Args: name (str): Name of the toolset to resolve visited (Set[str]): Set of already visited toolsets + (for cycle detection) include_registry (bool): When True (default), include tools that plugins/overlays + registered into a toolset. Platform reverse-mapping uses False so a registry-added tool cannot drop the + whole toolset from inference (see #49622 and ``_get_platform_tools``). + """ external_call = visited is None if external_call: memo_key = (name, include_registry, *_registry_generation()) diff --git a/tui_gateway/agent_callbacks.py b/tui_gateway/agent_callbacks.py index ea30f137c8..43f230c0fd 100644 --- a/tui_gateway/agent_callbacks.py +++ b/tui_gateway/agent_callbacks.py @@ -216,6 +216,8 @@ def _apply_personality_to_session( # Like the model-switch marker: role=user so strict providers accept it mid-conversation, # but `display_kind` keeps it out of the `truncate_before_user_ordinal` addressing space # (untagged, every rewind would land one turn early and hard-delete the difference). + # Untagged, it counts as a real user turn on the gateway side while no client counts it, so every later + # rewind resolves one turn too early and `replace_messages` hard-deletes the difference (#82756). with session["history_lock"]: session["history"].append({"role": "user", "content": marker, "display_kind": "personality_switch"}) session["history_version"] = int(session.get("history_version", 0)) + 1 diff --git a/tui_gateway/change_watcher.py b/tui_gateway/change_watcher.py index 9cb3dfa698..8eadec3e45 100644 --- a/tui_gateway/change_watcher.py +++ b/tui_gateway/change_watcher.py @@ -117,7 +117,12 @@ def _pet_changed_payload() -> dict: def _sessions_sig(): """Newest mtime across state.db + WAL: the one thing messaging-gateway turns and cron runs - all move. Served sibling profile homes are probed too, else a routed Bot Chat never refreshes.""" + all move. Served sibling profile homes are probed too, else a routed Bot Chat never refreshes. + + signal. Messaging-gateway turns and cron runs are written by OTHER processes that never touch this + gateway's transports; the shared SQLite file is the one thing they all move (#58671). A backend serving + several profiles owns one store per profile, so every served sibling home is + """ return _newest_mtime_ns( root / name for root in (_watcher_home(), *_served_profile_homes) @@ -149,7 +154,12 @@ _bot_relay_outbox_seen = 0 def _bot_relay_outbox_sig(): """Newest mtime across pending bot-relay outbox envelopes (monotone). Written by the AGENT - process, so the files are the only shared signal; the Desktop reacts with a debounced drain.""" + process, so the files are the only shared signal; the Desktop reacts with a debounced drain. + + Envelopes are written by the AGENT process (``message_agent`` → ``tools.bot_relay.enqueue_envelope``) — + a different process that never touches this gateway's transports — so the files are the only shared + signal, exactly like the pairing store. See #92760, #93091. + """ global _bot_relay_outbox_seen home = _watcher_home() root = home.parent.parent if home.parent.name == "profiles" else home diff --git a/tui_gateway/compute_host_bridge.py b/tui_gateway/compute_host_bridge.py index 5a5a0398d9..003c119b2e 100644 --- a/tui_gateway/compute_host_bridge.py +++ b/tui_gateway/compute_host_bridge.py @@ -16,6 +16,7 @@ _compute_host_supervisor_lock = threading.Lock() # Cap on how long session.compress blocks its RPC on the compute host. Must stay # below the desktop's SESSION_COMPRESS_TIMEOUT_MS (660s) so the client gets the # `pending` answer, not its own timeout; the late-ack path covers anything slower. +# See #97948. _COMPUTE_HOST_COMPRESS_WAIT_CAP_SECS = 630.0 @@ -242,7 +243,10 @@ def _send_compute_host_control( def _compute_host_compress_wait_seconds(cfg: dict | None = None) -> float: """RPC wait budget for a compute-host compress control: the configured compression ceiling plus slack, capped below the desktop's RPC timeout (a fixed waiter reported - false timeouts while the host kept working); slower acks land via the late-ack path.""" + false timeouts while the host kept working); slower acks land via the late-ack path. + + See #97948. + """ from agent.conversation_compression import resolve_context_compression_timeouts try: compression_cfg = (cfg if cfg is not None else _load_cfg()).get("compression", {}) diff --git a/tui_gateway/entry.py b/tui_gateway/entry.py index 209c48cbf9..725d17ac0c 100644 --- a/tui_gateway/entry.py +++ b/tui_gateway/entry.py @@ -153,6 +153,15 @@ def wait_for_mcp_discovery(timeout: "float | None" = None) -> None: return # Shared-owner path: re-invoke the idempotent spawn first so a zero-connected run gets # its retry instead of latching the process MCP-less (runs under the CALLER's profile). + # Discovery is spawned via the shared owner (ensure_mcp_discovery_started → hermes_cli.mcp_startup); + # wait on it so the first agent build still catches fast servers. Re-invoke the idempotent spawn first: + # if the previous run finished with zero connected servers, start_background_mcp_discovery's + # retry-after-zero-connected allowance kicks off a fresh discovery run here instead of leaving the + # process latched MCP-less for the session. In multi-profile processes this retry runs under the + # CALLER's profile context (agent build binds the session profile's HERMES_HOME first), so a launch + # profile with no mcp_servers no longer starves selected profiles of discovery (#67605). Gated on + # _mcp_discovery_enabled so non-MCP sessions never pay the tools.mcp_tool import on the per-agent-build + # wait path. if not _mcp_discovery_enabled: return _spawn_discovery(("debug", "TUI MCP discovery retry-spawn failed")) @@ -161,7 +170,16 @@ def wait_for_mcp_discovery(timeout: "float | None" = None) -> None: def mcp_discovery_in_flight() -> bool: """True if ANY background MCP discovery thread is still running: the late-refresh - scheduler calls this regardless of surface, so it MUST consult both owners.""" + scheduler calls this regardless of surface, so it MUST consult both owners. + + There are two independent discovery-thread owners by surface: the stdio ``hermes --tui`` path spawns ITS + thread here (``_mcp_discovery_thread``), while the desktop app + dashboard WebSocket sidecar + (``tui_gateway/ws.py``) and ``hermes dashboard`` spawn theirs via + ``hermes_cli.mcp_startup.start_background_mcp_discovery``. The late-refresh scheduler imports this + function regardless of surface, so it MUST consult both — checking only the entry thread left the + desktop/dashboard surfaces with no late refresh, so a slow MCP server's tools never surfaced for the + whole session (#51587). + """ thread = _mcp_discovery_thread if thread is not None and thread.is_alive(): return True @@ -170,7 +188,11 @@ def mcp_discovery_in_flight() -> bool: def join_mcp_discovery(timeout: float | None = None) -> bool: """Join both discovery owners; True once neither is alive. Accepts an unbounded wait - (off-critical-path late-refresh waiter); ``timeout`` bounds EACH join, entry thread first.""" + (off-critical-path late-refresh waiter); ``timeout`` bounds EACH join, entry thread first. + + Joins both discovery-thread owners (see ``mcp_discovery_in_flight``): the entry thread first, then the + ``hermes_cli.mcp_startup`` thread used by the desktop/dashboard surfaces. See #51587. + """ entry_done = True thread = _mcp_discovery_thread if thread is not None: @@ -192,7 +214,17 @@ def _has_configured_mcp_servers() -> bool: def ensure_mcp_discovery_started() -> None: """Start background MCP discovery for the current profile context, once. ``main()`` calls this for stdio; ``server._start_agent_build`` also calls it AFTER binding the session - profile's HERMES_HOME. MCP registration is process-global: the FIRST profile wins.""" + profile's HERMES_HOME. MCP registration is process-global: the FIRST profile wins. + + WebSocket/Desktop entrypoints can accept sessions without running ``main()``, so the agent-build path + (``server._start_agent_build``) also calls it AFTER binding the session profile's HERMES_HOME override — + the shared owner in ``hermes_cli.mcp_startup`` captures the caller's context-local override and + propagates it into the discovery thread, so discovery reads the SELECTED profile's ``mcp_servers``, not + the launch profile's (#67605). + Known limitation: MCP tool registration is process-global, so in a multi-profile process the FIRST + profile that builds an agent wins the discovery slot. Full per-profile MCP registries are tracked in + #67605. + """ global _mcp_discovery_enabled if not _has_configured_mcp_servers(): return diff --git a/tui_gateway/git_probe.py b/tui_gateway/git_probe.py index 875385c7d7..9c1430260e 100644 --- a/tui_gateway/git_probe.py +++ b/tui_gateway/git_probe.py @@ -21,7 +21,12 @@ _NEG_TTL = 30.0 # "not a git repo" TTL: a fresh `git init` shows within seconds def run_git(cwd: str, *args: str) -> str: """``git -C <cwd> <args>`` → stripped stdout, or ``""`` on any failure. ``bounded_git_probe`` - bounds post-kill cleanup on Windows (a killed git's suspended descendant held the pipes).""" + bounds post-kill cleanup on Windows (a killed git's suspended descendant held the pipes). + + Uses the shared :func:`bounded_git_probe` so the post-kill cleanup is bounded on Windows — a plain + ``subprocess.run(timeout=...)`` here deadlocked Desktop session readiness when a killed git left a + suspended descendant holding the pipe handles (issue #68609). + """ # A missing dir can only fail at the price of a fork; deleted worktrees dominate a long # session history's cwds, so the stat pays off. if not cwd or not os.path.isdir(cwd): diff --git a/tui_gateway/host_supervisor.py b/tui_gateway/host_supervisor.py index cd19373672..e7f4516f1e 100644 --- a/tui_gateway/host_supervisor.py +++ b/tui_gateway/host_supervisor.py @@ -37,6 +37,7 @@ _RESPAWN_WINDOW_SECS = 300.0 _SHUTDOWN_TIMEOUT_SECS = 10.0 # Late control-ack handlers: a compress that outlives its RPC waiter can run for the full # compression ceiling plus a stall-fallback retry, so keep registrations past that — bounded. +# See #97948. _LATE_CONTROL_TTL_SECS = 1800.0 _LATE_CONTROL_MAX = 64 # Host frames whose ``request_id`` resolves a pending/late control waiter. @@ -150,6 +151,8 @@ class HostSupervisor: self._pending_controls: dict[str, queue.Queue[dict]] = {} # request_id -> (registered_at, handler) for control waiters that timed out while their # host work still runs, so the eventual control.ack is not silently dropped. + # The host emits its control.ack whenever it finishes; without this the ack matched no queue and was + # silently dropped. See #97948. self._late_control_handlers: dict[str, tuple[float, Callable[[dict], None]]] = {} self._stderr_tail: list[str] = [] self._last_progress_counter = 0 diff --git a/tui_gateway/methods_bot_relay.py b/tui_gateway/methods_bot_relay.py index 59fae4ecb6..7571c33e6c 100644 --- a/tui_gateway/methods_bot_relay.py +++ b/tui_gateway/methods_bot_relay.py @@ -78,6 +78,7 @@ def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict: # fenced out by the single-owner lease and the payload dropped. Land the DM in the live # session via prompt.submit — the composer's choke point, so role alternation, persistence # and streaming behave as a typed message would. + # (Nested per method_ctx rebinding.) See #100523. from tools.bot_mode_probe import BOT_CHAT_TITLE live_home = _profile_home(resolved) want_home = str(live_home) if live_home is not None else None @@ -105,12 +106,16 @@ def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict: # Per-profile turn lock serializes with any other delivery turn into this profile and # covers only the turn window. Worst-case hold is lock wait (bot_mode.turn_wait_seconds, # default 120s) + the 600s turn timeout, doubled on one retry — callers tolerate ~1320s. + # Worst-case handler hold is lock wait (bot_mode.turn_wait_seconds, default 120s) + the 600s + # turn timeout below — doubled when the retry policy grants one bounded re-run — so clients + # calling bot_relay.deliver must tolerate ~1320s before assuming failure. See #93091. with acquire_turn_lock(root, resolved): proc = _run(resolved, tmp) if proc.returncode != 0: # Retry policy: transient classes re-run the SAME session once; context_overflow # too — the retried turn's pre-API compaction pass compacts the over-threshold # transcript first (no fresh session is minted). Auth/quota/config never retry. + # See #93091. from tools.bot_failure_reasons import ( RETRY_NONE, classify_agent_error, retry_action) if retry_action(classify_agent_error(_detail(proc))) != RETRY_NONE: diff --git a/tui_gateway/methods_complete.py b/tui_gateway/methods_complete.py index 328d3d9354..ab238e4027 100644 --- a/tui_gateway/methods_complete.py +++ b/tui_gateway/methods_complete.py @@ -304,6 +304,8 @@ def _(rid, params: dict) -> dict: return _err(rid, 4003, f"{pconfig.name} uses {pconfig.auth_type} auth — run `hermes model` to configure") if not pconfig.api_key_env_vars: return _err(rid, 4004, f"no env var defined for {pconfig.name}") + # Save the key to ~/.hermes/.env via the unified credential lifecycle so any stale config.yaml mirror of + # the previous key (model.api_key, custom_providers[*].api_key) is rotated in the same action (#62269). env_var = pconfig.api_key_env_vars[0] from hermes_cli.credential_lifecycle import save_provider_env_credential # also rotates stale config.yaml mirrors save_provider_env_credential(env_var, api_key) diff --git a/tui_gateway/methods_config.py b/tui_gateway/methods_config.py index f469215435..9fa325ef78 100644 --- a/tui_gateway/methods_config.py +++ b/tui_gateway/methods_config.py @@ -41,6 +41,7 @@ def _(rid, params: dict) -> dict: _reconcile_repo_discovery(pdb, conn, policy, _repo_discovery_policy_key(policy)) # `scan=true` (remote-gateway desktop): its native scan only sees its own filesystem, # so the host scans the policy roots so zero-session repos surface. + # See #81723. if params.get("scan") and policy["enabled"]: _scan_discovered_repos_remote(conn, policy) repos = _discover_repos_payload(db, conn=conn, include_cached=policy["enabled"]) diff --git a/tui_gateway/methods_config_set.py b/tui_gateway/methods_config_set.py index a18d5864de..a961286277 100644 --- a/tui_gateway/methods_config_set.py +++ b/tui_gateway/methods_config_set.py @@ -302,6 +302,11 @@ def _set_reasoning(rid, params, key, value, session): if scope == "global" or session is None: _write_config_key("agent.reasoning_effort", arg) if session is not None: + # /new is a full conversation boundary: session-scoped runtime overrides (/model, /reasoning, + # /fast) do NOT carry forward — the fresh agent re-derives model/provider, reasoning, and + # service tier from config.yaml (#48055, #23131). Session pins are cleared below so a rebuild + # can't resurrect them. (Global process state is still never touched — see the + # cross-session-contamination note in _apply_model_switch.) session.pop("create_reasoning_override", None) else: # session-scoped like the gateway's `/reasoning <level>`; a menu pick must not rewrite the global session["create_reasoning_override"] = parsed diff --git a/tui_gateway/methods_profiles.py b/tui_gateway/methods_profiles.py index 65165ce0b4..f56542448b 100644 --- a/tui_gateway/methods_profiles.py +++ b/tui_gateway/methods_profiles.py @@ -112,7 +112,10 @@ def _latest_message_preview(db, session_id): def _resurrect_recoverable_canonical(db, profile_path, session_id): """Un-archive an accidentally archived canonical row (judged read-only, written via a - short-lived writable handle); False otherwise.""" + short-lived writable handle); False otherwise. + + See #92687. + """ try: row = db.get_session(session_id) if not row or not row.get("archived"): @@ -134,13 +137,25 @@ def _resurrect_recoverable_canonical(db, profile_path, session_id): def _canonical_session_row(db, profile_path): """Summary of the profile's canonical "Bot Chat" row (identity is the NAME), or None. Lineages via ``get_compression_tip`` (NOT the resume walker's unmarked-child fallback); - worker sources count as absent. ``id`` is the registry row, ``resolved_id`` the live tip.""" + worker sources count as absent. ``id`` is the registry row, ``resolved_id`` the live tip. + + The canonical chat's identity is the NAME: the session titled exactly "Bot Chat" on this profile (core + UNIQUE(title) makes it a registry of at most one row). Complements ``last_session``: that field answers + "what is the newest conversation", this answers "where is the forever-chat" — so a roster row's preview + and its click target describe the same session (hermes-agent#88200) with no client-side pointer + involved. + """ try: row = db.get_session_by_title("Bot Chat") session_id = str((row or {}).get("id") or "").strip() if not session_id or _denied_source(row): return None # Archived = retired (absent), except accidental reaper archives: resurrect those. + # An archived canonical row usually means the user deliberately retired it — report absent. But the + # ws-orphan reaper / older agent cleanup can archive it by accident (#92687): resurrect those. Judge + # recoverability READ-ONLY first so the writable open (20s write-lock patience, the very stall this + # refactor removes from the 5s poll) is paid only in the rare accidental-archive case, then run the + # real predicate through unarchive_recoverable_session on a short-lived writable handle. if row.get("archived") and not _resurrect_recoverable_canonical(db, profile_path, session_id): return None tip = _try(lambda: db.get_compression_tip(session_id), None) or session_id @@ -158,7 +173,16 @@ def _canonical_session_row(db, profile_path): def _latest_profile_session_rows(db): """(newest human-facing session, newest worker session); the worker row lets rosters show a - profile as working (workers heartbeat ``last_activity_at`` every ≤60s).""" + profile as working (workers heartbeat ``last_activity_at`` every ≤60s). + + First element mirrors session.list's deny-list (drops ``tool`` sub-agent rows and ``kanban`` dispatcher + workers). Second element is the newest DENIED row — the freshest kanban/tool worker — so roster UIs can + show that a profile is actively working even though worker sessions never surface in conversation lists + (hermes-agent#90268). Workers heartbeat ``last_activity_at`` every ≤60s while running (#72016), so a + live worker's ``last_active`` stays fresh and the client can apply its own liveness window. Best-effort: + any failure (missing state.db, locked db, older schema) degrades to (None, None) rather than failing the + whole profiles.list call. + """ try: human = worker = None for s in db.list_sessions_rich(source=None, limit=20, order_by_last_active=True, compact_rows=True): @@ -269,6 +293,10 @@ def _mirror_voice_sections(path) -> bool: def _inherit_launch_model(path) -> bool: """Inherit launch model.provider/default when the new profile has none. Gate on the MODEL SECTION, not config.yaml existing: voice mirroring creates the file first.""" + # Gate on the MODEL SECTION being absent, not on config.yaml existing — earlier mirroring steps (voice + # sections, #85755) legitimately create the file first, and a file-existence gate silently skipped + # inheritance for every non-clone bot ("No inference provider configured" on first message, tester + # report). Clones bring their own model section and stay untouched. from hermes_cli.config import load_config_readonly, read_user_config_raw with _hermes_home_scope(path): dst_model = (read_user_config_raw() or {}).get("model") or {} @@ -457,6 +485,13 @@ def _configure_model(profile_dir, params, applied): if not (model and provider): return None confirm_message = None + # #95293 remainder: this is the Bots editor's model-switch path, and it used to write guarded + # (data-policy / expensive) models silently — the ONE surface that bypassed the selection guard every + # other switch path enforces. Same handshake contract as ``config.set model``: without + # ``confirm_expensive_model`` a guarded pick answers ``confirm_required`` + ``confirm_message`` and + # writes NOTHING; the client resends with ``confirm_expensive_model: true`` once the user confirms. A + # misbehaving guard must never break the save (treated as "no warning"), matching + # ``_apply_model_switch``. if not is_truthy_value(params.get("confirm_expensive_model", False)): warn = _lazy("hermes_cli.model_selection_guards", "combined_selection_warning") confirm_message = _try(lambda: getattr(warn(model, provider=provider or None), "message", None), None) diff --git a/tui_gateway/methods_projects.py b/tui_gateway/methods_projects.py index 0023eb4d5c..f8abc8fa6f 100644 --- a/tui_gateway/methods_projects.py +++ b/tui_gateway/methods_projects.py @@ -207,7 +207,14 @@ def _scan_discovered_repos_remote(conn, policy: dict) -> bool: """Backend-side disk scan of the policy roots into the discovery cache. Best-effort: failures log and leave the cache untouched. True only when the scan is authoritative (every root walked to completion, cap not hit) — only then is the cache write - ``replace=True``; a partial/errored scan must MERGE, or a failed refresh blanks the sidebar.""" + ``replace=True``; a partial/errored scan must MERGE, or a failed refresh blanks the sidebar. + + The desktop's native repo scan only runs on the local filesystem. On a remote gateway connection the + host must scan its own disk so repos with zero Hermes sessions still appear in the sidebar (#81723). + Mirrors the desktop's behavior: walk each root (bounded depth), find `.git` directories, record (root, + label) pairs into the discovery cache. + See #81723. + """ from hermes_cli import projects_db as pdb roots = policy.get("roots") or [] excludes = policy.get("exclude_paths") or [] diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index 99208d1f3c..0ffc6616eb 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -89,7 +89,12 @@ def _load_durable_truncation_history( def _resolve_truncate_row_id(session: dict, history: list, target_row_id: int): """Resolve ``truncate_before_row_id`` to ``(user_ordinal, history_index)``: in-memory stamps first, else the durable transcript mapped onto the live list by user ordinal. - Never falls back to a client-supplied ordinal — unknown row ids refuse.""" + Never falls back to a client-supplied ordinal — unknown row ids refuse. + + Prefer in-memory ``_row_id`` / ``row_id`` stamps. When a live turn rewrote ``session["history"]`` + without stamps (provider-format messages), load the session's durable transcript with + ``include_row_ids=True`` and map the matched user-turn ordinal onto the live list. See #82959. + """ if (hit := _find_user_turn_by_row_id(history, target_row_id)) is not None: return hit db_history = _load_durable_truncation_history(session) @@ -167,6 +172,10 @@ def _typed_stop_phrase_response(rid, text): if not (isinstance(text, str) and _voice_mode_enabled()): return None try: + # Typed bare stop phrase while backend voice mode is active ends the voice chat instead of sending + # "stop" to the agent — the typed twin of the spoken stop phrase (PR #73106), applied at the ONE + # server-side choke point every TUI submit passes through. (The desktop's voice conversation is + # renderer-owned and never flips the backend flag, so it handles its own typed stop client-side.) from tools.voice_mode import is_voice_stop_phrase if not is_voice_stop_phrase(text): return None @@ -376,6 +385,18 @@ def _truncate_history_for_submit(rid, sid, session, params, requested_rebind_ids if db is not None: try: # NULL session_key (old CLI-origin sessions) would trip an FK violation. + # active_only=True: replace only the live (active=1) rows. In-place compaction (#38763) + # keeps the pre-compaction transcript as active=0/compacted=1 rows under this same session + # key; a bare replace_messages() would DELETE that durable archive on every edit/regenerate + # — the same bug class #80216 fixed for /retry. On an uncompacted session all rows are + # active=1, so this is behaviorally identical to the full replace. archive_dropped: a rewind + # overwrites turns the user may not have meant to drop, and this write is the last step + # before they are gone — three reported incidents ended here with nothing to restore from + # (#70516, #80763, #82756). Soft-archiving keeps them on disk (active=0) and in the FTS + # index, so a mis-aimed cut is recoverable instead of terminal. The live transcript is + # unchanged. Fall back to session id when session_key is NULL — CLI-origin sessions created + # before the session_key default fix have no key, and replace_messages(None) triggers an FK + # violation. truncation_key = session.get("session_key") or sid old_active_row_ids = _row_ids_of(history) if requested_rebind_ids is not None: @@ -447,6 +468,9 @@ def _persist_session_row_for_submit(rid, session): def _run_after_agent_ready(rid, sid, session, text, display_kind, hosted_terminal_callback): """Turn thread body: patient wait for a deferred build (a slow build must not eat the accepted in-flight message), then run.""" + # The wait delivers the prompt when the still-running build completes, honors a cancel promptly, notices + # the user once past the slow threshold, and only errors when the build itself fails or the bounded cap + # expires. See #63078. err = _wait_agent_for_prompt(session, rid, sid) if err: # Terminal frame + retained snapshot (not a bare "error" event): the snapshot is @@ -583,6 +607,14 @@ def _(rid, params: dict) -> dict: # The truncation already happened inline above (memory + DB). isolated_response["result"].update(survivor_fields) return isolated_response + # An ordinal/id alone is not consent. A client that carries a leftover ordinal into an ORDINARY + # submit sends a request that is indistinguishable, field by field, from a real rewind — same + # method, same shape, an in-range target — and the cut it asks for is a destructive + # replace_messages() the user never requested (#80763: 296 -> 52 messages, 244 durable rows gone). + # Only the client knows whether this submit is a rewind/edit/regenerate, so it has to say so; refuse + # the cut when it doesn't. Consent is checked BEFORE target resolution: an unconfirmed + # (leaked-state) request must refuse with 4029 without paying the durable transcript read or + # heal-stamping live history dicts that row-id resolution performs. logger.warning( "compute-host dispatch failed for session %s; falling back inline: %s", sid, isolated_response["error"].get("message", "unknown error")) @@ -877,6 +909,16 @@ def _spawn_side_agent( def run(): session_tokens = _set_session_context(task_id, cwd=(cwd or _session_cwd(session))) + # Bug #50233: ephemeral agent threads don't inherit the session's HERMES_HOME override (the + # ContextVar set on the session-create thread doesn't propagate here), so a background turn under a + # non-default profile would run against the wrong home. Re-bind the override for the duration of + # this turn, exactly as the normal prompt turn does, and restore it afterward. + # Bug #50233: ephemeral preview-restart agent threads don't inherit the session's HERMES_HOME + # override (the ContextVar set on the session-create thread doesn't propagate here). Re-bind it for + # the duration of the turn, mirroring the normal prompt turn, then restore it. NOTE: we deliberately + # do NOT close this agent through task-wide process cleanup — the whole point of preview.restart is + # to leave a background server running under this task_id, and AIAgent.close() would kill every + # process for the task_id and tear down the very server the restart just started. profile_home = session.get("profile_home") home_token = set_hermes_home_override(profile_home) if profile_home else None try: @@ -1082,7 +1124,10 @@ def _(rid, params: dict) -> dict: def _approval_respond_session_fallback(params: dict): """Durable-identity fallback for a stale live sid (re-minted after a reconnect while the prompt stayed on screen): (1) the ``request_id`` against every live session's - pending approvals, then (2) ``session_id`` as a STORED id. Live session or None.""" + pending approvals, then (2) ``session_id`` as a STORED id. Live session or None. + + See #91684. + """ request_id = str(params.get("request_id") or "") if request_id: try: diff --git a/tui_gateway/methods_slash.py b/tui_gateway/methods_slash.py index 4588aa9ca5..5a7dc6751f 100644 --- a/tui_gateway/methods_slash.py +++ b/tui_gateway/methods_slash.py @@ -251,6 +251,9 @@ def _compress_live_with_feedback(sid: str, session: dict, agent, arg: str, *, sn if snapshot_kwargs: _compress_session_history(session, arg.strip() or None, **snapshot) else: + # The raw argument goes through unparsed: _compress_session_history (the choke point shared by + # all three manual-compress routes) parses the boundary-aware forms (here [N], up to here, + # --keep N) and does the partial head/tail split there (#35533). _compress_session_history(session, arg) except CompressionLockHeld as e: return describe_compression_lock_skip(e.holder) diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index fb81597aab..f4c849fdaf 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -590,6 +590,7 @@ def _cmd_moa(rid, params, session, name, arg): # Record the live identity for post-turn restore, then swap the agent's client in # place: session["model_override"] alone never switches an already-built agent. agent = session.get("agent") + # See #53444. session["moa_one_shot_restore"] = { "override": session.get("model_override"), "model": getattr(agent, "model", None), "provider": getattr(agent, "provider", None)} @@ -726,6 +727,10 @@ def _cmd_loop(rid, params, session, name, arg): def _cmd_undo(rid, params, session, name, arg): if not session: + # /undo [N]: back up N user turns (default 1), soft-delete the truncated rows on disk, and prefill + # the composer with the text of the user message we backed up to so it can be edited and + # resubmitted. N=1 is the Claude-Code-style single-step undo; /undo 3 backs up three user turns at + # once. See issue #21910. return _err(rid, 4001, "no active session to undo") if busy := _busy_error(rid, session, "undo"): return busy @@ -749,6 +754,7 @@ def _cmd_undo(rid, params, session, name, arg): # Notify memory providers (same hook /branch fires) with rewound=True so cached per-turn state invalidates. agent = session.get("agent") if agent is not None: + # See #6672 + #21910. mm = getattr(agent, "_memory_manager", None) for step in ( lambda: mm is not None and mm.on_session_switch( diff --git a/tui_gateway/methods_voice.py b/tui_gateway/methods_voice.py index 00b2395917..dd1f1d7041 100644 --- a/tui_gateway/methods_voice.py +++ b/tui_gateway/methods_voice.py @@ -279,7 +279,14 @@ def _speak_text_with_barge(text: str) -> None: def _voice_cfg_dict() -> dict: - """Shape-safe ``voice:`` block (no deep-merged defaults: any YAML shape possible; bad → {}).""" + """Shape-safe ``voice:`` block (no deep-merged defaults: any YAML shape possible; bad → {}). + + ``_load_cfg()`` does not deep-merge DEFAULT_CONFIG, so both the root AND ``voice`` may be any YAML + scalar / list / None. A hand-edit like ``voice: true`` or a malformed top-level config that parses to a + scalar would otherwise break ``.get("…")`` and take every ``voice.*`` branch down with it (Copilot + round-3..7 review on 19835). Coerce through ``isinstance`` at every level so malformed config falls back + to an empty dict instead of crashing /voice. See #19835. + """ cfg = _load_cfg() voice_cfg = cfg.get("voice") if isinstance(cfg, dict) else None return voice_cfg if isinstance(voice_cfg, dict) else {} @@ -723,6 +730,10 @@ def _(rid, params: dict) -> dict: set_voice_busy_probe(_any_session_running) # Shape-safe: malformed voice YAML falls back to documented defaults; an explicit numeric # max_recording_seconds <= 0 disables the cap (0.0). + # Shape-safe lookups: malformed ``voice:`` YAML (bool/scalar/list) must not crash /voice with a 5025 + # — fall back to VAD defaults. Exclude ``bool`` from the numeric check since Python's bool is a + # subclass of int — a hand-edit like ``silence_threshold: true`` would otherwise forward as ``1`` + # instead of falling back to the documented 200 / 3.0 defaults (Copilot round-12 on #19835). voice_cfg = _voice_cfg_dict() max_rec = _voice_cfg_number(voice_cfg.get("max_recording_seconds"), 120.0) # Hand the mic to STT if the wake detector holds it; a terminal capture event resumes it. diff --git a/tui_gateway/model_switch.py b/tui_gateway/model_switch.py index 02c458105c..56c2499636 100644 --- a/tui_gateway/model_switch.py +++ b/tui_gateway/model_switch.py @@ -178,6 +178,10 @@ def _commit_agent_switch(sid: str, session: dict, agent, result, current_model: except Exception as exc: # The in-place swap rolled the agent back and re-raised. Abort the whole commit (worker # restart, persist, marker, override, config write) or the session pins a broken model. + # Abort the commit: do NOT restart the slash worker, persist runtime, append the switch marker, set + # a session model_override, or persist to config — all of which would otherwise leave the session + # pinned to a broken model and kill the conversation on the next turn (#50163). A failed switch is a + # no-op; surface a clean error to the client. logger.warning("In-place model switch failed for TUI agent: %s", exc) raise ValueError(f"Model switch to {result.new_model} failed ({exc}); " f"staying on {getattr(agent, 'model', current_model)}.") from exc diff --git a/tui_gateway/prompt_turn.py b/tui_gateway/prompt_turn.py index 3cc33ad224..1322bdbbf0 100644 --- a/tui_gateway/prompt_turn.py +++ b/tui_gateway/prompt_turn.py @@ -85,6 +85,7 @@ def _admit_prompt_turn( """Ownership + liveness gate every turn source must cross; ``(images, agent)`` or None. Synthesized turns (auto-continue, wake-ups) call ``_run_prompt_submit`` directly — the bypass that once let a second backend run a duplicate turn.""" + # When the session already holds its lease this is a cheap dict check. See #94778. if (ownership_refusal := _ensure_active_session_slot(sid, session)) is not None: logger.info( "Refusing turn for session %s at _run_prompt_submit: %s", @@ -215,6 +216,13 @@ def _commit_turn_history( session["history"] = result["messages"] session["history_version"] = history_version + 1 return None + # History mutated externally during the turn. Check if the only mutation was a pivot marker the + # gateway itself inserted mid-turn (#76870). If so the agent output is still valid — merge it into + # the current history that now contains the marker. A personality change counts here too: unlike a + # model switch it has no pending queue, so `/personality` during a running turn lands immediately + # and used to read as a genuine desync, dropping the finished turn (#82756). + # _append_model_switch_marker strips prior markers in-place then appends a new one, so the delta is + # NOT a simple tail-slice — we must compare content, not indices. current_history = list(session["history"]) history_no_markers = [e for e in history if not _is_pivot_marker(e)] current_no_markers = [e for e in current_history if not _is_pivot_marker(e)] @@ -254,6 +262,9 @@ def _turn_outcome(result: Any) -> tuple[Any, str, str | None]: raw = f"Error: {result.get('error')}" # "Operation interrupted: waiting for model response (…)" is cancellation # metadata, not assistant prose (gateway/run.py and ACP suppress it too). + # "Operation interrupted: waiting for model response (…)" is cancellation metadata, not assistant prose. + # gateway/run.py and the ACP adapter already suppress this sentinel; without this the desktop paints it + # as the agent's reply whenever a stop/steer lands mid-request (#7921). if status == "interrupted" and isinstance(raw, str) and raw.strip().startswith( INTERRUPT_WAITING_FOR_MODEL_PREFIX): raw = "" @@ -446,6 +457,11 @@ def _prepare_turn_input(sid: str, session: dict, st: _TurnRun, text: Any, images # fall through to /dev/tty and hang the headless gateway (re-run is a no-op). _wire_callbacks(sid) if not st.one_turn_restore: + # Skip the config-model sync while a /model --once override is active: the once-model is + # intentionally not pinned as a session model_override (it must not persist), so without this guard + # the sync would see "agent model != config model" and clobber the once-override back to the config + # model before the turn runs (#29923 review defect). Any config.yaml change is adopted on the NEXT + # turn, after the finally-restore below. _apply_pending_model_switch(sid, session) _sync_agent_model_with_config(sid, session) _sync_agent_compression_with_config(sid, session) @@ -567,6 +583,7 @@ def _absorb_turn_result( # Undo a /moa one-shot through the switch path: resetting model_override alone # would leave the live client pinned to MoA after the in-place switch_model(). _restore = session.pop("moa_one_shot_restore", None) + # Restore the model the user was on before the /moa one-shot. See #53444. if isinstance(_restore, dict): _prev_override = _restore.get("override") _prev_model = _restore.get("model") @@ -596,6 +613,7 @@ def _absorb_turn_result( # Auto-compression may have rotated agent.session_id: sync session_key before # title/goal/finalize use it, keep pending_title (user intent), restart the slash # worker so worker-backed commands target the live session. + # Fix for #20001. _sync_session_key_after_compress( sid, session, clear_pending_title=False, restart_slash_worker=True) return status_note @@ -745,6 +763,12 @@ def _run_prompt_submit( # session_key and the agent's live session_id together. No prompt content is logged. _turn_started_monotonic = time.monotonic() logger.info( + # Desktop/TUI observability (#86647): this is the ONE INFO record proving a Desktop/TUI prompt was + # accepted by THIS process, and it ties together every id a rotation-mute trace needs — the UI + # session id, the gateway session_key, and the agent's live session_id (which compression rotates + # independently of the other two). Before this line a Desktop request left no trace in agent.log at + # all ("0 platform=desktop" — see #86647), so a muted window was structurally indistinguishable from + # a request that never arrived. "tui prompt accepted: ui_session=%s session_key=%s agent_session_id=%s " "kind=%s chars=%s images=%d", sid, session.get("session_key") or "", getattr(agent, "session_id", "") or "", diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 4f675ca7a6..43902800e5 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -115,6 +115,14 @@ def _ws_orphan_setting(env_var: str, cfg_key: str, default: float) -> float: return max(0.0, default) +# When a WebSocket client (the dashboard's embedded-chat tab / desktop app) disconnects, ``tui_gateway.ws`` +# detaches the transport but intentionally leaves the session parked so a quick reconnect can reattach it +# (see ws.py). That park is unbounded, though: a browser refresh spins up a brand-new ``session.create`` +# (new sid + a fresh _SlashWorker via _deferred_build) and never reattaches the OLD sid, so the old +# session's slash-worker subprocess lingers forever — one leaked python process per refresh (#38591 +# fallout). After this grace window, an orphaned WS session is interrupted if it is still running, then +# reaped once the normal turn-finalization path settles. Set to 0 to disable (park forever, pre-fix +# behaviour). def _resolve_ws_orphan_reap_grace() -> float: """Grace before an orphaned WS session is interrupted/reaped (0 = park forever): ws.py parks a disconnected session for a quick reattach, but a browser refresh mints a NEW sid and never @@ -129,6 +137,10 @@ _WS_ORPHAN_ACTIVITY_STALE_S = _ws_orphan_setting("HERMES_TUI_WS_ORPHAN_ACTIVITY_ _WS_ORPHAN_INTERRUPT_REAP_POLL_S = 1.0 # Interrupt-then-reap poll budget: a turn that never settles (thread hung in a syscall) would # reschedule the 1s poll forever; after this many polls, log loudly and force-reap. +# If an interrupted turn never settles (agent thread hung in a syscall, supervisor lost), each 1s poll would +# otherwise reschedule forever — trading the old leak-one-worker bug for leak-one-session-plus-timer-chain +# (review finding, PR #90373). After this many polls we log loudly and force-reap, mirroring the +# pre-existing stuck-`running` safety net's role of breaking the deadlock. _WS_ORPHAN_INTERRUPT_REAP_MAX_POLLS = 60 _TURN_SETTLE_BEFORE_CLOSE_SECONDS = 5.0 _DETAIL_SECTION_NAMES = ("thinking", "tools", "subagents", "activity") @@ -220,6 +232,11 @@ class _SlashWorker: argv = [sys.executable, "-m", "tui_gateway.slash_worker", "--session-key", session_key] + (["--model", model] if model else []) self._closed = False from hermes_cli._subprocess_compat import windows_hide_flags + # slash_worker runs the Hermes agent → needs provider credentials. Tier-1 secrets + # (gateway/GitHub/infra) are still stripped (#29157). Global-remote / multi-profile sessions: the + # worker must resolve config/skills/state against the session's profile home, not the gateway's + # launch HERMES_HOME (#40677). The override goes through the build_subprocess_env factory's `extra` + # (applied last, always wins) instead of a hand-rolled env["HERMES_HOME"] assignment. from tools.environments.local import build_subprocess_env # The worker runs the agent → needs provider credentials; tier-1 secrets (gateway/GitHub/ @@ -231,6 +248,9 @@ class _SlashWorker: # start_new_session: otherwise the worker inherits the gateway's pgid and mcp_tool's orphan # sweep, racing the spawn, killpg()s the TUI parent itself. errors="replace": bytes invalid # in the system locale (GBK Windows) must not raise UnicodeDecodeError in the drain threads. + # Prepend the Hermes venv bin dir and the user-local bin dir to PATH so slash_worker child processes + # can resolve Hermes-managed CLIs (browser-use, uvx) even when the parent gateway was launched with + # a minimal PATH (e.g. by the Desktop/Dashboard app). See #83845. self.proc = subprocess.Popen( argv, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding="utf-8", errors="replace", bufsize=1, cwd=os.getcwd(), env=env, @@ -372,6 +392,10 @@ def _transfer_db_to_agent(agent, db) -> bool: with contextlib.suppress(Exception): if agent is None or db is None or getattr(agent, "_session_db", None) is not db: return False + # Defense in depth (#91610): the shared launch handle must never transfer. Identity alone passes for + # it — a launch-profile agent IS holding that handle — and ownership would make session.close() tear + # down the process-wide database every other session shares. Refuse it explicitly even if a caller + # invokes the transfer incorrectly; the caller's own `owns_db` gate is the first line of defense. if db is _get_db(): logger.warning("Refused transfer of the shared launch SessionDB to a session " "agent — the caller's owns_db gate should have prevented this.") @@ -452,7 +476,24 @@ _served_profile_homes: set[Path] = set() def _profile_scoped(handler): """Bind ``params['profile']``'s HERMES_HOME around a handler (pets/projects resolve via - ``get_hermes_home``, so app-global remote mode still hits the focused profile). No-op for launch.""" + ``get_hermes_home``, so app-global remote mode still hits the focused profile). No-op for launch. + + Secondary-profile adapters are constructed inside ``_profile_runtime_scope`` (secret scope installed + + multiplex active) — the same discriminator the Buzz/SimpleX adapters use for this bug class (#98738). + The DEFAULT profile under multiplexing runs unscoped: ``os.environ`` holds its own bridge output there + and keeps its legacy precedence. + Same discriminator as the Buzz/SimpleX/Raft adapters (#98738): secret scope installed + multiplex + active. The DEFAULT profile under multiplexing (and every single-profile process) runs unscoped and + keeps its legacy ``os.environ`` precedence. + Secondary-profile adapters are constructed, connected, and reloaded inside ``_profile_runtime_scope`` + (secret scope installed + multiplex active) — the same discriminator as the Discord adapter's + ``_profile_scoped_config_load`` (#72348). The DEFAULT profile under multiplexing runs unscoped: + ``os.environ`` holds its own bridge output there and keeps its legacy precedence. + Secondary-profile adapters are constructed, connected, and reloaded inside ``_profile_runtime_scope`` + (secret scope installed + multiplex active) — the same discriminator the Buzz/SimpleX adapters use for + this bug class (#98738). The DEFAULT profile under multiplexing runs unscoped: ``os.environ`` holds its + own bridge output there and keeps its legacy precedence. + """ def wrapper(rid, params): home = _profile_home(params.get("profile") if isinstance(params, dict) else None) if home is None: @@ -483,7 +524,12 @@ def _configured_cwd_from_cfg(cfg: dict | None) -> str | None: def _profile_configured_cwd(profile_home: Path | None) -> str | None: """A non-launch profile's ``terminal.cwd`` from ITS config.yaml (fail-open → None): the process-global ``TERMINAL_CWD`` belongs to the *launch* profile, and load_config() resolves the ACTIVE profile, so - read the file directly through the _load_cfg pipeline.""" + read the file directly through the _load_cfg pipeline. + + A new session bound to another profile must take its workspace from THAT profile's config, not the stale + env var (issue #40334). Returns an absolute, existing directory, or None for placeholders / missing / + invalid paths. + """ if profile_home is None: return None with contextlib.suppress(Exception): @@ -615,7 +661,10 @@ def _pending_approval_request_payload(session_key: str) -> dict | None: def _emit_approval_request(sid: str, data: dict | None) -> None: """Emit ``approval.request`` with the command redacted: a credential-shaped value Tirith flagged would - otherwise echo verbatim to the TUI (third egress alongside chat platforms and the SSE/API stream).""" + otherwise echo verbatim to the TUI (third egress alongside chat platforms and the SSE/API stream). + + Reuse the shared gateway See #48456, #50767. + """ _emit("approval.request", sid, _approval_request_payload(data)) @@ -625,6 +674,7 @@ def _status_update(sid: str, kind: str, text: str | None = None): out_kind = kind if text is not None else "status" # Auto-compaction arrives as a generic "lifecycle" status; re-tag so drivers can show a # summarizing indicator — otherwise idle/preflight compaction looks like a hung turn. + # See #97239. if out_kind == "lifecycle": from agent.conversation_compression import is_compaction_progress_status if is_compaction_progress_status(body): @@ -760,7 +810,15 @@ def _wait_agent_for_prompt(session: dict, rid: str, sid: str) -> dict | None: first message IS the turn, while a cold build routinely outlives the flat 30s ceiling (timing out silently discarded it). Waits in short slices (cancel honored promptly), notifies once (keyed) past ``_AGENT_BUILD_SLOW_NOTICE_AFTER``, fails only on a dead build thread or the bounded cap. - Returns None on success OR cancel mid-wait (the caller's cancel branch owns that messaging).""" + Returns None on success OR cancel mid-wait (the caller's cancel branch owns that messaging). + + The flat 30s ``_wait_agent`` ceiling was a message-eating cliff (#63078): ``prompt.submit`` has already + returned ``{"status": "streaming"}``, the user's first message IS the turn in flight, and the deferred + agent build (MCP discovery with per-server retry backoff, synchronous model-metadata HTTP, skills + scanning) routinely outlives 30 seconds on cold starts. On timeout the old path emitted an error EVENT + and returned without ever calling ``_run_prompt_submit`` — the first message was permanently discarded + while the build finished successfully in the background, leaving the blank first session. + """ ready = session.get("agent_ready") if ready is None: return None @@ -992,6 +1050,12 @@ def _sess_nowait(params, rid): # Stale runtime id (reaped/evicted/TTL): the client should session.resume the STORED id. Logged so # "message vanished" reads as "arrived and was rejected". logger.warning("session-scoped RPC rejected: method=%s session_id=%r not in memory " + # A session-scoped RPC hit a runtime id the gateway no longer holds (detached on WS + # disconnect and orphan-reaped, LRU-evicted, or torn down after an idle TTL). The client + # is expected to recover via session.resume on the STORED session id, but a plain + # stale-id send leaves no trace anywhere when the resume never fires — every RPC in this + # class returned a silent 4001. Log it so a "message vanished" report is diagnosable as + # "request arrived and was rejected" instead of "request never arrived" (see #90428). "(detached/reaped runtime; client should resume the stored session), rid=%r", _current_rpc_method.get() or "?", sid, rid) return (None, _err(rid, 4001, "session not found")) @@ -1250,7 +1314,13 @@ def _tour_request(sid: str, payload: dict) -> str: """Bridge the tour tool callback onto _block without paying for a client that cannot answer: against an older app nobody calls ``tour.respond`` and each action would block the full deadline, stacking per turn. First action per session gets the short probe deadline; unanswered → bridge marked unavailable - for that session; once answered, the full deadline. Verdict lives on the record, so a new session re-probes.""" + for that session; once answered, the full deadline. Verdict lives on the record, so a new session re-probes. + + The renderer's ``tour.request`` handler ships in the desktop bundle, but the tool is offered by this + backend — and the two update on different clocks. The model then does what the schema tells it to and + tries the next action, so a single "give me a tour" turn stacks those waits (the timeouts reported + against #89620). + """ session = _sessions.get(sid) if session is None: # detached caller: throwaway record, plain bridge, unprobed ({} is falsy but a REAL record) session = {} @@ -1346,6 +1416,11 @@ def _resolve_startup_runtime() -> tuple[str, str | None]: # Bare billing buckets are not routable provider identities; restoring one as a session provider override # breaks resume. ``openrouter`` is deliberately NOT in this set (fully routable; agent_init's gate is a different set). +# (agent_init's fail-fast gate is a DIFFERENT set that also skips "openrouter" — there it means "default +# route, don't fail fast", not "unroutable".) ``openrouter`` is deliberately excluded here — it is a fully +# routable provider with its own API key and base_url. Sessions that used OpenRouter store +# ``billing_provider="openrouter"``; dropping it forces resume to the current global model (e.g. a custom +# endpoint), which is the wrong provider for the stored model. See #57588. from hermes_state import _BARE_BILLING_PROVIDERS @@ -1502,6 +1577,8 @@ def _persist_live_session_system_prompt(session: dict | None) -> None: agent, session_key, db = live # Re-bind the session's profile HERMES_HOME (the build's finally reset it → root profile's SOUL.md/skills) # and session context (on the RPC thread _SESSION_CWD is unset → the process TERMINAL_CWD would persist). + # Without this, _start_agent_build's finally block has already reset the override and the rebuilt prompt + # silently uses the root profile's SOUL.md and skills. See issue #50233. profile_home = session.get("profile_home") home_token = set_hermes_home_override(profile_home) if profile_home else None session_tokens = _set_session_context(session_key, cwd=_session_cwd(session)) @@ -1517,6 +1594,8 @@ def _persist_live_session_system_prompt(session: dict | None) -> None: # Stable leading text of the model-switch marker (builder + dedup); only the newest marker is meaningful. +# Only the newest marker is meaningful (it names the *currently* active model); older ones are stale and +# would otherwise be re-sent to the provider on every turn (#65891). _MODEL_SWITCH_MARKER_PREFIX = "[System: The active model for this chat has changed to " @@ -1537,7 +1616,10 @@ def _is_pivot_marker(entry: Any) -> bool: def _append_model_switch_marker(session: dict | None, *, model: str, provider: str) -> None: """Record a real system-history pivot after a live model switch. Only the newest marker is kept (each switch strips prior ones, so N switches leave one marker, not N re-sent every API call; self-healing - across resumes because the next switch collapses whatever a reload brought back).""" + across resumes because the next switch collapses whatever a reload brought back). + + See #65891. + """ session_key = str((session or {}).get("session_key") or "").strip() if not session_key: return @@ -1546,6 +1628,7 @@ def _append_model_switch_marker(session: dict | None, *, model: str, provider: s f"{_MODEL_SWITCH_MARKER_PREFIX}{model}{provider_part}. From this point forward, use this runtime " "metadata when answering questions about what model/provider is active.]") # A user message, not system: strict OpenAI-compatible providers (vLLM, Qwen) reject non-leading system messages. + # See #48338. entry = {"role": "user", "content": marker, "display_kind": "model_switch"} with session.get("history_lock") or contextlib.nullcontext(): history = session.setdefault("history", []) @@ -1621,7 +1704,10 @@ def _display_mouse_tracking(display: dict) -> str: def _load_reasoning_config(model: str = "") -> dict | None: """Via the shared chokepoint :func:`hermes_constants.resolve_reasoning_config` (per-model override > - global ``agent.reasoning_effort``; YAML False = disabled).""" + global ``agent.reasoning_effort``; YAML False = disabled). + + Closes #21256. + """ from hermes_constants import resolve_reasoning_config return resolve_reasoning_config(_load_cfg(), model) @@ -1750,6 +1836,9 @@ def _load_enabled_toolsets(platform: str | None = None) -> list[str] | None: cfg = load_config() # include_default_mcp_servers=True is the runtime variant (the agent must be able to call # default MCP servers); the config-editing variant would silently drop MCP tools from the TUI. + # Passing ``False`` here is the config-editing variant — used when we need to persist a toolset list + # without baking in implicit MCP defaults. Using the wrong variant at agent creation time makes MCP + # tools silently missing from the TUI. See PR #3252 for the original design split. enabled = _get_platform_tools(cfg, "cli", include_default_mcp_servers=True) if fallback_notice is not None: _tui_notice(fallback_notice) @@ -1778,6 +1867,12 @@ def _tool_lifecycle_required_for_ui(name: str) -> bool: def _restart_slash_worker(sid: str, session: dict): + # Close the slash-worker subprocess as part of finalize itself, not just in the callers. + # Defense-in-depth: every session-end path goes through _finalize_session (it's the single + # ``_finalized``-guarded chokepoint), so folding worker cleanup in here means a future code path that + # calls _finalize_session directly — without the surrounding _teardown_session / _shutdown_sessions + # worker.close() — can't reintroduce the #38095 leak. Idempotent: _SlashWorker.close() is + # poll()-guarded, so the explicit close() still in those callers is harmless. worker = session.get("slash_worker") if worker is None: return # never spawned one; spawning here would fork the per-worker MCP fleet for nothing @@ -1808,6 +1903,16 @@ def _get_usage(agent) -> dict: # context_used is *current-window* occupancy — never usage["total"] (cumulative: an external engine # showed 1.9m/120k clamped to 100%). Falsy last_prompt_tokens emits NO gauge; the -1 "compression # just ran" sentinel clamps to 0 (matches cli.py _get_status_bar_snapshot). + # Do NOT fall back to usage["total"] (cumulative lifetime session_total_tokens): for an external + # context engine that doesn't report last_prompt_tokens that substitution showed lifetime totals as + # the live context fill, yielding impossible readings such as 1.9m/120k clamped to 100% (#50421). + # Per the issue, populate context_used/percent only from a *real* current-occupancy value and "leave + # it unknown otherwise" — so a falsy last_prompt_tokens (0 or missing, i.e. an engine that doesn't + # track per-window occupancy) intentionally emits no gauge rather than a fabricated 0% or the old + # cumulative reading. The built-in compressor always reports a real last_prompt_tokens once a turn + # runs, so it is unaffected. Clamp the -1 "compression just ran, awaiting real usage" sentinel + # (conversation_compression.py) to 0 so the transitional turn reads as unknown (no gauge) instead of + # leaking context_used=-1. last_prompt = max(0, getattr(comp, "last_prompt_tokens", 0) or 0) ctx_max = getattr(comp, "context_length", 0) or 0 if ctx_max and last_prompt: @@ -1818,6 +1923,10 @@ def _get_usage(agent) -> dict: # Cache-hit ratio + rolling latency/tps (CLI status-bar parity). Omitted, not fabricated, when there is no # data (Codex reports no latency; zero cache reads shows no hit% rather than an alarming 0). with contextlib.suppress(Exception): + # Mirrors the classic CLI bar (cli.py _get_status_bar_snapshot / PR #98250): hit = + # session_cache_read_tokens / session_prompt_tokens (CanonicalUsage.prompt_tokens = input + + # cache_read + cache_write) latency/tps read the deque(maxlen=10) history maintained per API call in + # agent/conversation_loop.py. _prompt_total = int(getattr(agent, "session_prompt_tokens", 0) or 0) _cache_read = int(getattr(agent, "session_cache_read_tokens", 0) or 0) if _prompt_total > 0 and _cache_read > 0: @@ -2325,6 +2434,7 @@ def _claim_or_reuse_live(sid: str, session_key: str, record: dict, lease) -> tup """Register ``record`` as the live session for ``session_key`` under the resume lock, or — if a concurrent resume already won — release ``lease`` and return the winner for the caller to reuse.""" # A live runtime of the same stored id under ANOTHER profile is not a winner to reuse. + # See #100029. profile_home = record.get("profile_home") with _session_resume_lock: live = _find_live_session_by_key(session_key, profile_home) @@ -2416,6 +2526,9 @@ def _schedule_resume_hydration(sid: str, stored_id: str, db, *, close_db: bool = _emit("session.resume_progress", sid, {"phase": "history", "status": "loading"}) db.reopen_session(stored_id) raw_history, display_history, prefix = _load_resume_transcript(db, stored_id) + # Display keeps the full transcript; the model-fed history drops a dangling/interrupted + # tool-call tail so a session killed mid-loop does not replay the unanswered call forever + # (#29086). history = sanitize_replay_history(raw_history) if _sessions.get(sid) is not session: return @@ -2506,6 +2619,8 @@ def _session_lookup_key(session: dict, *, fallback: str = "") -> str: def _find_live_session_by_key(session_key: str, profile_home=_ANY_PROFILE) -> tuple[str, dict] | None: # Timestamp-based stored ids can exist in several profiles' stores; a bare-id match would hand # profile B's resume profile A's runtime, so profile-aware callers match on (profile_home, key). + # Profile-aware callers pass the home they resolved; the match must then be on (profile_home, + # session_key). See #100029. for sid, session in list(_sessions.items()): if (not session.get("_finalized") and _session_lookup_key(session, fallback=sid) == session_key and _live_profile_matches(session, profile_home)): @@ -2519,6 +2634,11 @@ def _fallback_session_info(session: dict) -> dict: return _session_info(agent) # The SESSION's own workspace, not the launch dir (wrong project in the desktop Files pane). `branch` is # always emitted ("" outside git) so a stale label clears; `desktop_contract` missing reads as "out of date". + # Reporting `_default_session_cwd()` here told a lazily-resumed session's client that its workspace was + # wherever the gateway process happened to start, so the desktop Files pane painted the wrong project + # even after the renderer rebound correctly (#71254). `branch` is always emitted ("" outside a git repo) + # so a client can clear a stale label instead of retaining it — the same contract `_lazy_session_info` + # above already follows. cwd = _session_cwd(session) return { "cwd": cwd, "branch": _git_branch_for_cwd(cwd), "project": _project_info_for_cwd(cwd), "lazy": True, @@ -2556,6 +2676,7 @@ def _live_visible_history(session: dict, db, in_memory_fallback: list[dict]) -> # conversation; without them a warm switch repainted the chat as summary + tail only. display = db.get_messages_as_conversation( key, include_ancestors=True, include_row_ids=True, include_compacted=True) + # See #92080. return _reconcile_display_with_live(display, in_memory_fallback) except Exception: logger.debug("live display projection read failed", exc_info=True) @@ -2573,9 +2694,12 @@ def _live_session_payload( # Every transport that showed this session (pop-outs resume the same sid); on disconnect the last # viewer becomes the transport instead of the drop sentinel. session.setdefault("viewers", {})[transport] = time.time() + # See #83716. if transport is not _detached_ws_transport: _cancel_ws_orphan_reap(sid) # the client is back — a pending ws-orphan reap must not fire if touch: + # #84417: do not re-fire the live turn's original user text from a stale server-queue + # self-duplicate after settle. session["last_active"] = time.time() in_memory_history = list(session.get("display_history_prefix") or []) + list(session.get("history") or []) inflight, queued = _inflight_snapshot(session), _queued_prompt_snapshot(session) @@ -2836,6 +2960,8 @@ def _spawn_tree_session_dir(session_id: str): # Per-session append-only JSONL index so `spawn_tree.list` needn't read every snapshot; a cache — a lost # line just means list() falls back to a directory scan. +# Read by `spawn_tree.list` so scanning doesn't require reading every full snapshot file (Copilot review on +# #14045). One JSON object per line. _SPAWN_TREE_INDEX = "_index.jsonl" @@ -2927,6 +3053,11 @@ def _respond(rid, params, key, *, allow_expired=False): def _session_processes(session: dict) -> list: """Background processes owned by this session (registry session_key match).""" + # Drain completion notifications that arrived during this turn. The background poller handles + # between-turn delivery; this is the safety net for events that arrived mid-turn. Ownership filter + # (#42674, #35652): a turn finishing in session B must not consume an event that belongs to session A. + # The registry requeues every addressed event this session cannot positively claim; the poller then + # delivers it to a live owner or drops an orphan. from tools.process_registry import process_registry key = str(session.get("session_key") or "") owned = [] diff --git a/tui_gateway/session_auto_continue.py b/tui_gateway/session_auto_continue.py index f6ab0c8670..291d494cd0 100644 --- a/tui_gateway/session_auto_continue.py +++ b/tui_gateway/session_auto_continue.py @@ -12,6 +12,9 @@ from .method_ctx import bind_module # finally; only a process death leaves it behind, so a marker at session.resume proves the turn never finished AND the # client never saw a terminal frame. Fresh: re-submit automatically (as the messaging gateway does). Stale: clear it # and let the partial transcript speak. +# If the interruption is fresh, re-submit the interrupted prompt automatically (the messaging gateway has +# done this for restart-interrupted sessions since #27856); if it's stale, clear the marker and let the +# recovered partial transcript speak for itself — the user can ask to continue manually. _AUTO_CONTINUE_FRESHNESS_MINUTES_DEFAULT = 15 @@ -91,6 +94,8 @@ def _maybe_schedule_auto_continue(sid: str, session: dict, session_key: str) -> session["last_active"] = time.time() # Ownership admission BEFORE message.start: a sibling backend sharing this HERMES_HOME may have written the # marker and still be mid-turn. Leave the marker so a later resume retries. + # Running the continuation anyway would be the double-writer this fence exists to prevent. See + # #94778. if _ensure_active_session_slot(sid, session) is not None: logger.info("auto-continue for %s refused: session has another live owner", session_key) with session["history_lock"]: @@ -125,6 +130,7 @@ def _enqueue_prompt(session: dict, text: Any, transport: Any, image_paths: list[ image_paths = list(image_paths or []) # Scrub live-turn self-duplicates first so the text merge below can't glue "{original}\n\n{later}" and re-fire the # original after a correction settles. + # See #84417. _drop_queued_duplicates_of_inflight_user(session) text_only = not image_paths and isinstance(text, str) # Never queue a text-only self-copy of the live prompt: draining it would restart it. @@ -145,7 +151,12 @@ def _enqueue_prompt(session: dict, text: Any, transport: Any, image_paths: list[ def _sanitize_queued_entry_vs_inflight_user(entry: Any, original: str) -> dict | None: """Drop (``None``) a text-only self-duplicate of the live user text, or rewrite a merged slot ``"{original}\\n\\n{later}"`` to ``later`` so the correction survives without re-firing the original. Image-bearing - envelopes are left alone (chronology is load-bearing).""" + envelopes are left alone (chronology is load-bearing). + + Returns ``None`` to drop the envelope, or a (possibly rewritten) dict to keep. A merged slot + ``"{original}\\n\\n{later}"`` (from ``_enqueue_prompt``'s consecutive text merge) is rewritten to just + ``later`` so a later correction is not lost and the original is not re-fired (#84417). + """ if not isinstance(entry, dict): return None text = entry.get("text") @@ -158,7 +169,13 @@ def _sanitize_queued_entry_vs_inflight_user(entry: Any, original: str) -> dict | def _drop_queued_duplicates_of_inflight_user(session: dict) -> None: """Remove server-queue copies of the live turn's original user text: a mid-turn ``prompt.submit`` of the same text - queued while redirect was unavailable must not drain and restart the original.""" + queued while redirect was unavailable must not drain and restart the original. + + A mid-turn ``prompt.submit`` of the same text can land in ``queued_prompt`` when redirect is not yet + available (model not active, build window, tool boundary). If the user then corrects the turn with a + different prompt via redirect, that stale self-duplicate must not ``_drain_queued_prompt`` after the + redirected turn completes — otherwise the original prompt restarts as a fresh agent turn (#84417). + """ if not (original := _ac_inflight_original(session)): return head = session.get("queued_prompt") @@ -250,6 +267,10 @@ def _handle_busy_submit(rid, sid: str, session: dict, text: Any, transport: Any, # Attachments need their own model invocation: queue without cancelling so the user gets both results in order. # ``steer`` must NEVER escalate to a hard interrupt: it would kill the live turn AND drop ``AIAgent._pending_steer`` # (earlier accepted steers); steer fall-throughs stay FIFO-queued. + # A burst of user messages while the agent is busy can land as a mix of accepted steers (stashed in + # ``AIAgent._pending_steer``) and fall-through queue envelopes (payload not steerable, ``steer()`` + # rejected/raised). A hard interrupt here kills the live turn AND ``AIAgent.interrupt()`` drops the + # pending steer buffer — silently destroying the earlier messages of the burst. See #86134. if mode == "interrupt" and not image_paths: _interrupt_busy_session(sid, session, agent) return _ok(rid, {"status": "queued"}) @@ -271,6 +292,7 @@ def _drain_queued_prompt(rid, sid: str, session: dict) -> bool: if int(session.get("_queued_prompt_generation", 0)) != queue_generation: # Generation bump cancelled the claim (Stop, compress re-anchor, …): don't dispatch, but restore the # envelope (claimed head first, then whatever advanced into the slot) so a legitimate follow-up isn't dropped. + # See #84417. advanced = session.get("queued_prompt") _ac_set_queue(session, [queued, *([advanced] if advanced else []), *(session.get("queued_prompts") or [])]) session["running"] = False diff --git a/tui_gateway/session_compression.py b/tui_gateway/session_compression.py index 0d76db9e3d..65b89db12d 100644 --- a/tui_gateway/session_compression.py +++ b/tui_gateway/session_compression.py @@ -22,7 +22,12 @@ def _tui_compression_config_signature(cfg: dict | None) -> tuple: def _compressor_ctor_default(name: str, fallback: Any) -> Any: """Default read off ContextCompressor.__init__'s REAL signature, so unset-key restoration uses the - construction path's derivation instead of a hardcoded copy that could drift.""" + construction path's derivation instead of a hardcoded copy that could drift. + + Unset restoration must go through the same derivation the construction path uses (#94724 review finding + on #95980) — pulling the default off ``ContextCompressor.__init__`` itself instead of hardcoding copies + keeps the two from drifting. + """ try: import inspect from agent.context_compressor import ContextCompressor @@ -67,7 +72,14 @@ _COMPRESSION_INT_KEYS = ( def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: """Update a live session's compressor in place from config.yaml. Every adopted key has UNSET semantics: a removed key restores the normalized default (or model-derived value) through the construction - path's own derivation — acting only on PRESENT keys would leave stale values active forever.""" + path's own derivation — acting only on PRESENT keys would leave stale values active forever. + + Every adopted key has UNSET semantics (#94724 review finding on the merged #95980): removing a key from + config.yaml restores the normalized default — or the model-derived value — on the next turn, through the + same derivation the construction path uses (ContextCompressor ctor defaults read off its real signature, + the Codex threshold autoraise via ``_resolve_compression_threshold``, context-length re-inference via + the deferred ``get_model_context_length`` resolution). + """ cfg = cfg if isinstance(cfg, dict) else {} compression = cfg.get("compression") if isinstance(cfg.get("compression"), dict) else {} model_cfg = cfg.get("model") if isinstance(cfg.get("model"), dict) else {} @@ -138,7 +150,11 @@ def _apply_live_compression_config(agent: Any, cfg: dict | None) -> None: def _sync_agent_compression_with_config(sid: str, session: dict) -> None: """Adopt compression.* / model.context_length edits at turn start (messaging gateways rebuild the - agent on these keys; Desktop/TUI keeps the live compressor, so it must be updated in place).""" + agent on these keys; Desktop/TUI keeps the live compressor, so it must be updated in place). + + Desktop/TUI only synced the model; the live compressor kept the threshold captured at agent creation + (#95151). + """ agent = session.get("agent") if agent is None: return @@ -185,7 +201,13 @@ def _compress_session_history( ) -> tuple[int, dict]: """Single choke point for all manual-compress routes. ``focus_topic`` is the RAW argument string after ``/compress``, parsed HERE (not per-route) so boundary forms (``here [N]``, ``up to here``, ``--keep N``) - trigger a partial compress on EVERY route instead of a FULL compress focused on the literal text.""" + trigger a partial compress on EVERY route instead of a FULL compress focused on the literal text. + + It is parsed here with :func:`parse_partial_compress_args` so boundary-aware forms (``here [N]``, ``up + to here``, ``--keep N``) trigger a partial compress — head summarized, most recent ``keep_last`` + exchanges kept verbatim — on EVERY route, mirroring cli.py's ``_manual_compress`` and + gateway/slash_commands.py (PR #35252). + """ from agent.conversation_compression import finalize_context_engine_compression_notification from agent.model_metadata import estimate_request_tokens_rough from hermes_cli.partial_compress import ( @@ -208,11 +230,18 @@ def _compress_session_history( head = history if approx_tokens is None: # Include system prompt + tool schemas so the figure reflects real request pressure. + # Include system prompt + tool schemas in the estimate — a transcript-only number understates real + # request pressure and can even appear to grow after compression because a dense handoff summary + # replaces many short turns (#6217). approx_tokens = estimate_request_tokens_rough( history, system_prompt=getattr(agent, "_cached_system_prompt", "") or "", tools=getattr(agent, "tools", None) or None ) # system_message=None: passing the cached prompt (already holding the identity block) would append the # identity twice. force=True: manual /compress bypasses the summary-failure cooldown like CLI/gateway. + # Pass system_message=None so AIAgent._compress_context rebuilds the system prompt cleanly via + # _build_system_prompt(None). Mirrors the CLI's _manual_compress fix for issue #15281. force=True: every + # caller of this helper is a manual /compress path (session.compress RPC, slash compress/compact, + # slash-worker mirror) — auto-compaction runs inside the agent loop, not here. try: compressed, _ = agent._compress_context( head, None, approx_tokens=approx_tokens, focus_topic=focus_topic or None, force=True, diff --git a/tui_gateway/session_history.py b/tui_gateway/session_history.py index 4230051e5c..998c79e16d 100644 --- a/tui_gateway/session_history.py +++ b/tui_gateway/session_history.py @@ -14,7 +14,15 @@ def _active_image_routing_identity(agent: Any) -> tuple[str, str]: def _build_image_ref_message(user_text: str, image_paths: list[str]) -> str: """Reference attached images by path so the agent analyzes them in-loop with ``vision_analyze``: pre- - analyzing with the auxiliary vision model blocked submit 60-90s/photo and poisoned auto-titles.""" + analyzing with the auxiliary vision model blocked submit 60-90s/photo and poisoned auto-titles. + + This used to pre-analyze every image with the auxiliary vision model *before* the turn was dispatched + (``_enrich_with_attached_images``): serial blocking calls on the submit path — 60-90s per large photo — + with failures silently swallowed and an interrupt during the window killing the turn with zero API calls + (#83291). It also prepended the vision description to the first user message, poisoning session + auto-titles (#82339). The CLI never gates turn dispatch on vision like this, which is why the same + message was seconds there and minutes on desktop. + """ prefix = "\n\n".join( f"[The user attached an image: {p.name}]\n[Examine it with the vision_analyze tool using image_url: {p}]" for p in map(Path, image_paths) if p.exists() @@ -120,7 +128,11 @@ def _is_text_only_busy_payload(content: Any) -> bool: def _is_display_hidden_marker(role: str | None, text: str) -> bool: """Gateway notices (model-switch, personality) persist as role=user ``[System: …]`` rows so strict providers accept them mid-history; they must never render as a user bubble. Filtering in this one projection hides - them everywhere (raw marker stays in ``session["history"]``) and keeps the desktop's user ordinals stable.""" + them everywhere (raw marker stays in ``session["history"]``) and keeps the desktop's user ordinals stable. + + It also removes the stored marker from the payload the desktop reconciles against, so it can no longer + shift user-message ordinals and duplicate the optimistic prompt (#67603). + """ return role == "user" and text.lstrip().startswith("[System:") @@ -199,6 +211,7 @@ def _history_to_messages(history: list[dict]) -> list[dict]: continue msg = {"role": role, "text": content_text} # Authoring time (Unix seconds) for display.timestamps; display-only. + # Display-only: never fed back into model context. See #41531. ts = m.get("timestamp") if isinstance(ts, (int, float)) and ts > 0: msg["timestamp"] = float(ts) @@ -333,7 +346,19 @@ def _strip_prompt_echo(message: str, prompt: Any) -> str: def _turn_failure_detail(error: Any, reason: Any = None, prompt: Any = None) -> str: """Why a turn failed, for the ``tui turn finished`` bookend: ``""`` when nothing to say, else a fragment with its own leading space. ``redact_sensitive_text`` removes credentials; ``_strip_prompt_echo`` removes a 4xx - body quoting ``prompt`` back. This record may gain failure detail, never the user's own content.""" + body quoting ``prompt`` back. This record may gain failure detail, never the user's own content. + + 86865 added the bookend to trace compression rotations, so it logs identities and a coarse ``status`` + and deliberately logs no content. 89117 is what the missing cause costs: a report consisting of two + lines reading ``status=error error_retained=True duration=0.9s`` with no way to tell a provider 4xx from + a budget wall from a crashed finalizer. The returned-error path -- the one a 0.9 s failure almost always + takes -- emits no other log line at all; only the exception path prints to stderr, which is why the + quiet failures are the ones that get filed. See #86865, #89117. + Content discipline follows #86865's, and it takes two separate steps because it is two separate + contracts. It does nothing about a 4xx body that quotes the request back, because ordinary private prose + is not pattern-shaped -- so ``_strip_prompt_echo`` removes that separately, using the submitted + ``prompt`` itself as the thing to look for. + """ reason_text = str(reason or "").strip() message = str(error or "").strip() if isinstance(error, BaseException): diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index 0868b05d49..c27b6e5419 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -36,6 +36,10 @@ def _claim_active_session_slot( except Exception as exc: logger.warning("Failed to claim active session slot: %s", exc) # Fail CLOSED: an errored claim has NOT proven the session unowned; lease-less = silent double-writer hole. + # Fail CLOSED regardless of surface: per-session exclusivity is a correctness guarantee (see + # PER_SESSION_EXCLUSIVE_SUBMIT), and a claim that errors out has NOT proven the session is unowned. + # Proceeding without a lease here is the silent double-writer hole flagged in the #94595 review + # (blocker 2). return (None, _SESSION_OWNERSHIP_UNAVAILABLE) @@ -142,6 +146,7 @@ def _transfer_active_session_slot(sid: str, session: dict, *, new_session_id: st return False # Fallback (entry pruned / pid-check transiently failed): reserve the new slot BEFORE releasing the old one so # a gateway at the cap can't grab the freed slot and leave this session lease-less; on failure KEEP the old lease. + # See #49041. new_lease, limit_message = _claim_active_session_slot( new_session_id, live_session_id=sid, surface=_session_source(session), profile_home=session.get("profile_home")) if new_lease is None: @@ -157,6 +162,8 @@ def _transfer_active_session_slot(sid: str, session: dict, *, new_session_id: st # Sources this backend must never end in state.db: the messaging gateway owns those sessions and the TUI is only # a viewer (ending one causes the Groundhog Day loop, see _finalize_session). Self-created/CLI sources are NOT gateway-owned. +# Sources the TUI backend itself creates ("tui", plus whatever a client passes as its own ``source``) and +# the CLI's own sessions are NOT gateway-owned. See #60609. _NON_GATEWAY_SOURCES = frozenset({ "", "tui", "cli", "webui", "desktop", "cron", "kanban", "subagent", "test", "local", "acp", "webhook", "api_server", "msgraph_webhook"}) @@ -227,6 +234,7 @@ def _finalize_session(session: dict | None, end_reason: str = "tui_close") -> No _notify_session_boundary("on_session_finalize", session_id, _session_source(session)) # End the state.db row so it doesn't linger as a ghost in /resume. Use session_id (agent.session_id), not # session_key: after compression the key may be the stale ended parent while session_id is the live continuation. + # Fix for #20001. if _desktop_automatic_cleanup and not session_id: _release_active_session_slot(session) _lifecycle_guard = (_other_runtime_lease_guard(session_id, session) @@ -391,7 +399,10 @@ def _interrupt_session_turn(sid: str, session: dict, *, request_id: str | None = def _session_has_active_delegations(sid: str, session: dict | None = None) -> bool: """True when UI session ``sid`` still owns live background work — by live UI sid AND, when the TUI owns the durable - lifecycle (never for gateway-viewer tabs), by session_key so a delegation from an earlier tab keeps it alive.""" + lifecycle (never for gateway-viewer tabs), by session_key so a delegation from an earlier tab keeps it alive. + + See #60609. + """ if session is None: with _sessions_lock: session = _sessions.get(sid) @@ -435,7 +446,12 @@ def _cancel_ws_orphan_reap(sid: str) -> None: def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: """Whether a detached RUNNING turn's activity clock (``_touch_activity``) is still fresh — the reaper must NOT interrupt healthy detached work (closed laptop). Conservative: disabled threshold, missing/opaque agent, unreadable - summary or never-stamped clock all report NOT fresh (eligible for interrupt-at-grace) to keep the wedged-turn net.""" + summary or never-stamped clock all report NOT fresh (eligible for interrupt-at-grace) to keep the wedged-turn net. + + Reuses the agent's existing activity summary (``_touch_activity`` is stamped by API waits, stream + tokens, and tool heartbeats — the same clock the turn-liveness watchdog samples; see + agent/turn_liveness.py). See #100325, #98028. + """ if _WS_ORPHAN_ACTIVITY_STALE_S <= 0: return False if not callable(summary_fn := getattr(session.get("agent"), "get_activity_summary", None)): @@ -478,6 +494,7 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: # Mid-turn detached sessions must never drop the single Timer: interrupt once after grace, then poll # until turn-finalization settles. polls = current["_client_gone_interrupt_polls"] = int(current.get("_client_gone_interrupt_polls") or 0) + 1 + # See #85578. if polls > _WS_ORPHAN_INTERRUPT_REAP_MAX_POLLS: # Never settled inside the budget — force-reap rather than park forever. logger.error( @@ -543,6 +560,7 @@ def _close_sessions_for_transport(transport, *, end_reason: str = "ws_disconnect # `hermes --tui` keeps real _stdio. UNLESS another window (pop-out viewer) still shows the session: # re-bind to the most recent surviving viewer instead. viewers = current.get("viewers") or {} + # See #83716. viewers.pop(transport, None) live = [vt for vt, ts in sorted(viewers.items(), key=lambda kv: kv[1]) if not _transport_is_dead(vt)] if live: diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index 2c080c60ec..10965da960 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -293,7 +293,10 @@ def _kb_poll_board(_kb, slug: str, session_key: str) -> list: def _collect_kanban_notifications(session: dict) -> list: """Claim unseen terminal kanban events for this session's ``platform="tui"`` subscriptions (``kanban_create`` auto-subscribes with ``chat_id=HERMES_SESSION_KEY``; no "tui" messaging adapter exists, so this poller is the - delivery path). Same atomic cursor-claim as the gateway notifier: exactly-once even if a gateway polls the same DB.""" + delivery path). Same atomic cursor-claim as the gateway notifier: exactly-once even if a gateway polls the same DB. + + See #59890. + """ session_key = str(session.get("session_key") or "") if not session_key or session.get("_finalized"): return [] @@ -399,7 +402,12 @@ def _notif_handle_event(sid, session, evt, emitted, registry, fmt, deferred) -> def _notification_poller_loop(stop_event: threading.Event, sid: str, session: dict) -> None: """Daemon thread (started by _init_session()) that drains the process-global completion_queue for this session (ownership routing: _notif_handle_event) and polls ``kanban_notify_subs`` every ``_KANBAN_POLL_SECONDS`` — the - delivery path for platform="tui" rows.""" + delivery path for platform="tui" rows. + + Also polls ``kanban_notify_subs`` every ``_KANBAN_POLL_SECONDS`` for this session's TUI kanban + subscriptions and delivers terminal task events the same way (status.update + agent turn) — the delivery + path tools/kanban_tools.py documents for platform="tui" rows (issue #59890). + """ from tools.process_registry import process_registry, format_process_notification queue = process_registry.completion_queue emitted: set = set() # dedup re-queued events so one completion isn't emitted 50 times while busy diff --git a/tui_gateway/session_reaper.py b/tui_gateway/session_reaper.py index 2cc7a3bbc5..a8eaccd91e 100644 --- a/tui_gateway/session_reaper.py +++ b/tui_gateway/session_reaper.py @@ -22,7 +22,10 @@ from .method_ctx import bind_module def _flush_session_messages(session: dict | None) -> bool: """Best-effort durable flush of one session's transcript via ``agent._persist_session`` (same marker-deduped - contract as ``_finalize_session``: repeated calls never duplicate rows).""" + contract as ``_finalize_session``: repeated calls never duplicate rows). + + See #13121. + """ agent = session.get("agent") if session else None snapshot = getattr(agent, "_session_messages", None) if hasattr(agent, "_persist_session") else None if not snapshot: @@ -245,6 +248,8 @@ def _schedule_session_cap_enforcement() -> None: # `ended_at IS NULL` forever. Scheduled once per process from both gateway entry points (stdio `entry.main`, WS # sidecar `handle_ws`). state.db is shared by sibling processes on the same profile, so eligibility is # conservative. Disable via `dashboard.startup_orphan_sweep: false`. +# This is the startup complement every other resource type already has (docker_orphan_reaper, compression +# orphans). See #65194. _ORPHAN_SWEEP_SOURCES = ("tui", "desktop", "subagent") _startup_orphan_sweep_ran = False _startup_orphan_sweep_lock = threading.Lock() @@ -367,7 +372,10 @@ def _start_backend_heartbeat_refresher() -> None: def _schedule_startup_orphan_sweep() -> None: """Schedule the once-per-process startup orphan sweep, delayed by the WS-orphan grace window so a client reconnecting right after a restart can ``session.resume`` its row first. Grace 0 (park forever), TTL 0 and - ``dashboard.startup_orphan_sweep: false`` all suppress the sweep.""" + ``dashboard.startup_orphan_sweep: false`` all suppress the sweep. + + See #65194. + """ global _startup_orphan_sweep_ran if _WS_ORPHAN_REAP_GRACE_S <= 0 or _SESSION_TTL_S <= 0 or not _session_orphan_reaper_enabled(): return diff --git a/tui_gateway/session_workdir.py b/tui_gateway/session_workdir.py index 04bb4fb651..8f8a9748ad 100644 --- a/tui_gateway/session_workdir.py +++ b/tui_gateway/session_workdir.py @@ -253,7 +253,10 @@ def _ensure_session_db_row(session: dict) -> bool: A cwd the user *chose* is always persisted. Otherwise the launch directory stands in only for terminal sessions (the user deliberately ``cd``'d there; dropping it left the sidebar with no cwd AND no git_repo_root); desktop - launch dirs (``/``, home) stay null -> "No workspace".""" + launch dirs (``/``, home) stay null -> "No workspace". + + See #98924. + """ if not (key := session.get("session_key")): return # Persist into the session's own profile db (global remote mode), not the launch profile's — otherwise the unified @@ -265,6 +268,8 @@ def _ensure_session_db_row(session: dict) -> bool: if db is None: # Fail loud ONLY when the store failed to open (_db_error records the SessionDB open exception); None with # no recorded error means "no store in this context" -> True. + # A None db with no recorded error means "no store in this context" (degraded harness, store + # deliberately absent) — that keeps the pinned best-effort contract and stays True. See #98924. return _db_error is None row_model, model_config = _workdir_row_model_config(session) try: @@ -273,6 +278,10 @@ def _ensure_session_db_row(session: dict) -> bool: parent_session_id=session.get("parent_session_id") or None, cwd=_persisted_session_cwd(session), # Self-describing rows: aggregators merging several profile DBs can't rely on which file a row came # from; a NULL is only repaired by the one-shot backfill. + # Stamp the launch profile explicitly instead of leaving NULL — NULL is exactly what the + # #94724 legacy-owner backfill exists to repair, and rows minted AFTER that one-shot + # backfill ran stayed NULL forever: profile-keyed matching then drops them from the sidebar + # and deep links can't resolve them (#99222). profile_name=Path(profile_home).name if profile_home else _current_profile_name()) # Born hidden (session.create hidden=true, or set_hidden before the row existed): apply the deferred intent. if session.get("pending_hidden"): @@ -319,6 +328,8 @@ def _persist_branch_seed(session: dict) -> None: try: # Chunked so each BEGIN IMMEDIATE stays short (a seed can be hundreds of rows); a mid-copy failure leaves a # partial seed with _branch_seed_persisted unset. + # Bounded-chunk transactions (see #23254): a branch seed can be hundreds of rows; chunking keeps + # each BEGIN IMMEDIATE short so concurrent writers aren't starved. db.append_messages_batch( key, [{"role": msg.get("role", "user"), **{f: msg.get(f) for f in _WORKDIR_SEED_FIELDS}} for msg in seed], chunk_rows=500) diff --git a/tui_gateway/slash_worker.py b/tui_gateway/slash_worker.py index 3dc0bbbd67..a478adec12 100644 --- a/tui_gateway/slash_worker.py +++ b/tui_gateway/slash_worker.py @@ -7,6 +7,9 @@ Protocol: reads JSON lines from stdin {id, command}, writes {id, ok, output|erro # top-level modules: this worker is spawned as ``-m tui_gateway.slash_worker`` with the user's CWD, so # ``import cli`` would otherwise resolve ``utils`` to a colliding local package and crash the child in a # retry loop. ``hermes_bootstrap`` lives at the repo root (no collision risk), so importing it first is safe. +# ``hermes_bootstrap`` lives at the repo root, so importing it is safe before the guard runs (its name won't +# collide with a user package), and it owns the canonical path-hardening logic shared with the other entry +# points — #51693 added the guard to ``entry.py``/``acp_adapter/entry.py`` but missed this child. import hermes_bootstrap hermes_bootstrap.harden_import_path() @@ -41,7 +44,10 @@ def _is_orphaned(original_ppid, getppid=os.getppid) -> bool: def _prepare_slash_worker_runtime() -> None: """Start bounded MCP discovery before HermesCLI snapshots tools: each slash_worker child is its - own process — the parent ``hermes serve`` discovery thread does not populate this registry.""" + own process — the parent ``hermes serve`` discovery thread does not populate this registry. + + See #61891. + """ from hermes_cli.mcp_startup import start_background_mcp_discovery, wait_for_mcp_discovery start_background_mcp_discovery(logger=logger, thread_name="slash-worker-mcp-discovery") wait_for_mcp_discovery() diff --git a/tui_gateway/tool_progress.py b/tui_gateway/tool_progress.py index cf3153dce8..e7f362a882 100644 --- a/tui_gateway/tool_progress.py +++ b/tui_gateway/tool_progress.py @@ -8,6 +8,10 @@ from .method_ctx import bind_module # Verbose tool text is capped to the Ink render budget (a hair more, so the "[omitted …]" label # stays informative): unbounded output fed a render-tree blowup that OOM-killed the TUI parent. # Full output stays in the agent context and the SQLite session, untouched. +# Tool Args/Result text shipped to the TUI for the verbose trail line. The TUI renders only a small +# persisted preview (ui-tui VERBOSE_TRAIL_MAX_CHARS), kept all session and expanded by default — so shipping +# more than that is pure pipe waste AND feeds the Ink render-tree blowup that silently OOM-killed the TUI +# parent (#34095). _TUI_VERBOSE_TEXT_MAX_CHARS = 1_000 _TUI_VERBOSE_TEXT_MAX_LINES = 16 @@ -270,6 +274,9 @@ def _progress_moa_reference(sid, name, preview, kw): def _progress_moa_progress(sid, name, preview, kw): # Drives the status-bar `MOA: 2/3 refs done`; both counters required for deterministic rendering. refs_done, refs_total = kw.get("moa_refs_done"), kw.get("moa_refs_total") + # Per-reference completion — drives the status-bar progress indicator (`MOA: 2/3 refs done`) requested + # in issue #59546. Only emitted when both counters are present so the client can render + # deterministically. if refs_done is None or refs_total is None: return _emit("moa.progress", sid, {"label": str(name or ""), "refs_done": int(refs_done), "refs_total": int(refs_total)}) diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index e2d2d85a24..fb7e866a97 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -49,6 +49,8 @@ def _sanitize_ws_text(text: str) -> str: ``json.dumps(..., ensure_ascii=False)`` happily emits lone UTF-16 surrogates; Starlette's ``send_text`` then raises ``UnicodeEncodeError``, which used to latch the whole connection closed. Same U+FFFD replacement every other Hermes transport applies. + + See #97288. """ return _sanitize_surrogates(text) if text else text diff --git a/utils.py b/utils.py index b9a9c534b2..ed9e35f37a 100644 --- a/utils.py +++ b/utils.py @@ -250,6 +250,9 @@ class IndentDumper(yaml.SafeDumper): PyYAML emits "indentless" sequences while ruamel (:func:`atomic_roundtrip_yaml_update`) indents them; mixing both in one ``config.yaml`` makes stricter parsers like ``js-yaml`` reject it, so every write path is forced to the same shape. + + Forcing ``indentless=False`` aligns the two serializers so all write paths emit byte-identical layouts + (#31999). """ def increase_indent(self, flow=False, indentless=False): # noqa: ARG002 @@ -301,6 +304,7 @@ def atomic_roundtrip_yaml_update(path: Union[str, Path], key_path: str, value: A # Honor escaped dots and prefer existing literal dotted keys (model IDs like ``glm-5.3``) over # blind splitting — same navigation as ``hermes config set``'s ``_set_nested``; otherwise # /model + TUI persistence wrote ``glm-5: {'3': ...}`` phantom siblings. + # See #91607. from hermes_cli.config import _greedy_literal_match, _split_key_path path = Path(path)