# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
549 lines
20 KiB
Python
549 lines
20 KiB
Python
"""Unit tests for the plugin LLM facade (``agent.plugin_llm``).
|
|
|
|
These tests exercise the trust gate, JSON parsing, schema validation,
|
|
image input encoding, and the auxiliary-client invocation contract.
|
|
The auxiliary client itself is stubbed via ``make_plugin_llm_for_test``
|
|
so we don't hit real providers.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import base64
|
|
from types import SimpleNamespace
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from agent.plugin_llm import (
|
|
PluginLlmCompleteResult,
|
|
PluginLlmImageInput,
|
|
PluginLlmStructuredResult,
|
|
PluginLlmTextInput,
|
|
PluginLlmTrustError,
|
|
_build_structured_messages,
|
|
_check_overrides,
|
|
_coerce_allowlist,
|
|
_parse_structured_text,
|
|
_strip_code_fences,
|
|
_TrustPolicy,
|
|
make_plugin_llm_for_test,
|
|
)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _fake_response(text: str, *, prompt: int = 4, completion: int = 6) -> SimpleNamespace:
|
|
"""Build an OpenAI-shaped response with the given text + token usage."""
|
|
return SimpleNamespace(
|
|
choices=[
|
|
SimpleNamespace(
|
|
message=SimpleNamespace(content=text, role="assistant"),
|
|
finish_reason="stop",
|
|
)
|
|
],
|
|
usage=SimpleNamespace(
|
|
prompt_tokens=prompt,
|
|
completion_tokens=completion,
|
|
total_tokens=prompt + completion,
|
|
),
|
|
)
|
|
|
|
def _trusted_policy(plugin_id: str = "trusted-plugin", **overrides: Any) -> _TrustPolicy:
|
|
defaults = dict(
|
|
allow_provider_override=True,
|
|
allowed_providers=None,
|
|
allow_any_provider=True,
|
|
allow_model_override=True,
|
|
allowed_models=None,
|
|
allow_any_model=True,
|
|
allow_agent_id_override=True,
|
|
allow_profile_override=True,
|
|
)
|
|
defaults.update(overrides)
|
|
return _TrustPolicy(plugin_id=plugin_id, **defaults)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Trust gate
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestTrustGate:
|
|
|
|
def test_overrides_independent(self):
|
|
"""Each override is gated independently — turning on
|
|
``allow_model_override`` does NOT also grant provider override."""
|
|
policy = _TrustPolicy(
|
|
plugin_id="model-only",
|
|
allow_model_override=True,
|
|
allow_any_model=True,
|
|
)
|
|
# model alone passes
|
|
_, m, _, _ = _check_overrides(
|
|
policy,
|
|
requested_provider=None,
|
|
requested_model="gpt-4o",
|
|
requested_agent_id=None,
|
|
requested_profile=None,
|
|
)
|
|
assert m == "gpt-4o"
|
|
# provider alone is still denied
|
|
with pytest.raises(PluginLlmTrustError, match="cannot override the provider"):
|
|
_check_overrides(
|
|
policy,
|
|
requested_provider="anthropic",
|
|
requested_model=None,
|
|
requested_agent_id=None,
|
|
requested_profile=None,
|
|
)
|
|
|
|
def test_provider_allowlist_accepts_listed_case_insensitively(self):
|
|
policy = _TrustPolicy(
|
|
plugin_id="restricted",
|
|
allow_provider_override=True,
|
|
allowed_providers=frozenset({"openrouter"}),
|
|
allow_any_provider=False,
|
|
)
|
|
p, _, _, _ = _check_overrides(
|
|
policy,
|
|
requested_provider="OpenRouter",
|
|
requested_model=None,
|
|
requested_agent_id=None,
|
|
requested_profile=None,
|
|
)
|
|
assert p == "OpenRouter"
|
|
|
|
def test_no_overrides_passes_through(self):
|
|
policy = _TrustPolicy(plugin_id="locked")
|
|
result = _check_overrides(
|
|
policy,
|
|
requested_provider=None,
|
|
requested_model=None,
|
|
requested_agent_id=None,
|
|
requested_profile=None,
|
|
)
|
|
assert result == (None, None, None, None)
|
|
|
|
def test_all_overrides_when_fully_trusted(self):
|
|
policy = _trusted_policy()
|
|
result = _check_overrides(
|
|
policy,
|
|
requested_provider="openrouter",
|
|
requested_model="anthropic/claude-3-5-sonnet",
|
|
requested_agent_id="ada",
|
|
requested_profile="work",
|
|
)
|
|
assert result == ("openrouter", "anthropic/claude-3-5-sonnet", "ada", "work")
|
|
|
|
class TestAllowlistCoercion:
|
|
|
|
def test_list_of_strings(self):
|
|
ranges, allow_any = _coerce_allowlist(["A", "B"])
|
|
assert ranges == frozenset({"a", "b"})
|
|
assert allow_any is False
|
|
|
|
def test_star_alone_means_any(self):
|
|
ranges, allow_any = _coerce_allowlist(["*"])
|
|
assert ranges == frozenset()
|
|
assert allow_any is True
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Structured message building
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestStructuredMessageBuilding:
|
|
def test_text_only_input(self):
|
|
messages = _build_structured_messages(
|
|
instructions="Extract the action items",
|
|
inputs=[PluginLlmTextInput(text="meeting notes go here")],
|
|
json_mode=False,
|
|
json_schema=None,
|
|
schema_name=None,
|
|
system_prompt=None,
|
|
)
|
|
assert len(messages) == 1
|
|
assert messages[0]["role"] == "user"
|
|
parts = messages[0]["content"]
|
|
assert parts[0]["type"] == "text"
|
|
assert "Extract the action items" in parts[0]["text"]
|
|
assert parts[1] == {"type": "text", "text": "meeting notes go here"}
|
|
|
|
def test_image_bytes_encoded_as_data_url(self):
|
|
png_bytes = b"\x89PNG\r\n\x1a\nfake"
|
|
messages = _build_structured_messages(
|
|
instructions="Read the image",
|
|
inputs=[
|
|
PluginLlmImageInput(data=png_bytes, mime_type="image/png"),
|
|
PluginLlmTextInput(text="prefer printed text"),
|
|
],
|
|
json_mode=False,
|
|
json_schema=None,
|
|
schema_name=None,
|
|
system_prompt=None,
|
|
)
|
|
parts = messages[0]["content"]
|
|
assert parts[1]["type"] == "image_url"
|
|
url = parts[1]["image_url"]["url"]
|
|
assert url.startswith("data:image/png;base64,")
|
|
decoded = base64.b64decode(url.split(",", 1)[1])
|
|
assert decoded == png_bytes
|
|
assert parts[2] == {"type": "text", "text": "prefer printed text"}
|
|
|
|
def test_image_url_passed_through(self):
|
|
messages = _build_structured_messages(
|
|
instructions="Caption this",
|
|
inputs=[PluginLlmImageInput(url="https://example.com/cat.jpg")],
|
|
json_mode=False,
|
|
json_schema=None,
|
|
schema_name=None,
|
|
system_prompt=None,
|
|
)
|
|
img_part = messages[0]["content"][1]
|
|
assert img_part["type"] == "image_url"
|
|
assert img_part["image_url"]["url"] == "https://example.com/cat.jpg"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# JSON parsing
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestJsonParsing:
|
|
def test_strip_code_fences_with_json_label(self):
|
|
assert _strip_code_fences('```json\n{"a":1}\n```') == '{"a":1}'
|
|
|
|
def test_parse_valid_json_with_json_mode(self):
|
|
parsed, ct = _parse_structured_text(
|
|
text='{"language": "French", "is_question": true}',
|
|
json_mode=True,
|
|
json_schema=None,
|
|
)
|
|
assert parsed == {"language": "French", "is_question": True}
|
|
assert ct == "json"
|
|
|
|
def test_schema_validation_accepts_match(self):
|
|
pytest.importorskip("jsonschema")
|
|
schema = {
|
|
"type": "object",
|
|
"properties": {"language": {"type": "string"}},
|
|
"required": ["language"],
|
|
}
|
|
parsed, ct = _parse_structured_text(
|
|
text='{"language": "French"}',
|
|
json_mode=False,
|
|
json_schema=schema,
|
|
)
|
|
assert parsed == {"language": "French"}
|
|
assert ct == "json"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# End-to-end facade
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestPluginLlmFacade:
|
|
def test_complete_uses_active_model_by_default(self):
|
|
captured: dict = {}
|
|
|
|
def fake_caller(**kwargs):
|
|
captured.update(kwargs)
|
|
return "auto", "default", _fake_response("Hello world.")
|
|
|
|
llm = make_plugin_llm_for_test(
|
|
plugin_id="my-plugin",
|
|
policy=_TrustPolicy(plugin_id="my-plugin"),
|
|
sync_caller=fake_caller,
|
|
)
|
|
result = llm.complete([{"role": "user", "content": "hi"}])
|
|
assert isinstance(result, PluginLlmCompleteResult)
|
|
assert result.text == "Hello world."
|
|
assert captured["provider_override"] is None
|
|
assert captured["model_override"] is None
|
|
assert captured["profile_override"] is None
|
|
assert result.usage.input_tokens == 4
|
|
assert result.usage.total_tokens == 10
|
|
|
|
def test_complete_passes_through_trusted_overrides(self):
|
|
captured: dict = {}
|
|
|
|
def fake_caller(**kwargs):
|
|
captured.update(kwargs)
|
|
return "anthropic", "claude-3-opus", _fake_response("ok")
|
|
|
|
llm = make_plugin_llm_for_test(
|
|
plugin_id="my-plugin",
|
|
policy=_trusted_policy("my-plugin"),
|
|
sync_caller=fake_caller,
|
|
)
|
|
result = llm.complete(
|
|
[{"role": "user", "content": "hi"}],
|
|
provider="anthropic",
|
|
model="claude-3-opus",
|
|
profile="work",
|
|
agent_id="ada",
|
|
temperature=0.0,
|
|
max_tokens=128,
|
|
timeout=10.0,
|
|
purpose="extract",
|
|
)
|
|
# The recorded provider/model in the result come from the override,
|
|
# since the stub caller echoed those values.
|
|
assert result.provider == "anthropic"
|
|
assert result.model == "claude-3-opus"
|
|
assert captured["provider_override"] == "anthropic"
|
|
assert captured["model_override"] == "claude-3-opus"
|
|
assert captured["profile_override"] == "work"
|
|
assert captured["temperature"] == 0.0
|
|
assert captured["max_tokens"] == 128
|
|
assert captured["timeout"] == 10.0
|
|
|
|
def test_complete_structured_returns_parsed_json(self):
|
|
def fake_caller(**_kwargs):
|
|
return "openai", "gpt-4o", _fake_response(
|
|
'{"language": "French", "is_question": true, "confidence": 0.99}'
|
|
)
|
|
|
|
llm = make_plugin_llm_for_test(
|
|
plugin_id="my-plugin",
|
|
policy=_TrustPolicy(plugin_id="my-plugin"),
|
|
sync_caller=fake_caller,
|
|
)
|
|
result = llm.complete_structured(
|
|
instructions="Detect language",
|
|
input=[PluginLlmTextInput(text="Comment ça va?")],
|
|
json_mode=True,
|
|
)
|
|
assert isinstance(result, PluginLlmStructuredResult)
|
|
assert result.parsed == {
|
|
"language": "French",
|
|
"is_question": True,
|
|
"confidence": 0.99,
|
|
}
|
|
assert result.content_type == "json"
|
|
|
|
def test_complete_structured_with_image_passes_image_url_part(self):
|
|
captured: dict = {}
|
|
|
|
def fake_caller(**kwargs):
|
|
captured.update(kwargs)
|
|
return "openai", "gpt-4o", _fake_response('{"caption": "ok"}')
|
|
|
|
llm = make_plugin_llm_for_test(
|
|
plugin_id="my-plugin",
|
|
policy=_TrustPolicy(plugin_id="my-plugin"),
|
|
sync_caller=fake_caller,
|
|
)
|
|
png = b"fake-bytes"
|
|
llm.complete_structured(
|
|
instructions="Caption this",
|
|
input=[PluginLlmImageInput(data=png, mime_type="image/png")],
|
|
json_mode=True,
|
|
)
|
|
msgs = captured["messages"]
|
|
user_msg = next(m for m in msgs if m["role"] == "user")
|
|
image_parts = [p for p in user_msg["content"] if p.get("type") == "image_url"]
|
|
assert len(image_parts) == 1
|
|
assert image_parts[0]["image_url"]["url"].startswith("data:image/png;base64,")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Async surface
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestAsyncSurface:
|
|
|
|
def test_acomplete_structured_parses_json(self):
|
|
async def fake_async(**_kwargs):
|
|
return "openai", "gpt-4o", _fake_response('{"x": 42}')
|
|
|
|
llm = make_plugin_llm_for_test(
|
|
plugin_id="my-plugin",
|
|
policy=_TrustPolicy(plugin_id="my-plugin"),
|
|
async_caller=fake_async,
|
|
)
|
|
|
|
async def _run() -> PluginLlmStructuredResult:
|
|
return await llm.acomplete_structured(
|
|
instructions="Extract x",
|
|
input=[PluginLlmTextInput(text="data")],
|
|
json_mode=True,
|
|
)
|
|
|
|
result = asyncio.run(_run())
|
|
assert result.parsed == {"x": 42}
|
|
assert result.content_type == "json"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Config-driven trust gate (round-trip via plugins.entries.<id>.llm)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestConfigDrivenPolicy:
|
|
def test_policy_loaded_from_yaml(self, tmp_path, monkeypatch):
|
|
from agent.plugin_llm import _resolve_trust_policy
|
|
|
|
hermes_home = tmp_path / ".hermes"
|
|
hermes_home.mkdir()
|
|
(hermes_home / "config.yaml").write_text(
|
|
"""
|
|
plugins:
|
|
entries:
|
|
my-plugin:
|
|
llm:
|
|
allow_provider_override: true
|
|
allowed_providers: [openrouter, anthropic]
|
|
allow_model_override: true
|
|
allowed_models:
|
|
- openai/gpt-4o-mini
|
|
- anthropic/claude-3-5-haiku
|
|
allow_profile_override: false
|
|
""",
|
|
encoding="utf-8",
|
|
)
|
|
monkeypatch.setenv("HERMES_HOME", str(hermes_home))
|
|
from hermes_cli import config as _config_mod
|
|
_config_mod._config_cache = None # type: ignore[attr-defined]
|
|
|
|
policy = _resolve_trust_policy("my-plugin")
|
|
assert policy.allow_provider_override is True
|
|
assert policy.allow_model_override is True
|
|
assert policy.allow_profile_override is False
|
|
assert policy.allowed_providers == frozenset({"openrouter", "anthropic"})
|
|
assert policy.allowed_models == frozenset({
|
|
"openai/gpt-4o-mini", "anthropic/claude-3-5-haiku",
|
|
})
|
|
|
|
def test_missing_plugin_entry_yields_default_deny(self, tmp_path, monkeypatch):
|
|
from agent.plugin_llm import _resolve_trust_policy
|
|
|
|
hermes_home = tmp_path / ".hermes"
|
|
hermes_home.mkdir()
|
|
(hermes_home / "config.yaml").write_text("plugins: {}\n", encoding="utf-8")
|
|
monkeypatch.setenv("HERMES_HOME", str(hermes_home))
|
|
from hermes_cli import config as _config_mod
|
|
_config_mod._config_cache = None # type: ignore[attr-defined]
|
|
|
|
policy = _resolve_trust_policy("never-configured")
|
|
assert policy.allow_provider_override is False
|
|
assert policy.allow_model_override is False
|
|
assert policy.allow_profile_override is False
|
|
assert policy.allow_agent_id_override is False
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Plugin context wiring
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestPluginContextIntegration:
|
|
|
|
def test_ctx_llm_uses_manifest_key_for_policy(self):
|
|
from hermes_cli.plugins import PluginContext, PluginManifest, PluginManager
|
|
|
|
manifest = PluginManifest(
|
|
name="bare-name", source="test", key="image_gen/openai"
|
|
)
|
|
manager = PluginManager()
|
|
ctx = PluginContext(manifest, manager)
|
|
assert ctx.llm._plugin_id == "image_gen/openai" # type: ignore[attr-defined]
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Attribution (result.provider / result.model / audit log)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestAttribution:
|
|
"""Verifies that the result object and the audit log carry the real
|
|
provider/model that ``call_llm`` ended up using, NOT the placeholder
|
|
fallbacks ('auto', 'default') from earlier drafts."""
|
|
|
|
def test_response_model_wins_over_model_override(self):
|
|
"""Providers often canonicalise the model name (e.g. ``gpt-4o``
|
|
→ ``gpt-4o-2024-08-06``). Whatever they actually returned wins
|
|
for the recorded model so the audit log reflects reality."""
|
|
from agent.plugin_llm import _resolve_attribution
|
|
|
|
response = SimpleNamespace(model="gpt-4o-2024-08-06", choices=[])
|
|
provider, model = _resolve_attribution(
|
|
provider_override="openrouter",
|
|
model_override="openai/gpt-4o",
|
|
response=response,
|
|
)
|
|
assert model == "gpt-4o-2024-08-06"
|
|
# Provider override is unaffected by response.model.
|
|
assert provider == "openrouter"
|
|
|
|
def test_response_model_used_even_when_no_overrides(self, monkeypatch):
|
|
"""The provider's canonical model name should still flow through
|
|
when no overrides are set."""
|
|
from agent import plugin_llm
|
|
import agent.auxiliary_client as ac
|
|
|
|
monkeypatch.setattr(ac, "_read_main_provider", lambda: "openrouter")
|
|
monkeypatch.setattr(ac, "_read_main_model", lambda: "openai/gpt-4o")
|
|
|
|
response = SimpleNamespace(model="openai/gpt-4o-2024-08-06", choices=[])
|
|
provider, model = plugin_llm._resolve_attribution(
|
|
provider_override=None,
|
|
model_override=None,
|
|
response=response,
|
|
)
|
|
assert provider == "openrouter"
|
|
assert model == "openai/gpt-4o-2024-08-06"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Hook-mode integration (ctx.llm called from a post_tool_call callback)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestHookMode:
|
|
"""The docs page promises ``ctx.llm`` works from inside lifecycle
|
|
hooks. This exercises that path: register a ``post_tool_call``
|
|
callback that calls ``ctx.llm.complete``, fire the hook through
|
|
the real ``invoke_hook`` machinery, and check the call landed."""
|
|
|
|
def test_complete_works_from_post_tool_call_hook(self):
|
|
from hermes_cli.plugins import PluginContext, PluginManifest, PluginManager
|
|
|
|
manifest = PluginManifest(name="hook-plugin", source="test", key="hook-plugin")
|
|
manager = PluginManager()
|
|
ctx = PluginContext(manifest, manager)
|
|
|
|
# Replace ctx.llm with a stub that records what the hook called.
|
|
captured: list = []
|
|
|
|
def fake_caller(**kwargs):
|
|
captured.append(kwargs)
|
|
return "openrouter", "openai/gpt-4o", _fake_response("rewrote it")
|
|
|
|
ctx._llm = make_plugin_llm_for_test( # type: ignore[attr-defined]
|
|
plugin_id="hook-plugin",
|
|
policy=_TrustPolicy(plugin_id="hook-plugin"),
|
|
sync_caller=fake_caller,
|
|
)
|
|
|
|
# Plugin registers a hook that runs ctx.llm.complete on every tool call.
|
|
def rewrite_error_hook(*, tool_name, args, result, **_):
|
|
if "Traceback" in (result or ""):
|
|
rewritten = ctx.llm.complete(
|
|
messages=[
|
|
{"role": "system", "content": "Rewrite errors plainly."},
|
|
{"role": "user", "content": result},
|
|
],
|
|
max_tokens=64,
|
|
purpose="hook-plugin.rewrite",
|
|
)
|
|
# Real hook would return the rewritten text via
|
|
# transform_tool_result; here we just capture for the assert.
|
|
captured.append({"hook_returned": rewritten.text})
|
|
|
|
ctx.register_hook("post_tool_call", rewrite_error_hook)
|
|
|
|
# Fire the hook the same way the agent core does it.
|
|
manager.invoke_hook(
|
|
"post_tool_call",
|
|
tool_name="terminal",
|
|
args={"command": "boom"},
|
|
result="Traceback (most recent call last):\n RuntimeError",
|
|
)
|
|
|
|
# Verify ctx.llm.complete fired through the hook.
|
|
assert len(captured) == 2 # one llm call + one hook return record
|
|
llm_call = captured[0]
|
|
assert "messages" in llm_call
|
|
assert any("rewrite" in m.get("content", "").lower()
|
|
for m in llm_call["messages"] if isinstance(m, dict))
|
|
hook_record = captured[1]
|
|
assert hook_record["hook_returned"] == "rewrote it"
|