Files
hermes-agent/tests/agent/test_insights.py
ethernet 890bbbda1f Merge remote-tracking branch 'origin/main' into ethie/pm-clean
# Conflicts:
#	apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts
#	apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts
#	apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts
#	apps/desktop/e2e/bot-mode-roster-localized.spec.ts
#	apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts
#	apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts
#	apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts
#	apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts
#	apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts
#	apps/desktop/e2e/bot-roster-user-sections.spec.ts
#	apps/desktop/e2e/bot-routines-pane-narrow.spec.ts
#	apps/desktop/e2e/bot-row-open-recent-session.spec.ts
#	apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts
#	apps/desktop/e2e/group-composer-auto-grow.spec.ts
#	apps/desktop/e2e/group-create-gate-remote-roster.spec.ts
#	apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts
#	apps/desktop/e2e/hosted-room-backend-continuity.spec.ts
#	apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts
#	apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts
#	apps/desktop/e2e/worktree-branch-status.spec.ts
#	apps/desktop/electron/backend-probes.test.ts
#	apps/desktop/electron/connection-apply.test.ts
#	apps/desktop/electron/desktop-electron-pin.test.ts
#	apps/desktop/electron/desktop-uninstall.test.ts
#	apps/desktop/electron/gateway-file-download-transport.test.ts
#	apps/desktop/electron/gateway-stop-before-update.test.ts
#	apps/desktop/electron/github-api-auth.test.ts
#	apps/desktop/electron/registry-primary-profile-scope.test.ts
#	apps/desktop/electron/update-api-check.test.ts
#	apps/desktop/electron/update-handoff-marker.test.ts
#	apps/desktop/electron/venv-blocker-scan.test.ts
#	apps/desktop/scripts/after-extract.test.mjs
#	apps/desktop/scripts/local-pack-publish.test.mjs
#	apps/desktop/scripts/tasks-scroll.test.mjs
#	apps/desktop/src/app/settings/model-settings.test.tsx
#	apps/desktop/src/app/updates-overlay.blockers.test.tsx
#	apps/desktop/src/components/desktop-install-overlay.test.tsx
#	apps/desktop/src/lib/update-copy.test.ts
#	scripts/ci/check_os_marker_fakes.py
#	tests-js/desktop-mac-usage-descriptions.test.ts
#	tests-js/node-engine-alignment.test.ts
#	tests/agent/lsp/test_install_and_lint_fixes.py
#	tests/agent/test_command_token_source.py
#	tests/agent/test_compression_boundary_hook.py
#	tests/agent/test_create_openai_client_ssl_verify.py
#	tests/agent/test_custom_provider_ca_probes.py
#	tests/agent/test_endpoint_blackhole.py
#	tests/agent/test_estimator_parity.py
#	tests/agent/test_in_place_compaction.py
#	tests/agent/test_moa_loop_mode.py
#	tests/agent/test_model_metadata.py
#	tests/agent/test_skill_session_platform_gate.py
#	tests/agent/test_skill_utils.py
#	tests/agent/test_ssl_ca_guard.py
#	tests/computer_use/test_doctor.py
#	tests/cron/test_codex_execution_paths.py
#	tests/cron/test_cron_bot_chat_delivery.py
#	tests/cron/test_cron_script.py
#	tests/cron/test_media_delivery_parity.py
#	tests/cron/test_misfire_catchup.py
#	tests/cron/test_parallel_pool.py
#	tests/cron/test_recurring_eagain_redispatch.py
#	tests/gateway/test_choice_picker.py
#	tests/gateway/test_control_socket_windows_live.py
#	tests/gateway/test_dingtalk.py
#	tests/gateway/test_feishu.py
#	tests/gateway/test_feishu_onboard.py
#	tests/gateway/test_gateway_shutdown.py
#	tests/gateway/test_matrix.py
#	tests/gateway/test_model_command_custom_providers.py
#	tests/gateway/test_reasoning_command.py
#	tests/gateway/test_runtime_footer.py
#	tests/gateway/test_session.py
#	tests/gateway/test_session_hygiene.py
#	tests/gateway/test_status.py
#	tests/gateway/test_teams.py
#	tests/gateway/test_turn_lease.py
#	tests/gateway/test_whatsapp_connect.py
#	tests/hermes_cli/test_approvals_command.py
#	tests/hermes_cli/test_auth_store_lock_concurrent.py
#	tests/hermes_cli/test_backup.py
#	tests/hermes_cli/test_banner_git_state.py
#	tests/hermes_cli/test_certifi_repair.py
#	tests/hermes_cli/test_cmd_update.py
#	tests/hermes_cli/test_compat_manifest_targets.py
#	tests/hermes_cli/test_computer_use_cli.py
#	tests/hermes_cli/test_cpr_local_leak.py
#	tests/hermes_cli/test_dashboard_auth_gate.py
#	tests/hermes_cli/test_dashboard_procs_kill_grace.py
#	tests/hermes_cli/test_desktop_lifecycle_windows_live.py
#	tests/hermes_cli/test_doctor.py
#	tests/hermes_cli/test_doctor_command_install.py
#	tests/hermes_cli/test_fleet_config_migration_windows_live.py
#	tests/hermes_cli/test_gateway.py
#	tests/hermes_cli/test_gateway_platform_gating.py
#	tests/hermes_cli/test_gateway_restart_loop.py
#	tests/hermes_cli/test_gateway_task_probe.py
#	tests/hermes_cli/test_gateway_wsl.py
#	tests/hermes_cli/test_gui_command.py
#	tests/hermes_cli/test_install_cua_driver.py
#	tests/hermes_cli/test_kanban_db.py
#	tests/hermes_cli/test_lazy_command_exports.py
#	tests/hermes_cli/test_lazy_refresh_venv_repair.py
#	tests/hermes_cli/test_linux_desktop_entry.py
#	tests/hermes_cli/test_local_runtime.py
#	tests/hermes_cli/test_local_runtime_updates.py
#	tests/hermes_cli/test_managed_uv.py
#	tests/hermes_cli/test_mcp_reload_confirm_gate.py
#	tests/hermes_cli/test_nous_subscription.py
#	tests/hermes_cli/test_npm_engine.py
#	tests/hermes_cli/test_personality_none.py
#	tests/hermes_cli/test_pet_toggle.py
#	tests/hermes_cli/test_plan_reconciliation_windows_live.py
#	tests/hermes_cli/test_plugin_event_bus.py
#	tests/hermes_cli/test_plugin_manifest_v2.py
#	tests/hermes_cli/test_plugin_packs.py
#	tests/hermes_cli/test_plugins_cmd.py
#	tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py
#	tests/hermes_cli/test_process_identity.py
#	tests/hermes_cli/test_profiles.py
#	tests/hermes_cli/test_profiles_sidebar_cache.py
#	tests/hermes_cli/test_pty_bridge.py
#	tests/hermes_cli/test_resolve_turn_limit.py
#	tests/hermes_cli/test_serve_runtime_inventory.py
#	tests/hermes_cli/test_session_vacuum_config.py
#	tests/hermes_cli/test_set_config_value.py
#	tests/hermes_cli/test_signal_handler_kanban_worker.py
#	tests/hermes_cli/test_slash_confirm_windows.py
#	tests/hermes_cli/test_stale_pid_guard.py
#	tests/hermes_cli/test_startup_fast_guards.py
#	tests/hermes_cli/test_status.py
#	tests/hermes_cli/test_telegram_managed_bot.py
#	tests/hermes_cli/test_tools_config.py
#	tests/hermes_cli/test_update_apply_shallow_count.py
#	tests/hermes_cli/test_update_autostash.py
#	tests/hermes_cli/test_update_concurrent_quarantine.py
#	tests/hermes_cli/test_update_fetch_failure_classifier.py
#	tests/hermes_cli/test_update_fleet_probe_resume_token.py
#	tests/hermes_cli/test_update_handoff_backend_reap.py
#	tests/hermes_cli/test_update_handoff_desktop_rebuild.py
#	tests/hermes_cli/test_update_head_moved_gate.py
#	tests/hermes_cli/test_update_host_obligation.py
#	tests/hermes_cli/test_update_import_guard.py
#	tests/hermes_cli/test_update_interrupted_recovery.py
#	tests/hermes_cli/test_update_inventory.py
#	tests/hermes_cli/test_update_launchd_unloaded_gateway.py
#	tests/hermes_cli/test_update_missing_configured_deps.py
#	tests/hermes_cli/test_update_modified_notice.py
#	tests/hermes_cli/test_update_multiplex_migration_hook.py
#	tests/hermes_cli/test_update_no_gateway_restart.py
#	tests/hermes_cli/test_update_orphan_backend_reap.py
#	tests/hermes_cli/test_update_parked_branch_guard.py
#	tests/hermes_cli/test_update_post_pull_syntax_guard.py
#	tests/hermes_cli/test_update_receipt.py
#	tests/hermes_cli/test_update_self_lock.py
#	tests/hermes_cli/test_update_shim_fail_closed.py
#	tests/hermes_cli/test_update_shim_self_lock.py
#	tests/hermes_cli/test_update_sqlite_remediation.py
#	tests/hermes_cli/test_update_stale_dashboard.py
#	tests/hermes_cli/test_update_stale_virtualenv.py
#	tests/hermes_cli/test_update_venv_health.py
#	tests/hermes_cli/test_update_venv_ownership_preflight.py
#	tests/hermes_cli/test_update_wedged_gateway.py
#	tests/hermes_cli/test_update_yes_flag.py
#	tests/hermes_cli/test_update_zip_two_phase.py
#	tests/hermes_cli/test_urllib_security.py
#	tests/hermes_cli/test_ux_messages_auth_config.py
#	tests/hermes_cli/test_ux_messages_startup.py
#	tests/hermes_cli/test_venv_holder_classifier.py
#	tests/hermes_cli/test_verify_console_scripts.py
#	tests/hermes_cli/test_verify_core_dependencies.py
#	tests/hermes_cli/test_web_server.py
#	tests/hermes_cli/test_web_server_console_ws.py
#	tests/hermes_cli/test_web_server_ws_ping.py
#	tests/hermes_cli/test_web_ui_build.py
#	tests/hermes_state/test_fts_rebuild_admission.py
#	tests/hermes_state/test_hermes_state.py
#	tests/plugins/memory/test_memory_lazy_install.py
#	tests/plugins/test_google_meet_plugin.py
#	tests/plugins/test_langfuse_plugin.py
#	tests/plugins/test_security_guidance_plugin.py
#	tests/plugins/test_transform_llm_output_hook.py
#	tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py
#	tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py
#	tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py
#	tests/scripts/install/test_install_clone_throttle_fallback.py
#	tests/scripts/install/test_install_lockfile_churn.py
#	tests/scripts/install/test_install_no_initial_commit.py
#	tests/scripts/install/test_install_sh_browser_install.py
#	tests/scripts/install/test_install_sh_node_prerelease.py
#	tests/scripts/install/test_install_sh_symlink_stomp.py
#	tests/scripts/install/test_install_sh_uv_lock_config.py
#	tests/scripts/install/test_install_unmerged_index.py
#	tests/scripts/test_contributor_map.py
#	tests/scripts/test_run_tests_parallel.py
#	tests/skills/test_competitor_news_monitor_skill.py
#	tests/skills/test_document_to_action_items_skill.py
#	tests/skills/test_google_workspace_setup.py
#	tests/skills/test_google_workspace_setup_deps.py
#	tests/skills/test_grounded_citations_skill.py
#	tests/skills/test_ip_as_logo_skill.py
#	tests/skills/test_live_dashboard_skill.py
#	tests/skills/test_mcp_oauth_remote_gateway_skill.py
#	tests/skills/test_office_document_skills.py
#	tests/skills/test_openclaw_migration.py
#	tests/skills/test_product_price_monitor_skill.py
#	tests/skills/test_scrollcraft_skill.py
#	tests/skills/test_setup_wizard_generator_skill.py
#	tests/skills/test_weekly_review_planning_skill.py
#	tests/test_engines_satisfiable.py
#	tests/test_fast_safe_load.py
#	tests/test_hermes_bootstrap.py
#	tests/test_hermes_constants.py
#	tests/test_hermes_logging.py
#	tests/test_managed_runtime_resolution.py
#	tests/test_model_tools_async_bridge.py
#	tests/test_packaging_build_guard.py
#	tests/test_packaging_metadata.py
#	tests/test_yaml_indent_consistency.py
#	tests/tools/test_approval_timeout_overflow.py
#	tests/tools/test_base_environment.py
#	tests/tools/test_bot_mode_dm.py
#	tests/tools/test_browser_chromium_check.py
#	tests/tools/test_browser_hardening.py
#	tests/tools/test_browser_homebrew_paths.py
#	tests/tools/test_browser_npx_warmup.py
#	tests/tools/test_browser_orphan_reaper.py
#	tests/tools/test_browser_real_profile.py
#	tests/tools/test_browser_use_cli.py
#	tests/tools/test_clipboard.py
#	tests/tools/test_code_execution.py
#	tests/tools/test_code_execution_modes.py
#	tests/tools/test_code_execution_windows_env.py
#	tests/tools/test_computer_use.py
#	tests/tools/test_delegate_liveness_timeout.py
#	tests/tools/test_execute_code_approval_cluster.py
#	tests/tools/test_execution_flag_detection.py
#	tests/tools/test_fal_common.py
#	tests/tools/test_file_operations.py
#	tests/tools/test_file_tools.py
#	tests/tools/test_file_tools_cwd_resolution.py
#	tests/tools/test_file_tools_live.py
#	tests/tools/test_lazy_deps.py
#	tests/tools/test_lazy_deps_durable_target.py
#	tests/tools/test_lazy_deps_managed.py
#	tests/tools/test_local_env_blocklist.py
#	tests/tools/test_local_tempdir.py
#	tests/tools/test_macos_protected_search.py
#	tests/tools/test_mcp_npx_cached_bin.py
#	tests/tools/test_oneshot_completion_linger.py
#	tests/tools/test_process_registry.py
#	tests/tools/test_read_file_schema_gating.py
#	tests/tools/test_skill_improvements.py
#	tests/tools/test_skills_sync.py
#	tests/tools/test_termux_api_detection.py
#	tests/tools/test_tirith_security.py
#	tests/tools/test_transcription_tools.py
#	tests/tools/test_tts_streaming.py
#	tests/tools/test_wake_word.py
#	tests/tui_gateway/test_compute_host_borrowed_lease.py
#	tests/tui_gateway/test_compute_host_turn_protocol.py
#	tests/tui_gateway/test_isolated_orphan_activity.py
#	tests/tui_gateway/test_protocol.py
#	tests/tui_gateway/test_slash_worker_profile_home.py
#	tests/tui_gateway/test_subprocess_encoding.py
#	tests/tui_gateway/test_tui_gateway_server.py
#	ui-tui/src/__tests__/terminalParity.test.ts
#	ui-tui/src/__tests__/termuxComposerLayout.test.ts
#	ui-tui/src/__tests__/textInputFastEcho.test.ts
2026-09-23 07:02:44 -04:00

593 lines
26 KiB
Python

"""Tests for agent/insights.py — InsightsEngine analytics and reporting."""
import time
import pytest
from hermes_state import SessionDB
from agent.insights import (
InsightsEngine,
_estimate_cost,
_bar_chart,
)
from agent.usage_pricing import (
format_duration_compact as _format_duration,
has_known_pricing as _has_known_pricing,
)
@pytest.fixture()
def db(tmp_path):
"""Create a SessionDB with a temp database file."""
db_path = tmp_path / "test_insights.db"
session_db = SessionDB(db_path=db_path)
yield session_db
session_db.close()
@pytest.fixture()
def populated_db(db):
"""Create a DB with realistic session data for insights testing."""
now = time.time()
day = 86400
# Session 1: CLI, claude-sonnet, ended, 2 days ago
db.create_session(
session_id="s1", source="cli",
model="anthropic/claude-sonnet-4-20250514", user_id="user1",
)
# Backdate the started_at
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's1'", (now - 2 * day,))
db.end_session("s1", end_reason="user_exit")
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's1'", (now - 2 * day + 3600,))
db.update_token_counts("s1", input_tokens=50000, output_tokens=15000)
db.append_message("s1", role="user", content="Hello, help me fix a bug")
db.append_message("s1", role="assistant", content="Sure, let me look into that.")
db.append_message("s1", role="assistant", content="Let me search the files.",
tool_calls=[{"function": {"name": "search_files"}}])
db.append_message("s1", role="tool", content="Found 3 matches", tool_name="search_files")
db.append_message("s1", role="assistant", content="Let me read the file.",
tool_calls=[{"function": {"name": "read_file"}}])
db.append_message("s1", role="tool", content="file contents...", tool_name="read_file")
db.append_message("s1", role="assistant", content="I found the bug. Let me fix it.",
tool_calls=[{"function": {"name": "patch"}}])
db.append_message("s1", role="tool", content="patched successfully", tool_name="patch")
db.append_message(
"s1",
role="assistant",
content="Let me load the PR workflow skill.",
tool_calls=[{"function": {"name": "skill_view", "arguments": '{"name":"github-pr-workflow"}'}}],
)
db.append_message("s1", role="user", content="Thanks!")
db.append_message("s1", role="assistant", content="You're welcome!")
# Session 2: Telegram, gpt-4o, ended, 5 days ago
db.create_session(
session_id="s2", source="telegram",
model="gpt-4o", user_id="user1",
)
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's2'", (now - 5 * day,))
db.end_session("s2", end_reason="timeout")
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's2'", (now - 5 * day + 1800,))
db.update_token_counts("s2", input_tokens=20000, output_tokens=8000)
db.append_message("s2", role="user", content="Search the web for something")
db.append_message("s2", role="assistant", content="Searching...",
tool_calls=[{"function": {"name": "web_search"}}])
db.append_message("s2", role="tool", content="results...", tool_name="web_search")
db.append_message("s2", role="assistant", content="Here's what I found")
# Session 3: CLI, deepseek-chat, ended, 10 days ago
db.create_session(
session_id="s3", source="cli",
model="deepseek-chat", user_id="user1",
)
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's3'", (now - 10 * day,))
db.end_session("s3", end_reason="user_exit")
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's3'", (now - 10 * day + 7200,))
db.update_token_counts("s3", input_tokens=100000, output_tokens=40000)
db.append_message("s3", role="user", content="Run this terminal command")
db.append_message("s3", role="assistant", content="Running...",
tool_calls=[{"function": {"name": "terminal"}}])
db.append_message("s3", role="tool", content="output...", tool_name="terminal")
db.append_message("s3", role="assistant", content="Let me run another",
tool_calls=[{"function": {"name": "terminal"}}])
db.append_message("s3", role="tool", content="more output...", tool_name="terminal")
db.append_message("s3", role="assistant", content="And search files",
tool_calls=[{"function": {"name": "search_files"}}])
db.append_message("s3", role="tool", content="found stuff", tool_name="search_files")
db.append_message(
"s3",
role="assistant",
content="Load the debugging skill.",
tool_calls=[{"function": {"name": "skill_view", "arguments": '{"name":"systematic-debugging"}'}}],
)
# Session 4: Discord, same model as s1, ended, 1 day ago
db.create_session(
session_id="s4", source="discord",
model="anthropic/claude-sonnet-4-20250514", user_id="user2",
)
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's4'", (now - 1 * day,))
db.end_session("s4", end_reason="user_exit")
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's4'", (now - 1 * day + 900,))
db.update_token_counts("s4", input_tokens=10000, output_tokens=5000)
db.append_message("s4", role="user", content="Quick question")
db.append_message("s4", role="assistant", content="Sure, go ahead")
db.append_message(
"s4",
role="assistant",
content="Load and update GitHub skills.",
tool_calls=[
{"function": {"name": "skill_view", "arguments": '{"name":"github-pr-workflow"}'}},
{"function": {"name": "skill_manage", "arguments": '{"name":"github-code-review"}'}},
],
)
# Session 5: Old session, 45 days ago (should be excluded from 30-day window)
db.create_session(
session_id="s_old", source="cli",
model="gpt-4o-mini", user_id="user1",
)
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's_old'", (now - 45 * day,))
db.end_session("s_old", end_reason="user_exit")
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's_old'", (now - 45 * day + 600,))
db.update_token_counts("s_old", input_tokens=5000, output_tokens=2000)
db.append_message("s_old", role="user", content="old message")
db.append_message("s_old", role="assistant", content="old reply")
db._conn.commit()
return db
class TestHasKnownPricing:
def test_unknown_custom_model(self):
assert _has_known_pricing("FP16_Hermes_4.5") is False
assert _has_known_pricing("my-custom-model") is False
assert _has_known_pricing("glm-5") is False
assert _has_known_pricing("") is False
assert _has_known_pricing(None) is False
def test_heuristic_matched_models_are_not_considered_known(self):
assert _has_known_pricing("some-opus-model") is False
assert _has_known_pricing("future-sonnet-v2") is False
class TestEstimateCost:
def test_cache_aware_usage(self):
cost, status = _estimate_cost(
"anthropic/claude-sonnet-4-20250514",
1000,
500,
cache_read_tokens=2000,
cache_write_tokens=400,
provider="anthropic",
)
assert status == "estimated"
expected = (1000 * 3.0 + 500 * 15.0 + 2000 * 0.30 + 400 * 3.75) / 1_000_000
assert cost == pytest.approx(expected, abs=0.0001)
# =========================================================================
# Format helpers
# =========================================================================
class TestFormatDuration:
def test_seconds(self):
assert _format_duration(45) == "45s"
def test_hours_with_minutes(self):
result = _format_duration(5400) # 1.5 hours
assert result == "1h 30m"
class TestBarChart:
def test_basic_bars(self):
bars = _bar_chart([10, 5, 0, 20], max_width=10)
assert len(bars) == 4
assert len(bars[3]) == 10 # max value gets full width
assert len(bars[0]) == 5 # half of max
assert bars[2] == "" # zero gets empty
def test_all_zeros(self):
bars = _bar_chart([0, 0, 0], max_width=10)
assert all(b == "" for b in bars)
# =========================================================================
# InsightsEngine — empty DB
# =========================================================================
class TestInsightsEmpty:
def test_empty_db_returns_empty_report(self, db):
engine = InsightsEngine(db)
report = engine.generate(days=30)
assert report["empty"] is True
assert report["overview"] == {}
# Both renderers must handle the empty report without crashing.
assert engine.format_terminal(report)
assert engine.format_gateway(report)
# =========================================================================
# InsightsEngine — populated DB
# =========================================================================
class TestInsightsPopulated:
def test_overview_token_totals(self, populated_db):
engine = InsightsEngine(populated_db)
report = engine.generate(days=30)
overview = report["overview"]
expected_input = 50000 + 20000 + 100000 + 10000
expected_output = 15000 + 8000 + 40000 + 5000
assert overview["total_input_tokens"] == expected_input
assert overview["total_output_tokens"] == expected_output
assert overview["total_tokens"] == expected_input + expected_output
def test_model_breakdown_splits_mid_session_switch(self, db):
"""A session that switches models mid-flight is split across both
models in the breakdown, not dumped on the initial model (#51607).
"""
now = time.time()
db.create_session(session_id="sw", source="cli",
model="deepseek/deepseek-v4-pro")
# 40k tokens on deepseek, then switch and 50k on opus.
db.update_token_counts("sw", input_tokens=40000, output_tokens=8000,
model="deepseek/deepseek-v4-pro",
billing_provider="deepseek", api_call_count=2)
db.update_session_model("sw", "anthropic/claude-opus-4.8")
db.update_token_counts("sw", input_tokens=50000, output_tokens=4000,
model="anthropic/claude-opus-4.8",
billing_provider="openrouter", api_call_count=3)
db._conn.commit()
report = InsightsEngine(db).generate(days=30)
models = {m["model"]: m for m in report["models"]}
assert "deepseek-v4-pro" in models
assert "claude-opus-4.8" in models
# Tokens attributed to the model that actually incurred them.
assert models["deepseek-v4-pro"]["input_tokens"] == 40000
assert models["claude-opus-4.8"]["input_tokens"] == 50000
assert models["claude-opus-4.8"]["api_calls"] == 3
# The summary row's single model would have hidden one of these.
assert models["deepseek-v4-pro"]["total_tokens"] == 48000
assert models["claude-opus-4.8"]["total_tokens"] == 54000
def test_overview_cost_matches_per_model_stored_cost(self, db):
db.create_session(session_id="cost", source="cli", model="model-a")
db.update_token_counts(
"cost", input_tokens=10, model="model-a", billing_provider="custom",
estimated_cost_usd=1.25, actual_cost_usd=1.0,
cost_status="estimated", cost_source="provider", api_call_count=1,
)
db.update_session_model("cost", "model-b")
db.update_session_billing_route("cost", provider="custom-b", base_url=None)
db.update_token_counts(
"cost", input_tokens=20, model="model-b", billing_provider="custom-b",
estimated_cost_usd=2.5, actual_cost_usd=2.0,
cost_status="estimated", cost_source="provider", api_call_count=1,
)
report = InsightsEngine(db).generate(days=30)
assert sum(m["cost"] for m in report["models"]) == pytest.approx(3.75)
assert report["overview"]["estimated_cost"] == pytest.approx(3.75)
assert report["overview"]["actual_cost"] == pytest.approx(3.0)
def test_tool_usage_sums_disjoint_sessions_without_double_counting_pairs(self, db):
"""One session records a call as tool_name only, another as tool_calls only: both count.
A session carrying BOTH representations of the same call still counts it once (#9814)."""
db.create_session(session_id="gw", source="gateway", model="m")
db.append_message("gw", role="tool", content="r", tool_name="search_files")
db.create_session(session_id="cli", source="cli", model="m")
db.append_message("cli", role="assistant", content="x",
tool_calls=[{"function": {"name": "search_files", "arguments": "{}"}}])
db.create_session(session_id="both", source="cli", model="m")
db.append_message("both", role="assistant", content="x",
tool_calls=[{"function": {"name": "search_files", "arguments": "{}"}}])
db.append_message("both", role="tool", content="r", tool_name="search_files")
db._conn.commit()
tools = InsightsEngine(db).generate(days=30)["tools"]
assert next(t["count"] for t in tools if t["tool"] == "search_files") == 3
def test_tool_breakdown(self, populated_db):
engine = InsightsEngine(populated_db)
report = engine.generate(days=30)
tools = report["tools"]
tool_names = [t["tool"] for t in tools]
assert "terminal" in tool_names
assert "search_files" in tool_names
assert "read_file" in tool_names
assert "patch" in tool_names
assert "web_search" in tool_names
# terminal was used 2x in s3
terminal = next(t for t in tools if t["tool"] == "terminal")
assert terminal["count"] == 2
# Percentages should sum to ~100%
total_pct = sum(t["percentage"] for t in tools)
assert total_pct == pytest.approx(100.0, abs=0.1)
def test_skill_breakdown(self, populated_db):
engine = InsightsEngine(populated_db)
report = engine.generate(days=30)
skills = report["skills"]
assert skills["summary"]["distinct_skills_used"] == 3
assert skills["summary"]["total_skill_loads"] == 3
assert skills["summary"]["total_skill_edits"] == 1
assert skills["summary"]["total_skill_actions"] == 4
top_skill = skills["top_skills"][0]
assert top_skill["skill"] == "github-pr-workflow"
assert top_skill["view_count"] == 2
assert top_skill["manage_count"] == 0
assert top_skill["total_count"] == 2
assert top_skill["last_used_at"] is not None
# The Insights assistant tool-call queries pin
# idx_messages_assistant_calls_by_session via INDEXED BY. These tests prove
# (a) the planner uses that index for BOTH the unfiltered and source-filtered
# branches on a fresh DB *without* ANALYZE, and (b) the index is a pure
# optimization — output is identical whether or not it is selected.
_INDEX = "idx_messages_assistant_calls_by_session"
_PINNED_QUERIES = (
("_GET_TOOL_CALLS_ALL", (0.0,)),
("_GET_TOOL_CALLS_WITH_SOURCE", (0.0, "cli")),
("_GET_SKILL_CALLS_ALL", (0.0,)),
("_GET_SKILL_CALLS_WITH_SOURCE", (0.0, "cli")),
)
def test_assistant_call_queries_use_partial_index_without_analyze(
self, populated_db
):
"""Every fixed-predicate branch selects the partial index on a fresh DB.
No ANALYZE is run, so this covers the default-statistics case a freshly
initialized state.db is actually in. Both the unfiltered and the
source-filtered (``s.source = ?``) branches are checked.
"""
# Guard against the fresh-DB planner regression the reviewers found:
# without INDEXED BY the source-filtered branch fell back to
# idx_messages_session_active.
assert "ANALYZE" not in "".join(
r["sql"] or ""
for r in populated_db._conn.execute(
"SELECT sql FROM sqlite_master WHERE type = 'index'"
)
)
for attr, params in self._PINNED_QUERIES:
sql = getattr(InsightsEngine, attr)
plan = "\n".join(
row["detail"]
for row in populated_db._conn.execute(
"EXPLAIN QUERY PLAN " + sql, params
).fetchall()
)
assert self._INDEX in plan, f"{attr} did not use the index:\n{plan}"
def test_assistant_call_rows_invariant_to_index_selection(self, populated_db):
"""The pinned index only changes the plan, never the result set.
For every branch, the index-pinned query and the un-pinned form (whose
plan the optimizer chooses freely) must return identical rows — proving
the index is a pure optimization — for both the unfiltered and
source-filtered scopes.
"""
assert populated_db._conn.execute(
"SELECT 1 FROM sqlite_master WHERE type = 'index' AND name = ?",
(self._INDEX,),
).fetchone() is not None
for attr, params in self._PINNED_QUERIES:
pinned_sql = getattr(InsightsEngine, attr)
unpinned_sql = pinned_sql.replace(f" INDEXED BY {self._INDEX}", "")
pinned = [
tuple(r) for r in
populated_db._conn.execute(pinned_sql, params).fetchall()
]
unpinned = [
tuple(r) for r in
populated_db._conn.execute(unpinned_sql, params).fetchall()
]
assert sorted(pinned) == sorted(unpinned), attr
def test_missing_index_falls_back_to_unpinned_queries(self, populated_db):
"""INDEXED BY would be a hard error if the index is missing — which
happens on read-only opens of a state.db written by an older version
(web dashboard analytics). The engine must probe and fall back to the
unpinned variants instead of crashing, returning identical rows."""
engine_pinned = InsightsEngine(populated_db)
tools_before = engine_pinned._get_tool_usage(0.0)
populated_db._conn.execute(f"DROP INDEX IF EXISTS {self._INDEX}")
populated_db._conn.commit()
engine = InsightsEngine(populated_db)
assert engine._has_assistant_calls_index is False
assert "INDEXED BY" not in engine._GET_TOOL_CALLS_ALL
tools_after = engine._get_tool_usage(0.0)
assert sorted(t["tool_name"] for t in tools_after) == sorted(
t["tool_name"] for t in tools_before
)
# And with the index present, the pin stays.
assert "INDEXED BY" in InsightsEngine._GET_TOOL_CALLS_ALL
def test_get_usage_breakdown_matches_full_generate(self, populated_db):
engine = InsightsEngine(populated_db)
full = engine.generate(days=30)
focused = engine.get_usage_breakdown(days=30)
assert focused["skills"] == full["skills"]
assert focused["tools"] == full["tools"]
def test_get_skill_breakdown_respects_source_filter(self, populated_db):
engine = InsightsEngine(populated_db)
# Only s1 (cli) has skill_view "github-pr-workflow"
focused = engine.get_usage_breakdown(days=30, source="cli")["skills"]
skill_names = [s["skill"] for s in focused["top_skills"]]
assert "github-pr-workflow" in skill_names
# github-code-review was in discord (s4), not cli
assert "github-code-review" not in skill_names
def test_get_skill_usage_prefilter_ignores_non_skill_substring(self, db):
# "my_skill_view_helper" contains "skill_view" as a substring; instr()
# will match but the Python-side name check keeps the set clean.
# More importantly, messages with no skill_* tools must be excluded.
db.create_session(session_id="sx", source="cli", model="gpt-4o")
db.append_message(
"sx",
role="assistant",
content="Just using read_file.",
tool_calls=[{"function": {"name": "read_file", "arguments": '{"path":"/tmp/x"}'}}],
)
db._conn.commit()
focused = InsightsEngine(db).get_usage_breakdown(days=30)["skills"]
assert focused["summary"]["total_skill_actions"] == 0
assert focused["top_skills"] == []
# =========================================================================
# Formatting
# =========================================================================
class TestTerminalFormatting:
def test_terminal_format_unknown_bucket_for_custom_models(self, db):
"""Custom models with no pricing surface as the Unknown bucket (#77223)."""
db.create_session(session_id="s1", source="cli", model="my-custom-model")
db.update_token_counts("s1", input_tokens=1000, output_tokens=500)
db._conn.commit()
engine = InsightsEngine(db)
report = engine.generate(days=30)
text = engine.format_terminal(report)
# Cost section surfaces unknown-cost sessions (#77223) instead of
# hiding them — a custom model with no pricing data shows in the
# Unknown bucket rather than silently reporting $0.
assert "Unknown" in text
class TestGatewayFormatting:
def test_gateway_format_is_shorter(self, populated_db):
engine = InsightsEngine(populated_db)
report = engine.generate(days=30)
terminal_text = engine.format_terminal(report)
gateway_text = engine.format_gateway(report)
assert len(gateway_text) < len(terminal_text)
# =========================================================================
# Edge cases
# =========================================================================
class TestEdgeCases:
def test_session_with_no_model(self, db):
"""Sessions with NULL model should not crash."""
db.create_session(session_id="s1", source="cli")
db.update_token_counts("s1", input_tokens=1000, output_tokens=500)
db._conn.commit()
engine = InsightsEngine(db)
report = engine.generate(days=30)
assert report["empty"] is False
models = report["models"]
assert len(models) == 1
assert models[0]["model"] == "unknown"
assert models[0]["has_pricing"] is False
def test_mixed_commercial_and_custom_models(self, db):
"""Mix of commercial and custom models: only commercial ones get costs."""
db.create_session(session_id="s1", source="cli", model="anthropic/claude-sonnet-4-20250514")
db.update_token_counts(
"s1",
input_tokens=10000,
output_tokens=5000,
billing_provider="anthropic",
)
db.create_session(session_id="s2", source="cli", model="my-local-llama")
db.update_token_counts("s2", input_tokens=10000, output_tokens=5000)
db._conn.commit()
engine = InsightsEngine(db)
report = engine.generate(days=30)
# Cost should only come from gpt-4o, not from the custom model
overview = report["overview"]
assert overview["estimated_cost"] > 0
assert "claude-sonnet-4-20250514" in overview["models_with_pricing"] # list now, not set
assert "my-local-llama" in overview["models_without_pricing"]
# Verify individual model entries
claude = next(m for m in report["models"] if m["model"] == "claude-sonnet-4-20250514")
assert claude["has_pricing"] is True
assert claude["cost"] > 0
llama = next(m for m in report["models"] if m["model"] == "my-local-llama")
assert llama["has_pricing"] is False
assert llama["cost"] == 0.0
def test_only_one_platform(self, db):
"""Single-platform usage should still work."""
db.create_session(session_id="s1", source="cli", model="test")
db._conn.commit()
engine = InsightsEngine(db)
report = engine.generate(days=30)
assert len(report["platforms"]) == 1
assert report["platforms"][0]["platform"] == "cli"
# Terminal format should NOT show platform section for single platform
text = engine.format_terminal(report)
# (it still shows platforms section if there's only cli and nothing else)
# Actually the condition is > 1 platforms OR non-cli, so single cli won't show
def test_cost_buckets_displayed_in_terminal_format(self, db):
"""#77223: included/estimated/unknown cost buckets surface in terminal."""
# Estimated cost session
db.create_session(session_id="est", source="cli", model="model-a")
db.update_token_counts(
"est", input_tokens=100, model="model-a",
billing_provider="custom",
estimated_cost_usd=1.50, actual_cost_usd=1.0,
cost_status="estimated", cost_source="provider", api_call_count=1,
)
# Included cost session (subscription)
db.create_session(session_id="inc", source="cli", model="gpt-5.4-mini")
db.update_token_counts(
"inc", input_tokens=200, model="gpt-5.4-mini",
billing_provider="openai-codex",
estimated_cost_usd=0.0, actual_cost_usd=0.0,
cost_status="included", cost_source="none", api_call_count=1,
)
engine = InsightsEngine(db)
report = engine.generate(days=30)
text = engine.format_terminal(report)
# The cost section should appear with the buckets this DB has
# (estimated + included; no unknown-cost session is created here)
assert "~$1.50" in text # estimated
assert "included" in text.lower()
assert "subscription" in text.lower()
def test_sub_cent_aggregate_estimated_cost_not_zero(self, db):
"""A sub-cent aggregate must not render 'Estimated: ~$0.00' (#79220).
The insights formatters share format_cost_label with per-response
labels; a cheap-model period totaling $0.0046 shows 4dp, not $0.00.
"""
db.create_session(session_id="est", source="cli", model="model-a")
db.update_token_counts(
"est", input_tokens=100, model="model-a",
billing_provider="custom",
estimated_cost_usd=0.0046, actual_cost_usd=0.0,
cost_status="estimated", cost_source="provider", api_call_count=1,
)
engine = InsightsEngine(db)
report = engine.generate(days=30)
terminal_text = engine.format_terminal(report)
gateway_text = engine.format_gateway(report)
assert "~$0.00\n" not in terminal_text
assert "~$0.0046" in terminal_text
assert "~$0.00 estimated" not in gateway_text
assert "~$0.0046 estimated" in gateway_text