# Conflicts: # apps/desktop/e2e/archived-hidden-session-recoverable.spec.ts # apps/desktop/e2e/bot-chat-message-agent-friendly-name.spec.ts # apps/desktop/e2e/bot-mailbox-unreadable-ticket.spec.ts # apps/desktop/e2e/bot-mode-roster-localized.spec.ts # apps/desktop/e2e/bot-mode-row-click-mirrors-registry.spec.ts # apps/desktop/e2e/bot-mode-tab-shows-bot-name.spec.ts # apps/desktop/e2e/bot-roster-group-row-organisation.spec.ts # apps/desktop/e2e/bot-roster-ignores-infra-dirs.spec.ts # apps/desktop/e2e/bot-roster-timestamp-meta.spec.ts # apps/desktop/e2e/bot-roster-user-sections.spec.ts # apps/desktop/e2e/bot-routines-pane-narrow.spec.ts # apps/desktop/e2e/bot-row-open-recent-session.spec.ts # apps/desktop/e2e/bot-tile-ignores-ambient-composer-model.spec.ts # apps/desktop/e2e/group-composer-auto-grow.spec.ts # apps/desktop/e2e/group-create-gate-remote-roster.spec.ts # apps/desktop/e2e/group-prompt-renamed-primary-handle.spec.ts # apps/desktop/e2e/hosted-room-backend-continuity.spec.ts # apps/desktop/e2e/hosted-room-legacy-store-migration.spec.ts # apps/desktop/e2e/settings-scope-chips-bot-title.spec.ts # apps/desktop/e2e/worktree-branch-status.spec.ts # apps/desktop/electron/backend-probes.test.ts # apps/desktop/electron/connection-apply.test.ts # apps/desktop/electron/desktop-electron-pin.test.ts # apps/desktop/electron/desktop-uninstall.test.ts # apps/desktop/electron/gateway-file-download-transport.test.ts # apps/desktop/electron/gateway-stop-before-update.test.ts # apps/desktop/electron/github-api-auth.test.ts # apps/desktop/electron/registry-primary-profile-scope.test.ts # apps/desktop/electron/update-api-check.test.ts # apps/desktop/electron/update-handoff-marker.test.ts # apps/desktop/electron/venv-blocker-scan.test.ts # apps/desktop/scripts/after-extract.test.mjs # apps/desktop/scripts/local-pack-publish.test.mjs # apps/desktop/scripts/tasks-scroll.test.mjs # apps/desktop/src/app/settings/model-settings.test.tsx # apps/desktop/src/app/updates-overlay.blockers.test.tsx # apps/desktop/src/components/desktop-install-overlay.test.tsx # apps/desktop/src/lib/update-copy.test.ts # scripts/ci/check_os_marker_fakes.py # tests-js/desktop-mac-usage-descriptions.test.ts # tests-js/node-engine-alignment.test.ts # tests/agent/lsp/test_install_and_lint_fixes.py # tests/agent/test_command_token_source.py # tests/agent/test_compression_boundary_hook.py # tests/agent/test_create_openai_client_ssl_verify.py # tests/agent/test_custom_provider_ca_probes.py # tests/agent/test_endpoint_blackhole.py # tests/agent/test_estimator_parity.py # tests/agent/test_in_place_compaction.py # tests/agent/test_moa_loop_mode.py # tests/agent/test_model_metadata.py # tests/agent/test_skill_session_platform_gate.py # tests/agent/test_skill_utils.py # tests/agent/test_ssl_ca_guard.py # tests/computer_use/test_doctor.py # tests/cron/test_codex_execution_paths.py # tests/cron/test_cron_bot_chat_delivery.py # tests/cron/test_cron_script.py # tests/cron/test_media_delivery_parity.py # tests/cron/test_misfire_catchup.py # tests/cron/test_parallel_pool.py # tests/cron/test_recurring_eagain_redispatch.py # tests/gateway/test_choice_picker.py # tests/gateway/test_control_socket_windows_live.py # tests/gateway/test_dingtalk.py # tests/gateway/test_feishu.py # tests/gateway/test_feishu_onboard.py # tests/gateway/test_gateway_shutdown.py # tests/gateway/test_matrix.py # tests/gateway/test_model_command_custom_providers.py # tests/gateway/test_reasoning_command.py # tests/gateway/test_runtime_footer.py # tests/gateway/test_session.py # tests/gateway/test_session_hygiene.py # tests/gateway/test_status.py # tests/gateway/test_teams.py # tests/gateway/test_turn_lease.py # tests/gateway/test_whatsapp_connect.py # tests/hermes_cli/test_approvals_command.py # tests/hermes_cli/test_auth_store_lock_concurrent.py # tests/hermes_cli/test_backup.py # tests/hermes_cli/test_banner_git_state.py # tests/hermes_cli/test_certifi_repair.py # tests/hermes_cli/test_cmd_update.py # tests/hermes_cli/test_compat_manifest_targets.py # tests/hermes_cli/test_computer_use_cli.py # tests/hermes_cli/test_cpr_local_leak.py # tests/hermes_cli/test_dashboard_auth_gate.py # tests/hermes_cli/test_dashboard_procs_kill_grace.py # tests/hermes_cli/test_desktop_lifecycle_windows_live.py # tests/hermes_cli/test_doctor.py # tests/hermes_cli/test_doctor_command_install.py # tests/hermes_cli/test_fleet_config_migration_windows_live.py # tests/hermes_cli/test_gateway.py # tests/hermes_cli/test_gateway_platform_gating.py # tests/hermes_cli/test_gateway_restart_loop.py # tests/hermes_cli/test_gateway_task_probe.py # tests/hermes_cli/test_gateway_wsl.py # tests/hermes_cli/test_gui_command.py # tests/hermes_cli/test_install_cua_driver.py # tests/hermes_cli/test_kanban_db.py # tests/hermes_cli/test_lazy_command_exports.py # tests/hermes_cli/test_lazy_refresh_venv_repair.py # tests/hermes_cli/test_linux_desktop_entry.py # tests/hermes_cli/test_local_runtime.py # tests/hermes_cli/test_local_runtime_updates.py # tests/hermes_cli/test_managed_uv.py # tests/hermes_cli/test_mcp_reload_confirm_gate.py # tests/hermes_cli/test_nous_subscription.py # tests/hermes_cli/test_npm_engine.py # tests/hermes_cli/test_personality_none.py # tests/hermes_cli/test_pet_toggle.py # tests/hermes_cli/test_plan_reconciliation_windows_live.py # tests/hermes_cli/test_plugin_event_bus.py # tests/hermes_cli/test_plugin_manifest_v2.py # tests/hermes_cli/test_plugin_packs.py # tests/hermes_cli/test_plugins_cmd.py # tests/hermes_cli/test_plugins_cmd_enable_disable_nested.py # tests/hermes_cli/test_process_identity.py # tests/hermes_cli/test_profiles.py # tests/hermes_cli/test_profiles_sidebar_cache.py # tests/hermes_cli/test_pty_bridge.py # tests/hermes_cli/test_resolve_turn_limit.py # tests/hermes_cli/test_serve_runtime_inventory.py # tests/hermes_cli/test_session_vacuum_config.py # tests/hermes_cli/test_set_config_value.py # tests/hermes_cli/test_signal_handler_kanban_worker.py # tests/hermes_cli/test_slash_confirm_windows.py # tests/hermes_cli/test_stale_pid_guard.py # tests/hermes_cli/test_startup_fast_guards.py # tests/hermes_cli/test_status.py # tests/hermes_cli/test_telegram_managed_bot.py # tests/hermes_cli/test_tools_config.py # tests/hermes_cli/test_update_apply_shallow_count.py # tests/hermes_cli/test_update_autostash.py # tests/hermes_cli/test_update_concurrent_quarantine.py # tests/hermes_cli/test_update_fetch_failure_classifier.py # tests/hermes_cli/test_update_fleet_probe_resume_token.py # tests/hermes_cli/test_update_handoff_backend_reap.py # tests/hermes_cli/test_update_handoff_desktop_rebuild.py # tests/hermes_cli/test_update_head_moved_gate.py # tests/hermes_cli/test_update_host_obligation.py # tests/hermes_cli/test_update_import_guard.py # tests/hermes_cli/test_update_interrupted_recovery.py # tests/hermes_cli/test_update_inventory.py # tests/hermes_cli/test_update_launchd_unloaded_gateway.py # tests/hermes_cli/test_update_missing_configured_deps.py # tests/hermes_cli/test_update_modified_notice.py # tests/hermes_cli/test_update_multiplex_migration_hook.py # tests/hermes_cli/test_update_no_gateway_restart.py # tests/hermes_cli/test_update_orphan_backend_reap.py # tests/hermes_cli/test_update_parked_branch_guard.py # tests/hermes_cli/test_update_post_pull_syntax_guard.py # tests/hermes_cli/test_update_receipt.py # tests/hermes_cli/test_update_self_lock.py # tests/hermes_cli/test_update_shim_fail_closed.py # tests/hermes_cli/test_update_shim_self_lock.py # tests/hermes_cli/test_update_sqlite_remediation.py # tests/hermes_cli/test_update_stale_dashboard.py # tests/hermes_cli/test_update_stale_virtualenv.py # tests/hermes_cli/test_update_venv_health.py # tests/hermes_cli/test_update_venv_ownership_preflight.py # tests/hermes_cli/test_update_wedged_gateway.py # tests/hermes_cli/test_update_yes_flag.py # tests/hermes_cli/test_update_zip_two_phase.py # tests/hermes_cli/test_urllib_security.py # tests/hermes_cli/test_ux_messages_auth_config.py # tests/hermes_cli/test_ux_messages_startup.py # tests/hermes_cli/test_venv_holder_classifier.py # tests/hermes_cli/test_verify_console_scripts.py # tests/hermes_cli/test_verify_core_dependencies.py # tests/hermes_cli/test_web_server.py # tests/hermes_cli/test_web_server_console_ws.py # tests/hermes_cli/test_web_server_ws_ping.py # tests/hermes_cli/test_web_ui_build.py # tests/hermes_state/test_fts_rebuild_admission.py # tests/hermes_state/test_hermes_state.py # tests/plugins/memory/test_memory_lazy_install.py # tests/plugins/test_google_meet_plugin.py # tests/plugins/test_langfuse_plugin.py # tests/plugins/test_security_guidance_plugin.py # tests/plugins/test_transform_llm_output_hook.py # tests/scripts/desktop_update/test_desktop_update_windows_gateway_flag.py # tests/scripts/desktop_update/test_desktop_update_windows_python_handoff.py # tests/scripts/desktop_update/test_desktop_update_windows_timestamp.py # tests/scripts/install/test_install_clone_throttle_fallback.py # tests/scripts/install/test_install_lockfile_churn.py # tests/scripts/install/test_install_no_initial_commit.py # tests/scripts/install/test_install_sh_browser_install.py # tests/scripts/install/test_install_sh_node_prerelease.py # tests/scripts/install/test_install_sh_symlink_stomp.py # tests/scripts/install/test_install_sh_uv_lock_config.py # tests/scripts/install/test_install_unmerged_index.py # tests/scripts/test_contributor_map.py # tests/scripts/test_run_tests_parallel.py # tests/skills/test_competitor_news_monitor_skill.py # tests/skills/test_document_to_action_items_skill.py # tests/skills/test_google_workspace_setup.py # tests/skills/test_google_workspace_setup_deps.py # tests/skills/test_grounded_citations_skill.py # tests/skills/test_ip_as_logo_skill.py # tests/skills/test_live_dashboard_skill.py # tests/skills/test_mcp_oauth_remote_gateway_skill.py # tests/skills/test_office_document_skills.py # tests/skills/test_openclaw_migration.py # tests/skills/test_product_price_monitor_skill.py # tests/skills/test_scrollcraft_skill.py # tests/skills/test_setup_wizard_generator_skill.py # tests/skills/test_weekly_review_planning_skill.py # tests/test_engines_satisfiable.py # tests/test_fast_safe_load.py # tests/test_hermes_bootstrap.py # tests/test_hermes_constants.py # tests/test_hermes_logging.py # tests/test_managed_runtime_resolution.py # tests/test_model_tools_async_bridge.py # tests/test_packaging_build_guard.py # tests/test_packaging_metadata.py # tests/test_yaml_indent_consistency.py # tests/tools/test_approval_timeout_overflow.py # tests/tools/test_base_environment.py # tests/tools/test_bot_mode_dm.py # tests/tools/test_browser_chromium_check.py # tests/tools/test_browser_hardening.py # tests/tools/test_browser_homebrew_paths.py # tests/tools/test_browser_npx_warmup.py # tests/tools/test_browser_orphan_reaper.py # tests/tools/test_browser_real_profile.py # tests/tools/test_browser_use_cli.py # tests/tools/test_clipboard.py # tests/tools/test_code_execution.py # tests/tools/test_code_execution_modes.py # tests/tools/test_code_execution_windows_env.py # tests/tools/test_computer_use.py # tests/tools/test_delegate_liveness_timeout.py # tests/tools/test_execute_code_approval_cluster.py # tests/tools/test_execution_flag_detection.py # tests/tools/test_fal_common.py # tests/tools/test_file_operations.py # tests/tools/test_file_tools.py # tests/tools/test_file_tools_cwd_resolution.py # tests/tools/test_file_tools_live.py # tests/tools/test_lazy_deps.py # tests/tools/test_lazy_deps_durable_target.py # tests/tools/test_lazy_deps_managed.py # tests/tools/test_local_env_blocklist.py # tests/tools/test_local_tempdir.py # tests/tools/test_macos_protected_search.py # tests/tools/test_mcp_npx_cached_bin.py # tests/tools/test_oneshot_completion_linger.py # tests/tools/test_process_registry.py # tests/tools/test_read_file_schema_gating.py # tests/tools/test_skill_improvements.py # tests/tools/test_skills_sync.py # tests/tools/test_termux_api_detection.py # tests/tools/test_tirith_security.py # tests/tools/test_transcription_tools.py # tests/tools/test_tts_streaming.py # tests/tools/test_wake_word.py # tests/tui_gateway/test_compute_host_borrowed_lease.py # tests/tui_gateway/test_compute_host_turn_protocol.py # tests/tui_gateway/test_isolated_orphan_activity.py # tests/tui_gateway/test_protocol.py # tests/tui_gateway/test_slash_worker_profile_home.py # tests/tui_gateway/test_subprocess_encoding.py # tests/tui_gateway/test_tui_gateway_server.py # ui-tui/src/__tests__/terminalParity.test.ts # ui-tui/src/__tests__/termuxComposerLayout.test.ts # ui-tui/src/__tests__/textInputFastEcho.test.ts
593 lines
26 KiB
Python
593 lines
26 KiB
Python
"""Tests for agent/insights.py — InsightsEngine analytics and reporting."""
|
|
|
|
import time
|
|
import pytest
|
|
|
|
from hermes_state import SessionDB
|
|
from agent.insights import (
|
|
InsightsEngine,
|
|
_estimate_cost,
|
|
_bar_chart,
|
|
)
|
|
from agent.usage_pricing import (
|
|
format_duration_compact as _format_duration,
|
|
has_known_pricing as _has_known_pricing,
|
|
)
|
|
|
|
@pytest.fixture()
|
|
def db(tmp_path):
|
|
"""Create a SessionDB with a temp database file."""
|
|
db_path = tmp_path / "test_insights.db"
|
|
session_db = SessionDB(db_path=db_path)
|
|
yield session_db
|
|
session_db.close()
|
|
|
|
@pytest.fixture()
|
|
def populated_db(db):
|
|
"""Create a DB with realistic session data for insights testing."""
|
|
now = time.time()
|
|
day = 86400
|
|
|
|
# Session 1: CLI, claude-sonnet, ended, 2 days ago
|
|
db.create_session(
|
|
session_id="s1", source="cli",
|
|
model="anthropic/claude-sonnet-4-20250514", user_id="user1",
|
|
)
|
|
# Backdate the started_at
|
|
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's1'", (now - 2 * day,))
|
|
db.end_session("s1", end_reason="user_exit")
|
|
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's1'", (now - 2 * day + 3600,))
|
|
db.update_token_counts("s1", input_tokens=50000, output_tokens=15000)
|
|
db.append_message("s1", role="user", content="Hello, help me fix a bug")
|
|
db.append_message("s1", role="assistant", content="Sure, let me look into that.")
|
|
db.append_message("s1", role="assistant", content="Let me search the files.",
|
|
tool_calls=[{"function": {"name": "search_files"}}])
|
|
db.append_message("s1", role="tool", content="Found 3 matches", tool_name="search_files")
|
|
db.append_message("s1", role="assistant", content="Let me read the file.",
|
|
tool_calls=[{"function": {"name": "read_file"}}])
|
|
db.append_message("s1", role="tool", content="file contents...", tool_name="read_file")
|
|
db.append_message("s1", role="assistant", content="I found the bug. Let me fix it.",
|
|
tool_calls=[{"function": {"name": "patch"}}])
|
|
db.append_message("s1", role="tool", content="patched successfully", tool_name="patch")
|
|
db.append_message(
|
|
"s1",
|
|
role="assistant",
|
|
content="Let me load the PR workflow skill.",
|
|
tool_calls=[{"function": {"name": "skill_view", "arguments": '{"name":"github-pr-workflow"}'}}],
|
|
)
|
|
db.append_message("s1", role="user", content="Thanks!")
|
|
db.append_message("s1", role="assistant", content="You're welcome!")
|
|
|
|
# Session 2: Telegram, gpt-4o, ended, 5 days ago
|
|
db.create_session(
|
|
session_id="s2", source="telegram",
|
|
model="gpt-4o", user_id="user1",
|
|
)
|
|
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's2'", (now - 5 * day,))
|
|
db.end_session("s2", end_reason="timeout")
|
|
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's2'", (now - 5 * day + 1800,))
|
|
db.update_token_counts("s2", input_tokens=20000, output_tokens=8000)
|
|
db.append_message("s2", role="user", content="Search the web for something")
|
|
db.append_message("s2", role="assistant", content="Searching...",
|
|
tool_calls=[{"function": {"name": "web_search"}}])
|
|
db.append_message("s2", role="tool", content="results...", tool_name="web_search")
|
|
db.append_message("s2", role="assistant", content="Here's what I found")
|
|
|
|
# Session 3: CLI, deepseek-chat, ended, 10 days ago
|
|
db.create_session(
|
|
session_id="s3", source="cli",
|
|
model="deepseek-chat", user_id="user1",
|
|
)
|
|
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's3'", (now - 10 * day,))
|
|
db.end_session("s3", end_reason="user_exit")
|
|
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's3'", (now - 10 * day + 7200,))
|
|
db.update_token_counts("s3", input_tokens=100000, output_tokens=40000)
|
|
db.append_message("s3", role="user", content="Run this terminal command")
|
|
db.append_message("s3", role="assistant", content="Running...",
|
|
tool_calls=[{"function": {"name": "terminal"}}])
|
|
db.append_message("s3", role="tool", content="output...", tool_name="terminal")
|
|
db.append_message("s3", role="assistant", content="Let me run another",
|
|
tool_calls=[{"function": {"name": "terminal"}}])
|
|
db.append_message("s3", role="tool", content="more output...", tool_name="terminal")
|
|
db.append_message("s3", role="assistant", content="And search files",
|
|
tool_calls=[{"function": {"name": "search_files"}}])
|
|
db.append_message("s3", role="tool", content="found stuff", tool_name="search_files")
|
|
db.append_message(
|
|
"s3",
|
|
role="assistant",
|
|
content="Load the debugging skill.",
|
|
tool_calls=[{"function": {"name": "skill_view", "arguments": '{"name":"systematic-debugging"}'}}],
|
|
)
|
|
|
|
# Session 4: Discord, same model as s1, ended, 1 day ago
|
|
db.create_session(
|
|
session_id="s4", source="discord",
|
|
model="anthropic/claude-sonnet-4-20250514", user_id="user2",
|
|
)
|
|
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's4'", (now - 1 * day,))
|
|
db.end_session("s4", end_reason="user_exit")
|
|
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's4'", (now - 1 * day + 900,))
|
|
db.update_token_counts("s4", input_tokens=10000, output_tokens=5000)
|
|
db.append_message("s4", role="user", content="Quick question")
|
|
db.append_message("s4", role="assistant", content="Sure, go ahead")
|
|
db.append_message(
|
|
"s4",
|
|
role="assistant",
|
|
content="Load and update GitHub skills.",
|
|
tool_calls=[
|
|
{"function": {"name": "skill_view", "arguments": '{"name":"github-pr-workflow"}'}},
|
|
{"function": {"name": "skill_manage", "arguments": '{"name":"github-code-review"}'}},
|
|
],
|
|
)
|
|
|
|
# Session 5: Old session, 45 days ago (should be excluded from 30-day window)
|
|
db.create_session(
|
|
session_id="s_old", source="cli",
|
|
model="gpt-4o-mini", user_id="user1",
|
|
)
|
|
db._conn.execute("UPDATE sessions SET started_at = ? WHERE id = 's_old'", (now - 45 * day,))
|
|
db.end_session("s_old", end_reason="user_exit")
|
|
db._conn.execute("UPDATE sessions SET ended_at = ? WHERE id = 's_old'", (now - 45 * day + 600,))
|
|
db.update_token_counts("s_old", input_tokens=5000, output_tokens=2000)
|
|
db.append_message("s_old", role="user", content="old message")
|
|
db.append_message("s_old", role="assistant", content="old reply")
|
|
|
|
db._conn.commit()
|
|
return db
|
|
|
|
class TestHasKnownPricing:
|
|
|
|
def test_unknown_custom_model(self):
|
|
assert _has_known_pricing("FP16_Hermes_4.5") is False
|
|
assert _has_known_pricing("my-custom-model") is False
|
|
assert _has_known_pricing("glm-5") is False
|
|
assert _has_known_pricing("") is False
|
|
assert _has_known_pricing(None) is False
|
|
|
|
def test_heuristic_matched_models_are_not_considered_known(self):
|
|
assert _has_known_pricing("some-opus-model") is False
|
|
assert _has_known_pricing("future-sonnet-v2") is False
|
|
|
|
class TestEstimateCost:
|
|
|
|
def test_cache_aware_usage(self):
|
|
cost, status = _estimate_cost(
|
|
"anthropic/claude-sonnet-4-20250514",
|
|
1000,
|
|
500,
|
|
cache_read_tokens=2000,
|
|
cache_write_tokens=400,
|
|
provider="anthropic",
|
|
)
|
|
assert status == "estimated"
|
|
expected = (1000 * 3.0 + 500 * 15.0 + 2000 * 0.30 + 400 * 3.75) / 1_000_000
|
|
assert cost == pytest.approx(expected, abs=0.0001)
|
|
|
|
# =========================================================================
|
|
# Format helpers
|
|
# =========================================================================
|
|
|
|
class TestFormatDuration:
|
|
def test_seconds(self):
|
|
assert _format_duration(45) == "45s"
|
|
|
|
def test_hours_with_minutes(self):
|
|
result = _format_duration(5400) # 1.5 hours
|
|
assert result == "1h 30m"
|
|
|
|
class TestBarChart:
|
|
def test_basic_bars(self):
|
|
bars = _bar_chart([10, 5, 0, 20], max_width=10)
|
|
assert len(bars) == 4
|
|
assert len(bars[3]) == 10 # max value gets full width
|
|
assert len(bars[0]) == 5 # half of max
|
|
assert bars[2] == "" # zero gets empty
|
|
|
|
def test_all_zeros(self):
|
|
bars = _bar_chart([0, 0, 0], max_width=10)
|
|
assert all(b == "" for b in bars)
|
|
|
|
# =========================================================================
|
|
# InsightsEngine — empty DB
|
|
# =========================================================================
|
|
|
|
class TestInsightsEmpty:
|
|
def test_empty_db_returns_empty_report(self, db):
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
assert report["empty"] is True
|
|
assert report["overview"] == {}
|
|
# Both renderers must handle the empty report without crashing.
|
|
assert engine.format_terminal(report)
|
|
assert engine.format_gateway(report)
|
|
|
|
# =========================================================================
|
|
# InsightsEngine — populated DB
|
|
# =========================================================================
|
|
|
|
class TestInsightsPopulated:
|
|
|
|
def test_overview_token_totals(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
report = engine.generate(days=30)
|
|
overview = report["overview"]
|
|
|
|
expected_input = 50000 + 20000 + 100000 + 10000
|
|
expected_output = 15000 + 8000 + 40000 + 5000
|
|
assert overview["total_input_tokens"] == expected_input
|
|
assert overview["total_output_tokens"] == expected_output
|
|
assert overview["total_tokens"] == expected_input + expected_output
|
|
|
|
def test_model_breakdown_splits_mid_session_switch(self, db):
|
|
"""A session that switches models mid-flight is split across both
|
|
models in the breakdown, not dumped on the initial model (#51607).
|
|
"""
|
|
now = time.time()
|
|
db.create_session(session_id="sw", source="cli",
|
|
model="deepseek/deepseek-v4-pro")
|
|
# 40k tokens on deepseek, then switch and 50k on opus.
|
|
db.update_token_counts("sw", input_tokens=40000, output_tokens=8000,
|
|
model="deepseek/deepseek-v4-pro",
|
|
billing_provider="deepseek", api_call_count=2)
|
|
db.update_session_model("sw", "anthropic/claude-opus-4.8")
|
|
db.update_token_counts("sw", input_tokens=50000, output_tokens=4000,
|
|
model="anthropic/claude-opus-4.8",
|
|
billing_provider="openrouter", api_call_count=3)
|
|
db._conn.commit()
|
|
|
|
report = InsightsEngine(db).generate(days=30)
|
|
models = {m["model"]: m for m in report["models"]}
|
|
assert "deepseek-v4-pro" in models
|
|
assert "claude-opus-4.8" in models
|
|
# Tokens attributed to the model that actually incurred them.
|
|
assert models["deepseek-v4-pro"]["input_tokens"] == 40000
|
|
assert models["claude-opus-4.8"]["input_tokens"] == 50000
|
|
assert models["claude-opus-4.8"]["api_calls"] == 3
|
|
# The summary row's single model would have hidden one of these.
|
|
assert models["deepseek-v4-pro"]["total_tokens"] == 48000
|
|
assert models["claude-opus-4.8"]["total_tokens"] == 54000
|
|
|
|
def test_overview_cost_matches_per_model_stored_cost(self, db):
|
|
db.create_session(session_id="cost", source="cli", model="model-a")
|
|
db.update_token_counts(
|
|
"cost", input_tokens=10, model="model-a", billing_provider="custom",
|
|
estimated_cost_usd=1.25, actual_cost_usd=1.0,
|
|
cost_status="estimated", cost_source="provider", api_call_count=1,
|
|
)
|
|
db.update_session_model("cost", "model-b")
|
|
db.update_session_billing_route("cost", provider="custom-b", base_url=None)
|
|
db.update_token_counts(
|
|
"cost", input_tokens=20, model="model-b", billing_provider="custom-b",
|
|
estimated_cost_usd=2.5, actual_cost_usd=2.0,
|
|
cost_status="estimated", cost_source="provider", api_call_count=1,
|
|
)
|
|
|
|
report = InsightsEngine(db).generate(days=30)
|
|
assert sum(m["cost"] for m in report["models"]) == pytest.approx(3.75)
|
|
assert report["overview"]["estimated_cost"] == pytest.approx(3.75)
|
|
assert report["overview"]["actual_cost"] == pytest.approx(3.0)
|
|
|
|
def test_tool_usage_sums_disjoint_sessions_without_double_counting_pairs(self, db):
|
|
"""One session records a call as tool_name only, another as tool_calls only: both count.
|
|
A session carrying BOTH representations of the same call still counts it once (#9814)."""
|
|
db.create_session(session_id="gw", source="gateway", model="m")
|
|
db.append_message("gw", role="tool", content="r", tool_name="search_files")
|
|
db.create_session(session_id="cli", source="cli", model="m")
|
|
db.append_message("cli", role="assistant", content="x",
|
|
tool_calls=[{"function": {"name": "search_files", "arguments": "{}"}}])
|
|
db.create_session(session_id="both", source="cli", model="m")
|
|
db.append_message("both", role="assistant", content="x",
|
|
tool_calls=[{"function": {"name": "search_files", "arguments": "{}"}}])
|
|
db.append_message("both", role="tool", content="r", tool_name="search_files")
|
|
db._conn.commit()
|
|
|
|
tools = InsightsEngine(db).generate(days=30)["tools"]
|
|
assert next(t["count"] for t in tools if t["tool"] == "search_files") == 3
|
|
|
|
def test_tool_breakdown(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
report = engine.generate(days=30)
|
|
tools = report["tools"]
|
|
|
|
tool_names = [t["tool"] for t in tools]
|
|
assert "terminal" in tool_names
|
|
assert "search_files" in tool_names
|
|
assert "read_file" in tool_names
|
|
assert "patch" in tool_names
|
|
assert "web_search" in tool_names
|
|
|
|
# terminal was used 2x in s3
|
|
terminal = next(t for t in tools if t["tool"] == "terminal")
|
|
assert terminal["count"] == 2
|
|
|
|
# Percentages should sum to ~100%
|
|
total_pct = sum(t["percentage"] for t in tools)
|
|
assert total_pct == pytest.approx(100.0, abs=0.1)
|
|
|
|
def test_skill_breakdown(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
report = engine.generate(days=30)
|
|
skills = report["skills"]
|
|
|
|
assert skills["summary"]["distinct_skills_used"] == 3
|
|
assert skills["summary"]["total_skill_loads"] == 3
|
|
assert skills["summary"]["total_skill_edits"] == 1
|
|
assert skills["summary"]["total_skill_actions"] == 4
|
|
|
|
top_skill = skills["top_skills"][0]
|
|
assert top_skill["skill"] == "github-pr-workflow"
|
|
assert top_skill["view_count"] == 2
|
|
assert top_skill["manage_count"] == 0
|
|
assert top_skill["total_count"] == 2
|
|
assert top_skill["last_used_at"] is not None
|
|
|
|
# The Insights assistant tool-call queries pin
|
|
# idx_messages_assistant_calls_by_session via INDEXED BY. These tests prove
|
|
# (a) the planner uses that index for BOTH the unfiltered and source-filtered
|
|
# branches on a fresh DB *without* ANALYZE, and (b) the index is a pure
|
|
# optimization — output is identical whether or not it is selected.
|
|
_INDEX = "idx_messages_assistant_calls_by_session"
|
|
_PINNED_QUERIES = (
|
|
("_GET_TOOL_CALLS_ALL", (0.0,)),
|
|
("_GET_TOOL_CALLS_WITH_SOURCE", (0.0, "cli")),
|
|
("_GET_SKILL_CALLS_ALL", (0.0,)),
|
|
("_GET_SKILL_CALLS_WITH_SOURCE", (0.0, "cli")),
|
|
)
|
|
|
|
def test_assistant_call_queries_use_partial_index_without_analyze(
|
|
self, populated_db
|
|
):
|
|
"""Every fixed-predicate branch selects the partial index on a fresh DB.
|
|
|
|
No ANALYZE is run, so this covers the default-statistics case a freshly
|
|
initialized state.db is actually in. Both the unfiltered and the
|
|
source-filtered (``s.source = ?``) branches are checked.
|
|
"""
|
|
# Guard against the fresh-DB planner regression the reviewers found:
|
|
# without INDEXED BY the source-filtered branch fell back to
|
|
# idx_messages_session_active.
|
|
assert "ANALYZE" not in "".join(
|
|
r["sql"] or ""
|
|
for r in populated_db._conn.execute(
|
|
"SELECT sql FROM sqlite_master WHERE type = 'index'"
|
|
)
|
|
)
|
|
for attr, params in self._PINNED_QUERIES:
|
|
sql = getattr(InsightsEngine, attr)
|
|
plan = "\n".join(
|
|
row["detail"]
|
|
for row in populated_db._conn.execute(
|
|
"EXPLAIN QUERY PLAN " + sql, params
|
|
).fetchall()
|
|
)
|
|
assert self._INDEX in plan, f"{attr} did not use the index:\n{plan}"
|
|
|
|
def test_assistant_call_rows_invariant_to_index_selection(self, populated_db):
|
|
"""The pinned index only changes the plan, never the result set.
|
|
|
|
For every branch, the index-pinned query and the un-pinned form (whose
|
|
plan the optimizer chooses freely) must return identical rows — proving
|
|
the index is a pure optimization — for both the unfiltered and
|
|
source-filtered scopes.
|
|
"""
|
|
assert populated_db._conn.execute(
|
|
"SELECT 1 FROM sqlite_master WHERE type = 'index' AND name = ?",
|
|
(self._INDEX,),
|
|
).fetchone() is not None
|
|
|
|
for attr, params in self._PINNED_QUERIES:
|
|
pinned_sql = getattr(InsightsEngine, attr)
|
|
unpinned_sql = pinned_sql.replace(f" INDEXED BY {self._INDEX}", "")
|
|
pinned = [
|
|
tuple(r) for r in
|
|
populated_db._conn.execute(pinned_sql, params).fetchall()
|
|
]
|
|
unpinned = [
|
|
tuple(r) for r in
|
|
populated_db._conn.execute(unpinned_sql, params).fetchall()
|
|
]
|
|
assert sorted(pinned) == sorted(unpinned), attr
|
|
|
|
def test_missing_index_falls_back_to_unpinned_queries(self, populated_db):
|
|
"""INDEXED BY would be a hard error if the index is missing — which
|
|
happens on read-only opens of a state.db written by an older version
|
|
(web dashboard analytics). The engine must probe and fall back to the
|
|
unpinned variants instead of crashing, returning identical rows."""
|
|
engine_pinned = InsightsEngine(populated_db)
|
|
tools_before = engine_pinned._get_tool_usage(0.0)
|
|
|
|
populated_db._conn.execute(f"DROP INDEX IF EXISTS {self._INDEX}")
|
|
populated_db._conn.commit()
|
|
|
|
engine = InsightsEngine(populated_db)
|
|
assert engine._has_assistant_calls_index is False
|
|
assert "INDEXED BY" not in engine._GET_TOOL_CALLS_ALL
|
|
tools_after = engine._get_tool_usage(0.0)
|
|
assert sorted(t["tool_name"] for t in tools_after) == sorted(
|
|
t["tool_name"] for t in tools_before
|
|
)
|
|
# And with the index present, the pin stays.
|
|
assert "INDEXED BY" in InsightsEngine._GET_TOOL_CALLS_ALL
|
|
|
|
def test_get_usage_breakdown_matches_full_generate(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
full = engine.generate(days=30)
|
|
focused = engine.get_usage_breakdown(days=30)
|
|
assert focused["skills"] == full["skills"]
|
|
assert focused["tools"] == full["tools"]
|
|
|
|
def test_get_skill_breakdown_respects_source_filter(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
# Only s1 (cli) has skill_view "github-pr-workflow"
|
|
focused = engine.get_usage_breakdown(days=30, source="cli")["skills"]
|
|
skill_names = [s["skill"] for s in focused["top_skills"]]
|
|
assert "github-pr-workflow" in skill_names
|
|
# github-code-review was in discord (s4), not cli
|
|
assert "github-code-review" not in skill_names
|
|
|
|
def test_get_skill_usage_prefilter_ignores_non_skill_substring(self, db):
|
|
# "my_skill_view_helper" contains "skill_view" as a substring; instr()
|
|
# will match but the Python-side name check keeps the set clean.
|
|
# More importantly, messages with no skill_* tools must be excluded.
|
|
db.create_session(session_id="sx", source="cli", model="gpt-4o")
|
|
db.append_message(
|
|
"sx",
|
|
role="assistant",
|
|
content="Just using read_file.",
|
|
tool_calls=[{"function": {"name": "read_file", "arguments": '{"path":"/tmp/x"}'}}],
|
|
)
|
|
db._conn.commit()
|
|
focused = InsightsEngine(db).get_usage_breakdown(days=30)["skills"]
|
|
assert focused["summary"]["total_skill_actions"] == 0
|
|
assert focused["top_skills"] == []
|
|
|
|
# =========================================================================
|
|
# Formatting
|
|
# =========================================================================
|
|
|
|
class TestTerminalFormatting:
|
|
|
|
def test_terminal_format_unknown_bucket_for_custom_models(self, db):
|
|
"""Custom models with no pricing surface as the Unknown bucket (#77223)."""
|
|
db.create_session(session_id="s1", source="cli", model="my-custom-model")
|
|
db.update_token_counts("s1", input_tokens=1000, output_tokens=500)
|
|
db._conn.commit()
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
text = engine.format_terminal(report)
|
|
|
|
# Cost section surfaces unknown-cost sessions (#77223) instead of
|
|
# hiding them — a custom model with no pricing data shows in the
|
|
# Unknown bucket rather than silently reporting $0.
|
|
assert "Unknown" in text
|
|
|
|
class TestGatewayFormatting:
|
|
def test_gateway_format_is_shorter(self, populated_db):
|
|
engine = InsightsEngine(populated_db)
|
|
report = engine.generate(days=30)
|
|
terminal_text = engine.format_terminal(report)
|
|
gateway_text = engine.format_gateway(report)
|
|
|
|
assert len(gateway_text) < len(terminal_text)
|
|
|
|
# =========================================================================
|
|
# Edge cases
|
|
# =========================================================================
|
|
|
|
class TestEdgeCases:
|
|
|
|
def test_session_with_no_model(self, db):
|
|
"""Sessions with NULL model should not crash."""
|
|
db.create_session(session_id="s1", source="cli")
|
|
db.update_token_counts("s1", input_tokens=1000, output_tokens=500)
|
|
db._conn.commit()
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
assert report["empty"] is False
|
|
|
|
models = report["models"]
|
|
assert len(models) == 1
|
|
assert models[0]["model"] == "unknown"
|
|
assert models[0]["has_pricing"] is False
|
|
|
|
def test_mixed_commercial_and_custom_models(self, db):
|
|
"""Mix of commercial and custom models: only commercial ones get costs."""
|
|
db.create_session(session_id="s1", source="cli", model="anthropic/claude-sonnet-4-20250514")
|
|
db.update_token_counts(
|
|
"s1",
|
|
input_tokens=10000,
|
|
output_tokens=5000,
|
|
billing_provider="anthropic",
|
|
)
|
|
db.create_session(session_id="s2", source="cli", model="my-local-llama")
|
|
db.update_token_counts("s2", input_tokens=10000, output_tokens=5000)
|
|
db._conn.commit()
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
|
|
# Cost should only come from gpt-4o, not from the custom model
|
|
overview = report["overview"]
|
|
assert overview["estimated_cost"] > 0
|
|
assert "claude-sonnet-4-20250514" in overview["models_with_pricing"] # list now, not set
|
|
assert "my-local-llama" in overview["models_without_pricing"]
|
|
|
|
# Verify individual model entries
|
|
claude = next(m for m in report["models"] if m["model"] == "claude-sonnet-4-20250514")
|
|
assert claude["has_pricing"] is True
|
|
assert claude["cost"] > 0
|
|
|
|
llama = next(m for m in report["models"] if m["model"] == "my-local-llama")
|
|
assert llama["has_pricing"] is False
|
|
assert llama["cost"] == 0.0
|
|
|
|
def test_only_one_platform(self, db):
|
|
"""Single-platform usage should still work."""
|
|
db.create_session(session_id="s1", source="cli", model="test")
|
|
db._conn.commit()
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
assert len(report["platforms"]) == 1
|
|
assert report["platforms"][0]["platform"] == "cli"
|
|
|
|
# Terminal format should NOT show platform section for single platform
|
|
text = engine.format_terminal(report)
|
|
# (it still shows platforms section if there's only cli and nothing else)
|
|
# Actually the condition is > 1 platforms OR non-cli, so single cli won't show
|
|
|
|
def test_cost_buckets_displayed_in_terminal_format(self, db):
|
|
"""#77223: included/estimated/unknown cost buckets surface in terminal."""
|
|
# Estimated cost session
|
|
db.create_session(session_id="est", source="cli", model="model-a")
|
|
db.update_token_counts(
|
|
"est", input_tokens=100, model="model-a",
|
|
billing_provider="custom",
|
|
estimated_cost_usd=1.50, actual_cost_usd=1.0,
|
|
cost_status="estimated", cost_source="provider", api_call_count=1,
|
|
)
|
|
# Included cost session (subscription)
|
|
db.create_session(session_id="inc", source="cli", model="gpt-5.4-mini")
|
|
db.update_token_counts(
|
|
"inc", input_tokens=200, model="gpt-5.4-mini",
|
|
billing_provider="openai-codex",
|
|
estimated_cost_usd=0.0, actual_cost_usd=0.0,
|
|
cost_status="included", cost_source="none", api_call_count=1,
|
|
)
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
text = engine.format_terminal(report)
|
|
|
|
# The cost section should appear with the buckets this DB has
|
|
# (estimated + included; no unknown-cost session is created here)
|
|
assert "~$1.50" in text # estimated
|
|
assert "included" in text.lower()
|
|
assert "subscription" in text.lower()
|
|
|
|
def test_sub_cent_aggregate_estimated_cost_not_zero(self, db):
|
|
"""A sub-cent aggregate must not render 'Estimated: ~$0.00' (#79220).
|
|
|
|
The insights formatters share format_cost_label with per-response
|
|
labels; a cheap-model period totaling $0.0046 shows 4dp, not $0.00.
|
|
"""
|
|
db.create_session(session_id="est", source="cli", model="model-a")
|
|
db.update_token_counts(
|
|
"est", input_tokens=100, model="model-a",
|
|
billing_provider="custom",
|
|
estimated_cost_usd=0.0046, actual_cost_usd=0.0,
|
|
cost_status="estimated", cost_source="provider", api_call_count=1,
|
|
)
|
|
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=30)
|
|
terminal_text = engine.format_terminal(report)
|
|
gateway_text = engine.format_gateway(report)
|
|
|
|
assert "~$0.00\n" not in terminal_text
|
|
assert "~$0.0046" in terminal_text
|
|
assert "~$0.00 estimated" not in gateway_text
|
|
assert "~$0.0046 estimated" in gateway_text
|