850 lines
37 KiB
Python
850 lines
37 KiB
Python
"""Read-only gateway introspection commands: /status, /context, /usage, /agents, /insights, /topup.
|
|
|
|
Split out of ``gateway/slash_commands.py``; bound onto ``GatewayRunner`` through
|
|
``GatewaySlashCommandsMixin``. Origin internals are imported lazily (``from gateway.slash_commands
|
|
import ...``) inside the bodies to avoid the import cycle.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import asyncio
|
|
import hashlib
|
|
import os
|
|
import re
|
|
import time
|
|
from typing import Any
|
|
|
|
from agent.account_usage import fetch_account_usage, render_account_usage_lines
|
|
from agent.i18n import t
|
|
from gateway.config import Platform
|
|
from gateway.platforms.base import MessageEvent
|
|
|
|
# Log-record parity with gateway/run.py and the origin module.
|
|
logger = logging.getLogger("gateway.run")
|
|
|
|
|
|
def _clean_str(value: Any) -> str:
|
|
"""Strip and return a non-empty string value, or empty string."""
|
|
return value.strip() if isinstance(value, str) and value.strip() else ""
|
|
|
|
|
|
def _int_value(value: Any) -> int:
|
|
"""Safely coerce to int."""
|
|
try:
|
|
return int(value)
|
|
except (TypeError, ValueError):
|
|
return 0
|
|
|
|
|
|
def _status_model_route(status_agent, persisted_route: dict, session_row: dict, session_entry):
|
|
"""``(model, provider, context_used, context_total)`` for /status.
|
|
|
|
Order: live/cached agent route -> persisted dominant route -> SessionDB row -> gateway config
|
|
(only loaded when something is still missing).
|
|
"""
|
|
from gateway.run import _AGENT_PENDING_SENTINEL, _load_gateway_config, _resolve_gateway_model
|
|
|
|
model_name = provider_name = ""
|
|
route_resolved = False
|
|
context_used = context_total = 0
|
|
if status_agent is not None and status_agent is not _AGENT_PENDING_SENTINEL:
|
|
live_model = _clean_str(getattr(status_agent, "model", ""))
|
|
live_provider = _clean_str(getattr(status_agent, "provider", ""))
|
|
if live_model and live_provider:
|
|
model_name, provider_name, route_resolved = live_model, live_provider, True
|
|
ctx = getattr(status_agent, "context_compressor", None)
|
|
if ctx is not None:
|
|
context_used = _int_value(getattr(ctx, "last_prompt_tokens", 0))
|
|
context_total = _int_value(getattr(ctx, "context_length", 0))
|
|
|
|
persisted_model = _clean_str(persisted_route.get("model"))
|
|
persisted_provider = _clean_str(persisted_route.get("billing_provider"))
|
|
if not route_resolved and persisted_model and persisted_provider:
|
|
model_name, provider_name, route_resolved = persisted_model, persisted_provider, True
|
|
if not route_resolved:
|
|
model_name = _clean_str(session_row.get("model"))
|
|
provider_name = _clean_str(session_row.get("billing_provider"))
|
|
context_used = context_used or _int_value(getattr(session_entry, "last_prompt_tokens", 0))
|
|
|
|
user_config: dict[str, Any] = {}
|
|
if not model_name or not provider_name or not context_total:
|
|
try:
|
|
user_config = _load_gateway_config()
|
|
except Exception:
|
|
user_config = {}
|
|
model_cfg = user_config.get("model", {}) if isinstance(user_config, dict) else {}
|
|
if not isinstance(model_cfg, dict):
|
|
model_cfg = {}
|
|
if not model_name:
|
|
model_name = _resolve_gateway_model(user_config)
|
|
if not provider_name:
|
|
provider_name = _clean_str(model_cfg.get("provider"))
|
|
if not context_total:
|
|
configured_context = model_cfg.get("context_length")
|
|
if isinstance(configured_context, int) and configured_context > 0:
|
|
context_total = configured_context
|
|
return model_name, provider_name, context_used, context_total
|
|
|
|
|
|
def _context_compressor_lines(agent, ctx, used: int) -> list[str]:
|
|
"""/context full view: auto-compression threshold/headroom, compression count + last savings,
|
|
and cumulative throughput (labelled as throughput, NOT context size)."""
|
|
lines: list[str] = []
|
|
threshold = getattr(ctx, "threshold_tokens", 0) or 0
|
|
threshold_pct = (getattr(ctx, "threshold_percent", 0) or 0) * 100
|
|
if threshold > 0:
|
|
if used >= threshold:
|
|
lines.append(
|
|
t("gateway.context.over_threshold", threshold=f"{threshold:,}", threshold_pct=f"{threshold_pct:.0f}")
|
|
)
|
|
else:
|
|
lines.append(
|
|
t(
|
|
"gateway.context.threshold",
|
|
threshold=f"{threshold:,}",
|
|
threshold_pct=f"{threshold_pct:.0f}",
|
|
to_go=f"{threshold - used:,}",
|
|
)
|
|
)
|
|
compressions = getattr(ctx, "compression_count", 0) or 0
|
|
lines.append(t("gateway.context.compressions", count=compressions))
|
|
if compressions:
|
|
savings = getattr(ctx, "_last_compression_savings_pct", None)
|
|
if savings is not None:
|
|
lines.append(t("gateway.context.last_savings", savings=f"{savings:.0f}"))
|
|
|
|
def _n(attr):
|
|
return getattr(agent, attr, 0) or 0
|
|
|
|
lines.append("")
|
|
lines.append(t("gateway.context.totals_header", calls=_n("session_api_calls")))
|
|
lines.append(
|
|
t(
|
|
"gateway.context.totals_line",
|
|
input=f"{_n('session_input_tokens'):,}",
|
|
output=f"{_n('session_output_tokens'):,}",
|
|
reasoning=f"{_n('session_reasoning_tokens'):,}",
|
|
)
|
|
)
|
|
lines.append(t("gateway.context.total_billed", total=f"{_n('session_total_tokens'):,}"))
|
|
lines.append(t("gateway.context.throughput_note"))
|
|
return lines
|
|
|
|
|
|
def _agents_delegation_lines(d: dict) -> list[str]:
|
|
"""/agents rows for one background delegation. Live per-child activity comes from the
|
|
registry's progress sampler: api calls, current tool, seconds since last activity."""
|
|
goal = " ".join(str(d.get("goal") or "").split())
|
|
if len(goal) > 70:
|
|
goal = goal[:67] + "..."
|
|
status = d.get("status", "?")
|
|
row = f"- `{d.get('delegation_id', '?')}` · {status}"
|
|
if status == "stalling":
|
|
quiet = d.get("stalled_after_quiet_seconds")
|
|
if quiet is not None:
|
|
row += f" · no progress {quiet:.0f}s"
|
|
elif d.get("seconds_since_progress", 0) >= 60:
|
|
row += f" · quiet {d['seconds_since_progress']:.0f}s"
|
|
if goal:
|
|
row += f" · {goal}"
|
|
lines = [row]
|
|
for i, child in enumerate(d.get("children_activity") or []):
|
|
if not isinstance(child, dict):
|
|
continue
|
|
tool = child.get("current_tool")
|
|
doing = f"`{tool}`" if tool else "between turns"
|
|
part = f" - child {i + 1}: {child.get('api_calls', '?')} api calls · {doing}"
|
|
idle = child.get("seconds_since_activity")
|
|
if idle is not None:
|
|
part += f" · active {idle:.0f}s ago"
|
|
lines.append(part)
|
|
return lines
|
|
|
|
|
|
def _usage_agent_stats_lines(agent) -> list[str]:
|
|
"""/usage session block for a live agent: rate limits, token breakdown (matches the CLI),
|
|
context window and compression count."""
|
|
lines: list[str] = []
|
|
rl_state = agent.get_rate_limit_state()
|
|
if rl_state and rl_state.has_data:
|
|
from agent.rate_limit_tracker import format_rate_limit_compact
|
|
lines.append(t("gateway.usage.rate_limits", state=format_rate_limit_compact(rl_state)))
|
|
lines.append("")
|
|
input_tokens = getattr(agent, "session_input_tokens", 0) or 0
|
|
output_tokens = getattr(agent, "session_output_tokens", 0) or 0
|
|
lines.append(t("gateway.usage.header_session"))
|
|
lines.append(t("gateway.usage.label_model", model=agent.model))
|
|
lines.append(t("gateway.usage.label_input_tokens", count=f"{input_tokens:,}"))
|
|
lines.append(t("gateway.usage.label_output_tokens", count=f"{output_tokens:,}"))
|
|
lines.append(t("gateway.usage.label_total", count=f"{agent.session_total_tokens:,}"))
|
|
lines.append(t("gateway.usage.label_api_calls", count=agent.session_api_calls))
|
|
ctx = agent.context_compressor
|
|
_lpt = ctx.last_prompt_tokens if ctx.last_prompt_tokens > 0 else 0
|
|
if _lpt:
|
|
pct = min(100, _lpt / ctx.context_length * 100) if ctx.context_length else 0
|
|
lines.append(t("gateway.usage.label_context", used=f"{_lpt:,}", total=f"{ctx.context_length:,}", pct=f"{pct:.0f}"))
|
|
if ctx.compression_count:
|
|
lines.append(t("gateway.usage.label_compressions", count=ctx.compression_count))
|
|
return lines
|
|
|
|
|
|
class GatewayStatusCommandsMixin:
|
|
"""Read-only gateway introspection commands: /status, /context, /usage, /agents, /insights, /topup."""
|
|
|
|
async def _handle_status_command(self, event: MessageEvent) -> str:
|
|
"""Handle /status command."""
|
|
from gateway.run import _AGENT_PENDING_SENTINEL
|
|
|
|
source = event.source
|
|
session_entry = await self.async_session_store.get_or_create_session(source)
|
|
|
|
connected_platforms = [p.value for p in self.adapters]
|
|
|
|
# Check if there's an active agent. Keep the sentinel distinct: a
|
|
# starting/pending run should not be treated as a fully usable agent for
|
|
# model/context display, but it still occupies the session slot.
|
|
session_key = session_entry.session_key
|
|
agent = self._running_agents.get(session_key)
|
|
is_running = agent is not None and agent is not _AGENT_PENDING_SENTINEL
|
|
|
|
# Count pending /queue follow-ups (slot + overflow).
|
|
adapter = self.adapters.get(source.platform) if source else None
|
|
queue_depth = self._queue_depth(session_key, adapter=adapter)
|
|
|
|
title, session_row, db_total_tokens, persisted_route = await self._status_session_db_facts(
|
|
session_entry.session_id
|
|
)
|
|
# Resolve model/context for cockpit-style status. Prefer the live or cached agent because it
|
|
# carries the actual runtime route and context compressor; fall back to SessionDB metadata +
|
|
# last_prompt_tokens so /status stays useful between turns without billing/account calls.
|
|
status_agent = agent if is_running else self._cached_agent_for(session_key)
|
|
model_name, provider_name, context_used, context_total = _status_model_route(
|
|
status_agent, persisted_route, session_row, session_entry
|
|
)
|
|
|
|
model_line = ""
|
|
if model_name:
|
|
if provider_name:
|
|
model_line = t("gateway.status.model_provider", model=model_name, provider=provider_name)
|
|
else:
|
|
model_line = t("gateway.status.model", model=model_name)
|
|
|
|
context_line = ""
|
|
if context_total:
|
|
pct = min(100, round((context_used / context_total) * 100)) if context_total else 0
|
|
context_line = t(
|
|
"gateway.status.context",
|
|
used=f"{context_used:,}",
|
|
total=f"{context_total:,}",
|
|
pct=f"{pct}",
|
|
)
|
|
elif context_used:
|
|
context_line = t("gateway.status.context_used", used=f"{context_used:,}")
|
|
|
|
lines = [
|
|
t("gateway.status.header"),
|
|
"",
|
|
t("gateway.status.session_id", session_id=session_entry.session_id),
|
|
]
|
|
if title:
|
|
lines.append(t("gateway.status.title", title=title))
|
|
lines.extend([
|
|
t("gateway.status.created", timestamp=session_entry.created_at.strftime('%Y-%m-%d %H:%M')),
|
|
t("gateway.status.last_activity", timestamp=session_entry.updated_at.strftime('%Y-%m-%d %H:%M')),
|
|
])
|
|
if model_line:
|
|
lines.append(model_line)
|
|
if context_line:
|
|
lines.append(context_line)
|
|
lines.extend([
|
|
t("gateway.status.tokens", tokens=f"{db_total_tokens:,}"),
|
|
t("gateway.status.agent_running", state=t("gateway.status.state_yes") if is_running else t("gateway.status.state_no")),
|
|
])
|
|
if queue_depth:
|
|
lines.append(t("gateway.status.queued", count=queue_depth))
|
|
if source.platform == Platform.MATRIX:
|
|
scope = getattr(self.adapters.get(Platform.MATRIX), "_matrix_session_scope", os.getenv("MATRIX_SESSION_SCOPE", "auto"))
|
|
thread = source.thread_id or "none"
|
|
lines.extend([
|
|
"",
|
|
t("gateway.status.matrix_scope_header"),
|
|
t("gateway.status.matrix_scope_room", room=source.chat_name or source.chat_id),
|
|
t("gateway.status.matrix_scope_room_id", room_id=source.chat_id),
|
|
t("gateway.status.matrix_scope_thread", thread_id=thread),
|
|
t("gateway.status.matrix_scope_mode", scope=scope),
|
|
t(
|
|
"gateway.status.matrix_scope_key",
|
|
session_key=self._redact_matrix_session_key(session_key),
|
|
),
|
|
])
|
|
lines.extend([
|
|
"",
|
|
t("gateway.status.platforms", platforms=', '.join(connected_platforms)),
|
|
])
|
|
|
|
return "\n".join(lines)
|
|
|
|
async def _status_session_db_facts(self, session_id: str):
|
|
"""``(title, session_row, db_total_tokens, persisted_route)`` for /status; each fail-open.
|
|
|
|
Token totals come from the SQLite session DB rather than the in-memory SessionStore: the
|
|
agent's per-turn token deltas are persisted into sessions_db (run_agent.py), not into
|
|
SessionEntry, so session_entry.total_tokens is always 0.
|
|
"""
|
|
title = None
|
|
session_row: dict[str, Any] = {}
|
|
db_total_tokens = 0
|
|
persisted_route: dict[str, Any] = {}
|
|
if not self._session_db:
|
|
return title, session_row, db_total_tokens, persisted_route
|
|
try:
|
|
title = await self._session_db.get_session_title(session_id)
|
|
except Exception:
|
|
title = None
|
|
try:
|
|
row = await self._session_db.get_session(session_id)
|
|
if isinstance(row, dict):
|
|
session_row = row
|
|
db_total_tokens = sum(
|
|
_int_value(row.get(k))
|
|
for k in ("input_tokens", "output_tokens", "cache_read_tokens", "cache_write_tokens", "reasoning_tokens")
|
|
)
|
|
except Exception:
|
|
db_total_tokens = 0
|
|
try:
|
|
route = await self._session_db.get_dominant_session_model_route(session_id)
|
|
if isinstance(route, dict):
|
|
persisted_route = route
|
|
except Exception:
|
|
persisted_route = {}
|
|
return title, session_row, db_total_tokens, persisted_route
|
|
|
|
@staticmethod
|
|
def _redact_matrix_session_key(session_key: str) -> str:
|
|
"""Return a stable Matrix session-key fingerprint for shared room status."""
|
|
text = str(session_key or "")
|
|
digest = hashlib.sha256(text.encode("utf-8")).hexdigest()[:12]
|
|
return f"sha256:{digest}"
|
|
|
|
async def _handle_context_command(self, event: MessageEvent) -> str:
|
|
"""Handle /context — the dedicated context-window view.
|
|
|
|
/status shows a one-line ``used / total`` summary; this command is the deep view: a usage
|
|
gauge, auto-compression threshold and headroom, compression count and last savings, and
|
|
cumulative throughput — the last clearly labelled as throughput, NOT context size.
|
|
Resolution order: running agent, cached agent, SessionStore/SessionDB metadata, and a
|
|
transcript estimate only as last resort. ``/context all`` adds per-skill/toolset listings.
|
|
"""
|
|
source = event.source
|
|
session_key = self._session_key_for_source(source)
|
|
session_entry = await self.async_session_store.get_or_create_session(source)
|
|
expanded = event.get_command_args().strip().lower() in {"all", "full", "details"}
|
|
|
|
# Running agent first (mid-turn), then cached agent (between turns).
|
|
agent = self._resident_agent_for(session_key)
|
|
has_agent = bool(agent)
|
|
|
|
ctx = getattr(agent, "context_compressor", None) if has_agent else None
|
|
used, context_length, model_name = await self._resolve_context_figures(
|
|
agent if has_agent else None, ctx, session_entry, source
|
|
)
|
|
|
|
# Gauge path: real current-context figure
|
|
if used > 0 and context_length > 0:
|
|
pct = min(100.0, used / context_length * 100)
|
|
headroom = max(0, context_length - used)
|
|
BAR_WIDTH = 24
|
|
filled = int(round(pct / 100 * BAR_WIDTH))
|
|
bar = "█" * max(0, filled) + "░" * max(0, BAR_WIDTH - filled)
|
|
|
|
lines = [
|
|
t("gateway.context.header"),
|
|
"",
|
|
t("gateway.context.model", model=model_name or "?"),
|
|
t("gateway.context.window", total=f"{context_length:,}"),
|
|
t(
|
|
"gateway.context.in_use",
|
|
used=f"{used:,}",
|
|
total=f"{context_length:,}",
|
|
pct=f"{pct:.0f}",
|
|
),
|
|
t("gateway.context.bar", bar=bar),
|
|
t("gateway.context.headroom", headroom=f"{headroom:,}"),
|
|
"",
|
|
]
|
|
# Full view — compression / throughput need the live agent.
|
|
if ctx is not None:
|
|
lines.extend(_context_compressor_lines(agent, ctx, used))
|
|
else:
|
|
lines.append(t("gateway.context.detail_after_first"))
|
|
|
|
# Per-category estimated breakdown (+ optional expanded listings). Same chars/4 engine
|
|
# the desktop popover and /usage use; plain text (no glyph grid — monospace isn't
|
|
# guaranteed on messaging platforms). Fail-open: rendering errors never break /context.
|
|
if has_agent:
|
|
breakdown = await asyncio.to_thread(
|
|
self._context_breakdown_block, agent, source, expanded
|
|
)
|
|
if breakdown:
|
|
lines.append("")
|
|
lines.extend(breakdown)
|
|
|
|
return "\n".join(lines)
|
|
|
|
# Last resort: rough estimate from transcript
|
|
history = await self.async_session_store.load_transcript(session_entry.session_id)
|
|
if history:
|
|
from agent.model_metadata import estimate_messages_tokens_rough
|
|
|
|
msgs = [
|
|
m
|
|
for m in history
|
|
if m.get("role") in {"user", "assistant"} and m.get("content")
|
|
]
|
|
approx = estimate_messages_tokens_rough(msgs)
|
|
return "\n".join(
|
|
[
|
|
t("gateway.context.header"),
|
|
"",
|
|
t(
|
|
"gateway.context.estimated",
|
|
count=f"{approx:,}",
|
|
messages=len(msgs),
|
|
),
|
|
t("gateway.context.detail_after_first"),
|
|
]
|
|
)
|
|
return t("gateway.context.no_data")
|
|
|
|
async def _resolve_context_figures(self, agent, ctx, session_entry, source):
|
|
"""``(used, context_length, model_name)`` for /context with cascading fallbacks.
|
|
|
|
used : compressor.last_prompt_tokens -> SessionStore.last_prompt_tokens
|
|
model : agent.model -> SessionDB row model
|
|
window: compressor.context_length -> effective gateway model route -> model metadata
|
|
"""
|
|
used = context_length = 0
|
|
if ctx is not None:
|
|
used = getattr(ctx, "last_prompt_tokens", 0) or 0
|
|
context_length = getattr(ctx, "context_length", 0) or 0
|
|
model_name = _clean_str(getattr(agent, "model", "")) if agent is not None else ""
|
|
if not used:
|
|
used = _int_value(getattr(session_entry, "last_prompt_tokens", 0))
|
|
if not model_name and self._session_db:
|
|
try:
|
|
row = await self._session_db.get_session(session_entry.session_id) or {}
|
|
if isinstance(row, dict):
|
|
model_name = _clean_str(row.get("model", ""))
|
|
except Exception:
|
|
model_name = ""
|
|
if not context_length:
|
|
try:
|
|
from gateway.run import _profile_runtime_scope, _resolve_gateway_model_context
|
|
|
|
def _resolve_nonresident_context():
|
|
if getattr(getattr(self, "config", None), "multiplex_profiles", False):
|
|
profile_home = self._resolve_profile_home_for_source(source)
|
|
with _profile_runtime_scope(profile_home):
|
|
return _resolve_gateway_model_context(model_name or None)
|
|
return _resolve_gateway_model_context(model_name or None)
|
|
|
|
resolved = await asyncio.to_thread(_resolve_nonresident_context)
|
|
model_name = model_name or resolved.model
|
|
context_length = _int_value(resolved.context_length)
|
|
except Exception:
|
|
context_length = 0
|
|
if not context_length and model_name:
|
|
try:
|
|
from agent.model_metadata import get_model_context_length
|
|
|
|
context_length = _int_value(await asyncio.to_thread(get_model_context_length, model_name))
|
|
except Exception:
|
|
context_length = 0
|
|
return used, context_length, model_name
|
|
|
|
async def _handle_agents_command(self, event: MessageEvent) -> str:
|
|
"""Handle /agents command - list active agents and running tasks."""
|
|
from gateway.run import _AGENT_PENDING_SENTINEL
|
|
from tools.process_registry import format_uptime_short, process_registry
|
|
|
|
now = time.time()
|
|
current_session_key = self._session_key_for_source(event.source)
|
|
|
|
running_agents: dict = getattr(self, "_running_agents", {}) or {}
|
|
running_started: dict = getattr(self, "_running_agents_ts", {}) or {}
|
|
|
|
agent_rows: list[dict] = []
|
|
for session_key, agent in running_agents.items():
|
|
started = float(running_started.get(session_key, now))
|
|
elapsed = max(0, int(now - started))
|
|
is_pending = agent is _AGENT_PENDING_SENTINEL
|
|
agent_rows.append(
|
|
{
|
|
"session_key": session_key,
|
|
"elapsed": elapsed,
|
|
"state": t("gateway.agents.state_starting") if is_pending else t("gateway.agents.state_running"),
|
|
"session_id": "" if is_pending else str(getattr(agent, "session_id", "") or ""),
|
|
"model": "" if is_pending else str(getattr(agent, "model", "") or ""),
|
|
}
|
|
)
|
|
|
|
agent_rows.sort(key=lambda row: row["elapsed"], reverse=True)
|
|
|
|
running_processes: list[dict] = []
|
|
try:
|
|
running_processes = [
|
|
p for p in process_registry.list_sessions()
|
|
if p.get("status") == "running"
|
|
]
|
|
except Exception:
|
|
running_processes = []
|
|
|
|
background_tasks = [
|
|
t for t in (getattr(self, "_background_tasks", set()) or set())
|
|
if hasattr(t, "done") and not t.done()
|
|
]
|
|
|
|
lines = [
|
|
t("gateway.agents.header"),
|
|
"",
|
|
t("gateway.agents.active_agents", count=len(agent_rows)),
|
|
]
|
|
|
|
if agent_rows:
|
|
for idx, row in enumerate(agent_rows[:12], 1):
|
|
current = t("gateway.agents.this_chat") if row["session_key"] == current_session_key else ""
|
|
sid = f" · `{row['session_id']}`" if row["session_id"] else ""
|
|
model = f" · `{row['model']}`" if row["model"] else ""
|
|
lines.append(
|
|
f"{idx}. `{row['session_key']}` · {row['state']} · "
|
|
f"{format_uptime_short(row['elapsed'])}{sid}{model}{current}"
|
|
)
|
|
if len(agent_rows) > 12:
|
|
lines.append(t("gateway.agents.more", count=len(agent_rows) - 12))
|
|
|
|
lines.extend(
|
|
[
|
|
"",
|
|
t("gateway.agents.running_processes", count=len(running_processes)),
|
|
]
|
|
)
|
|
if running_processes:
|
|
for proc in running_processes[:12]:
|
|
cmd = " ".join(str(proc.get("command", "")).split())
|
|
if len(cmd) > 90:
|
|
cmd = cmd[:87] + "..."
|
|
lines.append(
|
|
f"- `{proc.get('session_id', '?')}` · "
|
|
f"{format_uptime_short(int(proc.get('uptime_seconds', 0)))} · `{cmd}`"
|
|
)
|
|
if len(running_processes) > 12:
|
|
lines.append(t("gateway.agents.more", count=len(running_processes) - 12))
|
|
|
|
lines.extend(
|
|
[
|
|
"",
|
|
t("gateway.agents.async_jobs", count=len(background_tasks)),
|
|
]
|
|
)
|
|
|
|
# Background (async) delegations — delegate_task(background=true).
|
|
try:
|
|
from tools.async_delegation import list_async_delegations
|
|
delegations = [
|
|
d for d in list_async_delegations()
|
|
if d.get("status") in ("running", "stalling", "finalizing")
|
|
]
|
|
except Exception:
|
|
delegations = []
|
|
if delegations:
|
|
lines.extend(["", t("gateway.agents.background_delegations", count=len(delegations))])
|
|
for d in delegations[:12]:
|
|
lines.extend(_agents_delegation_lines(d))
|
|
if len(delegations) > 12:
|
|
lines.append(t("gateway.agents.more", count=len(delegations) - 12))
|
|
|
|
if (
|
|
not agent_rows
|
|
and not running_processes
|
|
and not background_tasks
|
|
and not delegations
|
|
):
|
|
lines.append("")
|
|
lines.append(t("gateway.agents.none"))
|
|
|
|
return "\n".join(lines)
|
|
|
|
async def _handle_topup_command(self, event: MessageEvent) -> str:
|
|
"""Handle /topup -- show the Nous balance and hand off to the portal.
|
|
|
|
Does NOT charge, confirm, or track payment — that happens in the browser; the next /topup
|
|
shows the new balance. Fetched off the event loop; fail-open.
|
|
"""
|
|
from agent.account_usage import build_credits_view
|
|
|
|
try:
|
|
view = await asyncio.to_thread(build_credits_view, markdown=True)
|
|
except Exception:
|
|
view = None
|
|
|
|
if view is None or not view.logged_in:
|
|
return t("gateway.credits.not_logged_in")
|
|
|
|
lines: list[str] = ["💳 **Nous balance**"]
|
|
for line in view.balance_lines:
|
|
if line.lstrip().startswith("📈"):
|
|
continue # drop the helper's header; we print our own
|
|
lines.append(line)
|
|
if view.identity_line:
|
|
lines.append("")
|
|
lines.append(view.identity_line)
|
|
if view.topup_url:
|
|
lines.append("")
|
|
lines.append(f"Manage billing on the portal: {view.topup_url}")
|
|
lines.append("Top up and manage billing in the browser — your balance updates here after.")
|
|
return "\n".join(lines)
|
|
|
|
def _context_breakdown_block(self, agent, source, expanded: bool) -> list[str]:
|
|
"""Render the /context per-category block (plain text, no grid).
|
|
|
|
Estimated (chars/4), same engine as /usage. Runs in a thread; returns [] and never raises.
|
|
"""
|
|
try:
|
|
from agent.context_breakdown import compute_context_details, render_context_breakdown_lines
|
|
|
|
payload = self._session_context_breakdown(agent, source)
|
|
if not (payload.get("categories") or []):
|
|
return []
|
|
|
|
details = None
|
|
if expanded:
|
|
try:
|
|
details = compute_context_details(agent)
|
|
except Exception:
|
|
details = {"skills": [], "toolsets": []}
|
|
|
|
return render_context_breakdown_lines(payload, details=details, grid=False)
|
|
except Exception:
|
|
return []
|
|
|
|
def _session_context_breakdown(self, agent, source) -> dict:
|
|
"""Per-category context estimate (chars/4) for *agent* over the session transcript (sync)."""
|
|
from agent.context_breakdown import compute_session_context_breakdown
|
|
|
|
history: list[dict] = []
|
|
try:
|
|
entry = self.session_store.get_or_create_session(source)
|
|
history = self.session_store.load_transcript(entry.session_id) or []
|
|
except Exception:
|
|
history = []
|
|
return compute_session_context_breakdown(agent, history)
|
|
|
|
def _context_breakdown_lines(self, agent, source) -> list[str]:
|
|
"""Render the per-category context breakdown for /usage.
|
|
|
|
Estimated (chars/4). Returns [] and never raises so /usage stays robust.
|
|
"""
|
|
try:
|
|
payload = self._session_context_breakdown(agent, source)
|
|
categories = payload.get("categories") or []
|
|
if not categories:
|
|
return []
|
|
|
|
total = payload.get("estimated_total") or 0
|
|
out = [t("gateway.usage.breakdown_header")]
|
|
for cat in categories:
|
|
tokens = int(cat.get("tokens") or 0)
|
|
if tokens <= 0:
|
|
continue
|
|
cat_id = str(cat.get("id") or "")
|
|
label = t(f"gateway.usage.breakdown_cat_{cat_id}")
|
|
# Missing key → t() echoes the key back; fall back to the
|
|
# English label the engine already provides.
|
|
if label.endswith(f"breakdown_cat_{cat_id}"):
|
|
label = str(cat.get("label") or cat_id)
|
|
pct = round(tokens / total * 100) if total else 0
|
|
out.append(
|
|
t("gateway.usage.breakdown_line", label=label, count=f"{tokens:,}", pct=pct)
|
|
)
|
|
return out if len(out) > 1 else []
|
|
except Exception:
|
|
return []
|
|
|
|
async def _handle_usage_command(self, event: MessageEvent) -> str:
|
|
"""Handle /usage command -- show token usage for the current session.
|
|
|
|
Checks both _running_agents (mid-turn) and _agent_cache (between turns) so details are
|
|
available whenever the user asks.
|
|
"""
|
|
source = event.source
|
|
session_key = self._session_key_for_source(source)
|
|
|
|
# `/usage reset [--force]` — redeem one banked Codex rate-limit reset
|
|
# credit. Parsed before the display path so it never mixes with the
|
|
# stats rendering below.
|
|
raw_args = event.get_command_args().strip()
|
|
args = [a.lower() for a in raw_args.split()] if raw_args else []
|
|
wants_reset = bool(args) and args[0] == "reset"
|
|
if args and not wants_reset:
|
|
return t("gateway.usage.unknown_subcommand", args=raw_args)
|
|
|
|
# Running agent first (mid-turn), then cached agent (between turns).
|
|
agent = self._resident_agent_for(session_key)
|
|
|
|
# Resolve provider/base_url/api_key for the account-usage fetch. Prefer the live agent; fall
|
|
# back to persisted billing data on the SessionDB row so `/usage` still returns account info
|
|
# between turns when no agent is resident.
|
|
provider = getattr(agent, "provider", None) if agent else None
|
|
base_url = getattr(agent, "base_url", None) if agent else None
|
|
api_key = getattr(agent, "api_key", None) if agent else None
|
|
if not provider and getattr(self, "_session_db", None) is not None:
|
|
provider, base_url = await self._persisted_billing_route(source)
|
|
|
|
if wants_reset:
|
|
normalized_provider = str(provider or "").strip().lower()
|
|
if normalized_provider != "openai-codex":
|
|
return t("gateway.usage.reset_wrong_provider")
|
|
force = "--force" in args[1:]
|
|
from agent.account_usage import redeem_codex_reset_credit
|
|
|
|
result = await asyncio.to_thread(
|
|
redeem_codex_reset_credit,
|
|
base_url=base_url,
|
|
api_key=api_key,
|
|
force=force,
|
|
)
|
|
return result.message
|
|
|
|
# Fetch account usage off the event loop so slow provider APIs don't
|
|
# block the gateway. Failures are non-fatal -- account_lines stays [].
|
|
account_lines: list[str] = []
|
|
credits_lines: list[str] = []
|
|
if provider:
|
|
try:
|
|
account_snapshot = await asyncio.to_thread(
|
|
fetch_account_usage,
|
|
provider,
|
|
base_url=base_url,
|
|
api_key=api_key,
|
|
)
|
|
except Exception:
|
|
account_snapshot = None
|
|
if account_snapshot:
|
|
account_lines = render_account_usage_lines(account_snapshot, markdown=True)
|
|
|
|
# ── Nous credits magnitudes + monthly-grant % gauge ─────────────
|
|
# Shared with CLI/TUI via nous_credits_lines(); run off the event loop. Gates on "a Nous
|
|
# account is logged in" — NOT the inference provider, NOT under `if provider:` — so a Nous
|
|
# user inferring elsewhere still sees a balance. No recovery trigger; fail-open.
|
|
try:
|
|
from agent.account_usage import nous_credits_lines
|
|
|
|
credits_lines = await asyncio.to_thread(nous_credits_lines, markdown=True)
|
|
except Exception:
|
|
credits_lines = [] # fail-open: never break /usage
|
|
|
|
def _with_account_blocks(lines: list[str]) -> str:
|
|
# Each block is preceded by a blank divider only when something precedes it.
|
|
for block in (account_lines, credits_lines):
|
|
if block:
|
|
if lines:
|
|
lines.append("")
|
|
lines.extend(block)
|
|
return "\n".join(lines)
|
|
|
|
if agent and hasattr(agent, "session_total_tokens") and agent.session_api_calls > 0:
|
|
lines = _usage_agent_stats_lines(agent)
|
|
# Per-category context breakdown (estimated — chars/4 heuristic). Same engine the
|
|
# desktop popover uses. The system prompt / tools / skills / memory slices read off the
|
|
# live agent; the conversation slice is estimated from the session transcript.
|
|
breakdown_lines = await asyncio.to_thread(self._context_breakdown_lines, agent, source)
|
|
if breakdown_lines:
|
|
lines.append("")
|
|
lines.extend(breakdown_lines)
|
|
return _with_account_blocks(lines)
|
|
|
|
# No agent at all -- check session history for a rough count
|
|
session_entry = await self.async_session_store.get_or_create_session(source)
|
|
history = await self.async_session_store.load_transcript(session_entry.session_id)
|
|
if history:
|
|
from agent.model_metadata import estimate_messages_tokens_rough
|
|
msgs = [m for m in history if m.get("role") in {"user", "assistant"} and m.get("content")]
|
|
approx = estimate_messages_tokens_rough(msgs)
|
|
return _with_account_blocks([
|
|
t("gateway.usage.header_session_info"),
|
|
t("gateway.usage.label_messages", count=len(msgs)),
|
|
t("gateway.usage.label_estimated_context", count=f"{approx:,}"),
|
|
t("gateway.usage.detailed_after_first"),
|
|
])
|
|
if account_lines or credits_lines:
|
|
return _with_account_blocks([])
|
|
return t("gateway.usage.no_data")
|
|
|
|
async def _persisted_billing_route(self, source):
|
|
"""``(provider, base_url)`` from the SessionDB row / dominant route when no agent is resident."""
|
|
try:
|
|
entry = await self.async_session_store.get_or_create_session(source)
|
|
persisted = await self._session_db.get_session(entry.session_id) or {}
|
|
route = await self._session_db.get_dominant_session_model_route(entry.session_id)
|
|
persisted_route = route if isinstance(route, dict) else {}
|
|
except Exception:
|
|
persisted = {}
|
|
persisted_route = {}
|
|
if persisted_route.get("billing_provider"):
|
|
return persisted_route["billing_provider"], persisted_route.get("billing_base_url")
|
|
return persisted.get("billing_provider"), persisted.get("billing_base_url")
|
|
|
|
async def _handle_insights_command(self, event: MessageEvent) -> str:
|
|
"""Handle /insights command -- show usage insights and analytics."""
|
|
args = event.get_command_args().strip()
|
|
|
|
# Normalize Unicode dashes (Telegram/iOS auto-converts -- to em/en dash)
|
|
args = re.sub(r'[\u2012\u2013\u2014\u2015](days|source)', r'--\1', args)
|
|
|
|
days = 30
|
|
source = None
|
|
|
|
# Parse simple args: /insights 7 or /insights --days 7
|
|
if args:
|
|
parts = args.split()
|
|
i = 0
|
|
while i < len(parts):
|
|
if parts[i] == "--days" and i + 1 < len(parts):
|
|
try:
|
|
days = int(parts[i + 1])
|
|
except ValueError:
|
|
return t("gateway.insights.invalid_days", value=parts[i + 1])
|
|
i += 2
|
|
elif parts[i] == "--source" and i + 1 < len(parts):
|
|
source = parts[i + 1]
|
|
i += 2
|
|
elif parts[i].isdigit():
|
|
days = int(parts[i])
|
|
i += 1
|
|
else:
|
|
i += 1
|
|
|
|
try:
|
|
from hermes_state import get_shared_session_db
|
|
from agent.insights import InsightsEngine
|
|
|
|
def _run_insights():
|
|
db = get_shared_session_db()
|
|
try:
|
|
engine = InsightsEngine(db)
|
|
report = engine.generate(days=days, source=source)
|
|
result = engine.format_gateway(report)
|
|
return result
|
|
finally:
|
|
from hermes_state import release_or_close
|
|
release_or_close(db)
|
|
|
|
# Not a bare hop: ``SessionDB()`` resolves ``get_hermes_home()`` at call time, which is
|
|
# a contextvar set by ``_profile_runtime_scope``; a default-executor hop starts with an
|
|
# EMPTY context and would read the DEFAULT profile's state.db.
|
|
return await self._run_in_executor_with_context(_run_insights)
|
|
except Exception as e:
|
|
logger.error("Insights command error: %s", e, exc_info=True)
|
|
return t("gateway.insights.error", error=e)
|