feat(desktop): status bar can show live cache-hit rate and tokens/sec (off by default)
Two new right-click-toggleable status bar items, mirroring the CLI/TUI
Pantheon status bar upgrades: prompt-cache hit rate ("87%") and rolling
output throughput ("42 t/s"). Both are hidden by default and enabled from
the bar's existing 'Show in status bar' context menu, like the context meter.
Renderer-only: the tui_gateway already emits cache_hit_pct and avg_tps in
every session.usage tick and message.complete payload, so the items ride
the same UsageStats the context meter reads — no new RPC, no polling.
Labels show a placeholder until the backend has data, never self-hide.
This commit is contained in:
@@ -14,9 +14,9 @@ import { Codicon } from '@/components/ui/codicon'
|
||||
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { displayPath, pathLeaf } from '@/lib/display-path'
|
||||
import { Activity, AlertCircle, Clock, Command, FolderOpen, Globe, Hash, Loader2, Terminal } from '@/lib/icons'
|
||||
import { Activity, AlertCircle, Clock, Command, FolderOpen, Globe, Hash, Layers3, Loader2, Terminal, Zap } from '@/lib/icons'
|
||||
import { runtimeReadinessDisplay, type RuntimeReadinessResult } from '@/lib/runtime-readiness'
|
||||
import { contextBarLabel, LiveDuration, usageContextLabel } from '@/lib/statusbar'
|
||||
import { cacheHitLabel, contextBarLabel, LiveDuration, tokensPerSecondLabel, usageContextLabel } from '@/lib/statusbar'
|
||||
import { useStoreSelector } from '@/lib/use-session-slice'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { resolveVersionStatus } from '@/lib/version-status'
|
||||
@@ -267,6 +267,10 @@ export function useStatusbarItems({
|
||||
|
||||
const contextUsage = useMemo(() => usageContextLabel(gaugeUsage), [gaugeUsage])
|
||||
const contextBar = useMemo(() => contextBarLabel(gaugeUsage), [gaugeUsage])
|
||||
// Both ride the same usage payload the context meter does (session.usage
|
||||
// ticks mid-turn, message.complete after) — no extra RPC, no polling.
|
||||
const cacheHit = cacheHitLabel(currentUsage)
|
||||
const tokensPerSecond = tokensPerSecondLabel(currentUsage)
|
||||
|
||||
const approvalModeItem = useApprovalModeStatusbarItem(activeGatewayProfile, requestGateway)
|
||||
const systemResourcesItem = useSystemResourcesStatusbarItem()
|
||||
@@ -562,6 +566,24 @@ export function useStatusbarItems({
|
||||
toggleLabel: copy.toggleContextUsage,
|
||||
variant: 'menu'
|
||||
},
|
||||
{
|
||||
icon: <Layers3 className="size-3" />,
|
||||
id: 'cache-hit-rate',
|
||||
// Same never-self-hide rule as the context meter: opted in means a
|
||||
// placeholder until the first cached turn reports, not a vanished item.
|
||||
label: cacheHit || '—',
|
||||
title: copy.cacheHitRateTitle,
|
||||
toggleLabel: copy.toggleCacheHitRate,
|
||||
variant: 'text'
|
||||
},
|
||||
{
|
||||
icon: <Zap className="size-3" />,
|
||||
id: 'tokens-per-second',
|
||||
label: tokensPerSecond || '—',
|
||||
title: copy.tokensPerSecondTitle,
|
||||
toggleLabel: copy.toggleTokensPerSecond,
|
||||
variant: 'text'
|
||||
},
|
||||
{
|
||||
detail: <LiveDuration since={sessionStartedAt} />,
|
||||
hidden: !sessionStartedAt,
|
||||
@@ -594,6 +616,7 @@ export function useStatusbarItems({
|
||||
approvalModeItem,
|
||||
backendVersionItem,
|
||||
busy,
|
||||
cacheHit,
|
||||
chatOpen,
|
||||
clientVersionItem,
|
||||
contextBar,
|
||||
@@ -606,6 +629,7 @@ export function useStatusbarItems({
|
||||
gatewayState,
|
||||
systemResourcesItem,
|
||||
terminalShowing,
|
||||
tokensPerSecond,
|
||||
turnStartedAt
|
||||
]
|
||||
)
|
||||
|
||||
@@ -99,21 +99,27 @@ describe('statusbar item visibility', () => {
|
||||
const statusbar = bar([
|
||||
item('running-timer', 'Turn timer', { variant: 'text' }),
|
||||
item('context-usage', 'Context meter', { variant: 'menu' }),
|
||||
item('cache-hit-rate', 'Cache hit rate', { variant: 'text' }),
|
||||
item('tokens-per-second', 'Tokens per second', { variant: 'text' }),
|
||||
item('session-timer', 'Session timer', { variant: 'text' }),
|
||||
item('gateway-health', 'Gateway')
|
||||
])
|
||||
|
||||
for (const label of ['Turn timer', 'Context meter', 'Session timer']) {
|
||||
for (const label of ['Turn timer', 'Context meter', 'Cache hit rate', 'Tokens per second', 'Session timer']) {
|
||||
expect(screen.queryByText(label)).toBeNull()
|
||||
}
|
||||
|
||||
openContextMenu(statusbar)
|
||||
|
||||
const row = await screen.findByRole('menuitemcheckbox', { name: 'Session timer' })
|
||||
fireEvent.click(row)
|
||||
for (const [id, label] of [
|
||||
['session-timer', 'Session timer'],
|
||||
['cache-hit-rate', 'Cache hit rate']
|
||||
]) {
|
||||
fireEvent.click(await screen.findByRole('menuitemcheckbox', { name: label }))
|
||||
|
||||
expect($statusbarHiddenIds.get()).not.toContain('session-timer')
|
||||
expect(within(statusbar).getByText('Session timer')).toBeTruthy()
|
||||
expect($statusbarHiddenIds.get()).not.toContain(id)
|
||||
expect(within(statusbar).getByText(label)).toBeTruthy()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -3068,13 +3068,17 @@ export const en: Translations = {
|
||||
resetStatusbar: 'Reset to defaults',
|
||||
toggleApprovalMode: 'Approvals',
|
||||
toggleBackendVersion: 'Backend version',
|
||||
toggleCacheHitRate: 'Cache hit rate',
|
||||
toggleCommandCenter: 'Command Center',
|
||||
toggleContextUsage: 'Context meter',
|
||||
toggleRunningTimer: 'Turn timer',
|
||||
toggleSessionTimer: 'Session timer',
|
||||
toggleTerminal: 'Terminal',
|
||||
toggleTokensPerSecond: 'Tokens per second',
|
||||
toggleVersion: 'Version & updates',
|
||||
toggleWorkspace: 'Workspace',
|
||||
cacheHitRateTitle: 'Prompt cache hit rate this session — cached tokens cost less, so higher is cheaper',
|
||||
tokensPerSecondTitle: 'Output tokens per second, averaged over the last 10 model calls',
|
||||
agents: 'Agents',
|
||||
closeAgents: 'Close agents',
|
||||
openAgents: 'Open agents',
|
||||
|
||||
@@ -3095,13 +3095,17 @@ export const ru = defineLocale({
|
||||
resetStatusbar: 'Сбросить к значениям по умолчанию',
|
||||
toggleApprovalMode: 'Подтверждения',
|
||||
toggleBackendVersion: 'Версия бэкенда',
|
||||
toggleCacheHitRate: 'Попадания в кэш',
|
||||
toggleCommandCenter: 'Командный центр',
|
||||
toggleContextUsage: 'Шкала контекста',
|
||||
toggleRunningTimer: 'Таймер хода',
|
||||
toggleSessionTimer: 'Таймер сеанса',
|
||||
toggleTerminal: 'Терминал',
|
||||
toggleTokensPerSecond: 'Токенов в секунду',
|
||||
toggleVersion: 'Версия и обновления',
|
||||
toggleWorkspace: 'Рабочее пространство',
|
||||
cacheHitRateTitle: 'Доля попаданий в кэш промпта за сеанс — кэшированные токены дешевле, чем выше, тем дешевле',
|
||||
tokensPerSecondTitle: 'Выходных токенов в секунду, среднее за последние 10 вызовов модели',
|
||||
agents: 'Агенты',
|
||||
closeAgents: 'Закрыть агентов',
|
||||
openAgents: 'Открыть агентов',
|
||||
|
||||
@@ -2615,13 +2615,17 @@ export interface Translations {
|
||||
resetStatusbar: string
|
||||
toggleApprovalMode: string
|
||||
toggleBackendVersion: string
|
||||
toggleCacheHitRate: string
|
||||
toggleCommandCenter: string
|
||||
toggleContextUsage: string
|
||||
toggleRunningTimer: string
|
||||
toggleSessionTimer: string
|
||||
toggleTerminal: string
|
||||
toggleTokensPerSecond: string
|
||||
toggleVersion: string
|
||||
toggleWorkspace: string
|
||||
cacheHitRateTitle: string
|
||||
tokensPerSecondTitle: string
|
||||
agents: string
|
||||
closeAgents: string
|
||||
openAgents: string
|
||||
|
||||
@@ -3218,13 +3218,17 @@ export const zh: Translations = {
|
||||
resetStatusbar: '恢复默认设置',
|
||||
toggleApprovalMode: '审批',
|
||||
toggleBackendVersion: '后端版本',
|
||||
toggleCacheHitRate: '缓存命中率',
|
||||
toggleCommandCenter: '命令中心',
|
||||
toggleContextUsage: '上下文用量',
|
||||
toggleRunningTimer: '回合计时',
|
||||
toggleSessionTimer: '会话计时',
|
||||
toggleTerminal: '终端',
|
||||
toggleTokensPerSecond: '每秒 token 数',
|
||||
toggleVersion: '版本与更新',
|
||||
toggleWorkspace: '工作区',
|
||||
cacheHitRateTitle: '本会话的提示缓存命中率 — 缓存 token 更便宜,越高越省',
|
||||
tokensPerSecondTitle: '每秒输出 token 数,取最近 10 次模型调用的平均值',
|
||||
agents: '代理',
|
||||
closeAgents: '关闭代理',
|
||||
openAgents: '打开代理',
|
||||
|
||||
17
apps/desktop/src/lib/statusbar.test.ts
Normal file
17
apps/desktop/src/lib/statusbar.test.ts
Normal file
@@ -0,0 +1,17 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { cacheHitLabel, tokensPerSecondLabel } from '@/lib/statusbar'
|
||||
|
||||
const base = { calls: 0, input: 0, output: 0, total: 0 }
|
||||
|
||||
describe('statusbar usage readouts', () => {
|
||||
it('paints the backend cache-hit and throughput fields, and stays blank when they are absent', () => {
|
||||
// The backend omits both fields (rather than sending 0) when it has no data
|
||||
// — a provider with no cache reads, or a session before its first call.
|
||||
expect(cacheHitLabel(base)).toBe('')
|
||||
expect(tokensPerSecondLabel(base)).toBe('')
|
||||
|
||||
expect(cacheHitLabel({ ...base, cache_hit_pct: 87 })).toBe('87%')
|
||||
expect(tokensPerSecondLabel({ ...base, avg_tps: 41.6 })).toBe('42 t/s')
|
||||
})
|
||||
})
|
||||
@@ -59,6 +59,22 @@ export function contextBarLabel(usage: UsageStats): string {
|
||||
return `[${contextBar(usage.context_percent)}] ${pct}%`
|
||||
}
|
||||
|
||||
/** `87%` for a reported hit rate; '' when the backend omitted it (no cache
|
||||
* reads yet, or a provider that doesn't report them). The backend already
|
||||
* clamps and rounds, so this only guards a malformed/absent field. */
|
||||
export function cacheHitLabel(usage: UsageStats): string {
|
||||
const pct = usage.cache_hit_pct
|
||||
|
||||
return typeof pct === 'number' && Number.isFinite(pct) ? `${Math.round(pct)}%` : ''
|
||||
}
|
||||
|
||||
/** `42 t/s` for the rolling throughput; '' before the first completed call. */
|
||||
export function tokensPerSecondLabel(usage: UsageStats): string {
|
||||
const tps = usage.avg_tps
|
||||
|
||||
return typeof tps === 'number' && Number.isFinite(tps) && tps > 0 ? `${Math.round(tps)} t/s` : ''
|
||||
}
|
||||
|
||||
export function LiveDuration({ since }: { since: number | null | undefined }) {
|
||||
const [now, setNow] = useState(() => Date.now())
|
||||
|
||||
|
||||
@@ -16,17 +16,20 @@ export function toggleStatusbarVisible() {
|
||||
// bar's job is to answer "is the backend healthy, where am I, what's it doing" —
|
||||
// route shortcuts (cron/webhooks/agents), the terminal toggle, and the approval
|
||||
// pill are navigation, not status, so they start out of the way. The per-turn
|
||||
// session readouts (running/session timers, context meter) are diagnostics most
|
||||
// users don't watch, so they start hidden too and the bar stays quiet mid-turn.
|
||||
// session readouts (running/session timers, context meter, cache hit rate,
|
||||
// tokens/sec) are diagnostics most users don't watch, so they start hidden too
|
||||
// and the bar stays quiet mid-turn.
|
||||
export const STATUSBAR_HIDDEN_BY_DEFAULT: readonly string[] = [
|
||||
'agents',
|
||||
'approval-mode',
|
||||
'cache-hit-rate',
|
||||
'context-usage',
|
||||
'cron',
|
||||
'running-timer',
|
||||
'session-timer',
|
||||
'system-resources',
|
||||
'terminal',
|
||||
'tokens-per-second',
|
||||
'webhooks'
|
||||
]
|
||||
|
||||
|
||||
@@ -731,6 +731,10 @@ export interface SessionRuntimeInfo {
|
||||
}
|
||||
|
||||
export interface UsageStats {
|
||||
/** Rolling tokens-per-second over the last ~10 API calls (tui_gateway `_get_usage`). */
|
||||
avg_tps?: number
|
||||
/** Session prompt-cache hit rate, 0–100. Omitted (not 0) when the provider reports no cache reads. */
|
||||
cache_hit_pct?: number
|
||||
calls: number
|
||||
context_max?: number
|
||||
context_percent?: number
|
||||
|
||||
@@ -55,7 +55,8 @@ The bar along the bottom of the chat shows live session state and exposes quick
|
||||
|
||||
- **Per-session YOLO toggle** — flip YOLO on or off for just this session (matching the TUI). YOLO bypasses the dangerous-command approval prompts, so know what you're turning off — see [Security → YOLO Mode](./security.md#yolo-mode).
|
||||
- **Context-usage meter** — a live "% full" meter of the session's context window. Click it to open the **Context Usage** popover with a token breakdown by category (system prompt, tool definitions, skills, memory, rules, MCP, subagent definitions, and the conversation itself) so you can see exactly what's eating the window before compression kicks in.
|
||||
- **Customizable items** — right-click the status bar (**Show in status bar**) to choose what appears: the context meter, workspace, model, approvals, turn/session timers, terminal, Command Center, backend version, and more — or hide the bar entirely (**Cmd/Ctrl+Shift+S** toggles it).
|
||||
- **Cache hit rate and tokens per second** — off by default; turn them on from the right-click menu. Cache hit rate is the share of this session's prompt tokens served from the provider's prompt cache (cached tokens cost less, so higher is cheaper — you can watch a session get cheaper as it warms up). Tokens per second is output throughput averaged over the last 10 model calls. Both update live during a turn.
|
||||
- **Customizable items** — right-click the status bar (**Show in status bar**) to choose what appears: the context meter, cache hit rate, tokens per second, workspace, model, approvals, turn/session timers, terminal, Command Center, backend version, and more — or hide the bar entirely (**Cmd/Ctrl+Shift+S** toggles it).
|
||||
|
||||
Chatting against a Hermes instance on another machine instead of the bundled local backend? See [Connecting to a remote backend](#connecting-to-a-remote-backend) below — and for the full picture of how the remote-hosted dashboard connection works (the auth gate, the `/api/ws` chat socket, and WebSocket close-code triage), see [Web Dashboard → Connecting Hermes Desktop to a remote backend](./features/web-dashboard.md#connecting-hermes-desktop-to-a-remote-backend).
|
||||
|
||||
|
||||
Reference in New Issue
Block a user