Files
hermes-agent/hermes_cli/config_defaults.py

2832 lines
177 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Default configuration data for Hermes Agent.
Pure-data leaf module: DEFAULT_CONFIG and OPTIONAL_ENV_VARS, extracted verbatim from
hermes_cli/config.py. Must not import from hermes_cli.config.
"""
def _aux(timeout, *, reasoning_effort=True, **extra):
"""Standard auxiliary-task model block (see DEFAULT_CONFIG["auxiliary"]).
reasoning_effort=False omits that key (MoA blocks configure depth per slot);
``extra`` keys are appended after the standard ones.
"""
d = {"provider": "auto", "model": "", "base_url": "", "api_key": "", "timeout": timeout, "extra_body": {}}
if reasoning_effort:
d["reasoning_effort"] = ""
d.update(extra)
return d
DEFAULT_CONFIG = {
"model": "",
"providers": {},
"fallback_providers": [],
"credential_pool_strategies": {},
"toolsets": ["hermes-cli"],
# journal_mode: SQLite journal mode for every Hermes DB. "wal" default; use "delete" on
# weak-fsync/shared filesystems where WAL is not crash-safe (macOS virtiofs, NFS, SMB).
"database": {
"journal_mode": "wal",
# WAL sizing pragmas (ints). None = SQLite defaults (autocheckpoint 1000 pages, no limit).
"wal_autocheckpoint": None,
"journal_size_limit": None,
},
# Soft fd limit for long-running server processes; clamped to OS hard limit. 0/false/null = off.
"runtime": {"nofile_soft_limit": 4096},
# Global active chat session cap across CLI, TUI/dashboard, and messaging. None/0 = unbounded.
"max_concurrent_sessions": None,
# Soft LRU cap on in-memory TUI/desktop/dashboard sessions. Above it the gateway evicts the
# least-recently-active DETACHED sessions (no live client); reopening re-resumes from disk.
# 0/null disables.
"max_live_sessions": 16,
"session": {
# Per-terminal `hermes -c`: each CLI session writes a breadcrumb under
# $HERMES_HOME/terminal-sessions/<terminal-id>, so bare -c/--continue resumes THIS
# terminal's session (tmux/kitty/wezterm pane, tty). false = resume globally most-recent.
"terminal_continue": True,
},
"agent": {
# Turn cap. null = unlimited (default; caps caused silent mid-task truncation). Positive
# int caps; "none"/"unlimited"/"inf"/0/-1 also mean unlimited (resolve_turn_limit).
"max_turns": None,
# Wall-clock budget (seconds) per run. null = off. When set: one-time wrap-up notice at
# 80% elapsed; implicit provider stale timeouts capped to remaining budget.
# CLI equivalent: `hermes chat --run-budget N`.
"run_budget_seconds": None,
# Gateway inactivity timeout (seconds). Only fires when the agent is completely idle —
# not while calling tools or receiving API responses. 0 = unlimited.
"gateway_timeout": 1800,
# Max seconds an alias routing key waits for the active turn holding the same session
# lease; on expiry the message is rejected with a resend notice. Keep short: Telegram
# dispatches sequentially, so a waiter delays unrelated topics. Non-positive -> 5s.
"gateway_turn_lease_timeout": 5,
# Per-session AIAgent cache in the gateway. Each entry keeps a warm prompt prefix AND the
# full transcript: too small re-pays uncached prompts, too large fills the heap.
"agent_cache": {
"max_size": 128, # LRU entry cap
"idle_ttl_secs": 3600, # evict agents idle this long
# Anonymous-RSS budget (MB) above which LRU transcripts are shed (reloaded from
# disk next turn). "auto" = derive from cgroup memory limit (or total RAM);
# number = explicit; 0/off = disable the pass.
"memory_high_mb": "auto",
# Max sessions shed per pass (teardown bursts can't stall the gateway) and the number
# of most-recently-used sessions the pass never touches.
"max_evictions_per_pass": 16,
"protect_recent": 8,
},
# Force-interrupt budget (seconds) once gateway stop()/drain has begun (SIGTERM, and the
# final phase of in-band restart). 0 = interrupt immediately. Keep under systemd
# TimeoutStopSec or risk SIGKILL mid-cleanup; for /restart prefer
# restart_after_turn_timeout so turns finish BEFORE stop().
"restart_drain_timeout": 0,
# Cron-only floor under the stop()/drain wait (seconds). Interrupted chat turns resume on
# the next message, but an interrupted cron run is recorded as a permanent failure, so it
# must not inherit restart_drain_timeout's 0. Clamped to the shutdown-watchdog leash
# minus teardown headroom (~50s unless TimeoutStopSec is raised). 0 = opt out.
"cron_drain_timeout": 30,
# In-band restart (/restart, SIGUSR1): refuse new work, then wait up to this many seconds
# for in-flight agents/cron/api runs to finish before stop(). 0 = enter stop() at once.
# 30 min is a safety valve for wedged agents, not a target; raise for long unattended turns.
"restart_after_turn_timeout": 1800,
# Max seconds a submitted prompt waits for the deferred agent build (MCP discovery, model
# metadata, skills scan) before failing visibly. The prompt is delivered as soon as the
# build completes (progress notice past 30s), so this only fires on a hung build. Raise
# for many slow/unreachable MCP servers.
"build_wait_timeout": 600,
# Hermes-level retry attempts for API errors (connection drops, timeouts, 5xx) wrapping
# the whole call; the OpenAI SDK also retries transient errors (max_retries=2). Set 1 for
# fast failover to fallback providers; raise to tolerate longer provider hiccups.
"api_max_retries": 3,
# Empty-response retry guard. Empty retries re-send the full input at full price; this
# stops re-billing deterministic empties (unsignaled refusals, zero output tokens) while
# failing open on ambiguous evidence (missing usage, any tokens, model/provider change).
"empty_response_guard": {
"enabled": True, # False = legacy fixed 3 retries unconditionally
# When one empty attempt's estimated input cost >= this USD, the streak's retry
# budget drops from 3 to 1. Unknown pricing / missing usage leaves it untouched.
"cost_threshold_usd": 0.25,
},
# Fast mode: "" / "normal" (off), "fast" (always), "auto" (first fast_auto_seconds of
# every turn), "cold" (first turn of a session only).
"service_tier": "",
"fast_auto_seconds": 60,
# System-prompt guidance telling the model to call tools instead of describing actions.
# "auto" = gpt/codex models; true/false = force for all models; or a list of model-name
# substrings (e.g. ["gpt", "codex", "gemini", "qwen"]).
"tool_use_enforcement": "auto",
# Execution-discipline prompt block (tool persistence, tools for arithmetic/system facts,
# read-back after external writes, count reconciliation, literal identifiers,
# verification-gated completion). Chosen once per session by model name (byte-stable).
# "auto" = gpt/codex/grok/deepseek/kimi/qwen/glm/minimax/mimo/mistral; true/false = force;
# or a list of model-name substrings.
"execution_guidance": "auto",
# When the model narrates an action ("I'll go check the logs...") but emits no tool call,
# inject a "continue now, execute the tools" nudge and loop (max 2 nudges/turn). Corrective
# sibling of tool_use_enforcement. "auto" = codex_responses api_mode only; true = all
# api_modes (fixes Gemini/Claude "stops after stating intent"); false = never; or a list
# of model-name substrings.
"intent_ack_continuation": "auto",
# Anti-stall guards: (1) identical-call loop breaker appends a notice when the same tool
# is called 3+ times with identical args AND results (never blocks; pollers like `process`
# exempt); (2) continue-intent extension of empty-response recovery re-prompts once when
# the model says it will continue but takes no action. False disables both.
"stall_guards": True,
# "Finish the job" prompt block for all models: don't stop at a stub, never fabricate output
# when the real path is blocked. ~80 cached tokens. False disables.
"task_completion_guidance": True,
# Prompt block for all models steering independent tool calls (reads, searches, fetches,
# read-only commands) into one batched turn; the runtime already runs them concurrently.
# ~70 cached tokens. False disables.
"parallel_tool_call_guidance": True,
# Toolchain probe: surfaces Python/pip/uv/PEP-668 state in the system prompt only when
# something non-default is detected (no pip module, pip/python mismatch, PEP 668 without
# uv); zero tokens when clean. Skipped for docker/modal/ssh backends (own probe).
"environment_probe": True,
# Bot Mode teammate-messaging protocol section (silent unless desktop Bot Mode manages it).
"bot_mode_protocol": True,
# Embedder-supplied text appended to the system prompt's environment-hints block, so a
# host wrapping Hermes (sandbox runner, managed platform) can describe proxy/credential/
# mount layout without editing SOUL.md. Env HERMES_ENVIRONMENT_HINT overrides it.
"environment_hint": "",
# Coding posture: on interactive coding surfaces (CLI, TUI, desktop, ACP) in a code
# workspace, add a coding brief + live git/workspace snapshot to the system prompt
# (agent/coding_context.py). "auto" = prompt-only when interactive AND cwd is a code
# workspace (toolsets untouched, messaging platforms unaffected); "focus" = auto + collapse
# toolset to the lean coding set (+ enabled MCP servers) + demote non-coding skill
# categories to names-only (explicit opt-in); "on" = force everywhere; "off" = disable.
"coding_context": "auto",
# Standing operator instructions (string or list) appended to the coding brief as an extra
# stable system block — project-wide workflow rules, e.g. "Don't run tsc/lint until I
# approve." Cache-safe: takes effect next session.
"coding_instructions": "",
# When verify-on-stop finds edits without fresh verification evidence, add guidance for
# creative UI work (no broad tsc/lint/test before visual approval) and clean-diff
# expectations. false = keep the evidence nudge terse.
"verify_guidance": True,
# Max consecutive `pre_verify` "continue" nudges per turn (hooks can't trap the loop).
"max_verify_nudges": 3,
# Verification closure: after code edits in a workspace, refuse a final answer until fresh
# verification evidence exists or the agent explains why it can't check (bounded loop,
# passive ledger). False (default) because the nudges proved more noise than signal;
# true = force on everywhere; "auto" = on for interactive coding surfaces and programmatic
# callers, off for messaging surfaces. Doc/markdown/skill-only edits never fire.
"verify_on_stop": False,
# Inactivity warning (seconds), once per run before gateway_timeout; no interrupt. 0 = off.
"gateway_timeout_warning": 900,
# Max seconds the gateway blocks an agent awaiting a clarify-tool reply; then it unblocks
# with "[user did not respond within Xm]". CLI clarify blocks indefinitely and ignores
# this. 1h because users step away and a shorter value evicted the entry mid-think so a
# later button tap hit a dead entry. Lower it to free the running-agent guard sooner.
"clarify_timeout": 3600,
# "Still working" status interval (seconds); 0 = off. Lower = faster feedback, more noise;
# 180 catches spinning weak-model runs before users /restart.
"gateway_notify_interval": 180,
# Session stall watchdog (seconds): RECOVERY notifier for an in-process AIAgent with an
# adapter-queued follow-up while its activity clock is stale — NOT a general stall
# detector (ignores startup restore, build sentinels, leases, debounce, other processes;
# scan cadence per AIAgent). Notify-only: tells the user to try /new. Distinct from
# gateway_timeout (kills the turn) and gateway_notify_interval. 0 = disable.
"session_stall_timeout": 300,
# Transcript-sanitiser heal escalation: after this many pre-send heal passes within a
# 10-minute window, log one ERROR and queue a ONE-TIME out-of-band notice pointing at
# /debug share or `hermes doctor` (status channel only; prompt cache untouched).
# 0 = no escalation (per-window WARNINGs still fire).
"sanitizer_heal_escalation_threshold": 3,
# Seconds of continuous reconnect failure before a platform gets needs_attention flagged
# in gateway status (`hermes status` / fleet monitoring). Retries never stop — a signal,
# not a circuit breaker. 0 = disable.
"reconnect_attention_after": 7200,
# Freshness window (seconds) for the auto-continue note. After a crash/restart mid-run the
# next user message gets "[System note: your previous turn was interrupted...]" prepended;
# only when the last persisted transcript row is younger than this, so stale markers don't
# revive an unrelated old task. Covers gateway_timeout (1800) plus slack. 0 = always inject.
"gateway_auto_continue_freshness": 3600,
# Max seconds the gateway waits for boot auto-resume turns before releasing the
# startup-restore inbound gate (all inbound is QUEUED while shut, so one long resumed
# turn would leave every channel unanswered). On timeout the gate opens and the resume
# keeps running in the background; duplicate-agent protection is unaffected because the
# resume slot is claimed synchronously first. 0 = wait forever.
"gateway_startup_restore_drain_timeout": 30,
# Max seconds the boot turn-machinery warm-up (run_agent import graph, tool schemas +
# availability probes, context-file tier) may hold the inbound gate shut, so an early
# message isn't served with a skeleton system prompt. On timeout the gate opens and
# warm-up finishes in the background. 0 = disable warm-up (lazy init).
"gateway_startup_warmup_timeout": 20,
# Stale-stream ceiling (seconds) for local providers (Ollama, oMLX, llama-cpp). Applied
# when the base stale timeout is at its 180s default and a local endpoint is detected, so
# a wedged local server eventually trips the detector instead of hanging forever.
# Env HERMES_LOCAL_STREAM_STALE_TIMEOUT overrides.
"local_stream_stale_timeout": 900,
# How user-attached images reach the main model (gateway, TUI, CLI /attach). "auto" =
# native when the model reports supports_vision=True AND auxiliary.vision.provider is not
# explicitly set, else text; "native" = always attach (non-vision models error at the
# provider or get a last-chance text fallback); "text" = always pre-analyze with
# vision_analyze and prepend the description. vision_analyze stays a tool regardless.
"image_input_mode": "auto",
"disabled_toolsets": [],
# Model name (any reasonable spelling) -> effort level; overrides agent.reasoning_effort
# when the current model matches. Edit in config.yaml (no CLI support: dots in keys).
"reasoning_overrides": {},
# Preserve assistant `reasoning_content` on history replay. Echo families (DeepSeek,
# Kimi/Moonshot, Xiaomi MiMo) are auto-detected by provider name/base-URL host; custom
# providers and OpenAI-compatible gateways proxying them are not. Set
# `reasoning_echo: true` on a `model:` entry or a `fallback_providers:` entry to opt in
# per provider. Default false: strict providers (Mistral, Groq, Cerebras) reject the field.
"reasoning_echo": False,
# Turn liveness watchdog: a turn with no observable progress for `timeout_s` seconds is
# logged, force-interrupted so the UI can retry, and its lease stops renewing so stale-turn
# cleanup can reclaim the session even if the interrupt can't unwind a wedged frame.
# timeout_s <= 0 disables; poll_s = sampling interval. Invalid values (NaN, Inf,
# non-positive poll) warn and fall back to defaults. See agent/turn_liveness.py.
"turn_liveness": {"timeout_s": 600.0, "poll_s": 15.0},
},
"terminal": {
"backend": "local",
"modal_mode": "auto",
# Remote-backend connection-class failures (SSH host unreachable, Docker daemon down):
# "warn" = structured degraded tool result with reason + retry hint; "fail" = raise
# error + traceback.
"degraded_mode": "warn",
"cwd": ".", # Use current directory
# Root for terminal session temp files (background logs/pid/exit files, code-exec
# sandboxes). Empty = TMPDIR/TMP/TEMP if set, else HERMES_HOME/cache/terminal
# (auto-pruned after 72h) — NOT tmpfs /tmp, which is RAM-capped and fills under load.
# Must be an existing absolute POSIX path; user-set paths are never auto-pruned.
"temp_dir": "",
# CSS font-family for the desktop app's xterm.js terminal (e.g. "'CaskaydiaCoveNerdFont',
# monospace"). Empty = built-in default ("'JetBrains Mono', 'Cascadia Code', 'SF Mono',
# Menlo, Consolas, monospace"). Lets users use a Nerd Font without patching the app.
"font_family": "",
"timeout": 180,
# Seconds between SIGTERM and escalated SIGKILL for host process trees (browser daemons).
# 0 = SIGTERM only.
"daemon_term_grace_seconds": 2.0,
# Max seconds a one-shot CLI run (-q/-Q/-z) lingers for tracked notify_on_complete
# background processes to finish. The dying parent owns their stdout pipes, so exiting
# immediately kills the delivery (e.g. Bot Mode handoff replies via message_agent /
# bot_relay). Plain background processes without notify_on_complete are never waited on.
# 0 disables.
"oneshot_completion_wait_seconds": 600.0,
# Env vars passed into sandboxed terminal/execute_code (skill-declared
# required_environment_variables pass through automatically).
"env_passthrough": [],
# HOME for host tool subprocesses: "auto" = host keeps the real OS-user HOME, containers
# use HERMES_HOME/home; "real" = force real HOME; "profile" = force HERMES_HOME/home when
# it exists (strict per-profile isolation).
"home_mode": "auto",
# Extra files sourced in the login shell when building the per-session env snapshot —
# for nvm/pyenv/asdf/PATH entries registered by files a bash login shell skips
# (~/.bashrc, ~/.zshrc, ~/.zprofile). Supports ~ and ${VAR}; missing files skipped.
# When empty and the shell is bash, ~/.profile, ~/.bash_profile, ~/.bashrc are
# auto-sourced in that order (see auto_source_bashrc).
"shell_init_files": [],
# Source ~/.profile, ~/.bash_profile, ~/.bashrc in the snapshot login shell to capture
# PATH additions, functions, and aliases that `bash -l -c` misses (bash skips bashrc when
# non-interactive; Debian/Ubuntu ~/.bashrc short-circuits). ~/.profile and ~/.bash_profile
# go first because n/nvm/asdf write PATH exports there without an interactivity guard.
# Turn off if an rc file misbehaves when sourced non-interactively (exits on TTY check).
"auto_source_bashrc": True,
"docker_image": "nikolaik/python-nodejs:python3.11-nodejs20",
"docker_forward_env": [],
# Exact key-value env pairs set inside Docker containers (unlike docker_forward_env, which
# reads host values) — useful under systemd without the user's shell env.
# Example: {"SSH_AUTH_SOCK": "/run/user/1000/ssh-agent.sock"}
"docker_env": {},
"singularity_image": "docker://nikolaik/python-nodejs:python3.11-nodejs20",
"modal_image": "nikolaik/python-nodejs:python3.11-nodejs20",
"daytona_image": "nikolaik/python-nodejs:python3.11-nodejs20",
"vercel_runtime": "node24", # vercel_sandbox backend only: node24 | node22 | python3.13
# Container limits (docker, singularity, modal, daytona, vercel_sandbox; not local/ssh).
"container_cpu": 1,
"container_memory": 5120, # MB (default 5GB)
"container_disk": 51200, # MB (default 50GB)
"container_persistent": True, # Persist filesystem across sessions
# Docker volume mounts, "host_path:container_path" (docker -v syntax), e.g.
# ["/home/user/.hermes/cache/documents:/output"]. For gateway MEDIA delivery, write to
# /output/... inside Docker and emit the host-visible path in MEDIA:, not the container one.
"docker_volumes": [],
"docker_mount_cwd_to_workspace": False, # mount host cwd at /workspace (weakens isolation)
"docker_network": True, # false = --network=none, no network access from commands
"docker_extra_args": [], # Extra flags passed verbatim to docker run
# /dev/shm size for the Docker sandbox. Docker's 64 MB default silently breaks
# Chromium/Playwright and PyTorch DataLoader workers; tmpfs is lazily allocated so the
# higher ceiling is free until used. "" or "0" = omit the flag (Docker default).
"docker_shm_size": "1g",
# Run the container as the host uid:gid (`--user`) so files written to bind mounts
# (docker_volumes, persistent workspace, mounted cwd) are owned by you, not root. Off by
# default for images whose entrypoints must start as root (e.g. the bundled Hermes image,
# which drops to `hermes` via s6-setuidgid). When on, SETUID/SETGID caps are omitted.
"docker_run_as_host_user": False,
# Trusted profiles sharing one Docker container identity; empty = per-profile boundary.
"docker_shared_container_key": "",
# Keep a long-lived bash shell across execute() calls so cwd/env/shell variables survive.
# Applies to non-local backends (SSH); local is opt-in via TERMINAL_LOCAL_PERSISTENT env.
"persistent_shell": True,
},
"web": {
"backend": "", # shared fallback — applies to both search and extract
"search_backend": "", # per-capability override for web_search (e.g. "searxng")
"extract_backend": "", # per-capability override for web_extract (e.g. "native")
# per-page char budget for web_extract; larger pages truncate, full text kept in cache/web
"extract_char_limit": 15000,
# Keyless free-tier ring: with NO web backend configured or keyed, web_search/web_extract
# rotate round-robin across exa, parallel, firecrawl, keenable public free tiers, failing
# over on rate limits. Never pre-empts a configured/keyed backend. false = disable.
"keyless_fallback": True,
# One-shot rescue: when the chosen/keyed backend fails a call, THAT call retries once on
# the keyless ring; the next call tries the chosen backend again (no sticky failover).
# Off when keyless_fallback is false.
"keyless_rescue": True,
# Per-vendor tier for vendors with both a keyless free endpoint and a keyed paid path
# (exa, parallel, firecrawl, keenable; tavily is opt-in keyless via `hermes tools`, not a
# ring member). Set by the `hermes tools` picker. "free" = always anonymous endpoint even
# with a key; "paid" = always keyed (missing key = error; vendor excluded from the ring);
# unset = keyed when the key is present, else the ring.
"provider_tier": {},
# TTL caching for web_search + web_extract: repeat searches (same query + provider) within
# the TTL come from an in-process memo; repeat extracts from the cache/web store.
# Concurrent identical searches coalesce into one vendor request. Only successes cached.
"cache_enabled": True,
"cache_ttl_minutes": 20,
# Hosts always fetched live, never from the extract cache (staging deploys, tunnel URLs,
# preview builds). Entries match exactly, as "*.wildcard", or as a domain suffix
# ("mysite.dev" also covers "preview.mysite.dev"). localhost/private IPs always exempt.
"cache_exempt_hosts": [],
},
"browser": {
# "" = Browser Use mode when the browser-use CLI (or uvx) is available, else built-in tools
# (Camofox setups always keep built-in tools: no CDP surface); "browser-use" = force one
# browser_exec tool driving the Browser Use CLI over any CDP backend (local Chrome, cloud);
# "off" = force the built-in browser_navigate/browser_click/... tools.
"backend": "",
"inactivity_timeout": 120,
"command_timeout": 30, # seconds per browser command (screenshot, navigate, etc.)
"snapshot_threshold": 15000, # max chars before snapshot truncate-and-store (min 1000)
"record_sessions": False, # auto-record browser sessions as WebM videos
# headed: visible Chromium window (local); skips per-turn cleanup, idle reaper still applies
"headed": False,
"allow_private_urls": False, # allow private/internal IPs (localhost, 192.168.x.x, ...)
# Local browser engine for both drivers. "auto" = Chrome; "lightpanda" = faster navigation,
# no screenshots (Browser Use mode spawns `lightpanda serve` per session; built-in tools
# pass `--engine <value>` to agent-browser with Chrome fallback); "chrome" = explicit.
# Ignored while a cloud provider, Camofox, cdp_url or use_real_profile is active.
# Also settable via AGENT_BROWSER_ENGINE.
"engine": "auto",
# With a cloud provider, auto-spawn local Chromium for LAN/localhost URLs instead
"auto_local_for_private_urls": True,
"cdp_url": "", # persistent CDP endpoint for attaching to an existing Chromium/Chrome
# Consent to browse with the user's REAL logins locally: runs on a Hermes-managed SNAPSHOT
# of the ACTIVE default-Chromium profile (Local State -> profile.last_used; cookies, logins,
# prefs copied and re-synced per fresh session) driven by Hermes' packaged Chromium. The
# snapshot dir sidesteps Chrome 136+'s default-profile debugging block and never contends
# with the running browser. Turning off deletes ~/.hermes/browser-profile/ so credentials
# don't outlive consent. Chromium-family only (Chrome, Edge, Brave, Brave Origin,
# Chromium); Firefox etc. fails closed. Also gates the browser_exec `local` argument
# (real-profile local session even under a cloud backend). Desktop Settings -> Browser.
"use_real_profile": False,
# Windows only: a running Chrome/Edge/Brave locks its cookie DB, so the profile can't be
# copied. When on, a locked profile still blocks and the agent ASKS first; on approval it
# runs `hermes browser close-profile` (kills that profile's browser tree, unsaved tabs
# lost) and retries once; still locked -> stays blocked, no auto-kill. No effect on
# macOS/Linux (copy-while-running works).
"real_profile_autoclose": False,
# Pin WHICH source profile directory is snapshotted for real-profile browsing (e.g.
# "Profile 2"). Empty = browser's last-used profile, which on multi-profile machines can
# hand the agent the wrong identity. A pin naming a missing directory FAILS CLOSED.
"real_profile_pin": "",
# restrict_evaluate: opt-in denylist blocking sensitive JS primitives (cookies/storage/
# clipboard/network/form values) in browser_console(expression=...);
# allow_unsafe_evaluate is the legacy override that bypasses that denylist entirely.
"allow_unsafe_evaluate": False,
"restrict_evaluate": False,
# CDP supervisor: dialog + frame detection over a persistent WebSocket; active only with a
# CDP-capable backend (Browserbase, or local Chrome via /browser connect).
# See website/docs/developer-guide/browser-supervisor.md.
"dialog_policy": "must_respond", # must_respond | auto_dismiss | auto_accept
"dialog_timeout_s": 300, # safety auto-dismiss after N seconds under must_respond
"camofox": {
# true = send a stable profile-scoped userId so Camofox maps it to a persistent
# Firefox profile; false = random ephemeral userId per session.
"managed_persistence": False,
# Externally managed Camofox identity, for when another app owns the visible browser.
"user_id": "",
"session_key": "",
"adopt_existing_tab": False, # rehydrate tab_id from Camofox before creating a tab
# Docker Camofox opens page URLs from inside the container: rewrite loopback page URLs
# (localhost/127.0.0.1/::1) to the host alias; CAMOFOX_URL itself is unchanged.
"rewrite_loopback_urls": False,
"loopback_host_alias": "host.docker.internal",
},
# Authenticated browser-extension controller lane: a registered extension can become the
# exact controller for a session's browser_* tools (fail-closed once bound). Local API
# registration also requires the API server bearer key. developer_mode gates the
# privileged browser_cdp / browser_evaluate capabilities.
"extension_control": {"enabled": False, "developer_mode": False},
},
# Filesystem checkpoints: snapshot the working directory once per turn (on the first
# write_file/patch call); restore with /rollback. Opt-in via `hermes chat --checkpoints` or
# enabled=True (most users never use /rollback). Single shared shadow store with real pruning.
"checkpoints": {
"enabled": False,
# Max checkpoints per working directory; enforced by ref rewrite + GC of older commits.
"max_snapshots": 20,
# Hard ceiling on total ~/.hermes/checkpoints/ size (MB); the oldest checkpoint per project
# is dropped round-robin until under the cap. 0 disables.
"max_total_size_mb": 500,
# Skip files larger than this (MB) when staging (datasets, model weights). 0 = no filter.
"max_file_size_mb": 10,
# Startup sweep (at most once per min_interval_hours): deletes projects whose last_touch
# is older than retention_days, GCs the shared store, enforces max_total_size_mb, deletes
# legacy-* archives older than retention_days. It NEVER deletes orphans (workdir missing
# on disk) — a missing workdir may just be an unmounted volume/VPN, and an unattended
# sweep must not guess. Orphans: `hermes checkpoints prune` (`--keep-orphans` to skip).
"auto_prune": True,
"retention_days": 7,
"min_interval_hours": 24,
},
# Hard cap (chars) for one auto-loaded context file (SOUL.md, AGENTS.md, CLAUDE.md,
# .hermes.md, .cursorrules) before head/tail truncation. null = scale with the model's
# context window (floor 20K, ceiling 500K); a positive int pins a fixed cap.
# Separate from read_file limits.
"context_file_max_chars": None,
# Max chars per read_file call; larger reads are rejected with offset+limit guidance.
# 100K chars ≈ 25–35K tokens.
"file_read_max_chars": 100_000,
# Seconds the first agent build waits for background MCP discovery before snapshotting
# its tool list. Returns the instant discovery completes (no MCP servers → ~0s); the
# bound only bites when a server is still connecting. Turn-1 latency knob only: a
# server that misses it is picked up by the between-turns refresh (agent/turn_context.py),
# so keep it small — a dead server adds this much to first-response latency.
"mcp_discovery_timeout": 1.5,
# Same bound for single-query mode (``hermes -q/-z``). With only ONE turn there is no
# between-turns refresh, so a server that misses the window is invisible for the whole
# session; the larger bound lets slow cold-start servers (npx, uvx, remote HTTP) land.
# Reachable servers still only wait their real handshake time.
"mcp_single_query_discovery_timeout": 15.0,
# MCP runtime behavior (distinct from mcp_servers: definitions and auxiliary.mcp).
"mcp": {
# Auto-reload MCP connections when config.yaml's mcp_servers changes (CLI watcher).
# Every reload rebuilds the tool surface and INVALIDATES the provider prompt cache
# (next message re-sends the full prefix) — costly on long-context models. When
# false the watcher still detects the change and prints /reload-mcp guidance.
"auto_reload_on_config_change": True,
},
# Tool-output truncation. max_bytes: terminal_tool output cap in chars (head+tail kept;
# 50_000 ≈ 12-15K tokens). max_lines: max `limit` one read_file call may request before
# clamping. max_line_length: per-line cap in read_file's line-numbered view (chars).
"tool_output": {"max_bytes": 50000, "max_lines": 2000, "max_line_length": 2000},
# Tool loop guardrails nudge models that repeat failed/non-progressing tool calls.
# Soft warnings are always on; hard stops are opt-in so interactive sessions keep flowing.
"tool_loop_guardrails": {
"warnings_enabled": True,
"hard_stop_enabled": False,
# Unattended gateway/cron platforms hard-stop by default (nobody can /stop a model
# that ignores warnings); interactive cli/tui/desktop/acp stay warning-only.
"non_interactive_hard_stop_enabled": True,
"warn_after": {"exact_failure": 2, "same_tool_failure": 3, "idempotent_no_progress": 2},
"hard_stop_after": {
"exact_failure": 5,
"same_tool_failure": 8,
"idempotent_no_progress": 5,
},
# Per-turn hard ceilings for runaway-prone tools; counters reset every turn, always
# on regardless of the thresholds above. Dozens of searches/subagents in ONE turn
# is already pathological, hence low defaults. 0 = unlimited.
"loop_caps": {
"max_web_searches": 50, # web_search calls per turn
"max_subagents": 50, # subagents spawned per turn
},
},
"compression": {
"enabled": True,
# checkpoint_required: fail closed before lossy compaction unless an active memory
# provider confirms checkpoint API compatibility and completes the checkpoint.
"checkpoint_required": False,
# progress_notices: when True, routine compression progress statuses (compacting/
# preflight/pre-API/idle/retry) reach chat gateways instead of being filtered as
# noise. Failure notices and manual /compress feedback are always visible.
"progress_notices": False,
# threshold: compress when context usage exceeds this ratio. Models with windows
# below 512K are floored at 0.75 (raise-only) so compaction doesn't fire with half
# the window free; set above 0.75 to override the floor.
"threshold": 0.50,
# threshold_tokens: absolute token cap — compression triggers at the lower of the
# ratio threshold and this count. Clamped to the model's context length.
"threshold_tokens": None,
"target_ratio": 0.20, # fraction of threshold to preserve as recent tail
# tail_mode: "lean" = clamped 2.5%-of-window tail (10K floor / 25K cap) plus chunked
# digests, anchor index, verbatim user messages and session_search pointers in the
# summary (~3x fewer retained tokens; a few extra summarizer calls at the boundary).
# "legacy" = 0.20×threshold verbatim tail (100-240K tokens on big windows).
"tail_mode": "lean",
"protect_last_n": 20, # minimum recent messages kept uncompressed
# min_tail_user_messages: REAL (actionable) user messages guaranteed to survive in
# the tail. 1 = single last-user anchor; raise (e.g. 3) when bulky tool outputs
# fill the tail budget.
"min_tail_user_messages": 1,
# max_attempts: retry rounds before a turn gives up with "max compression attempts
# reached". Raise (e.g. 6) for tool-schema-heavy sessions. Validated >= 1, cap 10.
"max_attempts": 3,
# proactive_prune_tokens: opt-in trigger (tokens) for the deterministic no-LLM
# tool-result prune, independent of `threshold` (which rarely fires on large
# windows, so old tool output is re-sent every turn); e.g. 48000 reclaims early.
# 0 = off. Tail protected by `protect_last_n`. Built-in compressor only. Each
# committed prune rewrites sent history and breaks the prompt-cache prefix — the
# min_reclaim gate below keeps those breaks episodic.
"proactive_prune_tokens": 0,
# Prune's summarize pass only touches tool results larger than this (chars);
# clamped >= 200 so a generated summary can't be re-summarized.
"proactive_prune_min_result_chars": 8000,
# A prune only commits when it reclaims at least this many tokens, then waits for a
# trigger-sized runway to regrow before rearming. 0 = no minimum-savings gate.
"proactive_prune_min_reclaim_tokens": 4096,
# micro_compact: opt-in — after each turn fold the oldest un-absorbed exchange into a
# rolling summary, amortizing compression cost. Off by default because every pass
# rewrites sent history and breaks the prompt-cache prefix EVERY turn; enable only
# if the amortized stall beats the cached-prefix discount. See docs/micro-compaction.md.
"micro_compact": False,
# Cadence: run a pass every Nth completed turn (1 = one cache break per turn, 5 =
# a fifth of the breaks). Clamped >= 1; ignored unless micro_compact is true.
"micro_compact_every_n_turns": 1,
# Once the rolling summary exceeds this many tokens, the next pass re-summarizes it.
"micro_compact_defrag_threshold_tokens": 2000,
# Gateway session-hygiene force-compress threshold, by message count.
"hygiene_hard_message_limit": 5000,
# Max seconds the gateway waits for pre-agent hygiene compression WITHOUT forward
# progress. Inactivity budget: a slow model still streaming tokens extends the wait.
"hygiene_timeout_seconds": 30,
# Absolute cap on the hygiene wait even while tokens are moving (bounds a trickle
# stream). Clamped >= hygiene_timeout_seconds.
"hygiene_total_ceiling_seconds": 600,
"hygiene_failure_cooldown_seconds": 300, # skip repeated failed hygiene attempts
# Max seconds an ARRIVING user turn is held while a streaming hygiene summary
# finishes; bounds user-visible latency (keep under chat idle timeouts, Telegram
# ~30s). On expiry the turn proceeds uncompressed; the detached worker keeps its
# watermark-fenced commit, so the summary is adopted at the next safe boundary.
"hygiene_max_turn_hold_seconds": 10,
# Inactivity budget for in-agent compress_context (loop, /compress, preflight);
# same progress-aware semantics as hygiene_timeout_seconds. 0 = disable the owned
# wrapper (callers passing commit_fence, e.g. gateway hygiene, never use it).
"context_timeout_seconds": 120,
# Absolute cap on the *pre-commit* compress_context wait (summary/stream phase) even
# while tokens move. Clamped >= context_timeout_seconds when that is > 0. A started
# SessionDB commit is never abandoned: past the ceiling it is logged (WARNING, then
# ERROR) and surfaced on the warning channel while the host keeps waiting.
"context_total_ceiling_seconds": 600,
# Non-system head messages always kept verbatim, in ADDITION to the (always
# protected) system prompt. 0 = pin nothing but system prompt + summary + tail.
"protect_first_n": 3,
# When True, auto-compression whose summary fails (aux error / non-JSON / timeout)
# aborts instead of dropping the middle with a "summary unavailable" placeholder;
# the session freezes at its size until /compress (bypasses the cooldown) or /new.
"abort_on_summary_failure": False,
# (Historical key name.) When True, gpt-5.4/5.5/5.6 on the ChatGPT Codex OAuth route
# raise their compaction trigger to 85%: Codex hard-caps them at a 272K window, so
# the global 50% would compact at ~136K. False = global `threshold`. Only that route;
# the same models via OpenAI direct, OpenRouter or Copilot keep the global value.
"codex_gpt55_autoraise": True,
# Show the one-time autoraise banner; False keeps the autoraise, hides the notice.
"codex_gpt55_autoraise_notice": True,
# Codex app-server thread compaction mode. The codex agent owns the thread context,
# so Hermes' summarizer cannot shrink it. native = codex decides; hermes = Hermes'
# threshold triggers thread/compact/start; off = never auto-trigger.
"codex_app_server_auto": "native",
# Opt in to OpenAI server-side compaction on the Responses API. Only gpt-5.6-family
# on api.openai.com or the Codex backend; local compression stays as fallback.
"codex_responses_native": False,
# Absolute server compaction trigger (input tokens). None follows the local trigger
# with a safety margin; explicit values only clamp downward so the server goes first.
"codex_responses_compact_threshold": None,
# in_place: compaction rewrites the message list and system prompt WITHOUT rotating
# the session id (no parent_session_id chain, no `name #N` renumbering), avoiding
# the session-rotation bug cluster. Pre-compaction turns are soft-archived under the
# same id (active=0, compacted=1) — still session_search-able. False = legacy
# rotating-compaction path.
"in_place": True,
# Per-model threshold overrides: keys substring-match the model name (longest wins),
# values replace the global `threshold`, e.g. {"glm-5.2": 0.40}. The <512K floor
# (0.75) still applies raise-only on top.
"model_thresholds": {},
# Opt-in idle compaction (0 = off): a session resuming after this many idle seconds
# compacts up front, before the first reply. Time-based complement to `threshold`;
# skipped when already at/below threshold × target_ratio; honors the same cooldown/
# anti-thrash/lock guards. Example: 1800 = 30 min.
"idle_compact_after_seconds": 0,
},
# Anthropic prompt caching (Claude via OpenRouter or native API). cache_ttl: "5m" | "1h";
# other non-falsy values are ignored; falsy (false, null, "off", "disabled", "no",
# "none") disables caching.
"prompt_caching": {"cache_ttl": "5m"},
# OpenRouter settings. response_cache: X-OpenRouter-Cache header — identical requests
# return cached responses at zero billing; independent of Anthropic prompt caching.
# response_cache_ttl: seconds (1-86400), only used when response_cache is on.
# min_coding_score (0.0-1.0): pareto-code router knob, applied only when model.model is
# "openrouter/pareto-code"; higher = stronger/pricier coders, 0.65 = mid-tier, "" = let
# OpenRouter pick the strongest. Docs: openrouter.ai/docs/guides/routing/routers/pareto-router
"openrouter": {"response_cache": True, "response_cache_ttl": 300, "min_coding_score": 0.65},
# AWS Bedrock; only used when model.provider is "bedrock".
"bedrock": {
"region": "", # empty = AWS_REGION env var → us-east-1
"discovery": {
"enabled": True, # auto-discover models via ListFoundationModels
"provider_filter": [], # restrict to these providers, e.g. ["anthropic", "amazon"]
"refresh_interval": 3600, # cache discovery results (seconds)
},
# Bedrock Guardrails: create one in the console, then set ID and version.
# https://docs.aws.amazon.com/bedrock/latest/userguide/guardrails.html
"guardrail": {
"guardrail_identifier": "", # e.g. "abc123def456"
"guardrail_version": "", # e.g. "1" or "DRAFT"
"stream_processing_mode": "async", # "sync" | "async"
"trace": "disabled", # "enabled" | "disabled" | "enabled_full"
},
},
# Auxiliary model config — provider/model per side task. provider "auto" = auto-detect;
# empty model = provider's default aux model; all tasks fall back to
# openrouter:google/gemini-3-flash-preview when the configured provider is unavailable.
# extra_body is forwarded verbatim as request body fields for that task, e.g. OpenRouter
# routing prefs / Pareto Code floor:
# auxiliary:
# compression:
# extra_body:
# provider: {order: [anthropic, google], sort: throughput} # or price | latency
# plugins: [{id: pareto-router, min_coding_score: 0.5}]
# Each task is independent — main-agent provider_routing and openrouter.min_coding_score
# do NOT propagate to aux calls by design.
"auxiliary": {
# Same-provider retries for a transient blip (reset/timeout/5xx/408) on ANY aux call
# before falling back; clamped [0,6]. Matters for pinned calls (MoA advisors) where
# provider fallback is not meaningful recovery.
"transient_retries": 2,
# When true, the auto-chain's OpenRouter step is skipped unless the fallback model
# ends in ":free" — a PAID lane is never used for background aux traffic even with
# OPENROUTER_API_KEY set.
"free_only": False,
# Override the auto-chain's OpenRouter fallback model (default google/gemini-3.6-flash,
# PAID). Pair e.g. "nvidia/nemotron-3-ultra-550b-a55b:free" with free_only: true.
# A one-time WARNING is logged whenever a non-":free" model is engaged.
"openrouter_model": "",
# Endpoints that reject NON-streaming chat (HTTP 400): aux calls are sent with
# stream=True and aggregated. Case-insensitive URL substrings; copilot.tencent.com
# is always stream-only.
"stream_only_base_urls": [],
# Per-task blocks share one shape (_aux): provider "auto" = inherit the main model;
# base_url overrides provider; api_key falls back to OPENAI_API_KEY; reasoning_effort:
# none|minimal|low|medium|high|xhigh|max|ultra ("" = provider default); extra_body =
# OpenAI-compatible request fields. Vision: download_timeout = image HTTP download (s).
"vision": _aux(120, download_timeout=30),
# web_extract and session_search no longer use an aux LLM; leftover blocks in user
# config are ignored. Compression: raise timeout for local models. max_output_tokens
# is only honored with a concrete provider/model AND ``reasoning_effort: none``;
# 0 = uncapped.
"compression": _aux(120, max_output_tokens=0),
"skills_hub": _aux(30),
"approval": _aux(30), # classifier — a fast/cheap model is recommended
# /review reviewer: a full subagent on the async delegation rail, credentials
# resolved like delegation.provider pins. "auto" + "" = main agent's model.
# api_mode forces transport: chat_completions | anthropic_messages | codex_responses.
"review": {"provider": "auto", "model": "", "base_url": "", "api_key": "", "api_mode": ""},
"mcp": _aux(30),
# prefer_fast_model opts in to the provider fast tier; auto otherwise = main model.
"title_generation": {
"enabled": True,
"provider": "auto",
"model": "",
"prefer_fast_model": False,
"base_url": "",
"api_key": "",
"timeout": 30,
"extra_body": {},
"reasoning_effort": "",
"language": "",
},
"memory_query_rewrite": _aux(8, reasoning_effort=False),
"tts_audio_tags": _aux(30),
# Kanban: triage_specifier expands a Triage one-liner into a spec (cheap model OK);
# kanban_decomposer emits a JSON graph of child tasks (more tokens).
"triage_specifier": _aux(120),
"kanban_decomposer": _aux(180),
"profile_describer": _aux(60), # 1-2 sentence profile blurb; short, cheap
"goal_judge": _aux(60), # /goal satisfaction + contract drafting; JSON calls
# Curator skill-usage review can take minutes on reasoning models (umbrellas over
# hundreds of skills); route cheaper via `hermes model` → auxiliary → Curator.
"curator": _aux(600),
"monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine
# Post-turn self-improvement fork (save memory / patch skill). "auto" = main model
# replaying the full conversation (warm cache); other models replay a compact digest
# (~3-5x cheaper). enabled=false skips auto spawns (/refine still works).
# max_input_tokens caps the SUM of replayed input tokens over the review loop
# (iterations capped at 16); the loop stops before crossing it. <= 0 = unlimited.
"background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000},
# No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset
# (moa.presets.<name>.reference_models[].reasoning_effort / aggregator.reasoning_effort).
"moa_reference": _aux(900, reasoning_effort=False),
"moa_aggregator": _aux(900, reasoning_effort=False),
},
"display": {
"compact": False,
"personality": "",
"resume_display": "full",
# Recap tuning for /resume and startup resume.
"resume_exchanges": 10, # max user+assistant pairs to show
"resume_max_user_chars": 300, # truncate user message text
"resume_max_assistant_chars": 200, # truncate non-last assistant text
"resume_max_assistant_lines": 3, # truncate non-last assistant lines
# Skip tool-call-only assistant entries in the recap so it isn't dominated by
# `[2 tool calls: ...]` lines; False shows them inline.
"resume_skip_tool_only": True,
"busy_input_mode": "interrupt", # interrupt | queue | steer
# steer mode: false hides only the "Steered into current run" bubble; steering
# itself still happens.
"busy_steer_ack_enabled": True,
# Classic CLI multiline beyond Alt+Enter: Ctrl+J newline, trailing backslash+Enter
# continues, Shift+Enter reported distinctly. False restores the c-j submit fallback
# for POSIX PTYs whose plain Enter arrives as LF.
"cli_multiline_shortcuts": True,
# Interface bare `hermes`/`hermes chat` launches: "cli" (prompt_toolkit REPL) | "tui"
# (Ink). Flags win: `--cli` forces the REPL, `--tui` / HERMES_TUI=1 forces the TUI.
"interface": "cli",
# `hermes --tui` auto-resumes the most recent human-facing session (like `hermes -c`).
# HERMES_TUI_RESUME=<id> always wins.
"tui_auto_resume_recent": False,
# Desktop reopens the last chat/page on cold start (also in Settings → Appearance).
"resume_last_session": True,
# One-time TUI hint ("subagents working · /agents to watch live") on first delegation.
"tui_agents_nudge": True,
"bell_on_complete": False,
"bell_on_prompt": False, # bell when a blocking prompt opens (clarify/approval/sudo)
# Stream reasoning live before the response; otherwise thinking models show only a
# spinner for tens of seconds.
"show_reasoning": True,
# Post-response "Reasoning" recap collapses to 10 lines; true prints it all
# (live streaming is always full).
"reasoning_full": False,
# Background self-improvement notices in chat: "off" (review still runs) | "on"
# (generic "💾 Memory updated") | "verbose" (content preview). Per-platform via
# display.platforms.<platform>.memory_notifications.
"memory_notifications": "on",
# Gateway notices when a terminal(background=true) process finishes: "concise"
# (one line; failures append an output tail) | "all" (running updates + final raw
# output) | "result" (final raw only) | "error" (raw only on non-zero exit) | "off".
"background_process_notifications": "concise",
"streaming": False,
"timestamps": False, # message timestamps (CLI labels, TUI rows, desktop transcript)
"timestamp_format": "%H:%M", # strftime format, e.g. "%b-%d %H:%M"
"final_response_markdown": "strip", # render | strip | raw
# Preserve recent classic-CLI output across Ctrl+L, /redraw and resize clears;
# disable if an emulator misbehaves with replayed scrollback.
"persistent_output": True,
"persistent_output_max_lines": 200,
# Also clear terminal scrollback on classic-CLI full redraw/resize recovery; enable
# when a terminal/tmux stack stamps stale prompt chrome into scrollback.
"cli_rebuild_scrollback_on_redraw": False,
# Print a one-line summary of resolved modal prompts (approval/clarify) to scrollback.
"persist_prompts": True,
"inline_diffs": True, # inline diff previews for write_file/patch/skill_manage
# Append a one-line advisory to the final response when a write_file/patch failed
# this turn and was never superseded by a successful write to the same path
# (catches "half the parallel patches failed, model claims success").
"file_mutation_verifier": True,
# Nous credits status-bar notices (usage bands, grant-spent, depleted/restored).
# False mutes them; balance data and /usage keep working.
"credits_notices": True,
# Append a one-line explanation when a turn ends with no usable reply (empty after
# retries, truncated stream, pending tool result, iteration/budget limit) instead of
# the bare "(empty)" sentinel.
"turn_completion_explainer": True,
"show_cost": False, # $ cost in the status bar
"battery": False, # battery read-out first in status bar; no-op w/o battery
# Focus view (/focus): display-only. Pins tool_progress to "off", reports per-turn
# hidden-line count, pins a "focus" status segment. focus_saved_tool_progress holds
# the mode /focus off restores. Never affects what the model sees (focus_view.py).
"focus_view": False,
"focus_saved_tool_progress": "all",
"skin": "default",
# UI language for static messages (approval prompts, some gateway slash replies); not
# agent responses/logs/tool outputs. en, zh, ja, de, es, fr, tr, uk; unknown → en.
"language": "en",
# TUI busy indicator: kaomoji | emoji | unicode (braille) | ascii. `/indicator <style>`.
"tui_status_indicator": "kaomoji",
# Seconds between idle prompt_toolkit redraws in the classic CLI; keeps wall-clock
# status-bar read-outs ticking and the bottom chrome from going stale. 0 disables it
# if it fights terminal auto-scroll in non-fullscreen mode.
"cli_refresh_interval": 1.0,
"user_message_preview": { # CLI: submitted user-message lines echoed to scrollback
"first_lines": 2,
"last_lines": 2,
},
# Gateway: natural mid-turn assistant status messages. Desktop: keep mid-turn
# narration between tool calls instead of collapsing to the final message.
"interim_assistant_messages": True,
# Codex Responses commentary channel: true delivers completed commentary as mid-turn
# interim updates; false routes it to reasoning (visible only with show_reasoning).
"show_commentary": True,
"tool_progress_command": False, # enable /verbose command in messaging gateway
# display.tool_progress_overrides is deprecated (use display.platforms); a user-set
# value is still honored at runtime and folded into platforms by migration.
"tool_preview_length": 0, # max chars for tool call previews (0 = no limit)
# Human-phrased status labels for built-in tools ("Reading <file>") in CLI spinner and
# gateway/desktop tool-progress; custom/plugin/MCP tools use the raw preview.
"friendly_tool_labels": True,
# CLI-only post-turn line: "⋯ 12.4s · edited 2 files +18 -3 · read 4 files · ran 3
# commands". Never in quiet/non-interactive or gateway surfaces (own footer).
"turn_summary": True,
# CLI-only: cumulative turn output tokens on the live spinner ("· ↓ 1.2k tok").
"spinner_token_flow": True,
# Gateway tool-progress grouping where edits are supported: "accumulate" edits one
# bubble | "separate" one message per tool (noisier). Needs tool_progress enabled.
# Per-platform: display.platforms.<platform>.tool_progress_grouping.
"tool_progress_grouping": "accumulate",
# Custom long-running status phrases. Defaults: gateway/assets/status_phrases.yaml.
# `path`/`paths` = HERMES_HOME-relative YAML files/dirs (or conventional
# status_phrases.yaml / status_phrases/*.yaml). Keys: status, generic. mode: "append"
# (default) | "replace". Per-platform: display.platforms.<platform>.status_phrases.
"status_phrases": {},
# Reasoning summary rendering: "code" (💭 fenced block) | "blockquote" ("> ") |
# "subtext" ("-# " Discord small grey text; Discord's default). Per-platform via
# display.platforms.<platform>.reasoning_style.
"reasoning_style": "code",
# Auto-delete EphemeralReply system notices ("✨ New session started!", …) after N
# seconds where deletion is supported (Telegram; others ignore). Agent responses are
# never touched. 0 = disabled.
"ephemeral_system_ttl": 0,
# Per-platform display/streaming overrides; unset keys fall through to the global.
# Telegram has smooth native draft streaming (on); Discord/Slack only edit-based
# streaming, which flickers (off). Gap-fillers only: explicit user values win, and the
# global streaming.enabled master switch still gates everything.
"platforms": {
"telegram": {"streaming": True},
"discord": {"streaming": False},
"slack": {"streaming": False},
# WeCom native streaming (msgtype "stream" via aibot_respond_msg).
"wecom": {"streaming": True},
},
# Gateway runtime footer on the FINAL message, e.g. `model · 68% · ~/projects/hermes`.
# Per-platform: display.platforms.<platform>.runtime_footer.
"runtime_footer": {
"enabled": False,
"fields": ["model", "context_pct", "cwd"], # order shown; drop any to hide
},
# CLI/TUI status bar fields. Non-empty = only listed fields show (built-in order kept,
# config controls visibility not ordering); empty = default set. Available: model,
# context_detail, context_pct, cache_hit, latency, tps, compressions, bg_tasks,
# bg_processes, bg_subagents, goal, duration, prompt_elapsed, idle_since, focus,
# yolo, stash, battery, title, total_tokens (session Σ, opt-in only). Narrow terminals
# still drop context_detail/prompt_elapsed/idle_since.
"status_bar": {
"fields": [],
},
"copy_shortcut": "auto", # "auto" (platform default) | ctrl_c | ctrl_shift_c | disabled
# Petdex animated mascot (github.com/crafter-station/petdex): cosmetic sprite across
# CLI/TUI/desktop, managed with `hermes pets`. No effect on prompt caching.
"pet": {
"enabled": False,
"slug": "", # active pet slug in get_hermes_home()/pets/; empty → first installed
# auto (detect kitty/iTerm2/sixel, else unicode half-blocks) | kitty | iterm |
# sixel | unicode | off
"render_mode": "auto",
# Size scalar relative to native 192×208 frames, shared by desktop canvas and
# CLI/TUI column width. Half-block fallback clamps to a legibility floor.
"scale": 0.33,
# Hard override for terminal column width; 0 = derive from scale.
"unicode_cols": 0,
},
},
# Web dashboard settings
"dashboard": {
# Visual theme: "default" | "midnight" | "ember" | "mono" | "cyberpunk" | "rose"
"theme": "default",
# Process-isolation rollout controls. Read via the raw config loader, so
# tui_gateway.server also owns explicit defaults.
"turn_isolation": False,
"compute_host_heartbeat_secs": 15,
"compute_host_respawn_max": 3,
# Token/cost analytics surfaces are hidden by default: the numbers are a local
# LOWER-BOUND estimate, not billing — only successful main-agent responses with a
# response.usage count; auxiliary calls, retries, fallbacks and cache writes are
# missed, so the total can be 10x-100x under the provider bill.
"show_token_analytics": False,
# IPs / bounded CIDRs of reverse proxies trusted to supply X-Forwarded-Proto/-For.
# Loopback always trusted; wildcards and /0 rejected (spoofing guard).
"trusted_proxies": [],
# WebSocket keepalive (seconds), NON-loopback binds only: loopback always disables
# the protocol ping so an event-loop stall never kills a healthy local connection.
"ws_ping_interval": 20.0,
"ws_ping_timeout": 20.0,
# Grace (seconds) before a WS-orphaned gateway session is interrupted/reaped after
# its client disconnects. 0 = park forever. Env: HERMES_TUI_WS_ORPHAN_REAP_GRACE_S.
"ws_orphan_reap_grace_s": 20.0,
# A detached RUNNING turn is only interrupted once its activity clock (API waits,
# stream tokens, tool heartbeats) has been idle this many seconds; an active turn
# runs to completion. Default = agent.turn_liveness.timeout_s. 0 = interrupt at grace.
"ws_orphan_activity_stale_s": 600.0,
# On gateway boot, close tui/desktop/subagent rows orphaned by a dead gateway
# (start AND newest message older than HERMES_TUI_SESSION_TTL_S, default 6h) with
# end_reason='startup_orphan_reap'; otherwise they stay phantom "active" forever.
# Messaging-gateway and live sessions are never touched; swept rows stay resumable.
"startup_orphan_sweep": True,
# OAuth gate (engaged when --host is set and --insecure is not), read by the Nous
# Portal plugin. Env HERMES_DASHBOARD_OAUTH_CLIENT_ID / HERMES_DASHBOARD_PORTAL_URL
# win when non-empty. Empty client_id = no provider; empty portal_url = production.
"oauth": {
"client_id": "", # agent:{instance_id} — Portal provisions this
"portal_url": "",
},
# Username/password gate (dashboard_auth/basic plugin, no OAuth IDP). Active when
# username plus password_hash (preferred) or password (hashed in-memory) are set;
# empty username = no-op. Env HERMES_DASHBOARD_BASIC_AUTH_USERNAME / _PASSWORD_HASH
# / _PASSWORD / _SECRET / _TTL_SECONDS win when non-empty. secret signs session
# tokens; empty = random per-process key (sessions die on restart, no multi-worker)
# — set 32+ random bytes. Hash: plugins.dashboard_auth.basic.hash_password('PW').
"basic_auth": {
"username": "",
"password_hash": "", # scrypt$...
"password": "",
"secret": "",
"session_ttl_seconds": 0, # 0 → plugin default (12h)
},
# Drain-control token auth (dashboard_auth/drain plugin). The secret is NOT here:
# env HERMES_DASHBOARD_DRAIN_SECRET; no-op unless >=256-bit, weak secrets rejected
# (fail-closed). scope = capability label; min_secret_chars in url-safe-b64 chars.
"drain_auth": {"scope": "drain", "min_secret_chars": 43},
# Public URL (env HERMES_DASHBOARD_PUBLIC_URL): full authority (scheme + host +
# optional prefix, e.g. https://example.com/hermes) for the OAuth redirect_uri;
# its hostname is trusted by Host/Origin guards and engages the auth gate when
# non-loopback. For proxies that don't forward X-Forwarded-Host/-Proto/-Prefix;
# X-Forwarded-Prefix is then IGNORED on the OAuth path. Empty or malformed (no
# http(s):// + host, or quote/angle/whitespace chars) = reconstruct from headers.
"public_url": "",
},
# Privacy settings
"privacy": {
"redact_pii": False, # hash user IDs and strip phone numbers from LLM context
},
# Text-to-speech. Each provider accepts an optional `max_text_length:` override for
# the per-request input-character cap; omit to use the provider's documented limit
# (OpenAI 4096, xAI 15000, MiniMax 10000, ElevenLabs 5k-40k model-aware, Gemini
# 32000, Edge 5000, Mistral 4000, NeuTTS/KittenTTS 2000).
"tts": {
# "edge" (free) | "elevenlabs" (premium) | "openai" | "xai" | "minimax" | "mistral"
# | "gemini" | "deepinfra" | "neutts" (local) | "kittentts" (local) | "piper" (local)
"provider": "edge",
"edge": {
# Popular: AriaNeural, JennyNeural, AndrewNeural, BrianNeural, SoniaNeural
"voice": "en-US-AriaNeural",
},
"elevenlabs": {
"voice_id": "pNInz6obpgDQGcFmaJgB", # Adam
"model_id": "eleven_multilingual_v2",
},
"openai": {
"model": "gpt-4o-mini-tts",
# gpt-4o-mini-tts voices: alloy, ash, ballad, cedar, coral, echo, fable,
# marin, nova, onyx, sage, shimmer, verse
"voice": "alloy",
},
"gemini": {
"model": "gemini-2.5-flash-preview-tts",
"voice": "Kore",
# Gemini 3.1: aux-model rewrite inserts [audio tags] into the TTS script only.
"audio_tags": False,
# Optional local text file with performance direction; may include a
# `{transcript}` placeholder, else the live transcript is appended.
"persona_prompt_file": "",
},
"xai": {
"voice_id": "eve", # or a custom voice ID (docs.x.ai custom voices)
"language": "en", # BCP-47 code ("en", "pt-BR") or "auto"
"speed": 1.0, # 0.7–1.5
"auto_speech_tags": False, # insert expressive audio tags via LLM rewrite
"optimize_streaming_latency": 0, # 0–2, trades quality for lower latency
"sample_rate": 24000, # 22050 / 24000 / 44100 / 48000
"bit_rate": 128000, # MP3 bitrate; only applies when codec=mp3
},
"mistral": {
"model": "voxtral-mini-tts-2603",
"voice_id": "c69964a6-ab8b-4f8a-9465-ec0925096ec8", # Paul - Neutral
},
"minimax": {"model": "speech-02-hd", "voice_id": "English_expressive_narrator"},
"kittentts": {
"model": "KittenML/kitten-tts-nano-0.8-int8", # nano 25MB; micro 41MB; mini 80MB
"voice": "Jasper",
},
"neutts": {
"ref_audio": "", # path to reference voice audio (empty = bundled default)
"ref_text": "", # path to reference voice transcript (empty = bundled default)
"model": "neuphonic/neutts-air-q4-gguf", # HuggingFace model repo
"device": "cpu", # cpu, cuda, or mps
},
"piper": {
# Voice name (downloaded on first use) or absolute path to a .onnx file; list:
# github.com/OHF-Voice/piper1-gpl/blob/main/docs/VOICES.md. Optional keys:
# voices_dir (~/.hermes/cache/piper-voices/), use_cuda, length_scale (2.0 =
# twice as slow), noise_scale, noise_w_scale, volume, normalize_audio.
"voice": "en_US-lessac-medium",
},
"deepinfra": {
"model": "", # empty = first tts-tagged model from the live catalog
"voice": "default",
# optional "base_url" key overrides DEEPINFRA_BASE_URL for TTS only
},
},
"stt": {
"enabled": True,
# Echo the raw transcript of gateway voice messages back as a 🎙️ message.
"echo_transcripts": True,
# No seeded "provider": a stored value counts as an explicit user pick; unset =
# autodetect ladder. Valid: "local" (faster-whisper) | "groq" | "openai" |
# "mistral" | "elevenlabs" | "deepinfra".
# Global language hint unless a per-provider language overrides it. "en" because
# Whisper auto-detect misreads short/accented clips; "" = auto; or "es", "zh", ...
"language": "en",
# Client-side ffmpeg silence trim before cloud upload (local whisper uses VAD):
# silence inflates upload time, billing and hallucinations. Failure = raw upload.
"cloud_trim_silence": True,
"cloud_trim_threshold_db": -40, # quieter than this counts as silence
"cloud_trim_keep_ms": 300, # how much of each pause survives (natural pacing)
"local": {
"model": "base", # tiny, base, small, medium, large-v3
"language": "", # auto-detect; set "en", "es", ... to force
"initial_prompt": "",
# Anti-hallucination (faster-whisper decodes junk from silence). vad: Silero
# filter (false = raw audio, for music/ambient). A segment is dropped only if
# no_speech_prob ABOVE no_speech_prob_threshold AND avg_logprob BELOW logprob_threshold.
"vad": True,
"vad_min_silence_ms": 500, # min silence (ms) that splits speech chunks
"no_speech_prob_threshold": 0.6,
"logprob_threshold": -1.0,
"unload_after_idle_seconds": 0, # 0 = never; e.g. 300 frees the model after 5min
},
"groq": {
# whisper-large-v3, whisper-large-v3-turbo, distil-whisper-large-v3-en
"model": "whisper-large-v3-turbo",
"language": "", # auto-detect; set "en", "es", ... to force
},
"openai": {
# whisper-1, gpt-4o-mini-transcribe, gpt-4o-transcribe, gpt-transcribe
"model": "whisper-1",
"language": "", # auto-detect; set "en", "es", ... to force
},
"mistral": {
"model": "voxtral-mini-latest", # voxtral-mini-latest, voxtral-mini-2602
"language": "", # auto-detect; set "en", "es", ... to force
},
"xai": {
"language": "", # auto-detect; set "en", "es", ... to force
},
"elevenlabs": {
"model_id": "scribe_v2", # scribe_v2, scribe_v1
"language_code": "", # auto-detect; set "eng", "spa", ... to force
"tag_audio_events": False,
"diarize": False,
},
"deepinfra": {
"model": "", # empty = first stt-tagged model from the live catalog
# optional "base_url" key overrides DEEPINFRA_BASE_URL for STT only
},
},
"voice": {
"record_key": "ctrl+b",
"submit_mode": "direct", # TUI: direct submits immediately; draft = editable transcript
"max_recording_seconds": 120,
"auto_tts": False,
# Desktop remote clients call STT/TTS providers DIRECTLY (config + key fetched
# over authenticated REST at session start) instead of relaying via the gateway.
"client_direct": True,
"beep_enabled": True, # record start/stop beeps in CLI voice mode
"beep_volume": 0.3, # beep amplitude multiplier, 0.0-1.0
"thinking_sound": True, # ambient bubble sound while the agent works (volume = beep_volume)
"silence_threshold": 200, # RMS below this = silence (0-32767)
"silence_duration": 3.0, # seconds of silence before auto-stop
"barge_in": True, # interrupt the agent / stop TTS when the user starts talking
# Trip suppression after TTS onset (mic stays live the whole turn).
"barge_in_grace_seconds": 0.5,
# Speech trigger = quiet-room floor x this (floor calibrated BEFORE playback).
"barge_in_threshold_multiplier": 3.0,
# Saying EXACTLY one of these (case-insensitive, punctuation ignored) ends the
# voice chat instead of going to the agent. [] disables.
"stop_phrases": ["stop"],
},
# "Hey Hermes" hands-free wake word: always-on, on-device hotword detection that
# starts a fresh voice session. Off by default; toggle with /wake.
"wake_word": {
"enabled": False,
"surface": "auto", # eligible surface: "auto" (first claimant) | "cli" | "tui" | "gui"
"input_device": None, # PortAudio input device index/name; null = process default
"capture": "auto", # auto | local | client (desktop streams mic via wake.feed)
# "openwakeword" (free, local) | "sherpa" (free, ANY phrase, no training) |
# "porcupine" (premium; needs PORCUPINE_ACCESS_KEY)
"provider": "openwakeword",
# sherpa: this IS the detected phrase; other engines: cosmetic label (detection
# is keyed by the model/keyword below)
"phrase": "hey hermes",
"sensitivity": 0.6, # 0.0-1.0 threshold, consistent across engines (higher = stricter)
# openWakeWord only: consecutive over-threshold frames to fire (higher = fewer
# false triggers, more latency; 1 = single-frame)
"confirmation_frames": 3,
"start_new_session": True, # fresh session on wake vs. continue the current one
# sherpa only: listen for every wake-enabled profile's phrase and route to it
"profile_routing": True,
"openwakeword": {
# "hey_hermes" | built-in openWakeWord name ("hey_jarvis", "alexa", ...) | path
# to a custom .onnx/.tflite model
"model": "hey_hermes",
# "" (auto: tflite on macOS ARM64, onnx elsewhere) | "onnx" | "tflite" — onnx
# scores near-zero on macOS ARM64 (arms but never fires)
"inference_framework": "",
},
"sherpa": {
# sherpa-onnx KWS model dir; empty = auto-download the small English zipformer
"model_dir": "",
},
"porcupine": {
# built-in keyword ("jarvis", "computer", ...) or path to a custom .ppn
"keyword": "jarvis",
},
},
"human_delay": {"mode": "off", "min_ms": 800, "max_ms": 2500},
# Context engine — how the context window is managed near the token limit.
# "compressor" = built-in lossy summarization; or a plugin name (e.g. "lcm") installed
# in plugins/context_engine/<name>/ or ~/.hermes/plugins/.
"context": {
"engine": "compressor",
# Return freed glibc pages at agent/TUI cleanup boundaries (no-op elsewhere).
"memory_trim": {
"enabled": True,
"cooldown_seconds": 60.0,
# INFO-log every Nth periodic trim; force paths always log.
"log_every_n": 1,
# Suppress INFO logs when the readable RSS delta is smaller; 0 = log all.
"info_log_min_delta_mb": 0.0,
},
},
# Persistent memory — bounded curated memory injected into the system prompt
"memory": {
"memory_enabled": True,
"user_profile_enabled": True,
# Approval gate for memory writes on BOTH foreground turns and the background
# review fork. true = foreground writes prompt inline; background writes are staged
# (/memory pending|approve <id>|reject <id>). To disable memory: memory_enabled.
"write_approval": False,
"memory_char_limit": 2200, # ~800 tokens at 2.75 chars/token
"user_char_limit": 1375, # ~500 tokens at 2.75 chars/token
# Periodic built-in memory review; 0 when an external provider auto-extracts.
"nudge_interval": 10,
# External memory provider plugin (empty = built-in only); only ONE at a time:
# "openviking", "mem0", "hindsight", "holographic", "retaindb", "byterover".
"provider": "",
},
# Subagent delegation — override the provider:model used by delegate_task so children
# run on a cheaper/faster model. Uses the same runtime provider resolution as
# CLI/gateway startup, so every configured provider is supported.
"delegation": {
"model": "", # e.g. "google/gemini-3-flash-preview" (empty = inherit parent)
"provider": "", # e.g. "openrouter" (empty = inherit parent provider + credentials)
"base_url": "", # direct OpenAI-compatible endpoint for subagents
"api_key": "", # key for delegation.base_url (falls back to OPENAI_API_KEY)
# Wire protocol for delegation.base_url: "chat_completions" | "codex_responses" |
# "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix); set
# explicitly for non-standard endpoints.
"api_mode": "",
# Per-child request settings on every delegation call (all resolution branches).
# Top-level keys = API kwargs (e.g. service_tier); "extra_body" sub-dict merges
# into extra_body, e.g. {"extra_body": {"provider": {"sort": "throughput"}}}.
# Explicit values win OVER runtime/parent overrides (extra_body deep-merged 1 level).
"request_overrides": {},
# When delegate_task narrows child toolsets, keep the parent's enabled MCP
# toolsets (so toolsets=["web"] doesn't strip MCP). false = strict intersection.
"inherit_mcp_toolsets": True,
# Per-subagent iteration cap (own budget, independent of the parent's).
"max_iterations": 250,
# Hard per-summary char ceiling on subagent results, layered on the dynamic budget
# (each summary is sized to the parent's remaining context headroom; trimmed text
# spills to ~/.hermes/cache/delegation/ with a head+tail window + read_file offset
# footer, nothing lost). 0 disables the ceiling; the dynamic budget still applies.
"max_summary_chars": 24000,
# Wall-clock cap per child (seconds, floor 30). 0 = no timeout: children fail
# only from real errors (API, tools, iteration budget).
"child_timeout_seconds": 0,
# Subagent effort: "ultra" | "max" | "xhigh" | "high" | "medium" | "low" |
# "minimal" | "none" (empty = inherit)
"reasoning_effort": "",
# Max parallel children per batch AND max concurrent background delegation units;
# async dispatches beyond it run synchronously. Floor 1, no ceiling.
"max_concurrent_children": 10,
# Orchestrator role controls. Depth floored at 1, no ceiling; each level multiplies cost.
"max_spawn_depth": 1, # 1 = flat, 2 = orchestrator→leaf, 3+ = deeper
"orchestrator_enabled": True, # kill switch for role="orchestrator"
# Subagent threads ALWAYS resolve approvals non-interactively (the parent TUI owns
# stdin; input() from a worker would deadlock). false = auto-deny, true =
# auto-approve "once"; both log a warning audit line. true only for trusted batch work.
"subagent_auto_approve": False,
# Subagent background processes (task_id "sa-...") route notify_on_complete /
# watch_pattern notifications to the PARENT; false suppresses them (the child's
# result is the deliverable). Async-delegation results are NEVER suppressed.
"surface_child_process_notifications": False,
},
# Ephemeral prefill messages file — JSON list of {role, content} dicts injected at the
# start of every API call for few-shot priming. Never saved to sessions/logs/trajectories.
"prefill_messages_file": "",
# Goals — persistent cross-turn /goal loop: after each turn an aux-model judge checks
# if the goal is satisfied, else a continuation prompt re-enters the session until
# done, budget exhausted, or paused. Judge failures fail OPEN; the budget is the backstop.
"goals": {
# Max continuation turns before auto-pause (/goal resume) — guards against judge
# false negatives and unbounded spend.
"max_turns": 20,
},
# Loops — /loop re-runs a prompt or slash command on a cadence in-session. Fixed
# interval fires on the user's clock; self-paced (no interval) starts at the floor and
# backs off exponentially while replies stop changing.
"loops": {
"min_interval_seconds": 30, # smallest fixed interval; tighter cadences raised to it
"max_ticks": 100, # auto-pause after this many wakeups unless --times set; 0 = unlimited
# Self-paced cadence bounds (seconds).
"self_paced_floor_seconds": 60,
"self_paced_ceiling_seconds": 900,
},
# Mixture of Agents — named presets used by /moa. A preset is an execution mode around
# the main model, not a model itself: references + aggregator synthesize private
# guidance before each main-model iteration.
"moa": {
"default_preset": "default",
"active_preset": "",
# Write each MoA turn (reference + aggregator exact input/output/usage) as JSONL
# to <hermes_home>/moa-traces/<session_id>.jsonl (or trace_dir) for auditing.
"save_traces": False,
"trace_dir": "",
# PII/credential redaction of advisor outputs: "" off | "display" (UI reference
# blocks + traces only; aggregator sees raw) | "full" (also the aggregator prompt).
"privacy_filter": "",
"presets": {
"default": {
"reference_models": [
{"provider": "openai-codex", "model": "gpt-5.5"},
{"provider": "openrouter", "model": "deepseek/deepseek-v4-pro"},
],
"aggregator": {"provider": "openrouter", "model": "anthropic/claude-opus-4.8"},
"max_tokens": 4096,
"enabled": True,
}
},
},
# Skills — external skill directories shared across tools/agents. Paths are expanded
# (~, ${VAR}) and resolved; read-only — creation goes to ~/.hermes/skills/ unless
# create_dir redirects it.
"skills": {
"external_dirs": [], # e.g. ["~/.agents/skills", "/shared/team-skills"]
# Where skill_manage-created skills go (empty = profile-local dir). When set, new
# skills land here AND agent-facing instructions name this path; expanded (~,
# ${VAR}), relative to HERMES_HOME, scanned alongside the local dir.
"create_dir": "",
# In a git checkout, <root>/.hermes/skills/ and <root>/.agents/skills/ load as the
# highest-precedence tier — ONLY if the root is in trusted_project_dirs. false =
# no scan, no untrusted-skills notice.
"project_discovery": True,
# Trusted project roots; managed by `hermes skills trust` / `untrust`.
"trusted_project_dirs": [],
# Substitute ${HERMES_SKILL_DIR} / ${HERMES_SESSION_ID} in SKILL.md content.
"template_vars": True,
# Pre-execute !`cmd` snippets in SKILL.md, inlining stdout (dates, git state...).
# Off: skill-author content would run on the host unapproved — trusted sources only.
"inline_shell": False,
"inline_shell_timeout": 10, # seconds per !`cmd` snippet
# Security-scan skills the agent writes via skill_manage. Off: the agent can run the
# same code via terminal() ungated, so it mostly blocks prose with risky keywords.
# On: a dangerous verdict is a tool error the agent can retry. Hub installs are
# always scanned.
"guard_agent_created": False,
# Advisory NVIDIA SkillEvaluator Tier 1 scan on `hermes skills install` (alongside
# the enforcing built-in guard), only if `skillevaluator` is on PATH
# (uv tool install "skillevaluator @ git+https://github.com/NVIDIA/SkillEvaluator.git").
# Informational, never blocking; secrets-class findings shown red. No-op without it.
"tier1_advisory": True,
# Approval gate for skill_manage mutations on BOTH foreground turns and the
# background review fork. true = ALWAYS stage (SKILL.md too large for an inline
# prompt): /skills pending, /skills diff <id>, /skills approve|reject <id>.
"write_approval": False,
# Audit ledger: every skill mutation appends to ~/.hermes/skills/.curator_ledger.jsonl
# with before/after hashes (blobs under ~/.hermes/.curator_backups/blobs/); powers
# `hermes curator ledger` / `rollback <entry-id>`. Never a gate — failures can't block.
"ledger": True,
},
# Curator — background maintenance of AGENT-CREATED skills (never hub-installed):
# marks long-unused skills stale, archives (never deletes) obsolete ones, optionally
# consolidates overlaps via a forked aux-model agent. Inactivity-triggered from session
# start, no cron daemon. `hermes curator status` shows the last run.
"curator": {
"enabled": True,
"interval_hours": 24 * 7, # hours between runs
"min_idle_hours": 2, # only run after the agent has been idle this long
"stale_after_days": 30, # mark "stale" after this many unused days
"archive_after_days": 90, # move to skills/.archive/ (recoverable) after this many
# LLM consolidation (umbrella-building) pass. OFF = deterministic inactivity prune
# only, no aux-model cost. `hermes curator run --consolidate` overrides once.
"consolidate": False,
# Also prune bundled built-ins (a suppression list stops `hermes update` restoring
# them); hub-installed skills are NEVER pruned. A built-in's clock starts when the
# curator first sees it, so never a mass-prune on the first run. false = keep all.
"prune_builtins": True,
# TTL purge of skills/.archive/: 0 = never; > 0 lets the explicit `hermes curator
# purge` delete older archived skills (never automatic; logged in the ledger).
"archive_ttl_days": 0,
# Before every real (non-dry-run) pass, snapshot ~/.hermes/skills/ to
# ~/.hermes/skills/.curator_backups/<utc-iso>/skills.tar.gz (`hermes curator rollback`).
"backup": {
"enabled": True,
"keep": 5, # retain last N regular snapshots
},
},
# Honcho AI-native memory — ~/.honcho/config.json is the source of truth (apiKey,
# workspace, peerName, sessions, enabled); hermes-specific overrides only here.
"honcho": {},
# IANA timezone (e.g. "Asia/Kolkata", "America/New_York"). Empty = server-local time.
"timezone": "",
# Slack platform settings (gateway mode)
"slack": {
"require_mention": True, # require @mention to respond in channels
"free_response_channels": "", # comma-separated channel IDs answered without mention
"allowed_channels": "", # if set, ONLY respond in these channel IDs (whitelist)
"require_mention_channels": "", # channel IDs where @mention is ALWAYS required
# Ignore messages whose first token @mentions another user unless the bot is also
# mentioned. Env: SLACK_IGNORE_OTHER_USER_MENTIONS.
"ignore_other_user_mentions": False,
"thread_require_mention": False, # require @mention in thread replies too
"channel_prompts": {}, # per-channel ephemeral system prompts
},
# Discord platform settings (gateway mode)
"discord": {
"require_mention": True, # require @mention to respond in server channels
"free_response_channels": "", # comma-separated channel IDs answered without mention
"allowed_channels": "", # if set, ONLY respond in these channel IDs (whitelist)
"auto_thread": True, # auto-create threads on @mention in channels (like Slack)
"thread_require_mention": False, # require @mention in threads too (multi-bot threads)
# Multi-bot rooms: another bot must type @thisbot (a reply/quote alone won't) to
# trigger a reply — stops two bots replying to each other forever. Humans unaffected.
"bots_require_inline_mention": False,
# Prepend recent channel scrollback when triggered (recovers messages gated out by
# require_mention); limit = max messages scanned.
"history_backfill": True,
"history_backfill_limit": 50,
# Replay messages missed while offline, after reconnect/startup.
"missed_message_backfill": {
"enabled": False,
"channels": "", # comma-separated channel IDs; empty uses free_response_channels
"window_seconds": 21600, # only inspect messages from the last 6 hours
"limit": 100, # global cap on messages scanned per reconnect
"max_dispatches": 10, # cap on recovered messages dispatched per reconnect
},
"reactions": True, # add 👀/✅/❌ reactions to messages during processing
# Gateway transport health probe: inspects the WebSocket's ready/open/heartbeat
# state (never REST) as proof events still arrive. Any value 0 disables it.
"websocket_liveness_interval_seconds": 15,
"websocket_liveness_failure_threshold": 2,
"websocket_heartbeat_ack_max_age_seconds": 60,
"websocket_max_latency_seconds": 30,
# per-channel ephemeral system prompts (forum parents apply to child threads)
"channel_prompts": {},
# Opt-in DM role auth: DISCORD_ALLOWED_ROLES normally authorizes guild messages
# only (DMs need DISCORD_ALLOWED_USERS). A guild ID here also authorizes DMs from
# that guild's members holding the allowed role. Unset / "" / 0 = off.
"dm_role_auth_guild": "",
# discord / discord_admin tools: allowed actions (comma string or YAML list; empty
# = all, subject to bot intents; unknown names dropped with a warning): list_guilds,
# server_info, list_channels, channel_info, list_roles, member_info, search_members,
# fetch_messages, list_pins, pin_message, unpin_message, create_thread, add_role,
# remove_role.
"server_actions": "",
# DEPRECATED no-op (uploads are always cached; messaging auth is the gate). Kept so
# existing configs don't error. Env: DISCORD_ALLOW_ANY_ATTACHMENT.
"allow_any_attachment": False,
# Max bytes per cached attachment (held in memory while written); 0 = no cap.
# Env: DISCORD_MAX_ATTACHMENT_BYTES.
"max_attachment_bytes": 33554432,
# Mention allowed users on approval prompts so owners notice them in shared
# channels. Env: DISCORD_APPROVAL_MENTIONS.
"approval_mentions": False,
# Voice-channel inactivity timeout (seconds); 0 = stay until `/voice leave`.
"voice_channel_inactivity_timeout_seconds": 300,
# Minimum seconds before force-stopping a VC playback; the adapter probes clip
# duration and extends this floor so long TTS isn't cut off.
"voice_playback_timeout_seconds": 120,
# Voice-channel software mixer (plugins/platforms/discord/voice_mixer.py): ambient
# "thinking" bed, verbal acks and TTS OVERLAP (ambient ducked) vs stop-and-swap.
"voice_fx": {
"enabled": False, # master switch for the mixer subsystem
"ambient_enabled": True, # play the idle "thinking" bed while tools run
"ambient_path": "", # custom loop audio file; "" = synthesised pad
"ambient_gain": 0.18, # idle bed loudness, 0.0–1.0
"duck_gain": 0.06, # ambient loudness while speech plays
"speech_gain": 1.0, # TTS / ack loudness, 0.0–1.0
"ack_enabled": True, # speak a short phrase before the first tool call
"ack_phrases": [ # picked at random; [] disables phrases
"Let me look into that.",
"One moment.",
"Checking on that now.",
"Give me a sec.",
"On it.",
],
},
},
# WhatsApp platform settings (gateway mode)
"whatsapp": {
# reply_prefix: None = built-in "⚕ *Hermes Agent*" header; "" disables; \n allowed.
},
# Telegram platform settings (gateway mode)
"telegram": {
"reactions": False, # add 👀/✅/❌ reactions to messages during processing
# per-chat/topic ephemeral system prompts (topics inherit from parent group)
"channel_prompts": {},
"allowed_chats": "", # if set, ONLY respond in these group/supergroup chat IDs
"extra": {
# Bot API 10.1 native rich messages (tables/task lists/math). Off = legacy
# MarkdownV2, since rich messages are hard to copy as plain text.
"rich_messages": False,
# Experimental rich draft previews while streaming DMs; off because Telegram
# Desktop/macOS can overlay draft frames until the chat redraws.
"rich_drafts": False,
},
},
# Mattermost platform settings (gateway mode)
"mattermost": {
"require_mention": True, # require @mention to respond in channels
"free_response_channels": "", # comma-separated channel IDs answered without mention
"allowed_channels": "", # if set, ONLY respond in these channel IDs (whitelist)
"channel_prompts": {}, # per-channel ephemeral system prompts
},
# Matrix platform settings (gateway mode)
"matrix": {
"require_mention": True, # require @mention to respond in rooms
"free_response_rooms": "", # comma-separated room IDs answered without mention
"allowed_rooms": "", # if set, ONLY respond in these room IDs (whitelist)
},
# Approvals for dangerous commands.
# mode: manual (always prompt) | smart (aux LLM auto-approves low-risk) | off (= --yolo)
# cron_mode / single_query_mode / unattended_mode: deny | approve — what to do when a
# cron job, a -q session (HERMES_INTERACTIVE=1 but nobody to answer), or an unattended
# platform (webhook, msgraph_webhook, api_server; no /approve channel) hits one.
# deny blocks instantly so the agent finds another way instead of waiting out the
# timeout and failing closed.
# timeout: seconds before an unanswered prompt fails closed (CLI and gateway). 60s
# proved too tight for Telegram/Discord push notifications, hence 300.
"approvals": {
"mode": "smart",
"timeout": 300,
"cron_mode": "deny",
"single_query_mode": "deny",
"unattended_mode": "deny",
# Extra rules appended to the smart-approval guardian's SYSTEM prompt, e.g.
# "Always ESCALATE commands touching /etc".
"smart_policy": "",
# After this many consecutive guardian DENYs in a session, the deny message
# escalates to a hard-stop (report to user / ask for /approve). Approval resets; 0 off.
"denial_breaker_threshold": 3,
# Case-insensitive fnmatch globs against terminal commands; a match blocks even
# under --yolo / mode=off. Quote in YAML when starting with * or containing {}/!/:
# e.g. "git push --force*".
"deny": [],
# /reload-mcp confirms before rebuilding the MCP tool set (it invalidates the
# prompt cache, so the next message re-sends full input). "Always Approve" → false.
"mcp_reload_confirm": True,
# /clear, /new, /reset, /undo confirm before discarding state (Approve Once /
# Always Approve / Cancel via tools.slash_confirm; native buttons on Telegram/
# Discord/Slack). "Always Approve" → false. HERMES_TUI_NO_CONFIRM=1 skips the TUI modal.
"destructive_slash_confirm": True,
},
# Permanently allowed dangerous command patterns (added via "always" approval).
"command_allowlist": [],
# User-defined quick commands that bypass the agent loop (type: exec only).
"quick_commands": {},
# Per-platform system-prompt hint overrides, keyed by platform name (whatsapp, slack,
# telegram, ...). Value: {"append": text} keeps the built-in hint and appends;
# {"replace": text} substitutes it; a bare string is shorthand for append.
# `replace` wins over `append` if both are given.
"platform_hints": {},
# Plugin system. `enabled`/`disabled` lists are written by `hermes plugins enable|disable`
# and deliberately omitted here so an empty default never clobbers a user allow-list.
"plugins": {
# Wall-clock cap (seconds) for one in-process Python plugin hook callback; shell hooks
# keep their own per-entry `timeout`. 0 = no cap (sync call on agent thread). Max 600.
"hook_callback_timeout": 30,
},
# Shell-script hooks: event name (pre_tool_call, post_tool_call, pre_llm_call,
# subagent_stop, ...) -> list of {matcher, command, timeout}. First run of a new command
# prompts for consent; approvals persist in ~/.hermes/shell-hooks-allowlist.json.
# Schema + examples: website/docs/user-guide/features/hooks.md.
"hooks": {},
# Auto-accept shell-hook registrations without a TTY prompt (also --accept-hooks or
# HERMES_ACCEPT_HOOKS=1). Gateway/cron/non-interactive runs need one of these to pick
# up newly-added hooks.
"hooks_auto_accept": False,
# Custom personalities: {"name": "system prompt"} or
# {"name": {"description", "system_prompt", "tone", "style"}}.
"personalities": {},
# Security: pre-exec scanning via tirith plus related guards.
"security": {
"allow_private_urls": False, # allow requests to private/internal IPs (OpenWrt, VPNs)
"redact_secrets": True,
# Persisted acknowledgement for unattended model overrides whose tier lets the vendor
# train on prompts. The startup guard still warns every run; cost guards are unaffected.
"allow_data_training_tiers_noninteractive": False,
# Human-approval presentation transport. "builtin" = CLI/TUI/gateway/ACP surfaces; a
# plugin transport is used only when named explicitly. Transport timeout/error/invalid
# response DENIES unless transport_fallback is "builtin". Presentation only: plugins
# cannot detect, suppress, or auto-approve commands outside a correlated human response.
"approval": {"transport": "builtin", "transport_fallback": "deny"},
# Writes to agent-instruction files (AGENTS.md/CLAUDE.md/SOUL.md/.cursorrules,
# project-local .hermes config) always need human approval, even under yolo.
# Extra patterns are fnmatch globs on the basename (e.g. "*.mdc").
"protected_instruction_files": True,
"protected_instruction_extra_patterns": [],
"tirith_enabled": True,
"tirith_path": "tirith",
"tirith_timeout": 5,
"tirith_fail_open": True,
"website_blocklist": {"enabled": False, "domains": [], "shared_files": []},
# IDs of supply-chain advisories the user has read and acted on; acked ones stop the
# startup banner. Add via `hermes doctor --ack <id>`; remove by editing the list.
# Catalog: hermes_cli/security_advisories.py.
"acked_advisories": [],
# Lazy-install opt-in backend packages from PyPI when a backend that needs them is first
# enabled (e.g. `elevenlabs`). False = require explicit pip install for everything beyond
# the base set (restricted/audited/air-gapped environments).
"allow_lazy_installs": True,
},
"cron": {
# Let cron-spawned agents use the cronjob toolset (the "cron-librarian" pattern). Off by
# default: policy-denied in cron context to prevent unattended scheduling loops. Jobs
# created this way are user-owned in the same flat jobs table. Interactive toolsets
# (messaging/clarify) stay denied in cron regardless.
"allow_agent_scheduling": False,
# Pre-dispatch validation: before building any agent machinery, verify the provider API
# key resolves (unless a fallback chain exists), attached skills are ready, and delivery
# platforms are configured. Failure -> last_status=blocked_config, ONE alert, no LLM call.
# False = fail during the run instead.
"preflight": True,
# Fail closed when an unpinned job's current global model/provider differs from its
# creation-time snapshot, so unattended jobs never silently inherit a paid default.
# False only when jobs should track changing global inference defaults.
"model_drift_guard": True,
# Default model for cron jobs (WHAT model runs). Fire-time resolution: per-job pin >
# cron.model > model.default. When set, unpinned jobs follow it deliberately and the
# drift guard does not engage for the model axis. "" = fall through to model.default.
"model": "",
# Inference provider paired with cron.model (NOT the scheduler provider below).
# "" = resolve from global config.
"model_provider": "",
# Cron SCHEDULER provider (WHEN a due job fires). "" = built-in in-process 60s ticker.
# Name an installed provider (plugins/cron_providers/<name>/ or $HERMES_HOME/plugins/
# <name>/), e.g. "chronos" (NAS-mediated managed cron for scale-to-zero). An unknown or
# unavailable provider falls back to the built-in so cron never loses its trigger.
"provider": "",
# Chronos settings; consulted only when provider == "chronos". All non-secret — the
# agent holds NO scheduler credentials (provision reuses the Nous Portal token).
"chronos": {
# NAS/portal base URL that arms/cancels one-shots and mints the inbound fire JWT
# (used as the expected issuer).
"portal_url": "https://portal.nousresearch.com",
# This agent's publicly reachable base URL; NAS POSTs {callback_url}/api/cron/fire.
# "" -> Chronos unavailable, resolver falls back to the built-in ticker.
"callback_url": "",
# Expected JWT audience (e.g. "agent:{instance_id}").
"expected_audience": "",
# NAS JWKS URL for verifying the fire JWT signature. "" -> the fire endpoint refuses
# all tokens (never an unsigned decode).
"nas_jwks_url": "",
},
# Wrap delivered cron responses with a task-name header and "The agent cannot see this
# message" footer. False = clean output.
"wrap_response": True,
# Delivery behaviour for cron output sent through a live gateway adapter.
"delivery": {
# Mark cron deliveries FINAL so the platform pushes them (Telegram's "important" mode
# otherwise sends with disable_notification=True and briefs look undelivered).
# False = silent, no-push deliveries.
"notify": True,
},
# Make cron deliveries CONTINUABLE (user can reply to a brief with it in context). False
# keeps deliveries isolated to the job's session; per-job `attach_to_session` overrides.
# Thread-capable platforms (Telegram topics, Discord/Slack threads) get a seeded thread
# per job via create_handoff_thread; DM-only platforms mirror the brief into the origin
# DM session. Appended at a turn boundary via mirror_to_session, cached system prompt
# untouched; fan-out/broadcast targets are never mirrored.
"mirror_delivery": False,
# Max due jobs run in parallel per tick. None/0 = unbounded (thread count only);
# 1 = serial. Env override: HERMES_CRON_MAX_PARALLEL.
"max_parallel_jobs": None,
# save_job_output keeps the N most recent .md files per job; 0 or negative disables
# pruning (for externally managed cleanup).
"output_retention": 50,
# Timeout (seconds) for a no-agent cron script. Env: HERMES_CRON_SCRIPT_TIMEOUT.
# Keep in sync with cron.scheduler._DEFAULT_SCRIPT_TIMEOUT.
"script_timeout_seconds": 3600,
# Timeout (seconds) for SessionDB() init inside cron jobs: state.db open/migrate has no
# timeout of its own against a wedged sqlite3.connect, and an unbounded hang wedges the
# job's dispatch guard forever. Env: HERMES_CRON_SESSION_DB_TIMEOUT. 0 = unlimited.
"session_db_timeout_seconds": 10,
# Timeout (seconds) per media attachment send during gateway delivery; large attachments
# (long TTS audio, big exports) need more than 30s. Env: HERMES_CRON_MEDIA_SEND_TIMEOUT.
# Keep in sync with cron.scheduler._DEFAULT_MEDIA_SEND_TIMEOUT.
"media_send_timeout_seconds": 300,
},
# Kanban multi-agent coordination. The dispatcher ticks every N seconds, reclaims stale
# claims, promotes dependency-satisfied todos to ready, and fires
# `hermes -p <assignee> chat -q ...` per claimable task. Run ONE dispatcher per profile;
# two on the same kanban.db race for claims.
"kanban": {
# Auto-subscribe the originating gateway/TUI session to completion + block events when
# kanban_create is called from a session with a persistent delivery channel. Disable
# for profiles that prefer explicit kanban_notify-subscribe calls per task.
"auto_subscribe_on_create": True,
# Run the dispatcher inside the gateway process (~300µs per idle tick). False only if
# you run it as a separate unit or don't want the gateway spawning workers.
"dispatch_in_gateway": True,
# Auto-claim tasks in the review column and spawn the assigned profile with the bundled
# sdlc-review skill. Disable where every review is done manually from the dashboard.
"review_dispatch": True,
# Seconds between dispatcher ticks. Lower = snappier pickup; higher = less SQL pressure.
"dispatch_interval_seconds": 60,
# Auto-block after this many consecutive non-success attempts (spawn_failed, timed_out,
# crashed) for the same task/profile. Reassignment resets the streak.
"failure_limit": 2,
# Worker stdout/stderr log rotation at spawn time (2 MiB + one backup). Raise to keep
# more early failure evidence from long-running workers.
"worker_log_rotate_bytes": 2 * 1024 * 1024,
"worker_log_backup_count": 1,
# Profile for the root/orchestration task after Triage decomposition; "" = default
# profile. Does not control the decomposer LLM path (see auxiliary.kanban_decomposer).
"orchestrator_profile": "",
# Assignee when the orchestrator can't match one to an installed profile; "" = default
# profile. A task never ends up with assignee=None.
"default_assignee": "",
# Global cap: positive int = the HOST never has more than N tasks 'running' across all
# boards and both dispatch lanes. None = ~MemTotal / 512 MiB clamped to [2, 8]; where
# MemTotal is unreadable (macOS/Windows) None means no cap.
"max_in_progress": None,
# Per-profile cap: positive int = no single profile runs more than N workers even if the
# global caps allow; blocked tasks defer to the next tick. None = no per-profile cap.
# Useful when fan-out would saturate one profile's model/API quota/browser pool.
"max_in_progress_per_profile": None,
# Auto-run the decomposer on Triage tasks every tick. False = manual via
# `hermes kanban decompose <id>` or the dashboard's Decompose button.
"auto_decompose": True,
# Max triage tasks decomposed per tick, bounding the aux-LLM burst from a bulk load.
# Excess defers to the next tick.
"auto_decompose_per_tick": 3,
# Running tasks with no heartbeat (last_heartbeat_at) for this many seconds are reclaimed
# to ready on the next tick; a still-running local worker is terminated first. 0 = off.
"dispatch_stale_timeout_seconds": 14400,
# Each tick, requeue 'running' cards with broken claim bookkeeping (claim_lock or
# claim_expires NULL with a dead worker) that TTL/crash/stale recovery can't see.
# False keeps orphans frozen for manual forensics.
"reconcile_orphans": True,
# Notify subscriptions survive `done` (completion is reversible) and are removed on
# archive. On boards that never archive, the notifier GC purges subscriptions for tasks
# done with no activity for this many days so stale rows aren't scanned forever. 0 = off.
"done_sub_retention_days": 30,
},
# Bot Mode cross-connection relay (tools/bot_relay.py): envelopes queued by message_agent
# for agents on other connections wait in an on-disk outbox until the Desktop drains them.
"bot_mode": {
# Drain-time TTL (seconds): older envelopes are NOT delivered on drain; the sender gets
# an error reply (reason 'queued_expired') so a DM can't land hours late as a zombie.
# 0 = no drain-time expiry (the 6h stale-artifact sweep still applies).
"envelope_ttl_seconds": 900,
# How long a second delivery into a busy target profile queues behind the current turn
# before failing with a structured 'target_busy' error. Deliveries are serialized per
# profile with a cross-process file lock.
"turn_wait_seconds": 120,
},
# execute_code settings (programmatic tool calls).
"code_execution": {
# project = run in the session cwd with the active venv/conda python so project deps and
# relative paths resolve. strict = isolated temp dir with hermes-agent's own python
# (sys.executable): max isolation, project deps/relative paths won't work. Env scrubbing
# (*_API_KEY, *_TOKEN, *_SECRET, ...) and the tool whitelist apply in both modes.
"mode": "project",
# Session kernels are always on locally (`kernel_mode` is ignored; remote backends run
# per-call). One kernel per (session owner, mode, interpreter, cwd, tool-set) keeps state
# across calls and turns; subagents get their own. Kernels die with the session, after
# kernel_idle_timeout idle seconds, or by LRU eviction past max_session_kernels. A
# timed-out/interrupted cell kills the kernel; env is frozen at spawn (reset=true after
# changing passthrough). Tool RPC authority (approval, session, allow-list, call budget)
# is rebound per cell — that runtime boundary is the cross-cell enforcement.
"kernel_idle_timeout": 1800,
"max_session_kernels": 4,
},
# Tool Search: deferrable (MCP / non-core plugin) tools are replaced in the model-facing
# array by tool_search / tool_describe / tool_call bridges and surfaced on demand. Core
# Hermes tools (terminal, file tools, todo, memory, browser_*, ...) are NEVER deferred.
"tools": {
"tool_search": {
# Tiered: tier 0 (no deferrable tools) = everything eager; tier 1 = bridge + a
# name+description manifest when it fits the budget (degrades to names-only);
# tier 2 (over budget even names-only, e.g. ~3,300-tool APIs) = bare bridge + a
# one-line-per-server summary (name + tool count).
# "auto"|"on" = activate when at least one deferrable tool exists ("auto" is an
# alias of "on" today, reserved for a future budget-gated mode; keep it the default
# so explicit "on"/"off" pins are unaffected). "off" = pass-through, no bridge.
"enabled": "auto",
# Listing budget as % of the model's context length; effective budget =
# min(this % of context, listing_max_tokens). Range 0..100.
"threshold_pct": 5,
# Hits per query when the model omits `limit`. Range 1..max_search_limit.
"search_default_limit": 5,
# Hard upper bound the model may request via `limit` (per query). Range 1..50.
"max_search_limit": 25,
# Catalog listing embedded in the bridge description (name + first sentence ≤60
# chars, grouped by server/toolset). "auto" = include when it fits (falls back to
# names-only, then bare tier-2 bridge); "on" = same rendering, explicit intent;
# "off" = always the bare bridge.
"listing": "auto",
# Absolute cap on the embedded listing in tokens (chars/4), regardless of context
# size. Range 200..60000.
"listing_max_tokens": 4000,
},
},
# File logging to ~/.hermes/logs/: agent.log captures INFO+, errors.log WARNING+.
"logging": {
"level": "INFO", # minimum level for agent.log: DEBUG, INFO, WARNING
"max_size_mb": 5, # max size per log file before rotation
"backup_count": 3, # rotated backups to keep
},
# Remote model-catalog manifest: curated OpenRouter / Nous Portal model lists fetched from
# this URL (falls back to the in-repo snapshot on network failure), so picker lists update
# without a release. Default URL is served by the docs-site GitHub Pages deploy.
"model_catalog": {
"enabled": True,
"url": "https://hermes-agent.nousresearch.com/docs/api/model-catalog.json",
# Disk cache TTL in minutes. The gateway refreshes in the background on this cadence;
# the CLI refetches on the next /model or `hermes model` once the cache is older.
# Network failures silently use the stale cache. Legacy `ttl_hours` is honoured if set.
"ttl_minutes": 20,
# Per-provider override URLs for self-hosted curation lists using the same schema,
# e.g. providers: {openrouter: {url: https://example.com/my-curation.json}}.
"providers": {},
},
# Per-model metadata overrides. Fields: context_window, max_output_tokens, supports_tools,
# supports_vision, supports_reasoning, model_family. <provider>.<model_id> wins over
# models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in
# agent/model_metadata.py). <provider>._default and top-level _default fill gaps ONLY for
# models the catalog does not know, so they never clamp known models. Unknown ids start
# from safe defaults (200K context, tools on, vision/reasoning off) and get patched.
# Provider keys: Hermes or models.dev id; model ids match case-insensitively.
# Example: {"custom:my-local-vllm": {"my-llava-model": {"context_window": 8192}}}
"model_overrides": {},
# models.dev registry (context windows, capabilities, pricing, modalities): fetched on
# startup, served from cache, refreshed by a background daemon with ETag conditional GET.
# Override `url` to point at a mirror (e.g. behind a corporate proxy).
"models_dev": {
"url": "", # empty = default https://models.dev/api.json
},
# Network workarounds.
"network": {
# Force IPv4. With broken/unreachable IPv6, Python tries AAAA first and hangs for the
# full TCP timeout before falling back. True skips IPv6 entirely.
"force_ipv4": False,
},
# Gateway monitoring: service health + redacted operational diagnostics exported over OTLP
# to an operator endpoint (OTEL Collector, DataDog, ...). Content-free by construction: no
# prompts, messages, tool args/results, session history, usage analytics, audit logs, or
# trajectories. Nothing is collected or sent until enabled with an endpoint.
"monitoring": {
# Stable install identifier on exported signals so operators can tell instances apart.
# "" = mint a fresh UUID on first use; clear it to rotate. Carries no account identity.
"install_id": "",
# Gateway health & diagnostics export.
"gateway_health_export": {
"enabled": False,
"metrics_enabled": True,
"diagnostic_events_enabled": True,
"warning_error_events_enabled": True,
"export_interval_seconds": 60,
"logs_export_interval_seconds": 5,
"resource_attributes": {
"service.name": "hermes-gateway",
"deployment.environment.name": "production",
},
},
# OTLP destination. headers_env maps header names to ENVIRONMENT VARIABLE NAMES (never
# secret values); values are read from the environment at export time.
"export": {"otlp": {"enabled": False, "endpoint": "", "headers_env": {}}},
},
# Gateway settings (messaging platforms: Telegram, Discord, Slack, ...).
"gateway": {
# Named-profile allowlist for multiplex mode. None = serve all; [] = default only.
"multiplex_profile_allowlist": None,
# Seconds to let a SIGTERM-interrupted gateway agent unwind before adapter/database
# teardown. Keep short so service-manager shutdowns don't exhaust their stop budget.
"signal_interrupt_grace_timeout": 1,
# Durable delivery-obligation ledger: final responses are recorded in state.db around
# the platform send; a gateway that died between finalize and platform ACK redelivers
# on next boot (ambiguous cases carry a "recovered reply — may be a duplicate" marker;
# at-least-once). Disable to lose in-flight final responses on crash/restart.
"delivery_ledger": True,
# Seconds to wait for one platform to connect at startup/reconnect; raise on "discord
# connect timed out" loops (many slash commands to sync). 0/negative = wait forever.
# Bridged to HERMES_GATEWAY_PLATFORM_CONNECT_TIMEOUT, which wins if set explicitly.
"platform_connect_timeout": 30,
# Event-loop liveness watchdog: a daemon thread probes the asyncio loop; after
# consecutive missed probes it dumps all-thread stacks and hard-exits with the
# service-restart code so systemd/launchd revives the process instead of leaving a
# wedged-but-alive zombie.
"loop_watchdog": True,
# Watchdog tuning (defaults mirror gateway/shutdown_watchdog.py): probe_interval =
# seconds between probes; probe_timeout = seconds before an unprocessed probe counts as
# a miss; max_strikes = consecutive misses before hard-exit 75 (~90-120s of sustained
# loop block at the defaults).
"loop_watchdog_probe_interval_s": 30.0,
"loop_watchdog_probe_timeout_s": 10.0,
"loop_watchdog_max_strikes": 3,
# Startup-liveness watchdog: stdlib-only daemon thread armed at process entry that
# hard-exits 75 if the loop isn't live within the deadline. Armed before config loads,
# so run_gateway() bridges these to HERMES_STARTUP_WATCHDOG /
# HERMES_STARTUP_WATCHDOG_TIMEOUT_S and re-arms the live handle; explicit env wins.
"startup_watchdog": True,
"startup_watchdog_timeout_seconds": 300,
# Keep writing the legacy ~/.hermes/sessions/sessions.json mirror of the routing index
# (primary copy: state.db gateway_routing table). True for external tooling and
# downgrade safety; False stops producing the file.
"write_sessions_json": True,
# Scale-to-zero idle TIMEOUT only. When an instance is opted in via the NAS "Labs"
# toggle (HERMES_SCALE_TO_ZERO env stamp) AND messaging is relay-only/absent AND a
# wakeUrl is registered, the relay transport goes dormant so the platform (e.g. Fly
# autostop) can suspend the machine; it wakes on the wakeUrl poke. Enablement is the
# Labs toggle, never a config key. 0/negative = default.
"scale_to_zero": {"idle_timeout_minutes": 2},
# Auto-resume restart-loop breaker. A supervisor-revived gateway auto-resumes the
# SIGTERM-interrupted session; if that turn keeps triggering the kill, boots no more than
# `max_gap_seconds` apart (floored by `window_seconds`) chain, and after `max_restarts`
# auto-resume is SKIPPED for that boot (inbound messages still served). Gap-based
# chaining also catches SLOW ~150s crash cycles. max_restarts=0 disables.
"restart_loop_guard": {"max_restarts": 3, "window_seconds": 60, "max_gap_seconds": 300},
# Respawn-storm circuit breaker (complements restart_loop_guard): counts (re)starts in a
# sliding window and sleeps an exponential backoff before booting so a crash-looping
# supervisor can't hammer the process. max_starts <= 0 disables. Env escape hatches:
# HERMES_GATEWAY_MAX_STARTS / HERMES_GATEWAY_START_WINDOW_S.
"respawn_storm": {"max_starts": 5, "window_seconds": 120},
# Prefix user messages IN THE MODEL'S CONTEXT with a timestamp (e.g. "[Tue 2026-04-28
# 13:40:53 CEST]") for temporal awareness. Persisted transcripts stay clean (timestamp
# is message metadata regardless), so enabling later surfaces past send-times too.
"message_timestamps": {"enabled": False},
# Max bytes of inbound image/audio/video the gateway buffers into RAM and caches to
# disk. Media is read fully into memory first, so unbounded uploads (Discord Nitro:
# 500 MB) or huge remote URLs can OOM-kill constrained deployments. Enforced in
# gateway/platforms/base.py for every adapter. 0 = no cap. Default 128 MiB.
"max_inbound_media_bytes": 134217728,
# Let adapters read HTTP_PROXY/HTTPS_PROXY/NO_PROXY/SSL_CERT_FILE from the environment
# and auto-detect generic/macOS system proxies. False when the gateway inherits a proxy
# it must not use (e.g. a scheduled task picking up a Clash/V2Ray HTTP_PROXY ->
# "Cannot connect to host 127.0.0.1:7890"). Per-platform vars (DISCORD_PROXY,
# TELEGRAM_PROXY, ...) are still honored.
"trust_env": True,
# Media delivery. False: any emitted file path is delivered natively unless under the
# credential/system denylist (/etc, /proc, ~/.ssh, ~/.aws, ~/.hermes/.env, auth.json).
# True: files must be under the Hermes cache, media_delivery_allow_dirs, or fresher than
# trust_recent_files_seconds — recommended for public-facing gateways so prompt
# injection can't exfiltrate host secrets. Bridged to HERMES_MEDIA_DELIVERY_STRICT.
"strict": False,
# Extra roots (project/scratch dirs, mounted shares) from which bare file paths may be
# uploaded; the Hermes cache is always trusted. List of absolute paths or one
# os.pathsep-separated string; tildes expanded. Bridged to HERMES_MEDIA_ALLOW_DIRS.
# Honored in both modes.
"media_delivery_allow_dirs": [],
# Trust files whose mtime is within trust_recent_files_seconds even outside the cache/
# allowlist (e.g. `pandoc -o /tmp/report.pdf`); system paths stay blocked. False =
# pure-allowlist mode. Bridged to HERMES_MEDIA_TRUST_RECENT_FILES. Only consulted when
# strict is true.
"trust_recent_files": True,
# Recency window in seconds; 600 covers a multi-tool turn. Bridged to
# HERMES_MEDIA_TRUST_RECENT_SECONDS. Only consulted when strict is true.
"trust_recent_files_seconds": 600,
# OpenAI-compatible API server platform (gateway/platforms/api_server.py).
"api_server": {
# Max concurrent agent runs. Requests to /v1/chat/completions, /v1/responses, and
# /v1/runs beyond this get HTTP 429 + Retry-After, bounding CPU/memory/LLM-quota
# exhaustion from a request flood. 0 = no cap.
"max_concurrent_runs": 10,
},
},
# Real-time token streaming to messaging platforms (gateway; restart after
# enabling). Off by default: costs extra edit/draft API calls per response.
"streaming": {
# When false, each response is delivered as one final message.
"enabled": False,
# auto = native drafts where supported (Telegram DMs, Bot API 9.5+), edits
# elsewhere (Discord, Slack, Matrix, Telegram groups); draft = drafts with
# edit fallback; edit = progressive editMessageText only; off = disabled.
"transport": "auto",
# Minimum seconds between progressive edits (Telegram's ~1 edit/s flood envelope).
"edit_interval": 0.8,
# Flush the buffer once this many chars accumulate, so short replies feel instant.
"buffer_threshold": 24,
# Cursor glyph appended to the in-progress message.
"cursor": " \u2589",
# Telegram only: when >0, if the preview was visible at least this many seconds
# the final edit is sent as a fresh message so the timestamp reflects completion.
"fresh_final_after_seconds": 0.0,
},
# Automatic cleanup of ~/.hermes/state.db, which otherwise grows without bound
# and slows FTS5 inserts, /resume listing, and insights queries.
"sessions": {
# Prune ENDED sessions inactive for retention_days (activity = latest message,
# else creation) about once per min_interval_hours at startup. Open, pinned,
# or mid-turn sessions are never deleted; stale automation sessions whose
# process died are *closed*, then get a full retention window before removal.
"auto_prune": True,
# Inactive days of ended-session history to keep (= `hermes sessions prune`).
"retention_days": 90,
# Auto-archive (soft-hide, never delete) sessions with no activity for
# auto_archive_days, once per min_interval_hours. Pinned sessions are exempt.
"auto_archive": False,
# Idle days before auto-archive hides a session (only when auto_archive is true).
"auto_archive_days": 3,
# VACUUM after a prune that deleted rows (SQLite never reclaims disk on
# DELETE). VACUUM blocks writes (~seconds per 100MB), so it runs only at
# startup, only when ≥1 session was deleted AND freelist/page_count > 25%.
"vacuum_after_prune": True,
# Minimum days between VACUUM rewrites; pruning keeps its normal cadence.
"min_vacuum_interval_days": 30,
# Minimum hours between auto-maintenance runs (tracked in state.db state_meta,
# shared across processes).
"min_interval_hours": 24,
# Legacy ~/.hermes/sessions/session_{sid}.json snapshots rewritten every turn.
# state.db is canonical (superset); snapshots consumed GBs on heavy users.
# Enable only for an external tool that reads the JSON files directly.
"write_json_snapshots": False,
# Notice about the compact FTS layout (reclaims ~60%+ of state.db). OPT-IN:
# legacy indexes stay until `hermes sessions optimize-storage` runs, since
# the rebuild is disk-heavy on large DBs. advise = `hermes update` prints a
# one-line notice with reclaimable size when a legacy index is detected;
# require = shown as a REQUIRED upgrade (tooling may gate on it); off = none.
"fts_optimize_notice": "advise",
# CJK-bigram search index (messages_fts_cjk). When the extension is built
# (native/fts5_cjk/build.sh → ~/.hermes/lib/libfts5_cjk.so), 1-2 char CJK
# terms get exact index matches instead of LIKE scans. True = use when present
# (inert otherwise); False = never load/serve it. Bridged to HERMES_CJK_FTS.
"cjk_fts": True,
# Slow session-search threshold (ms): searches at/above it log one INFO line
# with the routing path (fts_cjk / fts5 / trigram / like_scan). 0 logs every
# search. Bridged to HERMES_SEARCH_SLOW_MS.
"search_slow_ms": 1000,
# Transcript guards (a runaway 100k+ row session can exhaust memory when
# materialized at once; 0 disables). Max active messages (across the
# compression lineage) for interactive resume.
"max_resume_messages": 20000,
# Max active messages per session for in-memory export (`hermes sessions
# export`); checked per session, so full-DB backups of small sessions work.
"max_export_messages": 20000,
},
# First-touch onboarding hints (agent/onboarding.py). Each hint shows once and
# is latched under `seen`; wipe the section to re-see all hints.
"onboarding": {
"seen": {},
# First-ever gateway message: ask = offer to build a user profile (consent-
# gated; never reads connected accounts silently); off = plain intro only.
"profile_build": "ask",
},
# Privacy-safe aggregate metrics in this profile's local telemetry dir. Collection
# (`enabled`) and transmission to Nous (`send`) are SEPARATE opt-ins; see
# docs/observability/relay-shared-metrics.md Appendix A for consent/retention.
"telemetry": {
"shared_metrics": {
"enabled": False,
# Requires `enabled` (`send` alone logs an error). A package is sent only
# if its whole period is inside a recorded consent window.
"send": False,
# Ingest endpoint (override for staging/local). Deliberately NOT env-
# overridable. Non-HTTPS refused unless the host is localhost.
"endpoint": "https://telemetry.nousresearch.com/v1/telemetry",
},
},
# ``hermes doctor`` behaviour.
"doctor": {
# Per-probe timeout (seconds) for `hermes doctor --live` real-call probes.
"live_probe_timeout": 10,
},
# ``hermes update`` behaviour.
"updates": {
# Pre-update backup. quick = snapshot small critical state (pairing JSONs,
# cron jobs, config.yaml, .env, auth.json, profile DBs) into
# <HERMES_HOME>/state-snapshots/, skipping files >1 GiB; restore via
# ``/snapshot``. full = quick PLUS a ``hermes backup`` zip in
# <HERMES_HOME>/backups/ (``hermes import`` restores; slow on large homes;
# ``--backup`` forces once). off = none (``--no-backup`` forces once).
# Legacy booleans: true -> full, false -> off.
"pre_update_backup": "quick",
# Full backup zips to retain (older pruned after each success; floored to 1
# so the newest is always kept). The quick snapshot always keeps exactly 1.
"backup_keep": 5,
# Uncommitted source-tree changes during NON-interactive updates (desktop,
# gateway — no TTY; interactive updates always stash and ask). stash = stash,
# pull, restore on top (conflicts stay in a git stash). discard = stash and
# drop after the pull (stash-and-drop, not reset --hard + clean -fd, so
# ignored paths like node_modules/venv are never touched).
"non_interactive_local_changes": "stash",
# If the checkout is parked on a feature branch and the tree is clean, switch
# to the update target (commits stay on the branch; a loud notice names it) so
# non-interactive updates keep working. A DIRTY tree blocks the switch and the
# code update is SKIPPED with a loud warning. False = never auto-switch.
"auto_switch_parked_branch": True,
# Clean parked branch with unmerged commits: switch = move to the update
# target, commits stay on the branch (never conflicts). update_in_place = for
# a maintained custom branch: merge origin/<target> INTO it after leaving a
# pre-update-<stamp> tag; a conflict stops the update cleanly.
# `hermes update --switch-branch` overrides to switch for one run.
"parked_branch_strategy": "switch",
# Refresh an installed cua-driver during `hermes update` (best-effort, macOS
# only). Turn off e.g. on non-admin accounts where /Applications isn't writable.
"refresh_cua_driver": True,
},
# LSP diagnostics (pyright, gopls, rust-analyzer...) in the post-write lint check
# of write_file/patch. Runs only when the cwd or edited file is inside a git
# worktree; otherwise dormant and the in-process syntax check is the only tier.
"lsp": {
# False disables the whole subsystem: no servers, no event loop, no cost.
"enabled": True,
# document = wait up to wait_timeout seconds for the current file's
# diagnostics; full = also request workspace-wide diagnostics (slower).
"wait_mode": "document",
"wait_timeout": 5.0,
# Missing server binaries: auto = install via npm/go/pip into
# <HERMES_HOME>/lsp/bin/ on first use; manual = only binaries on PATH;
# off = alias for manual.
"install_strategy": "auto",
# Idle seconds before a server is shut down (respawned on demand), so long-
# running processes don't accumulate stale children (hundreds of MB + pipe
# FDs each) across worktrees. 0 = keep servers for process lifetime.
"idle_timeout": 600.0,
# Per-server overrides keyed by registry server_id (pyright, gopls...):
# disabled: true; command: ["path/to/server", "--stdio"] (bypasses auto-
# install); env: {...}; initialization_options: {...} (merged into LSP
# initializationOptions).
"servers": {},
},
# X (Twitter) Search via xAI's x_search Responses tool. Registers when xAI creds
# exist (SuperGrok OAuth or XAI_API_KEY) AND the toolset is enabled in `hermes tools`.
"x_search": {
# xAI model for the Responses call; any Grok model with x_search access works.
"model": "grok-4.5",
# Reasoning effort for models that support it; null keeps the model default.
"reasoning_effort": None,
# Request timeout in seconds (minimum 30); complex queries can take 60-120s.
"timeout_seconds": 180,
# Retries on 5xx / ReadTimeout / ConnectionError (backoff 1.5x attempt s, cap 5s).
"retries": 2,
},
# External secret sources — pull credentials from secret managers at startup
# instead of storing them in ~/.hermes/.env.
"secrets": {
# Optional ordering of enabled sources (e.g. [onepassword, bitwarden]); default
# registration order. Mapped sources (explicit VAR→ref) always beat bulk
# sources (BSM project dumps); first claim on a var wins, later ones warn.
# "sources": [],
"bitwarden": {
# When false, BSM is never contacted and bws is never auto-installed.
"enabled": False,
# Env var holding the machine-account token — the one bootstrap secret;
# lives in ~/.hermes/.env (or the shell), never in config.yaml.
"access_token_env": "BWS_ACCESS_TOKEN",
# UUID of the BSM project to sync from.
"project_id": "",
# Seconds to reuse a fresh disk/memory cache entry before contacting
# Bitwarden again. 0 disables fresh-cache reuse.
"cache_ttl_seconds": 300,
# Last-good fallback for NETWORK/TIMEOUT outages: AES-GCM cache under
# ~/.hermes/cache/, reused up to max_stale_seconds. Auth failures never
# fall back.
"encrypted_cache": {"enabled": False, "max_stale_seconds": 0},
# BSM values overwrite existing env vars, so rotating in Bitwarden takes
# effect without clearing the matching .env line.
"override_existing": True,
# Auto-download bws into ~/.hermes/bin/ on first use; False = bws must be on PATH.
"auto_install": True,
# Passed to bws as BWS_SERVER_URL. Empty = US Cloud (bws default);
# https://vault.bitwarden.eu for EU; your own URL for self-hosted.
"server_url": "",
},
"onepassword": {
# When false, the op CLI is never invoked.
"enabled": False,
# env-var name → op://vault/item/field; each resolved with one `op read`.
"env": {},
# Account shorthand / sign-in address for `op read --account`; empty = default.
"account": "",
# Env var holding a service-account token for headless auth (exported to
# op as OP_SERVICE_ACCOUNT_TOKEN). Unset = interactive/desktop op session.
"service_account_token_env": "OP_SERVICE_ACCOUNT_TOKEN",
# Absolute path to op, used verbatim (avoids trusting PATH). Empty = PATH.
"binary_path": "",
# Seconds to cache values in-process and on disk; 0 disables BOTH layers.
"cache_ttl_seconds": 300,
# Overwrite existing env vars so rotation takes effect; False lets .env win.
"override_existing": True,
},
},
# Paste collapse thresholds (TUI + CLI); 0 disables each. threshold: bracketed
# pastes with this many newlines collapse to a file reference. fallback: same
# test for terminals without bracketed paste, gated by chars/newlines-added
# heuristics. char_threshold: single-line pastes this long collapse too.
"paste_collapse_threshold": 5,
"paste_collapse_threshold_fallback": 5,
"paste_collapse_char_threshold": 2000,
# Computer Use (cua-driver) toolset settings.
"computer_use": {
# cua-driver's upstream PostHog telemetry defaults ON; Hermes sets
# CUA_DRIVER_RS_TELEMETRY_ENABLED=0 in every child env unless this is true.
"cua_telemetry": False,
# Cap driver screenshot longest edge (pixels) via set_config at session
# start; shrinks SOM multimodal payloads. 0 disables.
"max_image_dimension": 1456,
# capture_after mode: som = screenshot + overlays; ax = elements only, no
# PNG (faster); vision = pixels only.
"capture_after_mode": "som",
# Disable cua-driver's cursor overlay, which can peg a core when idle (macOS
# redraw loop; Linux/WSL2 idle spin). None = auto (off on macOS + headless/
# WSL2 Linux, on elsewhere); True = always disable; False = always enable.
"no_overlay": None,
# standard = cua-driver's own approval boundary; bounded = no runtime prompts,
# anything outside capability_manifest fails closed. `unrestricted` is NOT
# accepted here: it stays on the per-session YOLO toggle so config can't
# bypass approvals.
"permission_mode": "standard",
# Path (~ ok) to the reviewed manifest for permission_mode bounded; passed as
# --capability-manifest. See cua.ai/docs/reference/cua-driver/permission-modes
"capability_manifest": "",
# macOS only: allow an UNSIGNED CuaDriver.app for the private-session daemon.
# False fails closed unless signed with the official com.trycua.driver
# identity. Only for local driver development from source.
"allow_unsigned_driver": False,
},
# Egress credential-injection proxy (iron-proxy) for remote terminal sandboxes
# (Docker today): the sandbox sees opaque tokens and iron-proxy swaps in real
# credentials at egress, so a compromised sandbox leaks only tokens that work
# behind the trusted proxy. Configure with `hermes egress setup`.
"proxy": {
# When false, nothing starts, no docker mounts, no binary installs.
"enabled": False,
# Tunnel listener port; sandboxes get HTTPS_PROXY=http://<host>:<port>.
"tunnel_port": 9090,
# Auto-download the pinned binary into ~/.hermes/bin/; False = iron-proxy on PATH.
"auto_install": True,
# Upstream secret source: env = process env; bitwarden = refetch via `bws
# secret list` on each proxy restart (requires secrets.bitwarden.enabled).
"credential_source": "env",
# True: the Docker backend refuses to start a sandbox if the proxy is enabled
# but not running. False: fall back to direct outbound with real credentials.
"enforce_on_docker": True,
# With credential_source bitwarden, a missing BWS token/project_id or an
# empty fetch makes the daemon raise. True silently falls back to host env
# (useful mid-migration). A leftover fail_on_uncovered_providers key is ignored.
"allow_env_fallback": False,
# SSRF deny list. None/empty = loopback, link-local (incl. 169.254.169.254
# metadata), RFC1918. Explicit [] opts out (only for hermetic loopback tests).
"upstream_deny_cidrs": None,
# Extra allowed upstream hosts beyond the bundled major-provider defaults;
# wildcards (`*.foo.com`) supported.
"extra_allowed_hosts": [],
},
# Hermes Desktop (Electron) launch options; only affect `hermes desktop`.
"desktop": {
# Git repo discovery for the Projects sidebar; empty roots = bounded scan of $HOME.
"repo_scan_enabled": True,
"repo_scan_roots": [],
"repo_scan_exclude_paths": [],
# Extra Electron flags per launch, e.g. ["--ozone-platform=x11"] or GPU
# workarounds. List of strings; a single string is shell-split.
"electron_flags": [],
# Linux Ozone backend, bridged to ELECTRON_OZONE_PLATFORM_HINT (explicit env
# wins). auto = Chromium default; x11 = XWayland, for compositors that ignore
# always-on-top for Wayland clients (e.g. COSMIC) — also puts the HUD on the
# solid-window input path; wayland = force a native Wayland surface.
"ozone_platform_hint": "auto",
# Bridged to HERMES_DESKTOP_DISABLE_GPU: auto = disable GPU only on remote
# displays (SSH/VNC/RDP); true = always software rendering (no-GPU VMs where
# the GPU path hangs); false = always keep GPU on.
"disable_gpu": "auto",
# Linux keychain for token storage (Chromium --password-store). auto = detect
# KWallet (KDE env) or any org.freedesktop.secrets provider via D-Bus;
# gnome-libsecret|kwallet|kwallet5|kwallet6|basic force one (basic =
# unencrypted). Bridged to HERMES_DESKTOP_PASSWORD_STORE; ignored off-Linux.
"password_store": "auto",
# macOS only: code-signing identity (login-keychain cert; self-signed works)
# to re-sign locally rebuilt apps so the Designated Requirement — and thus
# TCC grants — survives updates. Empty = default ad-hoc identifier-pinned signing.
"macos_signing_identity": "",
# Auto-continue a turn killed by a crash: resuming re-submits the interrupted
# prompt if fresh; a stale one just shows the recovered partial transcript.
"auto_continue": {
"enabled": True,
# How recent the interruption must be to auto-continue (minutes).
"freshness_minutes": 15,
# Crash-loop breaker: max automatic re-runs of one interrupted turn.
"max_attempts": 2,
},
},
"nous": {
# Upper bound (seconds) on the Nous auth keepalive tick, which derives from
# the server-issued credential lifetime (raising above it has no effect).
# 0 disables the keepalive thread.
"keepalive_interval_seconds": 900,
},
# Google Vertex AI (Gemini). Auth is OAuth2 from a service-account JSON or ADC,
# NOT an API key; the credential path lives in .env (VERTEX_CREDENTIALS_PATH /
# GOOGLE_APPLICATION_CREDENTIALS). Bridged to VERTEX_PROJECT_ID / VERTEX_REGION.
"vertex": {
# GCP project ID. Empty → project_id from the service-account JSON (or ADC).
"project_id": "",
# "global" is required for Gemini 3.x preview models (regional endpoints
# silently 404); use e.g. "us-central1" only if your models are region-pinned.
"region": "global",
},
# Managed llama.cpp runtime (docs: user-guide/local-models): official binaries,
# one supervised llama-server in router mode. No context/VRAM knobs by design.
"local_runtime": {
# Off = detection-only (Hermes still finds an external llama-server you run).
"enabled": False,
# Pinned llama.cpp release tag; bumped by Hermes releases after validation.
"tag": "b10679",
# auto = CUDA on NVIDIA, Metal on macOS, Vulkan on other GPUs, else CPU.
# Explicit: cuda|metal|vulkan|hip|cpu.
"backend": "auto",
# Router process: how many models may be resident at once.
"models_max": 4,
# Port for the managed server. 0 = pick a free port at spawn.
"port": 0,
# Extra ports detection probes for an external llama-server (besides 8080).
"detect_ports": [],
},
# Config schema version - bump this when adding new required fields
"_config_version": 40,
}
def _env(description, prompt, **keys):
"""One OPTIONAL_ENV_VARS entry; keyword order is preserved as dict key order."""
return {"description": description, "prompt": prompt, **keys}
# Optional environment variables that enhance functionality. Feeds the dashboard keys page and
# setup checklists; category: provider|tool|skill|messaging|setting, advanced=True hides from
# checklists, tools=[...] lists the model tools the key unlocks.
OPTIONAL_ENV_VARS = {
# ── Provider (handled in provider selection, not shown in checklists) ──
"NOUS_BASE_URL": _env("Nous Portal base URL override", "Nous Portal base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"OPENROUTER_API_KEY": _env("OpenRouter API key (for vision, web scraping helpers, and MoA)",
"OpenRouter API key", url="https://openrouter.ai/keys", password=True, tools=["vision_analyze"],
category="provider", advanced=True),
"GOOGLE_API_KEY": _env("Google AI Studio API key (also recognized as GEMINI_API_KEY)",
"Google AI Studio API key", url="https://aistudio.google.com/app/apikey", password=True,
category="provider", advanced=True),
"GEMINI_API_KEY": _env("Google AI Studio API key (alias for GOOGLE_API_KEY)", "Gemini API key",
url="https://aistudio.google.com/app/apikey", password=True, category="provider", advanced=True),
"GEMINI_BASE_URL": _env("Google AI Studio base URL override", "Gemini base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"VERTEX_CREDENTIALS_PATH": _env(
"Path to a Google Cloud service account JSON for Vertex AI (Gemini). Vertex uses OAuth2, not a "
"static API key — this points at the credentials Hermes mints short-lived tokens from. Falls back "
"to GOOGLE_APPLICATION_CREDENTIALS, then to ADC (gcloud auth application-default login). Set "
"project/region under vertex: in config.yaml.",
"Vertex service account JSON path (leave empty to use ADC / GOOGLE_APPLICATION_CREDENTIALS)",
url="https://cloud.google.com/iam/docs/keys-create-delete", password=False, category="provider",
advanced=True),
"XAI_API_KEY": _env("xAI API key", "xAI API key", url="https://console.x.ai/", password=True,
category="provider", advanced=True),
"XAI_BASE_URL": _env("xAI base URL override", "xAI base URL (leave empty for default)", url=None,
password=False, category="provider", advanced=True),
"NVIDIA_API_KEY": _env("NVIDIA NIM API key (build.nvidia.com or local NIM endpoint)",
"NVIDIA NIM API key", url="https://build.nvidia.com/", password=True, category="provider",
advanced=True),
"NVIDIA_BASE_URL": _env("NVIDIA NIM base URL override (e.g. http://localhost:8000/v1 for local NIM)",
"NVIDIA NIM base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"LM_API_KEY": _env("LM Studio bearer token for auth-enabled local servers",
"LM Studio API key / bearer token", url=None, password=True, category="provider", advanced=True),
"LM_BASE_URL": _env("LM Studio base URL override", "LM Studio base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"GLM_API_KEY": _env("Z.AI / GLM API key (also recognized as ZAI_API_KEY / Z_AI_API_KEY)",
"Z.AI / GLM API key", url="https://z.ai/", password=True, category="provider", advanced=True),
"ZAI_API_KEY": _env("Z.AI API key (alias for GLM_API_KEY)", "Z.AI API key", url="https://z.ai/",
password=True, category="provider", advanced=True),
"Z_AI_API_KEY": _env("Z.AI API key (alias for GLM_API_KEY)", "Z.AI API key", url="https://z.ai/",
password=True, category="provider", advanced=True),
"GLM_BASE_URL": _env("Z.AI / GLM base URL override", "Z.AI / GLM base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"KIMI_API_KEY": _env("Kimi / Moonshot API key", "Kimi API key", url="https://platform.moonshot.cn/",
password=True, category="provider", advanced=True),
"KIMI_BASE_URL": _env("Kimi / Moonshot base URL override", "Kimi base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"KIMI_CN_API_KEY": _env("Kimi / Moonshot China API key", "Kimi (China) API key",
url="https://platform.moonshot.cn/", password=True, category="provider", advanced=True),
"STEPFUN_API_KEY": _env("StepFun Step Plan API key", "StepFun Step Plan API key",
url="https://platform.stepfun.com/", password=True, category="provider", advanced=True),
"STEPFUN_BASE_URL": _env("StepFun Step Plan base URL override",
"StepFun Step Plan base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"ARCEEAI_API_KEY": _env("Arcee AI API key", "Arcee AI API key", url="https://chat.arcee.ai/",
password=True, category="provider", advanced=True),
"ARCEE_BASE_URL": _env("Arcee AI base URL override", "Arcee base URL (leave empty for default)", url=None,
password=False, category="provider", advanced=True),
"GMI_API_KEY": _env("GMI Cloud API key", "GMI Cloud API key", url="https://www.gmicloud.ai/",
password=True, category="provider", advanced=True),
"GMI_BASE_URL": _env("GMI Cloud base URL override", "GMI Cloud base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"ACTUAL_API_KEY": _env("Actual Computer inference key (ac_...)", "Actual Computer inference key",
url="https://actual.inc/user/keys", password=True, category="provider", advanced=True),
"ACTUAL_BASE_URL": _env(
"Actual Computer base URL override (set to http://127.0.0.1:8080 for the local offline daemon)",
"Actual Computer base URL (leave empty for hosted relay)", url=None, password=False,
category="provider", advanced=True),
"FIREWORKS_API_KEY": _env("Fireworks AI API key", "Fireworks AI API key",
url="https://app.fireworks.ai/settings/users/api-keys", password=True, category="provider",
advanced=True),
"MINIMAX_API_KEY": _env("MiniMax API key (international)", "MiniMax API key",
url="https://www.minimax.io/", password=True, category="provider", advanced=True),
"MINIMAX_BASE_URL": _env("MiniMax base URL override", "MiniMax base URL (leave empty for default)",
url=None, password=False, category="provider", advanced=True),
"MINIMAX_CN_API_KEY": _env("MiniMax API key (China endpoint)", "MiniMax (China) API key",
url="https://www.minimaxi.com/", password=True, category="provider", advanced=True),
"MINIMAX_CN_BASE_URL": _env("MiniMax (China) base URL override",
"MiniMax (China) base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"DEEPSEEK_API_KEY": _env("DeepSeek API key for direct DeepSeek access", "DeepSeek API Key",
url="https://platform.deepseek.com/api_keys", password=True, category="provider"),
"DEEPSEEK_BASE_URL": _env("Custom DeepSeek API base URL (advanced)", "DeepSeek Base URL", url="",
password=False, category="provider"),
"DASHSCOPE_API_KEY": _env("Alibaba Cloud DashScope API key (Qwen + multi-provider models)",
"DashScope API Key", url="https://modelstudio.console.alibabacloud.com/", password=True,
category="provider"),
"DASHSCOPE_BASE_URL": _env("Custom DashScope base URL (default: coding-intl OpenAI-compat endpoint)",
"DashScope Base URL", url="", password=False, category="provider", advanced=True),
"HERMES_QWEN_BASE_URL": _env("Qwen Portal base URL override (default: https://portal.qwen.ai/v1)",
"Qwen Portal base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"OPENCODE_ZEN_API_KEY": _env("OpenCode Zen API key (pay-as-you-go access to curated models)",
"OpenCode Zen API key", url="https://opencode.ai/auth", password=True, category="provider",
advanced=True),
"COMMANDCODE_API_KEY": _env("CommandCode API key (GOAT/Pro/Max/Provider plans — 30+ models via one key)",
"CommandCode API key", url="https://commandcode.ai/studio/", password=True, category="provider",
advanced=True),
"OPENCODE_ZEN_BASE_URL": _env("OpenCode Zen base URL override",
"OpenCode Zen base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"OPENCODE_GO_API_KEY": _env("OpenCode Go API key ($10/month subscription for open models)",
"OpenCode Go API key", url="https://opencode.ai/auth", password=True, category="provider",
advanced=True),
"OPENCODE_GO_BASE_URL": _env("OpenCode Go base URL override",
"OpenCode Go base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"HF_TOKEN": _env("Hugging Face token for Inference Providers (20+ open models via router.huggingface.co)",
"Hugging Face Token", url="https://huggingface.co/settings/tokens", password=True,
category="provider"),
"HF_BASE_URL": _env("Hugging Face Inference Providers base URL override",
"HF base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"OLLAMA_API_KEY": _env("Ollama Cloud API key (ollama.com — cloud-hosted open models)",
"Ollama Cloud API key", url="https://ollama.com/settings", password=True, category="provider",
advanced=True),
"OLLAMA_BASE_URL": _env("Ollama Cloud base URL override (default: https://ollama.com/v1)",
"Ollama base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"XIAOMI_API_KEY": _env(
"Xiaomi MiMo API key for MiMo models (mimo-v2.5-pro, mimo-v2.5, mimo-v2-pro, mimo-v2-omni, "
"mimo-v2-flash)", "Xiaomi MiMo API Key", url="https://platform.xiaomimimo.com", password=True,
category="provider"),
"XIAOMI_BASE_URL": _env("Xiaomi MiMo base URL override (default: https://api.xiaomimimo.com/v1)",
"Xiaomi base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"UPSTAGE_API_KEY": _env("Upstage API key for Solar LLM models", "Upstage API Key",
url="https://console.upstage.ai/api-keys", password=True, category="provider"),
"UPSTAGE_BASE_URL": _env("Upstage base URL override (default: https://api.upstage.ai/v1)",
"Upstage base URL (leave empty for default)", url=None, password=False, category="provider",
advanced=True),
"AWS_REGION": _env("AWS region for Bedrock API calls (e.g. us-east-1, eu-central-1)", "AWS Region",
url="https://docs.aws.amazon.com/bedrock/latest/userguide/bedrock-regions.html", password=False,
category="provider", advanced=True),
"AWS_PROFILE": _env("AWS named profile for Bedrock authentication (from ~/.aws/credentials)",
"AWS Profile", url=None, password=False, category="provider", advanced=True),
"AZURE_FOUNDRY_API_KEY": _env("Azure Foundry API key for custom Azure endpoints", "Azure Foundry API Key",
url="https://ai.azure.com/", password=True, category="provider"),
"AZURE_FOUNDRY_BASE_URL": _env(
"Azure Foundry base URL (set via 'hermes model' for endpoint-specific config)",
"Azure Foundry base URL", url=None, password=False, category="provider", advanced=True),
# ── Tool API keys ──
"EXA_API_KEY": _env("Exa API key for AI-native web search and contents", "Exa API key",
url="https://exa.ai/", tools=["web_search", "web_extract"], password=True, category="tool"),
"PARALLEL_API_KEY": _env("Parallel API key for AI-native web search and extract", "Parallel API key",
url="https://parallel.ai/", tools=["web_search", "web_extract"], password=True, category="tool"),
"FIRECRAWL_API_KEY": _env("Firecrawl API key for web search and scraping", "Firecrawl API key",
url="https://firecrawl.dev/", tools=["web_search", "web_extract"], password=True, category="tool"),
"FIRECRAWL_API_URL": _env("Firecrawl API URL for self-hosted instances (optional)",
"Firecrawl API URL (leave empty for cloud)", url=None, password=False, category="tool",
advanced=True),
"FIRECRAWL_GATEWAY_URL": _env(
"Exact Firecrawl tool-gateway origin override for Nous Subscribers only (optional)",
"Firecrawl gateway URL (leave empty to derive from domain)", url=None, password=False,
category="tool", advanced=True),
"TOOL_GATEWAY_DOMAIN": _env(
"Shared tool-gateway domain suffix for Nous Subscribers only, used to derive vendor hosts, e.g. "
"nousresearch.com -> firecrawl-gateway.nousresearch.com", "Tool-gateway domain suffix", url=None,
password=False, category="tool", advanced=True),
"TOOL_GATEWAY_SCHEME": _env(
"Shared tool-gateway URL scheme for Nous Subscribers only, used to derive vendor hosts (`https` by "
"default, set `http` for local gateway testing)", "Tool-gateway URL scheme", url=None, password=False,
category="tool", advanced=True),
"TOOL_GATEWAY_USER_TOKEN": _env(
"Explicit Nous Subscriber access token for tool-gateway requests (optional; otherwise read from "
"the Hermes auth store)", "Tool-gateway user token", url=None, password=True, category="tool",
advanced=True),
"TAVILY_API_KEY": _env(
"Tavily API key for AI-native web search and extract (optional — keyless works when Tavily is "
"selected)", "Tavily API key", url="https://app.tavily.com/home", tools=["web_search", "web_extract"],
password=True, category="tool"),
"KEENABLE_API_KEY": _env(
"Keenable API key for fast independent-index web search and page fetch (optional — keyless free "
"tier works without it)", "Keenable API key", url="https://keenable.ai",
tools=["web_search", "web_extract"], password=True, category="tool"),
"SEARXNG_URL": _env("URL of your SearXNG instance for free self-hosted web search",
"SearXNG URL (e.g. http://localhost:8080)", url="https://searxng.github.io/searxng/",
tools=["web_search"], password=False, category="tool"),
"BRAVE_SEARCH_API_KEY": _env("Brave Search API subscription token (free tier: 2,000 queries/mo)",
"Brave Search subscription token", url="https://brave.com/search/api/", tools=["web_search"],
password=True, category="tool"),
"BROWSERBASE_API_KEY": _env(
"Browserbase API key for cloud browser (optional — local browser works without this)",
"Browserbase API key", url="https://browserbase.com/", tools=["browser_navigate", "browser_click"],
password=True, category="tool"),
"BROWSERBASE_PROJECT_ID": _env("Browserbase project ID (optional — only needed for cloud browser)",
"Browserbase project ID", url="https://browserbase.com/", tools=["browser_navigate", "browser_click"],
password=False, category="tool"),
"BROWSER_USE_API_KEY": _env(
"Browser Use API key for cloud browser (optional — local browser works without this)",
"Browser Use API key", url="https://browser-use.com/", tools=["browser_navigate", "browser_click"],
password=True, category="tool"),
"FIRECRAWL_BROWSER_TTL": _env("Firecrawl browser session TTL in seconds (optional, default 300)",
"Browser session TTL (seconds)", tools=["browser_navigate", "browser_click"], password=False,
category="tool"),
"AGENT_BROWSER_ENGINE": _env(
"Local browser engine: auto (default Chrome), lightpanda (faster, no screenshots; Browser Use mode "
"spawns lightpanda serve), chrome", "Browser engine (auto/lightpanda/chrome)",
url="https://lightpanda.io/docs/run-locally/installation/one-liner",
tools=["browser_exec", "browser_navigate", "browser_snapshot", "browser_click", "browser_vision"],
password=False, category="tool", advanced=True),
"CAMOFOX_URL": _env(
"Camofox browser server URL for local anti-detection browsing (e.g. http://localhost:9377)",
"Camofox server URL", url="https://github.com/jo-inc/camofox-browser",
tools=["browser_navigate", "browser_click"], password=False, category="tool"),
"CAMOFOX_API_KEY": _env(
"Optional bearer token sent as Authorization header to a remote/authenticated Camofox server",
"Camofox API key", url="https://github.com/jo-inc/camofox-browser",
tools=["browser_navigate", "browser_click"], password=True, category="tool", advanced=True),
"FAL_KEY": _env("FAL API key for image and video generation", "FAL API key", url="https://fal.ai/",
tools=["image_generate", "video_generate"], password=True, category="tool"),
"KREA_API_KEY": _env("Krea API key for Krea 2 image generation (Medium + Large)", "Krea API key",
url="https://www.krea.ai/settings/api-tokens", tools=["image_generate"], password=True,
category="tool"),
"VOICE_TOOLS_OPENAI_KEY": _env("OpenAI API key for voice transcription (Whisper) and OpenAI TTS",
"OpenAI API Key (for Whisper STT + TTS)", url="https://platform.openai.com/api-keys",
tools=["voice_transcription", "openai_tts"], password=True, category="tool"),
"ELEVENLABS_API_KEY": _env(
"ElevenLabs API key for premium text-to-speech voices and Scribe transcription", "ElevenLabs API key",
url="https://elevenlabs.io/", tools=["elevenlabs_tts", "voice_transcription"], password=True,
category="tool"),
"MISTRAL_API_KEY": _env("Mistral API key for Voxtral TTS and transcription (STT)", "Mistral API key",
url="https://console.mistral.ai/", password=True, category="tool"),
"PORCUPINE_ACCESS_KEY": _env(
"Picovoice access key for the Porcupine 'Hey Hermes' wake word engine (optional; openWakeWord is "
"the free default)", "Picovoice access key", url="https://console.picovoice.ai/", password=True,
category="tool"),
"GITHUB_TOKEN": _env("GitHub token for Skills Hub (higher API rate limits, skill publish)",
"GitHub Token", url="https://github.com/settings/tokens", password=True, category="tool"),
# ── Bundled skills (opt-in) ── category="skill" (not "tool") so the sandbox env blocklist in
# tools/environments/local.py does NOT rewrite them; skills need them passed through to curl
# via tools/env_passthrough.py.
"NOTION_API_KEY": _env("Notion integration token (used by the `notion` skill)", "Notion API key",
url="https://www.notion.so/my-integrations", password=True, category="skill", advanced=True),
"LINEAR_API_KEY": _env("Linear personal API key (used by the `linear` skill)", "Linear API key",
url="https://linear.app/settings/account/security", password=True, category="skill", advanced=True),
"AIRTABLE_API_KEY": _env("Airtable personal access token (used by the `airtable` skill)",
"Airtable API key", url="https://airtable.com/create/tokens", password=True, category="skill",
advanced=True),
"TENOR_API_KEY": _env("Tenor API key for GIF search (used by the `gif-search` skill)", "Tenor API key",
url="https://developers.google.com/tenor/guides/quickstart", password=True, category="skill",
advanced=True),
# ── Honcho ──
"HONCHO_API_KEY": _env("Honcho API key for AI-native persistent memory", "Honcho API key",
url="https://app.honcho.dev", tools=["honcho_context"], password=True, category="tool"),
"HONCHO_BASE_URL": _env("Base URL for self-hosted Honcho instances (no API key needed)",
"Honcho base URL (e.g. http://localhost:8000)", category="tool"),
# ── Hindsight ──
"HINDSIGHT_API_KEY": _env("Hindsight API key for graph-aware persistent memory", "Hindsight API key",
url="https://hindsight.vectorize.io", tools=["hindsight_recall"], password=True, category="tool"),
"HINDSIGHT_API_URL": _env("Base URL for the Hindsight API (default: https://api.hindsight.vectorize.io)",
"Hindsight API URL", category="tool", advanced=True),
# ── Supermemory ──
"SUPERMEMORY_API_KEY": _env("Supermemory API key for conversation-scoped persistent memory",
"Supermemory API key", url="https://supermemory.ai", tools=["supermemory_search"], password=True,
category="tool"),
# ── Mem0 ──
"MEM0_API_KEY": _env("Mem0 Platform API key for semantic persistent memory", "Mem0 API key",
url="https://app.mem0.ai", tools=["mem0_search"], password=True, category="tool"),
# ── RetainDB ──
"RETAINDB_API_KEY": _env("RetainDB API key for persistent memory", "RetainDB API key",
url="https://retaindb.com", tools=["retaindb_search"], password=True, category="tool"),
"RETAINDB_BASE_URL": _env(
"Base URL for self-hosted RetainDB instances (default: https://api.retaindb.com)",
"RetainDB base URL", category="tool", advanced=True),
# ── ByteRover ──
"BRV_API_KEY": _env("ByteRover API key (optional, for cloud sync — local-first by default)",
"ByteRover API key", url="https://app.byterover.dev", tools=["brv_query"], password=True,
category="tool"),
# ── OpenViking ──
"OPENVIKING_API_KEY": _env("OpenViking API key (leave blank for local dev mode)", "OpenViking API key",
tools=["viking_search"], password=True, category="tool"),
"OPENVIKING_ENDPOINT": _env("OpenViking server URL (default: http://127.0.0.1:1933)",
"OpenViking endpoint", category="tool", advanced=True),
# ── Langfuse observability ──
"HERMES_LANGFUSE_PUBLIC_KEY": _env("Langfuse project public key (pk-lf-...)", "Langfuse public key",
url="https://cloud.langfuse.com", password=False, category="tool"),
"HERMES_LANGFUSE_SECRET_KEY": _env("Langfuse project secret key (sk-lf-...)", "Langfuse secret key",
url="https://cloud.langfuse.com", password=True, category="tool"),
"HERMES_LANGFUSE_BASE_URL": _env("Langfuse server URL (default: https://cloud.langfuse.com)",
"Langfuse server URL (leave empty for cloud.langfuse.com)", url=None, password=False, category="tool",
advanced=True),
# ── Messaging platforms ──
"TELEGRAM_BOT_TOKEN": _env(
"Complete Telegram bot token created by @BotFather (numeric bot ID followed by a colon and secret)",
"Telegram bot token", url="https://t.me/BotFather", password=True, category="messaging"),
"TELEGRAM_ALLOWED_USERS": _env(
"Optional comma-separated numeric Telegram user IDs allowed immediately; leave blank to approve "
"new users through DM pairing", "Allowed Telegram user IDs (comma-separated)",
url="https://t.me/userinfobot", password=False, category="messaging"),
"TELEGRAM_PROXY": _env(
"Proxy URL for Telegram connections (overrides HTTPS_PROXY). Supports http://, https://, socks5://",
"Telegram proxy URL (optional)", password=False, category="messaging"),
"DISCORD_BOT_TOKEN": _env("Discord bot token from Developer Portal", "Discord bot token",
url="https://discord.com/developers/applications", password=True, category="messaging"),
"DISCORD_ALLOWED_USERS": _env("Comma-separated Discord user IDs allowed to use the bot",
"Allowed Discord user IDs (comma-separated)", url=None, password=False, category="messaging"),
"DISCORD_REPLY_TO_MODE": _env(
"Discord reply threading mode: 'off' (no reply references), 'first' (reply on first message only, "
"default), 'all' (reply on every chunk)", "Discord reply mode (off/first/all)", url=None,
password=False, category="messaging"),
"SLACK_BOT_TOKEN": _env(
"Slack bot token (xoxb-). Get from OAuth & Permissions after installing your app. Required scopes: "
"chat:write, app_mentions:read, channels:history, groups:history, im:history, im:read, im:write, "
"mpim:history, mpim:read, users:read, files:read, files:write", "Slack Bot Token (xoxb-...)",
help="In your Slack app, add the required bot scopes, install the app to the workspace, then copy OAuth & Permissions > Bot User OAuth Token.",
url="https://api.slack.com/apps", password=True, category="messaging"),
"SLACK_APP_TOKEN": _env(
"Slack app-level token (xapp-) for Socket Mode. Get from Basic Information → App-Level Tokens. "
"Also ensure Event Subscriptions include: message.im, message.channels, message.groups, "
"message.mpim, app_mention", "Slack App Token (xapp-...)",
help="In your Slack app, enable Socket Mode, then create Basic Information > App-Level Tokens with the connections:write scope.",
url="https://api.slack.com/apps", password=True, category="messaging"),
"SLACK_ALLOWED_USERS": _env(
"Comma-separated Slack member IDs allowed to use Hermes, e.g. U01ABC2DEF3. Without this, Slack may "
"connect but deny messages by default.", "Allowed Slack member IDs",
help="In Slack, open your profile, choose More or the three-dot menu, then Copy member ID. Add multiple IDs comma-separated.",
url="https://api.slack.com/apps", password=False, category="messaging"),
"MATTERMOST_URL": _env("Mattermost server URL (e.g. https://mm.example.com)", "Mattermost server URL",
url="https://mattermost.com/deploy/", password=False, category="messaging"),
"MATTERMOST_TOKEN": _env("Mattermost bot token or personal access token", "Mattermost bot token",
url=None, password=True, category="messaging"),
"MATTERMOST_ALLOWED_USERS": _env("Comma-separated Mattermost user IDs allowed to use the bot",
"Allowed Mattermost user IDs (comma-separated)", url=None, password=False, category="messaging"),
"MATTERMOST_REQUIRE_MENTION": _env(
"Require @mention in Mattermost channels (default: true). Set to false to respond to all messages.",
"Require @mention in channels", url=None, password=False, category="messaging"),
"MATTERMOST_FREE_RESPONSE_CHANNELS": _env(
"Comma-separated Mattermost channel IDs where bot responds without @mention",
"Free-response channel IDs (comma-separated)", url=None, password=False, category="messaging"),
"MATRIX_HOMESERVER": _env("Matrix homeserver URL (e.g. https://matrix.example.org)",
"Matrix homeserver URL", url="https://matrix.org/ecosystem/servers/", password=False,
category="messaging"),
"MATRIX_ACCESS_TOKEN": _env("Matrix access token (preferred over password login)", "Matrix access token",
url=None, password=True, category="messaging"),
"MATRIX_USER_ID": _env("Matrix user ID (e.g. @hermes:example.org)", "Matrix user ID (@user:server)",
url=None, password=False, category="messaging"),
"MATRIX_ALLOWED_USERS": _env(
"Comma-separated Matrix user IDs allowed to use the bot (@user:server format)",
"Allowed Matrix user IDs (comma-separated)", url=None, password=False, category="messaging"),
"MATRIX_REQUIRE_MENTION": _env(
"Require @mention in Matrix rooms (default: true). Set to false to respond to all messages.",
"Require @mention in rooms (true/false)", url=None, password=False, category="messaging",
advanced=True),
"MATRIX_FREE_RESPONSE_ROOMS": _env("Comma-separated Matrix room IDs where bot responds without @mention",
"Free-response room IDs (comma-separated)", url=None, password=False, category="messaging",
advanced=True),
"MATRIX_AUTO_THREAD": _env("Auto-create threads for messages in Matrix rooms (default: true)",
"Auto-create threads in rooms (true/false)", url=None, password=False, category="messaging",
advanced=True),
"MATRIX_DM_AUTO_THREAD": _env("Auto-create threads for DM messages in Matrix (default: false)",
"Auto-create threads in DMs (true/false)", url=None, password=False, category="messaging",
advanced=True),
"MATRIX_DEVICE_ID": _env("Stable Matrix device ID for E2EE persistence across restarts (e.g. HERMES_BOT)",
"Matrix device ID (stable across restarts)", url=None, password=False, category="messaging",
advanced=True),
"MATRIX_RECOVERY_KEY": _env(
"Matrix recovery key for cross-signing verification after device key rotation (from Element: "
"Settings → Security → Recovery Key)", "Matrix recovery key", url=None, password=True,
category="messaging", advanced=True),
"BLUEBUBBLES_SERVER_URL": _env(
"BlueBubbles server URL for iMessage integration (e.g. http://192.168.1.10:1234)",
"BlueBubbles server URL", url="https://bluebubbles.app/", password=False, category="messaging"),
"BLUEBUBBLES_PASSWORD": _env("BlueBubbles server password (from BlueBubbles Server → Settings → API)",
"BlueBubbles server password", url=None, password=True, category="messaging"),
"BLUEBUBBLES_ALLOWED_USERS": _env(
"Comma-separated iMessage addresses (email or phone) allowed to use the bot",
"Allowed iMessage addresses (comma-separated)", url=None, password=False, category="messaging"),
"BLUEBUBBLES_ALLOW_ALL_USERS": _env("Allow all BlueBubbles users without allowlist",
"Allow All BlueBubbles Users", category="messaging"),
"QQ_APP_ID": _env("QQ Bot App ID from QQ Open Platform (q.qq.com)", "QQ App ID", url="https://q.qq.com",
category="messaging"),
"QQ_CLIENT_SECRET": _env("QQ Bot Client Secret from QQ Open Platform", "QQ Client Secret", password=True,
category="messaging"),
"QQ_ALLOWED_USERS": _env("Comma-separated QQ user IDs allowed to use the bot", "QQ Allowed Users",
category="messaging"),
"QQ_GROUP_ALLOWED_USERS": _env("Comma-separated QQ group IDs allowed to interact with the bot",
"QQ Group Allowed Users", category="messaging"),
"QQ_ALLOW_ALL_USERS": _env("Allow all QQ users without an allowlist (true/false)", "Allow All QQ Users",
category="messaging"),
"QQBOT_HOME_CHANNEL": _env("Default QQ channel/group for cron delivery and notifications",
"QQ Home Channel", category="messaging"),
"QQBOT_HOME_CHANNEL_NAME": _env("Display name for the QQ home channel", "QQ Home Channel Name",
category="messaging"),
"QQ_SANDBOX": _env("Enable QQ sandbox mode for development testing (true/false)", "QQ Sandbox Mode",
category="messaging"),
"IRC_SERVER": _env("IRC server hostname (e.g. irc.libera.chat)", "IRC server", url=None, password=False,
category="messaging"),
"IRC_CHANNEL": _env("IRC channel to join (e.g. #hermes)", "IRC channel", url=None, password=False,
category="messaging"),
"IRC_NICKNAME": _env("Bot nickname on IRC (default: hermes-bot)", "IRC nickname", url=None,
password=False, category="messaging"),
"IRC_SERVER_PASSWORD": _env("IRC server password (if required)", "IRC server password", url=None,
password=True, category="messaging", advanced=True),
"IRC_NICKSERV_PASSWORD": _env("NickServ password for nick identification", "NickServ password", url=None,
password=True, category="messaging", advanced=True),
"GATEWAY_ALLOW_ALL_USERS": _env(
"Allow all users to interact with messaging bots (true/false). Default: false.",
"Allow all users (true/false)", url=None, password=False, category="messaging", advanced=True),
"API_SERVER_ENABLED": _env(
"Enable the OpenAI-compatible API server (true/false). Allows frontends like Open WebUI, LobeChat, "
"etc. to connect.", "Enable API server (true/false)", url=None, password=False, category="messaging",
advanced=True),
"API_SERVER_KEY": _env(
"Bearer token for API server authentication. Required whenever the API server is enabled; server "
"refuses to start without it.", "API server auth key", url=None, password=True, category="messaging",
advanced=True),
"API_SERVER_PORT": _env("Port for the API server (default: 8642).", "API server port", url=None,
password=False, category="messaging", advanced=True),
"API_SERVER_HOST": _env(
"Host/bind address for the API server (default: 127.0.0.1). API_SERVER_KEY is still required even "
"on loopback binds.", "API server host", url=None, password=False, category="messaging",
advanced=True),
"API_SERVER_MODEL_NAME": _env(
"Model name advertised on /v1/models. Defaults to the profile name (or 'hermes-agent' for the "
"default profile). Useful for multi-user setups with OpenWebUI.", "API server model name", url=None,
password=False, category="messaging", advanced=True),
"GATEWAY_PROXY_URL": _env(
"URL of a remote Hermes API server to forward messages to (proxy mode). When set, the gateway "
"handles platform I/O only — all agent work is delegated to the remote server. Use for Docker E2EE "
"containers that relay to a host agent. Also configurable via gateway.proxy_url in config.yaml.",
"Remote Hermes API server URL (e.g. http://192.168.1.100:8642)", url=None, password=False,
category="messaging", advanced=True),
"GATEWAY_PROXY_KEY": _env(
"Bearer token for authenticating with the remote Hermes API server (proxy mode). Must match the "
"API_SERVER_KEY on the remote host.", "Remote API server auth key", url=None, password=True,
category="messaging", advanced=True),
"WEBHOOK_ENABLED": _env(
"Enable the webhook platform adapter for receiving events from GitHub, GitLab, etc.",
"Enable webhooks (true/false)", url=None, password=False, category="messaging"),
"WEBHOOK_PORT": _env("Port for the webhook HTTP server (default: 8644).", "Webhook port", url=None,
password=False, category="messaging"),
"WEBHOOK_SECRET": _env(
"Global HMAC secret for webhook signature validation (overridable per route in config.yaml).",
"Webhook secret", url=None, password=True, category="messaging"),
# ── Agent settings ── (MESSAGING_CWD is gone: use terminal.cwd in config.yaml, which the
# gateway bridges to TERMINAL_CWD.)
"SUDO_PASSWORD": _env(
"Sudo password for terminal commands requiring root access; set to an explicit empty string to try "
"empty without prompting", "Sudo password", url=None, password=True, category="setting"),
# HERMES_TOOL_PROGRESS_MODE (deprecated; use display.tool_progress) is intentionally NOT
# listed: this dict feeds user-facing surfaces (dashboard keys page, setup checklists), so
# deprecated knobs stay in config._EXTRA_ENV_KEYS only. HERMES_TOOL_PROGRESS is unsupported.
"HERMES_PREFILL_MESSAGES_FILE": _env(
"Path to JSON file with ephemeral prefill messages for few-shot priming",
"Prefill messages file path", url=None, password=False, category="setting"),
"HERMES_EPHEMERAL_SYSTEM_PROMPT": _env(
"Ephemeral system prompt injected at API-call time (never persisted to sessions)",
"Ephemeral system prompt", url=None, password=False, category="setting"),
}