Files
hermes-agent/hermes_cli/oneshot.py
Siddharth Balyan 3e00a356a4 fix(mcp): one reader for mcp_servers.<name>.enabled (#119567)
The `enabled` key had four parsers. The MCP client (`_parse_boolish`) read
`enabled: 0` as on; the toolset resolver and editor (`_parse_enabled_flag`)
read it as off. The server list (`summarize_server`, `/api/mcp/servers`) read
any non-`False` value as on, so `enabled: "false"` showed on while the agent
skipped it. The catalog and `hermes mcp list` accepted only true/1/yes, so
`enabled: on` showed off while the server ran.

`tools/mcp_tool_common.py::mcp_server_enabled` is now the only reader, and
every surface calls it. `_parse_boolish` treats YAML numbers by truthiness
(0 off, other numbers on). Everything else keeps the client's semantics:
the off words are off, absent / null / junk stay on, with the existing
warning for junk.

The desktop MCP page mirrors the rule in `serverEnabled`
(`apps/desktop/src/lib/mcp-servers.ts`). One case table
(`mcp-enabled-cases.json`) drives the Python invariant test and the vitest
test, so the page and the runtime cannot drift apart again.
2026-09-22 22:00:24 +00:00

654 lines
30 KiB
Python

"""Oneshot (-z) mode: send a prompt, get the final content block, exit.
Toolsets = explicit --toolsets, else the user's "cli" toolsets from `hermes tools`. Rules /
memory / AGENTS.md / preloaded skills = same as a normal chat turn. Approvals are auto-bypassed
(HERMES_YOLO_MODE=1). Model/provider mirror `hermes chat`: both optional; only --model → auto-detect
the provider; only --provider → error (ambiguous).
"""
from __future__ import annotations
import json
import logging
import os
import sys
from contextlib import redirect_stderr, redirect_stdout
import dataclasses
from dataclasses import dataclass
from pathlib import Path
from typing import Optional
from gateway.session_context import declare_stateless_channel
from hermes_cli.fallback_config import get_fallback_chain
_ALL_TOOLSETS = {"all", "*"}
# Keys copied from the run result into the ``--usage-file`` report. ``service_tier`` is a
# billing-audit field: the tier REQUESTED via request_overrides.extra_body (None when unset), so
# batch pipelines can verify the tier they pay for went out on the wire. ``partial`` /
# ``interrupted`` / ``turn_exit_reason`` say WHY ``completed`` is false, so a pipeline can tell
# an iteration-budget stop from a Ctrl-C without parsing stderr (#111770).
_USAGE_KEYS = (
"estimated_cost_usd", "cost_status", "cost_source", "input_tokens", "output_tokens",
"cache_read_tokens", "cache_write_tokens", "reasoning_tokens", "total_tokens", "api_calls",
"model", "provider", "session_id", "completed", "partial", "interrupted", "turn_exit_reason",
)
# Counters summed per auxiliary task (vision, compression, title_generation, ...) into the
# ``auxiliary`` block of the report. The main-loop keys above stay main-loop-only (backward
# compatible); ``total_including_auxiliary`` carries the grand total pipelines bill on (#112848).
_AUX_COUNTERS = (
"api_calls", "input_tokens", "output_tokens", "cache_read_tokens", "cache_write_tokens",
"reasoning_tokens", "estimated_cost_usd",
)
def _auxiliary_usage(session_db, session_id: Optional[str]) -> dict[str, dict]:
"""Per-task aux usage recorded for *session_id*'s lineage (``{}`` without a store / session)."""
if session_db is None or not session_id:
return {}
try:
return session_db.auxiliary_usage_by_task(session_id)
except Exception:
logging.debug("oneshot: auxiliary usage read failed", exc_info=True)
return {}
def _attach_auxiliary_usage(result: dict, session_db, before: dict[str, dict],
fallback_session_id: Optional[str] = None) -> None:
"""Store this run's auxiliary usage on *result* as the delta against the pre-turn snapshot
(a resumed session already carries earlier runs' aux rows). Waits (bounded) for the auto-title
thread first: it bills from a daemon thread and can still be in flight when the turn returns.
Failed-turn dicts carry no ``session_id``; *fallback_session_id* keeps the delta readable then."""
from agent.title_generator import wait_for_title_upgrades
wait_for_title_upgrades()
after = _auxiliary_usage(session_db, result.get("session_id") or fallback_session_id)
by_task: dict[str, dict] = {}
for task, counters in after.items():
prior = before.get(task, {})
delta = {key: (counters.get(key) or 0) - (prior.get(key) or 0) for key in _AUX_COUNTERS}
if any(delta.values()):
by_task[task] = delta
result["auxiliary_usage"] = by_task
def _auxiliary_report(report: dict, by_task: dict[str, dict]) -> None:
"""Add the ``auxiliary`` breakdown and ``total_including_auxiliary`` to the ledger."""
totals = {key: sum(t.get(key) or 0 for t in by_task.values()) for key in _AUX_COUNTERS}
totals["total_tokens"] = totals["input_tokens"] + totals["output_tokens"]
report["auxiliary"] = {**totals, "by_task": by_task}
main_cost = report.get("estimated_cost_usd")
report["total_including_auxiliary"] = {
"estimated_cost_usd": None if main_cost is None else main_cost + totals["estimated_cost_usd"],
"total_tokens": (report.get("total_tokens") or 0) + totals["total_tokens"],
"api_calls": (report.get("api_calls") or 0) + totals["api_calls"],
}
# Exit code for a turn stopped by an interrupt (SIGINT convention, same as ``chat -Q``).
_INTERRUPTED_EXIT_CODE = 130
def _oneshot_exit_code(response: Optional[str], result: dict) -> int:
"""Map a finished ``-z`` turn onto its exit code: ``0`` only when the turn completed;
``130`` interrupted; ``2`` failed or stopped partway (``partial``, ``completed: False`` such
as the iteration budget); ``1`` a completed turn that produced no text at all.
The outcome is judged from the result, not from whether text was printed: a partial or
failed turn usually leaves an explanation on stdout, and exiting 0 for it made scripts treat
a half-done job (or a provider error summary) as success (#111770).
"""
if result.get("interrupted"):
return _INTERRUPTED_EXIT_CODE
if result.get("failed") or result.get("partial") or result.get("completed") is False:
return 2
if not (response or "").strip():
return 1
return 0
def _normalize_toolsets(toolsets: object = None) -> list[str] | None:
"""Split repeated/comma-separated toolset flags into a clean list (``None`` when empty)."""
if not toolsets:
return None
items = toolsets if isinstance(toolsets, (list, tuple)) else [toolsets]
parts = [str(item).split(",") if isinstance(item, str) else [str(item)] for item in items]
return [p.strip() for chunk in parts for p in chunk if p.strip()] or None
def _normalize_skills(skills: object = None) -> list[str]:
"""Normalize repeated/comma-separated skill flags and preserve order."""
return list(dict.fromkeys(_normalize_toolsets(skills) or []))
def _build_preloaded_skills_prompt(skills: object = None) -> str | None:
"""Load requested skills using the same partial-success contract as CLI chat."""
parsed_skills = _normalize_skills(skills)
if not parsed_skills:
return None
from agent.skill_commands import build_preloaded_skills_prompt
skills_prompt, loaded_skills, missing_skills = build_preloaded_skills_prompt(parsed_skills)
if missing_skills:
missing_display = ", ".join(missing_skills)
if not loaded_skills:
raise ValueError(f"Unknown skill(s): {missing_display}")
logging.warning(
"Unknown skill(s) requested, skipping: %s. Continuing with: %s. "
"List available skills with `hermes skills list`.",
missing_display,
", ".join(loaded_skills),
)
return skills_prompt or None
def _configured_mcp_servers() -> tuple[set[str], set[str]]:
"""``(enabled, disabled)`` MCP server names from config; both empty on any error."""
try:
from hermes_cli.config import read_raw_config
from tools.mcp_tool_common import mcp_server_enabled
cfg = read_raw_config()
mcp_servers = cfg.get("mcp_servers") if isinstance(cfg.get("mcp_servers"), dict) else {}
enabled: set[str] = set()
disabled: set[str] = set()
for name, server_cfg in mcp_servers.items():
if not isinstance(server_cfg, dict):
continue
target = enabled if mcp_server_enabled(server_cfg) else disabled
target.add(str(name))
return enabled, disabled
except Exception:
return set(), set()
def _validate_explicit_toolsets(toolsets: object = None) -> tuple[list[str] | None, str | None]:
normalized = _normalize_toolsets(toolsets)
if normalized is None:
return None, None
try:
from toolsets import validate_toolset
except Exception as exc:
return None, f"hermes -z: failed to validate --toolsets: {exc}\n"
built_in = [name for name in normalized if validate_toolset(name)]
unresolved = [name for name in normalized if name not in built_in]
if unresolved:
try:
from hermes_cli.plugins import discover_plugins
discover_plugins()
plugin_valid = [name for name in unresolved if validate_toolset(name)]
except Exception:
plugin_valid = []
built_in.extend(plugin_valid)
unresolved = [name for name in unresolved if name not in plugin_valid]
if any(name in _ALL_TOOLSETS for name in built_in):
ignored = [name for name in normalized if name not in _ALL_TOOLSETS]
if ignored:
sys.stderr.write(
"hermes -z: --toolsets all enables every toolset; "
f"ignoring additional entries: {', '.join(ignored)}\n"
)
return None, None
mcp_names, mcp_disabled = _configured_mcp_servers() if unresolved else (set(), set())
mcp_valid = [name for name in unresolved if name in mcp_names]
disabled = [name for name in unresolved if name in mcp_disabled]
unknown = [name for name in unresolved if name not in mcp_names and name not in mcp_disabled]
valid = built_in + mcp_valid
if unknown:
sys.stderr.write(f"hermes -z: ignoring unknown --toolsets entries: {', '.join(unknown)}\n")
if disabled:
sys.stderr.write(
"hermes -z: ignoring disabled MCP servers (set enabled: true in config.yaml to use): "
f"{', '.join(disabled)}\n"
)
if not valid:
return None, "hermes -z: --toolsets did not contain any valid toolsets.\n"
return valid, None
def _write_usage_file(path: Optional[str], result: dict, failure: Optional[str] = None) -> None:
"""Best-effort JSON usage report for pipelines (``-z --usage-file``).
Written even on failure so callers can always account for spend. Never raises — a broken usage
write must not mask the run's own outcome.
"""
if not path:
return
try:
report = {key: result.get(key) for key in _USAGE_KEYS}
report["failed"] = bool(result.get("failed")) or failure is not None
report["service_tier"] = result.get("service_tier")
if isinstance(result.get("auxiliary_usage"), dict):
_auxiliary_report(report, result["auxiliary_usage"])
if failure is not None:
report["failure"] = failure
out = Path(path).expanduser()
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
except Exception:
pass
def run_oneshot(
prompt: str,
model: Optional[str] = None,
provider: Optional[str] = None,
toolsets: object = None,
skills: object = None,
usage_file: Optional[str] = None,
resume: Optional[str] = None,
reasoning: object = None,
) -> int:
"""Execute a single prompt and print only the final content block.
Model/provider fall back to ``HERMES_INFERENCE_MODEL`` and config.yaml. ``usage_file`` gets a
JSON usage report even when the run fails. ``resume`` is a session id (already normalized by
the CLI layer: latest/title/--continue resolution) whose transcript is loaded and continued
by this turn. Returns the exit code; the caller owns process termination.
"""
# Silence every stdlib logger: AIAgent, tools and provider adapters log to stderr through the
# root logger. File handlers from setup_logging() keep working (level-independent).
logging.disable(logging.CRITICAL)
# --provider without --model is ambiguous (the provider may not host the configured model, and
# picking its catalog default hides the mismatch). Validate BEFORE the stderr redirect.
env_model_early = os.getenv("HERMES_INFERENCE_MODEL", "").strip()
if provider and not ((model or "").strip() or env_model_early):
sys.stderr.write(
"hermes -z: --provider requires --model (or HERMES_INFERENCE_MODEL). "
"Pass both explicitly, or neither to use your configured defaults.\n"
)
return 2
explicit_toolsets, toolsets_error = _validate_explicit_toolsets(toolsets)
if toolsets_error:
sys.stderr.write(toolsets_error)
return 2
use_config_toolsets = _normalize_toolsets(toolsets) is None
# Non-interactive by definition — an approval prompt would hang forever.
os.environ["HERMES_YOLO_MODE"] = "1"
os.environ["HERMES_ACCEPT_HOOKS"] = "1"
# Same finite-chat marker as `hermes chat -q` (cli.py): the session-source resolver uses it to drop an
# inherited tui/desktop transport label, and delegate dispatch to route detached results inline.
os.environ["HERMES_SINGLE_QUERY_SESSION"] = "1"
# Nothing here drains process_registry.completion_queue (only cli.py's process_loop and the
# gateway watchers do), so left unbound delegate_task would be forced background and every
# subagent result discarded. Stateless routes it to the inline/synchronous path.
declare_stateless_channel()
# Redirect stderr AND stdout for the entire call tree; the final response goes to the real
# stdout at the end.
real_stdout = sys.stdout
real_stderr = sys.stderr
response: Optional[str] = None
result: dict = {}
failure: BaseException | None = None
with open(os.devnull, "w", encoding="utf-8") as devnull, redirect_stdout(devnull), redirect_stderr(devnull):
try:
response, result = _run_agent(
prompt,
model=model,
provider=provider,
toolsets=explicit_toolsets,
use_config_toolsets=use_config_toolsets,
skills=skills,
resume=resume,
reasoning=reasoning,
ledger=bool(usage_file),
)
except BaseException as exc: # noqa: BLE001
# Capture anything escaping the agent (OSError from prompt_toolkit on a non-TTY pipe,
# KeyboardInterrupt, SystemExit, ...) so it reaches the real stderr instead of dying
# silently past the redirect — the worst failure mode in cron / SSH / subprocess use.
failure = exc
if failure is not None:
# Control-flow exceptions (Ctrl-C / sys.exit inside the agent) re-raise to the parent.
if isinstance(failure, (KeyboardInterrupt, SystemExit)):
_write_usage_file(usage_file, result, failure=repr(failure))
raise failure
_write_usage_file(usage_file, result, failure=str(failure))
real_stderr.write(f"hermes -z: agent failed: {failure}\n")
real_stderr.flush()
return 1
_write_usage_file(usage_file, result)
if response:
# Lone UTF-16 surrogates would raise UnicodeEncodeError on a real stdout and abort with
# exit 1 after the turn already completed — scrub to U+FFFD first.
# Model text can contain lone UTF-16 surrogates (invalid in UTF-8). See #80366.
from agent.message_sanitization import _sanitize_surrogates
response = _sanitize_surrogates(response)
real_stdout.write(response)
if not response.endswith("\n"):
real_stdout.write("\n")
real_stdout.flush()
exit_code = _oneshot_exit_code(response, result)
if exit_code == 1:
real_stderr.write("hermes -z: no final response was produced; treating the run as failed.\n")
real_stderr.flush()
return exit_code
def _create_session_db_for_oneshot():
"""Best-effort SessionDB — oneshot bypasses ``HermesCLI._init_agent()``, so it must wire the
SQLite store itself or ``session_search`` is advertised but always unavailable. The registry
handle is the one in-process tools (delegation, goals) acquire during the run, so the process
holds one writer; ``_close_agent``'s ``close()`` releases the refcount."""
try:
from hermes_state_registry import acquire
return acquire()
except Exception as exc:
logging.debug("SQLite session store not available for oneshot mode: %s", exc)
return None
@dataclass
class _ModelChoice:
model: str
provider: str | None
base_url: str | None = None
api_key: str | None = None
api_mode: str | None = None
def _configured_model(model_cfg: object) -> str:
if isinstance(model_cfg, str):
return model_cfg
raw = model_cfg.get("default") or model_cfg.get("model") or ""
if isinstance(raw, dict):
from hermes_cli.config import split_model_config_default
return split_model_config_default(raw)[0]
return str(raw or "")
def _resolve_model_and_provider(cfg: dict, model: Optional[str], provider: Optional[str]) -> _ModelChoice:
"""Effective model = arg → env → config; provider = arg → auto-detect → config/env.
Auto-detection only runs when the model was explicitly requested (arg or env var) — same
semantic as ``/model <name>`` — because the configured default provider may not host it.
Config-sourced models are the "use my defaults" path and keep the configured provider.
"""
from hermes_cli.models import detect_provider_for_model
model_cfg = cfg.get("model") or {}
env_model = os.getenv("HERMES_INFERENCE_MODEL", "").strip()
explicit_model = (model or "").strip() or env_model
choice = _ModelChoice(explicit_model or _configured_model(model_cfg), (provider or "").strip() or None)
if choice.provider is not None or not explicit_model:
return choice
# DIRECT_ALIASES (config.yaml ``model_aliases:``) map a user alias to (model, provider,
# base_url) for endpoints outside any catalog (local servers, custom proxies, ...).
from hermes_cli import model_switch as _ms
try:
_ms._ensure_direct_aliases()
direct = _ms.DIRECT_ALIASES.get(explicit_model.strip().lower())
except Exception:
direct = None
if direct is None:
cfg_provider = ""
if isinstance(model_cfg, dict):
cfg_provider = str(model_cfg.get("provider") or "").strip().lower()
current_provider = cfg_provider or os.getenv("HERMES_INFERENCE_PROVIDER", "").strip().lower() or "auto"
# Same owner as HermesCLI startup: a provider-qualified string (``custom:<name>:<model>``,
# ``<provider>/<model>``) selects that provider before auto-detection can hand the unsplit
# string to the configured default (#73943).
route = _ms.resolve_startup_model_route(
explicit_model, current_provider=current_provider,
user_providers=cfg.get("providers"), custom_providers=cfg.get("custom_providers"))
if route is not None:
choice.provider, choice.model = route.provider, route.model
return choice
detected = detect_provider_for_model(explicit_model, current_provider)
if detected:
choice.provider, choice.model = detected
return choice
choice.model = direct.model
choice.provider = direct.provider
# Resolve through the SAME owner the interactive `/model` path uses: passing `direct.provider`
# with a URL-bearing alias would let a label like `anthropic` keep the alias's base_url yet
# fall back to the live vendor token — a bearer credential crossing an origin boundary. The
# helper forces bare `custom` for URL-bearing aliases and carries the alias's own key.
try:
choice.provider, choice.api_key = _ms.direct_alias_runtime_request(direct)
except Exception:
choice.api_key = None
if direct.base_url:
choice.base_url = direct.base_url.rstrip("/")
return choice
def _load_resume_target(session_db, resume: Optional[str]) -> tuple[Optional[str], list, Optional[dict]]:
"""Resolve ``resume`` to ``(session_id, conversation_history, session_meta)`` for a oneshot turn.
Follows the same contract as the interactive CLI resume: compression-chain redirect via
``resolve_resume_session_id``, safe-resume guard, model-projection history with
``session_meta`` rows dropped. An unknown session raises (the user passed an explicit id;
silently starting a fresh session is the resume-dropped failure mode this exists to fix —
see #105892). An empty stored transcript still returns the resolved id: the turn replays
nothing but is recorded under the requested session — ``hermes -z "hello" -c <title>
--create-if-missing`` must fill the titled session it created, not mint a fresh id
(same contract as the interactive /resume of an empty session).
The resolved row is also reopened (best effort): the previous run stamped ``ended_at``,
the existing-row upsert never clears the end fields, and ``end_session()`` only writes
rows whose ``ended_at`` is null — so without this step the resumed turn would be recorded
under a session that stays closed and its new lifecycle boundary would be lost (same
reason the interactive resume calls ``reopen_session()`` before continuing).
"""
if not resume:
return None, [], None
if session_db is None:
raise RuntimeError(f"cannot resume session {resume}: session store unavailable")
resolved = session_db.resolve_resume_session_id(resume) or resume
session_meta = session_db.get_session(resolved)
if not session_meta:
raise RuntimeError(f"session not found: {resume}")
session_db.assert_resume_safe(resolved, tip_only=True)
restored, _display = session_db.get_resume_conversations(resolved)
history = [m for m in restored if m.get("role") != "session_meta"]
try:
session_db.reopen_session(resolved)
except Exception:
logging.debug("reopen_session failed for resumed one-shot session %s", resolved, exc_info=True)
return resolved, history, session_meta
def _apply_stored_session_runtime(
choice: _ModelChoice, session_meta: Optional[dict], *, explicit_model: bool,
) -> _ModelChoice:
"""Run a resumed one-shot on the session's stored runtime, not the ambient config — the same
contract as the interactive ``_restore_session_model``, via the shared ``stored_session_route``.
An explicit ``--model`` keeps the ambient choice. A changed provider drops the resolved
``api_key``: it belongs to the ambient endpoint and is never persisted, so runtime resolution
re-fetches credentials for the restored provider."""
if explicit_model:
return choice
from hermes_cli.cli_model_switch_mixin import stored_session_route
route = stored_session_route(session_meta, current_model=choice.model, current_provider=choice.provider)
if route is None:
return choice
choice.model, stored_provider, stored_base_url, stored_api_mode, provider_changed = route
if provider_changed:
choice.provider = stored_provider
choice.base_url = stored_base_url
choice.api_key = None
if stored_api_mode:
choice.api_mode = str(stored_api_mode)
return choice
def _run_agent(
prompt: str,
model: Optional[str] = None,
provider: Optional[str] = None,
toolsets: object = None,
use_config_toolsets: bool = True,
skills: object = None,
resume: Optional[str] = None,
reasoning: object = None,
ledger: bool = False,
) -> tuple[str, dict]:
"""Build an AIAgent exactly like a normal CLI chat turn, run one conversation, and return
``(final_response, run_result)``. Imports are local to keep CLI startup cheap. *ledger* (set when
``--usage-file`` is requested) attaches this run's auxiliary usage to the result."""
from hermes_cli.config import load_config
from hermes_cli.runtime_provider import resolve_runtime_with_fallback
from hermes_cli.tools_config import _get_platform_tools
from run_agent import AIAgent
cfg = load_config()
choice = _resolve_model_and_provider(cfg, model, provider)
# Resume resolves BEFORE the runtime provider: the session's stored model/route must
# replace the ambient config (see _apply_stored_session_runtime) and the ended row must
# be reopened before the agent can stamp a new lifecycle boundary.
session_db = _create_session_db_for_oneshot()
resume_sid, conversation_history, resume_meta = _load_resume_target(session_db, resume)
choice = _apply_stored_session_runtime(choice, resume_meta, explicit_model=bool((model or "").strip()))
# Resolution-time fallback (#81209): a quota-exhausted/expired primary raises AuthError here, before
# AIAgent (and its mid-session ``fallback_model`` wiring) exists, so walk the chain like the gateway.
runtime, fallback_entry = resolve_runtime_with_fallback(
cfg,
requested=choice.provider,
target_model=choice.model or None,
explicit_base_url=choice.base_url,
explicit_api_key=choice.api_key,
)
if fallback_entry is not None:
# The chosen entry names the model that will be sent; the primary's stored api_mode no longer applies.
choice = dataclasses.replace(choice, model=fallback_entry["model"], provider=runtime.get("provider"),
api_mode=None)
if choice.api_mode:
runtime["api_mode"] = choice.api_mode
from hermes_constants import parse_reasoning_effort, resolve_reasoning_config
reasoning_config = resolve_reasoning_config(cfg, choice.model)
if reasoning is not None and str(reasoning).strip():
parsed_reasoning = parse_reasoning_effort(reasoning)
if parsed_reasoning is None:
logging.warning("Unknown --reasoning '%s', keeping the configured level", reasoning)
else:
reasoning_config = parsed_reasoning
# sorted() gives stable ordering for config-derived sets; explicit values preserve user order.
toolsets_list = _normalize_toolsets(toolsets)
if toolsets_list is None and use_config_toolsets:
toolsets_list = sorted(_get_platform_tools(cfg, "cli"))
# Oneshot builds AIAgent directly, bypassing cli.py's MCP background discovery and
# _init_agent's wait, so the construction-time tool snapshot would miss late MCP servers.
# Idempotent start + bounded wait with the single-query bound (there is no later turn).
# Ensure MCP tools are discovered before building the agent. This helper starts discovery if needed
# (idempotent) and bounded-waits with the larger single-query bound (default 15s) because there is only
# ONE turn and no between-turns late-binding refresh (#38448).
from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build
ensure_mcp_discovery_before_agent_build(logger=logging.getLogger(__name__), single_query=True)
skills_prompt = _build_preloaded_skills_prompt(skills)
# The try spans agent construction (not just ``chat``) so the store is always closed, even when
# ``AIAgent(...)`` raises — the one-shot exit path hard-exits via os._exit and skips finalizers.
agent = None
try:
agent = AIAgent(
api_key=runtime.get("api_key"),
base_url=runtime.get("base_url"),
provider=runtime.get("provider"),
requested_provider=runtime.get("requested_provider"),
api_mode=runtime.get("api_mode"),
model=choice.model,
enabled_toolsets=toolsets_list,
quiet_mode=True,
platform="cli",
session_db=session_db,
session_id=resume_sid,
credential_pool=runtime.get("credential_pool"),
fallback_model=get_fallback_chain(cfg) or None,
ephemeral_system_prompt=skills_prompt,
reasoning_config=reasoning_config,
# The only interactive callback wired: no user sits at a terminal. Sudo prompts gate on
# HERMES_INTERACTIVE (never set), hook approval via HERMES_ACCEPT_HOOKS=1, dangerous
# commands via HERMES_YOLO_MODE=1, skill secret capture degrades gracefully.
clarify_callback=_oneshot_clarify_callback,
)
# Belt-and-braces: no streaming display callbacks may bypass our stdout capture.
agent.suppress_status_output = True
agent.stream_delta_callback = None
agent.tool_gen_callback = None
aux_before = _auxiliary_usage(session_db, resume_sid) if ledger else {}
result = agent.run_conversation(prompt, conversation_history=conversation_history or None)
if ledger:
_attach_auxiliary_usage(result, session_db, aux_before,
fallback_session_id=agent.session_id or resume_sid)
return (result.get("final_response") or "", result)
finally:
_close_agent(agent, session_db)
def _quietly(what: str, fn) -> None:
"""Run a cleanup step, logging (never raising) on failure."""
try:
fn()
except Exception:
logging.debug("oneshot %s failed", what, exc_info=True)
def _linger_for_background_completions() -> None:
# Linger (bounded) for background processes this turn spawned with notify_on_complete=true BEFORE
# agent.close(): close() calls process_registry.kill_all(task_id) and the dying parent owns the
# children's stdout pipes, so exiting now destroys in-flight deliveries — including Bot Mode handoff
# replies dispatched from a short-lived recipient (#90879).
from tools.process_registry import process_registry
process_registry.wait_for_pending_completions(None)
def _close_agent(agent, session_db) -> None:
"""Teardown mirroring gateway/run.py:_cleanup_agent_resources (NOT cli.py:_run_cleanup):
oneshot has no _active_agent_ref and the hard-exit path skips finalizers."""
if agent is not None:
# Linger (bounded) for notify_on_complete background processes BEFORE agent.close():
# close() kill_all()s the task and the dying parent owns the children's stdout pipes, so
# exiting now destroys in-flight deliveries (e.g. Bot Mode handoff replies).
_quietly("background completion wait", _linger_for_background_completions)
session_messages = getattr(agent, "_session_messages", None)
memory_args = (session_messages,) if isinstance(session_messages, list) else ()
_quietly("memory/context cleanup", lambda: agent.shutdown_memory_provider(*memory_args))
_quietly("agent cleanup", lambda: agent.close())
# agent.close() ends the session but leaves the connection open; close it to checkpoint the WAL.
if session_db is not None:
_quietly("session store cleanup", lambda: session_db.close())
def _oneshot_clarify_callback(question: str, choices=None, multi_select=False) -> str:
"""Clarify is disabled in oneshot mode — tell the agent to pick a default and proceed."""
if choices:
what = "subset" if multi_select else "option"
return (
f"[oneshot mode: no user available. Pick the best {what} from "
f"{choices} using your own judgment and continue.]"
)
return "[oneshot mode: no user available. Make the most reasonable assumption you can and continue.]"