diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index 7f9f4dba2f..568c897352 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -406,10 +406,18 @@ def _ensure_codex_session(agent) -> None: # _emit_interim_assistant_message). Without this, Discord/Telegram users see no live tool-progress or # interim commentary while codex_app_server is running — only the final answer (#33200). Supersedes the # narrower item/started-only bridge from #38835. + # A named custom provider (``providers.``) maps onto codex's own ``[model_providers.]`` + # table: send the stable id plus the active model and let codex resolve base_url/env_key itself, so + # Hermes' credential never enters the JSON-RPC payload (#75186). openai/openai-codex keep codex's defaults. + model_provider = None + if str(getattr(agent, "provider", "") or "").strip().lower() == "custom": + from hermes_cli.runtime_provider_custom import codex_model_provider_id + model_provider = codex_model_provider_id(str(getattr(agent, "requested_provider", "") or "")) agent._codex_session = CodexAppServerSession( cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), approval_callback=approval_callback, request_routing=_ServerRequestRouting(auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests), on_event=make_codex_app_server_event_bridge(agent), + model=getattr(agent, "model", None) if model_provider else None, model_provider=model_provider, ) diff --git a/agent/transports/codex_app_server_session.py b/agent/transports/codex_app_server_session.py index ede339a800..e82f9453d6 100644 --- a/agent/transports/codex_app_server_session.py +++ b/agent/transports/codex_app_server_session.py @@ -156,10 +156,15 @@ class CodexAppServerSession: on_event: Optional[Callable[[dict], None]] = None, request_routing: Optional[_ServerRequestRouting] = None, client_factory: Optional[Callable[..., CodexAppServerClient]] = None, + model: Optional[str] = None, model_provider: Optional[str] = None, ) -> None: self._cwd = cwd or os.getcwd() self._codex_bin = codex_bin self._codex_home = codex_home + # ``thread/start.model`` / ``.modelProvider``: select a provider from codex's own + # ``[model_providers.]`` table. Only the id travels; codex reads base_url/env_key itself. + self._model = (model or "").strip() or None + self._model_provider = (model_provider or "").strip() or None self._permission_profile = permission_profile or _HERMES_TO_CODEX_PERMISSION_PROFILE.get( os.environ.get("HERMES_TERMINAL_SECURITY_MODE", "auto"), "workspace-write" ) @@ -187,7 +192,12 @@ class CodexAppServerSession: self._client.initialize(client_name="hermes", client_title="Hermes Agent", client_version=_get_hermes_version()) # Permissions are NOT sent on thread/start: codex gates ``thread/start.permissions`` # behind experimentalApi + a matching ``[permissions]`` table in ~/.codex/config.toml. - result = self._client.request("thread/start", {"cwd": self._cwd}, timeout=15) + params: dict[str, Any] = {"cwd": self._cwd} + if self._model_provider: + params["modelProvider"] = self._model_provider + if self._model: + params["model"] = self._model + result = self._client.request("thread/start", params, timeout=15) # Different codex versions serialize the id under thread.id / sessionId / threadId. thread_obj = result.get("thread") or {} thread_id = thread_obj.get("id") or thread_obj.get("sessionId") or result.get("sessionId") or result.get("threadId") diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 68da8dec16..4ef290ad91 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -261,10 +261,15 @@ def _api_key_provider_api_mode(provider: str, model_cfg: Dict[str, Any], api_key return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=opencode_by_model) -def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model_cfg: Optional[Dict[str, Any]]) -> str: - """Opt-in rewrite to "codex_app_server" via ``model.openai_runtime``; only ``openai`` / - ``openai-codex`` are eligible. No-op when unset, "auto", or empty.""" - if model_cfg and provider in {"openai", "openai-codex"} and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server": +def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model_cfg: Optional[Dict[str, Any]], + requested_provider: str = "") -> str: + """Opt-in rewrite to "codex_app_server" via ``model.openai_runtime``. Eligible: ``openai`` / + ``openai-codex``, and a configured named custom provider (``providers.``) whose id codex + looks up in its own ``[model_providers.]`` table (#75186). Anonymous ``custom`` has no + stable id and stays ineligible. No-op when unset, "auto", or empty.""" + if not model_cfg or str(model_cfg.get("openai_runtime") or "").strip().lower() != "codex_app_server": + return api_mode + if provider in {"openai", "openai-codex"} or (provider == "custom" and codex_model_provider_id(requested_provider)): return "codex_app_server" return api_mode @@ -455,7 +460,7 @@ from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401 _LLAMACPP_ALIASES, _apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity, _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime, - _try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity, + _try_resolve_from_custom_pool, canonical_custom_identity, codex_model_provider_id, find_custom_provider_identity, find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider, ) from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 @@ -887,6 +892,18 @@ def _tag(runtime: Optional[Dict[str, Any]], requested_provider: str) -> Optional return runtime +def _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model) -> Optional[Dict[str, Any]]: + """Rung 3: a configured named custom provider. Honours the ``model.openai_runtime`` opt-in like the + pool path does for openai/openai-codex (codex resolves the provider from its own config by id).""" + runtime = _tag(_resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, + explicit_base_url=explicit_base_url, target_model=target_model), requested_provider) + if runtime and runtime.get("provider") == "custom": + runtime["api_mode"] = _maybe_apply_codex_app_server_runtime( + provider="custom", api_mode=runtime.get("api_mode") or "chat_completions", model_cfg=_get_model_config(), + requested_provider=requested_provider) + return runtime + + def _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) -> Dict[str, Any]: return _tag(_resolve_openrouter_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url), requested_provider) @@ -940,8 +957,7 @@ def _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, targe """Ladder rungs 2-8, yielded lazily so each is evaluated only when the previous one returned nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" yield _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) - yield _tag(_resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, - explicit_base_url=explicit_base_url, target_model=target_model), requested_provider) + yield _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model) # If provider is "auto" (or unset) but config.yaml has an explicit base_url pointing at a custom/local # endpoint (e.g. Ollama at localhost:11434), route through the OpenAI-compatible resolver instead of # letting resolve_provider() pick up an ANTHROPIC_API_KEY or OPENAI_API_KEY from the environment and diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 9b4219d7e2..f9b7151cb0 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -188,6 +188,22 @@ def has_named_custom_provider(requested_provider: str) -> bool: return False +def codex_model_provider_id(requested_provider: str) -> Optional[str]: + """Codex ``[model_providers.]`` key for a configured named custom provider — its ``custom:`` + identity without the prefix (the ``providers:`` config key; legacy ``custom_providers:`` entries + use their normalized display name). None for bare ``custom``, aliases that resolve to custom + (ollama, vllm, …) and unknown names: codex has no stable id to look up for those (#75186).""" + if _normalize_custom_provider_name(requested_provider or "") in {"", "custom"}: + return None + try: + entry = _rp()._get_named_custom_provider(requested_provider) + except Exception: + return None + if not entry: + return None + return custom_provider_slug(str(entry.get("name") or ""), str(entry.get("provider_key") or "")).split(":", 1)[1] or None + + # ── identity recovery (bare "custom" -> durable ``custom:``) ───────────────────────── diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 40bf6bcb17..09a1c9046e 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2344,6 +2344,7 @@ def _make_agent( session = _sessions.get(sid) agent = AIAgent( model=model, max_iterations=_cfg_max_turns(cfg, 500), provider=runtime.get("provider"), + requested_provider=runtime.get("requested_provider"), base_url=runtime.get("base_url"), api_key=runtime.get("api_key"), api_mode=runtime.get("api_mode"), acp_command=runtime.get("command"), acp_args=runtime.get("args"), credential_pool=runtime.get("credential_pool"), quiet_mode=True, diff --git a/website/docs/reference/slash-commands.md b/website/docs/reference/slash-commands.md index 1ecdcf5ab9..a3d351e394 100644 --- a/website/docs/reference/slash-commands.md +++ b/website/docs/reference/slash-commands.md @@ -78,7 +78,7 @@ Type `/` in the CLI to open the autocomplete menu. Built-in commands are case-in |---------|-------------| | `/config` | Show current configuration | | `/model [model-name]` | Show or change the current model. Supports: `/model claude-sonnet-4`, `/model provider:model` (switch providers), `/model custom:model` (custom endpoint), `/model custom:name:model` (named custom provider), `/model custom` (auto-detect from endpoint), OpenRouter account presets (`/model @preset/` or `/model @preset/` — presets are account-scoped, so they skip the public model-listing check), and user-defined aliases (`/model fav`, `/model grok` — see [Custom model aliases](#custom-model-aliases)). Flags: `--global` persists the change to config.yaml; `--session` forces session-only; `--once` applies to the next turn only; `--refresh` re-fetches the provider's model list; `--provider ` switches backend (session-only unless `--global`); `--reasoning ` sets the reasoning effort (`none`, `minimal` … `ultra`) in the same step and with the same scope as the pick. A plain `/model ` is session-only unless `model.persist_switch_by_default: true` is set — except when no `model.default`/`model.provider` is configured yet, in which case the first pick persists so the profile gets a real default. The same rule governs the desktop composer picker. **Interactive picker:** running `/model` with no arguments opens the provider→model picker; on the model list you can **type to fuzzy-filter** the models (e.g. type `grok` to narrow to matching models), Backspace to trim the filter, Esc to clear it (or close the picker). Selection always resolves to one concrete model — the filter only narrows the list, it never guesses. After the model, a third step offers the reasoning effort for that model (or **Keep current effort**); it is skipped when the catalog says the route has no reasoning control. **Note:** `/model` can only switch between already-configured providers. To add a new provider, exit the session and run `hermes model` from your terminal. **Cost note:** switching models mid-conversation resets the prompt cache — the cache key includes the model, so your next turn re-reads the entire conversation at full input price instead of the ~75%-discounted cached rate. Expected and unavoidable, but worth knowing on long sessions. | -| `/codex-runtime [auto\|codex_app_server\|on\|off]` | Toggle the optional [Codex app-server runtime](../user-guide/features/codex-app-server-runtime) for OpenAI/Codex models. `auto` (default) uses Hermes' standard chat completions; `codex_app_server` hands turns to a `codex app-server` subprocess for native shell, apply_patch, ChatGPT subscription auth, and migrated Codex plugins. Effective on next session. | +| `/codex-runtime [auto\|codex_app_server\|on\|off]` | Toggle the optional [Codex app-server runtime](../user-guide/features/codex-app-server-runtime) for OpenAI/Codex models and named custom providers that are also defined in `~/.codex/config.toml`. `auto` (default) uses Hermes' standard chat completions; `codex_app_server` hands eligible turns to a `codex app-server` subprocess for native shell, apply_patch, ChatGPT subscription auth, and migrated Codex plugins. Effective on next session. | | `/personality` | Set a predefined personality. `/personality none` (or `default` / `neutral`) clears the overlay and returns to base behavior. | | `/verbose` | Cycle tool progress display: off → new → all → verbose. Can be [enabled for messaging](#notes) via config. | | `/focus [on\|off\|status]` | Toggle **focus view** — a display-only reduced-output mode showing just your prompt and the final response. Composes with `/verbose`: turning it on snaps tool progress to `off` and remembers your previous mode, and `/focus off` restores it. Each turn ends with a dim recovery line (`⋯ 7 tool lines hidden · /focus off to show`) and a persistent `◉ focus` badge sits in the status bar so you always know you're in the reduced view. Nothing is sent differently to the model — detail is hidden, never discarded. | diff --git a/website/docs/user-guide/features/codex-app-server-runtime.md b/website/docs/user-guide/features/codex-app-server-runtime.md index 9c85c01891..cf74b27514 100644 --- a/website/docs/user-guide/features/codex-app-server-runtime.md +++ b/website/docs/user-guide/features/codex-app-server-runtime.md @@ -5,7 +5,7 @@ sidebar_label: Codex App-Server Runtime # Codex App-Server Runtime -Hermes can optionally hand `openai/*` and `openai-codex/*` turns to the [Codex CLI app-server](https://github.com/openai/codex) instead of running its own tool loop. When enabled, terminal commands, file edits, sandboxing, and MCP tool calls all execute inside Codex's runtime — Hermes becomes the shell around it (sessions DB, slash commands, gateway, memory and skill review). +Hermes can optionally hand `openai/*`, `openai-codex/*` and [named custom provider](#named-custom-providers) turns to the [Codex CLI app-server](https://github.com/openai/codex) instead of running its own tool loop. When enabled, terminal commands, file edits, sandboxing, and MCP tool calls all execute inside Codex's runtime — Hermes becomes the shell around it (sessions DB, slash commands, gateway, memory and skill review). This is **opt-in only**. Default Hermes behavior is unchanged unless you flip the flag. Hermes never auto-routes you onto this runtime. @@ -130,7 +130,8 @@ The kanban tools are gated by `HERMES_KANBAN_TASK` env var the dispatcher sets | Kanban worker dispatch | yes | yes (via callback) | | Kanban orchestrator tools | yes | yes (via callback) | | All gateway platforms | yes | yes | -| Non-OpenAI providers | yes | n/a — OpenAI/Codex-scoped | +| Named custom providers (`providers.`) | yes | yes — a matching `[model_providers.]` in `~/.codex/config.toml` is required | +| Other non-OpenAI providers | yes | n/a — not routed through codex | ### Live display @@ -160,6 +161,35 @@ uses: ``` Hermes' own `hermes auth add openai-codex` writes to `~/.hermes/auth.json` — that's a separate session. **Run `codex login` separately** if you haven't. + **Or: a named custom provider.** A `providers.` entry in Hermes config can use this runtime when the **same name** is defined as a Codex provider. Hermes config: + + ```yaml + providers: + my-gateway: + api: https://gateway.example.com/v1 + key_env: MY_GATEWAY_API_KEY + default_model: gpt-5.4 + + model: + provider: custom:my-gateway + default: gpt-5.4 + openai_runtime: codex_app_server + ``` + + and the matching table in `~/.codex/config.toml`: + + ```toml + [model_providers.my-gateway] + name = "My Gateway" + base_url = "https://gateway.example.com/v1" + env_key = "MY_GATEWAY_API_KEY" + wire_api = "responses" + ``` + + Hermes sends only `model` and `modelProvider = "my-gateway"` on `thread/start`; codex resolves `base_url` and reads the key from `env_key` in its own environment. **Hermes never forwards the API key**, so `MY_GATEWAY_API_KEY` must be present in the process environment Hermes runs in — `~/.hermes/.env` is loaded at startup and provider credentials are inherited by the codex subprocess. Auxiliary calls (titles, compression, memory review) still use Hermes' own `providers.my-gateway` entry. + + Caveats: the name after `custom:` is the `providers:` config key and must match the `[model_providers.]` table name exactly — if it does not exist on the codex side, codex reports an unknown provider rather than silently using the Hermes endpoint. Anonymous `provider: custom` (a bare `base_url`) is not eligible: it has no stable name to hand to codex, so it stays on Hermes' standard runtime. + 3. **(Optional) Install the Codex plugins you want.** When you enable the runtime, Hermes auto-migrates whichever curated plugins you've already installed via Codex CLI: ```bash codex plugin marketplace add openai-curated