From 92fa970e209c0bef23bfb8f56da43d244d45c0dd Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 23:36:44 -0700 Subject: [PATCH] test(aux): pin `responses` alias routing for azure-foundry vision; document the alias Two invariant tests (red on base): the task-level `auxiliary.vision.api_mode: responses` route resolves to `CodexAuxiliaryClient` with the azure-foundry identity intact, and an explicit `api_mode="responses"` kwarg to `resolve_provider_client` does the same. The Azure Foundry guide now lists `responses` as an accepted spelling for model, fallback and auxiliary routes. --- .../test_auxiliary_client_azure_foundry.py | 41 +++++++++++++++++++ website/docs/guides/azure-foundry.md | 1 + 2 files changed, 42 insertions(+) diff --git a/tests/agent/test_auxiliary_client_azure_foundry.py b/tests/agent/test_auxiliary_client_azure_foundry.py index 332be471d5..8d002215b2 100644 --- a/tests/agent/test_auxiliary_client_azure_foundry.py +++ b/tests/agent/test_auxiliary_client_azure_foundry.py @@ -360,3 +360,44 @@ class TestResolveProviderClientAzureFoundry: "azure-foundry" in rec.message and "hermes doctor" in rec.message for rec in caplog.records ) + + +# --------------------------------------------------------------------------- +# api_mode aliases — ``responses`` (user-facing spelling) must select the +# Responses adapter exactly like ``codex_responses`` (#39750) +# --------------------------------------------------------------------------- + + +class TestAzureFoundryResponsesAlias: + _AUX_VISION = { + "provider": "azure-foundry", "model": "gpt-5.4-nano", + "base_url": "https://r.services.ai.azure.com/openai/v1", "api_mode": "responses", + } + + def test_task_level_responses_alias_routes_vision_through_responses_adapter(self, monkeypatch): + """``auxiliary.vision.api_mode: responses`` on an azure-foundry route used to yield a + plain chat-completions client (401 from /chat/completions on a Responses-only + deployment, #39750); the alias must reach the Codex/Responses adapter and keep the + first-class provider identity.""" + from agent import auxiliary_client as _aux + + cfg = {"model": {"provider": "openrouter", "default": "x"}, "auxiliary": {"vision": dict(self._AUX_VISION)}} + monkeypatch.setattr("hermes_cli.config.load_config_readonly", lambda: cfg) + monkeypatch.setattr("hermes_cli.config.load_config", lambda: cfg) + monkeypatch.setenv("AZURE_FOUNDRY_API_KEY", "k") + + provider, client, model = _aux.resolve_vision_provider_client() + assert provider == "azure-foundry" + assert model == "gpt-5.4-nano" + assert isinstance(client, _aux.CodexAuxiliaryClient) + + def test_explicit_responses_alias_kwarg_wraps_in_codex_adapter(self, monkeypatch, patch_load_config): + """A caller-supplied ``api_mode="responses"`` is canonicalized at the resolver chokepoint.""" + from agent import auxiliary_client as _aux + + patch_load_config({"provider": "azure-foundry", "base_url": "https://r.services.ai.azure.com/openai/v1"}) + monkeypatch.setenv("AZURE_FOUNDRY_API_KEY", "k") + + client, model = _aux.resolve_provider_client("azure-foundry", "gpt-5.4-nano", api_mode="responses") + assert model == "gpt-5.4-nano" + assert isinstance(client, _aux.CodexAuxiliaryClient) diff --git a/website/docs/guides/azure-foundry.md b/website/docs/guides/azure-foundry.md index 63fb8125b7..b1098e5ba3 100644 --- a/website/docs/guides/azure-foundry.md +++ b/website/docs/guides/azure-foundry.md @@ -228,6 +228,7 @@ model: Important behaviour: - **GPT-5.x, codex, and o-series auto-route to the Responses API.** Microsoft Foundry deploys GPT-5 / codex / o1 / o3 / o4 models as Responses-API-only — calling `/chat/completions` against them returns `400 "The requested operation is unsupported."`. Hermes detects these model families by name and upgrades `api_mode` to `codex_responses` transparently, even when `config.yaml` still reads `api_mode: chat_completions`. GPT-4, GPT-4o, Llama, Mistral, and other deployments stay on `/chat/completions`. +- **`api_mode: responses` is accepted as a spelling of `codex_responses`.** The alias works on `model.api_mode`, on `fallback_providers` entries and on per-task `auxiliary..api_mode` (e.g. an `auxiliary.vision` route to a GPT-5.x deployment), and selects the same Responses adapter. - **`max_completion_tokens` is used automatically.** Azure OpenAI (like direct OpenAI) requires `max_completion_tokens` for gpt-4o, o-series, and gpt-5.x models. Hermes sends the right parameter based on the endpoint. - **Pre-v1 endpoints that require `api-version`.** If you have a legacy base URL like `https://.openai.azure.com/openai?api-version=2025-04-01-preview`, Hermes extracts the query string and forwards it via `default_query` on every request (the OpenAI SDK otherwise drops it when joining paths).