From 8a6b5b67a7483bc4958ad0973ed3febb80a578a9 Mon Sep 17 00:00:00 2001 From: Justin Bennington <43761886+somewheresy@users.noreply.github.com> Date: Thu, 10 Sep 2026 15:13:02 -0400 Subject: [PATCH] fix(providers): block Actual at Responses send sites (E-1047) --- agent/auxiliary_client.py | 28 ++++++- agent/codex_runtime.py | 9 +++ tests/agent/test_actual_auxiliary_routing.py | 81 ++++++++++++++++++++ 3 files changed, 114 insertions(+), 4 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 28d8cff386..a6af92bda3 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1449,6 +1449,15 @@ class _CodexCompletionsAdapter: return resp_kwargs, model, timeout def create(self, **kwargs) -> Any: + from hermes_cli.providers import is_actual_route + + if is_actual_route( + getattr(self._client, "_hermes_aux_effective_provider", ""), + str(getattr(self._client, "base_url", "") or ""), + ): + raise ValueError( + "Actual requests require Chat Completions; refusing to call /responses." + ) # Low-level ``responses.create(stream=True)`` and assemble the final response ourselves # from ``response.output_item.done``: the high-level ``responses.stream()`` rebuilds from # ``response.completed.response.output``, which Codex returns as ``null`` (SDK crash). @@ -4497,11 +4506,22 @@ def _wrap_transport(req: _ResolveRequest, client_obj: Any, final_model_str: str, explicit api_mode — api.openai.com + codex model. Anthropic (Messages): ``api_mode=anthropic_messages``, any ``/anthropic`` suffix, ``api.kimi.com/coding``, or ``api.anthropic.com``.""" if _is_actual_auxiliary_route(req, base_url_str): - return client_obj._real_client if isinstance(client_obj, CodexAuxiliaryClient) else client_obj - needs_codex = not (isinstance(client_obj, CodexAuxiliaryClient) or req.raw_codex) and ( + client = ( + client_obj._real_client + if isinstance(client_obj, CodexAuxiliaryClient) + else client_obj + ) + client._hermes_aux_effective_provider = "actual" + return client + needs_codex = not ( + isinstance(client_obj, CodexAuxiliaryClient) or req.raw_codex + ) and ( req.api_mode == "codex_responses" - or (not req.api_mode and base_url_hostname(base_url_str) == "api.openai.com" - and "codex" in (final_model_str or "").lower()) + or ( + not req.api_mode + and base_url_hostname(base_url_str) == "api.openai.com" + and "codex" in (final_model_str or "").lower() + ) ) if needs_codex: logger.debug("resolve_provider_client: wrapping client in CodexAuxiliaryClient " diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index e58638cca8..48a6c1a673 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -918,6 +918,15 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta return bool(agent._interrupt_requested) def _open_codex_stream(next_api_kwargs: dict[str, Any]): + from hermes_cli.providers import is_actual_route + + if is_actual_route( + getattr(agent, "provider", ""), + str(getattr(active_client, "base_url", "") or ""), + ): + raise ValueError( + "Actual requests require Chat Completions; refusing to call /responses." + ) stream_kwargs = _sanitize_consumer_codex_request(agent, next_api_kwargs) stream_kwargs["stream"] = True return active_client.responses.create(**_bypass_sdk_request_transform(stream_kwargs)) diff --git a/tests/agent/test_actual_auxiliary_routing.py b/tests/agent/test_actual_auxiliary_routing.py index 033bfeabaa..dfa2c9544d 100644 --- a/tests/agent/test_actual_auxiliary_routing.py +++ b/tests/agent/test_actual_auxiliary_routing.py @@ -348,6 +348,87 @@ def test_actual_runtime_transitions_reach_chat_completions( agent.client.close() +@pytest.mark.parametrize( + "provider,hosted", + [ + ("actual", False), + ("aci", False), + ("actual-computer", False), + ("actualcomputer", False), + ("custom", True), + ("custom:actual-relay", True), + ], +) +@pytest.mark.parametrize("send_site", ["main", "auxiliary", "async_auxiliary"]) +def test_actual_rejects_forced_responses_before_http( + tmp_path, monkeypatch, actual_endpoint, provider, hosted, send_site +): + from agent.auxiliary_client import ( + AsyncCodexAuxiliaryClient, + CodexAuxiliaryClient, + resolve_provider_client, + ) + from run_agent import AIAgent + + base_url, requests = actual_endpoint + if hosted: + base_url = base_url.replace("127.0.0.1", "api.actual.inc") + base_url += "/v1" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + config = { + "model": {"provider": provider, "base_url": base_url, "default": "test-model"}, + "providers": { + "actual-relay": {"base_url": base_url, "transport": "codex_responses"} + }, + } + (tmp_path / "config.yaml").write_text(yaml.safe_dump(config), encoding="utf-8") + if send_site == "main": + agent = AIAgent( + provider=provider, + base_url=base_url, + api_key="actual-test-key", + model="test-model", + enabled_toolsets=[], + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + save_trajectories=False, + ) + client = agent.client + agent.api_mode = "codex_responses" + + def force_responses(): + return agent._run_codex_stream( + {"model": "test-model", "input": "Reply briefly."}, client=client + ) + + else: + client, model = resolve_provider_client( + provider, + model="test-model", + explicit_base_url=base_url, + explicit_api_key="actual-test-key", + api_mode="codex_responses", + ) + wrapper = CodexAuxiliaryClient(client, model) + if send_site == "async_auxiliary": + wrapper = AsyncCodexAuxiliaryClient(wrapper) + + def force_responses(): + result = wrapper.chat.completions.create( + model=model, messages=[{"role": "user", "content": "Reply briefly."}] + ) + return asyncio.run(result) if send_site == "async_auxiliary" else result + + try: + initial_requests = list(requests) + with pytest.raises(ValueError, match="Actual.*Chat Completions"): + force_responses() + assert requests == initial_requests + finally: + client.close() + + @pytest.mark.parametrize("provider", ["actual", "aci", "custom:actual-relay"]) @pytest.mark.parametrize("async_mode", [False, True]) def test_actual_auxiliary_fallback_reaches_chat_completions(