diff --git a/plugins/model-providers/deepinfra/__init__.py b/plugins/model-providers/deepinfra/__init__.py index 2ba8fee8dd..21a3d0423b 100644 --- a/plugins/model-providers/deepinfra/__init__.py +++ b/plugins/model-providers/deepinfra/__init__.py @@ -3,11 +3,7 @@ their own plugin subsystems).""" from typing import Any -from agent.reasoning_effort import ( - OPENAI_COMPAT_WIRE_EFFORTS, - clamp_effort, - requested_effort, -) +from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort, requested_effort from providers import register_provider from providers.base import ProviderProfile @@ -19,25 +15,19 @@ class _DeepInfraProfile(ProviderProfile): def build_api_kwargs_extras( self, *, reasoning_config: dict | None = None, **context: Any ) -> tuple[dict[str, Any], dict[str, Any]]: - """Map Hermes reasoning controls to DeepInfra's top-level wire field. + """Map Hermes reasoning controls to DeepInfra's top-level ``reasoning_effort``. - DeepInfra applies a per-model default when the field is absent, while - ``none`` is its explicit off switch. This route does not advertise a - reasoning capability to the shared transport, so it must not be gated - on ``supports_reasoning``. + DeepInfra applies a per-model default when the field is absent (DeepSeek-V4.x off, + GLM/Qwen-Thinking on), so ``none`` is the only working off switch and an unset effort + is omitted rather than guessed. The core ``_supports_reasoning_extra_body`` allowlist + does not know this host, so the transport always passes ``supports_reasoning=False`` + here — gating on it would make the method a permanent no-op (#111872). """ - if ( - isinstance(reasoning_config, dict) - and reasoning_config.get("enabled") is False - ): + if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False: return {}, {"reasoning_effort": "none"} effort = requested_effort(reasoning_config) clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) - return ( - ({}, {"reasoning_effort": clamped}) - if clamped in OPENAI_COMPAT_WIRE_EFFORTS - else ({}, {}) - ) + return ({}, {"reasoning_effort": clamped}) if clamped in OPENAI_COMPAT_WIRE_EFFORTS else ({}, {}) def default_vision_model(self): # type: ignore[override] """First vision-capable *chat* model from the live catalog, or None. Key-gated so a box @@ -49,7 +39,6 @@ class _DeepInfraProfile(ProviderProfile): return None try: from hermes_cli.models import _fetch_deepinfra_models_by_tag - items = _fetch_deepinfra_models_by_tag("chat") except Exception: return None @@ -62,13 +51,9 @@ class _DeepInfraProfile(ProviderProfile): deepinfra = _DeepInfraProfile( - name="deepinfra", - aliases=("deep-infra", "deepinfra-ai"), - display_name="DeepInfra", - description="DeepInfra — 100+ open models, pay-per-use", - signup_url="https://deepinfra.com/dash/api_keys", - env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"), - base_url="https://api.deepinfra.com/v1/openai", + name="deepinfra", aliases=("deep-infra", "deepinfra-ai"), display_name="DeepInfra", + description="DeepInfra — 100+ open models, pay-per-use", signup_url="https://deepinfra.com/dash/api_keys", + env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"), base_url="https://api.deepinfra.com/v1/openai", auth_type="api_key", default_max_tokens=None, # DeepInfra applies its documented per-model limit # The only hardcoded DeepInfra model: aux resolution is synchronous, so it diff --git a/tests/plugins/model_providers/test_deepinfra_profile.py b/tests/plugins/model_providers/test_deepinfra_profile.py index f76c1fbb0d..cfbf280bfb 100644 --- a/tests/plugins/model_providers/test_deepinfra_profile.py +++ b/tests/plugins/model_providers/test_deepinfra_profile.py @@ -1,4 +1,11 @@ -"""Regression coverage for DeepInfra's top-level reasoning_effort wire field.""" +"""DeepInfra profile puts the reasoning switch on the wire (#111872). + +DeepInfra's OpenAI-compatible endpoint reads one top-level ``reasoning_effort`` field validated +against a gateway-wide enum (``none``..``max``; Hermes-internal ``ultra`` is rejected). The +profile is the ONLY source of that field on the transport's profile path, and the core +``_supports_reasoning_extra_body`` allowlist passes ``supports_reasoning=False`` for this host, +so the profile must emit without gating on it. +""" from __future__ import annotations @@ -7,8 +14,7 @@ import pytest @pytest.fixture def deepinfra_profile(): - """Resolve the registered profile through the normal plugin discovery path.""" - import model_tools # noqa: F401 + import model_tools # noqa: F401 (plugin discovery registers the profile) import providers profile = providers.get_provider_profile("deepinfra") @@ -16,59 +22,43 @@ def deepinfra_profile(): return profile -class TestDeepInfraReasoningEffort: - def test_default_preserves_the_model_default(self, deepinfra_profile): - assert deepinfra_profile.build_api_kwargs_extras(reasoning_config=None) == ( - {}, - {}, - ) - - @pytest.mark.parametrize( - "effort", ("minimal", "low", "medium", "high", "xhigh", "max") +@pytest.mark.parametrize( + "reasoning_config, expected_top_level", + [ + ({"enabled": True, "effort": "high"}, {"reasoning_effort": "high"}), + ({"enabled": True, "effort": "xhigh"}, {"reasoning_effort": "xhigh"}), # native, never folded into max + ({"enabled": True, "effort": "ultra"}, {"reasoning_effort": "max"}), # Hermes-internal tier clamps + ({"enabled": False}, {"reasoning_effort": "none"}), # the only off switch for default-on models + ({"enabled": True, "effort": "none"}, {"reasoning_effort": "none"}), + (None, {}), # nothing requested → keep DeepInfra's per-model default + ({"enabled": True}, {}), + ({"enabled": True, "effort": "future-tier"}, {}), # unknown level omitted rather than 422 + ], +) +def test_profile_translates_reasoning_config_to_top_level_effort(deepinfra_profile, reasoning_config, expected_top_level): + extra_body, top_level = deepinfra_profile.build_api_kwargs_extras( + reasoning_config=reasoning_config, supports_reasoning=False, model="deepseek-ai/DeepSeek-V4.1-Flash", ) - def test_explicit_efforts_are_sent_verbatim(self, deepinfra_profile, effort): - assert deepinfra_profile.build_api_kwargs_extras( - reasoning_config={"enabled": True, "effort": effort} - ) == ({}, {"reasoning_effort": effort}) + assert extra_body == {} + assert top_level == expected_top_level - def test_ultra_clamps_to_deepinfra_maximum(self, deepinfra_profile): - assert deepinfra_profile.build_api_kwargs_extras( - reasoning_config={"enabled": True, "effort": "ultra"} - ) == ({}, {"reasoning_effort": "max"}) - @pytest.mark.parametrize( - "reasoning_config", ({"enabled": False}, {"enabled": False, "effort": "high"}) +def test_transport_main_turn_carries_reasoning_effort_without_capability_gate(deepinfra_profile): + """The main turn builds through ``_build_kwargs_from_profile`` with ``supports_reasoning=False`` + (core allowlist excludes this host) — the field must still reach the request.""" + from agent.transports.chat_completions import ChatCompletionsTransport + + build = ChatCompletionsTransport().build_kwargs + on = build( + model="deepseek-ai/DeepSeek-V4.1-Flash", messages=[{"role": "user", "content": "ping"}], tools=None, + provider_profile=deepinfra_profile, provider_name="deepinfra", + reasoning_config={"enabled": True, "effort": "high"}, supports_reasoning=False, ) - def test_disabled_sends_the_explicit_off_value( - self, deepinfra_profile, reasoning_config - ): - assert deepinfra_profile.build_api_kwargs_extras( - reasoning_config=reasoning_config - ) == ({}, {"reasoning_effort": "none"}) - - @pytest.mark.parametrize( - "reasoning_config", - ({}, {"enabled": True}, {"enabled": True, "effort": "future-tier"}), + off = build( + model="zai-org/GLM-4.6", messages=[{"role": "user", "content": "ping"}], tools=None, + provider_profile=deepinfra_profile, provider_name="deepinfra", + reasoning_config={"enabled": False}, supports_reasoning=False, ) - def test_missing_or_unknown_effort_preserves_the_model_default( - self, deepinfra_profile, reasoning_config - ): - assert deepinfra_profile.build_api_kwargs_extras( - reasoning_config=reasoning_config - ) == ({}, {}) - - def test_transport_includes_top_level_reasoning_effort_without_capability_gate( - self, deepinfra_profile - ): - from agent.transports.chat_completions import ChatCompletionsTransport - - kwargs = ChatCompletionsTransport().build_kwargs( - model="deepseek-ai/DeepSeek-V4.1-Flash", - messages=[{"role": "user", "content": "ping"}], - tools=None, - provider_profile=deepinfra_profile, - provider_name="deepinfra", - reasoning_config={"enabled": True, "effort": "high"}, - supports_reasoning=False, - ) - assert kwargs["reasoning_effort"] == "high" + assert on["reasoning_effort"] == "high" + assert off["reasoning_effort"] == "none" + assert "reasoning" not in (on.get("extra_body") or {}) diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 759aed3dac..546e04392b 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -426,6 +426,10 @@ The model catalog is fetched dynamically from `ollama.com/v1/models` and cached Both speak the same OpenAI-compatible API. Cloud is a first-class provider (`--provider ollama-cloud`, `OLLAMA_API_KEY`); local Ollama is reached via the Custom Endpoint flow (base URL `http://localhost:11434/v1`, no key). Use cloud for large models you can't run locally; use local for privacy or offline work. ::: +### DeepInfra + +DeepInfra (`--provider deepinfra`, `DEEPINFRA_API_KEY`) is discovered live from its catalog. Reasoning is controlled through DeepInfra's top-level `reasoning_effort` field, so `agent.reasoning_effort`, `/reasoning `, `--reasoning` and per-model `agent.reasoning_overrides` work in **both directions**: an effort turns thinking on for models that default off (DeepSeek-V4.x), `/reasoning none` turns it off for models that default on (GLM-4.6, Qwen3-Thinking). Leaving reasoning unset keeps DeepInfra's per-model default; `xhigh` is native and `ultra` is sent as `max`. + ### AWS Bedrock Anthropic Claude, Amazon Nova, DeepSeek v3.2, Meta Llama 4, and other models via AWS Bedrock. Uses the AWS SDK (`boto3`) credential chain — no API key, just standard AWS auth.