fix: trim DeepInfra reasoning salvage to the invariant and document it

Follow-up to the cherry-picked #111876 (@KoNit-K), which shares the design
of the earlier #111875 by the issue author (@ats3v): emit DeepInfra's
top-level ``reasoning_effort`` from the provider profile, ungated on
``supports_reasoning``, ``none`` as the only off switch, ``xhigh`` native,
``ultra`` clamped to ``max`` via the shared vocabulary, unset/unknown omitted.

- drop the constructor/blank-line reformat churn (byte-identical to main)
- replace the 14-case test file with two invariant tests: the profile's
  config -> top-level field table, and the transport main-turn path with
  ``supports_reasoning=False`` (the gate the core allowlist actually passes)
- docs: DeepInfra subsection in integrations/providers.md describing the
  two-directional reasoning control

Offline kwargs probe: before every reasoning_config -> ({}, {}) and the
main turn carried no reasoning field; after ``high`` -> ``reasoning_effort:
high``, ``{'enabled': False}`` -> ``none``, ``ultra`` -> ``max``, unset and
unknown levels omitted, aux calls stop emitting the generic
``extra_body.reasoning`` for this provider.

Co-authored-by: Georgi Atsev <georgi@deepinfra.com>
This commit is contained in:
teknium1
2026-09-15 12:20:59 -07:00
committed by Teknium
parent 4fe3d288eb
commit b027a4658e
3 changed files with 60 additions and 81 deletions

View File

@@ -3,11 +3,7 @@ their own plugin subsystems)."""
from typing import Any
from agent.reasoning_effort import (
OPENAI_COMPAT_WIRE_EFFORTS,
clamp_effort,
requested_effort,
)
from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort, requested_effort
from providers import register_provider
from providers.base import ProviderProfile
@@ -19,25 +15,19 @@ class _DeepInfraProfile(ProviderProfile):
def build_api_kwargs_extras(
self, *, reasoning_config: dict | None = None, **context: Any
) -> tuple[dict[str, Any], dict[str, Any]]:
"""Map Hermes reasoning controls to DeepInfra's top-level wire field.
"""Map Hermes reasoning controls to DeepInfra's top-level ``reasoning_effort``.
DeepInfra applies a per-model default when the field is absent, while
``none`` is its explicit off switch. This route does not advertise a
reasoning capability to the shared transport, so it must not be gated
on ``supports_reasoning``.
DeepInfra applies a per-model default when the field is absent (DeepSeek-V4.x off,
GLM/Qwen-Thinking on), so ``none`` is the only working off switch and an unset effort
is omitted rather than guessed. The core ``_supports_reasoning_extra_body`` allowlist
does not know this host, so the transport always passes ``supports_reasoning=False``
here — gating on it would make the method a permanent no-op (#111872).
"""
if (
isinstance(reasoning_config, dict)
and reasoning_config.get("enabled") is False
):
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
return {}, {"reasoning_effort": "none"}
effort = requested_effort(reasoning_config)
clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS)
return (
({}, {"reasoning_effort": clamped})
if clamped in OPENAI_COMPAT_WIRE_EFFORTS
else ({}, {})
)
return ({}, {"reasoning_effort": clamped}) if clamped in OPENAI_COMPAT_WIRE_EFFORTS else ({}, {})
def default_vision_model(self): # type: ignore[override]
"""First vision-capable *chat* model from the live catalog, or None. Key-gated so a box
@@ -49,7 +39,6 @@ class _DeepInfraProfile(ProviderProfile):
return None
try:
from hermes_cli.models import _fetch_deepinfra_models_by_tag
items = _fetch_deepinfra_models_by_tag("chat")
except Exception:
return None
@@ -62,13 +51,9 @@ class _DeepInfraProfile(ProviderProfile):
deepinfra = _DeepInfraProfile(
name="deepinfra",
aliases=("deep-infra", "deepinfra-ai"),
display_name="DeepInfra",
description="DeepInfra — 100+ open models, pay-per-use",
signup_url="https://deepinfra.com/dash/api_keys",
env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"),
base_url="https://api.deepinfra.com/v1/openai",
name="deepinfra", aliases=("deep-infra", "deepinfra-ai"), display_name="DeepInfra",
description="DeepInfra — 100+ open models, pay-per-use", signup_url="https://deepinfra.com/dash/api_keys",
env_vars=("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL"), base_url="https://api.deepinfra.com/v1/openai",
auth_type="api_key",
default_max_tokens=None, # DeepInfra applies its documented per-model limit
# The only hardcoded DeepInfra model: aux resolution is synchronous, so it

View File

@@ -1,4 +1,11 @@
"""Regression coverage for DeepInfra's top-level reasoning_effort wire field."""
"""DeepInfra profile puts the reasoning switch on the wire (#111872).
DeepInfra's OpenAI-compatible endpoint reads one top-level ``reasoning_effort`` field validated
against a gateway-wide enum (``none``..``max``; Hermes-internal ``ultra`` is rejected). The
profile is the ONLY source of that field on the transport's profile path, and the core
``_supports_reasoning_extra_body`` allowlist passes ``supports_reasoning=False`` for this host,
so the profile must emit without gating on it.
"""
from __future__ import annotations
@@ -7,8 +14,7 @@ import pytest
@pytest.fixture
def deepinfra_profile():
"""Resolve the registered profile through the normal plugin discovery path."""
import model_tools # noqa: F401
import model_tools # noqa: F401 (plugin discovery registers the profile)
import providers
profile = providers.get_provider_profile("deepinfra")
@@ -16,59 +22,43 @@ def deepinfra_profile():
return profile
class TestDeepInfraReasoningEffort:
def test_default_preserves_the_model_default(self, deepinfra_profile):
assert deepinfra_profile.build_api_kwargs_extras(reasoning_config=None) == (
{},
{},
)
@pytest.mark.parametrize(
"effort", ("minimal", "low", "medium", "high", "xhigh", "max")
@pytest.mark.parametrize(
"reasoning_config, expected_top_level",
[
({"enabled": True, "effort": "high"}, {"reasoning_effort": "high"}),
({"enabled": True, "effort": "xhigh"}, {"reasoning_effort": "xhigh"}), # native, never folded into max
({"enabled": True, "effort": "ultra"}, {"reasoning_effort": "max"}), # Hermes-internal tier clamps
({"enabled": False}, {"reasoning_effort": "none"}), # the only off switch for default-on models
({"enabled": True, "effort": "none"}, {"reasoning_effort": "none"}),
(None, {}), # nothing requested → keep DeepInfra's per-model default
({"enabled": True}, {}),
({"enabled": True, "effort": "future-tier"}, {}), # unknown level omitted rather than 422
],
)
def test_profile_translates_reasoning_config_to_top_level_effort(deepinfra_profile, reasoning_config, expected_top_level):
extra_body, top_level = deepinfra_profile.build_api_kwargs_extras(
reasoning_config=reasoning_config, supports_reasoning=False, model="deepseek-ai/DeepSeek-V4.1-Flash",
)
def test_explicit_efforts_are_sent_verbatim(self, deepinfra_profile, effort):
assert deepinfra_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": effort}
) == ({}, {"reasoning_effort": effort})
assert extra_body == {}
assert top_level == expected_top_level
def test_ultra_clamps_to_deepinfra_maximum(self, deepinfra_profile):
assert deepinfra_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": "ultra"}
) == ({}, {"reasoning_effort": "max"})
@pytest.mark.parametrize(
"reasoning_config", ({"enabled": False}, {"enabled": False, "effort": "high"})
def test_transport_main_turn_carries_reasoning_effort_without_capability_gate(deepinfra_profile):
"""The main turn builds through ``_build_kwargs_from_profile`` with ``supports_reasoning=False``
(core allowlist excludes this host) — the field must still reach the request."""
from agent.transports.chat_completions import ChatCompletionsTransport
build = ChatCompletionsTransport().build_kwargs
on = build(
model="deepseek-ai/DeepSeek-V4.1-Flash", messages=[{"role": "user", "content": "ping"}], tools=None,
provider_profile=deepinfra_profile, provider_name="deepinfra",
reasoning_config={"enabled": True, "effort": "high"}, supports_reasoning=False,
)
def test_disabled_sends_the_explicit_off_value(
self, deepinfra_profile, reasoning_config
):
assert deepinfra_profile.build_api_kwargs_extras(
reasoning_config=reasoning_config
) == ({}, {"reasoning_effort": "none"})
@pytest.mark.parametrize(
"reasoning_config",
({}, {"enabled": True}, {"enabled": True, "effort": "future-tier"}),
off = build(
model="zai-org/GLM-4.6", messages=[{"role": "user", "content": "ping"}], tools=None,
provider_profile=deepinfra_profile, provider_name="deepinfra",
reasoning_config={"enabled": False}, supports_reasoning=False,
)
def test_missing_or_unknown_effort_preserves_the_model_default(
self, deepinfra_profile, reasoning_config
):
assert deepinfra_profile.build_api_kwargs_extras(
reasoning_config=reasoning_config
) == ({}, {})
def test_transport_includes_top_level_reasoning_effort_without_capability_gate(
self, deepinfra_profile
):
from agent.transports.chat_completions import ChatCompletionsTransport
kwargs = ChatCompletionsTransport().build_kwargs(
model="deepseek-ai/DeepSeek-V4.1-Flash",
messages=[{"role": "user", "content": "ping"}],
tools=None,
provider_profile=deepinfra_profile,
provider_name="deepinfra",
reasoning_config={"enabled": True, "effort": "high"},
supports_reasoning=False,
)
assert kwargs["reasoning_effort"] == "high"
assert on["reasoning_effort"] == "high"
assert off["reasoning_effort"] == "none"
assert "reasoning" not in (on.get("extra_body") or {})

View File

@@ -426,6 +426,10 @@ The model catalog is fetched dynamically from `ollama.com/v1/models` and cached
Both speak the same OpenAI-compatible API. Cloud is a first-class provider (`--provider ollama-cloud`, `OLLAMA_API_KEY`); local Ollama is reached via the Custom Endpoint flow (base URL `http://localhost:11434/v1`, no key). Use cloud for large models you can't run locally; use local for privacy or offline work.
:::
### DeepInfra
DeepInfra (`--provider deepinfra`, `DEEPINFRA_API_KEY`) is discovered live from its catalog. Reasoning is controlled through DeepInfra's top-level `reasoning_effort` field, so `agent.reasoning_effort`, `/reasoning <level>`, `--reasoning` and per-model `agent.reasoning_overrides` work in **both directions**: an effort turns thinking on for models that default off (DeepSeek-V4.x), `/reasoning none` turns it off for models that default on (GLM-4.6, Qwen3-Thinking). Leaving reasoning unset keeps DeepInfra's per-model default; `xhigh` is native and `ultra` is sent as `max`.
### AWS Bedrock
Anthropic Claude, Amazon Nova, DeepSeek v3.2, Meta Llama 4, and other models via AWS Bedrock. Uses the AWS SDK (`boto3`) credential chain — no API key, just standard AWS auth.