diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index 396e9fc0be..b6d89b0599 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -105,6 +105,9 @@ OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"} #: Tencent TokenHub: low/medium/high. TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high") +#: Nebius Token Factory: low/medium/high (top-level reasoning_effort knob). +NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high") + #: Kimi K3's vendor-documented translation quirks (platform.kimi.ai #: thinking-model guide): ``high`` is K3's positional middle AND server #: default, so ``medium`` rounds to it rather than down to ``low``; ``xhigh`` diff --git a/plugins/model-providers/nebius-token-factory/__init__.py b/plugins/model-providers/nebius-token-factory/__init__.py index 608b5b2b62..d54881d4cb 100644 --- a/plugins/model-providers/nebius-token-factory/__init__.py +++ b/plugins/model-providers/nebius-token-factory/__init__.py @@ -57,12 +57,12 @@ class NebiusTokenFactoryProfile(ProviderProfile): effort = str(raw_effort or "medium").strip().lower() if enabled is False or effort in {"none", "off", "disabled"}: return {}, {} - if effort in {"xhigh", "max"}: - effort = "high" - elif effort == "minimal": - effort = "low" - elif effort not in {"low", "medium", "high"}: - effort = "medium" + # Canonical clamp (nearest weaker supported level, never escalate, + # monotonic) — the hand-rolled map this replaces inverted the ladder: + # ultra fell through to medium while xhigh mapped to high. + from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort + + effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium" return {}, {"reasoning_effort": effort} diff --git a/tests/hermes_cli/test_nebius_token_factory_provider.py b/tests/hermes_cli/test_nebius_token_factory_provider.py index d1e3235465..5fc12a0b68 100644 --- a/tests/hermes_cli/test_nebius_token_factory_provider.py +++ b/tests/hermes_cli/test_nebius_token_factory_provider.py @@ -172,6 +172,28 @@ def test_nebius_reasoning_models_emit_top_level_reasoning_effort(): assert top_level == {"reasoning_effort": "high"} +def test_nebius_effort_clamp_is_monotonic(): + """Regression: the hand-rolled map sent ultra->medium while xhigh->high, + inverting the ladder. The canonical clamp_effort keeps stronger requests + at least as strong on the wire (all of xhigh/max/ultra clamp to high).""" + from providers import get_provider_profile + + profile = get_provider_profile("nebius-token-factory") + assert profile is not None + + def wire(effort): + _, top = profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": effort}, + model="deepseek-ai/DeepSeek-V4-Pro", + ) + return top.get("reasoning_effort") + + assert wire("ultra") == "high" + assert wire("max") == "high" + assert wire("xhigh") == "high" + assert wire("minimal") == "low" + + def test_nebius_reasoning_defaults_to_medium_for_known_reasoning_model(): from providers import get_provider_profile