fix(nebius): route effort through canonical clamp_effort — hand-rolled map inverted the ladder

Review finding on salvaged #28253: the hand-rolled mapping sent
ultra -> medium while xhigh -> high (stronger request, weaker wire
value). Declare NEBIUS_EFFORTS in agent/reasoning_effort.py and use
clamp_effort like the zai/kimi/tokenhub call sites; disable detection
stays ahead of the clamp since clamp_effort('none', ...) returns the
floor, not off. Adds a monotonicity regression test.
This commit is contained in:
kshitijk4poor
2026-08-29 19:32:30 +05:30
committed by kshitij
parent a8e11d21c0
commit b954547e72
3 changed files with 31 additions and 6 deletions

View File

@@ -105,6 +105,9 @@ OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: Tencent TokenHub: low/medium/high.
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Nebius Token Factory: low/medium/high (top-level reasoning_effort knob).
NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
#: Kimi K3's vendor-documented translation quirks (platform.kimi.ai
#: thinking-model guide): ``high`` is K3's positional middle AND server
#: default, so ``medium`` rounds to it rather than down to ``low``; ``xhigh``

View File

@@ -57,12 +57,12 @@ class NebiusTokenFactoryProfile(ProviderProfile):
effort = str(raw_effort or "medium").strip().lower()
if enabled is False or effort in {"none", "off", "disabled"}:
return {}, {}
if effort in {"xhigh", "max"}:
effort = "high"
elif effort == "minimal":
effort = "low"
elif effort not in {"low", "medium", "high"}:
effort = "medium"
# Canonical clamp (nearest weaker supported level, never escalate,
# monotonic) — the hand-rolled map this replaces inverted the ladder:
# ultra fell through to medium while xhigh mapped to high.
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium"
return {}, {"reasoning_effort": effort}

View File

@@ -172,6 +172,28 @@ def test_nebius_reasoning_models_emit_top_level_reasoning_effort():
assert top_level == {"reasoning_effort": "high"}
def test_nebius_effort_clamp_is_monotonic():
"""Regression: the hand-rolled map sent ultra->medium while xhigh->high,
inverting the ladder. The canonical clamp_effort keeps stronger requests
at least as strong on the wire (all of xhigh/max/ultra clamp to high)."""
from providers import get_provider_profile
profile = get_provider_profile("nebius-token-factory")
assert profile is not None
def wire(effort):
_, top = profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": effort},
model="deepseek-ai/DeepSeek-V4-Pro",
)
return top.get("reasoning_effort")
assert wire("ultra") == "high"
assert wire("max") == "high"
assert wire("xhigh") == "high"
assert wire("minimal") == "low"
def test_nebius_reasoning_defaults_to_medium_for_known_reasoning_model():
from providers import get_provider_profile