fix(nebius): route effort through canonical clamp_effort — hand-rolled map inverted the ladder
Review finding on salvaged #28253: the hand-rolled mapping sent ultra -> medium while xhigh -> high (stronger request, weaker wire value). Declare NEBIUS_EFFORTS in agent/reasoning_effort.py and use clamp_effort like the zai/kimi/tokenhub call sites; disable detection stays ahead of the clamp since clamp_effort('none', ...) returns the floor, not off. Adds a monotonicity regression test.
This commit is contained in:
@@ -105,6 +105,9 @@ OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
#: Tencent TokenHub: low/medium/high.
|
||||
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
#: Nebius Token Factory: low/medium/high (top-level reasoning_effort knob).
|
||||
NEBIUS_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
#: Kimi K3's vendor-documented translation quirks (platform.kimi.ai
|
||||
#: thinking-model guide): ``high`` is K3's positional middle AND server
|
||||
#: default, so ``medium`` rounds to it rather than down to ``low``; ``xhigh``
|
||||
|
||||
@@ -57,12 +57,12 @@ class NebiusTokenFactoryProfile(ProviderProfile):
|
||||
effort = str(raw_effort or "medium").strip().lower()
|
||||
if enabled is False or effort in {"none", "off", "disabled"}:
|
||||
return {}, {}
|
||||
if effort in {"xhigh", "max"}:
|
||||
effort = "high"
|
||||
elif effort == "minimal":
|
||||
effort = "low"
|
||||
elif effort not in {"low", "medium", "high"}:
|
||||
effort = "medium"
|
||||
# Canonical clamp (nearest weaker supported level, never escalate,
|
||||
# monotonic) — the hand-rolled map this replaces inverted the ladder:
|
||||
# ultra fell through to medium while xhigh mapped to high.
|
||||
from agent.reasoning_effort import NEBIUS_EFFORTS, clamp_effort
|
||||
|
||||
effort = clamp_effort(effort, NEBIUS_EFFORTS) or "medium"
|
||||
|
||||
return {}, {"reasoning_effort": effort}
|
||||
|
||||
|
||||
@@ -172,6 +172,28 @@ def test_nebius_reasoning_models_emit_top_level_reasoning_effort():
|
||||
assert top_level == {"reasoning_effort": "high"}
|
||||
|
||||
|
||||
def test_nebius_effort_clamp_is_monotonic():
|
||||
"""Regression: the hand-rolled map sent ultra->medium while xhigh->high,
|
||||
inverting the ladder. The canonical clamp_effort keeps stronger requests
|
||||
at least as strong on the wire (all of xhigh/max/ultra clamp to high)."""
|
||||
from providers import get_provider_profile
|
||||
|
||||
profile = get_provider_profile("nebius-token-factory")
|
||||
assert profile is not None
|
||||
|
||||
def wire(effort):
|
||||
_, top = profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": effort},
|
||||
model="deepseek-ai/DeepSeek-V4-Pro",
|
||||
)
|
||||
return top.get("reasoning_effort")
|
||||
|
||||
assert wire("ultra") == "high"
|
||||
assert wire("max") == "high"
|
||||
assert wire("xhigh") == "high"
|
||||
assert wire("minimal") == "low"
|
||||
|
||||
|
||||
def test_nebius_reasoning_defaults_to_medium_for_known_reasoning_model():
|
||||
from providers import get_provider_profile
|
||||
|
||||
|
||||
Reference in New Issue
Block a user