fix(zai): GLM-5.3 low/medium reasoning effort reaches the wire instead of clamping to high

GLM-5.3 accepts a graded low/medium/high/max reasoning_effort scale
(verified live in #91789: monotonic reasoning-token scaling, no 400s),
but the effort mapper reused GLM-5.2's two-level vocabulary, silently
rewriting low/medium to high. Adds GLM53_EFFORTS/GLM53_OVERRIDES and a
per-model vocabulary pick in the zai plugin; 5.2 keeps its high/max
clamp. Closes #91789. Also covers the gap noted when closing #86947
(credit @santhanakrishnan-d and @terje1965 for the graded-scale finding).
This commit is contained in:
Teknium
2026-08-21 14:47:04 -07:00
parent 3841910cee
commit 30f9955a44
3 changed files with 95 additions and 15 deletions

View File

@@ -117,6 +117,13 @@ KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"}
GLM52_EFFORTS: tuple[str, ...] = ("high", "max")
GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified
#: live on api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every
#: level accepted with monotonic reasoning-token scaling (low=4, medium=11,
#: high=98, max=125 on the probe prompt). ``xhigh`` requests the top tier.
GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")
GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"}
#: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh``
#: requests the top tier (matches the shipped profile mapping).
DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max")

View File

@@ -65,13 +65,30 @@ def _is_glm_5_2(model: str | None) -> bool:
)
def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None:
"""Map Hermes reasoning effort onto GLM-5.2/5.3's native ``high``/``max``.
def _is_glm_5_3(model: str | None) -> bool:
"""Detect GLM-5.3 specifically — it has a wider effort vocabulary.
These models only support two enabled effort levels. ``xhigh``/``max``/``ultra``
request the top tier; everything else that is enabled requests ``high``
(its minimum thinking level). When reasoning is explicitly disabled, or
no effort preference is supplied, the server default is left untouched.
5.2 accepts only ``high``/``max``; 5.3 accepts a graded
``low``/``medium``/``high``/``max`` scale (verified live, issue #91789),
so effort mapping must pick the vocabulary per model.
"""
m = (model or "").strip().lower()
if not m:
return False
return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3"))
def _glm_5_2_reasoning_effort(
reasoning_config: dict | None, *, model: str | None = None
) -> str | None:
"""Map Hermes reasoning effort onto GLM's native vocabulary.
GLM-5.2 supports two enabled effort levels (``high``/``max``);
GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale.
``xhigh``/``max``/``ultra`` request the top tier; anything below the
model's floor clamps to that floor. When reasoning is explicitly
disabled, or no effort preference is supplied, the server default is
left untouched.
"""
if not isinstance(reasoning_config, dict):
return None
@@ -82,14 +99,24 @@ def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None:
if not effort or effort == "none":
return None
# GLM-5.2's two-level vocabulary (high = its minimum thinking level,
# max = top tier) is declared in agent.reasoning_effort; xhigh rounds up
# to max. Everything at or below high clamps to high — GLM cannot think
# less than that.
from agent.reasoning_effort import GLM52_EFFORTS, GLM52_OVERRIDES, clamp_effort
# Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds
# up to max on both. 5.2 cannot think less than high; 5.3 accepts a
# graded scale down to low (issue #91789).
from agent.reasoning_effort import (
GLM52_EFFORTS,
GLM52_OVERRIDES,
GLM53_EFFORTS,
GLM53_OVERRIDES,
clamp_effort,
)
clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES)
return clamped if clamped in GLM52_EFFORTS else "high"
if _is_glm_5_3(model):
efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low"
else:
efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high"
clamped = clamp_effort(effort, efforts, overrides)
return clamped if clamped in efforts else floor
class ZaiProfile(ProviderProfile):
@@ -111,7 +138,7 @@ class ZaiProfile(ProviderProfile):
extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"}
if _is_glm_5_2(model):
effort = _glm_5_2_reasoning_effort(reasoning_config)
effort = _glm_5_2_reasoning_effort(reasoning_config, model=model)
if effort is not None:
top_level["reasoning_effort"] = effort

View File

@@ -106,7 +106,6 @@ class TestZaiGLM52ReasoningEffort:
assert extra_body == {"thinking": {"type": "disabled"}}
assert top_level == {}
@pytest.mark.parametrize(
"model",
[
@@ -136,6 +135,53 @@ class TestZaiGLM52ReasoningEffort:
assert top_level == {}
class TestZaiGLM53ReasoningEffort:
"""GLM-5.3's graded low/medium/high/max effort scale (issue #91789).
Verified live on api.z.ai/api/coding/paas/v4: all four levels accepted
with monotonic reasoning-token scaling. Unlike 5.2, low and medium must
reach the wire instead of clamping up to high.
"""
@pytest.mark.parametrize(
("effort", "expected"),
[
("low", "low"),
("medium", "medium"),
("high", "high"),
("max", "max"),
("xhigh", "max"),
("minimal", "low"),
],
)
def test_graded_efforts_pass_through(self, zai_profile, effort, expected):
extra_body, top_level = zai_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": effort},
model="glm-5.3",
)
assert extra_body == {"thinking": {"type": "enabled"}}
assert top_level == {"reasoning_effort": expected}
@pytest.mark.parametrize(
"model",
["z-ai/glm-5.3", "glm-5-3", "glm-5p3", "zai-org-glm-5-3"],
)
def test_alias_spellings_get_graded_scale(self, zai_profile, model):
_, top_level = zai_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": "low"},
model=model,
)
assert top_level == {"reasoning_effort": "low"}
def test_glm_5_2_still_clamps_low_to_high(self, zai_profile):
"""The 5.3 widening must not leak into 5.2's two-level wire."""
_, top_level = zai_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": "low"},
model="glm-5.2",
)
assert top_level == {"reasoning_effort": "high"}
class TestZaiModelGating:
"""GLM 4.5+ get thinking; earlier GLM models are left untouched."""