From 30f9955a44ec17f3d07100a008b9c2e2689a16bf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 14:47:04 -0700 Subject: [PATCH] fix(zai): GLM-5.3 low/medium reasoning effort reaches the wire instead of clamping to high GLM-5.3 accepts a graded low/medium/high/max reasoning_effort scale (verified live in #91789: monotonic reasoning-token scaling, no 400s), but the effort mapper reused GLM-5.2's two-level vocabulary, silently rewriting low/medium to high. Adds GLM53_EFFORTS/GLM53_OVERRIDES and a per-model vocabulary pick in the zai plugin; 5.2 keeps its high/max clamp. Closes #91789. Also covers the gap noted when closing #86947 (credit @santhanakrishnan-d and @terje1965 for the graded-scale finding). --- agent/reasoning_effort.py | 7 +++ plugins/model-providers/zai/__init__.py | 55 ++++++++++++++----- .../model_providers/test_zai_profile.py | 48 +++++++++++++++- 3 files changed, 95 insertions(+), 15 deletions(-) diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index e29c0273e5..396e9fc0be 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -117,6 +117,13 @@ KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"} GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} +#: GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified +#: live on api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every +#: level accepted with monotonic reasoning-token scaling (low=4, medium=11, +#: high=98, max=125 on the probe prompt). ``xhigh`` requests the top tier. +GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") +GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} + #: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh`` #: requests the top tier (matches the shipped profile mapping). DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") diff --git a/plugins/model-providers/zai/__init__.py b/plugins/model-providers/zai/__init__.py index 2828caff92..5038e55503 100644 --- a/plugins/model-providers/zai/__init__.py +++ b/plugins/model-providers/zai/__init__.py @@ -65,13 +65,30 @@ def _is_glm_5_2(model: str | None) -> bool: ) -def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None: - """Map Hermes reasoning effort onto GLM-5.2/5.3's native ``high``/``max``. +def _is_glm_5_3(model: str | None) -> bool: + """Detect GLM-5.3 specifically — it has a wider effort vocabulary. - These models only support two enabled effort levels. ``xhigh``/``max``/``ultra`` - request the top tier; everything else that is enabled requests ``high`` - (its minimum thinking level). When reasoning is explicitly disabled, or - no effort preference is supplied, the server default is left untouched. + 5.2 accepts only ``high``/``max``; 5.3 accepts a graded + ``low``/``medium``/``high``/``max`` scale (verified live, issue #91789), + so effort mapping must pick the vocabulary per model. + """ + m = (model or "").strip().lower() + if not m: + return False + return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3")) + + +def _glm_5_2_reasoning_effort( + reasoning_config: dict | None, *, model: str | None = None +) -> str | None: + """Map Hermes reasoning effort onto GLM's native vocabulary. + + GLM-5.2 supports two enabled effort levels (``high``/``max``); + GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale. + ``xhigh``/``max``/``ultra`` request the top tier; anything below the + model's floor clamps to that floor. When reasoning is explicitly + disabled, or no effort preference is supplied, the server default is + left untouched. """ if not isinstance(reasoning_config, dict): return None @@ -82,14 +99,24 @@ def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None: if not effort or effort == "none": return None - # GLM-5.2's two-level vocabulary (high = its minimum thinking level, - # max = top tier) is declared in agent.reasoning_effort; xhigh rounds up - # to max. Everything at or below high clamps to high — GLM cannot think - # less than that. - from agent.reasoning_effort import GLM52_EFFORTS, GLM52_OVERRIDES, clamp_effort + # Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds + # up to max on both. 5.2 cannot think less than high; 5.3 accepts a + # graded scale down to low (issue #91789). + from agent.reasoning_effort import ( + GLM52_EFFORTS, + GLM52_OVERRIDES, + GLM53_EFFORTS, + GLM53_OVERRIDES, + clamp_effort, + ) - clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES) - return clamped if clamped in GLM52_EFFORTS else "high" + if _is_glm_5_3(model): + efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low" + else: + efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high" + + clamped = clamp_effort(effort, efforts, overrides) + return clamped if clamped in efforts else floor class ZaiProfile(ProviderProfile): @@ -111,7 +138,7 @@ class ZaiProfile(ProviderProfile): extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"} if _is_glm_5_2(model): - effort = _glm_5_2_reasoning_effort(reasoning_config) + effort = _glm_5_2_reasoning_effort(reasoning_config, model=model) if effort is not None: top_level["reasoning_effort"] = effort diff --git a/tests/plugins/model_providers/test_zai_profile.py b/tests/plugins/model_providers/test_zai_profile.py index 58915b0daa..30b0aaf9a7 100644 --- a/tests/plugins/model_providers/test_zai_profile.py +++ b/tests/plugins/model_providers/test_zai_profile.py @@ -106,7 +106,6 @@ class TestZaiGLM52ReasoningEffort: assert extra_body == {"thinking": {"type": "disabled"}} assert top_level == {} - @pytest.mark.parametrize( "model", [ @@ -136,6 +135,53 @@ class TestZaiGLM52ReasoningEffort: assert top_level == {} +class TestZaiGLM53ReasoningEffort: + """GLM-5.3's graded low/medium/high/max effort scale (issue #91789). + + Verified live on api.z.ai/api/coding/paas/v4: all four levels accepted + with monotonic reasoning-token scaling. Unlike 5.2, low and medium must + reach the wire instead of clamping up to high. + """ + + @pytest.mark.parametrize( + ("effort", "expected"), + [ + ("low", "low"), + ("medium", "medium"), + ("high", "high"), + ("max", "max"), + ("xhigh", "max"), + ("minimal", "low"), + ], + ) + def test_graded_efforts_pass_through(self, zai_profile, effort, expected): + extra_body, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": effort}, + model="glm-5.3", + ) + assert extra_body == {"thinking": {"type": "enabled"}} + assert top_level == {"reasoning_effort": expected} + + @pytest.mark.parametrize( + "model", + ["z-ai/glm-5.3", "glm-5-3", "glm-5p3", "zai-org-glm-5-3"], + ) + def test_alias_spellings_get_graded_scale(self, zai_profile, model): + _, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": "low"}, + model=model, + ) + assert top_level == {"reasoning_effort": "low"} + + def test_glm_5_2_still_clamps_low_to_high(self, zai_profile): + """The 5.3 widening must not leak into 5.2's two-level wire.""" + _, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": "low"}, + model="glm-5.2", + ) + assert top_level == {"reasoning_effort": "high"} + + class TestZaiModelGating: """GLM 4.5+ get thinking; earlier GLM models are left untouched."""