From 65aaf16a4da478949901cf4fd2a098c6a6bbcc3e Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 16 Sep 2026 12:38:31 -0700 Subject: [PATCH] fix(aux): only treat a standalone reasoning field name as a reasoning rejection; let parameter rungs chain MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up on #113114. `_is_reasoning_field_rejection` fired on any 400 whose text contained "reasoning"/"think" next to a generic unsupported marker, so a route-gating 400 naming a thinking model ("The model kimi-k2-thinking is not supported when using this account") spent a strip-retry and then re-raised from `_param_rung_accepts` — the configured fallback chain, which main consulted for that error, was never reached. Require the token to be a standalone wire-field name (not a model-id segment, not "... with reasoning models"), and let `_param_rung_accepts` accept model-incompatible, reasoning-field and structured-output rejections so a temperature-strip retry that 400s on `reasoning_effort` reaches the reasoning rung and a gating 400 after any strip still reaches provider fallback. Also mirrors the new retry paragraph into the zh-Hans configuration page. --- agent/auxiliary_client.py | 20 ++++++++++++++++-- ...xiliary_reasoning_field_rejection_retry.py | 21 +++++++++++++++++++ .../current/user-guide/configuration.md | 2 ++ 3 files changed, 41 insertions(+), 2 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index c75fb7d9e7..f70c12312d 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -3235,7 +3235,19 @@ def _is_reasoning_field_rejection(exc: Exception) -> bool: status = getattr(exc, "status_code", None) if status is not None and status not in {400, 422}: return False - return any(_is_unsupported_parameter_error(exc, name) for name in ("reasoning", "think")) + if not any(_is_unsupported_parameter_error(exc, name) for name in ("reasoning", "think")): + return False + # The reasoning token must be a standalone wire-field name: not a model-id segment ("The model + # kimi-k2-thinking is not supported when using this account" is route gating that belongs to the + # provider-fallback rung) and not the adjective in "... not supported with reasoning models". + return _REASONING_FIELD_TOKEN.search(str(exc).lower()) is not None + + +# Reasoning wire-field names (the ``_PROFILE_REASONING_KEYS`` controls minus ``verbosity``), longest first. +_REASONING_FIELD_TOKEN = re.compile( + r"(? Optional[dict]: @@ -6991,7 +7003,11 @@ def _param_rung_accepts(exc: Exception) -> bool: """After a parameter-strip retry: fall through to the max_tokens/payment/auth chains with the stripped kwargs; re-raise anything those chains won't handle.""" return (_is_payment_error(exc) or _is_connection_error(exc) or _is_auth_error(exc) - or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc)) + or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc) + # Parameter rungs chain (temperature-strip retry 400s on reasoning_effort / response_format), + # and a route-gating 400 after a strip still reaches the provider-fallback rung. + or _is_reasoning_field_rejection(exc) or _is_structured_output_rejection(exc) + or _is_model_incompatible_error(exc)) def _credential_rung_accepts(exc: Exception) -> bool: diff --git a/tests/agent/test_auxiliary_reasoning_field_rejection_retry.py b/tests/agent/test_auxiliary_reasoning_field_rejection_retry.py index 2862ffa077..e758769a15 100644 --- a/tests/agent/test_auxiliary_reasoning_field_rejection_retry.py +++ b/tests/agent/test_auxiliary_reasoning_field_rejection_retry.py @@ -74,3 +74,24 @@ def test_unrelated_400_does_not_strip_reasoning_fields(): with pytest.raises(RuntimeError, match="Invalid value"): _call(False, client) assert client.chat.completions.create.call_count == 1 + + +def test_model_gating_400_naming_a_thinking_model_still_reaches_the_fallback_chain(): + """A route-gating 400 whose text merely contains a reasoning token inside the model id + ("kimi-k2-thinking is not supported when using this account") is not a field rejection: no + strip-retry is spent on it and the configured fallback chain is consulted exactly as on main.""" + client = MagicMock() + client.base_url = "https://relay.example/v1" + client.chat.completions.create.side_effect = RuntimeError( + "Error code: 400 - The model kimi-k2-thinking is not supported when using this account") + fb_client = MagicMock() + fb_client.chat.completions.create.return_value = {"fb": True} + p1, p2, p3, _p4 = _custom_route_patches(client) + with p1, p2, p3, patch("agent.auxiliary_client._try_configured_fallback_chain", + return_value=(fb_client, "fallback-model", "fallback")) as fallback: + result = call_llm(task="title_generation", messages=[{"role": "user", "content": "hi"}], + reasoning_config={"enabled": False}) + + assert result == {"fb": True} + assert client.chat.completions.create.call_count == 1 # no wasted reasoning-strip retry + assert fallback.called diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/configuration.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/configuration.md index ec754bc726..5a57e65e33 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/configuration.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/configuration.md @@ -852,6 +852,8 @@ Hermes 中的每个模型槽位 —— 辅助任务、压缩、回退 —— 使 `"main"` provider 选项表示"使用我的主 agent 使用的任何 provider" —— 它仅在 `auxiliary:`、`compression:` 和 `fallback_model:` 配置中有效。它**不是**顶级 `model.provider` 设置的有效值。如果您使用自定义 OpenAI 兼容端点,请在 `model:` 部分设置 `provider: custom`。所有主模型 provider 选项请参阅 [AI Providers](/integrations/providers)。 ::: +如果端点直接拒绝推理字段(例如 OpenAI 兼容中继后面的纯聊天模型返回 `400 Unrecognized request argument supplied: reasoning_effort`),辅助调用会去掉所有推理字段重试一次,因此该任务(例如会话标题)仍会以端点的默认行为完成。 + **后台审查有所不同:** 与主会话使用同一模型的审查分支始终继承主会话的推理强度;`auxiliary.background_review.reasoning_effort` 在这条路径上不会生效,即使显式指定了主会话的 provider/model 也一样。推理设置、系统 prompt、完整会话快照和工具定义保持逐字节一致,以复用 prompt 缓存前缀。没有用于同模型审查的独立推理强度开关。详见[同模型审查的推理强度](/user-guide/features/memory#same-model-review-reasoning)。路由到其他模型时的独立问题见 [#94825](https://github.com/NousResearch/hermes-agent/issues/94825)。 ### 完整辅助配置参考