fix(aux): only treat a standalone reasoning field name as a reasoning rejection; let parameter rungs chain
Review follow-up on #113114. `_is_reasoning_field_rejection` fired on any 400 whose text contained "reasoning"/"think" next to a generic unsupported marker, so a route-gating 400 naming a thinking model ("The model kimi-k2-thinking is not supported when using this account") spent a strip-retry and then re-raised from `_param_rung_accepts` — the configured fallback chain, which main consulted for that error, was never reached. Require the token to be a standalone wire-field name (not a model-id segment, not "... with reasoning models"), and let `_param_rung_accepts` accept model-incompatible, reasoning-field and structured-output rejections so a temperature-strip retry that 400s on `reasoning_effort` reaches the reasoning rung and a gating 400 after any strip still reaches provider fallback. Also mirrors the new retry paragraph into the zh-Hans configuration page.
This commit is contained in:
@@ -3235,7 +3235,19 @@ def _is_reasoning_field_rejection(exc: Exception) -> bool:
|
||||
status = getattr(exc, "status_code", None)
|
||||
if status is not None and status not in {400, 422}:
|
||||
return False
|
||||
return any(_is_unsupported_parameter_error(exc, name) for name in ("reasoning", "think"))
|
||||
if not any(_is_unsupported_parameter_error(exc, name) for name in ("reasoning", "think")):
|
||||
return False
|
||||
# The reasoning token must be a standalone wire-field name: not a model-id segment ("The model
|
||||
# kimi-k2-thinking is not supported when using this account" is route gating that belongs to the
|
||||
# provider-fallback rung) and not the adjective in "... not supported with reasoning models".
|
||||
return _REASONING_FIELD_TOKEN.search(str(exc).lower()) is not None
|
||||
|
||||
|
||||
# Reasoning wire-field names (the ``_PROFILE_REASONING_KEYS`` controls minus ``verbosity``), longest first.
|
||||
_REASONING_FIELD_TOKEN = re.compile(
|
||||
r"(?<![\w\-/])(?:reasoning_effort|thinking_config|thinking_budget|enable_thinking|thinkingconfig"
|
||||
r"|thinkingbudget|reasoning|thinking|think)(?![\w\-/])(?!\s+models?\b)"
|
||||
)
|
||||
|
||||
|
||||
def _without_reasoning_fields(kwargs: dict) -> Optional[dict]:
|
||||
@@ -6991,7 +7003,11 @@ def _param_rung_accepts(exc: Exception) -> bool:
|
||||
"""After a parameter-strip retry: fall through to the max_tokens/payment/auth
|
||||
chains with the stripped kwargs; re-raise anything those chains won't handle."""
|
||||
return (_is_payment_error(exc) or _is_connection_error(exc) or _is_auth_error(exc)
|
||||
or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc))
|
||||
or "max_tokens" in str(exc) or "unsupported_parameter" in str(exc)
|
||||
# Parameter rungs chain (temperature-strip retry 400s on reasoning_effort / response_format),
|
||||
# and a route-gating 400 after a strip still reaches the provider-fallback rung.
|
||||
or _is_reasoning_field_rejection(exc) or _is_structured_output_rejection(exc)
|
||||
or _is_model_incompatible_error(exc))
|
||||
|
||||
|
||||
def _credential_rung_accepts(exc: Exception) -> bool:
|
||||
|
||||
@@ -74,3 +74,24 @@ def test_unrelated_400_does_not_strip_reasoning_fields():
|
||||
with pytest.raises(RuntimeError, match="Invalid value"):
|
||||
_call(False, client)
|
||||
assert client.chat.completions.create.call_count == 1
|
||||
|
||||
|
||||
def test_model_gating_400_naming_a_thinking_model_still_reaches_the_fallback_chain():
|
||||
"""A route-gating 400 whose text merely contains a reasoning token inside the model id
|
||||
("kimi-k2-thinking is not supported when using this account") is not a field rejection: no
|
||||
strip-retry is spent on it and the configured fallback chain is consulted exactly as on main."""
|
||||
client = MagicMock()
|
||||
client.base_url = "https://relay.example/v1"
|
||||
client.chat.completions.create.side_effect = RuntimeError(
|
||||
"Error code: 400 - The model kimi-k2-thinking is not supported when using this account")
|
||||
fb_client = MagicMock()
|
||||
fb_client.chat.completions.create.return_value = {"fb": True}
|
||||
p1, p2, p3, _p4 = _custom_route_patches(client)
|
||||
with p1, p2, p3, patch("agent.auxiliary_client._try_configured_fallback_chain",
|
||||
return_value=(fb_client, "fallback-model", "fallback")) as fallback:
|
||||
result = call_llm(task="title_generation", messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_config={"enabled": False})
|
||||
|
||||
assert result == {"fb": True}
|
||||
assert client.chat.completions.create.call_count == 1 # no wasted reasoning-strip retry
|
||||
assert fallback.called
|
||||
|
||||
@@ -852,6 +852,8 @@ Hermes 中的每个模型槽位 —— 辅助任务、压缩、回退 —— 使
|
||||
`"main"` provider 选项表示"使用我的主 agent 使用的任何 provider" —— 它仅在 `auxiliary:`、`compression:` 和 `fallback_model:` 配置中有效。它**不是**顶级 `model.provider` 设置的有效值。如果您使用自定义 OpenAI 兼容端点,请在 `model:` 部分设置 `provider: custom`。所有主模型 provider 选项请参阅 [AI Providers](/integrations/providers)。
|
||||
:::
|
||||
|
||||
如果端点直接拒绝推理字段(例如 OpenAI 兼容中继后面的纯聊天模型返回 `400 Unrecognized request argument supplied: reasoning_effort`),辅助调用会去掉所有推理字段重试一次,因此该任务(例如会话标题)仍会以端点的默认行为完成。
|
||||
|
||||
**后台审查有所不同:** 与主会话使用同一模型的审查分支始终继承主会话的推理强度;`auxiliary.background_review.reasoning_effort` 在这条路径上不会生效,即使显式指定了主会话的 provider/model 也一样。推理设置、系统 prompt、完整会话快照和工具定义保持逐字节一致,以复用 prompt 缓存前缀。没有用于同模型审查的独立推理强度开关。详见[同模型审查的推理强度](/user-guide/features/memory#same-model-review-reasoning)。路由到其他模型时的独立问题见 [#94825](https://github.com/NousResearch/hermes-agent/issues/94825)。
|
||||
|
||||
### 完整辅助配置参考
|
||||
|
||||
Reference in New Issue
Block a user