fix(deepseek): deepseek-flash is the canonical Flash id; retired names fold onto it
DeepSeek retired deepseek-v4-flash on 2026-09-10 (V4.1-Flash release); the API's model name is now `deepseek-flash` and /v1/models lists only it. Hermes still folded every non-V-series name onto deepseek-v4-flash, so `/model deepseek-flash` on the DeepSeek provider was rewritten, then the validator "auto-corrected" it back against the live listing: "Auto-corrected deepseek-v4-flash -> deepseek-flash" on every switch. Retired aliases (deepseek-chat / -reasoner and other fuzzy names) now fold onto deepseek-flash; the curated catalog, profile fallback list, aux default, goal-judge hint and pricing snapshot (2026-09-10 off-peak USD) follow the docs. Dated deepseek-v4-* ids still pass through untouched. Builds on YipTszkwan's #107126 (earliest fix in the cluster).
This commit is contained in:
@@ -186,11 +186,11 @@ _SNAPSHOTS: tuple[tuple[str, Optional[str], str, dict], ...] = (
|
||||
"gpt-4.1-nano": ("0.10", "0.40", "0.025"), "o3": ("10.00", "40.00", "2.50"),
|
||||
"o3-mini": ("1.10", "4.40", "0.55"),
|
||||
}),
|
||||
# deepseek-chat / deepseek-reasoner are deprecated aliases of
|
||||
# deepseek-v4-flash's non-thinking / thinking modes — same rates.
|
||||
("deepseek", "https://api-docs.deepseek.com/quick_start/pricing", "deepseek-pricing-2026-07", {
|
||||
("deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"): ("0.14", "0.28", "0.0028"),
|
||||
"deepseek-v4-pro": ("0.435", "0.87", "0.003625"),
|
||||
# Off-peak USD rates (peak = 2x, Mon-Fri 01-04 + 06-10 UTC). ``deepseek-v4-flash`` and the
|
||||
# retired deepseek-chat / deepseek-reasoner aliases are served by V4.1-Flash at the Flash price.
|
||||
("deepseek", "https://api-docs.deepseek.com/quick_start/pricing", "deepseek-pricing-2026-09-10", {
|
||||
("deepseek-flash", "deepseek-v4-flash", "deepseek-chat", "deepseek-reasoner"): ("0.15", "0.60", "0.003"),
|
||||
"deepseek-v4-pro": ("0.66", "1.98", "0.022"),
|
||||
}),
|
||||
("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-09-02", {
|
||||
("gemini-3.8-flash", "gemini-3.7-flash"): ("0.75", "3.75", "0.075"),
|
||||
|
||||
@@ -1499,7 +1499,7 @@ class GoalManager:
|
||||
f"judge API unreachable {n_tx} turns in a row (check auxiliary.goal_judge provider/key in config.yaml)",
|
||||
"continue", reason,
|
||||
f"⏸ Goal paused — judge API returned errors ({n_tx} turns). Check the goal_judge provider/key in "
|
||||
+ _JUDGE_CONFIG_HINT.format(provider="deepseek", model="deepseek-v4-flash"),
|
||||
+ _JUDGE_CONFIG_HINT.format(provider="deepseek", model="deepseek-flash"),
|
||||
)
|
||||
if n_parse >= DEFAULT_MAX_CONSECUTIVE_PARSE_FAILURES:
|
||||
return self._pause_decision(
|
||||
|
||||
@@ -86,18 +86,17 @@ _CATALOGUE_PREFIX_REPAIR_PROVIDERS: frozenset[str] = frozenset({
|
||||
_LOWERCASE_MODEL_PROVIDERS: frozenset[str] = frozenset({
|
||||
"xiaomi"})
|
||||
|
||||
# DeepSeek's direct API only accepts first-class V-series IDs after the 2026-07-24 cut-off (HTTP 400
|
||||
# otherwise). Both retired aliases map to deepseek-v4-flash per the official docs (thinking mode is
|
||||
# controlled by extra_body.thinking on the profile), so saved configs can't keep sending them.
|
||||
# DeepSeek's direct API only accepts first-class ids after the 2026-07-24 cut-off (HTTP 400
|
||||
# otherwise). Retired aliases fold onto the version-less ``deepseek-flash`` (V4.1-Flash, 2026-09-10;
|
||||
# thinking mode is controlled by extra_body.thinking on the profile), so saved configs can't keep
|
||||
# sending them.
|
||||
_DEEPSEEK_RETIRED_ALIASES: frozenset[str] = frozenset({
|
||||
"deepseek-chat", "deepseek-reasoner"})
|
||||
|
||||
# ``deepseek-flash`` is the version-less canonical Flash id from the 2026-09 Flash refresh:
|
||||
# ``GET /v1/models`` reports it and the API accepts it directly. It carries no ``v<N>``
|
||||
# marker, so without an entry here it misses the V-series regex below and the id the user
|
||||
# picked is folded onto ``deepseek-v4-flash`` before it ever reaches the wire.
|
||||
# ``deepseek-flash`` carries no ``v<N>`` marker, so it needs an entry here or the V-series regex
|
||||
# below misses it and the id the user picked is rewritten before it reaches the wire.
|
||||
_DEEPSEEK_CANONICAL_MODELS: frozenset[str] = frozenset({
|
||||
"deepseek-v4-pro", "deepseek-v4-flash", "deepseek-flash"})
|
||||
"deepseek-flash", "deepseek-v4-pro"})
|
||||
|
||||
# First-class V-series IDs incl. future ``deepseek-v5-*`` and dated variants
|
||||
# (``deepseek-v4-flash-20260423``): verified real model ids, NOT aliases of ``deepseek-chat``.
|
||||
@@ -106,12 +105,12 @@ _DEEPSEEK_V_SERIES_RE = re.compile(r"^deepseek-v\d+([-.].+)?$")
|
||||
|
||||
def _normalize_for_deepseek(model_name: str) -> str:
|
||||
"""Map a model input to a DeepSeek-accepted id: canonicals and ``deepseek-v<digit>…`` pass
|
||||
through (future V-series work without a release); retired aliases and everything else become
|
||||
``deepseek-v4-flash``."""
|
||||
through (dated variants and future V-series work without a release); retired aliases and
|
||||
everything else become ``deepseek-flash``."""
|
||||
bare = _strip_vendor_prefix(model_name).lower()
|
||||
if bare in _DEEPSEEK_CANONICAL_MODELS or _DEEPSEEK_V_SERIES_RE.match(bare):
|
||||
return bare
|
||||
return "deepseek-v4-flash"
|
||||
return "deepseek-flash"
|
||||
|
||||
|
||||
def _strip_vendor_prefix(model_name: str) -> str:
|
||||
|
||||
@@ -205,7 +205,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
|
||||
"claude-sonnet-4-6", "claude-opus-4-5-20251101", "claude-sonnet-4-5-20250929",
|
||||
"claude-opus-4-20250514", "claude-sonnet-4-20250514", "claude-haiku-4-5-20251001",
|
||||
],
|
||||
"deepseek": ["deepseek-v4-pro", "deepseek-v4-flash"],
|
||||
"deepseek": ["deepseek-v4-pro", "deepseek-flash"],
|
||||
"xiaomi": ["mimo-v2.5-pro", "mimo-v2.5", "mimo-v2-pro", "mimo-v2-omni", "mimo-v2-flash"],
|
||||
"tencent-tokenhub": list(_TENCENT_MODELS),
|
||||
"tencent-tokenplan": list(_TENCENT_MODELS),
|
||||
|
||||
@@ -54,8 +54,8 @@ class DeepSeekProfile(ProviderProfile):
|
||||
deepseek = DeepSeekProfile(
|
||||
name="deepseek", aliases=("deepseek-chat",), env_vars=("DEEPSEEK_API_KEY",), display_name="DeepSeek",
|
||||
description="DeepSeek — native DeepSeek API", signup_url="https://platform.deepseek.com/",
|
||||
fallback_models=("deepseek-v4-pro", "deepseek-v4-flash"), base_url="https://api.deepseek.com/v1",
|
||||
default_aux_model="deepseek-v4-flash",
|
||||
fallback_models=("deepseek-v4-pro", "deepseek-flash"), base_url="https://api.deepseek.com/v1",
|
||||
default_aux_model="deepseek-flash",
|
||||
)
|
||||
|
||||
register_provider(deepseek)
|
||||
|
||||
@@ -123,11 +123,12 @@ def test_deepseek_v4_pro_pricing_entry_exists():
|
||||
)
|
||||
|
||||
assert entry is not None
|
||||
assert entry.input_cost_per_million is not None
|
||||
assert entry.output_cost_per_million is not None
|
||||
assert float(entry.input_cost_per_million) == 0.435
|
||||
assert float(entry.output_cost_per_million) == 0.87
|
||||
assert float(entry.cache_read_cost_per_million) == 0.003625
|
||||
assert entry.source == "official_docs_snapshot"
|
||||
# Pro is the premium tier: every rate must sit above the Flash row's.
|
||||
flash = get_pricing_entry("deepseek-flash", provider="deepseek")
|
||||
assert entry.input_cost_per_million > flash.input_cost_per_million
|
||||
assert entry.output_cost_per_million > flash.output_cost_per_million
|
||||
assert entry.cache_read_cost_per_million > flash.cache_read_cost_per_million
|
||||
|
||||
|
||||
def test_bundled_pricing_skips_endpoint_metadata(monkeypatch):
|
||||
@@ -174,14 +175,13 @@ def test_unknown_model_falls_back_to_endpoint_metadata(monkeypatch):
|
||||
|
||||
|
||||
|
||||
def test_deepseek_deprecated_aliases_price_as_v4_flash():
|
||||
"""Invariant: deepseek-chat / deepseek-reasoner are deprecated aliases for
|
||||
deepseek-v4-flash's non-thinking / thinking modes (deprecation 2026-07-24)
|
||||
— they must bill at identical rates to the flash entry, or sessions on the
|
||||
legacy names over/under-report cost."""
|
||||
flash = get_pricing_entry("deepseek-v4-flash", provider="deepseek")
|
||||
def test_deepseek_deprecated_aliases_price_as_flash():
|
||||
"""Invariant: deepseek-v4-flash / deepseek-chat / deepseek-reasoner are retired aliases
|
||||
served by the current Flash model — they must bill at identical rates to the
|
||||
``deepseek-flash`` entry, or sessions on the legacy names over/under-report cost."""
|
||||
flash = get_pricing_entry("deepseek-flash", provider="deepseek")
|
||||
assert flash is not None
|
||||
for alias in ("deepseek-chat", "deepseek-reasoner"):
|
||||
for alias in ("deepseek-v4-flash", "deepseek-chat", "deepseek-reasoner"):
|
||||
entry = get_pricing_entry(alias, provider="deepseek")
|
||||
assert entry is not None, alias
|
||||
assert entry.input_cost_per_million == flash.input_cost_per_million, alias
|
||||
|
||||
@@ -129,7 +129,7 @@ class TestDeepseekVSeriesPassThrough:
|
||||
# ── DeepSeek post-2026-07-24 alias remapping ───────────────────────────
|
||||
|
||||
class TestDeepseekCanonicalAndReasonerMapping:
|
||||
"""Retired aliases and fuzzy names rewrite to deepseek-v4-flash.
|
||||
"""Retired aliases and fuzzy names rewrite to deepseek-flash.
|
||||
|
||||
DeepSeek cut off ``deepseek-chat`` / ``deepseek-reasoner`` on
|
||||
2026-07-24; sending them on the wire returns HTTP 400.
|
||||
@@ -139,7 +139,7 @@ class TestDeepseekCanonicalAndReasonerMapping:
|
||||
def test_provider_path_rewrites_reasoner(self):
|
||||
assert (
|
||||
normalize_model_for_provider("deepseek-reasoner", "deepseek")
|
||||
== "deepseek-v4-flash"
|
||||
== "deepseek-flash"
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize("model", [
|
||||
@@ -149,8 +149,8 @@ class TestDeepseekCanonicalAndReasonerMapping:
|
||||
"deepseek-reasoning-preview",
|
||||
"deepseek-cot-experimental",
|
||||
])
|
||||
def test_reasoner_keywords_map_to_v4_flash(self, model):
|
||||
assert _normalize_for_deepseek(model) == "deepseek-v4-flash"
|
||||
def test_reasoner_keywords_map_to_flash(self, model):
|
||||
assert _normalize_for_deepseek(model) == "deepseek-flash"
|
||||
|
||||
|
||||
# ── Regression: issue #78796 ───────────────────────────────────────────
|
||||
|
||||
@@ -190,16 +190,15 @@ class TestDeepSeekAuxModel:
|
||||
system.
|
||||
"""
|
||||
|
||||
def test_profile_advertises_deepseek_v4_flash(self, deepseek_profile):
|
||||
assert deepseek_profile.default_aux_model == "deepseek-v4-flash"
|
||||
def test_profile_advertises_deepseek_flash(self, deepseek_profile):
|
||||
assert deepseek_profile.default_aux_model == "deepseek-flash"
|
||||
|
||||
def test_fallback_models_are_v4_only(self, deepseek_profile):
|
||||
assert deepseek_profile.fallback_models == (
|
||||
"deepseek-v4-pro",
|
||||
"deepseek-v4-flash",
|
||||
)
|
||||
def test_fallback_models_are_current_ids(self, deepseek_profile):
|
||||
from hermes_cli.model_normalize import _normalize_for_deepseek
|
||||
# Every advertised id must survive normalization unchanged (no retired alias in the picker).
|
||||
assert all(_normalize_for_deepseek(m) == m for m in deepseek_profile.fallback_models)
|
||||
|
||||
def test_consumer_api_returns_deepseek_v4_flash(self):
|
||||
def test_consumer_api_returns_deepseek_flash(self):
|
||||
from agent.auxiliary_client import _get_aux_model_for_provider
|
||||
assert _get_aux_model_for_provider("deepseek") == "deepseek-v4-flash"
|
||||
assert _get_aux_model_for_provider("deepseek") == "deepseek-flash"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user