fix(deepseek): deepseek-flash is the canonical Flash id; retired names fold onto it

DeepSeek retired deepseek-v4-flash on 2026-09-10 (V4.1-Flash release); the API's
model name is now `deepseek-flash` and /v1/models lists only it. Hermes still
folded every non-V-series name onto deepseek-v4-flash, so `/model deepseek-flash`
on the DeepSeek provider was rewritten, then the validator "auto-corrected" it
back against the live listing: "Auto-corrected deepseek-v4-flash -> deepseek-flash"
on every switch.

Retired aliases (deepseek-chat / -reasoner and other fuzzy names) now fold onto
deepseek-flash; the curated catalog, profile fallback list, aux default, goal-judge
hint and pricing snapshot (2026-09-10 off-peak USD) follow the docs. Dated
deepseek-v4-* ids still pass through untouched.

Builds on YipTszkwan's #107126 (earliest fix in the cluster).
This commit is contained in:
Teknium
2026-09-10 02:15:18 -07:00
parent 215fd0ecb9
commit aeecb110f8
8 changed files with 43 additions and 45 deletions

View File

@@ -186,11 +186,11 @@ _SNAPSHOTS: tuple[tuple[str, Optional[str], str, dict], ...] = (
"gpt-4.1-nano": ("0.10", "0.40", "0.025"), "o3": ("10.00", "40.00", "2.50"),
"o3-mini": ("1.10", "4.40", "0.55"),
}),
# deepseek-chat / deepseek-reasoner are deprecated aliases of
# deepseek-v4-flash's non-thinking / thinking modes — same rates.
("deepseek", "https://api-docs.deepseek.com/quick_start/pricing", "deepseek-pricing-2026-07", {
("deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"): ("0.14", "0.28", "0.0028"),
"deepseek-v4-pro": ("0.435", "0.87", "0.003625"),
# Off-peak USD rates (peak = 2x, Mon-Fri 01-04 + 06-10 UTC). ``deepseek-v4-flash`` and the
# retired deepseek-chat / deepseek-reasoner aliases are served by V4.1-Flash at the Flash price.
("deepseek", "https://api-docs.deepseek.com/quick_start/pricing", "deepseek-pricing-2026-09-10", {
("deepseek-flash", "deepseek-v4-flash", "deepseek-chat", "deepseek-reasoner"): ("0.15", "0.60", "0.003"),
"deepseek-v4-pro": ("0.66", "1.98", "0.022"),
}),
("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-09-02", {
("gemini-3.8-flash", "gemini-3.7-flash"): ("0.75", "3.75", "0.075"),

View File

@@ -1499,7 +1499,7 @@ class GoalManager:
f"judge API unreachable {n_tx} turns in a row (check auxiliary.goal_judge provider/key in config.yaml)",
"continue", reason,
f"⏸ Goal paused — judge API returned errors ({n_tx} turns). Check the goal_judge provider/key in "
+ _JUDGE_CONFIG_HINT.format(provider="deepseek", model="deepseek-v4-flash"),
+ _JUDGE_CONFIG_HINT.format(provider="deepseek", model="deepseek-flash"),
)
if n_parse >= DEFAULT_MAX_CONSECUTIVE_PARSE_FAILURES:
return self._pause_decision(

View File

@@ -86,18 +86,17 @@ _CATALOGUE_PREFIX_REPAIR_PROVIDERS: frozenset[str] = frozenset({
_LOWERCASE_MODEL_PROVIDERS: frozenset[str] = frozenset({
"xiaomi"})
# DeepSeek's direct API only accepts first-class V-series IDs after the 2026-07-24 cut-off (HTTP 400
# otherwise). Both retired aliases map to deepseek-v4-flash per the official docs (thinking mode is
# controlled by extra_body.thinking on the profile), so saved configs can't keep sending them.
# DeepSeek's direct API only accepts first-class ids after the 2026-07-24 cut-off (HTTP 400
# otherwise). Retired aliases fold onto the version-less ``deepseek-flash`` (V4.1-Flash, 2026-09-10;
# thinking mode is controlled by extra_body.thinking on the profile), so saved configs can't keep
# sending them.
_DEEPSEEK_RETIRED_ALIASES: frozenset[str] = frozenset({
"deepseek-chat", "deepseek-reasoner"})
# ``deepseek-flash`` is the version-less canonical Flash id from the 2026-09 Flash refresh:
# ``GET /v1/models`` reports it and the API accepts it directly. It carries no ``v<N>``
# marker, so without an entry here it misses the V-series regex below and the id the user
# picked is folded onto ``deepseek-v4-flash`` before it ever reaches the wire.
# ``deepseek-flash`` carries no ``v<N>`` marker, so it needs an entry here or the V-series regex
# below misses it and the id the user picked is rewritten before it reaches the wire.
_DEEPSEEK_CANONICAL_MODELS: frozenset[str] = frozenset({
"deepseek-v4-pro", "deepseek-v4-flash", "deepseek-flash"})
"deepseek-flash", "deepseek-v4-pro"})
# First-class V-series IDs incl. future ``deepseek-v5-*`` and dated variants
# (``deepseek-v4-flash-20260423``): verified real model ids, NOT aliases of ``deepseek-chat``.
@@ -106,12 +105,12 @@ _DEEPSEEK_V_SERIES_RE = re.compile(r"^deepseek-v\d+([-.].+)?$")
def _normalize_for_deepseek(model_name: str) -> str:
"""Map a model input to a DeepSeek-accepted id: canonicals and ``deepseek-v<digit>…`` pass
through (future V-series work without a release); retired aliases and everything else become
``deepseek-v4-flash``."""
through (dated variants and future V-series work without a release); retired aliases and
everything else become ``deepseek-flash``."""
bare = _strip_vendor_prefix(model_name).lower()
if bare in _DEEPSEEK_CANONICAL_MODELS or _DEEPSEEK_V_SERIES_RE.match(bare):
return bare
return "deepseek-v4-flash"
return "deepseek-flash"
def _strip_vendor_prefix(model_name: str) -> str:

View File

@@ -205,7 +205,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
"claude-sonnet-4-6", "claude-opus-4-5-20251101", "claude-sonnet-4-5-20250929",
"claude-opus-4-20250514", "claude-sonnet-4-20250514", "claude-haiku-4-5-20251001",
],
"deepseek": ["deepseek-v4-pro", "deepseek-v4-flash"],
"deepseek": ["deepseek-v4-pro", "deepseek-flash"],
"xiaomi": ["mimo-v2.5-pro", "mimo-v2.5", "mimo-v2-pro", "mimo-v2-omni", "mimo-v2-flash"],
"tencent-tokenhub": list(_TENCENT_MODELS),
"tencent-tokenplan": list(_TENCENT_MODELS),

View File

@@ -54,8 +54,8 @@ class DeepSeekProfile(ProviderProfile):
deepseek = DeepSeekProfile(
name="deepseek", aliases=("deepseek-chat",), env_vars=("DEEPSEEK_API_KEY",), display_name="DeepSeek",
description="DeepSeek — native DeepSeek API", signup_url="https://platform.deepseek.com/",
fallback_models=("deepseek-v4-pro", "deepseek-v4-flash"), base_url="https://api.deepseek.com/v1",
default_aux_model="deepseek-v4-flash",
fallback_models=("deepseek-v4-pro", "deepseek-flash"), base_url="https://api.deepseek.com/v1",
default_aux_model="deepseek-flash",
)
register_provider(deepseek)

View File

@@ -123,11 +123,12 @@ def test_deepseek_v4_pro_pricing_entry_exists():
)
assert entry is not None
assert entry.input_cost_per_million is not None
assert entry.output_cost_per_million is not None
assert float(entry.input_cost_per_million) == 0.435
assert float(entry.output_cost_per_million) == 0.87
assert float(entry.cache_read_cost_per_million) == 0.003625
assert entry.source == "official_docs_snapshot"
# Pro is the premium tier: every rate must sit above the Flash row's.
flash = get_pricing_entry("deepseek-flash", provider="deepseek")
assert entry.input_cost_per_million > flash.input_cost_per_million
assert entry.output_cost_per_million > flash.output_cost_per_million
assert entry.cache_read_cost_per_million > flash.cache_read_cost_per_million
def test_bundled_pricing_skips_endpoint_metadata(monkeypatch):
@@ -174,14 +175,13 @@ def test_unknown_model_falls_back_to_endpoint_metadata(monkeypatch):
def test_deepseek_deprecated_aliases_price_as_v4_flash():
"""Invariant: deepseek-chat / deepseek-reasoner are deprecated aliases for
deepseek-v4-flash's non-thinking / thinking modes (deprecation 2026-07-24)
— they must bill at identical rates to the flash entry, or sessions on the
legacy names over/under-report cost."""
flash = get_pricing_entry("deepseek-v4-flash", provider="deepseek")
def test_deepseek_deprecated_aliases_price_as_flash():
"""Invariant: deepseek-v4-flash / deepseek-chat / deepseek-reasoner are retired aliases
served by the current Flash model — they must bill at identical rates to the
``deepseek-flash`` entry, or sessions on the legacy names over/under-report cost."""
flash = get_pricing_entry("deepseek-flash", provider="deepseek")
assert flash is not None
for alias in ("deepseek-chat", "deepseek-reasoner"):
for alias in ("deepseek-v4-flash", "deepseek-chat", "deepseek-reasoner"):
entry = get_pricing_entry(alias, provider="deepseek")
assert entry is not None, alias
assert entry.input_cost_per_million == flash.input_cost_per_million, alias

View File

@@ -129,7 +129,7 @@ class TestDeepseekVSeriesPassThrough:
# ── DeepSeek post-2026-07-24 alias remapping ───────────────────────────
class TestDeepseekCanonicalAndReasonerMapping:
"""Retired aliases and fuzzy names rewrite to deepseek-v4-flash.
"""Retired aliases and fuzzy names rewrite to deepseek-flash.
DeepSeek cut off ``deepseek-chat`` / ``deepseek-reasoner`` on
2026-07-24; sending them on the wire returns HTTP 400.
@@ -139,7 +139,7 @@ class TestDeepseekCanonicalAndReasonerMapping:
def test_provider_path_rewrites_reasoner(self):
assert (
normalize_model_for_provider("deepseek-reasoner", "deepseek")
== "deepseek-v4-flash"
== "deepseek-flash"
)
@pytest.mark.parametrize("model", [
@@ -149,8 +149,8 @@ class TestDeepseekCanonicalAndReasonerMapping:
"deepseek-reasoning-preview",
"deepseek-cot-experimental",
])
def test_reasoner_keywords_map_to_v4_flash(self, model):
assert _normalize_for_deepseek(model) == "deepseek-v4-flash"
def test_reasoner_keywords_map_to_flash(self, model):
assert _normalize_for_deepseek(model) == "deepseek-flash"
# ── Regression: issue #78796 ───────────────────────────────────────────

View File

@@ -190,16 +190,15 @@ class TestDeepSeekAuxModel:
system.
"""
def test_profile_advertises_deepseek_v4_flash(self, deepseek_profile):
assert deepseek_profile.default_aux_model == "deepseek-v4-flash"
def test_profile_advertises_deepseek_flash(self, deepseek_profile):
assert deepseek_profile.default_aux_model == "deepseek-flash"
def test_fallback_models_are_v4_only(self, deepseek_profile):
assert deepseek_profile.fallback_models == (
"deepseek-v4-pro",
"deepseek-v4-flash",
)
def test_fallback_models_are_current_ids(self, deepseek_profile):
from hermes_cli.model_normalize import _normalize_for_deepseek
# Every advertised id must survive normalization unchanged (no retired alias in the picker).
assert all(_normalize_for_deepseek(m) == m for m in deepseek_profile.fallback_models)
def test_consumer_api_returns_deepseek_v4_flash(self):
def test_consumer_api_returns_deepseek_flash(self):
from agent.auxiliary_client import _get_aux_model_for_provider
assert _get_aux_model_for_provider("deepseek") == "deepseek-v4-flash"
assert _get_aux_model_for_provider("deepseek") == "deepseek-flash"