diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 723174f87d..8171aec797 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -146,8 +146,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> return thinking_config if effort not in {"minimal", "low", "medium", "high", "xhigh", "max", "ultra"}: effort = "medium" - # Gemini 3 Flash documents low/medium/high; Gemini 3 Pro only low/high. - if normalized_model.startswith(("gemini-3", "gemini-3.1")): + # Gemini 3 Flash documents low/medium/high thinking levels; Gemini 3 Pro + # is stricter (low/high). Clamp Hermes' wider effort set to what each + # family accepts so we never forward an undocumented level verbatim. + if normalized_model.startswith("gemini-3"): if "flash" in normalized_model: thinking_config["thinkingLevel"] = ( "low" if effort in {"minimal", "low"} else "high" if effort in _HIGH_EFFORTS else "medium" diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 0f73ce36f9..9830e2e28b 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -191,6 +191,9 @@ _SNAPSHOTS: tuple[tuple[str, Optional[str], str, dict], ...] = ( ("deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"): ("0.14", "0.28", "0.0028"), "deepseek-v4-pro": ("0.435", "0.87", "0.003625"), }), + ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-09-02", { + ("gemini-3.8-flash", "gemini-3.7-flash"): ("0.75", "3.75", "0.075"), + }), ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-07-28", { "gemini-3.6-flash": ("1.50", "7.50", "0.15"), "gemini-3.5-flash-lite": ("0.30", "2.50", "0.03"), }), diff --git a/hermes_cli/models_catalog_static.py b/hermes_cli/models_catalog_static.py index 97d96b9924..4772edb1ed 100644 --- a/hermes_cli/models_catalog_static.py +++ b/hermes_cli/models_catalog_static.py @@ -170,6 +170,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro", ], "gemini": [ + "gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash", "gemini-3.1-flash-lite-preview", ], "zai": [ @@ -222,8 +223,8 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1", "gpt-5.1-codex", "gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex", "gpt-5-nano", "claude-fable-5", "claude-opus-5", "claude-sonnet-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5", - "claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash", - "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash", + "claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.8-flash", + "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash", "grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3", "minimax-m2.7", "minimax-m2.5", "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-free", "qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free", @@ -280,6 +281,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # only shows the configured model. IDs carry the "google/" publisher prefix Vertex expects # (see hermes_cli/model_setup_flows.py); validated live against a GCP project (global region). "vertex": [ + "google/gemini-3.8-flash", "google/gemini-3.7-flash", "google/gemini-3.1-pro-preview", "google/gemini-3-pro-preview", "google/gemini-3.6-flash", "google/gemini-3.5-flash", "google/gemini-3.5-flash-lite", "google/gemini-3-flash-preview", "google/gemini-3.1-flash-lite-preview", "google/gemini-3.1-flash-lite", diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index 6e82adfffd..a81ce3a12c 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -201,7 +201,7 @@ openrouter = OpenRouterProfile( base_url="https://openrouter.ai/api/v1", models_url="https://openrouter.ai/api/v1/models", fallback_models=( "anthropic/claude-sonnet-4.6", "openai/gpt-5.4", "deepseek/deepseek-chat", "google/gemini-3.8-flash", - "qwen/qwen3-plus", + "google/gemini-3.7-flash", "qwen/qwen3-plus", ), ) diff --git a/tests/agent/transports/test_chat_completions.py b/tests/agent/transports/test_chat_completions.py index 27f1767b9f..2531b10ea4 100644 --- a/tests/agent/transports/test_chat_completions.py +++ b/tests/agent/transports/test_chat_completions.py @@ -393,6 +393,20 @@ class TestChatCompletionsBuildKwargs: assert kw["max_tokens"] == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS assert kw["extra_body"]["thinking_config"]["thinkingLevel"] == "high" + # Also verify gemini-3.8-flash gets headroom and correct thinking level + kw38 = transport.build_kwargs( + model="gemini-3.8-flash", + messages=[{"role": "user", "content": "Hi"}], + provider_profile=profile, + provider_name="gemini", + base_url=profile.base_url, + max_tokens=4096, + max_tokens_param_fn=lambda n: {"max_tokens": n}, + reasoning_config={"enabled": True, "effort": "medium"}, + ) + assert kw38["max_tokens"] == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS + assert kw38["extra_body"]["thinking_config"]["thinkingLevel"] == "medium" + def test_gemini_without_thinking_keeps_explicit_max_tokens(self, transport): from providers import get_provider_profile