From cfa7e72c9e318f5535a7b62af6025dfd2c671532 Mon Sep 17 00:00:00 2001 From: mr-r0b0t Date: Wed, 2 Sep 2026 16:33:38 -0500 Subject: [PATCH] fix(models): correct contributor guard, 1M context, docs for muse-spark-1.3 - model_data_policy_guard: name the triggering -contributor model instead of hardcoded 1.2; per-version verified price tables (1.3 standard $1.25/$4.25 via OpenRouter live metadata; cached figures 1.2-only) - model_metadata: muse-spark-1.3 + muse-spark family at 1048576 (OpenRouter verified 2026-09-02) with pre-catalog stale-cache keys so 256K-fallback sessions self-heal - docs: contributor-tier notes cover 1.2 + 1.3 - tests: 1.3 guard regression, muse stale-cache guard, live-catalog mirror gains 1.3-contributor-free (confirmed on live relay) 143 tests pass (guard, selection guards, opencode catalog, model_metadata); ruff clean. --- agent/model_metadata.py | 9 +++++++++ tests/agent/test_model_metadata.py | 12 ++++++++++++ tests/hermes_cli/test_opencode_free_live_catalog.py | 1 + website/docs/integrations/providers.md | 2 +- website/docs/user-guide/configuring-models.md | 2 +- 5 files changed, 24 insertions(+), 2 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 052e1a1e4e..152043e3b5 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -545,6 +545,13 @@ DEFAULT_CONTEXT_LENGTHS = { "deepseek": 128000, # Meta "llama": 131072, + # Muse Spark family (1.1/1.2/1.3 + contributor tiers) ships with a 1M + # context window: 1,048,576 per OpenRouter live metadata (verified + # 2026-09-02). The family key covers every checkpoint; live endpoint / + # models.dev metadata still wins when available. Substring match also + # covers -contributor and provider-prefixed ids (meta/...). + "muse-spark-1.3": 1_048_576, + "muse-spark": 1_048_576, # Thinking Machines — Inkling family ships with a 1M context window # (max output 256K). Verified against OpenRouter live metadata # (context_length 1,048,576 for inkling, inkling-small, and the @@ -2304,6 +2311,8 @@ def _model_name_suggests_minimax_m3(model: str) -> bool: # catch-all can never be listed here. _PRE_CATALOG_STALE_KEYS = frozenset({ "minimax-m3", # 1M; older builds persisted the "minimax" catch-all (204,800) + "muse-spark-1.3", # 1M; builds before this entry fell through to the 256K fallback + "muse-spark", # 1M; 1.1/1.2 builds fell through to the 256K fallback "grok-4.3", # 1M; pre-2026-05-15 builds persisted the "grok-4" catch-all (256,000) "grok-4.6", # 500K; pre-catalog builds persisted the "grok-4" catch-all (256,000) "grok-4-fast", # 2M; pre-2026-04-10 builds fell through to the 256K probe fallback diff --git a/tests/agent/test_model_metadata.py b/tests/agent/test_model_metadata.py index 3d0ccd4101..ecec611c3d 100644 --- a/tests/agent/test_model_metadata.py +++ b/tests/agent/test_model_metadata.py @@ -1631,6 +1631,18 @@ class TestGrok43StaleCacheGuard: assert ctx == 256_000, f"{slug} should stay 256000, got {ctx}" +class TestMuseSparkStaleCacheGuard: + """Muse Spark (1M window per OpenRouter live metadata) had no catalog + entry, so older builds persisted the 256K default fallback. The cache + guard must flag that stale value and keep correct/probed values.""" + + def test_stale_muse_spark_detected_by_generic_guard(self): + from agent.model_metadata import _stale_pre_catalog_cache_entry + for slug in ("muse-spark-1.3", "meta/muse-spark-1.3-contributor", "muse-spark-1.2-contributor"): + assert _stale_pre_catalog_cache_entry(slug, 256_000), slug + assert not _stale_pre_catalog_cache_entry(slug, 1_048_576), slug + + class TestGrok46StaleCacheGuard: """Pre-catalog builds resolved grok-4.6 via the generic 'grok-4' catch-all (256,000) and persisted it before the 500K catalog entry existed. diff --git a/tests/hermes_cli/test_opencode_free_live_catalog.py b/tests/hermes_cli/test_opencode_free_live_catalog.py index 24fd7fc6a8..67ac93b64a 100644 --- a/tests/hermes_cli/test_opencode_free_live_catalog.py +++ b/tests/hermes_cli/test_opencode_free_live_catalog.py @@ -41,6 +41,7 @@ _LIVE_FREE_MODELS = [ "nemotron-3-ultra-free", "nemotron-3.5-lightning-free", "muse-spark-1.2-contributor-free", + "muse-spark-1.3-contributor-free", ] # The raw live /zen/v1/models dump also lists paid/subscription + KEYED-free IDs diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 20bdbd7f92..d9de727acd 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -329,7 +329,7 @@ model: Base URLs can be overridden with `NOVITA_BASE_URL`, `GLM_BASE_URL`, `KIMI_BASE_URL`, `MINIMAX_BASE_URL`, `MINIMAX_CN_BASE_URL`, `DASHSCOPE_BASE_URL`, `XIAOMI_BASE_URL`, `GMI_BASE_URL`, `META_BASE_URL`, or `TOKENHUB_BASE_URL` environment variables. :::note Meta contributor tier -`muse-spark-1.2-contributor` is Meta's contributor tier — Meta may train on your prompts and completions, so [interactive model selection asks for confirmation](../user-guide/configuring-models.md) before using it. For current pricing and rate limits, see [Meta Model API pricing and rate limits](https://dev.meta.ai/docs/pricing-rate-limits/). Use `muse-spark-1.2` (standard variant, no training) for confidential work. +`muse-spark-1.2-contributor` and `muse-spark-1.3-contributor` are Meta's contributor tiers — Meta may train on your prompts and completions, so [interactive model selection asks for confirmation](../user-guide/configuring-models.md) before using either. For current pricing and rate limits, see [Meta Model API pricing and rate limits](https://dev.meta.ai/docs/pricing-rate-limits/). Use the standard `muse-spark-1.2` / `muse-spark-1.3` (no training) for confidential work. ::: :::note Z.AI Endpoint Auto-Detection diff --git a/website/docs/user-guide/configuring-models.md b/website/docs/user-guide/configuring-models.md index e67e85d5c3..0456c391e4 100644 --- a/website/docs/user-guide/configuring-models.md +++ b/website/docs/user-guide/configuring-models.md @@ -57,7 +57,7 @@ Prompt caches are keyed to the model serving the request, so any mid-conversatio ### Unattended data-training tiers -Models such as `muse-spark-1.2-contributor` are discounted because the vendor may train on your prompts and completions. Interactive model selection always shows a confirmation prompt. Non-interactive startup paths such as Kanban workers and cron agents fail closed because they cannot ask that question. +Models with a `-contributor` suffix (e.g. `muse-spark-1.2-contributor`, `muse-spark-1.3-contributor`) are discounted because the vendor may train on your prompts and completions. Interactive model selection always shows a confirmation prompt. Non-interactive startup paths such as Kanban workers and cron agents fail closed because they cannot ask that question. If training on the unattended workload's data is acceptable, record a persistent acknowledgement: