diff --git a/plugins/video_gen/fal/__init__.py b/plugins/video_gen/fal/__init__.py index e4aaa7431d..3ce20e3d5e 100644 --- a/plugins/video_gen/fal/__init__.py +++ b/plugins/video_gen/fal/__init__.py @@ -7,28 +7,32 @@ called without ``image_url``, and to its image-to-video endpoint when ``image_url`` is provided. The agent never sees the routing — it just calls ``video_generate(prompt=..., image_url=...)``. -Model families (each with t2v + i2v endpoints): +Model families (most expose both t2v + i2v; gemini-omni-flash is image-to-video only): Cheap tier: - ltx-2.3 fal-ai/ltx-2.3-22b/text-to-video / fal-ai/ltx-2.3-22b/image-to-video - pixverse-v6 fal-ai/pixverse/v6/text-to-video / fal-ai/pixverse/v6/image-to-video + ltx-2.3 fal-ai/ltx-2.3-22b/text-to-video / fal-ai/ltx-2.3-22b/image-to-video + pixverse-v6 fal-ai/pixverse/v6/text-to-video / fal-ai/pixverse/v6/image-to-video + seedance-2.0-mini bytedance/seedance-2.0/mini/text-to-video / bytedance/seedance-2.0/mini/image-to-video Premium tier: - veo3.1 fal-ai/veo3.1 / fal-ai/veo3.1/image-to-video - seedance-2.0 bytedance/seedance-2.0/text-to-video / bytedance/seedance-2.0/image-to-video - seedance-2.5 bytedance/seedance-2.5/text-to-video / bytedance/seedance-2.5/image-to-video - minimax-h3 minimax/h3/text-to-video / minimax/h3/image-to-video - kling-v3-4k fal-ai/kling-video/v3/4k/text-to-video / fal-ai/kling-video/v3/4k/image-to-video - happy-horse alibaba/happy-horse/text-to-video / alibaba/happy-horse/image-to-video + veo3.1 fal-ai/veo3.1 / fal-ai/veo3.1/image-to-video + seedance-2.0 bytedance/seedance-2.0/text-to-video / bytedance/seedance-2.0/image-to-video + seedance-2.5 bytedance/seedance-2.5/text-to-video / bytedance/seedance-2.5/image-to-video + minimax-h3 minimax/h3/text-to-video / minimax/h3/image-to-video + flux-3 blackforestlabs/flux-3/text-to-video / blackforestlabs/flux-3/image-to-video + grok-imagine-1.5 xai/grok-imagine-video/v1.5/text-to-video / xai/grok-imagine-video/v1.5/image-to-video + kling-v3-4k fal-ai/kling-video/v3/4k/text-to-video / fal-ai/kling-video/v3/4k/image-to-video + happy-horse alibaba/happy-horse/text-to-video / alibaba/happy-horse/image-to-video - Cheap tier (continued): - seedance-2.0-mini bytedance/seedance-2.0/mini/text-to-video / bytedance/seedance-2.0/mini/image-to-video + Image-to-video only (no text_endpoint): + gemini-omni-flash google/gemini-omni-flash/image-to-video Selection precedence for the active family: 1. ``model=`` arg from the tool call 2. ``FAL_VIDEO_MODEL`` env var 3. ``video_gen.fal.model`` in ``config.yaml`` - 4. ``video_gen.model`` in ``config.yaml`` (when it's one of our family IDs) + 4. ``video_gen.model`` in ``config.yaml`` (when it's one of our family IDs + or a full endpoint path that contains a family ID) 5. ``DEFAULT_MODEL`` Authentication via ``FAL_KEY`` or the managed Nous gateway. Output is an @@ -68,6 +72,10 @@ logger = logging.getLogger(__name__) # (heuristic: 2-element with gap > 1 is a range) # audio : True if generate_audio is supported # negative : True if negative_prompt is supported +# seed : False when the endpoint declares no `seed` field +# (absent = True, so existing families keep sending it) +# duration_int : True when FAL types duration as an integer rather than +# the usual queue-API string FAL_FAMILIES: Dict[str, Dict[str, Any]] = { # ─── Cheap / fast tier ───────────────────────────────────────────── @@ -114,6 +122,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (4, 15), "audio": True, "negative": False, + "seed": False, }, # ─── Expensive / premium tier ────────────────────────────────────── "veo3.1": { @@ -146,6 +155,8 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (4, 15), "audio": True, "negative": False, + # FAL input schema has no `seed` (only returned on output). + "seed": False, }, "seedance-2.5": { "display": "Seedance 2.5", @@ -164,6 +175,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (4, 30), "audio": True, "negative": False, + "seed": False, }, "minimax-h3": { "display": "MiniMax H3", @@ -190,6 +202,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (5, 15), "audio": False, # audio is native/always-on; no generate_audio key "negative": False, + "seed": False, }, "flux-3": { "display": "FLUX 3 (via FAL)", @@ -206,6 +219,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (5, 20), "audio": True, "negative": False, + "seed": False, }, "grok-imagine-1.5": { "display": "Grok Imagine 1.5 (via FAL)", @@ -223,6 +237,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (1, 15), "audio": False, # audio is native; no generate_audio key "negative": False, + "seed": False, }, "gemini-omni-flash": { "display": "Gemini Omni Flash (via FAL)", @@ -239,6 +254,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "durations": (3, 10), "audio": False, # audio is native; no generate_audio key "negative": False, + "seed": False, }, "kling-v3-4k": { "display": "Kling v3 4K", @@ -323,6 +339,61 @@ def _load_video_gen_section() -> Dict[str, Any]: return {} +_ENDPOINT_MODALITY_LEAVES = frozenset({"text-to-video", "image-to-video"}) + + +def _normalize_family_key(c: str) -> Optional[str]: + """Try to extract a known family ID from a model string. + + Handles bare IDs (``seedance-2.5``), full endpoint paths + (``bytedance/seedance-2.5/text-to-video``), truncated endpoint stems + (``minimax/h3``, ``bytedance/seedance-2.0/mini``), and provider-prefixed + names (``bytedance/seedance-2.5``). + """ + c = c.strip() + if not c: + return None + if c in FAL_FAMILIES: + return c + + # Exact declared endpoint — unambiguous, and beats any segment scan + # that would otherwise see "seedance-2.0" inside ".../seedance-2.0/mini/...". + for fid, meta in FAL_FAMILIES.items(): + if c in (meta.get("text_endpoint"), meta.get("image_endpoint")): + return fid + + # Truncated stem of a declared endpoint: "minimax/h3" or + # "bytedance/seedance-2.0/mini". The next path segment after ``c`` must + # be a modality leaf so "bytedance/seedance-2.0" does not also match the + # Mini family's deeper ".../seedance-2.0/mini/text-to-video" path. + stem_hits: List[Tuple[int, str]] = [] + for fid, meta in FAL_FAMILIES.items(): + for endpoint in (meta.get("text_endpoint"), meta.get("image_endpoint")): + if not isinstance(endpoint, str): + continue + if not endpoint.startswith(c + "/"): + continue + first = endpoint[len(c) + 1:].split("/", 1)[0] + if first in _ENDPOINT_MODALITY_LEAVES: + stem_hits.append((len(c), fid)) + break + if stem_hits: + stem_hits.sort(key=lambda item: item[0], reverse=True) + return stem_hits[0][1] + + # Longest family-id path-segment match ("bytedance/seedance-2.5" → + # seedance-2.5; prefers seedance-2.0-mini over seedance-2.0 when both + # somehow appear). + parts = set(c.split("/")) + best_fid: Optional[str] = None + best_len = -1 + for fid in FAL_FAMILIES: + if fid in parts and len(fid) > best_len: + best_fid = fid + best_len = len(fid) + return best_fid + + def _resolve_family(explicit: Optional[str]) -> Tuple[str, Dict[str, Any]]: """Decide which FAL family to use. Returns ``(family_id, meta)``.""" candidates: List[Optional[str]] = [] @@ -338,9 +409,10 @@ def _resolve_family(explicit: Optional[str]) -> Tuple[str, Dict[str, Any]]: candidates.append(top) for c in candidates: - if isinstance(c, str) and c.strip() and c.strip() in FAL_FAMILIES: - fid = c.strip() - return fid, FAL_FAMILIES[fid] + if isinstance(c, str) and c.strip(): + fid = _normalize_family_key(c) + if fid: + return fid, FAL_FAMILIES[fid] return DEFAULT_MODEL, FAL_FAMILIES[DEFAULT_MODEL] @@ -373,7 +445,10 @@ def _build_payload( # declare an override. key = family.get("image_param_key") or "image_url" payload[key] = image_url - if seed is not None: + # Several newer endpoints (seedance 2.x, minimax h3, flux-3, grok, gemini) + # declare no `seed` field, and the managed gateway forwards whatever we + # send — so gate it on the family rather than leaking an unknown key. + if seed is not None and family.get("seed", True): payload["seed"] = seed if family.get("aspect_ratios"): @@ -604,7 +679,7 @@ class FALVideoGenProvider(VideoGenProvider): modalities.append("text") if meta.get("image_endpoint"): modalities.append("image") - out.append({ + entry: Dict[str, Any] = { "id": fid, "display": meta["display"], "speed": meta["speed"], @@ -612,7 +687,15 @@ class FALVideoGenProvider(VideoGenProvider): "price": meta["price"], "tier": meta.get("tier", "premium"), "modalities": modalities, - }) + } + durs = meta.get("durations") + if durs: + if _is_duration_range(durs): + entry["min_duration"], entry["max_duration"] = durs + else: + entry["min_duration"] = min(durs) + entry["max_duration"] = max(durs) + out.append(entry) return out def default_model(self) -> Optional[str]: @@ -622,7 +705,7 @@ class FALVideoGenProvider(VideoGenProvider): return { "name": "FAL", "badge": "paid", - "tag": "LTX, Pixverse, Seedance 2.0/2.5 + Mini, MiniMax H3, Veo 3.1, Kling 4K, Happy Horse — text-to-video & image-to-video", + "tag": "LTX, Pixverse, Seedance 2.0/2.5/Mini, Veo 3.1, MiniMax H3, FLUX 3, Kling 4K, Happy Horse, Grok Imagine, Gemini Omni — text-to-video & image-to-video", "env_vars": [ { "key": "FAL_KEY", @@ -633,12 +716,26 @@ class FALVideoGenProvider(VideoGenProvider): } def capabilities(self) -> Dict[str, Any]: + # Union across families so the tool schema doesn't understate the + # longest-running models (Seedance 2.5 = 30s, FLUX 3 = 20s). + max_dur = 1 + min_dur: Optional[int] = None + for meta in FAL_FAMILIES.values(): + durs = meta.get("durations") + if not durs: + continue + if _is_duration_range(durs): + lo, hi = durs + else: + lo, hi = min(durs), max(durs) + max_dur = max(max_dur, hi) + min_dur = lo if min_dur is None else min(min_dur, lo) return { "modalities": ["text", "image"], "aspect_ratios": ["16:9", "9:16", "1:1"], "resolutions": ["360p", "540p", "720p", "1080p"], - "max_duration": 15, - "min_duration": 1, + "max_duration": max_dur, + "min_duration": min_dur if min_dur is not None else 1, "supports_audio": True, "supports_negative_prompt": True, "max_reference_images": 0, diff --git a/tests/plugins/video_gen/test_fal_plugin.py b/tests/plugins/video_gen/test_fal_plugin.py index 45771cc1a1..0ab8802ab8 100644 --- a/tests/plugins/video_gen/test_fal_plugin.py +++ b/tests/plugins/video_gen/test_fal_plugin.py @@ -233,6 +233,47 @@ class TestFamilyRouting: assert with_fake_fal["arguments"]["image_url"] == "https://example.com/dog.png" +class TestFamilyKeyNormalization: + def test_full_endpoint_paths_resolve_to_their_own_family(self): + """A configured endpoint path must resolve to the family that declares + it. The segment scan alone reads the "seedance-2.0" in + ".../seedance-2.0/mini/..." and bills the full-price family.""" + from plugins.video_gen.fal import FAL_FAMILIES, _normalize_family_key + + for fid, meta in FAL_FAMILIES.items(): + for key in ("text_endpoint", "image_endpoint"): + endpoint = meta.get(key) + if endpoint: + assert _normalize_family_key(endpoint) == fid, endpoint + + def test_bare_and_prefixed_ids_still_resolve(self): + from plugins.video_gen.fal import _normalize_family_key + + assert _normalize_family_key("seedance-2.5") == "seedance-2.5" + assert _normalize_family_key("bytedance/seedance-2.5") == "seedance-2.5" + assert _normalize_family_key(" pixverse-v6 ") == "pixverse-v6" + assert _normalize_family_key("nonsense/thing") is None + + def test_truncated_endpoint_stems_resolve(self): + """Config often stores the FAL app path without the modality leaf.""" + from plugins.video_gen.fal import _normalize_family_key + + assert _normalize_family_key("bytedance/seedance-2.0/mini") == "seedance-2.0-mini" + assert _normalize_family_key("bytedance/seedance-2.0") == "seedance-2.0" + assert _normalize_family_key("minimax/h3") == "minimax-h3" + assert _normalize_family_key("xai/grok-imagine-video/v1.5") == "grok-imagine-1.5" + assert _normalize_family_key("google/gemini-omni-flash") == "gemini-omni-flash" + assert _normalize_family_key("blackforestlabs/flux-3") == "flux-3" + + def test_capabilities_span_longest_family_duration(self): + """Provider caps must not understate Seedance 2.5's 30s ceiling.""" + from plugins.video_gen.fal import FALVideoGenProvider + + caps = FALVideoGenProvider().capabilities() + assert caps["max_duration"] >= 30 + assert caps["min_duration"] <= 1 + + class TestPayloadBuilder: def test_drops_unsupported_keys(self): """Veo enum-clamps duration, supports aspect+resolution+audio+neg.""" @@ -275,6 +316,100 @@ class TestPayloadBuilder: ) assert p["duration"] == "15" + @pytest.mark.parametrize( + "family_id", + [ + "seedance-2.0", + "seedance-2.0-mini", + "seedance-2.5", + "minimax-h3", + "flux-3", + "grok-imagine-1.5", + "gemini-omni-flash", + ], + ) + def test_seed_dropped_for_families_without_seed_support(self, family_id): + """These FAL endpoints declare no `seed`; the gateway forwards whatever + we send, so an unknown key would reach the vendor.""" + from plugins.video_gen.fal import FAL_FAMILIES, _build_payload + + p = _build_payload( + FAL_FAMILIES[family_id], + prompt="x", + image_url="https://i.png", + duration=None, + aspect_ratio="16:9", + resolution="720p", + negative_prompt=None, + audio=None, + seed=42, + ) + assert "seed" not in p + + def test_minimax_h3_uses_uppercase_resolution_enum(self): + """FAL spells MiniMax H3 resolutions "768P"/"2K"/"4K"; tool-style + values like "720p" are aliased via resolution_aliases.""" + from plugins.video_gen.fal import FAL_FAMILIES, _build_payload + + meta = FAL_FAMILIES["minimax-h3"] + accepted = _build_payload( + meta, prompt="x", image_url=None, duration=7, aspect_ratio="16:9", + resolution="2K", negative_prompt=None, audio=None, seed=None, + ) + assert accepted["resolution"] == "2K" + assert accepted["duration"] == 7 + + aliased = _build_payload( + meta, prompt="x", image_url=None, duration=7, aspect_ratio="16:9", + resolution="720p", negative_prompt=None, audio=None, seed=None, + ) + assert aliased["resolution"] == "768P" + + def test_audio_only_sent_for_families_that_declare_it(self): + """minimax-h3 and the i2v-only families have no generate_audio field.""" + from plugins.video_gen.fal import FAL_FAMILIES, _build_payload + + for family_id in ("minimax-h3", "grok-imagine-1.5", "gemini-omni-flash"): + p = _build_payload( + FAL_FAMILIES[family_id], + prompt="x", image_url="https://i.png", duration=None, + aspect_ratio="16:9", resolution="720p", negative_prompt="ugly", + audio=True, seed=None, + ) + assert "generate_audio" not in p, family_id + assert "negative_prompt" not in p, family_id + + @pytest.mark.parametrize( + "family_id,expected", + [ + ("minimax-h3", 7), # FAL types duration as an integer + ("flux-3", 7), # mixed ["auto", 5, 6, ...] literal enum + ("grok-imagine-1.5", 7), + ("gemini-omni-flash", 7), + ("seedance-2.5", "7"), # FAL enum is strings: "auto","4",... + ("seedance-2.0-mini", "7"), + ("pixverse-v6", "7"), # unchanged legacy string form + ("veo3.1", "6s"), # unchanged suffix form (7 snaps to 6) + ], + ) + def test_duration_is_emitted_in_the_form_fal_declares(self, family_id, expected): + from plugins.video_gen.fal import FAL_FAMILIES, _build_payload + + p = _build_payload( + FAL_FAMILIES[family_id], + prompt="x", image_url=None, duration=7, aspect_ratio="16:9", + resolution="720p", negative_prompt=None, audio=None, seed=None, + ) + assert p["duration"] == expected + assert type(p["duration"]) is type(expected) + + def test_i2v_only_families_declare_no_text_endpoint(self): + """Catalog invariant: Gemini Omni Flash animates an existing image only.""" + from plugins.video_gen.fal import FAL_FAMILIES + + meta = FAL_FAMILIES["gemini-omni-flash"] + assert meta.get("text_endpoint") is None + assert meta["image_endpoint"] def test_ltx_omits_duration_aspect_resolution(self): """LTX 2.3 doesn't declare duration/aspect/resolution enums — diff --git a/tests/tools/test_image_generation.py b/tests/tools/test_image_generation.py index ded1c3e2ba..89e9b28738 100644 --- a/tests/tools/test_image_generation.py +++ b/tests/tools/test_image_generation.py @@ -76,6 +76,93 @@ class TestFalCatalog: f"{mid} should default to upscale=True (sub-2MP native)" + def test_edit_capable_entries_declare_a_full_edit_contract(self, image_tool): + """An `edit_endpoint` is useless without the whitelist and the + reference-image cap that `_build_fal_edit_payload` reads.""" + for mid, meta in image_tool.FAL_MODELS.items(): + if "edit_endpoint" not in meta: + continue + assert meta.get("edit_supports"), f"{mid} has edit_endpoint but no edit_supports" + assert "image_urls" in meta["edit_supports"], \ + f"{mid} edit_supports must allow image_urls" + cap = meta.get("max_reference_images") + assert isinstance(cap, int) and cap > 0, \ + f"{mid} needs a positive max_reference_images" + + +class TestAugust2026Catalog: + """The Aug 2026 FAL catalog expansion, surfaced in the model picker.""" + + NEW_MODELS = ( + "bytedance/seedream/v5/pro/text-to-image", + "bytedance/seedream/v5/lite/text-to-image", + "ideogram/v4/instant", + "ideogram/v4/fast", + "alibaba/qwen-image-3/text-to-image", + "microsoft/mai-image-2.5-pro", + "google/nano-banana-2-lite", + "fal-ai/recraft/v4.1/text-to-image", + "fal-ai/nano-banana-2", + ) + + def test_new_models_are_in_the_catalog(self, image_tool): + missing = [m for m in self.NEW_MODELS if m not in image_tool.FAL_MODELS] + assert not missing, f"missing from FAL_MODELS: {missing}" + + def test_paired_edit_endpoints_are_wired(self, image_tool): + expected = { + "bytedance/seedream/v5/pro/text-to-image": "bytedance/seedream/v5/pro/edit", + "alibaba/qwen-image-3/text-to-image": "alibaba/qwen-image-3/edit", + "google/nano-banana-2-lite": "google/nano-banana-2-lite/edit", + "fal-ai/nano-banana-2": "fal-ai/nano-banana-2/edit", + } + for model_id, edit_endpoint in expected.items(): + assert image_tool.FAL_MODELS[model_id]["edit_endpoint"] == edit_endpoint + + def test_text_only_models_declare_no_edit_endpoint(self, image_tool): + """These have no `/edit` app on FAL; claiming one would 404 mid-request.""" + for model_id in ( + "bytedance/seedream/v5/lite/text-to-image", + "ideogram/v4/instant", + "ideogram/v4/fast", + "microsoft/mai-image-2.5-pro", + "fal-ai/recraft/v4.1/text-to-image", + ): + assert "edit_endpoint" not in image_tool.FAL_MODELS[model_id] + + def test_recraft_v41_omits_keys_its_schema_lacks(self, image_tool): + """Recraft V4.1 exposes no num_images/output_format/seed — the + `supports` whitelist has to drop them rather than pass them upstream.""" + p = image_tool._build_fal_payload( + "fal-ai/recraft/v4.1/text-to-image", "hello", "landscape" + ) + assert p["image_size"] == "landscape_16_9" + for absent in ("num_images", "output_format", "seed"): + assert absent not in p + + def test_nano_banana_2_pins_the_1k_billing_tier(self, image_tool): + p = image_tool._build_fal_payload("fal-ai/nano-banana-2", "hello", "landscape") + assert p["resolution"] == "1K" + assert p["aspect_ratio"] == "16:9" + assert "image_size" not in p + + def test_nano_banana_2_lite_has_no_resolution_knob(self, image_tool): + """The lite tier renders at a fixed 1K and declares no `resolution`.""" + meta = image_tool.FAL_MODELS["google/nano-banana-2-lite"] + assert "resolution" not in meta["supports"] + assert "resolution" not in meta["defaults"] + p = image_tool._build_fal_payload("google/nano-banana-2-lite", "hello", "square") + assert "resolution" not in p + assert p["aspect_ratio"] == "1:1" + + def test_seedream_lite_uses_documented_size_presets(self, image_tool): + """Lite accepts FAL's preset enum; custom ImageSize dicts are unnecessary.""" + p = image_tool._build_fal_payload( + "bytedance/seedream/v5/lite/text-to-image", "hello", "landscape" + ) + assert p["image_size"] == "landscape_16_9" + + # --------------------------------------------------------------------------- # Payload building — three size families # --------------------------------------------------------------------------- diff --git a/tests/tools/test_video_generation_dynamic_schema.py b/tests/tools/test_video_generation_dynamic_schema.py index 33e210ceff..54a8fa8599 100644 --- a/tests/tools/test_video_generation_dynamic_schema.py +++ b/tests/tools/test_video_generation_dynamic_schema.py @@ -103,3 +103,59 @@ class TestDynamicSchemaBuilder: assert entry.dynamic_schema_overrides is not None out = entry.dynamic_schema_overrides() assert "description" in out + + def test_both_modalities_model_claims_both(self, cfg_home): + from tools.video_generation_tool import _build_dynamic_video_schema + + video_gen_registry.register_provider(_BothModalitiesProvider()) + _write_cfg(cfg_home, {"video_gen": {"provider": "both", "model": "family-a"}}) + + desc = _build_dynamic_video_schema()["description"] + assert "supports both text-to-video" in desc + assert "duration range: 1-15s" in desc + + def test_i2v_only_model_does_not_claim_text_to_video(self, cfg_home): + """A dual-modality backend with an i2v-only active model must not + contradict the model caveat with a 'supports both' line.""" + from tools.video_generation_tool import _build_dynamic_video_schema + + class _DualBackendI2VModel(VideoGenProvider): + @property + def name(self) -> str: + return "dual-i2v" + + def is_available(self) -> bool: + return True + + def list_models(self): + return [{ + "id": "gemini-like", + "modalities": ["image"], + "min_duration": 3, + "max_duration": 10, + }] + + def default_model(self): + return "gemini-like" + + def capabilities(self): + return { + "modalities": ["text", "image"], + "min_duration": 1, + "max_duration": 30, + } + + def generate(self, prompt, **kwargs): + return {"success": True} + + video_gen_registry.register_provider(_DualBackendI2VModel()) + _write_cfg( + cfg_home, + {"video_gen": {"provider": "dual-i2v", "model": "gemini-like"}}, + ) + + desc = _build_dynamic_video_schema()["description"] + assert "image-to-video only" in desc + assert "supports both text-to-video" not in desc + # Prefer the active model's duration window over the backend union. + assert "duration range: 3-10s" in desc diff --git a/tests/tools/test_video_generation_tool_surface_matrix.py b/tests/tools/test_video_generation_tool_surface_matrix.py index 88d2e5e781..96f7e74f12 100644 --- a/tests/tools/test_video_generation_tool_surface_matrix.py +++ b/tests/tools/test_video_generation_tool_surface_matrix.py @@ -137,7 +137,19 @@ def _all_fal_families(): return list(FAL_FAMILIES.keys()) -@pytest.mark.parametrize("family_id", _all_fal_families()) +def _t2v_fal_families(): + """Families that advertise a text-to-video endpoint.""" + from plugins.video_gen.fal import FAL_FAMILIES + return [fid for fid, meta in FAL_FAMILIES.items() if meta.get("text_endpoint")] + + +def _i2v_only_fal_families(): + """Families that only animate an existing image (no text_endpoint).""" + from plugins.video_gen.fal import FAL_FAMILIES + return [fid for fid, meta in FAL_FAMILIES.items() if not meta.get("text_endpoint")] + + +@pytest.mark.parametrize("family_id", _t2v_fal_families()) def test_fal_text_only_routes_to_text_endpoint(matrix_env, family_id): home, fal_calls, _ = matrix_env from plugins.video_gen.fal import FAL_FAMILIES @@ -171,6 +183,54 @@ def test_fal_text_only_routes_to_text_endpoint(matrix_env, family_id): assert not image_keys, f"{family_id} text-only leaked image keys: {image_keys}" +@pytest.mark.parametrize("family_id", _i2v_only_fal_families()) +def test_fal_i2v_only_family_refuses_text_only(matrix_env, family_id): + """An i2v-only family must refuse a text-only call rather than guess an endpoint.""" + home, fal_calls, _ = matrix_env + + result = _invoke_tool( + home, + {"video_gen": {"provider": "fal", "model": family_id}}, + {"prompt": "a dog running"}, + ) + + assert result["success"] is False, f"{family_id} has no text-to-video route" + assert result.get("error_type") == "modality_unsupported" + assert not fal_calls, f"{family_id} must not reach FAL for an unsupported modality" + + +def _i2v_fal_families(): + """Every family that can animate an existing image.""" + from plugins.video_gen.fal import FAL_FAMILIES + return [fid for fid, meta in FAL_FAMILIES.items() if meta.get("image_endpoint")] + + +@pytest.mark.parametrize("family_id", _i2v_fal_families()) +def test_fal_image_to_video_routes_to_image_endpoint(matrix_env, family_id): + home, fal_calls, _ = matrix_env + from plugins.video_gen.fal import FAL_FAMILIES + + result = _invoke_tool( + home, + {"video_gen": {"provider": "fal", "model": family_id}}, + {"prompt": "animate this", "image_url": "https://example.com/i.png"}, + ) + + meta = FAL_FAMILIES[family_id] + assert result["success"] is True, f"{family_id}: {result.get('error')}" + assert result["modality"] == "image" + assert len(fal_calls) == 1 + assert fal_calls[0]["endpoint"] == meta["image_endpoint"] + + # The image must land under the family's declared key and no other + # (kling v3 4k wants start_image_url; sending both would be a 422). + payload = fal_calls[0]["arguments"] or {} + image_key = meta.get("image_param_key") or "image_url" + assert payload.get(image_key) == "https://example.com/i.png" + other_keys = [k for k in payload if "image" in k and "url" in k and k != image_key] + assert not other_keys, f"{family_id} sent extra image keys: {other_keys}" + + # ───────────────────────────────────────────────────────────────────────── # xAI: text-only / text+image both go to /videos/generations # (xAI uses one endpoint with an optional 'image' field, not separate URLs) diff --git a/tools/image_generation_tool.py b/tools/image_generation_tool.py index be485db9f3..c7c5ccbf93 100644 --- a/tools/image_generation_tool.py +++ b/tools/image_generation_tool.py @@ -451,6 +451,11 @@ FAL_MODELS: Dict[str, Dict[str, Any]] = { }, "upscale": False, }, + # ─── Aug 2026 catalog expansion ──────────────────────────────────────── + # Endpoint ids, `supports` whitelists and enum defaults below are taken + # from each model's FAL OpenAPI schema, so a key we send is a key the + # vendor declares. Paired `/edit` apps hang off their text-to-image entry + # rather than appearing as separate picker rows. "bytedance/seedream/v5/pro/text-to-image": { "display": "Seedream 5.0 Pro", "speed": "~10s", @@ -488,11 +493,13 @@ FAL_MODELS: Dict[str, Dict[str, Any]] = { "strengths": "Fast/cheap Seedream tier, high-res output", "price": "$0.035/image", "size_style": "image_size_preset", - # Lite wants total pixels between 2560x1440 and 4096x4096. + # Lite wants total pixels between 2560x1440 and 4096x4096. Use the + # documented presets (FAL auto-scales if a preset is under the floor) + # instead of hand-rolled ImageSize dicts that drift from the schema. "sizes": { - "landscape": {"width": 3840, "height": 2160}, - "square": {"width": 2048, "height": 2048}, - "portrait": {"width": 2160, "height": 3840}, + "landscape": "landscape_16_9", + "square": "square_hd", + "portrait": "portrait_16_9", }, "defaults": { "num_images": 1, diff --git a/tools/video_generation_tool.py b/tools/video_generation_tool.py index 36e1b2dd85..1b4864151b 100644 --- a/tools/video_generation_tool.py +++ b/tools/video_generation_tool.py @@ -537,11 +537,15 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: for c in _format_model_caveats(model_meta, caps): parts.append(f"- {c}") - # Backend modality summary — only useful when the backend supports - # both text and image. Single-modality backends are already covered by - # the model caveat above. - modalities = set(caps.get("modalities") or []) - if "text" in modalities and "image" in modalities and not model_meta.get("modality"): + # Prefer the active model's modalities over the backend union. An + # i2v-only family on a dual-modality backend (e.g. gemini-omni-flash + # on FAL) must not also claim text-to-video support. + model_modalities = set(model_meta.get("modalities") or []) + modality = model_meta.get("modality") + if modality: + model_modalities.add(modality) + effective_modalities = model_modalities or set(caps.get("modalities") or []) + if "text" in effective_modalities and "image" in effective_modalities: parts.append( "- supports both text-to-video (omit image_url) and " "image-to-video (pass image_url) — routes automatically" @@ -551,9 +555,11 @@ def _build_dynamic_video_schema() -> Dict[str, Any]: parts.append(f"- aspect_ratio choices: {', '.join(caps['aspect_ratios'])}") if caps.get("resolutions"): parts.append(f"- resolution choices: {', '.join(caps['resolutions'])}") - if caps.get("min_duration") and caps.get("max_duration"): + min_duration = model_meta.get("min_duration", caps.get("min_duration")) + max_duration = model_meta.get("max_duration", caps.get("max_duration")) + if min_duration and max_duration: parts.append( - f"- duration range: {caps['min_duration']}-{caps['max_duration']}s" + f"- duration range: {min_duration}-{max_duration}s" ) if caps.get("supports_audio"): parts.append("- audio: pass `audio=true` to enable native audio (pricing tier)")