diff --git a/plugins/video_gen/deepinfra/__init__.py b/plugins/video_gen/deepinfra/__init__.py index 4dcb391e94..b37d256714 100644 --- a/plugins/video_gen/deepinfra/__init__.py +++ b/plugins/video_gen/deepinfra/__init__.py @@ -1,15 +1,10 @@ """DeepInfra video generation backend. -DeepInfra serves video over the OpenAI-compatible ``/v1/openai/videos`` -endpoint (async job: ``create`` → poll → ``download_content``), so all the -SDK plumbing lives in -:class:`agent.video_gen_provider.OpenAICompatibleVideoGenProvider`. This -plugin only declares DeepInfra's identity, credentials, and live model -discovery — no hardcoded model ids, so retired models drop out of hermes the -next time the catalog is fetched without a patch. - -Mirrors ``plugins/image_gen/deepinfra`` (which does the same for -``/v1/openai/images/generations``). +DeepInfra serves video over the OpenAI-compatible ``/v1/openai/videos`` endpoint (async job: +``create`` → poll → ``download_content``), so all SDK plumbing lives in +:class:`agent.video_gen_provider.OpenAICompatibleVideoGenProvider`. This plugin only declares +identity, credentials, and live model discovery — no hardcoded model ids, so retired models drop +out without a patch. Mirrors ``plugins/image_gen/deepinfra``. """ from __future__ import annotations @@ -34,41 +29,27 @@ class DeepInfraVideoGenProvider(OpenAICompatibleVideoGenProvider): return "DeepInfra" def list_models(self) -> List[Dict[str, Any]]: - """Return ``video-gen``-tagged DeepInfra models from the live catalog. - - Empty list when the catalog is unreachable — the picker then shows no - options rather than routing to a possibly-retired model. - """ + """``video-gen``-tagged models from the live catalog; empty when it is + unreachable so the picker shows nothing rather than a retired model.""" try: from hermes_cli.models import _fetch_deepinfra_models_by_tag except Exception as exc: # noqa: BLE001 — never break the picker logger.debug("Cannot import _fetch_deepinfra_models_by_tag: %s", exc) return [] - items = _fetch_deepinfra_models_by_tag("video-gen") or [] - out: List[Dict[str, Any]] = [] - for item in items: - mid = item.get("id") - if not mid: - continue - meta = item.get("metadata", {}) if isinstance(item, dict) else {} - out.append({ - "id": mid, - "display": mid.split("/")[-1], - "strengths": (meta.get("description") or "")[:80], - }) - return out + return [ + {"id": item["id"], "display": item["id"].split("/")[-1], + "strengths": ((item.get("metadata", {}) or {}).get("description") or "")[:80]} + for item in (_fetch_deepinfra_models_by_tag("video-gen") or []) if item.get("id") + ] def capabilities(self) -> Dict[str, Any]: return { "modalities": ["text", "image"], "aspect_ratios": ["16:9", "9:16", "1:1"], "resolutions": ["480p", "720p", "1080p"], - "max_duration": 10, - "min_duration": 1, - "supports_audio": False, - "supports_negative_prompt": True, - "supports_seed": True, - "supports_upscale": False, + "max_duration": 10, "min_duration": 1, + "supports_audio": False, "supports_negative_prompt": True, + "supports_seed": True, "supports_upscale": False, "max_reference_images": 0, } @@ -78,11 +59,7 @@ class DeepInfraVideoGenProvider(OpenAICompatibleVideoGenProvider): "badge": "paid", "tag": "Wan, p-video, … — live catalog from api.deepinfra.com; text-to-video & image-to-video", "env_vars": [ - { - "key": "DEEPINFRA_API_KEY", - "prompt": "DeepInfra API key", - "url": "https://deepinfra.com/dash/api_keys", - }, + {"key": "DEEPINFRA_API_KEY", "prompt": "DeepInfra API key", "url": "https://deepinfra.com/dash/api_keys"}, ], } diff --git a/plugins/video_gen/fal/__init__.py b/plugins/video_gen/fal/__init__.py index e63e2e16e3..6172da9f4c 100644 --- a/plugins/video_gen/fal/__init__.py +++ b/plugins/video_gen/fal/__init__.py @@ -1,621 +1,305 @@ """FAL.ai video generation backend. -User-facing surface: pick a **model family** (e.g. "Pixverse v6", -"Veo 3.1", "Seedance 2.0", "Kling v3 4K", "LTX 2.3", "Happy Horse"). -The plugin auto-routes to the family's text-to-video endpoint when -called without ``image_url``, and to its image-to-video endpoint when -``image_url`` is provided. The agent never sees the routing — it just -calls ``video_generate(prompt=..., image_url=...)``. - -Model families (most expose both t2v + i2v; gemini-omni-flash is image-to-video only): - - Cheap tier: - ltx-2.3 fal-ai/ltx-2.3-22b/text-to-video / fal-ai/ltx-2.3-22b/image-to-video - pixverse-v6 fal-ai/pixverse/v6/text-to-video / fal-ai/pixverse/v6/image-to-video - seedance-2.0-mini bytedance/seedance-2.0/mini/text-to-video / bytedance/seedance-2.0/mini/image-to-video - - Premium tier: - veo3.1 fal-ai/veo3.1 / fal-ai/veo3.1/image-to-video - seedance-2.0 bytedance/seedance-2.0/text-to-video / bytedance/seedance-2.0/image-to-video - seedance-2.5 bytedance/seedance-2.5/text-to-video / bytedance/seedance-2.5/image-to-video - minimax-h3 minimax/h3/text-to-video / minimax/h3/image-to-video - minimax-h3-max minimax/h3-max/text-to-video / minimax/h3-max/image-to-video - flux-3 blackforestlabs/flux-3/text-to-video / blackforestlabs/flux-3/image-to-video - grok-imagine-1.5 xai/grok-imagine-video/v1.5/text-to-video / xai/grok-imagine-video/v1.5/image-to-video - kling-v3-4k fal-ai/kling-video/v3/4k/text-to-video / fal-ai/kling-video/v3/4k/image-to-video - happy-horse alibaba/happy-horse/text-to-video / alibaba/happy-horse/image-to-video - - Image-to-video only (no text_endpoint): - gemini-omni-flash google/gemini-omni-flash/image-to-video - -Selection precedence for the active family: - 1. ``model=`` arg from the tool call - 2. ``FAL_VIDEO_MODEL`` env var - 3. ``video_gen.fal.model`` in ``config.yaml`` - 4. ``video_gen.model`` in ``config.yaml`` (when it's one of our family IDs - or a full endpoint path that contains a family ID) - 5. ``DEFAULT_MODEL`` - -Authentication via ``FAL_KEY`` or the managed Nous gateway. Output is an -HTTPS URL from FAL's CDN; the gateway downloads and delivers it. +The user picks a **model family** (e.g. "Pixverse v6", "Veo 3.1"); the plugin routes to the +family's text-to-video endpoint when called without ``image_url`` and to its image-to-video +endpoint otherwise (gemini-omni-flash is i2v only). Active-family precedence: tool ``model=`` +arg → ``FAL_VIDEO_MODEL`` env → ``video_gen.fal.model`` → ``video_gen.model`` (a family id or +an endpoint path containing one) → ``DEFAULT_MODEL``. Auth via ``FAL_KEY`` or the managed Nous +gateway. Output is an HTTPS URL from FAL's CDN; the gateway downloads it. """ from __future__ import annotations import logging -import os import threading import uuid from typing import Any, Dict, List, Optional, Tuple -from agent.video_gen_provider import ( - VideoGenProvider, - error_response, - success_response, -) +from agent.video_gen_provider import VideoGenProvider, error_response, success_response logger = logging.getLogger(__name__) -# --------------------------------------------------------------------------- -# Family catalog -# --------------------------------------------------------------------------- -# -# Each family declares both endpoints (when available) plus a per-family -# capability sheet derived from FAL's OpenAPI schemas. Capability flags -# drive which keys get added to the request payload — keys a family doesn't -# advertise are dropped before send. -# -# Capabilities: -# aspect_ratios : tuple of supported ratios (None = endpoint decides) -# resolutions : tuple of supported resolutions (None = endpoint decides) -# durations : tuple of supported durations OR (min, max) range -# (heuristic: 2-element with gap > 1 is a range) -# audio : True if generate_audio is supported -# negative : True if negative_prompt is supported -# seed : False when the endpoint declares no `seed` field -# (absent = True, so existing families keep sending it) -# duration_int : True when FAL types duration as an integer rather than -# the usual queue-API string +# Family catalog. Capability flags gate which keys reach the payload — keys a family +# doesn't advertise are never sent (the managed gateway forwards everything verbatim). +# ``_family`` defaults every enum to None and every flag to False. Per-family keys: +# aspect_ratios / resolutions : supported enums (None = endpoint decides) +# durations : enum tuple OR (min, max) range (2 ints with gap > 1) +# audio / audio_native : generate_audio toggle / audio always on (description line only) +# negative / seed : negative_prompt / seed accepted +# duration_int / duration_suffix : send duration as JSON int (default: queue-API string) / "4s" suffix +# image_param_key / image_drop_keys : i2v image key when not `image_url` / keys the i2v endpoint rejects +# resolution_aliases / static_payload : tool-style resolution → endpoint enum / constants always required +def _family(display: str, speed: str, tier: str, strengths: str, text: Optional[str], image: str, **caps: Any) -> Dict[str, Any]: + return { + "display": display, "speed": speed, "price": tier, "tier": tier, "strengths": strengths, + "text_endpoint": text, "image_endpoint": image, + "aspect_ratios": None, "resolutions": None, "durations": None, "audio": False, "negative": False, "seed": False, + **caps, + } + + +_SIX_ASPECTS = ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16") +# MiniMax H3 uses capitalized/2K-style resolution enums; aliases map the tool's usual values. +_H3_ALIASES = {"480p": "768P", "540p": "768P", "720p": "768P", "768p": "768P", "1080p": "2K", "2k": "2K", "4k": "4K", "2160p": "4K"} +_H3_MAX_ALIASES = {"480p": "480P", "540p": "480P", "720p": "768P", "768p": "768P", "1080p": "768P", "2k": "768P", "4k": "768P", "2160p": "768P"} FAL_FAMILIES: Dict[str, Dict[str, Any]] = { # ─── Cheap / fast tier ───────────────────────────────────────────── - "ltx-2.3": { - "display": "LTX 2.3 (22B)", - "speed": "~30-60s", - "price": "cheap", - "strengths": "22B model with native audio generation. Affordable.", - "tier": "cheap", - "text_endpoint": "fal-ai/ltx-2.3-22b/text-to-video", - "image_endpoint": "fal-ai/ltx-2.3-22b/image-to-video", - # LTX docs don't expose duration/aspect/resolution enums — leave - # blank so we don't send unrecognized payload keys. - "aspect_ratios": None, - "resolutions": None, - "durations": None, - "audio": True, - "negative": True, - "seed": True, - }, - "pixverse-v6": { - "display": "Pixverse v6", - "speed": "~30-90s", - "price": "cheap", - "strengths": "Affordable. Negative prompts. 1-15s durations.", - "tier": "cheap", - "text_endpoint": "fal-ai/pixverse/v6/text-to-video", - "image_endpoint": "fal-ai/pixverse/v6/image-to-video", - "aspect_ratios": None, - "resolutions": ("360p", "540p", "720p", "1080p"), - "durations": (1, 15), - "audio": True, - "negative": True, - "seed": True, - }, - "seedance-2.0-mini": { - "display": "Seedance 2.0 Mini", - "speed": "~30-90s", - "price": "cheap", - "strengths": "ByteDance. Faster/cheaper Seedance tier, audio + lip-sync, 4-15s.", - "tier": "cheap", - "text_endpoint": "bytedance/seedance-2.0/mini/text-to-video", - "image_endpoint": "bytedance/seedance-2.0/mini/image-to-video", - "aspect_ratios": ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16"), - "resolutions": ("480p", "720p"), - "durations": (4, 15), - "audio": True, - "negative": False, - "seed": False, - }, + "ltx-2.3": _family( # LTX docs expose no duration/aspect/resolution enums. + "LTX 2.3 (22B)", "~30-60s", "cheap", "22B model with native audio generation. Affordable.", + "fal-ai/ltx-2.3-22b/text-to-video", "fal-ai/ltx-2.3-22b/image-to-video", audio=True, negative=True, seed=True, + ), + "pixverse-v6": _family( + "Pixverse v6", "~30-90s", "cheap", "Affordable. Negative prompts. 1-15s durations.", + "fal-ai/pixverse/v6/text-to-video", "fal-ai/pixverse/v6/image-to-video", + resolutions=("360p", "540p", "720p", "1080p"), durations=(1, 15), audio=True, negative=True, seed=True, + ), + "seedance-2.0-mini": _family( + "Seedance 2.0 Mini", "~30-90s", "cheap", "ByteDance. Faster/cheaper Seedance tier, audio + lip-sync, 4-15s.", + "bytedance/seedance-2.0/mini/text-to-video", "bytedance/seedance-2.0/mini/image-to-video", + aspect_ratios=_SIX_ASPECTS, resolutions=("480p", "720p"), durations=(4, 15), audio=True, + ), # ─── Expensive / premium tier ────────────────────────────────────── - "veo3.1": { - "display": "Veo 3.1", - "speed": "~60-120s", - "price": "premium", - "strengths": "Google DeepMind. Cinematic, native audio, strong prompt adherence.", - "tier": "premium", - "text_endpoint": "fal-ai/veo3.1", - "image_endpoint": "fal-ai/veo3.1/image-to-video", - "aspect_ratios": ("16:9", "9:16"), - "resolutions": ("720p", "1080p", "4k"), - "durations": (4, 6, 8), - "duration_suffix": "s", # FAL veo3.1 wants "4s" not "4" - "audio": True, - "negative": True, - "seed": True, - }, - "seedance-2.0": { - "display": "Seedance 2.0", - "speed": "~60-120s", - "price": "premium", - "strengths": "ByteDance. Cinematic, synchronized audio + lip-sync, 4-15s.", - "tier": "premium", - "text_endpoint": "bytedance/seedance-2.0/text-to-video", - "image_endpoint": "bytedance/seedance-2.0/image-to-video", - # Seedance accepts "auto" too — we omit it from the enum so the - # agent can't pass it; the endpoint defaults handle the rest. - "aspect_ratios": ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16"), - "resolutions": ("480p", "720p", "1080p"), - "durations": (4, 15), - "audio": True, - "negative": False, - # FAL input schema has no `seed` (only returned on output). - "seed": False, - }, - "seedance-2.5": { - "display": "Seedance 2.5", - "speed": "~60-180s", - "price": "premium", - "strengths": "ByteDance flagship. Native 30s single-pass, audio in the same latent space, lip-sync.", - "tier": "premium", - "text_endpoint": "bytedance/seedance-2.5/text-to-video", - "image_endpoint": "bytedance/seedance-2.5/image-to-video", - # i2v accepts only "auto" for aspect_ratio (it follows the input - # image), so aspect_ratio is dropped for image jobs via - # image_drop_keys. - "image_drop_keys": ("aspect_ratio",), - "aspect_ratios": ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16"), - "resolutions": ("480p", "720p"), - "durations": (4, 30), - "audio": True, - "negative": False, - "seed": False, - }, - "minimax-h3": { - "display": "MiniMax H3", - "speed": "~60-180s", - "price": "premium", - "strengths": "MiniMax frontier. Native 2K (up to 4K), 5-15s, seven aspect ratios.", - "tier": "premium", - "text_endpoint": "minimax/h3/text-to-video", - "image_endpoint": "minimax/h3/image-to-video", - # H3 takes duration as a JSON integer, not the stringified form - # most FAL endpoints use. - "duration_int": True, - # i2v derives the aspect ratio from the input image and rejects - # the key entirely. - "image_drop_keys": ("aspect_ratio",), - "aspect_ratios": ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16"), - # H3 uses capitalized/2K-style resolution enums — mapped from the - # tool's usual 720p/1080p-style values via resolution_aliases. - "resolutions": ("768P", "2K", "4K"), - "resolution_aliases": { - "480p": "768P", "540p": "768P", "720p": "768P", "768p": "768P", - "1080p": "2K", "2k": "2K", "4k": "4K", "2160p": "4K", - }, - "durations": (5, 15), - "audio": False, # no generate_audio TOGGLE — audio is always on - "audio_native": True, # native audio in every generation (fal docs) # audio is native/always-on; no generate_audio key - "negative": False, - "seed": False, - }, - "minimax-h3-max": { - "display": "MiniMax H3 Max (fal post-train)", - "speed": "~5-30s", - "price": "premium", - "strengths": "fal's post-trained MiniMax H3. Top-ranked quality/prompt adherence/aesthetics, 768p in seconds, 5-15s.", - "tier": "premium", - "text_endpoint": "minimax/h3-max/text-to-video", - "image_endpoint": "minimax/h3-max/image-to-video", - # Same wire quirks as base H3: integer duration, i2v derives the - # aspect ratio from the input image (t2v-only key on Max: the i2v - # schema doesn't declare aspect_ratio at all). - "duration_int": True, - "image_drop_keys": ("aspect_ratio",), - "aspect_ratios": ("21:9", "16:9", "4:3", "1:1", "3:4", "9:16"), - # Max tops out at 768P (no 2K/4K tiers like base H3); map the - # tool's usual values onto the two capitalized enums. - "resolutions": ("480P", "768P"), - "resolution_aliases": { - "480p": "480P", "540p": "480P", - "720p": "768P", "768p": "768P", "1080p": "768P", - "2k": "768P", "4k": "768P", "2160p": "768P", - }, - "durations": (5, 15), - # `prompt_expansion_mode` is in the schema's required array (with a - # "balanced" default) — always send it. - "static_payload": {"prompt_expansion_mode": "balanced"}, - "audio": False, # no generate_audio TOGGLE — audio is always on - "audio_native": True, # native audio in every generation (fal docs) # audio is native/always-on; no generate_audio key - "negative": False, - "seed": True, - # Unlike base H3, Max declares `seed` on both endpoints. - }, - "flux-3": { - "display": "FLUX 3 (via FAL)", - "speed": "~60-120s", - "price": "premium", - "strengths": "Black Forest Labs frontier video. Native audio, 5-20s, 8 aspect ratios.", - "tier": "premium", - "text_endpoint": "blackforestlabs/flux-3/text-to-video", - "image_endpoint": "blackforestlabs/flux-3/image-to-video", - # FLUX 3 duration enum is "auto" | 5..20 as JSON integers. - "duration_int": True, - "aspect_ratios": ("21:9", "2:1", "16:9", "4:3", "1:1", "3:4", "9:16"), - "resolutions": ("720p", "1080p"), - "durations": (5, 20), - "audio": True, - "negative": False, - "seed": False, - }, - "grok-imagine-1.5": { - "display": "Grok Imagine 1.5 (via FAL)", - "speed": "~30-90s", - "price": "premium", - "strengths": "xAI. Fast stylized video with audio, 1-15s, cheap per second.", - "tier": "premium", - "text_endpoint": "xai/grok-imagine-video/v1.5/text-to-video", - "image_endpoint": "xai/grok-imagine-video/v1.5/image-to-video", - "duration_int": True, - # i2v derives aspect from the input image; the key is t2v-only. - "image_drop_keys": ("aspect_ratio",), - "aspect_ratios": ("16:9", "4:3", "3:2", "1:1", "2:3", "3:4", "9:16"), - "resolutions": ("480p", "720p", "1080p"), - "durations": (1, 15), - "audio": False, # no generate_audio TOGGLE — audio is always on - "audio_native": True, # native audio in every generation (fal docs) # audio is native; no generate_audio key - "negative": False, - "seed": False, - }, - "gemini-omni-flash": { - "display": "Gemini Omni Flash (via FAL)", - "speed": "~60-120s", - "price": "premium", - "strengths": "Google. Image-to-video with audio, physics-grounded motion, 3-10s.", - "tier": "premium", - # No text-to-video endpoint on FAL — image/reference only. - "text_endpoint": None, - "image_endpoint": "google/gemini-omni-flash/image-to-video", - "duration_int": True, - "aspect_ratios": ("16:9", "9:16"), - "resolutions": None, - "durations": (3, 10), - "audio": False, # no generate_audio TOGGLE — audio is always on - "audio_native": True, # native audio in every generation (fal docs) # audio is native; no generate_audio key - "negative": False, - "seed": False, - }, - "kling-v3-4k": { - "display": "Kling v3 4K", - "speed": "~120-300s", - "price": "premium", - "strengths": "4K output, native audio (Chinese/English), 3-15s.", - "tier": "premium", - "text_endpoint": "fal-ai/kling-video/v3/4k/text-to-video", - "image_endpoint": "fal-ai/kling-video/v3/4k/image-to-video", - # Kling 4K image-to-video uses `start_image_url` instead of - # `image_url`. Handled in _build_payload via image_param_key. - "image_param_key": "start_image_url", - "aspect_ratios": ("16:9", "9:16", "1:1"), - "resolutions": None, # 4K is implicit - "durations": (3, 15), - "audio": True, - "negative": True, - "seed": True, - }, - "happy-horse": { - "display": "Happy Horse 1.0", - "speed": "~60-120s", - "price": "premium", - "strengths": "Alibaba. New model, sparse public docs — conservative defaults.", - "tier": "premium", - "text_endpoint": "alibaba/happy-horse/text-to-video", - "image_endpoint": "alibaba/happy-horse/image-to-video", - # Docs don't expose duration/aspect/resolution — let the endpoint - # apply its own defaults. - "aspect_ratios": None, - "resolutions": None, - "durations": None, - "audio": False, # no generate_audio TOGGLE — audio is always on - "audio_native": True, # native audio in every generation (fal docs) - "negative": False, - "seed": True, - }, + "veo3.1": _family( + "Veo 3.1", "~60-120s", "premium", "Google DeepMind. Cinematic, native audio, strong prompt adherence.", + "fal-ai/veo3.1", "fal-ai/veo3.1/image-to-video", + aspect_ratios=("16:9", "9:16"), resolutions=("720p", "1080p", "4k"), + durations=(4, 6, 8), duration_suffix="s", # wants "4s" not "4" + audio=True, negative=True, seed=True, + ), + "seedance-2.0": _family( + "Seedance 2.0", "~60-120s", "premium", "ByteDance. Cinematic, synchronized audio + lip-sync, 4-15s.", + "bytedance/seedance-2.0/text-to-video", "bytedance/seedance-2.0/image-to-video", + # "auto" aspect is deliberately omitted so the agent can't pass it; input schema has no `seed`. + aspect_ratios=_SIX_ASPECTS, resolutions=("480p", "720p", "1080p"), durations=(4, 15), audio=True, + ), + "seedance-2.5": _family( + "Seedance 2.5", "~60-180s", "premium", + "ByteDance flagship. Native 30s single-pass, audio in the same latent space, lip-sync.", + "bytedance/seedance-2.5/text-to-video", "bytedance/seedance-2.5/image-to-video", + image_drop_keys=("aspect_ratio",), # i2v accepts only "auto" aspect (follows the input image) + aspect_ratios=_SIX_ASPECTS, resolutions=("480p", "720p"), durations=(4, 30), audio=True, + ), + "minimax-h3": _family( + "MiniMax H3", "~60-180s", "premium", "MiniMax frontier. Native 2K (up to 4K), 5-15s, seven aspect ratios.", + "minimax/h3/text-to-video", "minimax/h3/image-to-video", + duration_int=True, image_drop_keys=("aspect_ratio",), # i2v derives aspect from the input image + aspect_ratios=_SIX_ASPECTS, resolutions=("768P", "2K", "4K"), resolution_aliases=_H3_ALIASES, + durations=(5, 15), audio_native=True, + ), + "minimax-h3-max": _family( + "MiniMax H3 Max (fal post-train)", "~5-30s", "premium", + "fal's post-trained MiniMax H3. Top-ranked quality/prompt adherence/aesthetics, 768p in seconds, 5-15s.", + "minimax/h3-max/text-to-video", "minimax/h3-max/image-to-video", + duration_int=True, image_drop_keys=("aspect_ratio",), # i2v schema doesn't declare aspect_ratio + # Max tops out at 768P (no 2K/4K tiers like base H3). + aspect_ratios=_SIX_ASPECTS, resolutions=("480P", "768P"), resolution_aliases=_H3_MAX_ALIASES, + durations=(5, 15), + static_payload={"prompt_expansion_mode": "balanced"}, # in the schema's required array + audio_native=True, seed=True, # unlike base H3, Max declares `seed` on both endpoints + ), + "flux-3": _family( + "FLUX 3 (via FAL)", "~60-120s", "premium", "Black Forest Labs frontier video. Native audio, 5-20s, 8 aspect ratios.", + "blackforestlabs/flux-3/text-to-video", "blackforestlabs/flux-3/image-to-video", + duration_int=True, # enum is "auto" | 5..20 as JSON integers + aspect_ratios=("21:9", "2:1", "16:9", "4:3", "1:1", "3:4", "9:16"), resolutions=("720p", "1080p"), + durations=(5, 20), audio=True, + ), + "grok-imagine-1.5": _family( + "Grok Imagine 1.5 (via FAL)", "~30-90s", "premium", "xAI. Fast stylized video with audio, 1-15s, cheap per second.", + "xai/grok-imagine-video/v1.5/text-to-video", "xai/grok-imagine-video/v1.5/image-to-video", + duration_int=True, image_drop_keys=("aspect_ratio",), # t2v-only key; i2v follows the image + aspect_ratios=("16:9", "4:3", "3:2", "1:1", "2:3", "3:4", "9:16"), resolutions=("480p", "720p", "1080p"), + durations=(1, 15), audio_native=True, + ), + "gemini-omni-flash": _family( + "Gemini Omni Flash (via FAL)", "~60-120s", "premium", "Google. Image-to-video with audio, physics-grounded motion, 3-10s.", + None, "google/gemini-omni-flash/image-to-video", # image/reference only on FAL + duration_int=True, aspect_ratios=("16:9", "9:16"), durations=(3, 10), audio_native=True, + ), + "kling-v3-4k": _family( + "Kling v3 4K", "~120-300s", "premium", "4K output, native audio (Chinese/English), 3-15s.", + "fal-ai/kling-video/v3/4k/text-to-video", "fal-ai/kling-video/v3/4k/image-to-video", + image_param_key="start_image_url", aspect_ratios=("16:9", "9:16", "1:1"), durations=(3, 15), + audio=True, negative=True, seed=True, + ), + "happy-horse": _family( + "Happy Horse 1.0", "~60-120s", "premium", "Alibaba. New model, sparse public docs — conservative defaults.", + "alibaba/happy-horse/text-to-video", "alibaba/happy-horse/image-to-video", audio_native=True, seed=True, + ), } DEFAULT_MODEL = "pixverse-v6" # cheap, both modalities, sane defaults - -def _is_duration_range(durations: Any) -> bool: - """Heuristic: a 2-tuple of ints with a gap > 1 is treated as ``(min, max)``.""" - if not isinstance(durations, tuple) or len(durations) != 2: - return False - if not all(isinstance(d, int) for d in durations): - return False - return durations[1] - durations[0] > 1 - - -def _clamp_duration(family: Dict[str, Any], duration: Optional[int]) -> Optional[int]: - durations = family.get("durations") - if not durations: - return duration - if duration is None: - # Range families (e.g. pixverse-v6 (1,15)) should omit the field so - # the FAL endpoint applies its own default rather than receiving the - # minimum value. Enum families (e.g. veo3.1 (4,6,8)) keep sending - # their first entry as the default. - return None if _is_duration_range(durations) else durations[0] - if _is_duration_range(durations): - lo, hi = durations - return max(lo, min(hi, duration)) - # enum - if duration in durations: - return duration - return min(durations, key=lambda d: abs(d - duration)) - - -# --------------------------------------------------------------------------- -# Config / model resolution -# --------------------------------------------------------------------------- - - -def _load_video_gen_section() -> Dict[str, Any]: - try: - from hermes_cli.config import load_config - - cfg = load_config() - section = cfg.get("video_gen") if isinstance(cfg, dict) else None - return section if isinstance(section, dict) else {} - except Exception as exc: - logger.debug("Could not load video_gen config: %s", exc) - return {} - - _ENDPOINT_MODALITY_LEAVES = frozenset({"text-to-video", "image-to-video"}) -def _normalize_family_key(c: str) -> Optional[str]: - """Try to extract a known family ID from a model string. +def _is_duration_range(durations: Tuple[int, ...]) -> bool: + """Heuristic: a 2-tuple of ints with a gap > 1 is treated as ``(min, max)``.""" + return len(durations) == 2 and all(isinstance(d, int) for d in durations) and durations[1] - durations[0] > 1 - Handles bare IDs (``seedance-2.5``), full endpoint paths - (``bytedance/seedance-2.5/text-to-video``), truncated endpoint stems - (``minimax/h3``, ``bytedance/seedance-2.0/mini``), and provider-prefixed - names (``bytedance/seedance-2.5``). - """ + +def _duration_bounds(durations: Tuple[int, ...]) -> Tuple[int, int]: + """``(lo, hi)`` for a non-empty durations spec (range or enum).""" + return (durations[0], durations[1]) if _is_duration_range(durations) else (min(durations), max(durations)) + + +def _modalities(meta: Dict[str, Any]) -> List[str]: + return [m for m in ("text", "image") if meta[f"{m}_endpoint"]] + + +def _clamp_duration(durations: Tuple[int, ...], duration: Optional[int]) -> Optional[int]: + """Clamp into a range, or snap to the nearest enum entry. ``None`` stays None for + range families (the endpoint applies its own default) but becomes the first enum entry.""" + is_range = _is_duration_range(durations) + if duration is None: + return None if is_range else durations[0] + if is_range: + return max(durations[0], min(durations[1], duration)) + return min(durations, key=lambda d: abs(d - duration)) + + +def _normalize_family_key(c: str) -> Optional[str]: + """Extract a known family ID from a bare id, full endpoint path, + truncated endpoint stem (``minimax/h3``) or provider-prefixed name.""" c = c.strip() if not c: return None if c in FAL_FAMILIES: return c - - # Exact declared endpoint — unambiguous, and beats any segment scan - # that would otherwise see "seedance-2.0" inside ".../seedance-2.0/mini/...". - for fid, meta in FAL_FAMILIES.items(): - if c in (meta.get("text_endpoint"), meta.get("image_endpoint")): - return fid - - # Truncated stem of a declared endpoint: "minimax/h3" or - # "bytedance/seedance-2.0/mini". The next path segment after ``c`` must - # be a modality leaf so "bytedance/seedance-2.0" does not also match the - # Mini family's deeper ".../seedance-2.0/mini/text-to-video" path. - stem_hits: List[Tuple[int, str]] = [] - for fid, meta in FAL_FAMILIES.items(): - for endpoint in (meta.get("text_endpoint"), meta.get("image_endpoint")): - if not isinstance(endpoint, str): - continue - if not endpoint.startswith(c + "/"): - continue - first = endpoint[len(c) + 1:].split("/", 1)[0] - if first in _ENDPOINT_MODALITY_LEAVES: - stem_hits.append((len(c), fid)) - break - if stem_hits: - stem_hits.sort(key=lambda item: item[0], reverse=True) - return stem_hits[0][1] - - # Longest family-id path-segment match ("bytedance/seedance-2.5" → - # seedance-2.5; prefers seedance-2.0-mini over seedance-2.0 when both - # somehow appear). - parts = set(c.split("/")) - best_fid: Optional[str] = None - best_len = -1 - for fid in FAL_FAMILIES: - if fid in parts and len(fid) > best_len: - best_fid = fid - best_len = len(fid) - return best_fid + endpoints = [(fid, ep) for fid, meta in FAL_FAMILIES.items() + for ep in (meta["text_endpoint"], meta["image_endpoint"]) if isinstance(ep, str)] + # Exact declared endpoint beats any segment scan (which would see "seedance-2.0" inside ".../seedance-2.0/mini/..."). + exact = [fid for fid, ep in endpoints if c == ep] + # Truncated stem: the segment after ``c`` must be a modality leaf so "bytedance/seedance-2.0" skips Mini's deeper path. + stem = [fid for fid, ep in endpoints + if ep.startswith(c + "/") and ep[len(c) + 1:].split("/", 1)[0] in _ENDPOINT_MODALITY_LEAVES] + if exact or stem: + return (exact or stem)[0] + # Longest family-id path-segment match (prefers seedance-2.0-mini over seedance-2.0 when both appear). + hits = [fid for fid in FAL_FAMILIES if fid in c.split("/")] + return max(hits, key=len) if hits else None def _resolve_family(explicit: Optional[str]) -> Tuple[str, Dict[str, Any]]: """Decide which FAL family to use. Returns ``(family_id, meta)``.""" - candidates: List[Optional[str]] = [] - candidates.append(explicit) - candidates.append(os.environ.get("FAL_VIDEO_MODEL")) + import os - cfg = _load_video_gen_section() + try: + from hermes_cli.config import load_config + + cfg = load_config() + cfg = cfg.get("video_gen") if isinstance(cfg, dict) else None + cfg = cfg if isinstance(cfg, dict) else {} + except Exception as exc: + logger.debug("Could not load video_gen config: %s", exc) + cfg = {} fal_cfg = cfg.get("fal") if isinstance(cfg.get("fal"), dict) else {} - if isinstance(fal_cfg, dict): - candidates.append(fal_cfg.get("model")) - top = cfg.get("model") - if isinstance(top, str): - candidates.append(top) - - for c in candidates: - if isinstance(c, str) and c.strip(): - fid = _normalize_family_key(c) - if fid: - return fid, FAL_FAMILIES[fid] - + for c in (explicit, os.environ.get("FAL_VIDEO_MODEL"), fal_cfg.get("model"), cfg.get("model")): + fid = _normalize_family_key(c) if isinstance(c, str) else None + if fid: + return fid, FAL_FAMILIES[fid] return DEFAULT_MODEL, FAL_FAMILIES[DEFAULT_MODEL] -# --------------------------------------------------------------------------- -# Payload construction -# --------------------------------------------------------------------------- - - def _build_payload( - family: Dict[str, Any], - *, - prompt: str, - image_url: Optional[str], - duration: Optional[int], - aspect_ratio: str, - resolution: str, - negative_prompt: Optional[str], - audio: Optional[bool], - seed: Optional[int], + family: Dict[str, Any], *, prompt: str, image_url: Optional[str], duration: Optional[int], aspect_ratio: str, + resolution: str, negative_prompt: Optional[str], audio: Optional[bool], seed: Optional[int], ) -> Dict[str, Any]: """Build a family-specific payload, dropping keys the family doesn't declare.""" payload: Dict[str, Any] = {} - if prompt: payload["prompt"] = prompt if image_url: - # Some endpoints (e.g. Kling v3 4K image-to-video) expect - # `start_image_url` instead of `image_url`. The family entry can - # declare an override. - key = family.get("image_param_key") or "image_url" - payload[key] = image_url - # Several newer endpoints (seedance 2.x, minimax h3, flux-3, grok, gemini) - # declare no `seed` field, and the managed gateway forwards whatever we - # send — so gate it on the family rather than leaking an unknown key. + payload[family.get("image_param_key") or "image_url"] = image_url + # Newer endpoints declare no `seed` and the managed gateway forwards whatever we send — gate on the family. if seed is not None and family.get("seed", True): payload["seed"] = seed - - if family.get("aspect_ratios"): - if aspect_ratio in family["aspect_ratios"]: - payload["aspect_ratio"] = aspect_ratio - # otherwise let the endpoint auto-crop / use its default - - if family.get("resolutions"): - # Some families use non-standard resolution enums (e.g. MiniMax H3's - # "768P"/"2K"/"4K"); resolution_aliases maps the tool's usual - # 720p/1080p-style values onto them. - aliases = family.get("resolution_aliases") or {} - resolved = aliases.get((resolution or "").lower(), resolution) - if resolved in family["resolutions"]: - payload["resolution"] = resolved - # else: let the endpoint default - - clamped = _clamp_duration(family, duration) - if clamped is not None and family.get("durations"): - if family.get("duration_int"): - # A few endpoints (MiniMax H3) require duration as a JSON integer. - payload["duration"] = clamped - else: - # FAL exposes duration as a string in the queue API ("8" not 8). - # Some families (e.g. veo3.1) require a unit suffix ("4s" not "4"). - suffix = family.get("duration_suffix", "") - payload["duration"] = f"{clamped}{suffix}" - - if family.get("audio") and audio is not None: + # Unsupported aspect/resolution values are dropped so the endpoint defaults. + if family["aspect_ratios"] and aspect_ratio in family["aspect_ratios"]: + payload["aspect_ratio"] = aspect_ratio + resolved = (family.get("resolution_aliases") or {}).get((resolution or "").lower(), resolution) + if family["resolutions"] and resolved in family["resolutions"]: + payload["resolution"] = resolved + clamped = _clamp_duration(family["durations"], duration) if family["durations"] else None + if clamped is not None: + # FAL's queue API types duration as a string ("8" not 8) unless the family says int; + # some families (veo3.1) also need a unit suffix ("4s" not "4"). + payload["duration"] = clamped if family.get("duration_int") else f"{clamped}{family.get('duration_suffix', '')}" + if family["audio"] and audio is not None: payload["generate_audio"] = bool(audio) - - if family.get("negative") and negative_prompt: + if family["negative"] and negative_prompt: payload["negative_prompt"] = negative_prompt - - # Keys the family's image-to-video endpoint rejects outright (e.g. - # Seedance 2.5 / MiniMax H3 derive aspect_ratio from the input image). - if image_url: - for key in family.get("image_drop_keys", ()): # type: ignore[assignment] - payload.pop(key, None) - - # Constant keys the endpoint requires on every request (e.g. MiniMax - # H3 Max lists `prompt_expansion_mode` in its required array). + # Keys the i2v endpoint rejects outright, then constants it always requires. + for key in family.get("image_drop_keys", ()) if image_url else (): + payload.pop(key, None) for key, value in (family.get("static_payload") or {}).items(): payload.setdefault(key, value) - return payload -# --------------------------------------------------------------------------- -# fal_client lazy import (shared with image_generation_tool via fal_common) -# --------------------------------------------------------------------------- +def _video_url_from_result(result: Any) -> Tuple[Any, Optional[str]]: + """Return ``(video_field, url)`` from a FAL result dict (url None if absent).""" + video = result.get("video") if isinstance(result, dict) else None + url = video.get("url") if isinstance(video, dict) else video if isinstance(video, str) else None + return video, url or None + + +# ---- fal_client lazy import + managed FAL gateway (Nous Subscription) --------- _fal_client: Any = None _fal_client_lock = threading.Lock() - -def _load_fal_client() -> Any: - """Lazy-load the ``fal_client`` SDK and cache it on this module. - - Delegates the actual import to :func:`tools.fal_common.import_fal_client` - so the ``lazy_deps`` ensure-install handling stays in one place. - - Thread-safe via double-checked locking: concurrent first calls import - the SDK exactly once instead of each racing thread re-running the import. - """ - global _fal_client - if _fal_client is not None: - return _fal_client - with _fal_client_lock: - if _fal_client is not None: # re-check inside the lock - return _fal_client - from tools.fal_common import import_fal_client - _fal_client = import_fal_client() - return _fal_client - - -# --------------------------------------------------------------------------- -# Managed FAL gateway (Nous Subscription) -# --------------------------------------------------------------------------- - _managed_fal_video_client: Any = None _managed_fal_video_client_config: Any = None _managed_fal_video_client_lock = threading.Lock() -def _resolve_managed_fal_video_gateway(): - """Resolve the FAL video route from the stored selection. +def _load_fal_client() -> Any: + """Lazy-load ``fal_client`` once via ``tools.fal_common``.""" + global _fal_client + with _fal_client_lock: + if _fal_client is None: + from tools.fal_common import import_fal_client - Plain switch on the stored ``video_gen`` provider string — mirrors the - image FAL resolver: ``"nous"`` (or legacy ``use_gateway: true``) → - managed only (unentitled ⇒ selection-naming error); any other stored - provider → direct only (missing FAL_KEY ⇒ selection-naming error); - never-configured category → legacy credential autodetect. + _fal_client = import_fal_client() + return _fal_client + + +def _resolve_managed_fal_video_gateway(): + """Resolve the FAL video route from the stored ``video_gen`` selection. + + ``"nous"`` → managed only (unentitled ⇒ selection-naming error); any other stored provider → + direct only (missing FAL_KEY ⇒ selection-naming error); never-configured → legacy autodetect. """ from tools.managed_tool_gateway import resolve_managed_tool_gateway - from tools.tool_backend_helpers import ( - NOUS_MANAGED_PROVIDER, - fal_key_is_configured, - read_selection, - selection_error, - ) + from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER, fal_key_is_configured, read_selection, selection_error selected = read_selection("video_gen") if selected == NOUS_MANAGED_PROVIDER: gateway = resolve_managed_tool_gateway("fal-queue") if gateway is None: raise ValueError(selection_error( - "video_gen", - NOUS_MANAGED_PROVIDER, - "the Nous Tool Gateway is not available (not entitled or " - "unreachable)", + "video_gen", NOUS_MANAGED_PROVIDER, "the Nous Tool Gateway is not available (not entitled or unreachable)", )) return gateway if selected is not None: if not fal_key_is_configured(): - raise ValueError(selection_error( - "video_gen", - selected, - "FAL_KEY is not set", - )) + raise ValueError(selection_error("video_gen", selected, "FAL_KEY is not set")) return None - # Never-configured category: legacy credential autodetect (do NOT persist). - if fal_key_is_configured(): - return None - return resolve_managed_tool_gateway("fal-queue") + return None if fal_key_is_configured() else resolve_managed_tool_gateway("fal-queue") + + +def _check_fal_video_available() -> bool: + """True if the selected (or, never-configured, any) FAL backend is reachable. Never raises on a + stored-but-broken selection — the honest selection-naming error surfaces at call time.""" + from tools.tool_backend_helpers import fal_key_is_configured + + try: + return _resolve_managed_fal_video_gateway() is not None or fal_key_is_configured() + except ValueError: + return False def _get_managed_fal_video_client(managed_gateway): @@ -623,140 +307,83 @@ def _get_managed_fal_video_client(managed_gateway): global _managed_fal_video_client, _managed_fal_video_client_config from tools.fal_common import _ManagedFalSyncClient - client_config = ( - managed_gateway.gateway_origin.rstrip("/"), - managed_gateway.nous_user_token, - ) + client_config = (managed_gateway.gateway_origin.rstrip("/"), managed_gateway.nous_user_token) with _managed_fal_video_client_lock: - if _managed_fal_video_client is not None and _managed_fal_video_client_config == client_config: - return _managed_fal_video_client - - _load_fal_client() - _managed_fal_video_client = _ManagedFalSyncClient( - _fal_client, - key=managed_gateway.nous_user_token, - queue_run_origin=managed_gateway.gateway_origin, - ) - _managed_fal_video_client_config = client_config + if _managed_fal_video_client is None or _managed_fal_video_client_config != client_config: + _managed_fal_video_client = _ManagedFalSyncClient( + _load_fal_client(), key=managed_gateway.nous_user_token, queue_run_origin=managed_gateway.gateway_origin, + ) + _managed_fal_video_client_config = client_config return _managed_fal_video_client def _submit_fal_video_request(endpoint: str, arguments: Dict[str, Any]): - """Submit a FAL video request using direct credentials or the managed queue gateway. - - Returns a request handle whose ``.get()`` blocks until the result is ready. - """ - _load_fal_client() - request_headers = {"x-idempotency-key": str(uuid.uuid4())} + """Submit via direct credentials or the managed queue gateway; ``.get()`` blocks.""" + client = _load_fal_client() + headers = {"x-idempotency-key": str(uuid.uuid4())} managed_gateway = _resolve_managed_fal_video_gateway() if managed_gateway is None: - return _fal_client.submit(endpoint, arguments=arguments, headers=request_headers) - - managed_client = _get_managed_fal_video_client(managed_gateway) + return client.submit(endpoint, arguments=arguments, headers=headers) try: - return managed_client.submit( - endpoint, - arguments=arguments, - headers=request_headers, - ) + return _get_managed_fal_video_client(managed_gateway).submit(endpoint, arguments=arguments, headers=headers) except Exception as exc: from tools.fal_common import _extract_http_status status = _extract_http_status(exc) if status is not None and 400 <= status < 500: raise ValueError( - f"Nous Subscription gateway rejected endpoint '{endpoint}' " - f"(HTTP {status}). This model may not yet be enabled on " - f"the Nous Portal's FAL proxy. Either:\n" + f"Nous Subscription gateway rejected endpoint '{endpoint}' (HTTP {status}). This model may not yet " + f"be enabled on the Nous Portal's FAL proxy. Either:\n" f" • Set FAL_KEY in your environment to use FAL.ai directly, or\n" f" • Pick a different model via `hermes tools` → Video Generation." ) from exc raise -def _check_fal_video_available() -> bool: - """True if the FAL video backend selected via `hermes tools` (or, on a - never-configured install, any FAL backend) is reachable. - - Never raises — a stored-but-broken selection reports False here; the - honest selection-naming error surfaces at call time from - ``_resolve_managed_fal_video_gateway``. - """ - from tools.managed_tool_gateway import resolve_managed_tool_gateway - from tools.tool_backend_helpers import ( - NOUS_MANAGED_PROVIDER, - fal_key_is_configured, - read_selection, - ) - - selected = read_selection("video_gen") - if selected == NOUS_MANAGED_PROVIDER: - return resolve_managed_tool_gateway("fal-queue") is not None - if selected is not None: - return fal_key_is_configured() - if fal_key_is_configured(): - return True - return resolve_managed_tool_gateway("fal-queue") is not None - - -# --------------------------------------------------------------------------- -# Upscaler (SeedVR2 — video upscale pass) -# --------------------------------------------------------------------------- - -# ByteDance SeedVR2 on FAL: $0.001/megapixel of output video. A 5s 720p→1440p -# 2x pass is roughly $0.44. Faithful restoration-style upscaler (the same -# model family Krea exposes as its "SeedVR2" video enhancer). +# ByteDance SeedVR2 on FAL: $0.001/megapixel of output; a 5s 720p→1440p 2x pass is roughly $0.44. UPSCALER_ENDPOINT = "fal-ai/seedvr/upscale/video" UPSCALER_FACTOR = 2 -def _upscale_video( - video_url: str, - source_request_id: Optional[str] = None, -) -> Optional[str]: - """Upscale a generated video via SeedVR2; return the new URL or None. - - Best-effort: any failure logs and returns ``None`` so the caller falls - back to the native-resolution video — an upscale failure must never - destroy an already-successful generation. - """ +def _upscale_video(video_url: str, source_request_id: Optional[str] = None) -> Optional[str]: + """Best-effort SeedVR2 upscale; returns the new URL or None (never raises).""" try: logger.info("Upscaling video with SeedVR2 (%dx)...", UPSCALER_FACTOR) - arguments: Dict[str, Any] = { - "video_url": video_url, - "upscale_mode": "factor", - "upscale_factor": UPSCALER_FACTOR, - } + arguments: Dict[str, Any] = {"video_url": video_url, "upscale_mode": "factor", "upscale_factor": UPSCALER_FACTOR} if _resolve_managed_fal_video_gateway() is not None: if not source_request_id: raise RuntimeError("Managed SeedVR upscale requires the source FAL request id") arguments["source_request_id"] = source_request_id - handle = _submit_fal_video_request(UPSCALER_ENDPOINT, arguments) - result = handle.get() + result = _submit_fal_video_request(UPSCALER_ENDPOINT, arguments).get() except Exception as exc: # noqa: BLE001 logger.warning("Video upscale failed: %s", exc) return None - - video = (result or {}).get("video") if isinstance(result, dict) else None - if isinstance(video, dict) and video.get("url"): - return video["url"] - if isinstance(video, str) and video: - return video - logger.warning("Video upscaler returned no URL") - return None + _video, url = _video_url_from_result(result) + if not url: + logger.warning("Video upscaler returned no URL") + return url -# --------------------------------------------------------------------------- -# Provider -# --------------------------------------------------------------------------- +# ---- Provider --------------------------------------------------------------- + +_NO_BACKEND_MSG = ( + "No FAL backend available. Either set FAL_KEY (run `hermes tools` → Video Generation → FAL to configure) " + "or sign in to Nous (`hermes setup`) for managed gateway access." +) +_MODALITY_MISSING_MSG = { + "image": "FAL family {fid} has no image-to-video endpoint. Pick a family with image-to-video support via " + "`hermes tools` → Video Generation.", + "text": "FAL family {fid} has no text-to-video endpoint. Pass an image_url to use its image-to-video endpoint, " + "or pick a different family.", +} + + +def _fal_error(error: str, error_type: str, prompt: str, model: str = "", aspect_ratio: str = "") -> Dict[str, Any]: + return error_response(error=error, error_type=error_type, provider="fal", model=model, prompt=prompt, aspect_ratio=aspect_ratio) class FALVideoGenProvider(VideoGenProvider): - """FAL.ai multi-family video generation backend. - - Routes between text-to-video and image-to-video endpoints automatically - based on whether ``image_url`` was provided. - """ + """FAL.ai multi-family backend; routes t2v/i2v on ``image_url`` presence.""" @property def name(self) -> str: @@ -775,27 +402,10 @@ class FALVideoGenProvider(VideoGenProvider): def list_models(self) -> List[Dict[str, Any]]: out: List[Dict[str, Any]] = [] for fid, meta in FAL_FAMILIES.items(): - modalities: List[str] = [] - if meta.get("text_endpoint"): - modalities.append("text") - if meta.get("image_endpoint"): - modalities.append("image") - entry: Dict[str, Any] = { - "id": fid, - "display": meta["display"], - "speed": meta["speed"], - "strengths": meta["strengths"], - "price": meta["price"], - "tier": meta.get("tier", "premium"), - "modalities": modalities, - } - durs = meta.get("durations") - if durs: - if _is_duration_range(durs): - entry["min_duration"], entry["max_duration"] = durs - else: - entry["min_duration"] = min(durs) - entry["max_duration"] = max(durs) + entry: Dict[str, Any] = {"id": fid, **{k: meta[k] for k in ("display", "speed", "strengths", "price", "tier")}, + "modalities": _modalities(meta)} + if meta["durations"]: + entry["min_duration"], entry["max_duration"] = _duration_bounds(meta["durations"]) out.append(entry) return out @@ -804,258 +414,105 @@ class FALVideoGenProvider(VideoGenProvider): def get_setup_schema(self) -> Dict[str, Any]: return { - "name": "FAL", - "badge": "paid", - "tag": "LTX, Pixverse, Seedance 2.0/2.5/Mini, Veo 3.1, MiniMax H3, FLUX 3, Kling 4K, Happy Horse, Grok Imagine, Gemini Omni — text-to-video & image-to-video", - "env_vars": [ - { - "key": "FAL_KEY", - "prompt": "FAL.ai API key", - "url": "https://fal.ai/dashboard/keys", - }, - ], + "name": "FAL", "badge": "paid", + "tag": "LTX, Pixverse, Seedance 2.0/2.5/Mini, Veo 3.1, MiniMax H3, FLUX 3, Kling 4K, Happy Horse, Grok Imagine, " + "Gemini Omni — text-to-video & image-to-video", + "env_vars": [{"key": "FAL_KEY", "prompt": "FAL.ai API key", "url": "https://fal.ai/dashboard/keys"}], } def capabilities(self) -> Dict[str, Any]: - # Active-model-aware (mirrors the image_gen fal plugin, #97057): - # report the RESOLVED family's actual surface so the dynamic tool - # schema gates params on what the selected model honors, not a - # union that overstates every axis. Falls back to the cross-family - # union if resolution fails (never raises). + # Report the RESOLVED family's surface so the dynamic tool schema gates params on what + # the selected model honors; fall back to the cross-family union if resolution fails (never raises). try: _family_id, family = _resolve_family(None) except Exception: # noqa: BLE001 family = None if family: - modalities = [] - if family.get("text_endpoint"): - modalities.append("text") - if family.get("image_endpoint"): - modalities.append("image") - durs = family.get("durations") or (1, 1) - if _is_duration_range(durs): - lo, hi = durs - else: - lo, hi = min(durs), max(durs) + lo, hi = _duration_bounds(family["durations"] or (1, 1)) return { - "modalities": modalities or ["text"], - "aspect_ratios": list(family.get("aspect_ratios") or []), - "resolutions": list(family.get("resolutions") or []), - "max_duration": hi, - "min_duration": lo, - "supports_audio": bool(family.get("audio")), - # Always-on native audio (no toggle): surfaces as a - # description line, not a param. Verified per-family - # against fal model pages (H3, Grok 1.5, Happy Horse, - # Gemini Omni Flash all return native audio every run). - "audio_always_on": bool(family.get("audio_native")), - "supports_negative_prompt": bool(family.get("negative")), - # Explicit per-family key (contract-tested); absent would - # mean a catalog bug, so fail closed here. - "supports_seed": bool(family.get("seed", False)), - # SeedVR upscaler chains for any FAL video family. - "supports_upscale": True, + "modalities": _modalities(family) or ["text"], + "aspect_ratios": list(family["aspect_ratios"] or []), "resolutions": list(family["resolutions"] or []), + "max_duration": hi, "min_duration": lo, "supports_audio": bool(family["audio"]), + "audio_always_on": bool(family.get("audio_native")), # no toggle: description line, not a param + "supports_negative_prompt": bool(family["negative"]), + "supports_seed": bool(family["seed"]), + "supports_upscale": True, # SeedVR chains for any family "max_reference_images": 0, } - # Fallback: union across families (legacy shape). - max_dur = 1 - min_dur: Optional[int] = None - for meta in FAL_FAMILIES.values(): - durs = meta.get("durations") - if not durs: - continue - if _is_duration_range(durs): - lo, hi = durs - else: - lo, hi = min(durs), max(durs) - max_dur = max(max_dur, hi) - min_dur = lo if min_dur is None else min(min_dur, lo) + bounds = [_duration_bounds(m["durations"]) for m in FAL_FAMILIES.values() if m["durations"]] return { - "modalities": ["text", "image"], - "aspect_ratios": ["16:9", "9:16", "1:1"], - "resolutions": ["360p", "540p", "720p", "1080p"], - "max_duration": max_dur, - "min_duration": min_dur if min_dur is not None else 1, - "supports_audio": True, - "supports_negative_prompt": True, - "supports_seed": True, - "supports_upscale": True, - "max_reference_images": 0, + "modalities": ["text", "image"], "aspect_ratios": ["16:9", "9:16", "1:1"], + "resolutions": ["360p", "540p", "720p", "1080p"], "max_duration": max([1] + [hi for _lo, hi in bounds]), + "min_duration": min([lo for lo, _hi in bounds], default=1), "supports_audio": True, + "supports_negative_prompt": True, "supports_seed": True, "supports_upscale": True, "max_reference_images": 0, } def generate( - self, - prompt: str, - *, - model: Optional[str] = None, - image_url: Optional[str] = None, - reference_image_urls: Optional[List[str]] = None, - duration: Optional[int] = None, - aspect_ratio: str = "16:9", - resolution: str = "720p", - negative_prompt: Optional[str] = None, - audio: Optional[bool] = None, - seed: Optional[int] = None, - upscale: Optional[bool] = None, - **kwargs: Any, + self, prompt: str, *, model: Optional[str] = None, image_url: Optional[str] = None, + reference_image_urls: Optional[List[str]] = None, duration: Optional[int] = None, + aspect_ratio: str = "16:9", resolution: str = "720p", negative_prompt: Optional[str] = None, + audio: Optional[bool] = None, seed: Optional[int] = None, upscale: Optional[bool] = None, **kwargs: Any, ) -> Dict[str, Any]: if not _check_fal_video_available(): from tools.tool_backend_helpers import read_selection if read_selection("video_gen") is not None: - # A stored selection that cannot run gets the honest - # selection-naming error from the strict resolver. + # A stored selection that cannot run gets the honest selection-naming error from the strict resolver. try: _resolve_managed_fal_video_gateway() except ValueError as exc: - return error_response( - error=str(exc), - error_type="auth_required", - provider="fal", - prompt=prompt, - ) - return error_response( - error=( - "No FAL backend available. Either set FAL_KEY " - "(run `hermes tools` → Video Generation → FAL to configure) " - "or sign in to Nous (`hermes setup`) for managed gateway access." - ), - error_type="auth_required", - provider="fal", - prompt=prompt, - ) - + return _fal_error(str(exc), "auth_required", prompt) + return _fal_error(_NO_BACKEND_MSG, "auth_required", prompt) try: _load_fal_client() except ImportError: - return error_response( - error="fal_client Python package not installed (pip install fal-client)", - error_type="missing_dependency", - provider="fal", - prompt=prompt, - ) + return _fal_error("fal_client Python package not installed (pip install fal-client)", "missing_dependency", prompt) prompt = (prompt or "").strip() family_id, family = _resolve_family(model) - - # Route: image_url → image-to-video endpoint; else → text-to-video. image_url_norm = (image_url or "").strip() or None - if image_url_norm: - endpoint = family.get("image_endpoint") - modality_used = "image" - if not endpoint: - return error_response( - error=( - f"FAL family {family_id} has no image-to-video " - f"endpoint. Pick a family with image-to-video support " - f"via `hermes tools` → Video Generation." - ), - error_type="modality_unsupported", - provider="fal", model=family_id, prompt=prompt, - ) - else: - endpoint = family.get("text_endpoint") - modality_used = "text" - if not endpoint: - return error_response( - error=( - f"FAL family {family_id} has no text-to-video " - f"endpoint. Pass an image_url to use its " - f"image-to-video endpoint, or pick a different family." - ), - error_type="modality_unsupported", - provider="fal", model=family_id, prompt=prompt, - ) - + modality_used = "image" if image_url_norm else "text" # routes to the i2v vs t2v endpoint + endpoint = family[f"{modality_used}_endpoint"] + if not endpoint: + msg = _MODALITY_MISSING_MSG[modality_used].format(fid=family_id) + return _fal_error(msg, "modality_unsupported", prompt, model=family_id) if not prompt: - return error_response( - error="prompt is required.", - error_type="missing_prompt", - provider="fal", model=family_id, prompt=prompt, - ) + return _fal_error("prompt is required.", "missing_prompt", prompt, model=family_id) payload = _build_payload( - family, - prompt=prompt, - image_url=image_url_norm, - duration=duration, - aspect_ratio=aspect_ratio, - resolution=resolution, - negative_prompt=negative_prompt, - audio=audio, - seed=seed, + family, prompt=prompt, image_url=image_url_norm, duration=duration, aspect_ratio=aspect_ratio, + resolution=resolution, negative_prompt=negative_prompt, audio=audio, seed=seed, ) - try: handle = _submit_fal_video_request(endpoint, payload) source_request_id = getattr(handle, "request_id", None) - result = handle.get() + video, url = _video_url_from_result(handle.get()) except Exception as exc: - logger.warning( - "FAL video gen failed (family=%s, endpoint=%s): %s", - family_id, endpoint, exc, exc_info=True, - ) - return error_response( - error=f"FAL video generation failed: {exc}", - error_type="api_error", - provider="fal", model=family_id, prompt=prompt, - aspect_ratio=aspect_ratio, - ) - - video = (result or {}).get("video") if isinstance(result, dict) else None - url: Optional[str] = None - if isinstance(video, dict): - url = video.get("url") - elif isinstance(video, str): - url = video - + logger.warning("FAL video gen failed (family=%s, endpoint=%s): %s", family_id, endpoint, exc, exc_info=True) + return _fal_error(f"FAL video generation failed: {exc}", "api_error", prompt, model=family_id, aspect_ratio=aspect_ratio) if not url: - return error_response( - error="FAL returned no video URL in response", - error_type="empty_response", - provider="fal", model=family_id, prompt=prompt, - ) - - # Optional high-resolution pass (SeedVR2). Explicit agent/user opt-in - # only; best-effort — failure falls back to the native-resolution - # video rather than failing the generation. - upscaled = False - if upscale: - upscaled_url = _upscale_video(url, source_request_id) - if upscaled_url: - url = upscaled_url - upscaled = True - else: - logger.warning( - "Video upscale pass failed — returning native-resolution video" - ) + return _fal_error("FAL returned no video URL in response", "empty_response", prompt, model=family_id) + # Optional SeedVR2 pass — explicit opt-in, best-effort: failure falls back to the native video. + upscaled_url = _upscale_video(url, source_request_id) if upscale else None + upscaled = bool(upscaled_url) + if upscale and not upscaled: + logger.warning("Video upscale pass failed — returning native-resolution video") extra: Dict[str, Any] = {"endpoint": endpoint, "upscaled": upscaled} if upscaled: - extra["upscale_factor"] = UPSCALER_FACTOR + url, extra["upscale_factor"] = upscaled_url, UPSCALER_FACTOR if isinstance(video, dict): - if video.get("file_size") and not upscaled: - extra["file_size"] = video["file_size"] - if video.get("content_type"): - extra["content_type"] = video["content_type"] - + extra.update({k: video[k] for k in ("file_size", "content_type") if video.get(k)}) + if upscaled: + extra.pop("file_size", None) # native-resolution size no longer applies return success_response( - video=url, - model=family_id, - prompt=prompt, - modality=modality_used, + video=url, model=family_id, prompt=prompt, modality=modality_used, aspect_ratio=aspect_ratio if "aspect_ratio" in payload else "", duration=int("".join(c for c in str(payload["duration"]) if c.isdigit()) or "0") if "duration" in payload else 0, - provider="fal", - extra=extra, + provider="fal", extra=extra, ) -# --------------------------------------------------------------------------- -# Plugin entry point -# --------------------------------------------------------------------------- - - def register(ctx) -> None: """Plugin entry point — wire ``FALVideoGenProvider`` into the registry.""" ctx.register_video_gen_provider(FALVideoGenProvider()) diff --git a/plugins/video_gen/xai/__init__.py b/plugins/video_gen/xai/__init__.py index 4e45302e5e..bd0baac3e9 100644 --- a/plugins/video_gen/xai/__init__.py +++ b/plugins/video_gen/xai/__init__.py @@ -1,19 +1,13 @@ """xAI Grok-Imagine video generation backend. -Surface: text-to-video, image-to-video, and reference-to-video through the -unified video provider. xAI edit/extend are exposed through separate tools. +Surface: text-, image- and reference-to-video through the unified video provider; xAI +edit/extend are exposed by ``tools.xai_video_tools`` via ``run_xai_video_edit`` / ``run_xai_video_extend``. -Originally salvaged from PR #10600 by @Jaaneek; reshaped into the -:class:`VideoGenProvider` plugin interface and trimmed to the -generate-only surface. - -Authentication: xAI Grok OAuth tokens (preferred — billed against the -user's SuperGrok or X Premium+ subscription) or ``XAI_API_KEY``. Both routes are -resolved through ``tools.xai_http.resolve_xai_http_credentials`` so a -single login covers chat + TTS + image gen + video gen + transcription. -When xAI storage is enabled, the primary ``video`` / ``public_url`` fields are the -stored files-cdn HTTPS link. Pass that public MP4 URL as ``video_url`` for -edit/extend; it is sent to xAI as ``video.url``. +Authentication: xAI Grok OAuth tokens (preferred — billed to the user's SuperGrok / X Premium+ +subscription) or ``XAI_API_KEY``, both via ``tools.xai_http.resolve_xai_http_credentials`` so one +login covers chat + TTS + image gen + video gen + transcription. When xAI storage is enabled, the +primary ``video`` / ``public_url`` fields are the stored files-cdn HTTPS link; pass that public MP4 +URL as ``video_url`` for edit/extend (sent to xAI as ``video.url``). """ from __future__ import annotations @@ -25,23 +19,15 @@ import mimetypes import os import uuid from pathlib import Path -from typing import Any, Dict, List, Optional, Tuple +from typing import Any, Callable, Coroutine, Dict, List, Optional, Tuple import httpx -from agent.video_gen_provider import ( - VideoGenProvider, - error_response, - success_response, -) +from agent.video_gen_provider import VideoGenProvider, error_response, success_response logger = logging.getLogger(__name__) -# --------------------------------------------------------------------------- -# Constants -# --------------------------------------------------------------------------- - DEFAULT_XAI_BASE_URL = "https://api.x.ai/v1" DEFAULT_TEXT_TO_VIDEO_MODEL = "grok-imagine-video" DEFAULT_IMAGE_TO_VIDEO_MODEL = "grok-imagine-video-1.5" @@ -57,41 +43,37 @@ VALID_ASPECT_RATIOS = {"1:1", "16:9", "9:16", "4:3", "3:4", "3:2", "2:3"} VALID_RESOLUTIONS = {"480p", "720p"} MAX_REFERENCE_IMAGES = 7 +_REMOTE_PREFIXES = ("http://", "https://") +_TERMINAL_POLL_STATUSES = {"done", "failed", "error", "expired", "cancelled"} _MODELS: Dict[str, Dict[str, Any]] = { "grok-imagine-video": { - "display": "Grok Imagine Video", - "speed": "~60-240s", - "strengths": "Text-to-video; legacy image-to-video fallback.", - "price": "see https://docs.x.ai/developers/models/grok-imagine-video", - "modalities": ["text", "image"], + "display": "Grok Imagine Video", "speed": "~60-240s", "strengths": "Text-to-video; legacy image-to-video fallback.", + "price": "see https://docs.x.ai/developers/models/grok-imagine-video", "modalities": ["text", "image"], }, "grok-imagine-video-1.5": { - "display": "Grok Imagine Video 1.5", - "speed": "~60-240s", - "strengths": "Latest xAI image-to-video model.", - "price": "see https://docs.x.ai/developers/pricing", - "modalities": ["image"], + "display": "Grok Imagine Video 1.5", "speed": "~60-240s", "strengths": "Latest xAI image-to-video model.", + "price": "see https://docs.x.ai/developers/pricing", "modalities": ["image"], }, } -_IMAGE_TO_VIDEO_COMPAT_MODEL_IDS = { - "grok-imagine-video-1.5-preview", - "grok-imagine-video-1.5-2026-05-30", -} +_IMAGE_TO_VIDEO_COMPAT_MODEL_IDS = {"grok-imagine-video-1.5-preview", "grok-imagine-video-1.5-2026-05-30"} + +_AUTH_REQUIRED_MSG = ( + "No xAI credentials found. Sign in via `hermes auth add xai-oauth` " + "(SuperGrok / Premium+) or set XAI_API_KEY from https://console.x.ai/." +) +_PUBLIC_URL_HINT = "(e.g. the `image`/`public_url` from a prior Imagine result)" -# --------------------------------------------------------------------------- -# HTTP helpers -# --------------------------------------------------------------------------- +# ---- Credentials / HTTP helpers ------------------------------------------- def _resolve_xai_credentials() -> Tuple[str, str]: """Return ``(api_key, base_url)`` from the shared xAI credential resolver. - Order: runtime provider (xai-oauth pool entry) → singleton ``auth.json`` - OAuth tokens → ``XAI_API_KEY`` env var. ``api_key`` is empty when no - credential source is available; callers must check before using it. + Order: runtime provider (xai-oauth pool entry) → singleton ``auth.json`` OAuth tokens → + ``XAI_API_KEY`` env var. ``api_key`` is empty when no source is available; callers must check. """ try: from tools.xai_http import resolve_xai_http_credentials @@ -100,198 +82,79 @@ def _resolve_xai_credentials() -> Tuple[str, str]: except Exception as exc: logger.debug("xAI credential resolver failed: %s", exc) creds = {} - - api_key = str(creds.get("api_key") or os.getenv("XAI_API_KEY", "")).strip() - base_url = str( - creds.get("base_url") - or os.getenv("XAI_BASE_URL") - or DEFAULT_XAI_BASE_URL - ).strip().rstrip("/") - return api_key, base_url - - -def _xai_user_agent() -> str: - try: - from tools.xai_http import hermes_xai_user_agent - - return hermes_xai_user_agent() - except Exception: - return "hermes-agent/video_gen" + base_url = str(creds.get("base_url") or os.getenv("XAI_BASE_URL") or DEFAULT_XAI_BASE_URL) + return str(creds.get("api_key") or os.getenv("XAI_API_KEY", "")).strip(), base_url.strip().rstrip("/") def _xai_headers(api_key: str) -> Dict[str, str]: - return { - "Authorization": f"Bearer {api_key}", - "Content-Type": "application/json", - "User-Agent": _xai_user_agent(), - } + try: + from tools.xai_http import hermes_xai_user_agent + + user_agent = hermes_xai_user_agent() + except Exception: + user_agent = "hermes-agent/video_gen" + return {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json", "User-Agent": user_agent} -def _raise_if_blocked_local_input(ref: str) -> None: - """Refuse to read a local media path that Hermes' read deny-list blocks. +def _xai_error(error: str, error_type: str, prompt: str, model: str = "", aspect_ratio: str = "") -> Dict[str, Any]: + return error_response(error=error, error_type=error_type, provider="xai", model=model, prompt=prompt, aspect_ratio=aspect_ratio) - Thin wrapper over the shared ``agent.file_safety.raise_if_read_blocked`` - chokepoint so xAI video inputs enforce the same credential-store guard as - the image providers. Fails open if the guard machinery is unavailable - (defense-in-depth, per the denylist's own framing). + +# ---- Input normalization -------------------------------------------------- + + +def _media_ref_to_xai_url(value: str, *, kind: str, fallback_mime: str) -> str: + """Return a URL/data URI accepted by xAI for ``kind`` (``image``/``video``) inputs. + + Remote URLs and matching data URIs pass through; a readable local file of the right MIME + class is inlined as base64; anything else is returned as-is so the caller's URL check rejects + it clearly. Local reads go through Hermes' read deny-list (same credential-store guard as the + image providers), which fails open if its machinery is unavailable. """ + ref = (value or "").strip() + if not ref or ref.lower().startswith(_REMOTE_PREFIXES + (f"data:{kind}/",)): + return ref + path = Path(ref).expanduser() + if not path.is_file(): + return ref try: from agent.file_safety import raise_if_read_blocked except Exception as exc: # noqa: BLE001 - guard must never break loading logger.debug("xAI media input read guard unavailable: %s", exc) - return - raise_if_read_blocked(ref) - - -def _image_ref_to_xai_url(value: str) -> str: - """Return a URL/data URI accepted by xAI for image inputs.""" - ref = (value or "").strip() - if not ref: - return "" - lower = ref.lower() - if lower.startswith(("http://", "https://", "data:image/")): + else: + raise_if_read_blocked(ref) + mime = mimetypes.guess_type(path.name)[0] or fallback_mime + if not mime.startswith(f"{kind}/"): return ref - - path = Path(ref).expanduser() - if not path.is_file(): - return ref - - _raise_if_blocked_local_input(ref) - - mime = mimetypes.guess_type(path.name)[0] or "application/octet-stream" - if not mime.startswith("image/"): - return ref - - encoded = base64.b64encode(path.read_bytes()).decode("ascii") - return f"data:{mime};base64,{encoded}" + return f"data:{mime};base64,{base64.b64encode(path.read_bytes()).decode('ascii')}" def _image_ref_to_xai_input(value: str) -> Optional[Dict[str, str]]: - ref = _image_ref_to_xai_url(value) - if not ref: - return None - lower = ref.lower() - if lower.startswith(("http://", "https://", "data:image/")): - return {"url": ref} - return None + ref = _media_ref_to_xai_url(value, kind="image", fallback_mime="application/octet-stream") + return {"url": ref} if ref and ref.lower().startswith(_REMOTE_PREFIXES + ("data:image/",)) else None -def _xai_video_output_urls( - video: Dict[str, Any], -) -> Tuple[str, Optional[str], Optional[str]]: - """Return ``(public_video_url, temporary_url, stored_public_url)``. - - ``public_video_url`` is the stored files-cdn HTTPS MP4 (``public_url``) when - storage is enabled; otherwise xAI's temporary ``video.url``. Pass this value - as ``video_url`` for edit/extend chaining. - """ - file_output = video.get("file_output") if isinstance(video.get("file_output"), dict) else {} - file_output = file_output or {} - stored_public = file_output.get("public_url") - stored_public = stored_public.strip() if isinstance(stored_public, str) else None - temporary = video.get("url") - temporary = temporary.strip() if isinstance(temporary, str) else None - public_video_url = stored_public or temporary or "" - temporary_out = ( - temporary - if temporary and stored_public and temporary != stored_public - else None - ) - return public_video_url, temporary_out, stored_public - - -def _video_ref_to_xai_url(value: str) -> str: - """Return a URL/data URI accepted by xAI for video inputs.""" - ref = (value or "").strip() - if not ref: - return "" - lower = ref.lower() - if lower.startswith(("http://", "https://", "data:video/")): - return ref - - path = Path(ref).expanduser() - if not path.is_file(): - return ref - - _raise_if_blocked_local_input(ref) - - mime = mimetypes.guess_type(path.name)[0] or "video/mp4" - if not mime.startswith("video/"): - return ref - - encoded = base64.b64encode(path.read_bytes()).decode("ascii") - return f"data:{mime};base64,{encoded}" - - -async def _video_input_from_public_url( - value: str, - *, - api_key: str, - base_url: str, -) -> Optional[Dict[str, str]]: +async def _video_input_from_public_url(value: str, *, api_key: str, base_url: str) -> Optional[Dict[str, str]]: """Build xAI ``video`` input using a public HTTPS URL (``url`` field only).""" ref = (value or "").strip() - if not ref: - return None - - path = Path(ref).expanduser() - if path.is_file(): - data_ref = _video_ref_to_xai_url(ref) - return {"url": data_ref} if data_ref else None - - lower = ref.lower() - if not lower.startswith(("http://", "https://")): - return None - - return {"url": ref} + if ref and Path(ref).expanduser().is_file(): + ref = _media_ref_to_xai_url(ref, kind="video", fallback_mime="video/mp4") + return {"url": ref} if ref else None + return {"url": ref} if ref.lower().startswith(_REMOTE_PREFIXES) else None -def _normalize_reference_images( - reference_image_urls: Optional[List[str]], -) -> Tuple[Optional[List[Dict[str, str]]], Optional[str]]: - refs: List[Dict[str, str]] = [] - for url in reference_image_urls or []: - cleaned = (url or "").strip() - if not cleaned: - continue - normalized = _image_ref_to_xai_input(cleaned) - if not normalized: - return None, ( - "reference_image_urls must be public HTTPS URLs or data URIs " - "(e.g. the `image`/`public_url` from a prior Imagine result)" - ) - refs.append(normalized) - return (refs if refs else None), None +def _clamp_duration(duration: Optional[int], *, has_reference_images: bool = False, max_seconds: int = 15, + default: int = DEFAULT_DURATION) -> int: + """Clamp to ``[1, max_seconds]``; reference-to-video additionally caps at 10s.""" + value = max(1, min(max_seconds, duration if duration is not None else default)) + return min(value, 10) if has_reference_images else value -def _clamp_duration( - duration: Optional[int], - *, - has_reference_images: bool = False, - max_seconds: int = 15, - default: int = DEFAULT_DURATION, -) -> int: - value = duration if duration is not None else default - if value < 1: - value = 1 - if value > max_seconds: - value = max_seconds - if has_reference_images and value > 10: - value = 10 - return value - - -def _resolve_model_for_modality( - model: Optional[str], - *, - modality: str, - explicit_model: bool, -) -> str: +def _resolve_model_for_modality(model: Optional[str], *, modality: str, explicit_model: bool) -> str: """Select xAI's text/video model without treating config as a prompt override. - ``grok-imagine-video-1.5`` currently rejects text-only video - generation, but it is the desired image-to-video backend. Explicit tool - ``model=`` still wins for users who intentionally request another model. + ``grok-imagine-video-1.5`` rejects text-only generation but is the desired image-to-video + backend. Explicit tool ``model=`` still wins for users who intentionally request another model. """ requested = (model or "").strip() if explicit_model and requested: @@ -303,64 +166,7 @@ def _resolve_model_for_modality( return requested or DEFAULT_TEXT_TO_VIDEO_MODEL -async def _submit( - client: httpx.AsyncClient, - payload: Dict[str, Any], - *, - api_key: str, - base_url: str, - endpoint: str = "generations", -) -> str: - """POST to one of xAI's async video endpoints and return request_id.""" - response = await client.post( - f"{base_url}/videos/{endpoint}", - headers={**_xai_headers(api_key), "x-idempotency-key": str(uuid.uuid4())}, - json=payload, - timeout=60, - ) - response.raise_for_status() - body = response.json() - request_id = body.get("request_id") - if not request_id: - raise RuntimeError("xAI video response did not include request_id") - return request_id - - -async def _poll( - client: httpx.AsyncClient, - request_id: str, - *, - api_key: str, - base_url: str, - timeout_seconds: int, - poll_interval: int, -) -> Dict[str, Any]: - elapsed = 0.0 - last_status = "queued" - while elapsed < timeout_seconds: - response = await client.get( - f"{base_url}/videos/{request_id}", - headers=_xai_headers(api_key), - timeout=30, - ) - response.raise_for_status() - body = response.json() - last_status = (body.get("status") or "").lower() - - if last_status == "done": - return {"status": "done", "body": body} - if last_status in {"failed", "error", "expired", "cancelled"}: - return {"status": last_status, "body": body} - - await asyncio.sleep(poll_interval) - elapsed += poll_interval - - return {"status": "timeout", "body": {"status": last_status}} - - -# --------------------------------------------------------------------------- -# Provider -# --------------------------------------------------------------------------- +# ---- Provider --------------------------------------------------------------- class XAIVideoGenProvider(VideoGenProvider): @@ -375,8 +181,7 @@ class XAIVideoGenProvider(VideoGenProvider): return "xAI" def is_available(self) -> bool: - api_key, _ = _resolve_xai_credentials() - return bool(api_key) + return has_xai_video_credentials() def list_models(self) -> List[Dict[str, Any]]: return [{"id": mid, **meta} for mid, meta in _MODELS.items()] @@ -385,11 +190,8 @@ class XAIVideoGenProvider(VideoGenProvider): return DEFAULT_MODEL def get_setup_schema(self) -> Dict[str, Any]: - # Auth resolution lives entirely in the shared ``xai_grok`` post_setup - # hook (``hermes_cli/tools_config.py``) so the picker doesn't blindly - # prompt for an API key when the user is already signed in via xAI - # Grok OAuth (SuperGrok / Premium+) — TTS / image gen / video gen - # all share the same credential resolver. The hook offers an + # Auth resolution lives in the shared ``xai_grok`` post_setup hook (hermes_cli/tools_config.py) so the + # picker doesn't prompt for an API key when already signed in via xAI Grok OAuth; the hook offers an # OAuth-vs-API-key choice when neither is configured. try: from tools.xai_http import xai_storage_notice_text @@ -398,528 +200,260 @@ class XAIVideoGenProvider(VideoGenProvider): except Exception: storage_notice = "" tag = ( - "grok-imagine-video for text/reference; " - "grok-imagine-video-1.5 for image-to-video; " - "edit/extend: pass the stored public HTTPS MP4 (`video` / " - "`public_url` from a prior Imagine result); uses xAI Grok OAuth " + "grok-imagine-video for text/reference; grok-imagine-video-1.5 for image-to-video; edit/extend: pass " + "the stored public HTTPS MP4 (`video` / `public_url` from a prior Imagine result); uses xAI Grok OAuth " "or XAI_API_KEY" ) if storage_notice: tag += f". {storage_notice}" - return { - "name": "xAI Grok Imagine", - "badge": "paid", - "tag": tag, - "env_vars": [], - "post_setup": "xai_grok", - } + return {"name": "xAI Grok Imagine", "badge": "paid", "tag": tag, "env_vars": [], "post_setup": "xai_grok"} def capabilities(self) -> Dict[str, Any]: return { - "modalities": ["text", "image"], - "aspect_ratios": sorted(VALID_ASPECT_RATIOS), - "resolutions": sorted(VALID_RESOLUTIONS), - "max_duration": 15, - "min_duration": 1, - "supports_audio": False, - "supports_negative_prompt": False, - "supports_seed": True, - "supports_upscale": False, - "max_reference_images": MAX_REFERENCE_IMAGES, + "modalities": ["text", "image"], "aspect_ratios": sorted(VALID_ASPECT_RATIOS), + "resolutions": sorted(VALID_RESOLUTIONS), "max_duration": 15, "min_duration": 1, + "supports_audio": False, "supports_negative_prompt": False, "supports_seed": True, + "supports_upscale": False, "max_reference_images": MAX_REFERENCE_IMAGES, } def generate( - self, - prompt: str, - *, - model: Optional[str] = None, - image_url: Optional[str] = None, - reference_image_urls: Optional[List[str]] = None, - duration: Optional[int] = None, - aspect_ratio: str = DEFAULT_ASPECT_RATIO, - resolution: str = DEFAULT_RESOLUTION, - negative_prompt: Optional[str] = None, - audio: Optional[bool] = None, - seed: Optional[int] = None, + self, prompt: str, *, model: Optional[str] = None, image_url: Optional[str] = None, + reference_image_urls: Optional[List[str]] = None, duration: Optional[int] = None, + aspect_ratio: str = DEFAULT_ASPECT_RATIO, resolution: str = DEFAULT_RESOLUTION, + negative_prompt: Optional[str] = None, audio: Optional[bool] = None, seed: Optional[int] = None, **kwargs: Any, ) -> Dict[str, Any]: - return run_xai_video_generation( - prompt=prompt, - model=model, - explicit_model=bool(kwargs.get("_model_override_explicit")), - image_url=image_url, - reference_image_urls=reference_image_urls, - duration=duration, - aspect_ratio=aspect_ratio, - resolution=resolution, + return _run_xai_video_coroutine( + lambda api_key, base_url: _generate_xai_video_async( + api_key=api_key, base_url=base_url, prompt=prompt, model=model, + explicit_model=bool(kwargs.get("_model_override_explicit")), image_url=image_url, + reference_image_urls=reference_image_urls, duration=duration, aspect_ratio=aspect_ratio, + resolution=resolution, + ), + operation_label="generation", model=model, prompt=prompt, aspect_ratio=aspect_ratio, ) +# ---- Sync entry points (provider + tools.xai_video_tools) ------------------- + + def has_xai_video_credentials() -> bool: - api_key, _ = _resolve_xai_credentials() - return bool(api_key) + return bool(_resolve_xai_credentials()[0]) -def run_xai_video_generation( - *, - prompt: str, - model: Optional[str], - explicit_model: bool, - image_url: Optional[str], - reference_image_urls: Optional[List[str]], - duration: Optional[int], - aspect_ratio: str, - resolution: str, -) -> Dict[str, Any]: - return _run_xai_video_coroutine( - _generate_xai_video_async( - prompt=prompt, - model=model, - explicit_model=explicit_model, - image_url=image_url, - reference_image_urls=reference_image_urls, - duration=duration, - aspect_ratio=aspect_ratio, - resolution=resolution, - ), - operation_label="generation", - model=model, - prompt=prompt, - aspect_ratio=aspect_ratio, +def run_xai_video_edit(*, prompt: str, video_url: str, model: Optional[str] = None) -> Dict[str, Any]: + return _run_xai_video_mutation(prompt, video_url, model, endpoint="edits", operation="edit", duration=DEFAULT_DURATION) + + +def run_xai_video_extend(*, prompt: str, video_url: str, duration: Optional[int] = None, + model: Optional[str] = None) -> Dict[str, Any]: + return _run_xai_video_mutation( + prompt, video_url, model, endpoint="extensions", operation="extend", + duration=_clamp_duration(duration, max_seconds=10, default=DEFAULT_EXTEND_DURATION), ) -def run_xai_video_edit( - *, - prompt: str, - video_url: str, - model: Optional[str] = None, -) -> Dict[str, Any]: +def _run_xai_video_mutation(prompt: str, video_url: str, model: Optional[str], *, endpoint: str, operation: str, + duration: int) -> Dict[str, Any]: return _run_xai_video_coroutine( - _edit_xai_video_async(prompt=prompt, video_url=video_url, model=model), - operation_label="edit", - model=model, - prompt=prompt, - aspect_ratio=DEFAULT_ASPECT_RATIO, - ) - - -def run_xai_video_extend( - *, - prompt: str, - video_url: str, - duration: Optional[int] = None, - model: Optional[str] = None, -) -> Dict[str, Any]: - return _run_xai_video_coroutine( - _extend_xai_video_async( - prompt=prompt, - video_url=video_url, - duration=duration, - model=model, + lambda api_key, base_url: _mutate_xai_video_async( + api_key=api_key, base_url=base_url, prompt=prompt, video_url=video_url, model=model, + endpoint=endpoint, operation=operation, duration=duration, ), - operation_label="extend", - model=model, - prompt=prompt, - aspect_ratio=DEFAULT_ASPECT_RATIO, + operation_label=operation, model=model, prompt=prompt, aspect_ratio=DEFAULT_ASPECT_RATIO, ) def _run_xai_video_coroutine( - coro, - *, - operation_label: str, - model: Optional[str], - prompt: str, - aspect_ratio: str, + start: Callable[[str, str], Coroutine[Any, Any, Dict[str, Any]]], *, operation_label: str, + model: Optional[str], prompt: str, aspect_ratio: str, ) -> Dict[str, Any]: + """Resolve credentials, then drive ``start(api_key, base_url)`` on a fresh event loop; + any escaped exception → api_error response.""" + api_key, base_url = _resolve_xai_credentials() + if not api_key: + return _xai_error(_AUTH_REQUIRED_MSG, "auth_required", prompt) try: loop = asyncio.new_event_loop() try: - return loop.run_until_complete(coro) + return loop.run_until_complete(start(api_key, base_url)) finally: loop.close() except Exception as exc: logger.warning("xAI video %s unexpected failure: %s", operation_label, exc, exc_info=True) - return error_response( - error=f"xAI video {operation_label} failed: {exc}", - error_type="api_error", - provider="xai", - model=model or DEFAULT_MODEL, - prompt=prompt, - aspect_ratio=aspect_ratio, + return _xai_error( + f"xAI video {operation_label} failed: {exc}", "api_error", prompt, + model=model or DEFAULT_MODEL, aspect_ratio=aspect_ratio, ) +# ---- Async flows ------------------------------------------------------------ + + async def _generate_xai_video_async( - *, - prompt: str, - model: Optional[str], - explicit_model: bool, - image_url: Optional[str], - reference_image_urls: Optional[List[str]], - duration: Optional[int], - aspect_ratio: str, - resolution: str, + *, api_key: str, base_url: str, prompt: str, model: Optional[str], explicit_model: bool, image_url: Optional[str], + reference_image_urls: Optional[List[str]], duration: Optional[int], aspect_ratio: str, resolution: str, ) -> Dict[str, Any]: - api_key, base_url = _resolve_xai_credentials() - if not api_key: - return _auth_required_response(prompt) - prompt = (prompt or "").strip() - image_input = None - if (image_url or "").strip(): - image_input = _image_ref_to_xai_input(image_url) - if not image_input: - return error_response( - error=( - "image_url must be a public HTTPS URL or data URI " - "(e.g. the `image`/`public_url` from a prior Imagine result)" - ), - error_type="invalid_image_url", - provider="xai", - prompt=prompt, - ) - normalized_aspect_ratio = (aspect_ratio or DEFAULT_ASPECT_RATIO).strip() - normalized_resolution = (resolution or DEFAULT_RESOLUTION).strip().lower() - refs, refs_error = _normalize_reference_images(reference_image_urls) - if refs_error: - return error_response( - error=refs_error, - error_type="invalid_reference_image_urls", - provider="xai", - prompt=prompt, + image_input = _image_ref_to_xai_input(image_url) if (image_url or "").strip() else None + if (image_url or "").strip() and not image_input: + return _xai_error(f"image_url must be a public HTTPS URL or data URI {_PUBLIC_URL_HINT}", "invalid_image_url", prompt) + aspect_ratio = (aspect_ratio or DEFAULT_ASPECT_RATIO).strip() + resolution = (resolution or DEFAULT_RESOLUTION).strip().lower() + refs = [_image_ref_to_xai_input(url.strip()) for url in reference_image_urls or [] if (url or "").strip()] + if not all(refs): + return _xai_error( + f"reference_image_urls must be public HTTPS URLs or data URIs {_PUBLIC_URL_HINT}", "invalid_reference_image_urls", prompt, ) - if not prompt: - return error_response( - error="prompt is required for xAI video generation", - error_type="missing_prompt", - provider="xai", prompt=prompt, - ) - if refs and len(refs) > MAX_REFERENCE_IMAGES: - return error_response( - error=f"reference_image_urls supports at most {MAX_REFERENCE_IMAGES} images on xAI", - error_type="too_many_references", - provider="xai", prompt=prompt, + return _xai_error("prompt is required for xAI video generation", "missing_prompt", prompt) + if len(refs) > MAX_REFERENCE_IMAGES: + return _xai_error( + f"reference_image_urls supports at most {MAX_REFERENCE_IMAGES} images on xAI", "too_many_references", prompt, ) if image_input and refs: - return error_response( - error="image_url and reference_image_urls cannot be combined on xAI", - error_type="conflicting_inputs", - provider="xai", prompt=prompt, - ) + return _xai_error("image_url and reference_image_urls cannot be combined on xAI", "conflicting_inputs", prompt) - if normalized_aspect_ratio not in VALID_ASPECT_RATIOS: - normalized_aspect_ratio = DEFAULT_ASPECT_RATIO - if normalized_resolution not in VALID_RESOLUTIONS: - normalized_resolution = DEFAULT_RESOLUTION + # Unsupported values silently fall back to defaults rather than erroring. + aspect_ratio = aspect_ratio if aspect_ratio in VALID_ASPECT_RATIOS else DEFAULT_ASPECT_RATIO + resolution = resolution if resolution in VALID_RESOLUTIONS else DEFAULT_RESOLUTION modality_used = "reference" if refs else ("image" if image_input else "text") - resolved_model = _resolve_model_for_modality( - model, - modality=modality_used, - explicit_model=explicit_model, - ) + resolved_model = _resolve_model_for_modality(model, modality=modality_used, explicit_model=explicit_model) + # Reference-to-video only exists on the text model: explicit other model = error, implicit (config) = corrected. if refs and resolved_model != DEFAULT_TEXT_TO_VIDEO_MODEL: if explicit_model: - return error_response( - error=( - "xAI reference-to-video requires " - f"{DEFAULT_TEXT_TO_VIDEO_MODEL}; got {resolved_model}" - ), - error_type="unsupported_model", - provider="xai", - model=resolved_model, - prompt=prompt, + return _xai_error( + f"xAI reference-to-video requires {DEFAULT_TEXT_TO_VIDEO_MODEL}; got {resolved_model}", + "unsupported_model", prompt, model=resolved_model, ) resolved_model = DEFAULT_TEXT_TO_VIDEO_MODEL clamped_duration = _clamp_duration(duration, has_reference_images=bool(refs)) - payload = { - "model": resolved_model, - "prompt": prompt, - "duration": clamped_duration, - "aspect_ratio": normalized_aspect_ratio, - "resolution": normalized_resolution, - } - if image_input: - payload["image"] = image_input - if refs: - payload["reference_images"] = refs - + payload = {"model": resolved_model, "prompt": prompt, "duration": clamped_duration, "aspect_ratio": aspect_ratio, + "resolution": resolution} + payload.update({k: v for k, v in (("image", image_input), ("reference_images", refs)) if v}) return await _submit_xai_video_payload( - api_key=api_key, - base_url=base_url, - endpoint="generations", - payload=payload, - prompt=prompt, - resolved_model=resolved_model, - modality=modality_used, - aspect_ratio=normalized_aspect_ratio, - duration=clamped_duration, - operation="generate", - resolution=normalized_resolution, + api_key=api_key, base_url=base_url, endpoint="generations", payload=payload, + prompt=prompt, resolved_model=resolved_model, modality=modality_used, + aspect_ratio=aspect_ratio, duration=clamped_duration, operation="generate", resolution=resolution, ) -async def _run_xai_video_mutation( - *, - prompt: str, - video_url: str, - model: Optional[str], - endpoint: str, - operation: str, +async def _mutate_xai_video_async( + *, api_key: str, base_url: str, prompt: str, video_url: str, model: Optional[str], endpoint: str, operation: str, duration: int, ) -> Dict[str, Any]: """Edit or extend using a public HTTPS ``video_url`` input (``url`` on the wire).""" - api_key, base_url = _resolve_xai_credentials() - if not api_key: - return _auth_required_response(prompt) - prompt = (prompt or "").strip() - video_input = await _video_input_from_public_url( - video_url or "", - api_key=api_key, - base_url=base_url, - ) + video_input = await _video_input_from_public_url(video_url or "", api_key=api_key, base_url=base_url) if not prompt: - return error_response( - error="prompt is required for xAI video edit/extend", - error_type="missing_prompt", - provider="xai", - prompt=prompt, - ) + return _xai_error("prompt is required for xAI video edit/extend", "missing_prompt", prompt) if not video_input: - return error_response( - error=( - "video_url must be a public HTTPS MP4 URL " - "(the `video`/`public_url` from a prior Imagine result)" - ), - error_type="missing_video", - provider="xai", - prompt=prompt, - ) - - resolved_model = _resolve_model_for_modality( - model, - modality="text", - explicit_model=bool(model), - ) - payload: Dict[str, Any] = { - "model": resolved_model, - "prompt": prompt, - "video": video_input, - } + msg = "video_url must be a public HTTPS MP4 URL (the `video`/`public_url` from a prior Imagine result)" + return _xai_error(msg, "missing_video", prompt) + resolved_model = _resolve_model_for_modality(model, modality="text", explicit_model=bool(model)) + payload: Dict[str, Any] = {"model": resolved_model, "prompt": prompt, "video": video_input} if endpoint == "extensions": payload["duration"] = duration - return await _submit_xai_video_payload( - api_key=api_key, - base_url=base_url, - endpoint=endpoint, - payload=payload, - prompt=prompt, - resolved_model=resolved_model, - modality=operation, - aspect_ratio=DEFAULT_ASPECT_RATIO, - duration=duration, - operation=operation, - ) - - -async def _edit_xai_video_async( - *, - prompt: str, - video_url: str, - model: Optional[str], -) -> Dict[str, Any]: - return await _run_xai_video_mutation( - prompt=prompt, - video_url=video_url, - model=model, - endpoint="edits", - operation="edit", - duration=DEFAULT_DURATION, - ) - - -async def _extend_xai_video_async( - *, - prompt: str, - video_url: str, - duration: Optional[int], - model: Optional[str], -) -> Dict[str, Any]: - clamped_duration = _clamp_duration( - duration, - max_seconds=10, - default=DEFAULT_EXTEND_DURATION, - ) - return await _run_xai_video_mutation( - prompt=prompt, - video_url=video_url, - model=model, - endpoint="extensions", - operation="extend", - duration=clamped_duration, - ) - - -def _auth_required_response(prompt: str) -> Dict[str, Any]: - return error_response( - error=( - "No xAI credentials found. Sign in via `hermes auth add xai-oauth` " - "(SuperGrok / Premium+) or set XAI_API_KEY from " - "https://console.x.ai/." - ), - error_type="auth_required", - provider="xai", prompt=prompt, + api_key=api_key, base_url=base_url, endpoint=endpoint, payload=payload, + prompt=prompt, resolved_model=resolved_model, modality=operation, + aspect_ratio=DEFAULT_ASPECT_RATIO, duration=duration, operation=operation, ) async def _submit_xai_video_payload( - *, - api_key: str, - base_url: str, - endpoint: str, - payload: Dict[str, Any], - prompt: str, - resolved_model: str, - modality: str, - aspect_ratio: str, - duration: int, - operation: str, - resolution: Optional[str] = None, + *, api_key: str, base_url: str, endpoint: str, payload: Dict[str, Any], prompt: str, resolved_model: str, + modality: str, aspect_ratio: str, duration: int, operation: str, resolution: Optional[str] = None, ) -> Dict[str, Any]: + """POST ``payload`` to ``/videos/{endpoint}``, poll ``/videos/{request_id}`` to a terminal status, shape the response.""" try: - from tools.xai_http import ( - build_xai_storage_options, - maybe_mark_xai_storage_notice_seen, - read_xai_imagine_storage_config, - ) + from tools.xai_http import build_xai_storage_options, maybe_mark_xai_storage_notice_seen, read_xai_imagine_storage_config - storage_options = build_xai_storage_options( - "video_gen", - filename_prefix="hermes-xai-video", - extension="mp4", - ) + storage_options = build_xai_storage_options("video_gen", filename_prefix="hermes-xai-video", extension="mp4") storage_notice = maybe_mark_xai_storage_notice_seen("video_gen") storage_cfg = read_xai_imagine_storage_config("video_gen") except Exception: - storage_options = None - storage_notice = None - storage_cfg = {"enabled": False} + storage_options, storage_notice, storage_cfg = None, None, {"enabled": False} if storage_options is not None: payload["storage_options"] = storage_options + headers = _xai_headers(api_key) async with httpx.AsyncClient() as client: try: - request_id = await _submit( - client, payload, api_key=api_key, base_url=base_url, - endpoint=endpoint, + response = await client.post( + f"{base_url}/videos/{endpoint}", headers={**headers, "x-idempotency-key": str(uuid.uuid4())}, + json=payload, timeout=60, ) + response.raise_for_status() except httpx.HTTPStatusError as exc: detail = "" try: detail = exc.response.text[:500] except Exception: pass - return error_response( - error=f"xAI submit failed ({exc.response.status_code}): {detail or exc}", - error_type="api_error", - provider="xai", + return _xai_error( + f"xAI submit failed ({exc.response.status_code}): {detail or exc}", "api_error", prompt, model=resolved_model, + ) + request_id = response.json().get("request_id") + if not request_id: + raise RuntimeError("xAI video response did not include request_id") + + elapsed = 0.0 + status, body = "queued", {} + while elapsed < DEFAULT_TIMEOUT_SECONDS: + response = await client.get(f"{base_url}/videos/{request_id}", headers=headers, timeout=30) + response.raise_for_status() + body = response.json() + status = (body.get("status") or "").lower() + if status in _TERMINAL_POLL_STATUSES: + break + await asyncio.sleep(DEFAULT_POLL_INTERVAL_SECONDS) + elapsed += DEFAULT_POLL_INTERVAL_SECONDS + else: + return _xai_error( + f"Timed out waiting for xAI video request after {DEFAULT_TIMEOUT_SECONDS}s", "timeout", prompt, model=resolved_model, - prompt=prompt, ) - poll_result = await _poll( - client, request_id, - api_key=api_key, base_url=base_url, - timeout_seconds=DEFAULT_TIMEOUT_SECONDS, - poll_interval=DEFAULT_POLL_INTERVAL_SECONDS, - ) + if status != "done": + message = (body.get("error", {}) or {}).get("message") or body.get("message") + return _xai_error(message or f"xAI video request ended with status '{status}'", f"xai_{status}", prompt, model=resolved_model) - status = poll_result["status"] - body = poll_result["body"] - - if status == "done": - video = body.get("video") or {} - if not isinstance(video, dict): - video = {} - file_output = video.get("file_output") if isinstance(video.get("file_output"), dict) else {} - file_output = file_output or {} - public_video_url, temporary_url, stored_public_url = _xai_video_output_urls(video) - if not public_video_url: - return error_response( - error="xAI video request completed without a video URL", - error_type="empty_response", - provider="xai", - model=body.get("model") or resolved_model, - prompt=prompt, - ) - extra: Dict[str, Any] = { - "request_id": request_id, - "operation": operation, - "storage_enabled": bool(storage_cfg.get("enabled")), - } - if resolution: - extra["resolution"] = resolution - if storage_notice: - extra["storage_notice"] = storage_notice - if stored_public_url: - extra["public_url"] = stored_public_url - if temporary_url: - extra["temporary_url"] = temporary_url - if file_output: - for key in ( - "filename", - "expires_at", - "public_url_expires_at", - "public_url_error", - "storage_error", - ): - if key in file_output: - extra[key] = file_output[key] - if body.get("usage"): - extra["usage"] = body["usage"] - return success_response( - video=public_video_url, + video = body.get("video") if isinstance(body.get("video"), dict) else {} + # Primary URL is the stored files-cdn HTTPS MP4 (``public_url``) when storage is enabled, else xAI's + # temporary ``video.url``; pass it as ``video_url`` for edit/extend chaining. The temporary URL is only + # reported when it differs from the stored one. + file_output = video.get("file_output") + file_output = file_output if isinstance(file_output, dict) else {} + stored_public = file_output.get("public_url") + stored_public = stored_public.strip() if isinstance(stored_public, str) else None + temporary = video.get("url") + temporary = temporary.strip() if isinstance(temporary, str) else None + public_video_url = stored_public or temporary or "" + if not public_video_url: + return _xai_error( + "xAI video request completed without a video URL", "empty_response", prompt, model=body.get("model") or resolved_model, - prompt=prompt, - modality=modality, - aspect_ratio=aspect_ratio, - duration=video.get("duration") or duration, - provider="xai", - extra=extra, ) - - if status == "timeout": - return error_response( - error=f"Timed out waiting for xAI video request after {DEFAULT_TIMEOUT_SECONDS}s", - error_type="timeout", - provider="xai", - model=resolved_model, - prompt=prompt, - ) - - message = ( - (body.get("error", {}) or {}).get("message") - or body.get("message") - or f"xAI video request ended with status '{status}'" + extra: Dict[str, Any] = {"request_id": request_id, "operation": operation, "storage_enabled": bool(storage_cfg.get("enabled"))} + if resolution: + extra["resolution"] = resolution + if storage_notice: + extra["storage_notice"] = storage_notice + if stored_public: + extra["public_url"] = stored_public + if temporary and temporary != stored_public: + extra["temporary_url"] = temporary + extra.update({k: file_output[k] for k in ("filename", "expires_at", "public_url_expires_at", "public_url_error", "storage_error") + if k in file_output}) + if body.get("usage"): + extra["usage"] = body["usage"] + return success_response( + video=public_video_url, model=body.get("model") or resolved_model, prompt=prompt, modality=modality, + aspect_ratio=aspect_ratio, duration=video.get("duration") or duration, provider="xai", extra=extra, ) - return error_response( - error=message, - error_type=f"xai_{status}", - provider="xai", - model=resolved_model, - prompt=prompt, - ) - - -# --------------------------------------------------------------------------- -# Plugin entry point -# --------------------------------------------------------------------------- def register(ctx) -> None: