fix(video_gen): make the duration enum explicit instead of sniffing tuple shape
`durations` carried two meanings told apart only by len==2 and gap>1: a (min, max) range to clamp, or an enum to snap. A family with exactly two legal values would have been misread as a range (review finding on #91311). `durations` is now always the (min, max) window (what capabilities()/ list_models() read) and families with discrete values add `duration_enum`; _clamp_duration takes the family and branches on the key, not the shape. Also: restore the exact v1.1 endpoint assertion in the gateway namespace test (a startswith/endswith check would not catch a silent version drift), add the ltx-2.5 i2v snap case, and keep happy-horse on audio_native (the schema test forbids audio+audio_native together, and v1.1 audio is always on).
This commit is contained in:
@@ -18,8 +18,9 @@ from agent.video_gen_provider import VideoGenProvider, error_response, success_r
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Family catalog. Capability flags gate which keys reach the payload — keys a family doesn't advertise are never sent (the
|
||||
# managed gateway forwards everything verbatim). Enums default to None (endpoint decides), flags to False. ``durations`` is an
|
||||
# enum tuple OR a ``(min, max)`` range (2 ints with gap > 1). Extras: audio_native (always on; description line only),
|
||||
# managed gateway forwards everything verbatim). Enums default to None (endpoint decides), flags to False. ``durations`` is always a
|
||||
# ``(min, max)`` range (clamp); a family whose endpoint only accepts discrete values adds ``duration_enum`` (snap to nearest; None →
|
||||
# first entry). Extras: audio_native (always on; description line only),
|
||||
# duration_int (JSON int, default queue-API string), duration_suffix ("4s"), image_param_key (i2v key when not `image_url`),
|
||||
# image_drop_keys (i2v endpoint rejects), audio_param_key (toggle key when not `generate_audio`), resolution_aliases (tool value → endpoint enum), static_payload (always required).
|
||||
def _family(display: str, speed: str, tier: str, strengths: str, text: Optional[str], image: str, **caps: Any) -> Dict[str, Any]:
|
||||
@@ -41,7 +42,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = {
|
||||
"ltx-2.5": _family("LTX 2.5", "~30-90s", "cheap", "Lightricks open-source audio-video model. Native audio, up to 20s / 4K (i2v), camera-motion presets.",
|
||||
"lightricks/ltx-2.5/text-to-video/fast", "lightricks/ltx-2.5/image-to-video/fast", duration_int=True, aspect_ratios=("16:9", "9:16"),
|
||||
resolutions=("720p", "1080p", "1440p", "2160p"), resolution_aliases={"2k": "1440p", "4k": "2160p"},
|
||||
durations=(6, 8, 10, 12, 14, 16, 18, 20), audio=True),
|
||||
durations=(6, 20), duration_enum=tuple(range(6, 21, 2)), audio=True),
|
||||
"pixverse-v6": _family("Pixverse v6", "~30-90s", "cheap", "Affordable. Negative prompts. 1-15s durations.", "fal-ai/pixverse/v6/text-to-video",
|
||||
"fal-ai/pixverse/v6/image-to-video", resolutions=("360p", "540p", "720p", "1080p"), durations=(1, 15), audio=True, negative=True, seed=True),
|
||||
"seedance-2.0-mini": _family("Seedance 2.0 Mini", "~30-90s", "cheap", "ByteDance. Faster/cheaper Seedance tier, audio + lip-sync, 4-15s.",
|
||||
@@ -49,7 +50,7 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = {
|
||||
resolutions=("480p", "720p"), durations=(4, 15), audio=True),
|
||||
# ─── Expensive / premium tier ──────────────────────────────────────
|
||||
"veo3.1": _family("Veo 3.1", "~60-120s", "premium", "Google DeepMind. Cinematic, native audio, strong prompt adherence.", "fal-ai/veo3.1",
|
||||
"fal-ai/veo3.1/image-to-video", aspect_ratios=("16:9", "9:16"), resolutions=("720p", "1080p", "4k"), durations=(4, 6, 8),
|
||||
"fal-ai/veo3.1/image-to-video", aspect_ratios=("16:9", "9:16"), resolutions=("720p", "1080p", "4k"), durations=(4, 8), duration_enum=(4, 6, 8),
|
||||
duration_suffix="s", audio=True, negative=True, seed=True), # wants "4s" not "4"
|
||||
"seedance-2.0": _family("Seedance 2.0", "~60-120s", "premium", "ByteDance. Cinematic, synchronized audio + lip-sync, 4-15s.", # no "auto" aspect, no `seed`
|
||||
"bytedance/seedance-2.0/text-to-video", "bytedance/seedance-2.0/image-to-video", aspect_ratios=_SIX_ASPECTS,
|
||||
@@ -120,12 +121,14 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = {
|
||||
DEFAULT_MODEL = "pixverse-v6" # cheap, both modalities, sane defaults
|
||||
|
||||
|
||||
def _clamp_duration(durations: Tuple[int, ...], duration: Optional[int]) -> Optional[int]:
|
||||
"""Clamp into a ``(min, max)`` range (None stays None: endpoint default) or snap to the nearest enum entry (None → first).
|
||||
Range heuristic: a 2-tuple of ints with a gap > 1."""
|
||||
if len(durations) == 2 and all(isinstance(d, int) for d in durations) and durations[1] - durations[0] > 1:
|
||||
return None if duration is None else max(durations[0], min(durations[1], duration))
|
||||
return durations[0] if duration is None else min(durations, key=lambda d: abs(d - duration))
|
||||
def _clamp_duration(family: Dict[str, Any], duration: Optional[int]) -> Optional[int]:
|
||||
"""Snap to the nearest ``duration_enum`` entry (None → first) when the family declares one, else clamp into the
|
||||
``durations`` ``(min, max)`` range (None stays None: endpoint default)."""
|
||||
enum = family.get("duration_enum")
|
||||
if enum:
|
||||
return enum[0] if duration is None else min(enum, key=lambda d: abs(d - duration))
|
||||
lo, hi = family["durations"]
|
||||
return None if duration is None else max(lo, min(hi, duration))
|
||||
|
||||
|
||||
def _modalities(meta: Dict[str, Any]) -> List[str]:
|
||||
@@ -171,7 +174,7 @@ def _build_payload(family: Dict[str, Any], *, prompt: str, image_url: Optional[s
|
||||
resolution: str, negative_prompt: Optional[str], audio: Optional[bool], seed: Optional[int]) -> Dict[str, Any]:
|
||||
"""Build a family-specific payload, dropping keys the family doesn't declare (unsupported enums → endpoint default)."""
|
||||
resolved = (family.get("resolution_aliases") or {}).get((resolution or "").lower(), resolution)
|
||||
clamped = _clamp_duration(family["durations"], duration) if family["durations"] else None
|
||||
clamped = _clamp_duration(family, duration) if family["durations"] else None
|
||||
payload: Dict[str, Any] = {key: value for ok, key, value in (
|
||||
(prompt, "prompt", prompt),
|
||||
(image_url, family.get("image_param_key") or "image_url", image_url),
|
||||
|
||||
@@ -661,6 +661,12 @@ class TestPayloadBuilder:
|
||||
}
|
||||
assert isinstance(p["duration"], int)
|
||||
|
||||
# i2v: the fast i2v endpoint takes the same even enum (19 snaps to 18 — a tie picks the lower entry, like veo 7→6),
|
||||
# no seed key.
|
||||
p = _build_payload(meta, prompt="animate", image_url="https://example.com/f.png", duration=19, aspect_ratio="9:16",
|
||||
resolution="720p", negative_prompt=None, audio=None, seed=7)
|
||||
assert p == {"prompt": "animate", "image_url": "https://example.com/f.png", "aspect_ratio": "9:16", "resolution": "720p", "duration": 18}
|
||||
|
||||
def test_kling_o3_payload(self):
|
||||
"""Kling O3: string duration, i2v drops aspect_ratio, no seed."""
|
||||
from plugins.video_gen.fal import FAL_FAMILIES, _build_payload
|
||||
|
||||
@@ -346,7 +346,5 @@ def test_video_gen_happy_horse_uses_alibaba_namespace():
|
||||
spec.loader.exec_module(plugin_mod)
|
||||
|
||||
hh = plugin_mod.FAL_FAMILIES["happy-horse"]
|
||||
assert hh["text_endpoint"].startswith("alibaba/happy-horse/")
|
||||
assert hh["image_endpoint"].startswith("alibaba/happy-horse/")
|
||||
assert hh["text_endpoint"].endswith("/text-to-video")
|
||||
assert hh["image_endpoint"].endswith("/image-to-video")
|
||||
assert hh["text_endpoint"] == "alibaba/happy-horse/v1.1/text-to-video"
|
||||
assert hh["image_endpoint"] == "alibaba/happy-horse/v1.1/image-to-video"
|
||||
|
||||
Reference in New Issue
Block a user