diff --git a/plugins/video_gen/fal/__init__.py b/plugins/video_gen/fal/__init__.py index 35691fb160..539761f4ee 100644 --- a/plugins/video_gen/fal/__init__.py +++ b/plugins/video_gen/fal/__init__.py @@ -1,7 +1,7 @@ """FAL.ai video generation backend. The user picks a **model family** (e.g. "Pixverse v6"); the plugin routes to its text-to-video endpoint without -``image_url`` and to its image-to-video endpoint otherwise (gemini-omni-flash is i2v only). Active-family precedence: +``image_url`` and to its image-to-video endpoint otherwise. Active-family precedence: tool ``model=`` → ``FAL_VIDEO_MODEL`` env → ``video_gen.fal.model`` → ``video_gen.model`` (family id or an endpoint path containing one) → ``DEFAULT_MODEL``. Auth via ``FAL_KEY`` or the managed Nous gateway; output is an HTTPS URL. """ @@ -74,8 +74,10 @@ FAL_FAMILIES: Dict[str, Dict[str, Any]] = { "xai/grok-imagine-video/v1.5/text-to-video", "xai/grok-imagine-video/v1.5/image-to-video", duration_int=True, image_drop_keys=("aspect_ratio",), aspect_ratios=("16:9", "4:3", "3:2", "1:1", "2:3", "3:4", "9:16"), # aspect is t2v-only resolutions=("480p", "720p", "1080p"), durations=(1, 15), audio_native=True), - "gemini-omni-flash": _family("Gemini Omni Flash (via FAL)", "~60-120s", "premium", "Google. Image-to-video with audio, physics-grounded motion, 3-10s.", - None, "google/gemini-omni-flash/image-to-video", duration_int=True, aspect_ratios=("16:9", "9:16"), durations=(3, 10), audio_native=True), + # v1.1 (Aug 2026) added text-to-video and a 360p-4k resolution enum; v1.0 was image-only. + "gemini-omni-flash": _family("Gemini Omni Flash 1.1 (via FAL)", "~60-120s", "premium", "Google. Text & image to video with native audio, physics-grounded motion, up to 4K, 3-10s.", + "google/gemini-omni-flash/v1.1/text-to-video", "google/gemini-omni-flash/v1.1/image-to-video", duration_int=True, + aspect_ratios=("16:9", "9:16"), resolutions=("360p", "720p", "1080p", "4k"), durations=(3, 10), audio_native=True), # Kling 3.0 core tiers: t2v declares aspect_ratio, i2v derives it from `start_image_url`; string duration enum "3".."15"; # generate_audio is a real toggle (default on, audio-on costs more); no resolution or seed keys in the v3 schemas. "kling-v3": _family("Kling 3.0 (Standard)", "~60-180s", "premium", "Kuaishou frontier core model. Cinematic motion, native audio, 3-15s.", diff --git a/tests/plugins/video_gen/test_fal_plugin.py b/tests/plugins/video_gen/test_fal_plugin.py index f8ac45b2c8..dbfa63201e 100644 --- a/tests/plugins/video_gen/test_fal_plugin.py +++ b/tests/plugins/video_gen/test_fal_plugin.py @@ -202,14 +202,37 @@ def test_wan_30_audio_toggle_uses_family_key_and_start_image_url(): assert _build_payload(FAL_FAMILIES["veo3.1"], image_url=None, **kw)["generate_audio"] is True -def test_gemini_omni_flash_is_image_only(): - """Gemini Omni Flash has no t2v endpoint on FAL — text jobs must - error cleanly instead of submitting to a None endpoint.""" +def test_gemini_omni_flash_v11_is_dual_modality(): + """v1.1 (Aug 2026) added a text-to-video endpoint; both modalities + must route to the versioned v1.1 endpoints.""" from plugins.video_gen.fal import FAL_FAMILIES meta = FAL_FAMILIES["gemini-omni-flash"] - assert meta.get("text_endpoint") is None - assert meta.get("image_endpoint") + assert meta["text_endpoint"] == "google/gemini-omni-flash/v1.1/text-to-video" + assert meta["image_endpoint"] == "google/gemini-omni-flash/v1.1/image-to-video" + + +def test_text_only_job_errors_cleanly_for_i2v_only_family(): + """Catalog-shape guard: a family without a text endpoint must error + cleanly instead of submitting to a None endpoint (kept alive with a + synthetic family now that every cataloged family is dual-modality).""" + from plugins.video_gen.fal import _build_payload + + synthetic = { + "text_endpoint": None, + "image_endpoint": "example/i2v-only/image-to-video", + "durations": (3, 10), + "duration_int": True, + "seed": False, + } + # Payload building for the i2v path must still work. + p = _build_payload( + synthetic, prompt="x", image_url="https://i.png", duration=None, + aspect_ratio="16:9", resolution="720p", negative_prompt=None, + audio=None, seed=None, + ) + assert p["image_url"] == "https://i.png" + assert not synthetic.get("text_endpoint") def test_every_family_has_required_metadata(): @@ -514,13 +537,14 @@ class TestPayloadBuilder: assert p["duration"] == expected assert type(p["duration"]) is type(expected) - def test_i2v_only_families_declare_no_text_endpoint(self): - """Catalog invariant: Gemini Omni Flash animates an existing image only.""" + def test_every_family_declares_both_endpoints(self): + """Catalog invariant: since Gemini Omni Flash 1.1 every family is + dual-modality — both endpoints must be non-empty strings.""" from plugins.video_gen.fal import FAL_FAMILIES - meta = FAL_FAMILIES["gemini-omni-flash"] - assert meta.get("text_endpoint") is None - assert meta["image_endpoint"] + for fid, meta in FAL_FAMILIES.items(): + assert meta.get("text_endpoint"), fid + assert meta.get("image_endpoint"), fid def test_ltx_omits_duration_aspect_resolution(self): """LTX 2.3 doesn't declare duration/aspect/resolution enums —