refactor(video_generate): capability-gated dynamic schema (~814 → 458/377 tok/call) (#97095)

* refactor(video_generate): capability-gated dynamic schema — 6 optional args render only when the active provider/model honors them; fleet capability declarations + declaration<->implementation contract tests

* fix(video_gen): H3/Grok/Happy-Horse/Gemini audio is ALWAYS-ON native, not absent — new audio_native family key + audio_always_on capability surfaces as description line (maintainer catch)

* test(video_gen): duration-span test pins the active-model contract — resolved family's real window, short families not inflated, union fallback still spans 30s
This commit is contained in:
Teknium
2026-08-28 05:10:33 -07:00
committed by GitHub
parent 902cd05147
commit 536adb35c4
9 changed files with 499 additions and 139 deletions
+126 -117
View File
@@ -61,12 +61,13 @@ logger = logging.getLogger(__name__)
VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
"name": "video_generate",
# Placeholder — the real description is built dynamically at
# get_tool_definitions() time so it reflects the active backend's
# actual capabilities (which modalities / resolutions / duration
# ranges the user's currently-selected model supports).
# See _build_dynamic_video_schema() below and the dynamic-tool-schemas
# skill at github/hermes-agent-dev/references/dynamic-tool-schemas.md.
# Placeholder — description AND params are rebuilt dynamically at
# get_tool_definitions() time from the active provider's declared
# capabilities() and the active model's catalog entry. Optional args
# (image_url, reference_image_urls, negative_prompt, audio, seed,
# upscale) are advertised ONLY when the active backend/model honors
# them; the handler accepts them regardless (replay compat — providers
# clamp/ignore). See _build_dynamic_video_schema().
"description": "(rebuilt at get_definitions() time — see _build_dynamic_video_schema)",
"parameters": {
"type": "object",
@@ -78,93 +79,36 @@ VIDEO_GENERATE_SCHEMA: Dict[str, Any] = {
"subject, style, camera movement, etc."
),
},
"image_url": {
"type": "string",
"description": (
"Optional public HTTPS URL of a still image. When provided, "
"the active backend routes to its image-to-video "
"endpoint (animate the image); when omitted, it routes "
"to text-to-video. For xAI chaining, use the `image` or "
"`public_url` HTTPS URL from a prior Imagine result."
),
},
"reference_image_urls": {
"type": "array",
"items": {"type": "string"},
"description": (
"Optional list of public HTTPS reference image URLs "
"(style or character refs). For xAI chaining, use "
"`image` or `public_url` from prior Imagine results."
),
},
"duration": {
"type": "integer",
"description": (
"Desired video duration in seconds. Providers clamp to "
"their supported range (commonly 4-15s). Omit to use the "
"provider's default."
"their supported range. Omit for the provider default."
),
},
"aspect_ratio": {
"type": "string",
"enum": list(COMMON_ASPECT_RATIOS),
"description": (
"Output aspect ratio. Providers clamp to their supported "
"set."
),
"description": "Output aspect ratio.",
"default": DEFAULT_ASPECT_RATIO,
},
"resolution": {
"type": "string",
"enum": list(COMMON_RESOLUTIONS),
"description": (
"Output resolution. Providers clamp to their supported "
"set."
),
"description": "Output resolution.",
"default": DEFAULT_RESOLUTION,
},
"negative_prompt": {
"type": "string",
"description": (
"Optional negative prompt — content to avoid in the "
"output. Supported by Pixverse, Kling, and similar; "
"ignored by providers that do not support it."
),
},
"audio": {
"type": "boolean",
"description": (
"Optional audio generation toggle. Supported by Veo3 and "
"Pixverse (affects pricing tier); ignored elsewhere."
),
},
"seed": {
"type": "integer",
"description": (
"Optional seed for reproducible outputs (provider-"
"dependent)."
),
},
"upscale": {
"type": "boolean",
"description": (
"Optional high-resolution pass: when true, the generated "
"video is run through the active backend's video upscaler "
"(extra cost and latency, roughly 2x resolution). Use when "
"the user asks for high-res / 4K output. Omit for the "
"model's native resolution. Ignored by backends without "
"an upscaler."
),
},
"model": {
"type": "string",
"description": (
"Optional model override. If omitted, the user's "
"configured ``video_gen.model`` (set via `hermes tools` "
"→ Video Generation) is used. Models that the active "
"provider does not know are rejected."
"Optional model override; defaults to the configured "
"``video_gen.model``. Unknown models are rejected."
),
},
# NOTE (schema diet, #95681): image_url / reference_image_urls /
# negative_prompt / audio / seed / upscale are added
# per-capability by _build_dynamic_video_schema. Do not re-add
# them statically.
},
"required": ["prompt"],
},
@@ -487,22 +431,18 @@ def _format_model_caveats(
def _build_dynamic_video_schema() -> Dict[str, Any]:
"""Build a description that reflects the active backend's actual surface.
"""Render description AND params from the active backend's declared surface.
Cheap: reads config (already memoized by the caller), asks the active
provider for `capabilities()` and the active model's catalog entry,
and formats a few lines of prose. Falls back to the generic
description when no provider is configured or registered.
Optional args are advertised only when the resolved provider/model
honors them (capabilities() + the model's catalog entry — coverage is
contract-tested per provider); enums and duration bounds tighten to
the active model's actual sets. The handler still accepts unadvertised
args (replay compat): providers clamp or ignore, as before.
"""
static_props = VIDEO_GENERATE_SCHEMA["parameters"]["properties"]
parts: List[str] = [_GENERIC_DESCRIPTION]
configured_model = _read_configured_video_model()
# Reflect the *resolved* active provider (same resolution the handler uses
# in _resolve_active_provider): an explicit ``video_gen.provider``, or —
# when unset — the single available registered backend. Keeping the
# description in sync with execution stops the agent from being told
# "no backend configured" while a call would actually succeed.
provider = _resolve_active_provider()
if provider is None:
@@ -510,7 +450,14 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
"\nNo video backend is available. Calls will return an error "
"until the user picks one via `hermes tools` → Video Generation."
)
return {"description": "\n".join(parts)}
return {
"description": "\n".join(parts),
"parameters": {
"type": "object",
"properties": {"prompt": static_props["prompt"]},
"required": ["prompt"],
},
}
try:
caps = provider.capabilities() or {}
@@ -527,47 +474,22 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
{},
)
backend_label = provider.display_name
line = f"\nActive backend: {backend_label}"
if active_model:
line += f" · model: {active_model}"
parts.append(line)
# Model-specific caveats (the high-signal stuff)
# ---- description -------------------------------------------------
for c in _format_model_caveats(model_meta, caps):
parts.append(f"- {c}")
# Prefer the active model's modalities over the backend union. An
# i2v-only family on a dual-modality backend (e.g. gemini-omni-flash
# on FAL) must not also claim text-to-video support.
model_modalities = set(model_meta.get("modalities") or [])
modality = model_meta.get("modality")
if modality:
model_modalities.add(modality)
effective_modalities = model_modalities or set(caps.get("modalities") or [])
if "text" in effective_modalities and "image" in effective_modalities:
parts.append(
"- supports both text-to-video (omit image_url) and "
"image-to-video (pass image_url) — routes automatically"
)
can_i2v = "image" in effective_modalities
t2v = "text" in effective_modalities
if can_i2v and not t2v:
parts.append("- image-to-video only: image_url is REQUIRED")
elif not can_i2v:
parts.append("- text-to-video only (no image input)")
if caps.get("aspect_ratios"):
parts.append(f"- aspect_ratio choices: {', '.join(caps['aspect_ratios'])}")
if caps.get("resolutions"):
parts.append(f"- resolution choices: {', '.join(caps['resolutions'])}")
min_duration = model_meta.get("min_duration", caps.get("min_duration"))
max_duration = model_meta.get("max_duration", caps.get("max_duration"))
if min_duration and max_duration:
parts.append(
f"- duration range: {min_duration}-{max_duration}s"
)
if caps.get("supports_audio"):
parts.append("- audio: pass `audio=true` to enable native audio (pricing tier)")
if caps.get("supports_negative_prompt"):
parts.append("- negative_prompt: supported")
max_refs = caps.get("max_reference_images") or 0
if max_refs:
parts.append(f"- reference_image_urls: up to {max_refs} images")
if provider.name == "xai":
parts.append(
"- chaining: for edit/extend pass the public HTTPS MP4 in `video` "
@@ -584,7 +506,94 @@ def _build_dynamic_video_schema() -> Dict[str, Any]:
if notice:
parts.append(f"- storage: {notice}")
return {"description": "\n".join(parts)}
# ---- params ------------------------------------------------------
properties: Dict[str, Any] = {"prompt": static_props["prompt"]}
if can_i2v:
properties["image_url"] = {
"type": "string",
"description": (
"Public HTTPS URL of a still image to animate "
"(image-to-video). Omit for text-to-video."
),
}
max_refs = int(caps.get("max_reference_images") or 0)
if max_refs > 0:
properties["reference_image_urls"] = {
"type": "array",
"items": {"type": "string"},
"maxItems": max_refs,
"description": (
f"Up to {max_refs} public HTTPS reference image URLs "
"(style or character refs)."
),
}
min_duration = model_meta.get("min_duration", caps.get("min_duration"))
max_duration = model_meta.get("max_duration", caps.get("max_duration"))
duration_param = dict(static_props["duration"])
if min_duration and max_duration:
duration_param["minimum"] = int(min_duration)
duration_param["maximum"] = int(max_duration)
duration_param["description"] = (
f"Video duration in seconds ({min_duration}-{max_duration}). "
"Omit for the provider default."
)
properties["duration"] = duration_param
# Tighten enums to the active backend's actual sets when declared.
aspect_param = dict(static_props["aspect_ratio"])
if caps.get("aspect_ratios"):
aspect_param["enum"] = list(caps["aspect_ratios"])
properties["aspect_ratio"] = aspect_param
resolution_param = dict(static_props["resolution"])
if caps.get("resolutions"):
resolution_param["enum"] = list(caps["resolutions"])
properties["resolution"] = resolution_param
if caps.get("supports_negative_prompt"):
properties["negative_prompt"] = {
"type": "string",
"description": "Content to avoid in the output.",
}
if caps.get("supports_audio"):
properties["audio"] = {
"type": "boolean",
"description": (
"Enable native audio generation (affects pricing tier)."
),
}
elif caps.get("audio_always_on"):
parts.append(
"- audio: native stereo audio is generated with every video "
"(always on; no toggle) — describe the desired sound in the "
"prompt"
)
if caps.get("supports_seed"):
properties["seed"] = {
"type": "integer",
"description": "Seed for reproducible outputs.",
}
if caps.get("supports_upscale"):
properties["upscale"] = {
"type": "boolean",
"description": (
"High-resolution pass via the backend's video upscaler "
"(~2x, extra cost/latency). Omit for native resolution."
),
}
properties["model"] = static_props["model"]
return {
"description": "\n".join(parts),
"parameters": {
"type": "object",
"properties": properties,
"required": ["prompt"],
},
}
# ---------------------------------------------------------------------------