From ac0a8cd281faca3f727a6cc7a5c5c5405631cdbd Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 00:21:56 -0700 Subject: [PATCH] feat(image-gen): xAI Grok image catalog goes live-driven; grok-imagine-image-2.0 selectable - plugins/image_gen/xai: merge the live /v1/image-generation-models catalog (5-min cache, 10s timeout, static-table fallback when offline/unauth) into the picker so new xAI Imagine models appear automatically the day they launch, with generic metadata until curated text is added. - Add grok-imagine-image-2.0 to the curated static table (typography/ layout-aware model, API-available since Aug 8 2026). - Edits honor an explicitly selected image-input-capable model (e.g. grok-imagine-image-2.0) instead of always forcing grok-imagine-image-quality; quality remains the default edit baseline. - Tests: hermetic autouse fixture keeps unit runs offline; new coverage for live-merge, unknown-future-model selection, offline fallback, and edit-model resolution. Docs model table updated (en + zh-Hans). Live-verified: /image-generation-models returns grok-imagine-image, grok-imagine-image-2.0, grok-imagine-image-quality; real generation with 2.0 succeeded end to end. --- plugins/image_gen/xai/__init__.py | 142 ++++++++++++++++-- tests/plugins/image_gen/test_xai_provider.py | 79 ++++++++++ website/docs/guides/xai-grok-oauth.md | 1 + .../current/guides/xai-grok-oauth.md | 1 + 4 files changed, 212 insertions(+), 11 deletions(-) diff --git a/plugins/image_gen/xai/__init__.py b/plugins/image_gen/xai/__init__.py index 5ce9f26cb2..354e3addc7 100644 --- a/plugins/image_gen/xai/__init__.py +++ b/plugins/image_gen/xai/__init__.py @@ -55,6 +55,11 @@ _MODELS: Dict[str, Dict[str, Any]] = { "speed": "~5-10s", "strengths": "Fast, high-quality", }, + "grok-imagine-image-2.0": { + "display": "Grok Imagine Image 2.0", + "speed": "~10-20s", + "strengths": "Typography/layout-aware; legible small text; strongest quality.", + }, "grok-imagine-image-quality": { "display": "Grok Imagine Image (Quality)", "speed": "~10-20s", @@ -64,6 +69,95 @@ _MODELS: Dict[str, Dict[str, Any]] = { DEFAULT_MODEL = "grok-imagine-image" +# Live catalog cache: (models_dict, fetched_monotonic). xAI's +# ``/image-generation-models`` endpoint is the source of truth so newly +# released Imagine models appear in the picker without a code change; +# the static ``_MODELS`` table is the offline fallback and supplies curated +# speed/strengths text for the models we know about. +_LIVE_CACHE: Optional[Tuple[Dict[str, Dict[str, Any]], float]] = None +_LIVE_CACHE_TTL = 300.0 +_LIVE_TIMEOUT = 10.0 + + +def _fetch_live_models() -> Dict[str, Dict[str, Any]]: + """Fetch image models from xAI's ``/image-generation-models`` endpoint. + + Returns ``{model_id: {"input_modalities": [...], "aliases": [...]}}``. + Raises on any failure — callers treat that as "use the static table". + """ + creds = resolve_xai_http_credentials() + api_key = str(creds.get("api_key") or "").strip() + if not api_key: + raise RuntimeError("no xAI credentials") + base_url = str(creds.get("base_url") or "https://api.x.ai/v1").strip().rstrip("/") + response = requests.get( + f"{base_url}/image-generation-models", + headers={ + "Authorization": f"Bearer {api_key}", + "User-Agent": hermes_xai_user_agent(), + }, + timeout=_LIVE_TIMEOUT, + ) + response.raise_for_status() + payload = response.json() + entries = payload.get("models") or payload.get("data") or [] + out: Dict[str, Dict[str, Any]] = {} + for entry in entries: + if not isinstance(entry, dict): + continue + model_id = entry.get("id") or entry.get("name") + if not isinstance(model_id, str) or not model_id.strip(): + continue + out[model_id.strip()] = { + "input_modalities": entry.get("input_modalities") or [], + "aliases": entry.get("aliases") or [], + } + return out + + +def _live_models() -> Dict[str, Dict[str, Any]]: + """Cached live catalog (``{}`` when unreachable).""" + global _LIVE_CACHE + import time + + if _LIVE_CACHE is not None and time.monotonic() - _LIVE_CACHE[1] < _LIVE_CACHE_TTL: + return _LIVE_CACHE[0] + try: + live = _fetch_live_models() + except Exception as exc: # noqa: BLE001 - offline/unauth → static fallback + logger.debug("xAI live image model catalog unavailable: %s", exc) + live = {} + _LIVE_CACHE = (live, time.monotonic()) + return live + + +def _catalog() -> Dict[str, Dict[str, Any]]: + """Merged model catalog: live endpoint IDs + curated static metadata. + + Known models keep their curated display/speed/strengths; models xAI + ships after this file was written still show up (with generic metadata) + so users can pick them the day they launch. Static table alone when the + API is unreachable. + """ + live = _live_models() + if not live: + return dict(_MODELS) + merged: Dict[str, Dict[str, Any]] = {} + for model_id in live: + meta = _MODELS.get(model_id) + if meta is None: + meta = { + "display": model_id, + "speed": "", + "strengths": "New xAI Imagine model (from live xAI catalog)", + } + merged[model_id] = dict(meta) + merged[model_id]["input_modalities"] = live[model_id].get("input_modalities") or [] + # Keep curated entries that the live list may momentarily omit. + for model_id, meta in _MODELS.items(): + merged.setdefault(model_id, dict(meta)) + return merged + # xAI aspect ratios (more options than FAL/OpenAI) _XAI_ASPECT_RATIOS = { "landscape": "16:9", @@ -101,17 +195,41 @@ def _load_xai_config() -> Dict[str, Any]: def _resolve_model() -> Tuple[str, Dict[str, Any]]: - """Decide which model to use and return ``(model_id, meta)``.""" + """Decide which model to use and return ``(model_id, meta)``. + + Overrides are validated against the merged live+static catalog, so a + newly released xAI model can be selected the day it appears in the + live catalog — no code change required. + """ + catalog = _catalog() env_override = os.environ.get("XAI_IMAGE_MODEL") - if env_override and env_override in _MODELS: - return env_override, _MODELS[env_override] + if env_override and env_override in catalog: + return env_override, catalog[env_override] cfg = _load_xai_config() candidate = cfg.get("model") if isinstance(cfg.get("model"), str) else None - if candidate and candidate in _MODELS: - return candidate, _MODELS[candidate] + if candidate and candidate in catalog: + return candidate, catalog[candidate] - return DEFAULT_MODEL, _MODELS[DEFAULT_MODEL] + return DEFAULT_MODEL, catalog.get(DEFAULT_MODEL, _MODELS[DEFAULT_MODEL]) + + +def _resolve_edit_model() -> str: + """Model for ``/v1/images/edits`` requests. + + An explicitly selected model (env or config) that accepts image input + is honored for edits; otherwise fall back to the quality model, which + xAI documents as the edit-capable baseline. + """ + catalog = _catalog() + explicit = os.environ.get("XAI_IMAGE_MODEL") or ( + _load_xai_config().get("model") if isinstance(_load_xai_config().get("model"), str) else None + ) + if explicit and explicit in catalog: + modalities = catalog[explicit].get("input_modalities") or [] + if "image" in modalities: + return explicit + return "grok-imagine-image-quality" def _resolve_resolution() -> str: @@ -179,7 +297,7 @@ class XAIImageGenProvider(ImageGenProvider): "speed": meta.get("speed", ""), "strengths": meta.get("strengths", ""), } - for model_id, meta in _MODELS.items() + for model_id, meta in _catalog().items() ] def get_setup_schema(self) -> Dict[str, Any]: @@ -296,10 +414,12 @@ class XAIImageGenProvider(ImageGenProvider): storage_cfg = read_xai_imagine_storage_config("image_gen") if is_edit: - # Editing requires the quality model per xAI docs. The source - # image may be a public URL or a base64 data URI; local file paths - # are converted to a data URI here. - edit_model = "grok-imagine-image-quality" + # Editing needs an image-input-capable model. An explicit user + # selection that accepts image input (e.g. grok-imagine-image-2.0) + # is honored; otherwise the documented quality baseline is used. + # The source image may be a public URL or a base64 data URI; + # local file paths are converted to a data URI here. + edit_model = _resolve_edit_model() try: image_fields = [_xai_image_field(source) for source in source_images] except Exception as exc: diff --git a/tests/plugins/image_gen/test_xai_provider.py b/tests/plugins/image_gen/test_xai_provider.py index 8b55b647a6..99f33333f1 100644 --- a/tests/plugins/image_gen/test_xai_provider.py +++ b/tests/plugins/image_gen/test_xai_provider.py @@ -29,6 +29,25 @@ def _fake_api_key(monkeypatch, tmp_path): pass +@pytest.fixture(autouse=True) +def _no_live_catalog(monkeypatch): + """Keep unit tests hermetic: never hit xAI's live model-list endpoint. + + The fake XAI_API_KEY above would otherwise let ``_fetch_live_models`` + fire a real GET. Individual tests that exercise the live-merge path + re-patch ``_fetch_live_models`` themselves. + """ + import plugins.image_gen.xai as xai_mod + + def _offline(): + raise RuntimeError("offline (test)") + + monkeypatch.setattr(xai_mod, "_fetch_live_models", _offline) + monkeypatch.setattr(xai_mod, "_LIVE_CACHE", None) + yield + xai_mod._LIVE_CACHE = None + + # --------------------------------------------------------------------------- # Provider class tests # --------------------------------------------------------------------------- @@ -106,6 +125,66 @@ class TestConfig: assert model_id == "grok-imagine-image" +# --------------------------------------------------------------------------- +# Live catalog merge tests +# --------------------------------------------------------------------------- + + +class TestLiveCatalog: + def test_static_catalog_includes_image_2_0(self): + """Curated table carries the 2.0 model even offline.""" + from plugins.image_gen.xai import XAIImageGenProvider + + ids = [m["id"] for m in XAIImageGenProvider().list_models()] + assert "grok-imagine-image-2.0" in ids + + def test_unknown_live_model_appears_in_catalog(self, monkeypatch): + """A model xAI ships tomorrow shows up without a code change.""" + import plugins.image_gen.xai as xai_mod + + live = { + "grok-imagine-image": {"input_modalities": ["text", "image"], "aliases": []}, + "grok-imagine-image-3.0": {"input_modalities": ["text", "image"], "aliases": []}, + } + monkeypatch.setattr(xai_mod, "_fetch_live_models", lambda: live) + monkeypatch.setattr(xai_mod, "_LIVE_CACHE", None) + + catalog = xai_mod._catalog() + assert "grok-imagine-image-3.0" in catalog + # Curated metadata survives the merge for known models. + assert catalog["grok-imagine-image"]["display"] == "Grok Imagine Image" + # And the new model is selectable end to end. + monkeypatch.setenv("XAI_IMAGE_MODEL", "grok-imagine-image-3.0") + model_id, _ = xai_mod._resolve_model() + assert model_id == "grok-imagine-image-3.0" + + def test_live_failure_falls_back_to_static(self, monkeypatch): + import plugins.image_gen.xai as xai_mod + + monkeypatch.setattr(xai_mod, "_LIVE_CACHE", None) + catalog = xai_mod._catalog() # autouse fixture makes fetch raise + assert set(catalog) == set(xai_mod._MODELS) + + def test_edit_model_honors_image_capable_selection(self, monkeypatch): + import plugins.image_gen.xai as xai_mod + + live = { + "grok-imagine-image-2.0": {"input_modalities": ["text", "image"], "aliases": []}, + "grok-imagine-image-quality": {"input_modalities": ["text", "image"], "aliases": []}, + } + monkeypatch.setattr(xai_mod, "_fetch_live_models", lambda: live) + monkeypatch.setattr(xai_mod, "_LIVE_CACHE", None) + monkeypatch.setenv("XAI_IMAGE_MODEL", "grok-imagine-image-2.0") + assert xai_mod._resolve_edit_model() == "grok-imagine-image-2.0" + + def test_edit_model_defaults_to_quality(self, monkeypatch): + import plugins.image_gen.xai as xai_mod + + monkeypatch.setattr(xai_mod, "_LIVE_CACHE", None) + monkeypatch.delenv("XAI_IMAGE_MODEL", raising=False) + assert xai_mod._resolve_edit_model() == "grok-imagine-image-quality" + + # --------------------------------------------------------------------------- # Generate tests # --------------------------------------------------------------------------- diff --git a/website/docs/guides/xai-grok-oauth.md b/website/docs/guides/xai-grok-oauth.md index df21a21249..6645229a0c 100644 --- a/website/docs/guides/xai-grok-oauth.md +++ b/website/docs/guides/xai-grok-oauth.md @@ -161,6 +161,7 @@ The `x_search` toolset auto-enables whenever xAI credentials (a SuperGrok / X Pr | Chat | `grok-4.20-0309-non-reasoning` | Non-reasoning variant | | Chat | `grok-4.20-multi-agent-0309` | Multi-agent variant | | Image | `grok-imagine-image` | Default; ~5–10 s | +| Image | `grok-imagine-image-2.0` | Typography/layout-aware; strongest quality; ~10–20 s | | Image | `grok-imagine-image-quality` | Higher fidelity; ~10–20 s | | Video | `grok-imagine-video` | Text-to-video | | Video | `grok-imagine-video-1.5-preview` | Image-to-video; dated alias `grok-imagine-video-1.5-2026-05-30` | diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/guides/xai-grok-oauth.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/guides/xai-grok-oauth.md index 381917e9b9..2e2ace01d8 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/guides/xai-grok-oauth.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/guides/xai-grok-oauth.md @@ -161,6 +161,7 @@ hermes tools | 对话 | `grok-4.20-0309-non-reasoning` | 非推理变体 | | 对话 | `grok-4.20-multi-agent-0309` | 多 agent 变体 | | 图像 | `grok-imagine-image` | 默认;约 5–10 秒 | +| 图像 | `grok-imagine-image-2.0` | 版式/排版感知;最强画质;约 10–20 秒 | | 图像 | `grok-imagine-image-quality` | 更高保真度;约 10–20 秒 | | 视频 | `grok-imagine-video` | 文本转视频 | | 视频 | `grok-imagine-video-1.5-preview` | 图像转视频;日期别名 `grok-imagine-video-1.5-2026-05-30` |