diff --git a/docs/PROVIDERS.md b/docs/PROVIDERS.md index 7ad51bad..0afd5af1 100644 --- a/docs/PROVIDERS.md +++ b/docs/PROVIDERS.md @@ -393,7 +393,7 @@ allowance). OpenMontage estimates cost from the transcribed audio duration. See ### Google — TTS + Imagen + Music + Video (Shared Key) -> **One key, five tools.** Google Cloud TTS has 700+ voices in 50+ languages — the strongest localization option. Imagen 4 generates high-quality images. Google Lyria generates high-quality background music. Gemini Omni Flash supports conversational video editing, and direct Veo generation covers premium short video clips. +> **One key, five tools.** Google Cloud TTS has 700+ voices in 50+ languages — the strongest localization option. `google_imagen` supports both Imagen 4 and Gemini 2.5 Flash Image, including projects without Imagen catalog access. Google Lyria generates high-quality background music. Gemini Omni Flash supports conversational video editing, and direct Veo generation covers premium short video clips. **Tools unlocked:** `google_tts`, `google_imagen`, `google_music`, `gemini_omni_video`, `veo_video` **Env var:** `GOOGLE_API_KEY` (or `GEMINI_API_KEY` — either works; `GEMINI_API_KEY` takes precedence) @@ -434,9 +434,15 @@ The free tiers apply *independently* — you get 1M Standard AND 1M WaveNet AND | Imagen 4 Fast | $0.02 | | Imagen 4 Standard | $0.04 | | Imagen 4 Ultra | $0.06 | +| Gemini 2.5 Flash Image (`gemini-2.5-flash-image`) | $0.039 | **Free tier for Imagen:** None. Paid tier only. +To select the Gemini backend through the governed `image_selector`, pass +`preferred_provider: "google_imagen"` and +`model_name: "gemini-2.5-flash-image"`. The selector maps its neutral +`model_name` field to the provider's `model` input. + #### Gemini Omni Video Pricing | Model | Price | Notes | diff --git a/tests/tools/test_google_imagen_gemini_backend.py b/tests/tools/test_google_imagen_gemini_backend.py index 9b899b1d..fc7e960b 100644 --- a/tests/tools/test_google_imagen_gemini_backend.py +++ b/tests/tools/test_google_imagen_gemini_backend.py @@ -89,6 +89,31 @@ def test_gemini_model_routes_to_generate_content(imagen_tool, tmp_path): assert calls[0]["config"].image_config.aspect_ratio == "16:9" +def test_image_selector_maps_model_name_to_google_model( + imagen_tool, monkeypatch, tmp_path +): + """The governed selector must be able to reach the Gemini backend.""" + from tools.graphics.image_selector import ImageSelector + + tool, calls = imagen_tool + selector = ImageSelector() + monkeypatch.setattr(selector, "_providers", lambda: [tool]) + + result = selector.execute( + { + "prompt": "a flower", + "preferred_provider": "google_imagen", + "model_name": "gemini-2.5-flash-image", + "output_path": str(tmp_path / "selected.png"), + } + ) + + assert result.success, result.error + assert calls[0]["model"] == "gemini-2.5-flash-image" + assert result.data["selected_tool"] == "google_imagen" + assert result.data["model"] == "gemini-2.5-flash-image" + + def test_gemini_cost_estimate_is_per_image(): from tools.graphics.google_imagen import GoogleImagen diff --git a/tools/graphics/image_selector.py b/tools/graphics/image_selector.py index b106fbde..965679cc 100644 --- a/tools/graphics/image_selector.py +++ b/tools/graphics/image_selector.py @@ -242,6 +242,14 @@ class ImageSelector(BaseTool): props = tool.input_schema.get("properties", {}) if "query" in props and "query" not in adapted: adapted["query"] = adapted.get("prompt", "") + # The selector exposes a provider-neutral ``model_name`` field, + # while several providers call the same input ``model``. + if ( + "model_name" in adapted + and "model" in props + and "model" not in adapted + ): + adapted["model"] = adapted["model_name"] # Strip selector-only keys that downstream tools don't understand adapted.pop("preferred_provider", None)