"""Gemini image-generation connector — direct API wrapper. Generates images through Google's gemini-2.5-flash-image model. The key comes from GEMINI_API_KEY in the environment only (endpoints.get_api_key). **Every generate call costs real money**; `health` only lists models and is free. Formerly tooling/db/image_connector.py (T-1290). Unchanged except that failures raise a ReachError instead of printing `{"ok": false}` and exiting 1, and a network failure is reported as the network rather than as the API. The key travels in the URL, so no error message ever includes the URL's query. """ from __future__ import annotations import base64 import json import os import urllib.request from tooling.core import console from tooling.core.errors import ReachError from tooling.domains.assets import endpoints SERVICE = "Gemini API" MODEL = "gemini-2.5-flash-image" API = "https://generativelanguage.googleapis.com/v1beta" DEFAULT_OUTPUT_DIR = os.path.expanduser("~/Pictures/mcp-images") # Real API values for imageConfig.aspectRatio (not a prompt hint). # https://ai.google.dev/gemini-api/docs/image-generation ASPECT_RATIOS = ("1:1", "3:2", "2:3", "3:4", "4:3", "4:5", "5:4", "9:16", "16:9", "21:9") MIME_TYPES = {".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".webp": "image/webp"} def get_api_key() -> str: return endpoints.get_api_key("GEMINI_API_KEY") def health() -> dict: """Is the Gemini API reachable with the configured key? (Free — lists models.)""" key = get_api_key() data = endpoints.call_json( urllib.request.Request(f"{API}/models?key={key}", method="GET"), service=SERVICE, what="the model listing", timeout=10, ) models = [ m.get("name", "") for m in data.get("models", []) if "imagen" in m.get("name", "").lower() or "flash" in m.get("name", "").lower() ] return {"ok": True, "api": "gemini", "image_capable_models": models[:5]} def default_output(prompt: str) -> str: safe = "".join(c if c.isalnum() or c in "-_ " else "" for c in prompt[:40]) safe = safe.strip().replace(" ", "_").lower() return os.path.join(DEFAULT_OUTPUT_DIR, f"{safe}.png") def build_request_body( prompt: str, aspect_ratio: str | None = "1:1", image_size: str | None = None, input_image: str | None = None, ) -> dict: """The generateContent payload — pure, so it can be tested without a call.""" parts = [] if input_image: if not os.path.isfile(input_image): raise ReachError( f"input image not found: {input_image}", fix="pass --input with an existing .png/.jpg/.webp", ) with open(input_image, "rb") as f: data = base64.b64encode(f.read()).decode("utf-8") mime = MIME_TYPES.get(os.path.splitext(input_image)[1].lower(), "image/png") parts.append({"inlineData": {"mimeType": mime, "data": data}}) # The size is a best-effort prompt hint only: this model has no resolution # parameter, unlike the aspect ratio below. parts.append({"text": prompt + (f" Resolution: {image_size}." if image_size else "")}) generation_config: dict = {"responseModalities": ["TEXT", "IMAGE"]} if aspect_ratio: generation_config["imageConfig"] = {"aspectRatio": aspect_ratio} return {"contents": [{"parts": parts}], "generationConfig": generation_config} def generate( prompt: str, output: str | None = None, aspect_ratio: str = "1:1", image_size: str | None = None, input_image: str | None = None, ) -> dict: """Generate one image. COSTS MONEY. Returns the result dict.""" key = get_api_key() body = build_request_body(prompt, aspect_ratio, image_size, input_image) output = output or default_output(prompt) console.event(f"Generating image: {prompt!r}" + (f" (from {input_image})" if input_image else "")) result = endpoints.call_json( urllib.request.Request( f"{API}/models/{MODEL}:generateContent?key={key}", data=json.dumps(body).encode(), headers={"Content-Type": "application/json"}, method="POST", ), service=SERVICE, what="the generation request", timeout=120, ) candidates = result.get("candidates", []) if not candidates: raise ReachError( f"Gemini returned no candidates: {json.dumps(result)[:500]}", fix="the prompt may have been blocked by safety filters — rephrase it and re-run", ) image_saved = False text_response = "" for candidate in candidates: for part in candidate.get("content", {}).get("parts", []): if "inlineData" in part: os.makedirs(os.path.dirname(os.path.abspath(output)), exist_ok=True) with open(output, "wb") as f: f.write(base64.b64decode(part["inlineData"]["data"])) image_saved = True elif "text" in part: text_response += part["text"] if not image_saved: raise ReachError( f"Gemini answered with text but no image: {text_response[:500]!r}", fix="make the prompt ask for an image explicitly, then re-run", ) return { "ok": True, "file": output, "size_bytes": os.path.getsize(output), "prompt": prompt, "aspect_ratio": aspect_ratio, }