tooling/db/ (a misnamed directory: connectors, not database work),
trellis-batch.sh and synth_ui_sounds.py become `reach assets`:
audio {health,generate,batch,post {convert,normalize,trim,pipeline}},
image {health,generate}, trellis {health,generate,batch}, and synth-ui.
The four audio bash wrappers are retired, and tooling/db/ is gone.
Parity, from baselines taken before anything moved:
- the four UI-sound WAVs and the harmonic-synth WAVs (exponential and linear
decay) are byte-identical
- the ffmpeg pipeline's decoded PCM is identical. Its .ogg bytes are not,
even between two runs of the OLD code: Ogg picks a random stream serial,
so the encoded file was never the right thing to compare
- the network success paths can't be run in a gate (Stable Audio and Trellis
are kept off, Gemini costs money), so tooling/test_assets.py stands up a
fake Gradio and pins every payload: the audio submit, Trellis's six-call
session sequence with its 9-input image_to_3d, and the Gemini body. It
failed when one Trellis value was mutated (7.5 → 7.0)
Failure classification, in endpoints.py, is the point of the port. The
services are OFF by design (VRAM on tower-of-joy, D-17), and the topology doc
warns against "fixing" one by restarting it. So a refused connection says OFF
and asks for the service to be turned on rather than restarted; a 4xx/5xx says
the request was rejected; 401/403 says credentials; 429 says quota; and an
unreachable Gemini blames the network, not VRAM.
Behaviour changes, each a failure that used to read as success or crash:
- audio batch and trellis batch exited 0 with failures in their summaries;
they now print the summary and exit 1
- trellis generate on a missing image crashed with a TypeError
(print(..., indent=2)); it now names the file, and checks it before the
service so a typo is not reported as an outage
- the ffmpeg pipeline left its intermediates behind when a step failed
Structure: the connectors called each other as subprocesses (batch spawned
the connector, which spawned audio_post) and parsed each other's stdout. They
are now function calls, and ffmpeg is the only exec, through core/process.
ensure_venv() is removed: it os.execv'd into .venv, which D-263's exec rule
forbids, and reach declares the dependencies itself. config.json moved into
the domain deliberately, and the local-services rule follows it.
Output contract: results are still JSON on stdout with the same keys, so skill
readers keep working. Failures are an exit status with a Fix line, never
{"ok": false}. The audio-gen, glb-gen and image-gen skills, Araminta's agent
file and the allow-list are updated to match. glb-gen's "trellis-batch.sh is
hardcoded to one category" caveat is gone: batch takes --input-dir or --names.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
150 lines
5.3 KiB
Python
Executable File
150 lines
5.3 KiB
Python
Executable File
"""Gemini image-generation connector — direct API wrapper.
|
|
|
|
Generates images through Google's gemini-2.5-flash-image model. The key comes
|
|
from GEMINI_API_KEY in the environment only (endpoints.get_api_key). **Every
|
|
generate call costs real money**; `health` only lists models and is free.
|
|
|
|
Formerly tooling/db/image_connector.py (T-1290). Unchanged except that
|
|
failures raise a ReachError instead of printing `{"ok": false}` and exiting 1,
|
|
and a network failure is reported as the network rather than as the API.
|
|
The key travels in the URL, so no error message ever includes the URL's query.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import json
|
|
import os
|
|
import urllib.request
|
|
|
|
from tooling.core import console
|
|
from tooling.core.errors import ReachError
|
|
from tooling.domains.assets import endpoints
|
|
|
|
SERVICE = "Gemini API"
|
|
MODEL = "gemini-2.5-flash-image"
|
|
API = "https://generativelanguage.googleapis.com/v1beta"
|
|
DEFAULT_OUTPUT_DIR = os.path.expanduser("~/Pictures/mcp-images")
|
|
|
|
# Real API values for imageConfig.aspectRatio (not a prompt hint).
|
|
# https://ai.google.dev/gemini-api/docs/image-generation
|
|
ASPECT_RATIOS = ("1:1", "3:2", "2:3", "3:4", "4:3", "4:5", "5:4", "9:16", "16:9", "21:9")
|
|
|
|
MIME_TYPES = {".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg", ".webp": "image/webp"}
|
|
|
|
|
|
def get_api_key() -> str:
|
|
return endpoints.get_api_key("GEMINI_API_KEY")
|
|
|
|
|
|
def health() -> dict:
|
|
"""Is the Gemini API reachable with the configured key? (Free — lists models.)"""
|
|
key = get_api_key()
|
|
data = endpoints.call_json(
|
|
urllib.request.Request(f"{API}/models?key={key}", method="GET"),
|
|
service=SERVICE,
|
|
what="the model listing",
|
|
timeout=10,
|
|
)
|
|
models = [
|
|
m.get("name", "")
|
|
for m in data.get("models", [])
|
|
if "imagen" in m.get("name", "").lower() or "flash" in m.get("name", "").lower()
|
|
]
|
|
return {"ok": True, "api": "gemini", "image_capable_models": models[:5]}
|
|
|
|
|
|
def default_output(prompt: str) -> str:
|
|
safe = "".join(c if c.isalnum() or c in "-_ " else "" for c in prompt[:40])
|
|
safe = safe.strip().replace(" ", "_").lower()
|
|
return os.path.join(DEFAULT_OUTPUT_DIR, f"{safe}.png")
|
|
|
|
|
|
def build_request_body(
|
|
prompt: str,
|
|
aspect_ratio: str | None = "1:1",
|
|
image_size: str | None = None,
|
|
input_image: str | None = None,
|
|
) -> dict:
|
|
"""The generateContent payload — pure, so it can be tested without a call."""
|
|
parts = []
|
|
if input_image:
|
|
if not os.path.isfile(input_image):
|
|
raise ReachError(
|
|
f"input image not found: {input_image}",
|
|
fix="pass --input with an existing .png/.jpg/.webp",
|
|
)
|
|
with open(input_image, "rb") as f:
|
|
data = base64.b64encode(f.read()).decode("utf-8")
|
|
mime = MIME_TYPES.get(os.path.splitext(input_image)[1].lower(), "image/png")
|
|
parts.append({"inlineData": {"mimeType": mime, "data": data}})
|
|
|
|
# The size is a best-effort prompt hint only: this model has no resolution
|
|
# parameter, unlike the aspect ratio below.
|
|
parts.append({"text": prompt + (f" Resolution: {image_size}." if image_size else "")})
|
|
|
|
generation_config: dict = {"responseModalities": ["TEXT", "IMAGE"]}
|
|
if aspect_ratio:
|
|
generation_config["imageConfig"] = {"aspectRatio": aspect_ratio}
|
|
|
|
return {"contents": [{"parts": parts}], "generationConfig": generation_config}
|
|
|
|
|
|
def generate(
|
|
prompt: str,
|
|
output: str | None = None,
|
|
aspect_ratio: str = "1:1",
|
|
image_size: str | None = None,
|
|
input_image: str | None = None,
|
|
) -> dict:
|
|
"""Generate one image. COSTS MONEY. Returns the result dict."""
|
|
key = get_api_key()
|
|
body = build_request_body(prompt, aspect_ratio, image_size, input_image)
|
|
output = output or default_output(prompt)
|
|
|
|
console.event(f"Generating image: {prompt!r}" + (f" (from {input_image})" if input_image else ""))
|
|
result = endpoints.call_json(
|
|
urllib.request.Request(
|
|
f"{API}/models/{MODEL}:generateContent?key={key}",
|
|
data=json.dumps(body).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
),
|
|
service=SERVICE,
|
|
what="the generation request",
|
|
timeout=120,
|
|
)
|
|
|
|
candidates = result.get("candidates", [])
|
|
if not candidates:
|
|
raise ReachError(
|
|
f"Gemini returned no candidates: {json.dumps(result)[:500]}",
|
|
fix="the prompt may have been blocked by safety filters — rephrase it and re-run",
|
|
)
|
|
|
|
image_saved = False
|
|
text_response = ""
|
|
for candidate in candidates:
|
|
for part in candidate.get("content", {}).get("parts", []):
|
|
if "inlineData" in part:
|
|
os.makedirs(os.path.dirname(os.path.abspath(output)), exist_ok=True)
|
|
with open(output, "wb") as f:
|
|
f.write(base64.b64decode(part["inlineData"]["data"]))
|
|
image_saved = True
|
|
elif "text" in part:
|
|
text_response += part["text"]
|
|
|
|
if not image_saved:
|
|
raise ReachError(
|
|
f"Gemini answered with text but no image: {text_response[:500]!r}",
|
|
fix="make the prompt ask for an image explicitly, then re-run",
|
|
)
|
|
|
|
return {
|
|
"ok": True,
|
|
"file": output,
|
|
"size_bytes": os.path.getsize(output),
|
|
"prompt": prompt,
|
|
"aspect_ratio": aspect_ratio,
|
|
}
|