feat: add FAL GPT Image 2.5 generation and editing selections

This commit is contained in:
Teknium
2026-09-08 13:34:50 -07:00
parent 7777f8c350
commit b1f003e186
3 changed files with 82 additions and 3 deletions

View File

@@ -30,6 +30,29 @@ def image_tool():
# Catalog integrity
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("variant", ["flare", "sunburst"])
@pytest.mark.parametrize("aspect,size", [
("landscape", "landscape_4_3"), ("square", "square_hd"), ("portrait", "portrait_4_3"),
])
def test_image_25_selection_routes_generation_and_edits(image_tool, monkeypatch, variant, aspect, size):
model = f"openai/gpt-image-2.5/{variant}/text-to-image"
monkeypatch.setenv("FAL_IMAGE_MODEL", model)
monkeypatch.setenv("FAL_KEY", "test-key")
selected, meta = image_tool._resolve_fal_model()
assert selected == model
refs = [f"https://example.com/{i}.png" for i in range(17)]
for sources, endpoint in (([], model), (refs, f"openai/gpt-image-2.5/{variant}/edit")):
actual, payload = image_tool._prepare_fal_request(
selected, meta, "a cup", aspect, 42, {"guidance_scale": 9}, sources,
)
assert actual == endpoint
assert payload["quality"] == "medium"
assert payload["image_size"] == size
assert "seed" not in payload and "guidance_scale" not in payload
assert payload.get("image_urls", []) == sources[:16]
assert meta["upscale"] is False
class TestFalCatalog:
"""Every FAL_MODELS entry must have a consistent shape."""

View File

@@ -153,6 +153,31 @@ FAL_MODELS: Dict[str, Dict[str, Any]] = {
},
max_reference_images=16,
),
# Same minimum pixel count as GPT Image 2; keep medium quality explicit
# rather than inheriting FAL's higher-cost high default.
**{
f"openai/gpt-image-2.5/{variant}/text-to-image": _model(
f"GPT Image 2.5 {variant.title()}", speed, strengths, "Token-based pricing",
sizes={
"landscape": "landscape_4_3", "square": "square_hd", "portrait": "portrait_4_3",
},
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
supports={
"prompt", "image_size", "quality", "num_images", "output_format", "background",
"output_compression", "sync_mode",
},
edit_endpoint=f"openai/gpt-image-2.5/{variant}/edit",
edit_supports={
"prompt", "image_urls", "image_size", "quality", "num_images", "output_format",
"background", "output_compression", "sync_mode", "mask_url", "input_fidelity",
},
max_reference_images=16,
)
for variant, speed, strengths in (
("flare", "Fast", "Everyday creation, natural lighting and textures"),
("sunburst", "Slower", "Precision editing, subject and composition consistency"),
)
},
"fal-ai/ideogram/v3": _model(
"Ideogram V3", "~5s", "Best typography", "$0.03-0.09/image",
defaults={"rendering_speed": "BALANCED", "expand_prompt": True, "style": "AUTO"},

View File

@@ -126,6 +126,37 @@ Auth reuses the same env vars as the Meta chat provider — `MODEL_API_KEY`
as aliases. Set `META_BASE_URL` to point at a proxy or alternate host. Text-to-image
only for now; responses are saved to `$HERMES_HOME/cache/images/`.
## FAL: GPT Image 2.5
Select **GPT Image 2.5 Flare** or **GPT Image 2.5 Sunburst** under
`hermes tools` → Image Generation → FAL.ai. The model IDs are:
- `openai/gpt-image-2.5/flare/text-to-image`
- `openai/gpt-image-2.5/sunburst/text-to-image`
For example:
```bash
hermes config set image_gen.provider fal
hermes config set image_gen.model openai/gpt-image-2.5/flare/text-to-image
```
Providing `image_url` or reference images automatically selects the corresponding
`openai/gpt-image-2.5/flare/edit` or `openai/gpt-image-2.5/sunburst/edit` endpoint.
Both accept up to 16 source images. Hermes pins quality to `medium`, matching its
existing FAL GPT Image policy rather than FAL's higher-cost `high` default.
Landscape and portrait use 4:3 presets to satisfy the minimum pixel count;
square uses `square_hd`. Upscaling remains off unless requested.
FAL bills by tokens, not a fixed image price: $5/M text input, $1.25/M cached
text input, $10/M text output, $8/M image input, $2/M cached image input, and
$30/M image output, rounded up to $0.0001 per request. See the
[Flare](https://fal.ai/models/openai/gpt-image-2.5/flare/text-to-image) and
[Sunburst](https://fal.ai/models/openai/gpt-image-2.5/sunburst/text-to-image)
pages. Direct FAL requires a funded `FAL_KEY`; managed-gateway availability
depends on that gateway's endpoint allowlist and is not implied by FAL availability.
Existing provider and model defaults are unchanged.
## OpenAI API: GPT Image 2.5
The **OpenAI** provider supports GPT Image 2.5 Flare (fast everyday creation)
@@ -154,7 +185,7 @@ does not estimate 2.5 token consumption. See the official
The **OpenAI (Codex auth)** provider remains separate: its backend can accept
an image-model value without honoring that selection, so a successful image
alone does not verify Flare or Sunburst routing. These selections are offered
only through the direct OpenAI API provider, not Codex auth or FAL.
through the direct OpenAI API provider and FAL, not as verified Codex-auth selections.
## Usage
@@ -196,7 +227,7 @@ Two inputs drive the edit:
| Backend | Image-to-image | Reference cap | How |
|---|---|---|---|
| **FAL.ai** (edit-capable models below) | ✓ | up to 9 | routes to the model's `/edit` endpoint |
| **FAL.ai** (edit-capable models below) | ✓ | up to 16 (per model) | routes to the model's `/edit` endpoint |
| **OpenAI** (GPT Image 2 / 2.5 Flare / Sunburst) | ✓ | up to 16 | `images.edit()` |
| **xAI** (Grok Imagine) | ✓ | 1 | `/v1/images/edits` (`grok-imagine-image-quality`) |
| **Krea** (`Krea 2`) | ✓ | up to 10 | reference-guided generation (`image_style_references`) |
@@ -205,7 +236,7 @@ Two inputs drive the edit:
FAL models with an editing endpoint: `flux-2/klein/9b`, `flux-2-pro`,
`nano-banana-pro`, `gpt-image-1.5`, `gpt-image-2`, `ideogram/v3`, and
`qwen-image`. Pure text-to-image FAL models (`z-image/turbo`, `recraft`,
`qwen-image`, plus GPT Image 2.5 Flare and Sunburst above. Pure text-to-image FAL models (`z-image/turbo`, `recraft`,
`krea/*`) reject image inputs with a clear error pointing you at an
edit-capable model.