feat: add FAL GPT Image 2.5 generation and editing selections
This commit is contained in:
@@ -30,6 +30,29 @@ def image_tool():
|
||||
# Catalog integrity
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("variant", ["flare", "sunburst"])
|
||||
@pytest.mark.parametrize("aspect,size", [
|
||||
("landscape", "landscape_4_3"), ("square", "square_hd"), ("portrait", "portrait_4_3"),
|
||||
])
|
||||
def test_image_25_selection_routes_generation_and_edits(image_tool, monkeypatch, variant, aspect, size):
|
||||
model = f"openai/gpt-image-2.5/{variant}/text-to-image"
|
||||
monkeypatch.setenv("FAL_IMAGE_MODEL", model)
|
||||
monkeypatch.setenv("FAL_KEY", "test-key")
|
||||
selected, meta = image_tool._resolve_fal_model()
|
||||
assert selected == model
|
||||
refs = [f"https://example.com/{i}.png" for i in range(17)]
|
||||
for sources, endpoint in (([], model), (refs, f"openai/gpt-image-2.5/{variant}/edit")):
|
||||
actual, payload = image_tool._prepare_fal_request(
|
||||
selected, meta, "a cup", aspect, 42, {"guidance_scale": 9}, sources,
|
||||
)
|
||||
assert actual == endpoint
|
||||
assert payload["quality"] == "medium"
|
||||
assert payload["image_size"] == size
|
||||
assert "seed" not in payload and "guidance_scale" not in payload
|
||||
assert payload.get("image_urls", []) == sources[:16]
|
||||
assert meta["upscale"] is False
|
||||
|
||||
|
||||
class TestFalCatalog:
|
||||
"""Every FAL_MODELS entry must have a consistent shape."""
|
||||
|
||||
|
||||
@@ -153,6 +153,31 @@ FAL_MODELS: Dict[str, Dict[str, Any]] = {
|
||||
},
|
||||
max_reference_images=16,
|
||||
),
|
||||
# Same minimum pixel count as GPT Image 2; keep medium quality explicit
|
||||
# rather than inheriting FAL's higher-cost high default.
|
||||
**{
|
||||
f"openai/gpt-image-2.5/{variant}/text-to-image": _model(
|
||||
f"GPT Image 2.5 {variant.title()}", speed, strengths, "Token-based pricing",
|
||||
sizes={
|
||||
"landscape": "landscape_4_3", "square": "square_hd", "portrait": "portrait_4_3",
|
||||
},
|
||||
defaults={"quality": "medium", "num_images": 1, "output_format": "png"},
|
||||
supports={
|
||||
"prompt", "image_size", "quality", "num_images", "output_format", "background",
|
||||
"output_compression", "sync_mode",
|
||||
},
|
||||
edit_endpoint=f"openai/gpt-image-2.5/{variant}/edit",
|
||||
edit_supports={
|
||||
"prompt", "image_urls", "image_size", "quality", "num_images", "output_format",
|
||||
"background", "output_compression", "sync_mode", "mask_url", "input_fidelity",
|
||||
},
|
||||
max_reference_images=16,
|
||||
)
|
||||
for variant, speed, strengths in (
|
||||
("flare", "Fast", "Everyday creation, natural lighting and textures"),
|
||||
("sunburst", "Slower", "Precision editing, subject and composition consistency"),
|
||||
)
|
||||
},
|
||||
"fal-ai/ideogram/v3": _model(
|
||||
"Ideogram V3", "~5s", "Best typography", "$0.03-0.09/image",
|
||||
defaults={"rendering_speed": "BALANCED", "expand_prompt": True, "style": "AUTO"},
|
||||
|
||||
@@ -126,6 +126,37 @@ Auth reuses the same env vars as the Meta chat provider — `MODEL_API_KEY`
|
||||
as aliases. Set `META_BASE_URL` to point at a proxy or alternate host. Text-to-image
|
||||
only for now; responses are saved to `$HERMES_HOME/cache/images/`.
|
||||
|
||||
## FAL: GPT Image 2.5
|
||||
|
||||
Select **GPT Image 2.5 Flare** or **GPT Image 2.5 Sunburst** under
|
||||
`hermes tools` → Image Generation → FAL.ai. The model IDs are:
|
||||
|
||||
- `openai/gpt-image-2.5/flare/text-to-image`
|
||||
- `openai/gpt-image-2.5/sunburst/text-to-image`
|
||||
|
||||
For example:
|
||||
|
||||
```bash
|
||||
hermes config set image_gen.provider fal
|
||||
hermes config set image_gen.model openai/gpt-image-2.5/flare/text-to-image
|
||||
```
|
||||
|
||||
Providing `image_url` or reference images automatically selects the corresponding
|
||||
`openai/gpt-image-2.5/flare/edit` or `openai/gpt-image-2.5/sunburst/edit` endpoint.
|
||||
Both accept up to 16 source images. Hermes pins quality to `medium`, matching its
|
||||
existing FAL GPT Image policy rather than FAL's higher-cost `high` default.
|
||||
Landscape and portrait use 4:3 presets to satisfy the minimum pixel count;
|
||||
square uses `square_hd`. Upscaling remains off unless requested.
|
||||
|
||||
FAL bills by tokens, not a fixed image price: $5/M text input, $1.25/M cached
|
||||
text input, $10/M text output, $8/M image input, $2/M cached image input, and
|
||||
$30/M image output, rounded up to $0.0001 per request. See the
|
||||
[Flare](https://fal.ai/models/openai/gpt-image-2.5/flare/text-to-image) and
|
||||
[Sunburst](https://fal.ai/models/openai/gpt-image-2.5/sunburst/text-to-image)
|
||||
pages. Direct FAL requires a funded `FAL_KEY`; managed-gateway availability
|
||||
depends on that gateway's endpoint allowlist and is not implied by FAL availability.
|
||||
Existing provider and model defaults are unchanged.
|
||||
|
||||
## OpenAI API: GPT Image 2.5
|
||||
|
||||
The **OpenAI** provider supports GPT Image 2.5 Flare (fast everyday creation)
|
||||
@@ -154,7 +185,7 @@ does not estimate 2.5 token consumption. See the official
|
||||
The **OpenAI (Codex auth)** provider remains separate: its backend can accept
|
||||
an image-model value without honoring that selection, so a successful image
|
||||
alone does not verify Flare or Sunburst routing. These selections are offered
|
||||
only through the direct OpenAI API provider, not Codex auth or FAL.
|
||||
through the direct OpenAI API provider and FAL, not as verified Codex-auth selections.
|
||||
|
||||
## Usage
|
||||
|
||||
@@ -196,7 +227,7 @@ Two inputs drive the edit:
|
||||
|
||||
| Backend | Image-to-image | Reference cap | How |
|
||||
|---|---|---|---|
|
||||
| **FAL.ai** (edit-capable models below) | ✓ | up to 9 | routes to the model's `/edit` endpoint |
|
||||
| **FAL.ai** (edit-capable models below) | ✓ | up to 16 (per model) | routes to the model's `/edit` endpoint |
|
||||
| **OpenAI** (GPT Image 2 / 2.5 Flare / Sunburst) | ✓ | up to 16 | `images.edit()` |
|
||||
| **xAI** (Grok Imagine) | ✓ | 1 | `/v1/images/edits` (`grok-imagine-image-quality`) |
|
||||
| **Krea** (`Krea 2`) | ✓ | up to 10 | reference-guided generation (`image_style_references`) |
|
||||
@@ -205,7 +236,7 @@ Two inputs drive the edit:
|
||||
|
||||
FAL models with an editing endpoint: `flux-2/klein/9b`, `flux-2-pro`,
|
||||
`nano-banana-pro`, `gpt-image-1.5`, `gpt-image-2`, `ideogram/v3`, and
|
||||
`qwen-image`. Pure text-to-image FAL models (`z-image/turbo`, `recraft`,
|
||||
`qwen-image`, plus GPT Image 2.5 Flare and Sunburst above. Pure text-to-image FAL models (`z-image/turbo`, `recraft`,
|
||||
`krea/*`) reject image inputs with a clear error pointing you at an
|
||||
edit-capable model.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user