feat: add GPT Image 2.5 generation and editing to OpenAI provider

This commit is contained in:
Teknium
2026-09-08 12:21:41 -07:00
parent 7568fb6727
commit 7777f8c350
5 changed files with 90 additions and 23 deletions

View File

@@ -1,4 +1,4 @@
"""OpenAI ``gpt-image-2`` at three quality tiers (virtual ids ``gpt-image-2-low/-medium/-high``);
"""OpenAI GPT Image 2 and 2.5 Flare/Sunburst quality tiers;
base64 output → image cache. Selection: ``OPENAI_IMAGE_MODEL`` → ``image_gen.openai.model`` →
``image_gen.model`` → :data:`DEFAULT_MODEL`."""
@@ -18,9 +18,29 @@ from plugins.image_gen._common import (
logger = logging.getLogger(__name__)
# Keep subscription routing independent: Codex does not verify explicit image model selection.
MODELS = {
**{key: {**meta, "api_model": API_MODEL} for key, meta in GPT_IMAGE_2_TIERS.items()},
**{
model if quality == "auto" else f"{model}-{quality}": {
"display": f"GPT Image 2.5 {name} ({quality.title()})",
"speed": speed,
"strengths": strengths,
"api_model": model,
"quality": quality,
}
for model, name, speed, strengths in (
("gpt-image-2.5-flare", "Flare", "Fast", "Everyday image generation and editing"),
("gpt-image-2.5-sunburst", "Sunburst", "Slower", "Precision generation and editing"),
)
for quality in ("auto", "low", "medium", "high", "xhigh", "max")
},
}
def _resolve_model() -> Tuple[str, Dict[str, Any]]:
return resolve_static_model(
GPT_IMAGE_2_TIERS, DEFAULT_MODEL, env_var="OPENAI_IMAGE_MODEL", config_key="openai")
MODELS, DEFAULT_MODEL, env_var="OPENAI_IMAGE_MODEL", config_key="openai")
def _load_image_bytes(ref: str) -> Tuple[bytes, str]:
@@ -57,16 +77,16 @@ def _named_bytes_io(ref: str) -> io.BytesIO:
class OpenAIImageGenProvider(StaticImageGenProvider):
"""OpenAI ``images.generate`` / ``images.edit`` backend — gpt-image-2."""
"""OpenAI ``images.generate`` / ``images.edit`` backend with selectable API models."""
provider_id = "openai"
label = "OpenAI"
models = GPT_IMAGE_2_TIERS
models = MODELS
default_model_id = DEFAULT_MODEL
price = "varies"
setup = dict(
name="OpenAI", badge="paid",
tag="gpt-image-2 at low/medium/high quality tiers — text-to-image & image editing",
tag="GPT Image 2 / 2.5 Flare / 2.5 Sunburst — text-to-image & image editing",
key="OPENAI_API_KEY", prompt="OpenAI API key", url="https://platform.openai.com/api-keys")
def is_available(self) -> bool:
@@ -106,7 +126,7 @@ class OpenAIImageGenProvider(StaticImageGenProvider):
# gpt-image-2 returns b64_json unconditionally and REJECTS
# ``response_format`` as an unknown parameter. Don't send it.
request: Dict[str, Any] = dict(
model=API_MODEL, prompt=prompt, size=size, n=1, quality=meta["quality"])
model=meta["api_model"], prompt=prompt, size=size, n=1, quality=meta["quality"])
if is_edit:
try:
files = [_named_bytes_io(ref) for ref in sources]

View File

@@ -1,6 +1,6 @@
name: openai
version: 1.0.0
description: "OpenAI image generation backend (gpt-image-2). Saves generated images to $HERMES_HOME/cache/images/."
description: "OpenAI image generation backend (GPT Image 2 and GPT Image 2.5 Flare/Sunburst). Saves generated images to $HERMES_HOME/cache/images/."
author: NousResearch
kind: backend
requires_env:

View File

@@ -57,9 +57,10 @@ class TestMetadata:
def test_default_model(self, provider):
assert provider.default_model() == "gpt-image-2-medium"
def test_list_models_three_tiers(self, provider):
def test_picker_matches_resolvable_catalog(self, provider):
ids = [m["id"] for m in provider.list_models()]
assert ids == ["gpt-image-2-low", "gpt-image-2-medium", "gpt-image-2-high"]
assert set(ids) == set(provider.models)
assert provider.default_model() in ids
def test_catalog_entries_have_display_speed_strengths(self, provider):
for entry in provider.list_models():
@@ -171,24 +172,40 @@ class TestGenerate:
# gpt-image-2 rejects response_format — we must NOT send it.
assert "response_format" not in call_kwargs
@pytest.mark.parametrize("tier,expected_quality", [
("gpt-image-2-low", "low"),
("gpt-image-2-medium", "medium"),
("gpt-image-2-high", "high"),
@pytest.mark.parametrize("api_model,quality", [
("gpt-image-2", quality) for quality in ("low", "medium", "high")
] + [
(model, quality)
for model in ("gpt-image-2.5-flare", "gpt-image-2.5-sunburst")
for quality in ("auto", "low", "medium", "high", "xhigh", "max")
])
def test_tier_maps_to_quality(self, provider, monkeypatch, tier, expected_quality):
monkeypatch.setenv("OPENAI_IMAGE_MODEL", tier)
@pytest.mark.parametrize("editing", [False, True])
def test_selection_reaches_image_request(
self, provider, monkeypatch, tmp_path, api_model, quality, editing
):
import yaml
tier = api_model if quality == "auto" else f"{api_model}-{quality}"
monkeypatch.delenv("OPENAI_IMAGE_MODEL", raising=False)
(tmp_path / "config.yaml").write_text(yaml.safe_dump({
"image_gen": {"openai": {"model": tier}}
}))
source = tmp_path / "source.png"
source.write_bytes(bytes.fromhex(_PNG_HEX))
fake_client = MagicMock()
fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
call = fake_client.images.edit if editing else fake_client.images.generate
call.return_value = _fake_response(b64=_b64_png())
with _patched_openai(fake_client):
result = provider.generate("a cat")
result = provider.generate("a cat", image_url=str(source) if editing else None)
assert result["success"] is True
assert result["model"] == tier
assert result["quality"] == expected_quality
assert fake_client.images.generate.call_args.kwargs["quality"] == expected_quality
# Always the same underlying API model regardless of tier.
assert fake_client.images.generate.call_args.kwargs["model"] == "gpt-image-2"
assert result["quality"] == quality
assert call.call_args.kwargs["quality"] == quality
assert call.call_args.kwargs["model"] == api_model
assert "response_format" not in call.call_args.kwargs
assert Path(result["image"]).read_bytes() == bytes.fromhex(_PNG_HEX)
@pytest.mark.parametrize("aspect,expected_size", [
("landscape", "1536x1024"),

View File

@@ -61,7 +61,7 @@ The repo ships these bundled plugins under `plugins/`. All are opt-in — enable
| `teams_pipeline` | standalone | Microsoft Teams meeting pipeline — Graph-backed, transcript-first meeting summaries |
| `spotify` | backend (7 tools) | Native Spotify playback, queue, search, playlists, albums, library |
| `google_meet` | standalone | Join Meet calls, live-caption transcription, optional realtime duplex audio |
| `image_gen/openai` | image backend | OpenAI `gpt-image-2` image generation backend (alternative to FAL) |
| `image_gen/openai` | image backend | OpenAI GPT Image 2 and 2.5 Flare/Sunburst generation and editing (API key) |
| `image_gen/openai-codex` | image backend | OpenAI image generation via Codex OAuth |
| `image_gen/xai` | image backend | xAI `grok-2-image` backend |
| `hermes-achievements` | dashboard tab | Steam-style collectible badges generated from your real Hermes session history |

View File

@@ -126,6 +126,36 @@ Auth reuses the same env vars as the Meta chat provider — `MODEL_API_KEY`
as aliases. Set `META_BASE_URL` to point at a proxy or alternate host. Text-to-image
only for now; responses are saved to `$HERMES_HOME/cache/images/`.
## OpenAI API: GPT Image 2.5
The **OpenAI** provider supports GPT Image 2.5 Flare (fast everyday creation)
and Sunburst (precision generation and editing), using `OPENAI_API_KEY`.
Select them through `hermes tools` → Image Generation → OpenAI, or set:
```bash
hermes config set image_gen.provider openai
hermes config set image_gen.openai.model gpt-image-2.5-flare
```
`gpt-image-2.5-flare` and `gpt-image-2.5-sunburst` use automatic quality.
Append `-low`, `-medium`, `-high`, `-xhigh`, or `-max` to select a fixed quality,
for example `gpt-image-2.5-sunburst-high`. Both support generation and editing
with up to 16 reference images. Existing GPT Image 2 selections and the
`gpt-image-2-medium` default are unchanged.
This is paid API usage, separate from a ChatGPT/Codex subscription. Both models
cost $5 per million text-input tokens, $8 per million image-input tokens, and
$30 per million image-output tokens (cached input rates are $1.25 and $2,
respectively). Per-image cost varies with usage; the GPT Image 2 calculator
does not estimate 2.5 token consumption. See the official
[Flare](https://developers.openai.com/api/docs/models/gpt-image-2.5-flare) and
[Sunburst](https://developers.openai.com/api/docs/models/gpt-image-2.5-sunburst) docs.
The **OpenAI (Codex auth)** provider remains separate: its backend can accept
an image-model value without honoring that selection, so a successful image
alone does not verify Flare or Sunburst routing. These selections are offered
only through the direct OpenAI API provider, not Codex auth or FAL.
## Usage
The agent-facing schema is intentionally minimal — the model picks up whatever you've configured:
@@ -167,7 +197,7 @@ Two inputs drive the edit:
| Backend | Image-to-image | Reference cap | How |
|---|---|---|---|
| **FAL.ai** (edit-capable models below) | ✓ | up to 9 | routes to the model's `/edit` endpoint |
| **OpenAI** (`gpt-image-2`) | ✓ | up to 16 | `images.edit()` |
| **OpenAI** (GPT Image 2 / 2.5 Flare / Sunburst) | ✓ | up to 16 | `images.edit()` |
| **xAI** (Grok Imagine) | ✓ | 1 | `/v1/images/edits` (`grok-imagine-image-quality`) |
| **Krea** (`Krea 2`) | ✓ | up to 10 | reference-guided generation (`image_style_references`) |
| **OpenAI (Codex auth)** | ✓ | up to 16 | Codex Responses `image_generation` tool with `input_image` content parts |