feat: add GPT Image 2.5 generation and editing to OpenAI provider
This commit is contained in:
@@ -1,4 +1,4 @@
|
||||
"""OpenAI ``gpt-image-2`` at three quality tiers (virtual ids ``gpt-image-2-low/-medium/-high``);
|
||||
"""OpenAI GPT Image 2 and 2.5 Flare/Sunburst quality tiers;
|
||||
base64 output → image cache. Selection: ``OPENAI_IMAGE_MODEL`` → ``image_gen.openai.model`` →
|
||||
``image_gen.model`` → :data:`DEFAULT_MODEL`."""
|
||||
|
||||
@@ -18,9 +18,29 @@ from plugins.image_gen._common import (
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Keep subscription routing independent: Codex does not verify explicit image model selection.
|
||||
MODELS = {
|
||||
**{key: {**meta, "api_model": API_MODEL} for key, meta in GPT_IMAGE_2_TIERS.items()},
|
||||
**{
|
||||
model if quality == "auto" else f"{model}-{quality}": {
|
||||
"display": f"GPT Image 2.5 {name} ({quality.title()})",
|
||||
"speed": speed,
|
||||
"strengths": strengths,
|
||||
"api_model": model,
|
||||
"quality": quality,
|
||||
}
|
||||
for model, name, speed, strengths in (
|
||||
("gpt-image-2.5-flare", "Flare", "Fast", "Everyday image generation and editing"),
|
||||
("gpt-image-2.5-sunburst", "Sunburst", "Slower", "Precision generation and editing"),
|
||||
)
|
||||
for quality in ("auto", "low", "medium", "high", "xhigh", "max")
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _resolve_model() -> Tuple[str, Dict[str, Any]]:
|
||||
return resolve_static_model(
|
||||
GPT_IMAGE_2_TIERS, DEFAULT_MODEL, env_var="OPENAI_IMAGE_MODEL", config_key="openai")
|
||||
MODELS, DEFAULT_MODEL, env_var="OPENAI_IMAGE_MODEL", config_key="openai")
|
||||
|
||||
|
||||
def _load_image_bytes(ref: str) -> Tuple[bytes, str]:
|
||||
@@ -57,16 +77,16 @@ def _named_bytes_io(ref: str) -> io.BytesIO:
|
||||
|
||||
|
||||
class OpenAIImageGenProvider(StaticImageGenProvider):
|
||||
"""OpenAI ``images.generate`` / ``images.edit`` backend — gpt-image-2."""
|
||||
"""OpenAI ``images.generate`` / ``images.edit`` backend with selectable API models."""
|
||||
|
||||
provider_id = "openai"
|
||||
label = "OpenAI"
|
||||
models = GPT_IMAGE_2_TIERS
|
||||
models = MODELS
|
||||
default_model_id = DEFAULT_MODEL
|
||||
price = "varies"
|
||||
setup = dict(
|
||||
name="OpenAI", badge="paid",
|
||||
tag="gpt-image-2 at low/medium/high quality tiers — text-to-image & image editing",
|
||||
tag="GPT Image 2 / 2.5 Flare / 2.5 Sunburst — text-to-image & image editing",
|
||||
key="OPENAI_API_KEY", prompt="OpenAI API key", url="https://platform.openai.com/api-keys")
|
||||
|
||||
def is_available(self) -> bool:
|
||||
@@ -106,7 +126,7 @@ class OpenAIImageGenProvider(StaticImageGenProvider):
|
||||
# gpt-image-2 returns b64_json unconditionally and REJECTS
|
||||
# ``response_format`` as an unknown parameter. Don't send it.
|
||||
request: Dict[str, Any] = dict(
|
||||
model=API_MODEL, prompt=prompt, size=size, n=1, quality=meta["quality"])
|
||||
model=meta["api_model"], prompt=prompt, size=size, n=1, quality=meta["quality"])
|
||||
if is_edit:
|
||||
try:
|
||||
files = [_named_bytes_io(ref) for ref in sources]
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name: openai
|
||||
version: 1.0.0
|
||||
description: "OpenAI image generation backend (gpt-image-2). Saves generated images to $HERMES_HOME/cache/images/."
|
||||
description: "OpenAI image generation backend (GPT Image 2 and GPT Image 2.5 Flare/Sunburst). Saves generated images to $HERMES_HOME/cache/images/."
|
||||
author: NousResearch
|
||||
kind: backend
|
||||
requires_env:
|
||||
|
||||
@@ -57,9 +57,10 @@ class TestMetadata:
|
||||
def test_default_model(self, provider):
|
||||
assert provider.default_model() == "gpt-image-2-medium"
|
||||
|
||||
def test_list_models_three_tiers(self, provider):
|
||||
def test_picker_matches_resolvable_catalog(self, provider):
|
||||
ids = [m["id"] for m in provider.list_models()]
|
||||
assert ids == ["gpt-image-2-low", "gpt-image-2-medium", "gpt-image-2-high"]
|
||||
assert set(ids) == set(provider.models)
|
||||
assert provider.default_model() in ids
|
||||
|
||||
def test_catalog_entries_have_display_speed_strengths(self, provider):
|
||||
for entry in provider.list_models():
|
||||
@@ -171,24 +172,40 @@ class TestGenerate:
|
||||
# gpt-image-2 rejects response_format — we must NOT send it.
|
||||
assert "response_format" not in call_kwargs
|
||||
|
||||
@pytest.mark.parametrize("tier,expected_quality", [
|
||||
("gpt-image-2-low", "low"),
|
||||
("gpt-image-2-medium", "medium"),
|
||||
("gpt-image-2-high", "high"),
|
||||
@pytest.mark.parametrize("api_model,quality", [
|
||||
("gpt-image-2", quality) for quality in ("low", "medium", "high")
|
||||
] + [
|
||||
(model, quality)
|
||||
for model in ("gpt-image-2.5-flare", "gpt-image-2.5-sunburst")
|
||||
for quality in ("auto", "low", "medium", "high", "xhigh", "max")
|
||||
])
|
||||
def test_tier_maps_to_quality(self, provider, monkeypatch, tier, expected_quality):
|
||||
monkeypatch.setenv("OPENAI_IMAGE_MODEL", tier)
|
||||
@pytest.mark.parametrize("editing", [False, True])
|
||||
def test_selection_reaches_image_request(
|
||||
self, provider, monkeypatch, tmp_path, api_model, quality, editing
|
||||
):
|
||||
import yaml
|
||||
|
||||
tier = api_model if quality == "auto" else f"{api_model}-{quality}"
|
||||
monkeypatch.delenv("OPENAI_IMAGE_MODEL", raising=False)
|
||||
(tmp_path / "config.yaml").write_text(yaml.safe_dump({
|
||||
"image_gen": {"openai": {"model": tier}}
|
||||
}))
|
||||
source = tmp_path / "source.png"
|
||||
source.write_bytes(bytes.fromhex(_PNG_HEX))
|
||||
fake_client = MagicMock()
|
||||
fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
|
||||
call = fake_client.images.edit if editing else fake_client.images.generate
|
||||
call.return_value = _fake_response(b64=_b64_png())
|
||||
|
||||
with _patched_openai(fake_client):
|
||||
result = provider.generate("a cat")
|
||||
result = provider.generate("a cat", image_url=str(source) if editing else None)
|
||||
|
||||
assert result["success"] is True
|
||||
assert result["model"] == tier
|
||||
assert result["quality"] == expected_quality
|
||||
assert fake_client.images.generate.call_args.kwargs["quality"] == expected_quality
|
||||
# Always the same underlying API model regardless of tier.
|
||||
assert fake_client.images.generate.call_args.kwargs["model"] == "gpt-image-2"
|
||||
assert result["quality"] == quality
|
||||
assert call.call_args.kwargs["quality"] == quality
|
||||
assert call.call_args.kwargs["model"] == api_model
|
||||
assert "response_format" not in call.call_args.kwargs
|
||||
assert Path(result["image"]).read_bytes() == bytes.fromhex(_PNG_HEX)
|
||||
|
||||
@pytest.mark.parametrize("aspect,expected_size", [
|
||||
("landscape", "1536x1024"),
|
||||
|
||||
@@ -61,7 +61,7 @@ The repo ships these bundled plugins under `plugins/`. All are opt-in — enable
|
||||
| `teams_pipeline` | standalone | Microsoft Teams meeting pipeline — Graph-backed, transcript-first meeting summaries |
|
||||
| `spotify` | backend (7 tools) | Native Spotify playback, queue, search, playlists, albums, library |
|
||||
| `google_meet` | standalone | Join Meet calls, live-caption transcription, optional realtime duplex audio |
|
||||
| `image_gen/openai` | image backend | OpenAI `gpt-image-2` image generation backend (alternative to FAL) |
|
||||
| `image_gen/openai` | image backend | OpenAI GPT Image 2 and 2.5 Flare/Sunburst generation and editing (API key) |
|
||||
| `image_gen/openai-codex` | image backend | OpenAI image generation via Codex OAuth |
|
||||
| `image_gen/xai` | image backend | xAI `grok-2-image` backend |
|
||||
| `hermes-achievements` | dashboard tab | Steam-style collectible badges generated from your real Hermes session history |
|
||||
|
||||
@@ -126,6 +126,36 @@ Auth reuses the same env vars as the Meta chat provider — `MODEL_API_KEY`
|
||||
as aliases. Set `META_BASE_URL` to point at a proxy or alternate host. Text-to-image
|
||||
only for now; responses are saved to `$HERMES_HOME/cache/images/`.
|
||||
|
||||
## OpenAI API: GPT Image 2.5
|
||||
|
||||
The **OpenAI** provider supports GPT Image 2.5 Flare (fast everyday creation)
|
||||
and Sunburst (precision generation and editing), using `OPENAI_API_KEY`.
|
||||
Select them through `hermes tools` → Image Generation → OpenAI, or set:
|
||||
|
||||
```bash
|
||||
hermes config set image_gen.provider openai
|
||||
hermes config set image_gen.openai.model gpt-image-2.5-flare
|
||||
```
|
||||
|
||||
`gpt-image-2.5-flare` and `gpt-image-2.5-sunburst` use automatic quality.
|
||||
Append `-low`, `-medium`, `-high`, `-xhigh`, or `-max` to select a fixed quality,
|
||||
for example `gpt-image-2.5-sunburst-high`. Both support generation and editing
|
||||
with up to 16 reference images. Existing GPT Image 2 selections and the
|
||||
`gpt-image-2-medium` default are unchanged.
|
||||
|
||||
This is paid API usage, separate from a ChatGPT/Codex subscription. Both models
|
||||
cost $5 per million text-input tokens, $8 per million image-input tokens, and
|
||||
$30 per million image-output tokens (cached input rates are $1.25 and $2,
|
||||
respectively). Per-image cost varies with usage; the GPT Image 2 calculator
|
||||
does not estimate 2.5 token consumption. See the official
|
||||
[Flare](https://developers.openai.com/api/docs/models/gpt-image-2.5-flare) and
|
||||
[Sunburst](https://developers.openai.com/api/docs/models/gpt-image-2.5-sunburst) docs.
|
||||
|
||||
The **OpenAI (Codex auth)** provider remains separate: its backend can accept
|
||||
an image-model value without honoring that selection, so a successful image
|
||||
alone does not verify Flare or Sunburst routing. These selections are offered
|
||||
only through the direct OpenAI API provider, not Codex auth or FAL.
|
||||
|
||||
## Usage
|
||||
|
||||
The agent-facing schema is intentionally minimal — the model picks up whatever you've configured:
|
||||
@@ -167,7 +197,7 @@ Two inputs drive the edit:
|
||||
| Backend | Image-to-image | Reference cap | How |
|
||||
|---|---|---|---|
|
||||
| **FAL.ai** (edit-capable models below) | ✓ | up to 9 | routes to the model's `/edit` endpoint |
|
||||
| **OpenAI** (`gpt-image-2`) | ✓ | up to 16 | `images.edit()` |
|
||||
| **OpenAI** (GPT Image 2 / 2.5 Flare / Sunburst) | ✓ | up to 16 | `images.edit()` |
|
||||
| **xAI** (Grok Imagine) | ✓ | 1 | `/v1/images/edits` (`grok-imagine-image-quality`) |
|
||||
| **Krea** (`Krea 2`) | ✓ | up to 10 | reference-guided generation (`image_style_references`) |
|
||||
| **OpenAI (Codex auth)** | ✓ | up to 16 | Codex Responses `image_generation` tool with `input_image` content parts |
|
||||
|
||||
Reference in New Issue
Block a user