feat(openai): add GPT Image 2 (image-generation modality) (#71)

Introduces image-generation models as a first-class [[models]] entry via
a new `modality` field on the model schema ("text" default, "image",
"audio"). When modality != "text", context_window / max_output_tokens
are optional since no conventional context gate exists — OpenAI's
gpt-image-2 docs omit them.

Adds `image_input_cost_per_m` / `image_output_cost_per_m` alongside
existing text token cost fields to cover the 4-price structure OpenAI
uses for image generation (text $5/$10, image $8/$30 per 1M tokens).

Validator updated to:
- accept any modality in {text, image, audio}
- require context_window/max_output_tokens only for modality=text
- range-check the two new cost fields

gpt-image-2 entry added to providers/openai.toml with pricing sourced
from https://developers.openai.com/api/docs/pricing. Snapshot
gpt-image-2-2026-04-21 listed as alias.
This commit is contained in:
Evan authored and GitHub committed 2026-04-25 13:30:07 +09:00
1 parent 65cb852632
commit 5909b024c1
3 files changed
+67 -9

No files matched your search

+24 -4
View File
@@ -72,30 +72,50 @@ description = "Capability tier: frontier (cutting-edge, most capable), smart (co
options = ["frontier", "smart", "balanced", "fast", "local"]
example = "smart"
[provider.sections.models.fields.modality]
type = "string"
required = false
description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model (context_window/max_output_tokens become optional); 'audio' is a speech model."
options = ["text", "image", "audio"]
default = "text"
example = "text"
[provider.sections.models.fields.context_window]
type = "number"
required = true
description = "Maximum input tokens"
description = "Maximum input tokens. Required for modality='text'; optional for image/audio models where no published context gate exists."
example = 128000
[provider.sections.models.fields.max_output_tokens]
type = "number"
required = true
description = "Maximum output tokens"
description = "Maximum output tokens. Required for modality='text'; optional for image/audio models."
example = 16384
[provider.sections.models.fields.input_cost_per_m]
type = "number"
required = true
description = "USD per million input tokens (0.0 for free/local)"
description = "USD per million text input tokens (0.0 for free/local)"
example = 2.5
[provider.sections.models.fields.output_cost_per_m]
type = "number"
required = true
description = "USD per million output tokens (0.0 for free/local)"
description = "USD per million text output tokens (0.0 for free/local)"
example = 10.0
[provider.sections.models.fields.image_input_cost_per_m]
type = "number"
required = false
description = "USD per million image input tokens (for image/multimodal models)"
example = 8.0
[provider.sections.models.fields.image_output_cost_per_m]
type = "number"
required = false
description = "USD per million image output tokens (for image-generation models)"
example = 30.0
[provider.sections.models.fields.last_verified]
type = "string"
required = false