From 5909b024c1cdc51cb323d57c8aa8afbb82a1ed25 Mon Sep 17 00:00:00 2001 From: Evan Date: Sat, 25 Apr 2026 13:30:07 +0900 Subject: [PATCH] feat(openai): add GPT Image 2 (image-generation modality) (#71) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduces image-generation models as a first-class [[models]] entry via a new `modality` field on the model schema ("text" default, "image", "audio"). When modality != "text", context_window / max_output_tokens are optional since no conventional context gate exists — OpenAI's gpt-image-2 docs omit them. Adds `image_input_cost_per_m` / `image_output_cost_per_m` alongside existing text token cost fields to cover the 4-price structure OpenAI uses for image generation (text $5/$10, image $8/$30 per 1M tokens). Validator updated to: - accept any modality in {text, image, audio} - require context_window/max_output_tokens only for modality=text - range-check the two new cost fields gpt-image-2 entry added to providers/openai.toml with pricing sourced from https://developers.openai.com/api/docs/pricing. Snapshot gpt-image-2-2026-04-21 listed as alias. --- providers/openai.toml | 20 +++++++++++++++++++- schema.toml | 28 ++++++++++++++++++++++++---- scripts/validate.py | 28 ++++++++++++++++++++++++---- 3 files changed, 67 insertions(+), 9 deletions(-) diff --git a/providers/openai.toml b/providers/openai.toml index 6826d9f..145f4c8 100644 --- a/providers/openai.toml +++ b/providers/openai.toml @@ -1,5 +1,5 @@ # OpenAI — https://openai.com -# Models: 17 (including 2 Codex variants) +# Models: 18 (including 2 Codex variants + 1 image-generation model) [provider] id = "openai" @@ -237,3 +237,21 @@ supports_vision = true supports_streaming = true supports_thinking = true aliases = ["codex-o4"] + +# Image generation — priced per token (text + image), no conventional context +# window. Snapshot gpt-image-2-2026-04-21 released 2026-04-21. +# Source: https://developers.openai.com/api/docs/models/gpt-image-2 +# Pricing: https://developers.openai.com/api/docs/pricing +[[models]] +id = "gpt-image-2" +display_name = "GPT Image 2" +tier = "frontier" +modality = "image" +input_cost_per_m = 5.00 +output_cost_per_m = 10.00 +image_input_cost_per_m = 8.00 +image_output_cost_per_m = 30.00 +supports_tools = false +supports_vision = true +supports_streaming = false +aliases = ["gpt-image-2-2026-04-21"] diff --git a/schema.toml b/schema.toml index 68f1480..88e1ca3 100644 --- a/schema.toml +++ b/schema.toml @@ -72,30 +72,50 @@ description = "Capability tier: frontier (cutting-edge, most capable), smart (co options = ["frontier", "smart", "balanced", "fast", "local"] example = "smart" +[provider.sections.models.fields.modality] +type = "string" +required = false +description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model (context_window/max_output_tokens become optional); 'audio' is a speech model." +options = ["text", "image", "audio"] +default = "text" +example = "text" + [provider.sections.models.fields.context_window] type = "number" required = true -description = "Maximum input tokens" +description = "Maximum input tokens. Required for modality='text'; optional for image/audio models where no published context gate exists." example = 128000 [provider.sections.models.fields.max_output_tokens] type = "number" required = true -description = "Maximum output tokens" +description = "Maximum output tokens. Required for modality='text'; optional for image/audio models." example = 16384 [provider.sections.models.fields.input_cost_per_m] type = "number" required = true -description = "USD per million input tokens (0.0 for free/local)" +description = "USD per million text input tokens (0.0 for free/local)" example = 2.5 [provider.sections.models.fields.output_cost_per_m] type = "number" required = true -description = "USD per million output tokens (0.0 for free/local)" +description = "USD per million text output tokens (0.0 for free/local)" example = 10.0 +[provider.sections.models.fields.image_input_cost_per_m] +type = "number" +required = false +description = "USD per million image input tokens (for image/multimodal models)" +example = 8.0 + +[provider.sections.models.fields.image_output_cost_per_m] +type = "number" +required = false +description = "USD per million image output tokens (for image-generation models)" +example = 30.0 + [provider.sections.models.fields.last_verified] type = "string" required = false diff --git a/scripts/validate.py b/scripts/validate.py index da031e2..5b33517 100755 --- a/scripts/validate.py +++ b/scripts/validate.py @@ -34,6 +34,7 @@ except ImportError: sys.exit(1) VALID_TIERS = {"frontier", "smart", "balanced", "fast", "local"} +VALID_MODALITIES = {"text", "image", "audio"} VALID_HAND_CATEGORIES = { "communication", "content", "data", "development", "devops", "finance", "productivity", "research", "social", @@ -41,11 +42,14 @@ VALID_HAND_CATEGORIES = { VALID_SKILL_RUNTIMES = {"promptonly", "python", "node", "shell"} REQUIRED_PROVIDER_FIELDS = {"id", "display_name", "api_key_env", "base_url", "key_required"} -REQUIRED_MODEL_FIELDS = { + +# Required for all modalities. +REQUIRED_MODEL_FIELDS_COMMON = { "id", "display_name", "tier", - "context_window", "max_output_tokens", "input_cost_per_m", "output_cost_per_m", } +# Additionally required when modality='text' (the default). +REQUIRED_MODEL_FIELDS_TEXT_ONLY = {"context_window", "max_output_tokens"} VALID_CONTENT_TYPES = {"providers", "agents", "hands", "mcp", "skills", "plugins", "aliases"} @@ -79,7 +83,18 @@ def validate_provider_file(filepath: Path) -> list[str]: for i, model in enumerate(models): label = model.get("id", f"models[{i}]") - for field in REQUIRED_MODEL_FIELDS: + + modality = model.get("modality", "text") + if modality not in VALID_MODALITIES: + errors.append( + f"{filepath.name}: Model '{label}' invalid modality '{modality}' " + f"(valid: {', '.join(sorted(VALID_MODALITIES))})" + ) + + required_fields = set(REQUIRED_MODEL_FIELDS_COMMON) + if modality == "text": + required_fields |= REQUIRED_MODEL_FIELDS_TEXT_ONLY + for field in required_fields: if field not in model: errors.append(f"{filepath.name}: Model '{label}' missing field '{field}'") @@ -87,7 +102,12 @@ def validate_provider_file(filepath: Path) -> list[str]: if tier is not None and tier not in VALID_TIERS: errors.append(f"{filepath.name}: Model '{label}' invalid tier '{tier}'") - for cost_field in ("input_cost_per_m", "output_cost_per_m"): + for cost_field in ( + "input_cost_per_m", + "output_cost_per_m", + "image_input_cost_per_m", + "image_output_cost_per_m", + ): val = model.get(cost_field) if val is not None and (not isinstance(val, (int, float)) or val < 0): errors.append(f"{filepath.name}: Model '{label}' invalid {cost_field}: {val}")