From 82d5a6ecd57a5dc185b50ef2c2a04eb7f61aaf35 Mon Sep 17 00:00:00 2001 From: Evan Date: Mon, 27 Apr 2026 09:44:14 +0900 Subject: [PATCH] feat(minimax): add image/audio/video/music model entries (#77) Extend modality enum to support video and music, then register the non-text MiniMax models that were already declared in media_capabilities but had no concrete entries: - image-01 ($0.0035/image) - speech-2.8/2.6 hd & turbo ($60-$100 per 1M chars) - Hailuo 2.3 Fast / 2.3 / 02 video models ($0.10-$0.56 per video) - music-2.6, lyrics_generation Per-call pricing is documented in inline comments since the schema's token-based cost fields don't naturally fit per-call billing. schema.toml and scripts/validate.py both updated; the change is additive (existing modality values remain valid). --- providers/minimax.toml | 145 ++++++++++++++++++++++++++++++++++++++++- schema.toml | 4 +- scripts/validate.py | 2 +- 3 files changed, 147 insertions(+), 4 deletions(-) diff --git a/providers/minimax.toml b/providers/minimax.toml index 35d43ff..48b533e 100644 --- a/providers/minimax.toml +++ b/providers/minimax.toml @@ -1,6 +1,11 @@ # MiniMax — https://minimax.io -# Models: 4 +# Models: 14 (4 text + 1 image + 4 audio TTS + 3 video + 2 music) # Regions: international (default), china (minimaxi.com) +# +# Non-text models are billed per call rather than per token, so +# input_cost_per_m / output_cost_per_m are set to 0.0 and the actual +# rate is documented in the inline comment above each entry. +# Source: https://platform.minimax.io/docs/guides/pricing-paygo [provider] id = "minimax" @@ -70,3 +75,141 @@ supports_tools = true supports_vision = true supports_streaming = true aliases = ["minimax-m2.5-highspeed", "m2.5-highspeed"] + +# --- Image generation --- + +# $0.0035 per image +[[models]] +id = "image-01" +display_name = "MiniMax Image 01" +tier = "smart" +modality = "image" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["minimax-image"] + +# --- Text-to-speech --- + +# $100 per 1M characters +[[models]] +id = "speech-2.8-hd" +display_name = "MiniMax Speech 2.8 HD" +tier = "frontier" +modality = "audio" +input_cost_per_m = 100.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = true +aliases = ["minimax-tts-hd", "speech-hd"] + +# $60 per 1M characters +[[models]] +id = "speech-2.8-turbo" +display_name = "MiniMax Speech 2.8 Turbo" +tier = "fast" +modality = "audio" +input_cost_per_m = 60.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = true +aliases = ["minimax-tts-turbo", "speech-turbo"] + +# $100 per 1M characters +[[models]] +id = "speech-2.6-hd" +display_name = "MiniMax Speech 2.6 HD" +tier = "smart" +modality = "audio" +input_cost_per_m = 100.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = true +aliases = [] + +# $60 per 1M characters +[[models]] +id = "speech-2.6-turbo" +display_name = "MiniMax Speech 2.6 Turbo" +tier = "fast" +modality = "audio" +input_cost_per_m = 60.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = true +aliases = [] + +# --- Video generation (Hailuo) --- + +# Per-call: $0.19 (768P/6s) up to $0.33 (1080P/6s) +[[models]] +id = "MiniMax-Hailuo-2.3-Fast" +display_name = "Hailuo 2.3 Fast" +tier = "fast" +modality = "video" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["hailuo-2.3-fast", "hailuo-fast"] + +# Per-call: $0.28 (768P/6s) up to $0.56 (1080P/6s) +[[models]] +id = "MiniMax-Hailuo-2.3" +display_name = "Hailuo 2.3" +tier = "frontier" +modality = "video" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["hailuo-2.3", "hailuo"] + +# Per-call: $0.10 (512P/6s) up to $0.56 (1080P/6s) +[[models]] +id = "MiniMax-Hailuo-02" +display_name = "Hailuo 02" +tier = "smart" +modality = "video" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["hailuo-02"] + +# --- Music generation --- + +# $0.15 per up-to-5-minute track +[[models]] +id = "music-2.6" +display_name = "MiniMax Music 2.6" +tier = "frontier" +modality = "music" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["minimax-music"] + +# $0.01 per song +[[models]] +id = "lyrics_generation" +display_name = "MiniMax Lyrics Generation" +tier = "fast" +modality = "music" +input_cost_per_m = 0.0 +output_cost_per_m = 0.0 +supports_tools = false +supports_vision = false +supports_streaming = false +aliases = ["minimax-lyrics"] diff --git a/schema.toml b/schema.toml index 88e1ca3..6fc7b31 100644 --- a/schema.toml +++ b/schema.toml @@ -75,8 +75,8 @@ example = "smart" [provider.sections.models.fields.modality] type = "string" required = false -description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model (context_window/max_output_tokens become optional); 'audio' is a speech model." -options = ["text", "image", "audio"] +description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model; 'audio' is a speech / TTS model; 'video' is a video-generation model; 'music' is a music / lyrics generation model. For non-text modalities, context_window / max_output_tokens are optional and input_cost_per_m / output_cost_per_m may be 0 when the model is billed per call." +options = ["text", "image", "audio", "video", "music"] default = "text" example = "text" diff --git a/scripts/validate.py b/scripts/validate.py index 5b33517..f892b00 100755 --- a/scripts/validate.py +++ b/scripts/validate.py @@ -34,7 +34,7 @@ except ImportError: sys.exit(1) VALID_TIERS = {"frontier", "smart", "balanced", "fast", "local"} -VALID_MODALITIES = {"text", "image", "audio"} +VALID_MODALITIES = {"text", "image", "audio", "video", "music"} VALID_HAND_CATEGORIES = { "communication", "content", "data", "development", "devops", "finance", "productivity", "research", "social",