feat(minimax): add image/audio/video/music model entries (#77)
Extend modality enum to support video and music, then register the non-text MiniMax models that were already declared in media_capabilities but had no concrete entries: - image-01 ($0.0035/image) - speech-2.8/2.6 hd & turbo ($60-$100 per 1M chars) - Hailuo 2.3 Fast / 2.3 / 02 video models ($0.10-$0.56 per video) - music-2.6, lyrics_generation Per-call pricing is documented in inline comments since the schema's token-based cost fields don't naturally fit per-call billing. schema.toml and scripts/validate.py both updated; the change is additive (existing modality values remain valid).
This commit is contained in:
3 files changed
+147
-4
No files matched your search
+2
-2
@@ -75,8 +75,8 @@ example = "smart"
|
||||
[provider.sections.models.fields.modality]
|
||||
type = "string"
|
||||
required = false
|
||||
description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model (context_window/max_output_tokens become optional); 'audio' is a speech model."
|
||||
options = ["text", "image", "audio"]
|
||||
description = "Model modality. 'text' (default) is a chat/LLM; 'image' is an image-generation model; 'audio' is a speech / TTS model; 'video' is a video-generation model; 'music' is a music / lyrics generation model. For non-text modalities, context_window / max_output_tokens are optional and input_cost_per_m / output_cost_per_m may be 0 when the model is billed per call."
|
||||
options = ["text", "image", "audio", "video", "music"]
|
||||
default = "text"
|
||||
example = "text"
|
||||
|
||||
|
||||
Reference in new issue
Block a user