feat(hands): add Creator Hand for media generation (#17)
This commit is contained in:
3 files changed
+715
No files matched your search
@@ -0,0 +1,419 @@
|
|||||||
|
id = "creator"
|
||||||
|
version = "1.0.0"
|
||||||
|
name = "Creator Hand"
|
||||||
|
description = "AI media studio — generates images, videos, music, and speech from text prompts"
|
||||||
|
|
||||||
|
category = "content"
|
||||||
|
icon = "\U0001F3A8"
|
||||||
|
tools = [
|
||||||
|
"image_generate",
|
||||||
|
"video_generate",
|
||||||
|
"video_status",
|
||||||
|
"music_generate",
|
||||||
|
"text_to_speech",
|
||||||
|
"file_read",
|
||||||
|
"file_write",
|
||||||
|
"file_list",
|
||||||
|
"web_fetch",
|
||||||
|
"memory_store",
|
||||||
|
"memory_recall",
|
||||||
|
]
|
||||||
|
|
||||||
|
[routing]
|
||||||
|
aliases = [
|
||||||
|
"generate image",
|
||||||
|
"create image",
|
||||||
|
"make a picture",
|
||||||
|
"generate video",
|
||||||
|
"create video",
|
||||||
|
"make a video",
|
||||||
|
"generate music",
|
||||||
|
"create music",
|
||||||
|
"compose music",
|
||||||
|
"text to speech",
|
||||||
|
"generate speech",
|
||||||
|
"voice over",
|
||||||
|
"media generation",
|
||||||
|
]
|
||||||
|
weak_aliases = [
|
||||||
|
"illustration",
|
||||||
|
"artwork",
|
||||||
|
"album cover",
|
||||||
|
"background music",
|
||||||
|
"jingle",
|
||||||
|
"narration",
|
||||||
|
"audio",
|
||||||
|
"thumbnail",
|
||||||
|
"poster",
|
||||||
|
"banner",
|
||||||
|
]
|
||||||
|
|
||||||
|
# ---- Requirements ----------------------------------------------------------------
|
||||||
|
|
||||||
|
[[requires]]
|
||||||
|
key = "media_provider"
|
||||||
|
label = "At least one media provider API key must be set"
|
||||||
|
requirement_type = "any_env_var"
|
||||||
|
check_value = "OPENAI_API_KEY,MINIMAX_API_KEY"
|
||||||
|
description = "Creator Hand needs at least one configured media provider. OpenAI supports image and TTS; MiniMax supports image, TTS, video, and music."
|
||||||
|
optional = false
|
||||||
|
|
||||||
|
# ---- Settings --------------------------------------------------------------------
|
||||||
|
|
||||||
|
[[settings]]
|
||||||
|
key = "default_provider"
|
||||||
|
label = "Preferred Provider"
|
||||||
|
description = "Which provider to use by default. Auto will pick the first configured provider for each capability."
|
||||||
|
setting_type = "select"
|
||||||
|
default = "auto"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "auto"
|
||||||
|
label = "Auto-detect (best available)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "openai"
|
||||||
|
label = "OpenAI (image, TTS)"
|
||||||
|
provider_env = "OPENAI_API_KEY"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "minimax"
|
||||||
|
label = "MiniMax (image, TTS, video, music)"
|
||||||
|
provider_env = "MINIMAX_API_KEY"
|
||||||
|
|
||||||
|
[[settings]]
|
||||||
|
key = "image_model"
|
||||||
|
label = "Image Model"
|
||||||
|
description = "Model to use for image generation"
|
||||||
|
setting_type = "select"
|
||||||
|
default = "auto"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "auto"
|
||||||
|
label = "Provider default"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "gpt-image-1"
|
||||||
|
label = "GPT Image 1 (OpenAI)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "dall-e-3"
|
||||||
|
label = "DALL-E 3 (OpenAI)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "image-01"
|
||||||
|
label = "Image-01 (MiniMax)"
|
||||||
|
|
||||||
|
[[settings]]
|
||||||
|
key = "image_size"
|
||||||
|
label = "Default Image Size"
|
||||||
|
description = "Default resolution for generated images"
|
||||||
|
setting_type = "select"
|
||||||
|
default = "1024x1024"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "1024x1024"
|
||||||
|
label = "1024x1024 (square)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "1792x1024"
|
||||||
|
label = "1792x1024 (landscape)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "1024x1792"
|
||||||
|
label = "1024x1792 (portrait)"
|
||||||
|
|
||||||
|
[[settings]]
|
||||||
|
key = "tts_voice"
|
||||||
|
label = "TTS Voice"
|
||||||
|
description = "Default voice for text-to-speech"
|
||||||
|
setting_type = "select"
|
||||||
|
default = "alloy"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "alloy"
|
||||||
|
label = "Alloy (neutral)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "echo"
|
||||||
|
label = "Echo (male)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "fable"
|
||||||
|
label = "Fable (storytelling)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "nova"
|
||||||
|
label = "Nova (female)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "onyx"
|
||||||
|
label = "Onyx (deep male)"
|
||||||
|
|
||||||
|
[[settings.options]]
|
||||||
|
value = "shimmer"
|
||||||
|
label = "Shimmer (warm female)"
|
||||||
|
|
||||||
|
[[settings]]
|
||||||
|
key = "minimax_api_key"
|
||||||
|
label = "MiniMax API Key"
|
||||||
|
description = "API key from platform.minimaxi.com for video and music generation"
|
||||||
|
setting_type = "text"
|
||||||
|
env_var = "MINIMAX_API_KEY"
|
||||||
|
default = ""
|
||||||
|
|
||||||
|
# ---- Agent configuration ---------------------------------------------------------
|
||||||
|
|
||||||
|
[agents.main]
|
||||||
|
coordinator = true
|
||||||
|
name = "creator-hand"
|
||||||
|
description = "AI media studio — generates images, videos, music, and speech from natural language"
|
||||||
|
module = "builtin:chat"
|
||||||
|
provider = "default"
|
||||||
|
model = "default"
|
||||||
|
max_tokens = 8192
|
||||||
|
temperature = 0.5
|
||||||
|
max_iterations = 30
|
||||||
|
system_prompt = """You are Creator Hand — an AI media studio that generates images, videos, music, and speech from natural language requests.
|
||||||
|
|
||||||
|
## Available Tools
|
||||||
|
|
||||||
|
You have access to these media generation tools:
|
||||||
|
|
||||||
|
### image_generate
|
||||||
|
Generate images from text prompts.
|
||||||
|
- Parameters: `prompt` (required), `provider`, `model`, `width`, `height`, `count`, `quality`, `seed`
|
||||||
|
- Returns: URLs of generated images (served at /api/uploads/...)
|
||||||
|
- Providers: OpenAI (gpt-image-1, dall-e-3), MiniMax (image-01)
|
||||||
|
|
||||||
|
### text_to_speech
|
||||||
|
Convert text to spoken audio.
|
||||||
|
- Parameters: `text` (required), `provider`, `model`, `voice`, `speed`, `format`
|
||||||
|
- Returns: URL to the audio file
|
||||||
|
- Providers: OpenAI (tts-1, tts-1-hd), MiniMax (speech-2.8-hd)
|
||||||
|
|
||||||
|
### video_generate
|
||||||
|
Submit a video generation task (asynchronous).
|
||||||
|
- Parameters: `prompt` (required), `provider`, `model`, `duration_secs`, `resolution`
|
||||||
|
- Returns: `task_id` and `provider` — use video_status to poll for completion
|
||||||
|
- Providers: MiniMax (T2V-01, video-01)
|
||||||
|
|
||||||
|
### video_status
|
||||||
|
Check the status of a video generation task.
|
||||||
|
- Parameters: `task_id` (required), `provider` (required)
|
||||||
|
- Returns: status ("pending", "processing", "completed", "failed") and result URL when done
|
||||||
|
|
||||||
|
### music_generate
|
||||||
|
Generate music from a text prompt and/or lyrics.
|
||||||
|
- Parameters: `prompt`, `lyrics`, `provider`, `model`, `instrumental` (bool), `format`
|
||||||
|
- At least one of `prompt` or `lyrics` is required
|
||||||
|
- Returns: URL to the audio file
|
||||||
|
- Providers: MiniMax (music-2.5)
|
||||||
|
|
||||||
|
## Workflow Guidelines
|
||||||
|
|
||||||
|
1. **Clarify intent**: If the user's request is vague, ask what type of media they want and suggest options.
|
||||||
|
|
||||||
|
2. **Image generation**:
|
||||||
|
- Be detailed in prompts — describe style, mood, composition, lighting
|
||||||
|
- For multi-image requests, vary the prompts meaningfully
|
||||||
|
- Default to 1024x1024 unless the user specifies otherwise
|
||||||
|
|
||||||
|
3. **Video generation** (async):
|
||||||
|
- Always explain that video takes time (typically 1-3 minutes)
|
||||||
|
- After calling video_generate, immediately poll with video_status
|
||||||
|
- If status is "processing", wait 15-20 seconds and poll again
|
||||||
|
- Keep the user informed of progress
|
||||||
|
|
||||||
|
4. **Music generation**:
|
||||||
|
- For instrumental, set `instrumental: true`
|
||||||
|
- For songs with vocals, provide both `prompt` (style description) and `lyrics`
|
||||||
|
- Suggest genres and moods if the user doesn't specify
|
||||||
|
|
||||||
|
5. **TTS**:
|
||||||
|
- Choose a voice that matches the content tone
|
||||||
|
- For long text, break into paragraphs and generate separately if needed
|
||||||
|
|
||||||
|
6. **Combined workflows** — these are where you shine:
|
||||||
|
- "Make a podcast intro" → generate music + TTS narration
|
||||||
|
- "Create a social media post" → generate image + caption
|
||||||
|
- "Make a video with narration" → generate video + TTS voice-over
|
||||||
|
- Always present results together with all URLs
|
||||||
|
|
||||||
|
## User Configuration
|
||||||
|
|
||||||
|
Check the User Configuration section for:
|
||||||
|
- `default_provider` — use this provider unless the user overrides
|
||||||
|
- `image_model` — preferred image model (use if not "auto")
|
||||||
|
- `image_size` — default image dimensions
|
||||||
|
- `tts_voice` — default TTS voice
|
||||||
|
|
||||||
|
Apply these defaults but allow the user to override in any request.
|
||||||
|
|
||||||
|
## Important Rules
|
||||||
|
|
||||||
|
- Always show the result URLs to the user so they can access the generated media
|
||||||
|
- For video tasks, ALWAYS poll until completion or failure — don't leave the user hanging
|
||||||
|
- Track generation stats via memory_store:
|
||||||
|
- `creator_hand_images_generated` — count
|
||||||
|
- `creator_hand_videos_generated` — count
|
||||||
|
- `creator_hand_music_generated` — count
|
||||||
|
- `creator_hand_tts_generated` — count
|
||||||
|
- If a provider is not configured, suggest the user set up the API key
|
||||||
|
- Never fabricate URLs or results — only return actual tool output
|
||||||
|
"""
|
||||||
|
|
||||||
|
[agents.prompt_writer]
|
||||||
|
invoke_hint = "Creative prompt engineering — writing detailed, effective prompts for image, video, and music generation"
|
||||||
|
name = "prompt-writer"
|
||||||
|
description = "Prompt engineer that crafts detailed, effective prompts for media generation"
|
||||||
|
module = "builtin:chat"
|
||||||
|
provider = "default"
|
||||||
|
model = "default"
|
||||||
|
max_tokens = 4096
|
||||||
|
temperature = 0.8
|
||||||
|
system_prompt = """You are Prompt Writer, a creative prompt engineer within the Creator Hand.
|
||||||
|
|
||||||
|
Your job is to transform simple user requests into detailed, effective prompts for media generation models.
|
||||||
|
|
||||||
|
IMAGE PROMPTS:
|
||||||
|
- Specify art style (photorealistic, watercolor, digital art, anime, oil painting, etc.)
|
||||||
|
- Include composition details (wide shot, close-up, bird's eye view, etc.)
|
||||||
|
- Describe lighting (golden hour, studio lighting, dramatic shadows, etc.)
|
||||||
|
- Add mood/atmosphere (serene, chaotic, mysterious, vibrant, etc.)
|
||||||
|
- Include technical details when relevant (depth of field, lens type, etc.)
|
||||||
|
|
||||||
|
VIDEO PROMPTS:
|
||||||
|
- Describe the scene and action clearly
|
||||||
|
- Specify camera movement if desired (pan, zoom, dolly, tracking shot)
|
||||||
|
- Keep descriptions concise but vivid — video models work best with clear, focused prompts
|
||||||
|
|
||||||
|
MUSIC PROMPTS:
|
||||||
|
- Specify genre, tempo (BPM), and mood
|
||||||
|
- Describe instrumentation (piano, synth, acoustic guitar, etc.)
|
||||||
|
- Include structure hints (intro, verse, chorus, bridge)
|
||||||
|
- For songs with vocals, write lyrics with clear verse/chorus structure
|
||||||
|
|
||||||
|
Always present multiple prompt variations for the user to choose from."""
|
||||||
|
|
||||||
|
# ---- Dashboard metrics -----------------------------------------------------------
|
||||||
|
|
||||||
|
[dashboard]
|
||||||
|
[[dashboard.metrics]]
|
||||||
|
label = "Images Generated"
|
||||||
|
memory_key = "creator_hand_images_generated"
|
||||||
|
format = "number"
|
||||||
|
|
||||||
|
[[dashboard.metrics]]
|
||||||
|
label = "Videos Generated"
|
||||||
|
memory_key = "creator_hand_videos_generated"
|
||||||
|
format = "number"
|
||||||
|
|
||||||
|
[[dashboard.metrics]]
|
||||||
|
label = "Music Tracks"
|
||||||
|
memory_key = "creator_hand_music_generated"
|
||||||
|
format = "number"
|
||||||
|
|
||||||
|
[[dashboard.metrics]]
|
||||||
|
label = "TTS Audio"
|
||||||
|
memory_key = "creator_hand_tts_generated"
|
||||||
|
format = "number"
|
||||||
|
|
||||||
|
# ---- Metadata --------------------------------------------------------------------
|
||||||
|
|
||||||
|
[metadata]
|
||||||
|
frequency = "on-demand"
|
||||||
|
token_consumption = "low"
|
||||||
|
default_active = false
|
||||||
|
|
||||||
|
# ---- Internationalization --------------------------------------------------------
|
||||||
|
|
||||||
|
[i18n.zh]
|
||||||
|
name = "创作 Hand"
|
||||||
|
description = "AI 媒体工作室 — 通过文本指令生成图像、视频、音乐和语音"
|
||||||
|
category = "内容"
|
||||||
|
|
||||||
|
[i18n.zh.settings.default_provider]
|
||||||
|
label = "首选提供商"
|
||||||
|
description = "默认使用哪个提供商。自动模式会为每种能力选择首个可用的提供商。"
|
||||||
|
|
||||||
|
[i18n.zh.settings.image_model]
|
||||||
|
label = "图像模型"
|
||||||
|
description = "用于图像生成的模型"
|
||||||
|
|
||||||
|
[i18n.zh.settings.image_size]
|
||||||
|
label = "默认图像尺寸"
|
||||||
|
description = "生成图像的默认分辨率"
|
||||||
|
|
||||||
|
[i18n.zh.settings.tts_voice]
|
||||||
|
label = "TTS 语音"
|
||||||
|
description = "文字转语音的默认声音"
|
||||||
|
|
||||||
|
[i18n.zh.settings.minimax_api_key]
|
||||||
|
label = "MiniMax API 密钥"
|
||||||
|
description = "来自 platform.minimaxi.com 的 API 密钥,用于视频和音乐生成"
|
||||||
|
|
||||||
|
[i18n.ja]
|
||||||
|
name = "クリエイター Hand"
|
||||||
|
description = "AIメディアスタジオ — テキストプロンプトから画像、動画、音楽、音声を生成"
|
||||||
|
category = "コンテンツ"
|
||||||
|
|
||||||
|
[i18n.ja.settings.default_provider]
|
||||||
|
label = "優先プロバイダー"
|
||||||
|
description = "デフォルトで使用するプロバイダー。自動は各機能で最初に設定されたプロバイダーを選択。"
|
||||||
|
|
||||||
|
[i18n.ja.settings.image_model]
|
||||||
|
label = "画像モデル"
|
||||||
|
description = "画像生成に使用するモデル"
|
||||||
|
|
||||||
|
[i18n.ja.settings.image_size]
|
||||||
|
label = "デフォルト画像サイズ"
|
||||||
|
description = "生成画像のデフォルト解像度"
|
||||||
|
|
||||||
|
[i18n.ja.settings.tts_voice]
|
||||||
|
label = "TTS音声"
|
||||||
|
description = "テキスト読み上げのデフォルト音声"
|
||||||
|
|
||||||
|
[i18n.ja.settings.minimax_api_key]
|
||||||
|
label = "MiniMax APIキー"
|
||||||
|
description = "platform.minimaxi.comからのAPIキー。動画と音楽生成に使用。"
|
||||||
|
|
||||||
|
[i18n.ko]
|
||||||
|
name = "크리에이터 Hand"
|
||||||
|
description = "AI 미디어 스튜디오 — 텍스트로 이미지, 동영상, 음악, 음성 생성"
|
||||||
|
category = "콘텐츠"
|
||||||
|
|
||||||
|
[i18n.ko.settings.default_provider]
|
||||||
|
label = "기본 공급자"
|
||||||
|
description = "기본으로 사용할 공급자. 자동은 각 기능에 대해 첫 번째 구성된 공급자를 선택."
|
||||||
|
|
||||||
|
[i18n.ko.settings.image_model]
|
||||||
|
label = "이미지 모델"
|
||||||
|
description = "이미지 생성에 사용할 모델"
|
||||||
|
|
||||||
|
[i18n.ko.settings.image_size]
|
||||||
|
label = "기본 이미지 크기"
|
||||||
|
description = "생성 이미지의 기본 해상도"
|
||||||
|
|
||||||
|
[i18n.ko.settings.tts_voice]
|
||||||
|
label = "TTS 음성"
|
||||||
|
description = "텍스트-음성 변환의 기본 음성"
|
||||||
|
|
||||||
|
[i18n.ko.settings.minimax_api_key]
|
||||||
|
label = "MiniMax API 키"
|
||||||
|
description = "platform.minimaxi.com에서 발급한 API 키. 동영상 및 음악 생성에 사용."
|
||||||
|
|
||||||
|
[i18n.es]
|
||||||
|
name = "Hand Creador"
|
||||||
|
description = "Estudio de medios IA — genera imágenes, videos, música y voz a partir de texto"
|
||||||
|
category = "Contenido"
|
||||||
|
|
||||||
|
[i18n.fr]
|
||||||
|
name = "Hand Cr\u00e9ateur"
|
||||||
|
description = "Studio m\u00e9dia IA — g\u00e9n\u00e8re images, vid\u00e9os, musique et voix \u00e0 partir de texte"
|
||||||
|
category = "Contenu"
|
||||||
|
|
||||||
|
[i18n.de]
|
||||||
|
name = "Kreator-Hand"
|
||||||
|
description = "KI-Medienstudio — erzeugt Bilder, Videos, Musik und Sprache aus Text"
|
||||||
|
category = "Inhalt"
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
# Creator Hand
|
||||||
|
|
||||||
|
AI media studio -- generates images, videos, music, and speech from natural language prompts.
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
| Field | Value |
|
||||||
|
|-------|-------|
|
||||||
|
| Category | `content` |
|
||||||
|
| Agents | `creator-hand` (coordinator), `prompt-writer` |
|
||||||
|
| Routing | `generate image`, `create video`, `make music`, `text to speech`, `media generation` |
|
||||||
|
|
||||||
|
## Integrations
|
||||||
|
|
||||||
|
- **OpenAI API** -- Image generation (gpt-image-1, DALL-E 3) and text-to-speech (tts-1).
|
||||||
|
- **MiniMax API** -- Image, TTS, video generation (Hailuo T2V-01), and music generation (music-2.5).
|
||||||
|
|
||||||
|
## Provider Capabilities
|
||||||
|
|
||||||
|
| Provider | Image | TTS | Video | Music |
|
||||||
|
|----------|-------|-----|-------|-------|
|
||||||
|
| OpenAI | gpt-image-1, dall-e-3 | tts-1, tts-1-hd | -- | -- |
|
||||||
|
| MiniMax | image-01 | speech-2.8-hd | T2V-01 | music-2.5 |
|
||||||
|
|
||||||
|
## Settings
|
||||||
|
|
||||||
|
- **Preferred Provider** -- `auto`, `openai`, `minimax`
|
||||||
|
- **Image Model** -- `auto`, `gpt-image-1`, `dall-e-3`, `image-01`
|
||||||
|
- **Default Image Size** -- `1024x1024`, `1792x1024`, `1024x1792`
|
||||||
|
- **TTS Voice** -- `alloy`, `echo`, `fable`, `nova`, `onyx`, `shimmer`
|
||||||
|
- **MiniMax API Key** -- API key from platform.minimaxi.com
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```bash
|
||||||
|
librefang hand run creator
|
||||||
|
```
|
||||||
|
|
||||||
|
### Examples
|
||||||
|
|
||||||
|
```
|
||||||
|
> Generate a watercolor painting of a mountain lake at sunrise
|
||||||
|
> Create a 5-second video of ocean waves crashing on rocks
|
||||||
|
> Make an upbeat electronic jingle, 15 seconds, instrumental
|
||||||
|
> Read this text aloud in a warm female voice: "Welcome to..."
|
||||||
|
> Make a podcast intro: jingle + voice saying "Welcome to Tech Talk"
|
||||||
|
```
|
||||||
@@ -0,0 +1,249 @@
|
|||||||
|
---
|
||||||
|
name: media-generation-skill
|
||||||
|
version: "1.0.0"
|
||||||
|
description: "Expert knowledge for AI media generation — image prompting, video workflows, music composition, and TTS best practices"
|
||||||
|
runtime: prompt_only
|
||||||
|
---
|
||||||
|
|
||||||
|
# Media Generation Expert Knowledge
|
||||||
|
|
||||||
|
## Tool Reference
|
||||||
|
|
||||||
|
### image_generate
|
||||||
|
|
||||||
|
Generate images from text prompts via OpenAI or MiniMax.
|
||||||
|
|
||||||
|
**Parameters:**
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Default | Description |
|
||||||
|
|-----------|------|----------|---------|-------------|
|
||||||
|
| `prompt` | string | yes | — | Text description of the desired image |
|
||||||
|
| `provider` | string | no | auto | `openai` or `minimax` |
|
||||||
|
| `model` | string | no | provider default | `gpt-image-1`, `dall-e-3`, `image-01` |
|
||||||
|
| `width` | int | no | 1024 | Image width in pixels |
|
||||||
|
| `height` | int | no | 1024 | Image height in pixels |
|
||||||
|
| `count` | int | no | 1 | Number of images (1-4) |
|
||||||
|
| `quality` | string | no | `auto` | `low`, `medium`, `high`, `auto` |
|
||||||
|
| `seed` | int | no | random | Reproducibility seed |
|
||||||
|
|
||||||
|
**Provider-specific notes:**
|
||||||
|
|
||||||
|
- **OpenAI gpt-image-1**: Best for photorealistic and creative images. Supports inpainting hints in prompt. Sizes: 1024x1024, 1792x1024, 1024x1792.
|
||||||
|
- **OpenAI dall-e-3**: Good quality, may revise your prompt (check `revised_prompt` in response). Only generates 1 image per call.
|
||||||
|
- **MiniMax image-01**: Fast generation, good for illustrations and concept art. Supports arbitrary aspect ratios.
|
||||||
|
|
||||||
|
**Result:** Returns `images` array with `url` fields pointing to `/api/uploads/{id}`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### text_to_speech
|
||||||
|
|
||||||
|
Convert text to spoken audio.
|
||||||
|
|
||||||
|
**Parameters:**
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Default | Description |
|
||||||
|
|-----------|------|----------|---------|-------------|
|
||||||
|
| `text` | string | yes | — | Text to speak (max ~4096 chars per call) |
|
||||||
|
| `provider` | string | no | auto | `openai` or `minimax` |
|
||||||
|
| `model` | string | no | provider default | `tts-1`, `tts-1-hd`, `speech-2.8-hd` |
|
||||||
|
| `voice` | string | no | `alloy` | Voice selection (see table below) |
|
||||||
|
| `speed` | float | no | 1.0 | Playback speed (0.25 - 4.0) |
|
||||||
|
| `format` | string | no | `mp3` | `mp3`, `wav`, `flac`, `opus`, `aac` |
|
||||||
|
|
||||||
|
**OpenAI voices:**
|
||||||
|
|
||||||
|
| Voice | Character |
|
||||||
|
|-------|-----------|
|
||||||
|
| `alloy` | Neutral, balanced |
|
||||||
|
| `echo` | Male, warm |
|
||||||
|
| `fable` | Storytelling, expressive |
|
||||||
|
| `nova` | Female, friendly |
|
||||||
|
| `onyx` | Deep male, authoritative |
|
||||||
|
| `shimmer` | Warm female, gentle |
|
||||||
|
|
||||||
|
**MiniMax voices:**
|
||||||
|
|
||||||
|
| Voice | Character |
|
||||||
|
|-------|-----------|
|
||||||
|
| `English_Graceful_Lady` | Female, elegant |
|
||||||
|
| `English_Calm_Man` | Male, composed |
|
||||||
|
| `English_Energetic_Girl` | Female, upbeat |
|
||||||
|
|
||||||
|
**Tips:**
|
||||||
|
- For long content, split at paragraph boundaries to keep natural pacing
|
||||||
|
- `tts-1-hd` is higher quality but slower; use `tts-1` for drafts
|
||||||
|
- Speed 0.8-0.9 works well for narration; 1.1-1.2 for summaries
|
||||||
|
|
||||||
|
**Result:** Returns `url` to the audio file, `format`, `duration_ms`, `sample_rate`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### video_generate
|
||||||
|
|
||||||
|
Submit an asynchronous video generation task. Video generation takes 1-3 minutes.
|
||||||
|
|
||||||
|
**Parameters:**
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Default | Description |
|
||||||
|
|-----------|------|----------|---------|-------------|
|
||||||
|
| `prompt` | string | yes | — | Scene description |
|
||||||
|
| `provider` | string | no | auto | Currently only `minimax` |
|
||||||
|
| `model` | string | no | `T2V-01` | Video model |
|
||||||
|
| `duration_secs` | int | no | 5 | Video duration (5-10 seconds) |
|
||||||
|
| `resolution` | string | no | `1080p` | `720p`, `1080p` |
|
||||||
|
|
||||||
|
**Prompt writing for video:**
|
||||||
|
- Be specific about the scene, subject, and action
|
||||||
|
- Describe camera movement explicitly: "slow pan left", "zoom in", "static shot"
|
||||||
|
- Keep it focused — one scene per generation works best
|
||||||
|
- Include lighting and atmosphere: "golden hour lighting", "neon-lit street at night"
|
||||||
|
- Avoid complex multi-character interactions (current models handle single subjects best)
|
||||||
|
|
||||||
|
**Good prompts:**
|
||||||
|
- "A golden retriever running through a wheat field at sunset, slow motion, cinematic"
|
||||||
|
- "Close-up of coffee being poured into a ceramic cup, steam rising, warm morning light"
|
||||||
|
- "Aerial drone shot flying over a tropical coastline, turquoise water, white sand beach"
|
||||||
|
|
||||||
|
**Bad prompts:**
|
||||||
|
- "A video" (too vague)
|
||||||
|
- "Two people having a conversation at a cafe while a dog runs by and a car crashes outside" (too complex)
|
||||||
|
|
||||||
|
**Result:** Returns `task_id` and `provider`. You MUST poll with `video_status`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### video_status
|
||||||
|
|
||||||
|
Poll the status of a video generation task.
|
||||||
|
|
||||||
|
**Parameters:**
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Description |
|
||||||
|
|-----------|------|----------|-------------|
|
||||||
|
| `task_id` | string | yes | From video_generate response |
|
||||||
|
| `provider` | string | yes | Must match the provider from video_generate |
|
||||||
|
|
||||||
|
**Statuses:**
|
||||||
|
|
||||||
|
| Status | Meaning | Action |
|
||||||
|
|--------|---------|--------|
|
||||||
|
| `pending` | Queued, not started | Wait 10-15s, poll again |
|
||||||
|
| `processing` | Actively generating | Wait 15-20s, poll again |
|
||||||
|
| `completed` | Done | Result includes `file_url` |
|
||||||
|
| `failed` | Generation failed | Check error message, may retry with different prompt |
|
||||||
|
|
||||||
|
**Polling pattern:**
|
||||||
|
1. Call video_generate → get task_id
|
||||||
|
2. Wait 10 seconds
|
||||||
|
3. Call video_status with task_id + provider
|
||||||
|
4. If not completed, wait 15-20 seconds and poll again
|
||||||
|
5. Maximum ~10 polls (about 3 minutes total)
|
||||||
|
6. Always inform the user of current status
|
||||||
|
|
||||||
|
**Result (completed):** Returns `file_url`, `width`, `height`, `duration_secs`, `provider`, `model`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### music_generate
|
||||||
|
|
||||||
|
Generate music from a text prompt and/or lyrics.
|
||||||
|
|
||||||
|
**Parameters:**
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Default | Description |
|
||||||
|
|-----------|------|----------|---------|-------------|
|
||||||
|
| `prompt` | string | no* | — | Style/mood description |
|
||||||
|
| `lyrics` | string | no* | — | Song lyrics with structure |
|
||||||
|
| `provider` | string | no | auto | Currently only `minimax` |
|
||||||
|
| `model` | string | no | `music-2.5` | Music model |
|
||||||
|
| `instrumental` | bool | no | false | Generate without vocals |
|
||||||
|
| `format` | string | no | `mp3` | `mp3`, `wav`, `flac` |
|
||||||
|
|
||||||
|
*At least one of `prompt` or `lyrics` is required.
|
||||||
|
|
||||||
|
**Prompt writing for music:**
|
||||||
|
|
||||||
|
For instrumentals, describe:
|
||||||
|
- Genre: electronic, jazz, classical, hip-hop, rock, ambient, lo-fi
|
||||||
|
- Tempo: slow (60-80 BPM), medium (100-120 BPM), fast (130-160 BPM)
|
||||||
|
- Mood: uplifting, melancholic, energetic, relaxing, dramatic, mysterious
|
||||||
|
- Instruments: piano, synth, acoustic guitar, strings, drums, bass
|
||||||
|
|
||||||
|
For songs with vocals, provide lyrics with structure markers:
|
||||||
|
|
||||||
|
```
|
||||||
|
[Verse 1]
|
||||||
|
Walking down the empty street
|
||||||
|
Moonlight dancing at my feet
|
||||||
|
|
||||||
|
[Chorus]
|
||||||
|
This is where the night begins
|
||||||
|
Let the music pull us in
|
||||||
|
|
||||||
|
[Verse 2]
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
**Good prompts:**
|
||||||
|
- `prompt`: "Chill lo-fi hip-hop beat, vinyl crackle, mellow piano chords, 85 BPM, rainy day vibe"
|
||||||
|
- `prompt`: "Epic orchestral trailer music, building tension, brass and strings, 140 BPM"
|
||||||
|
- `prompt` + `lyrics`: "Indie folk acoustic ballad, fingerpicking guitar, gentle male vocals" with lyrics
|
||||||
|
|
||||||
|
**Result:** Returns `url` to audio file, `format`, `duration_ms`, `sample_rate`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Combined Workflow Recipes
|
||||||
|
|
||||||
|
### Podcast Intro
|
||||||
|
1. `music_generate` — instrumental jingle, 10-15 seconds, upbeat
|
||||||
|
2. `text_to_speech` — "Welcome to [show name]..." with energetic voice
|
||||||
|
3. Report both URLs to user
|
||||||
|
|
||||||
|
### Social Media Post
|
||||||
|
1. `image_generate` — eye-catching visual for the post
|
||||||
|
2. Suggest caption text based on the image
|
||||||
|
3. Optionally `text_to_speech` for accessibility audio version
|
||||||
|
|
||||||
|
### Video with Narration
|
||||||
|
1. `text_to_speech` — generate narration audio
|
||||||
|
2. `video_generate` — generate matching video clip
|
||||||
|
3. `video_status` — poll until complete
|
||||||
|
4. Report both URLs (user can combine with ffmpeg or editing tools)
|
||||||
|
|
||||||
|
### Album Art + Preview
|
||||||
|
1. `image_generate` — album cover artwork
|
||||||
|
2. `music_generate` — short preview track matching the artwork mood
|
||||||
|
3. Present together
|
||||||
|
|
||||||
|
### Audiobook Chapter
|
||||||
|
1. Split text into sections (~500 words each)
|
||||||
|
2. `text_to_speech` for each section with consistent voice
|
||||||
|
3. Report all audio URLs in order
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Error Handling
|
||||||
|
|
||||||
|
| Error | Cause | Fix |
|
||||||
|
|-------|-------|-----|
|
||||||
|
| `missing_key` | API key not configured | Ask user to set OPENAI_API_KEY or MINIMAX_API_KEY |
|
||||||
|
| `not_supported` | Provider doesn't support this modality | Switch to a provider that does |
|
||||||
|
| `content_filtered` | Safety filter rejected the prompt | Rephrase without prohibited content |
|
||||||
|
| `rate_limited` | Too many requests | Wait 30-60 seconds and retry |
|
||||||
|
| `invalid_request` | Bad parameters | Check parameter ranges (e.g., count 1-4, speed 0.25-4.0) |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Provider Capability Matrix
|
||||||
|
|
||||||
|
| Capability | OpenAI | MiniMax |
|
||||||
|
|------------|--------|---------|
|
||||||
|
| Image generation | gpt-image-1, dall-e-3 | image-01 |
|
||||||
|
| Text-to-speech | tts-1, tts-1-hd | speech-2.8-hd |
|
||||||
|
| Video generation | — | T2V-01, video-01 |
|
||||||
|
| Music generation | — | music-2.5 |
|
||||||
|
|
||||||
|
**Auto-detection priority:** OpenAI > MiniMax (for capabilities both support).
|
||||||
|
If only MiniMax key is set, all 4 modalities are available through MiniMax.
|
||||||
Reference in new issue
Block a user