feat(hands): add Creator Hand for media generation (#17)
This commit is contained in:
3 files changed
+715
No files matched your search
@@ -0,0 +1,419 @@
|
||||
id = "creator"
|
||||
version = "1.0.0"
|
||||
name = "Creator Hand"
|
||||
description = "AI media studio — generates images, videos, music, and speech from text prompts"
|
||||
|
||||
category = "content"
|
||||
icon = "\U0001F3A8"
|
||||
tools = [
|
||||
"image_generate",
|
||||
"video_generate",
|
||||
"video_status",
|
||||
"music_generate",
|
||||
"text_to_speech",
|
||||
"file_read",
|
||||
"file_write",
|
||||
"file_list",
|
||||
"web_fetch",
|
||||
"memory_store",
|
||||
"memory_recall",
|
||||
]
|
||||
|
||||
[routing]
|
||||
aliases = [
|
||||
"generate image",
|
||||
"create image",
|
||||
"make a picture",
|
||||
"generate video",
|
||||
"create video",
|
||||
"make a video",
|
||||
"generate music",
|
||||
"create music",
|
||||
"compose music",
|
||||
"text to speech",
|
||||
"generate speech",
|
||||
"voice over",
|
||||
"media generation",
|
||||
]
|
||||
weak_aliases = [
|
||||
"illustration",
|
||||
"artwork",
|
||||
"album cover",
|
||||
"background music",
|
||||
"jingle",
|
||||
"narration",
|
||||
"audio",
|
||||
"thumbnail",
|
||||
"poster",
|
||||
"banner",
|
||||
]
|
||||
|
||||
# ---- Requirements ----------------------------------------------------------------
|
||||
|
||||
[[requires]]
|
||||
key = "media_provider"
|
||||
label = "At least one media provider API key must be set"
|
||||
requirement_type = "any_env_var"
|
||||
check_value = "OPENAI_API_KEY,MINIMAX_API_KEY"
|
||||
description = "Creator Hand needs at least one configured media provider. OpenAI supports image and TTS; MiniMax supports image, TTS, video, and music."
|
||||
optional = false
|
||||
|
||||
# ---- Settings --------------------------------------------------------------------
|
||||
|
||||
[[settings]]
|
||||
key = "default_provider"
|
||||
label = "Preferred Provider"
|
||||
description = "Which provider to use by default. Auto will pick the first configured provider for each capability."
|
||||
setting_type = "select"
|
||||
default = "auto"
|
||||
|
||||
[[settings.options]]
|
||||
value = "auto"
|
||||
label = "Auto-detect (best available)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "openai"
|
||||
label = "OpenAI (image, TTS)"
|
||||
provider_env = "OPENAI_API_KEY"
|
||||
|
||||
[[settings.options]]
|
||||
value = "minimax"
|
||||
label = "MiniMax (image, TTS, video, music)"
|
||||
provider_env = "MINIMAX_API_KEY"
|
||||
|
||||
[[settings]]
|
||||
key = "image_model"
|
||||
label = "Image Model"
|
||||
description = "Model to use for image generation"
|
||||
setting_type = "select"
|
||||
default = "auto"
|
||||
|
||||
[[settings.options]]
|
||||
value = "auto"
|
||||
label = "Provider default"
|
||||
|
||||
[[settings.options]]
|
||||
value = "gpt-image-1"
|
||||
label = "GPT Image 1 (OpenAI)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "dall-e-3"
|
||||
label = "DALL-E 3 (OpenAI)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "image-01"
|
||||
label = "Image-01 (MiniMax)"
|
||||
|
||||
[[settings]]
|
||||
key = "image_size"
|
||||
label = "Default Image Size"
|
||||
description = "Default resolution for generated images"
|
||||
setting_type = "select"
|
||||
default = "1024x1024"
|
||||
|
||||
[[settings.options]]
|
||||
value = "1024x1024"
|
||||
label = "1024x1024 (square)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "1792x1024"
|
||||
label = "1792x1024 (landscape)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "1024x1792"
|
||||
label = "1024x1792 (portrait)"
|
||||
|
||||
[[settings]]
|
||||
key = "tts_voice"
|
||||
label = "TTS Voice"
|
||||
description = "Default voice for text-to-speech"
|
||||
setting_type = "select"
|
||||
default = "alloy"
|
||||
|
||||
[[settings.options]]
|
||||
value = "alloy"
|
||||
label = "Alloy (neutral)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "echo"
|
||||
label = "Echo (male)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "fable"
|
||||
label = "Fable (storytelling)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "nova"
|
||||
label = "Nova (female)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "onyx"
|
||||
label = "Onyx (deep male)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "shimmer"
|
||||
label = "Shimmer (warm female)"
|
||||
|
||||
[[settings]]
|
||||
key = "minimax_api_key"
|
||||
label = "MiniMax API Key"
|
||||
description = "API key from platform.minimaxi.com for video and music generation"
|
||||
setting_type = "text"
|
||||
env_var = "MINIMAX_API_KEY"
|
||||
default = ""
|
||||
|
||||
# ---- Agent configuration ---------------------------------------------------------
|
||||
|
||||
[agents.main]
|
||||
coordinator = true
|
||||
name = "creator-hand"
|
||||
description = "AI media studio — generates images, videos, music, and speech from natural language"
|
||||
module = "builtin:chat"
|
||||
provider = "default"
|
||||
model = "default"
|
||||
max_tokens = 8192
|
||||
temperature = 0.5
|
||||
max_iterations = 30
|
||||
system_prompt = """You are Creator Hand — an AI media studio that generates images, videos, music, and speech from natural language requests.
|
||||
|
||||
## Available Tools
|
||||
|
||||
You have access to these media generation tools:
|
||||
|
||||
### image_generate
|
||||
Generate images from text prompts.
|
||||
- Parameters: `prompt` (required), `provider`, `model`, `width`, `height`, `count`, `quality`, `seed`
|
||||
- Returns: URLs of generated images (served at /api/uploads/...)
|
||||
- Providers: OpenAI (gpt-image-1, dall-e-3), MiniMax (image-01)
|
||||
|
||||
### text_to_speech
|
||||
Convert text to spoken audio.
|
||||
- Parameters: `text` (required), `provider`, `model`, `voice`, `speed`, `format`
|
||||
- Returns: URL to the audio file
|
||||
- Providers: OpenAI (tts-1, tts-1-hd), MiniMax (speech-2.8-hd)
|
||||
|
||||
### video_generate
|
||||
Submit a video generation task (asynchronous).
|
||||
- Parameters: `prompt` (required), `provider`, `model`, `duration_secs`, `resolution`
|
||||
- Returns: `task_id` and `provider` — use video_status to poll for completion
|
||||
- Providers: MiniMax (T2V-01, video-01)
|
||||
|
||||
### video_status
|
||||
Check the status of a video generation task.
|
||||
- Parameters: `task_id` (required), `provider` (required)
|
||||
- Returns: status ("pending", "processing", "completed", "failed") and result URL when done
|
||||
|
||||
### music_generate
|
||||
Generate music from a text prompt and/or lyrics.
|
||||
- Parameters: `prompt`, `lyrics`, `provider`, `model`, `instrumental` (bool), `format`
|
||||
- At least one of `prompt` or `lyrics` is required
|
||||
- Returns: URL to the audio file
|
||||
- Providers: MiniMax (music-2.5)
|
||||
|
||||
## Workflow Guidelines
|
||||
|
||||
1. **Clarify intent**: If the user's request is vague, ask what type of media they want and suggest options.
|
||||
|
||||
2. **Image generation**:
|
||||
- Be detailed in prompts — describe style, mood, composition, lighting
|
||||
- For multi-image requests, vary the prompts meaningfully
|
||||
- Default to 1024x1024 unless the user specifies otherwise
|
||||
|
||||
3. **Video generation** (async):
|
||||
- Always explain that video takes time (typically 1-3 minutes)
|
||||
- After calling video_generate, immediately poll with video_status
|
||||
- If status is "processing", wait 15-20 seconds and poll again
|
||||
- Keep the user informed of progress
|
||||
|
||||
4. **Music generation**:
|
||||
- For instrumental, set `instrumental: true`
|
||||
- For songs with vocals, provide both `prompt` (style description) and `lyrics`
|
||||
- Suggest genres and moods if the user doesn't specify
|
||||
|
||||
5. **TTS**:
|
||||
- Choose a voice that matches the content tone
|
||||
- For long text, break into paragraphs and generate separately if needed
|
||||
|
||||
6. **Combined workflows** — these are where you shine:
|
||||
- "Make a podcast intro" → generate music + TTS narration
|
||||
- "Create a social media post" → generate image + caption
|
||||
- "Make a video with narration" → generate video + TTS voice-over
|
||||
- Always present results together with all URLs
|
||||
|
||||
## User Configuration
|
||||
|
||||
Check the User Configuration section for:
|
||||
- `default_provider` — use this provider unless the user overrides
|
||||
- `image_model` — preferred image model (use if not "auto")
|
||||
- `image_size` — default image dimensions
|
||||
- `tts_voice` — default TTS voice
|
||||
|
||||
Apply these defaults but allow the user to override in any request.
|
||||
|
||||
## Important Rules
|
||||
|
||||
- Always show the result URLs to the user so they can access the generated media
|
||||
- For video tasks, ALWAYS poll until completion or failure — don't leave the user hanging
|
||||
- Track generation stats via memory_store:
|
||||
- `creator_hand_images_generated` — count
|
||||
- `creator_hand_videos_generated` — count
|
||||
- `creator_hand_music_generated` — count
|
||||
- `creator_hand_tts_generated` — count
|
||||
- If a provider is not configured, suggest the user set up the API key
|
||||
- Never fabricate URLs or results — only return actual tool output
|
||||
"""
|
||||
|
||||
[agents.prompt_writer]
|
||||
invoke_hint = "Creative prompt engineering — writing detailed, effective prompts for image, video, and music generation"
|
||||
name = "prompt-writer"
|
||||
description = "Prompt engineer that crafts detailed, effective prompts for media generation"
|
||||
module = "builtin:chat"
|
||||
provider = "default"
|
||||
model = "default"
|
||||
max_tokens = 4096
|
||||
temperature = 0.8
|
||||
system_prompt = """You are Prompt Writer, a creative prompt engineer within the Creator Hand.
|
||||
|
||||
Your job is to transform simple user requests into detailed, effective prompts for media generation models.
|
||||
|
||||
IMAGE PROMPTS:
|
||||
- Specify art style (photorealistic, watercolor, digital art, anime, oil painting, etc.)
|
||||
- Include composition details (wide shot, close-up, bird's eye view, etc.)
|
||||
- Describe lighting (golden hour, studio lighting, dramatic shadows, etc.)
|
||||
- Add mood/atmosphere (serene, chaotic, mysterious, vibrant, etc.)
|
||||
- Include technical details when relevant (depth of field, lens type, etc.)
|
||||
|
||||
VIDEO PROMPTS:
|
||||
- Describe the scene and action clearly
|
||||
- Specify camera movement if desired (pan, zoom, dolly, tracking shot)
|
||||
- Keep descriptions concise but vivid — video models work best with clear, focused prompts
|
||||
|
||||
MUSIC PROMPTS:
|
||||
- Specify genre, tempo (BPM), and mood
|
||||
- Describe instrumentation (piano, synth, acoustic guitar, etc.)
|
||||
- Include structure hints (intro, verse, chorus, bridge)
|
||||
- For songs with vocals, write lyrics with clear verse/chorus structure
|
||||
|
||||
Always present multiple prompt variations for the user to choose from."""
|
||||
|
||||
# ---- Dashboard metrics -----------------------------------------------------------
|
||||
|
||||
[dashboard]
|
||||
[[dashboard.metrics]]
|
||||
label = "Images Generated"
|
||||
memory_key = "creator_hand_images_generated"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Videos Generated"
|
||||
memory_key = "creator_hand_videos_generated"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Music Tracks"
|
||||
memory_key = "creator_hand_music_generated"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "TTS Audio"
|
||||
memory_key = "creator_hand_tts_generated"
|
||||
format = "number"
|
||||
|
||||
# ---- Metadata --------------------------------------------------------------------
|
||||
|
||||
[metadata]
|
||||
frequency = "on-demand"
|
||||
token_consumption = "low"
|
||||
default_active = false
|
||||
|
||||
# ---- Internationalization --------------------------------------------------------
|
||||
|
||||
[i18n.zh]
|
||||
name = "创作 Hand"
|
||||
description = "AI 媒体工作室 — 通过文本指令生成图像、视频、音乐和语音"
|
||||
category = "内容"
|
||||
|
||||
[i18n.zh.settings.default_provider]
|
||||
label = "首选提供商"
|
||||
description = "默认使用哪个提供商。自动模式会为每种能力选择首个可用的提供商。"
|
||||
|
||||
[i18n.zh.settings.image_model]
|
||||
label = "图像模型"
|
||||
description = "用于图像生成的模型"
|
||||
|
||||
[i18n.zh.settings.image_size]
|
||||
label = "默认图像尺寸"
|
||||
description = "生成图像的默认分辨率"
|
||||
|
||||
[i18n.zh.settings.tts_voice]
|
||||
label = "TTS 语音"
|
||||
description = "文字转语音的默认声音"
|
||||
|
||||
[i18n.zh.settings.minimax_api_key]
|
||||
label = "MiniMax API 密钥"
|
||||
description = "来自 platform.minimaxi.com 的 API 密钥,用于视频和音乐生成"
|
||||
|
||||
[i18n.ja]
|
||||
name = "クリエイター Hand"
|
||||
description = "AIメディアスタジオ — テキストプロンプトから画像、動画、音楽、音声を生成"
|
||||
category = "コンテンツ"
|
||||
|
||||
[i18n.ja.settings.default_provider]
|
||||
label = "優先プロバイダー"
|
||||
description = "デフォルトで使用するプロバイダー。自動は各機能で最初に設定されたプロバイダーを選択。"
|
||||
|
||||
[i18n.ja.settings.image_model]
|
||||
label = "画像モデル"
|
||||
description = "画像生成に使用するモデル"
|
||||
|
||||
[i18n.ja.settings.image_size]
|
||||
label = "デフォルト画像サイズ"
|
||||
description = "生成画像のデフォルト解像度"
|
||||
|
||||
[i18n.ja.settings.tts_voice]
|
||||
label = "TTS音声"
|
||||
description = "テキスト読み上げのデフォルト音声"
|
||||
|
||||
[i18n.ja.settings.minimax_api_key]
|
||||
label = "MiniMax APIキー"
|
||||
description = "platform.minimaxi.comからのAPIキー。動画と音楽生成に使用。"
|
||||
|
||||
[i18n.ko]
|
||||
name = "크리에이터 Hand"
|
||||
description = "AI 미디어 스튜디오 — 텍스트로 이미지, 동영상, 음악, 음성 생성"
|
||||
category = "콘텐츠"
|
||||
|
||||
[i18n.ko.settings.default_provider]
|
||||
label = "기본 공급자"
|
||||
description = "기본으로 사용할 공급자. 자동은 각 기능에 대해 첫 번째 구성된 공급자를 선택."
|
||||
|
||||
[i18n.ko.settings.image_model]
|
||||
label = "이미지 모델"
|
||||
description = "이미지 생성에 사용할 모델"
|
||||
|
||||
[i18n.ko.settings.image_size]
|
||||
label = "기본 이미지 크기"
|
||||
description = "생성 이미지의 기본 해상도"
|
||||
|
||||
[i18n.ko.settings.tts_voice]
|
||||
label = "TTS 음성"
|
||||
description = "텍스트-음성 변환의 기본 음성"
|
||||
|
||||
[i18n.ko.settings.minimax_api_key]
|
||||
label = "MiniMax API 키"
|
||||
description = "platform.minimaxi.com에서 발급한 API 키. 동영상 및 음악 생성에 사용."
|
||||
|
||||
[i18n.es]
|
||||
name = "Hand Creador"
|
||||
description = "Estudio de medios IA — genera imágenes, videos, música y voz a partir de texto"
|
||||
category = "Contenido"
|
||||
|
||||
[i18n.fr]
|
||||
name = "Hand Cr\u00e9ateur"
|
||||
description = "Studio m\u00e9dia IA — g\u00e9n\u00e8re images, vid\u00e9os, musique et voix \u00e0 partir de texte"
|
||||
category = "Contenu"
|
||||
|
||||
[i18n.de]
|
||||
name = "Kreator-Hand"
|
||||
description = "KI-Medienstudio — erzeugt Bilder, Videos, Musik und Sprache aus Text"
|
||||
category = "Inhalt"
|
||||
@@ -0,0 +1,47 @@
|
||||
# Creator Hand
|
||||
|
||||
AI media studio -- generates images, videos, music, and speech from natural language prompts.
|
||||
|
||||
## Configuration
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| Category | `content` |
|
||||
| Agents | `creator-hand` (coordinator), `prompt-writer` |
|
||||
| Routing | `generate image`, `create video`, `make music`, `text to speech`, `media generation` |
|
||||
|
||||
## Integrations
|
||||
|
||||
- **OpenAI API** -- Image generation (gpt-image-1, DALL-E 3) and text-to-speech (tts-1).
|
||||
- **MiniMax API** -- Image, TTS, video generation (Hailuo T2V-01), and music generation (music-2.5).
|
||||
|
||||
## Provider Capabilities
|
||||
|
||||
| Provider | Image | TTS | Video | Music |
|
||||
|----------|-------|-----|-------|-------|
|
||||
| OpenAI | gpt-image-1, dall-e-3 | tts-1, tts-1-hd | -- | -- |
|
||||
| MiniMax | image-01 | speech-2.8-hd | T2V-01 | music-2.5 |
|
||||
|
||||
## Settings
|
||||
|
||||
- **Preferred Provider** -- `auto`, `openai`, `minimax`
|
||||
- **Image Model** -- `auto`, `gpt-image-1`, `dall-e-3`, `image-01`
|
||||
- **Default Image Size** -- `1024x1024`, `1792x1024`, `1024x1792`
|
||||
- **TTS Voice** -- `alloy`, `echo`, `fable`, `nova`, `onyx`, `shimmer`
|
||||
- **MiniMax API Key** -- API key from platform.minimaxi.com
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
librefang hand run creator
|
||||
```
|
||||
|
||||
### Examples
|
||||
|
||||
```
|
||||
> Generate a watercolor painting of a mountain lake at sunrise
|
||||
> Create a 5-second video of ocean waves crashing on rocks
|
||||
> Make an upbeat electronic jingle, 15 seconds, instrumental
|
||||
> Read this text aloud in a warm female voice: "Welcome to..."
|
||||
> Make a podcast intro: jingle + voice saying "Welcome to Tech Talk"
|
||||
```
|
||||
@@ -0,0 +1,249 @@
|
||||
---
|
||||
name: media-generation-skill
|
||||
version: "1.0.0"
|
||||
description: "Expert knowledge for AI media generation — image prompting, video workflows, music composition, and TTS best practices"
|
||||
runtime: prompt_only
|
||||
---
|
||||
|
||||
# Media Generation Expert Knowledge
|
||||
|
||||
## Tool Reference
|
||||
|
||||
### image_generate
|
||||
|
||||
Generate images from text prompts via OpenAI or MiniMax.
|
||||
|
||||
**Parameters:**
|
||||
|
||||
| Parameter | Type | Required | Default | Description |
|
||||
|-----------|------|----------|---------|-------------|
|
||||
| `prompt` | string | yes | — | Text description of the desired image |
|
||||
| `provider` | string | no | auto | `openai` or `minimax` |
|
||||
| `model` | string | no | provider default | `gpt-image-1`, `dall-e-3`, `image-01` |
|
||||
| `width` | int | no | 1024 | Image width in pixels |
|
||||
| `height` | int | no | 1024 | Image height in pixels |
|
||||
| `count` | int | no | 1 | Number of images (1-4) |
|
||||
| `quality` | string | no | `auto` | `low`, `medium`, `high`, `auto` |
|
||||
| `seed` | int | no | random | Reproducibility seed |
|
||||
|
||||
**Provider-specific notes:**
|
||||
|
||||
- **OpenAI gpt-image-1**: Best for photorealistic and creative images. Supports inpainting hints in prompt. Sizes: 1024x1024, 1792x1024, 1024x1792.
|
||||
- **OpenAI dall-e-3**: Good quality, may revise your prompt (check `revised_prompt` in response). Only generates 1 image per call.
|
||||
- **MiniMax image-01**: Fast generation, good for illustrations and concept art. Supports arbitrary aspect ratios.
|
||||
|
||||
**Result:** Returns `images` array with `url` fields pointing to `/api/uploads/{id}`.
|
||||
|
||||
---
|
||||
|
||||
### text_to_speech
|
||||
|
||||
Convert text to spoken audio.
|
||||
|
||||
**Parameters:**
|
||||
|
||||
| Parameter | Type | Required | Default | Description |
|
||||
|-----------|------|----------|---------|-------------|
|
||||
| `text` | string | yes | — | Text to speak (max ~4096 chars per call) |
|
||||
| `provider` | string | no | auto | `openai` or `minimax` |
|
||||
| `model` | string | no | provider default | `tts-1`, `tts-1-hd`, `speech-2.8-hd` |
|
||||
| `voice` | string | no | `alloy` | Voice selection (see table below) |
|
||||
| `speed` | float | no | 1.0 | Playback speed (0.25 - 4.0) |
|
||||
| `format` | string | no | `mp3` | `mp3`, `wav`, `flac`, `opus`, `aac` |
|
||||
|
||||
**OpenAI voices:**
|
||||
|
||||
| Voice | Character |
|
||||
|-------|-----------|
|
||||
| `alloy` | Neutral, balanced |
|
||||
| `echo` | Male, warm |
|
||||
| `fable` | Storytelling, expressive |
|
||||
| `nova` | Female, friendly |
|
||||
| `onyx` | Deep male, authoritative |
|
||||
| `shimmer` | Warm female, gentle |
|
||||
|
||||
**MiniMax voices:**
|
||||
|
||||
| Voice | Character |
|
||||
|-------|-----------|
|
||||
| `English_Graceful_Lady` | Female, elegant |
|
||||
| `English_Calm_Man` | Male, composed |
|
||||
| `English_Energetic_Girl` | Female, upbeat |
|
||||
|
||||
**Tips:**
|
||||
- For long content, split at paragraph boundaries to keep natural pacing
|
||||
- `tts-1-hd` is higher quality but slower; use `tts-1` for drafts
|
||||
- Speed 0.8-0.9 works well for narration; 1.1-1.2 for summaries
|
||||
|
||||
**Result:** Returns `url` to the audio file, `format`, `duration_ms`, `sample_rate`.
|
||||
|
||||
---
|
||||
|
||||
### video_generate
|
||||
|
||||
Submit an asynchronous video generation task. Video generation takes 1-3 minutes.
|
||||
|
||||
**Parameters:**
|
||||
|
||||
| Parameter | Type | Required | Default | Description |
|
||||
|-----------|------|----------|---------|-------------|
|
||||
| `prompt` | string | yes | — | Scene description |
|
||||
| `provider` | string | no | auto | Currently only `minimax` |
|
||||
| `model` | string | no | `T2V-01` | Video model |
|
||||
| `duration_secs` | int | no | 5 | Video duration (5-10 seconds) |
|
||||
| `resolution` | string | no | `1080p` | `720p`, `1080p` |
|
||||
|
||||
**Prompt writing for video:**
|
||||
- Be specific about the scene, subject, and action
|
||||
- Describe camera movement explicitly: "slow pan left", "zoom in", "static shot"
|
||||
- Keep it focused — one scene per generation works best
|
||||
- Include lighting and atmosphere: "golden hour lighting", "neon-lit street at night"
|
||||
- Avoid complex multi-character interactions (current models handle single subjects best)
|
||||
|
||||
**Good prompts:**
|
||||
- "A golden retriever running through a wheat field at sunset, slow motion, cinematic"
|
||||
- "Close-up of coffee being poured into a ceramic cup, steam rising, warm morning light"
|
||||
- "Aerial drone shot flying over a tropical coastline, turquoise water, white sand beach"
|
||||
|
||||
**Bad prompts:**
|
||||
- "A video" (too vague)
|
||||
- "Two people having a conversation at a cafe while a dog runs by and a car crashes outside" (too complex)
|
||||
|
||||
**Result:** Returns `task_id` and `provider`. You MUST poll with `video_status`.
|
||||
|
||||
---
|
||||
|
||||
### video_status
|
||||
|
||||
Poll the status of a video generation task.
|
||||
|
||||
**Parameters:**
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `task_id` | string | yes | From video_generate response |
|
||||
| `provider` | string | yes | Must match the provider from video_generate |
|
||||
|
||||
**Statuses:**
|
||||
|
||||
| Status | Meaning | Action |
|
||||
|--------|---------|--------|
|
||||
| `pending` | Queued, not started | Wait 10-15s, poll again |
|
||||
| `processing` | Actively generating | Wait 15-20s, poll again |
|
||||
| `completed` | Done | Result includes `file_url` |
|
||||
| `failed` | Generation failed | Check error message, may retry with different prompt |
|
||||
|
||||
**Polling pattern:**
|
||||
1. Call video_generate → get task_id
|
||||
2. Wait 10 seconds
|
||||
3. Call video_status with task_id + provider
|
||||
4. If not completed, wait 15-20 seconds and poll again
|
||||
5. Maximum ~10 polls (about 3 minutes total)
|
||||
6. Always inform the user of current status
|
||||
|
||||
**Result (completed):** Returns `file_url`, `width`, `height`, `duration_secs`, `provider`, `model`.
|
||||
|
||||
---
|
||||
|
||||
### music_generate
|
||||
|
||||
Generate music from a text prompt and/or lyrics.
|
||||
|
||||
**Parameters:**
|
||||
|
||||
| Parameter | Type | Required | Default | Description |
|
||||
|-----------|------|----------|---------|-------------|
|
||||
| `prompt` | string | no* | — | Style/mood description |
|
||||
| `lyrics` | string | no* | — | Song lyrics with structure |
|
||||
| `provider` | string | no | auto | Currently only `minimax` |
|
||||
| `model` | string | no | `music-2.5` | Music model |
|
||||
| `instrumental` | bool | no | false | Generate without vocals |
|
||||
| `format` | string | no | `mp3` | `mp3`, `wav`, `flac` |
|
||||
|
||||
*At least one of `prompt` or `lyrics` is required.
|
||||
|
||||
**Prompt writing for music:**
|
||||
|
||||
For instrumentals, describe:
|
||||
- Genre: electronic, jazz, classical, hip-hop, rock, ambient, lo-fi
|
||||
- Tempo: slow (60-80 BPM), medium (100-120 BPM), fast (130-160 BPM)
|
||||
- Mood: uplifting, melancholic, energetic, relaxing, dramatic, mysterious
|
||||
- Instruments: piano, synth, acoustic guitar, strings, drums, bass
|
||||
|
||||
For songs with vocals, provide lyrics with structure markers:
|
||||
|
||||
```
|
||||
[Verse 1]
|
||||
Walking down the empty street
|
||||
Moonlight dancing at my feet
|
||||
|
||||
[Chorus]
|
||||
This is where the night begins
|
||||
Let the music pull us in
|
||||
|
||||
[Verse 2]
|
||||
...
|
||||
```
|
||||
|
||||
**Good prompts:**
|
||||
- `prompt`: "Chill lo-fi hip-hop beat, vinyl crackle, mellow piano chords, 85 BPM, rainy day vibe"
|
||||
- `prompt`: "Epic orchestral trailer music, building tension, brass and strings, 140 BPM"
|
||||
- `prompt` + `lyrics`: "Indie folk acoustic ballad, fingerpicking guitar, gentle male vocals" with lyrics
|
||||
|
||||
**Result:** Returns `url` to audio file, `format`, `duration_ms`, `sample_rate`.
|
||||
|
||||
---
|
||||
|
||||
## Combined Workflow Recipes
|
||||
|
||||
### Podcast Intro
|
||||
1. `music_generate` — instrumental jingle, 10-15 seconds, upbeat
|
||||
2. `text_to_speech` — "Welcome to [show name]..." with energetic voice
|
||||
3. Report both URLs to user
|
||||
|
||||
### Social Media Post
|
||||
1. `image_generate` — eye-catching visual for the post
|
||||
2. Suggest caption text based on the image
|
||||
3. Optionally `text_to_speech` for accessibility audio version
|
||||
|
||||
### Video with Narration
|
||||
1. `text_to_speech` — generate narration audio
|
||||
2. `video_generate` — generate matching video clip
|
||||
3. `video_status` — poll until complete
|
||||
4. Report both URLs (user can combine with ffmpeg or editing tools)
|
||||
|
||||
### Album Art + Preview
|
||||
1. `image_generate` — album cover artwork
|
||||
2. `music_generate` — short preview track matching the artwork mood
|
||||
3. Present together
|
||||
|
||||
### Audiobook Chapter
|
||||
1. Split text into sections (~500 words each)
|
||||
2. `text_to_speech` for each section with consistent voice
|
||||
3. Report all audio URLs in order
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
| Error | Cause | Fix |
|
||||
|-------|-------|-----|
|
||||
| `missing_key` | API key not configured | Ask user to set OPENAI_API_KEY or MINIMAX_API_KEY |
|
||||
| `not_supported` | Provider doesn't support this modality | Switch to a provider that does |
|
||||
| `content_filtered` | Safety filter rejected the prompt | Rephrase without prohibited content |
|
||||
| `rate_limited` | Too many requests | Wait 30-60 seconds and retry |
|
||||
| `invalid_request` | Bad parameters | Check parameter ranges (e.g., count 1-4, speed 0.25-4.0) |
|
||||
|
||||
---
|
||||
|
||||
## Provider Capability Matrix
|
||||
|
||||
| Capability | OpenAI | MiniMax |
|
||||
|------------|--------|---------|
|
||||
| Image generation | gpt-image-1, dall-e-3 | image-01 |
|
||||
| Text-to-speech | tts-1, tts-1-hd | speech-2.8-hd |
|
||||
| Video generation | — | T2V-01, video-01 |
|
||||
| Music generation | — | music-2.5 |
|
||||
|
||||
**Auto-detection priority:** OpenAI > MiniMax (for capabilities both support).
|
||||
If only MiniMax key is set, all 4 modalities are available through MiniMax.
|
||||
Reference in new issue
Block a user