From ecc5215caed4ab14111d2fb5bf3f5ce1378c892d Mon Sep 17 00:00:00 2001 From: Evan Date: Mon, 23 Mar 2026 14:09:51 +0900 Subject: [PATCH] feat(hands): add Creator Hand for media generation (#17) --- hands/creator/HAND.toml | 419 ++++++++++++++++++++++++++++++++++++++++ hands/creator/README.md | 47 +++++ hands/creator/SKILL.md | 249 ++++++++++++++++++++++++ 3 files changed, 715 insertions(+) create mode 100644 hands/creator/HAND.toml create mode 100644 hands/creator/README.md create mode 100644 hands/creator/SKILL.md diff --git a/hands/creator/HAND.toml b/hands/creator/HAND.toml new file mode 100644 index 0000000..fc02124 --- /dev/null +++ b/hands/creator/HAND.toml @@ -0,0 +1,419 @@ +id = "creator" +version = "1.0.0" +name = "Creator Hand" +description = "AI media studio — generates images, videos, music, and speech from text prompts" + +category = "content" +icon = "\U0001F3A8" +tools = [ + "image_generate", + "video_generate", + "video_status", + "music_generate", + "text_to_speech", + "file_read", + "file_write", + "file_list", + "web_fetch", + "memory_store", + "memory_recall", +] + +[routing] +aliases = [ + "generate image", + "create image", + "make a picture", + "generate video", + "create video", + "make a video", + "generate music", + "create music", + "compose music", + "text to speech", + "generate speech", + "voice over", + "media generation", +] +weak_aliases = [ + "illustration", + "artwork", + "album cover", + "background music", + "jingle", + "narration", + "audio", + "thumbnail", + "poster", + "banner", +] + +# ---- Requirements ---------------------------------------------------------------- + +[[requires]] +key = "media_provider" +label = "At least one media provider API key must be set" +requirement_type = "any_env_var" +check_value = "OPENAI_API_KEY,MINIMAX_API_KEY" +description = "Creator Hand needs at least one configured media provider. OpenAI supports image and TTS; MiniMax supports image, TTS, video, and music." +optional = false + +# ---- Settings -------------------------------------------------------------------- + +[[settings]] +key = "default_provider" +label = "Preferred Provider" +description = "Which provider to use by default. Auto will pick the first configured provider for each capability." +setting_type = "select" +default = "auto" + +[[settings.options]] +value = "auto" +label = "Auto-detect (best available)" + +[[settings.options]] +value = "openai" +label = "OpenAI (image, TTS)" +provider_env = "OPENAI_API_KEY" + +[[settings.options]] +value = "minimax" +label = "MiniMax (image, TTS, video, music)" +provider_env = "MINIMAX_API_KEY" + +[[settings]] +key = "image_model" +label = "Image Model" +description = "Model to use for image generation" +setting_type = "select" +default = "auto" + +[[settings.options]] +value = "auto" +label = "Provider default" + +[[settings.options]] +value = "gpt-image-1" +label = "GPT Image 1 (OpenAI)" + +[[settings.options]] +value = "dall-e-3" +label = "DALL-E 3 (OpenAI)" + +[[settings.options]] +value = "image-01" +label = "Image-01 (MiniMax)" + +[[settings]] +key = "image_size" +label = "Default Image Size" +description = "Default resolution for generated images" +setting_type = "select" +default = "1024x1024" + +[[settings.options]] +value = "1024x1024" +label = "1024x1024 (square)" + +[[settings.options]] +value = "1792x1024" +label = "1792x1024 (landscape)" + +[[settings.options]] +value = "1024x1792" +label = "1024x1792 (portrait)" + +[[settings]] +key = "tts_voice" +label = "TTS Voice" +description = "Default voice for text-to-speech" +setting_type = "select" +default = "alloy" + +[[settings.options]] +value = "alloy" +label = "Alloy (neutral)" + +[[settings.options]] +value = "echo" +label = "Echo (male)" + +[[settings.options]] +value = "fable" +label = "Fable (storytelling)" + +[[settings.options]] +value = "nova" +label = "Nova (female)" + +[[settings.options]] +value = "onyx" +label = "Onyx (deep male)" + +[[settings.options]] +value = "shimmer" +label = "Shimmer (warm female)" + +[[settings]] +key = "minimax_api_key" +label = "MiniMax API Key" +description = "API key from platform.minimaxi.com for video and music generation" +setting_type = "text" +env_var = "MINIMAX_API_KEY" +default = "" + +# ---- Agent configuration --------------------------------------------------------- + +[agents.main] +coordinator = true +name = "creator-hand" +description = "AI media studio — generates images, videos, music, and speech from natural language" +module = "builtin:chat" +provider = "default" +model = "default" +max_tokens = 8192 +temperature = 0.5 +max_iterations = 30 +system_prompt = """You are Creator Hand — an AI media studio that generates images, videos, music, and speech from natural language requests. + +## Available Tools + +You have access to these media generation tools: + +### image_generate +Generate images from text prompts. +- Parameters: `prompt` (required), `provider`, `model`, `width`, `height`, `count`, `quality`, `seed` +- Returns: URLs of generated images (served at /api/uploads/...) +- Providers: OpenAI (gpt-image-1, dall-e-3), MiniMax (image-01) + +### text_to_speech +Convert text to spoken audio. +- Parameters: `text` (required), `provider`, `model`, `voice`, `speed`, `format` +- Returns: URL to the audio file +- Providers: OpenAI (tts-1, tts-1-hd), MiniMax (speech-2.8-hd) + +### video_generate +Submit a video generation task (asynchronous). +- Parameters: `prompt` (required), `provider`, `model`, `duration_secs`, `resolution` +- Returns: `task_id` and `provider` — use video_status to poll for completion +- Providers: MiniMax (T2V-01, video-01) + +### video_status +Check the status of a video generation task. +- Parameters: `task_id` (required), `provider` (required) +- Returns: status ("pending", "processing", "completed", "failed") and result URL when done + +### music_generate +Generate music from a text prompt and/or lyrics. +- Parameters: `prompt`, `lyrics`, `provider`, `model`, `instrumental` (bool), `format` +- At least one of `prompt` or `lyrics` is required +- Returns: URL to the audio file +- Providers: MiniMax (music-2.5) + +## Workflow Guidelines + +1. **Clarify intent**: If the user's request is vague, ask what type of media they want and suggest options. + +2. **Image generation**: + - Be detailed in prompts — describe style, mood, composition, lighting + - For multi-image requests, vary the prompts meaningfully + - Default to 1024x1024 unless the user specifies otherwise + +3. **Video generation** (async): + - Always explain that video takes time (typically 1-3 minutes) + - After calling video_generate, immediately poll with video_status + - If status is "processing", wait 15-20 seconds and poll again + - Keep the user informed of progress + +4. **Music generation**: + - For instrumental, set `instrumental: true` + - For songs with vocals, provide both `prompt` (style description) and `lyrics` + - Suggest genres and moods if the user doesn't specify + +5. **TTS**: + - Choose a voice that matches the content tone + - For long text, break into paragraphs and generate separately if needed + +6. **Combined workflows** — these are where you shine: + - "Make a podcast intro" → generate music + TTS narration + - "Create a social media post" → generate image + caption + - "Make a video with narration" → generate video + TTS voice-over + - Always present results together with all URLs + +## User Configuration + +Check the User Configuration section for: +- `default_provider` — use this provider unless the user overrides +- `image_model` — preferred image model (use if not "auto") +- `image_size` — default image dimensions +- `tts_voice` — default TTS voice + +Apply these defaults but allow the user to override in any request. + +## Important Rules + +- Always show the result URLs to the user so they can access the generated media +- For video tasks, ALWAYS poll until completion or failure — don't leave the user hanging +- Track generation stats via memory_store: + - `creator_hand_images_generated` — count + - `creator_hand_videos_generated` — count + - `creator_hand_music_generated` — count + - `creator_hand_tts_generated` — count +- If a provider is not configured, suggest the user set up the API key +- Never fabricate URLs or results — only return actual tool output +""" + +[agents.prompt_writer] +invoke_hint = "Creative prompt engineering — writing detailed, effective prompts for image, video, and music generation" +name = "prompt-writer" +description = "Prompt engineer that crafts detailed, effective prompts for media generation" +module = "builtin:chat" +provider = "default" +model = "default" +max_tokens = 4096 +temperature = 0.8 +system_prompt = """You are Prompt Writer, a creative prompt engineer within the Creator Hand. + +Your job is to transform simple user requests into detailed, effective prompts for media generation models. + +IMAGE PROMPTS: +- Specify art style (photorealistic, watercolor, digital art, anime, oil painting, etc.) +- Include composition details (wide shot, close-up, bird's eye view, etc.) +- Describe lighting (golden hour, studio lighting, dramatic shadows, etc.) +- Add mood/atmosphere (serene, chaotic, mysterious, vibrant, etc.) +- Include technical details when relevant (depth of field, lens type, etc.) + +VIDEO PROMPTS: +- Describe the scene and action clearly +- Specify camera movement if desired (pan, zoom, dolly, tracking shot) +- Keep descriptions concise but vivid — video models work best with clear, focused prompts + +MUSIC PROMPTS: +- Specify genre, tempo (BPM), and mood +- Describe instrumentation (piano, synth, acoustic guitar, etc.) +- Include structure hints (intro, verse, chorus, bridge) +- For songs with vocals, write lyrics with clear verse/chorus structure + +Always present multiple prompt variations for the user to choose from.""" + +# ---- Dashboard metrics ----------------------------------------------------------- + +[dashboard] +[[dashboard.metrics]] +label = "Images Generated" +memory_key = "creator_hand_images_generated" +format = "number" + +[[dashboard.metrics]] +label = "Videos Generated" +memory_key = "creator_hand_videos_generated" +format = "number" + +[[dashboard.metrics]] +label = "Music Tracks" +memory_key = "creator_hand_music_generated" +format = "number" + +[[dashboard.metrics]] +label = "TTS Audio" +memory_key = "creator_hand_tts_generated" +format = "number" + +# ---- Metadata -------------------------------------------------------------------- + +[metadata] +frequency = "on-demand" +token_consumption = "low" +default_active = false + +# ---- Internationalization -------------------------------------------------------- + +[i18n.zh] +name = "创作 Hand" +description = "AI 媒体工作室 — 通过文本指令生成图像、视频、音乐和语音" +category = "内容" + +[i18n.zh.settings.default_provider] +label = "首选提供商" +description = "默认使用哪个提供商。自动模式会为每种能力选择首个可用的提供商。" + +[i18n.zh.settings.image_model] +label = "图像模型" +description = "用于图像生成的模型" + +[i18n.zh.settings.image_size] +label = "默认图像尺寸" +description = "生成图像的默认分辨率" + +[i18n.zh.settings.tts_voice] +label = "TTS 语音" +description = "文字转语音的默认声音" + +[i18n.zh.settings.minimax_api_key] +label = "MiniMax API 密钥" +description = "来自 platform.minimaxi.com 的 API 密钥,用于视频和音乐生成" + +[i18n.ja] +name = "クリエイター Hand" +description = "AIメディアスタジオ — テキストプロンプトから画像、動画、音楽、音声を生成" +category = "コンテンツ" + +[i18n.ja.settings.default_provider] +label = "優先プロバイダー" +description = "デフォルトで使用するプロバイダー。自動は各機能で最初に設定されたプロバイダーを選択。" + +[i18n.ja.settings.image_model] +label = "画像モデル" +description = "画像生成に使用するモデル" + +[i18n.ja.settings.image_size] +label = "デフォルト画像サイズ" +description = "生成画像のデフォルト解像度" + +[i18n.ja.settings.tts_voice] +label = "TTS音声" +description = "テキスト読み上げのデフォルト音声" + +[i18n.ja.settings.minimax_api_key] +label = "MiniMax APIキー" +description = "platform.minimaxi.comからのAPIキー。動画と音楽生成に使用。" + +[i18n.ko] +name = "크리에이터 Hand" +description = "AI 미디어 스튜디오 — 텍스트로 이미지, 동영상, 음악, 음성 생성" +category = "콘텐츠" + +[i18n.ko.settings.default_provider] +label = "기본 공급자" +description = "기본으로 사용할 공급자. 자동은 각 기능에 대해 첫 번째 구성된 공급자를 선택." + +[i18n.ko.settings.image_model] +label = "이미지 모델" +description = "이미지 생성에 사용할 모델" + +[i18n.ko.settings.image_size] +label = "기본 이미지 크기" +description = "생성 이미지의 기본 해상도" + +[i18n.ko.settings.tts_voice] +label = "TTS 음성" +description = "텍스트-음성 변환의 기본 음성" + +[i18n.ko.settings.minimax_api_key] +label = "MiniMax API 키" +description = "platform.minimaxi.com에서 발급한 API 키. 동영상 및 음악 생성에 사용." + +[i18n.es] +name = "Hand Creador" +description = "Estudio de medios IA — genera imágenes, videos, música y voz a partir de texto" +category = "Contenido" + +[i18n.fr] +name = "Hand Cr\u00e9ateur" +description = "Studio m\u00e9dia IA — g\u00e9n\u00e8re images, vid\u00e9os, musique et voix \u00e0 partir de texte" +category = "Contenu" + +[i18n.de] +name = "Kreator-Hand" +description = "KI-Medienstudio — erzeugt Bilder, Videos, Musik und Sprache aus Text" +category = "Inhalt" diff --git a/hands/creator/README.md b/hands/creator/README.md new file mode 100644 index 0000000..d180deb --- /dev/null +++ b/hands/creator/README.md @@ -0,0 +1,47 @@ +# Creator Hand + +AI media studio -- generates images, videos, music, and speech from natural language prompts. + +## Configuration + +| Field | Value | +|-------|-------| +| Category | `content` | +| Agents | `creator-hand` (coordinator), `prompt-writer` | +| Routing | `generate image`, `create video`, `make music`, `text to speech`, `media generation` | + +## Integrations + +- **OpenAI API** -- Image generation (gpt-image-1, DALL-E 3) and text-to-speech (tts-1). +- **MiniMax API** -- Image, TTS, video generation (Hailuo T2V-01), and music generation (music-2.5). + +## Provider Capabilities + +| Provider | Image | TTS | Video | Music | +|----------|-------|-----|-------|-------| +| OpenAI | gpt-image-1, dall-e-3 | tts-1, tts-1-hd | -- | -- | +| MiniMax | image-01 | speech-2.8-hd | T2V-01 | music-2.5 | + +## Settings + +- **Preferred Provider** -- `auto`, `openai`, `minimax` +- **Image Model** -- `auto`, `gpt-image-1`, `dall-e-3`, `image-01` +- **Default Image Size** -- `1024x1024`, `1792x1024`, `1024x1792` +- **TTS Voice** -- `alloy`, `echo`, `fable`, `nova`, `onyx`, `shimmer` +- **MiniMax API Key** -- API key from platform.minimaxi.com + +## Usage + +```bash +librefang hand run creator +``` + +### Examples + +``` +> Generate a watercolor painting of a mountain lake at sunrise +> Create a 5-second video of ocean waves crashing on rocks +> Make an upbeat electronic jingle, 15 seconds, instrumental +> Read this text aloud in a warm female voice: "Welcome to..." +> Make a podcast intro: jingle + voice saying "Welcome to Tech Talk" +``` diff --git a/hands/creator/SKILL.md b/hands/creator/SKILL.md new file mode 100644 index 0000000..463c1d9 --- /dev/null +++ b/hands/creator/SKILL.md @@ -0,0 +1,249 @@ +--- +name: media-generation-skill +version: "1.0.0" +description: "Expert knowledge for AI media generation — image prompting, video workflows, music composition, and TTS best practices" +runtime: prompt_only +--- + +# Media Generation Expert Knowledge + +## Tool Reference + +### image_generate + +Generate images from text prompts via OpenAI or MiniMax. + +**Parameters:** + +| Parameter | Type | Required | Default | Description | +|-----------|------|----------|---------|-------------| +| `prompt` | string | yes | — | Text description of the desired image | +| `provider` | string | no | auto | `openai` or `minimax` | +| `model` | string | no | provider default | `gpt-image-1`, `dall-e-3`, `image-01` | +| `width` | int | no | 1024 | Image width in pixels | +| `height` | int | no | 1024 | Image height in pixels | +| `count` | int | no | 1 | Number of images (1-4) | +| `quality` | string | no | `auto` | `low`, `medium`, `high`, `auto` | +| `seed` | int | no | random | Reproducibility seed | + +**Provider-specific notes:** + +- **OpenAI gpt-image-1**: Best for photorealistic and creative images. Supports inpainting hints in prompt. Sizes: 1024x1024, 1792x1024, 1024x1792. +- **OpenAI dall-e-3**: Good quality, may revise your prompt (check `revised_prompt` in response). Only generates 1 image per call. +- **MiniMax image-01**: Fast generation, good for illustrations and concept art. Supports arbitrary aspect ratios. + +**Result:** Returns `images` array with `url` fields pointing to `/api/uploads/{id}`. + +--- + +### text_to_speech + +Convert text to spoken audio. + +**Parameters:** + +| Parameter | Type | Required | Default | Description | +|-----------|------|----------|---------|-------------| +| `text` | string | yes | — | Text to speak (max ~4096 chars per call) | +| `provider` | string | no | auto | `openai` or `minimax` | +| `model` | string | no | provider default | `tts-1`, `tts-1-hd`, `speech-2.8-hd` | +| `voice` | string | no | `alloy` | Voice selection (see table below) | +| `speed` | float | no | 1.0 | Playback speed (0.25 - 4.0) | +| `format` | string | no | `mp3` | `mp3`, `wav`, `flac`, `opus`, `aac` | + +**OpenAI voices:** + +| Voice | Character | +|-------|-----------| +| `alloy` | Neutral, balanced | +| `echo` | Male, warm | +| `fable` | Storytelling, expressive | +| `nova` | Female, friendly | +| `onyx` | Deep male, authoritative | +| `shimmer` | Warm female, gentle | + +**MiniMax voices:** + +| Voice | Character | +|-------|-----------| +| `English_Graceful_Lady` | Female, elegant | +| `English_Calm_Man` | Male, composed | +| `English_Energetic_Girl` | Female, upbeat | + +**Tips:** +- For long content, split at paragraph boundaries to keep natural pacing +- `tts-1-hd` is higher quality but slower; use `tts-1` for drafts +- Speed 0.8-0.9 works well for narration; 1.1-1.2 for summaries + +**Result:** Returns `url` to the audio file, `format`, `duration_ms`, `sample_rate`. + +--- + +### video_generate + +Submit an asynchronous video generation task. Video generation takes 1-3 minutes. + +**Parameters:** + +| Parameter | Type | Required | Default | Description | +|-----------|------|----------|---------|-------------| +| `prompt` | string | yes | — | Scene description | +| `provider` | string | no | auto | Currently only `minimax` | +| `model` | string | no | `T2V-01` | Video model | +| `duration_secs` | int | no | 5 | Video duration (5-10 seconds) | +| `resolution` | string | no | `1080p` | `720p`, `1080p` | + +**Prompt writing for video:** +- Be specific about the scene, subject, and action +- Describe camera movement explicitly: "slow pan left", "zoom in", "static shot" +- Keep it focused — one scene per generation works best +- Include lighting and atmosphere: "golden hour lighting", "neon-lit street at night" +- Avoid complex multi-character interactions (current models handle single subjects best) + +**Good prompts:** +- "A golden retriever running through a wheat field at sunset, slow motion, cinematic" +- "Close-up of coffee being poured into a ceramic cup, steam rising, warm morning light" +- "Aerial drone shot flying over a tropical coastline, turquoise water, white sand beach" + +**Bad prompts:** +- "A video" (too vague) +- "Two people having a conversation at a cafe while a dog runs by and a car crashes outside" (too complex) + +**Result:** Returns `task_id` and `provider`. You MUST poll with `video_status`. + +--- + +### video_status + +Poll the status of a video generation task. + +**Parameters:** + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `task_id` | string | yes | From video_generate response | +| `provider` | string | yes | Must match the provider from video_generate | + +**Statuses:** + +| Status | Meaning | Action | +|--------|---------|--------| +| `pending` | Queued, not started | Wait 10-15s, poll again | +| `processing` | Actively generating | Wait 15-20s, poll again | +| `completed` | Done | Result includes `file_url` | +| `failed` | Generation failed | Check error message, may retry with different prompt | + +**Polling pattern:** +1. Call video_generate → get task_id +2. Wait 10 seconds +3. Call video_status with task_id + provider +4. If not completed, wait 15-20 seconds and poll again +5. Maximum ~10 polls (about 3 minutes total) +6. Always inform the user of current status + +**Result (completed):** Returns `file_url`, `width`, `height`, `duration_secs`, `provider`, `model`. + +--- + +### music_generate + +Generate music from a text prompt and/or lyrics. + +**Parameters:** + +| Parameter | Type | Required | Default | Description | +|-----------|------|----------|---------|-------------| +| `prompt` | string | no* | — | Style/mood description | +| `lyrics` | string | no* | — | Song lyrics with structure | +| `provider` | string | no | auto | Currently only `minimax` | +| `model` | string | no | `music-2.5` | Music model | +| `instrumental` | bool | no | false | Generate without vocals | +| `format` | string | no | `mp3` | `mp3`, `wav`, `flac` | + +*At least one of `prompt` or `lyrics` is required. + +**Prompt writing for music:** + +For instrumentals, describe: +- Genre: electronic, jazz, classical, hip-hop, rock, ambient, lo-fi +- Tempo: slow (60-80 BPM), medium (100-120 BPM), fast (130-160 BPM) +- Mood: uplifting, melancholic, energetic, relaxing, dramatic, mysterious +- Instruments: piano, synth, acoustic guitar, strings, drums, bass + +For songs with vocals, provide lyrics with structure markers: + +``` +[Verse 1] +Walking down the empty street +Moonlight dancing at my feet + +[Chorus] +This is where the night begins +Let the music pull us in + +[Verse 2] +... +``` + +**Good prompts:** +- `prompt`: "Chill lo-fi hip-hop beat, vinyl crackle, mellow piano chords, 85 BPM, rainy day vibe" +- `prompt`: "Epic orchestral trailer music, building tension, brass and strings, 140 BPM" +- `prompt` + `lyrics`: "Indie folk acoustic ballad, fingerpicking guitar, gentle male vocals" with lyrics + +**Result:** Returns `url` to audio file, `format`, `duration_ms`, `sample_rate`. + +--- + +## Combined Workflow Recipes + +### Podcast Intro +1. `music_generate` — instrumental jingle, 10-15 seconds, upbeat +2. `text_to_speech` — "Welcome to [show name]..." with energetic voice +3. Report both URLs to user + +### Social Media Post +1. `image_generate` — eye-catching visual for the post +2. Suggest caption text based on the image +3. Optionally `text_to_speech` for accessibility audio version + +### Video with Narration +1. `text_to_speech` — generate narration audio +2. `video_generate` — generate matching video clip +3. `video_status` — poll until complete +4. Report both URLs (user can combine with ffmpeg or editing tools) + +### Album Art + Preview +1. `image_generate` — album cover artwork +2. `music_generate` — short preview track matching the artwork mood +3. Present together + +### Audiobook Chapter +1. Split text into sections (~500 words each) +2. `text_to_speech` for each section with consistent voice +3. Report all audio URLs in order + +--- + +## Error Handling + +| Error | Cause | Fix | +|-------|-------|-----| +| `missing_key` | API key not configured | Ask user to set OPENAI_API_KEY or MINIMAX_API_KEY | +| `not_supported` | Provider doesn't support this modality | Switch to a provider that does | +| `content_filtered` | Safety filter rejected the prompt | Rephrase without prohibited content | +| `rate_limited` | Too many requests | Wait 30-60 seconds and retry | +| `invalid_request` | Bad parameters | Check parameter ranges (e.g., count 1-4, speed 0.25-4.0) | + +--- + +## Provider Capability Matrix + +| Capability | OpenAI | MiniMax | +|------------|--------|---------| +| Image generation | gpt-image-1, dall-e-3 | image-01 | +| Text-to-speech | tts-1, tts-1-hd | speech-2.8-hd | +| Video generation | — | T2V-01, video-01 | +| Music generation | — | music-2.5 | + +**Auto-detection priority:** OpenAI > MiniMax (for capabilities both support). +If only MiniMax key is set, all 4 modalities are available through MiniMax.