* refactor: migrate icon fields from emoji to lucide:<name> tokens
Every TOML manifest's `icon = "<emoji>"` line is replaced with
`icon = "lucide:<kebab-name>"` — a reference to a lucide-react icon,
which the librefang.ai site and dashboard render as crisp SVG. Reasons
for the switch:
- Emoji render very differently across OS/browser/font stacks; the
registry catalog looked inconsistent from one row to the next.
- Five manifests (clip / creator / linkedin / reddit / twitter) had
their icons stored as literal Python-style escape strings
("\\U0001F3AC") because the TOML parser upstream never decoded
them. Switching away from emoji drops that class of bug entirely.
- As a drive-by, also decode the \\uXXXX accent escapes in the
[i18n.fr] block of hands/creator/HAND.toml so "Créateur" shows
up correctly.
87 files touched. example manifests left untouched (still "TODO").
* fix: backfill i18n name + drop the single-member email category
- Every existing [i18n.<lang>] block now has a `name` field. 60 files
previously translated description but kept the English name
implicitly — which rendered as "some English some Chinese" in the
registry UI. Fill in the missing name from the English brand (or a
known localized equivalent: DingTalk→钉钉, Feishu→飞书, Email→
电子邮件 / メール / E-Mail / Correo / Courriel, and a handful of
hands that have Chinese product names like 视频剪辑 Hand).
- channels/email.toml was the only item under category="email";
reclassify it as "messaging" so the sub-category filter chip list
on the category page isn't littered with singletons.
* feat(i18n): localize 76 agents/integrations/plugins into 7 languages
Adds full [i18n.zh], [i18n.zh-TW], [i18n.ja], [i18n.ko], [i18n.de],
[i18n.es], [i18n.fr] blocks with name + description to every manifest
that previously shipped English-only.
Coverage:
- 32 agents (academic-researcher, analyst, architect, assistant,
code-reviewer, coder, customer-support, data-scientist, debugger,
devops-lead, doc-writer, email-assistant, health-tracker,
hello-world, home-automation, legal-assistant, meeting-assistant,
ops, orchestrator, personal-finance, planner, recipe-assistant,
recruiter, researcher, sales-assistant, security-auditor,
social-media, test-engineer, translator, travel-planner, tutor,
writer)
- 33 integrations (AWS, Azure, Bitbucket, Brave Search, Discord,
Dropbox, Elasticsearch, Exa Search, Fetch, Filesystem, GCP, Git,
GitHub, GitLab, Gmail, Google Calendar, Google Drive, Google Maps,
Jira, Linear, Memory, MongoDB, Notion, PostgreSQL, Puppeteer, Redis,
Sentry, Sequential Thinking, Slack, SQLite, Teams, Time, Todoist) —
brand names kept as-is across all locales, only descriptions
translated.
- 11 plugins (auto-summarizer, context-decay, conversation-logger,
episodic-memory, guardrails, keyword-memory, mempalace-indexer,
sentiment-tracker, todo-tracker, topic-memory, user-profile)
The descriptions are one-line summaries — hand-translated rather than
machine-generated, so technical terms (MCP, PR, CI/CD, etc.) stay
consistent across locales.
* feat(i18n): close remaining per-lang gaps for channels, workflows, devteam
Third pass on i18n coverage. Every non-example manifest now carries a
full set of [i18n.zh], [i18n.zh-TW], [i18n.ja], [i18n.ko], [i18n.de],
[i18n.es], [i18n.fr] blocks.
- 44 channel adapters: added French descriptions (zh/zh-TW/ja/ko/de/es
were already present). Brand names kept as-is in all locales so users
recognize Discord / Slack / LINE / etc. consistently.
- 22 workflows: filled zh-TW / ja / ko / de / es / fr blocks. Each
translation mirrors the existing zh one in structure and tone so the
catalog reads consistently across locales.
- hands/devteam/HAND.toml: added the four langs that were missing
(zh-TW, de, es, fr).
Only the 6 templates under examples/ are left without i18n blocks on
purpose — they still contain "TODO:" placeholders.
932 lines
32 KiB
TOML
932 lines
32 KiB
TOML
id = "devops"
|
||
version = "1.1.0"
|
||
name = "DevOps Hand"
|
||
description = "Autonomous DevOps engineer — CI/CD management, infrastructure monitoring, deployment automation, and incident response"
|
||
|
||
category = "development"
|
||
icon = "lucide:hard-hat"
|
||
|
||
tools = [
|
||
"shell_exec",
|
||
"file_read",
|
||
"file_write",
|
||
"file_list",
|
||
"web_fetch",
|
||
"web_search",
|
||
"memory_store",
|
||
"memory_recall",
|
||
"schedule_create",
|
||
"schedule_list",
|
||
"schedule_delete",
|
||
"knowledge_add_entity",
|
||
"knowledge_add_relation",
|
||
"knowledge_query",
|
||
"event_publish",
|
||
]
|
||
|
||
[[requires]]
|
||
key = "curl"
|
||
label = "curl must be installed"
|
||
requirement_type = "binary"
|
||
check_value = "curl"
|
||
description = "curl is used for HTTP health checks, GitHub API calls, and service endpoint monitoring."
|
||
|
||
[requires.install]
|
||
macos = "brew install curl"
|
||
linux_apt = "sudo apt install curl"
|
||
linux_dnf = "sudo dnf install curl"
|
||
linux_pacman = "sudo pacman -S curl"
|
||
windows = "winget install cURL.cURL"
|
||
estimated_time = "1 min"
|
||
|
||
[[requires]]
|
||
key = "git"
|
||
label = "git must be installed"
|
||
requirement_type = "binary"
|
||
check_value = "git"
|
||
description = "git is used for deployment history, version control operations, and CI/CD pipeline management."
|
||
|
||
[requires.install]
|
||
macos = "brew install git"
|
||
linux_apt = "sudo apt install git"
|
||
linux_dnf = "sudo dnf install git"
|
||
linux_pacman = "sudo pacman -S git"
|
||
windows = "winget install Git.Git"
|
||
estimated_time = "1-2 min"
|
||
|
||
[[requires]]
|
||
key = "docker"
|
||
label = "Docker (optional — needed for container workloads)"
|
||
requirement_type = "binary"
|
||
check_value = "docker"
|
||
optional = true
|
||
description = "Docker is used for container status checks, image management, and service orchestration. Only needed if your infrastructure uses containers."
|
||
|
||
[requires.install]
|
||
macos = "brew install --cask docker"
|
||
linux_apt = "sudo apt install docker.io"
|
||
linux_dnf = "sudo dnf install docker"
|
||
linux_pacman = "sudo pacman -S docker"
|
||
windows = "winget install Docker.DockerDesktop"
|
||
manual_url = "https://docs.docker.com/get-docker/"
|
||
estimated_time = "5-10 min"
|
||
|
||
[[requires]]
|
||
key = "GITHUB_TOKEN"
|
||
label = "GitHub Token (optional — needed for GitHub Actions)"
|
||
requirement_type = "api_key"
|
||
check_value = "GITHUB_TOKEN"
|
||
optional = true
|
||
description = "A GitHub personal access token for accessing GitHub Actions API, checking pipeline status, and triggering workflows."
|
||
|
||
[requires.install]
|
||
signup_url = "https://github.com/settings/tokens"
|
||
docs_url = "https://docs.github.com/en/authentication/keeping-your-account-and-data-secure/managing-your-personal-access-tokens"
|
||
env_example = "GITHUB_TOKEN=ghp_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx"
|
||
estimated_time = "2-5 min"
|
||
steps = [
|
||
"Go to GitHub Settings → Developer settings → Personal access tokens → Fine-grained tokens",
|
||
"Click 'Generate new token'",
|
||
"Select repository access scope and permissions (Actions: read, Contents: read)",
|
||
"Copy the token and set it as GITHUB_TOKEN environment variable",
|
||
]
|
||
|
||
[routing]
|
||
aliases = [
|
||
"ci/cd",
|
||
"pipeline",
|
||
"github actions",
|
||
"infrastructure monitoring",
|
||
"deployment automation",
|
||
"incident response",
|
||
]
|
||
weak_aliases = [
|
||
"deploy",
|
||
"kubernetes",
|
||
"docker",
|
||
"container",
|
||
"terraform",
|
||
"helm",
|
||
]
|
||
|
||
# ─── Configurable settings ───────────────────────────────────────────────────
|
||
|
||
[[settings]]
|
||
key = "infrastructure"
|
||
label = "Infrastructure Type"
|
||
description = "Primary infrastructure platform"
|
||
setting_type = "select"
|
||
default = "cloud"
|
||
|
||
[[settings.options]]
|
||
value = "cloud"
|
||
label = "Cloud (AWS/GCP/Azure)"
|
||
|
||
[[settings.options]]
|
||
value = "kubernetes"
|
||
label = "Kubernetes"
|
||
|
||
[[settings.options]]
|
||
value = "docker"
|
||
label = "Docker / Docker Compose"
|
||
|
||
[[settings.options]]
|
||
value = "bare_metal"
|
||
label = "Bare Metal / VPS"
|
||
|
||
[[settings.options]]
|
||
value = "serverless"
|
||
label = "Serverless"
|
||
|
||
[[settings]]
|
||
key = "ci_platform"
|
||
label = "CI/CD Platform"
|
||
description = "Primary CI/CD platform"
|
||
setting_type = "select"
|
||
default = "github_actions"
|
||
|
||
[[settings.options]]
|
||
value = "github_actions"
|
||
label = "GitHub Actions"
|
||
|
||
[[settings.options]]
|
||
value = "gitlab_ci"
|
||
label = "GitLab CI"
|
||
|
||
[[settings.options]]
|
||
value = "jenkins"
|
||
label = "Jenkins"
|
||
|
||
[[settings.options]]
|
||
value = "circleci"
|
||
label = "CircleCI"
|
||
|
||
[[settings.options]]
|
||
value = "other"
|
||
label = "Other"
|
||
|
||
[[settings]]
|
||
key = "monitoring_focus"
|
||
label = "Monitoring Focus"
|
||
description = "Primary monitoring and alerting focus"
|
||
setting_type = "select"
|
||
default = "balanced"
|
||
|
||
[[settings.options]]
|
||
value = "uptime"
|
||
label = "Uptime & Availability"
|
||
|
||
[[settings.options]]
|
||
value = "performance"
|
||
label = "Performance & Latency"
|
||
|
||
[[settings.options]]
|
||
value = "security"
|
||
label = "Security & Compliance"
|
||
|
||
[[settings.options]]
|
||
value = "cost"
|
||
label = "Cost Optimization"
|
||
|
||
[[settings.options]]
|
||
value = "balanced"
|
||
label = "Balanced (all areas)"
|
||
|
||
[[settings]]
|
||
key = "auto_monitor"
|
||
label = "Auto Monitor"
|
||
description = "Automatically monitor infrastructure and alert on issues"
|
||
setting_type = "toggle"
|
||
default = "false"
|
||
|
||
[[settings]]
|
||
key = "check_interval"
|
||
label = "Health Check Interval"
|
||
description = "How often to run automated health checks"
|
||
setting_type = "select"
|
||
default = "5min"
|
||
|
||
[[settings.options]]
|
||
value = "1min"
|
||
label = "Every minute"
|
||
|
||
[[settings.options]]
|
||
value = "5min"
|
||
label = "Every 5 minutes"
|
||
|
||
[[settings.options]]
|
||
value = "15min"
|
||
label = "Every 15 minutes"
|
||
|
||
[[settings.options]]
|
||
value = "1hour"
|
||
label = "Every hour"
|
||
|
||
[[settings]]
|
||
key = "service_urls"
|
||
label = "Service URLs"
|
||
description = "Comma-separated URLs to monitor (e.g. https://api.example.com/health,https://app.example.com)"
|
||
setting_type = "text"
|
||
default = ""
|
||
|
||
[[settings]]
|
||
key = "alert_on_failure"
|
||
label = "Alert on Failure"
|
||
description = "Publish events when health checks fail"
|
||
setting_type = "toggle"
|
||
default = "true"
|
||
|
||
[[settings]]
|
||
key = "rollback_strategy"
|
||
label = "Rollback Strategy"
|
||
description = "Default rollback approach for failed deployments"
|
||
setting_type = "select"
|
||
default = "manual"
|
||
|
||
[[settings.options]]
|
||
value = "manual"
|
||
label = "Manual (alert and wait for user)"
|
||
|
||
[[settings.options]]
|
||
value = "auto_previous"
|
||
label = "Auto-rollback to previous version"
|
||
|
||
[[settings.options]]
|
||
value = "blue_green"
|
||
label = "Blue-green switch back"
|
||
|
||
[[settings]]
|
||
key = "approval_mode"
|
||
label = "Approval Mode"
|
||
description = "Queue deployment and infrastructure actions for your review instead of executing directly"
|
||
setting_type = "toggle"
|
||
default = "true"
|
||
|
||
# ─── Agent configuration ─────────────────────────────────────────────────────
|
||
|
||
[agents.main]
|
||
coordinator = true
|
||
name = "devops-hand"
|
||
description = "AI DevOps engineer — manages CI/CD pipelines, monitors infrastructure, automates deployments, and handles incident response"
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 16384
|
||
temperature = 0.2
|
||
max_iterations = 60
|
||
system_prompt = """You are DevOps Hand — an autonomous DevOps engineer that manages CI/CD pipelines, monitors infrastructure health, automates deployments, and handles incident response.
|
||
|
||
## Phase 0 — Environment Detection (ALWAYS DO THIS FIRST)
|
||
|
||
Detect the operating system and available tools:
|
||
```
|
||
python -c "import platform; print(platform.system())"
|
||
```
|
||
|
||
Check available DevOps tools:
|
||
```
|
||
docker --version 2>/dev/null
|
||
kubectl version --client 2>/dev/null
|
||
terraform --version 2>/dev/null
|
||
git --version
|
||
curl --version | head -1
|
||
```
|
||
|
||
Load context:
|
||
1. memory_recall `devops_hand_state` — load previous monitoring data and incident history
|
||
2. Read **User Configuration** for infrastructure, ci_platform, service_urls, approval_mode, etc.
|
||
3. file_read `devops_queue.json` if it exists — pending deployment/remediation actions
|
||
4. knowledge_query for known infrastructure topology and previous incidents
|
||
|
||
---
|
||
|
||
## Phase 1 — Infrastructure Health Check
|
||
|
||
Check the health of all configured services:
|
||
|
||
For each URL in `service_urls`:
|
||
```
|
||
curl -s -o /dev/null -w "%{http_code} %{time_total}" --max-time 10 "$URL"
|
||
```
|
||
|
||
Record:
|
||
- HTTP status code
|
||
- Response time
|
||
- SSL certificate expiry (if HTTPS)
|
||
- DNS resolution time
|
||
|
||
For Docker environments:
|
||
```
|
||
docker ps --format "table {{.Names}}\t{{.Status}}\t{{.Ports}}"
|
||
docker stats --no-stream --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}"
|
||
```
|
||
|
||
For Kubernetes environments:
|
||
```
|
||
kubectl get pods --all-namespaces -o wide
|
||
kubectl top pods --all-namespaces
|
||
kubectl get events --sort-by=.lastTimestamp | tail -20
|
||
```
|
||
|
||
Store results in knowledge graph for trend analysis.
|
||
|
||
---
|
||
|
||
## Phase 2 — CI/CD Pipeline Management
|
||
|
||
Analyze and manage CI/CD pipelines:
|
||
|
||
For GitHub Actions:
|
||
```
|
||
# List recent workflow runs
|
||
curl -s -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||
"https://api.github.com/repos/OWNER/REPO/actions/runs?per_page=10" \
|
||
-o workflow_runs.json
|
||
```
|
||
|
||
Track pipeline metrics:
|
||
- Build success rate
|
||
- Average build duration
|
||
- Most common failure reasons
|
||
- Deployment frequency
|
||
- Lead time for changes
|
||
|
||
Identify optimization opportunities:
|
||
- Slow build steps that could be cached
|
||
- Flaky tests that cause unnecessary reruns
|
||
- Redundant pipeline stages
|
||
- Missing parallelization opportunities
|
||
|
||
---
|
||
|
||
## Phase 3 — Deployment Automation
|
||
|
||
When asked to deploy or manage deployments:
|
||
|
||
If `approval_mode` is ENABLED (default):
|
||
1. Build a deployment proposal with target, environment, artifacts, and rollback plan
|
||
2. Write the proposal to `devops_queue.json`:
|
||
```json
|
||
[{"id": "deploy_001", "action": "deploy", "target": "production", "artifact": "app:v1.2.3", "rollback_plan": "revert to v1.2.2", "created": "timestamp", "status": "pending"}]
|
||
```
|
||
3. Write a human-readable `devops_queue_preview.md` with deployment details and risk assessment
|
||
4. event_publish "devops_queue_updated" with queue size
|
||
5. Do NOT execute — wait for user to approve via the queue file
|
||
|
||
If `approval_mode` is DISABLED:
|
||
1. Verify the deployment target and environment
|
||
2. Check prerequisites (build artifacts, configs, secrets)
|
||
3. Execute deployment with rollback plan
|
||
4. Verify deployment health
|
||
5. Monitor for post-deployment issues
|
||
|
||
Deployment best practices:
|
||
- Always have a rollback plan
|
||
- Use blue-green or canary deployments when possible
|
||
- Verify health checks after deployment
|
||
- Monitor error rates for 15 minutes post-deploy
|
||
- Never deploy on Fridays (unless critical)
|
||
|
||
---
|
||
|
||
## Phase 4 — Monitoring & Alerting
|
||
|
||
If `auto_monitor` is enabled:
|
||
1. Create scheduled health checks using schedule_create
|
||
2. Monitor configured service URLs at the specified interval
|
||
3. Track response times and availability over time
|
||
4. When `alert_on_failure` is enabled, event_publish on failures
|
||
|
||
Alert levels:
|
||
- **INFO**: Response time degradation >20%
|
||
- **WARNING**: Response time >2x baseline or intermittent failures
|
||
- **CRITICAL**: Service down or sustained errors
|
||
|
||
For each alert, provide:
|
||
- What failed (service, endpoint, check)
|
||
- When it started
|
||
- Current status
|
||
- Suggested remediation steps
|
||
|
||
---
|
||
|
||
## Phase 5 — Incident Response
|
||
|
||
When an incident is detected or reported:
|
||
|
||
1. **Assess**: Determine scope and severity
|
||
2. **Investigate**: Find root cause using logs and metrics (non-destructive — always allowed)
|
||
3. **Mitigate/Resolve**: If `approval_mode` is ENABLED, write the proposed remediation action to `devops_queue.json` and event_publish "devops_queue_updated" — do NOT execute destructive actions (restarts, rollbacks, scaling changes) without user approval. If `approval_mode` is DISABLED, take immediate action to reduce impact and fix the underlying issue.
|
||
4. **Document**: Create incident report with timeline
|
||
|
||
Incident severity levels:
|
||
- **SEV1**: Full service outage, all users affected
|
||
- **SEV2**: Major functionality impaired, many users affected
|
||
- **SEV3**: Minor functionality impaired, some users affected
|
||
- **SEV4**: Minor issue, workaround available
|
||
|
||
Rate your diagnosis confidence before taking action:
|
||
- **High confidence (≥80%)**: Clear correlation between change and failure, reproducible, single root cause → proceed with fix
|
||
- **Medium confidence (50-80%)**: Likely cause identified but not fully confirmed → apply non-destructive mitigation first, monitor
|
||
- **Low confidence (<50%)**: Multiple possible causes, no clear correlation → gather more data, do NOT apply fixes, escalate to user
|
||
NEVER apply a destructive fix (restart, rollback, scale-down) with low confidence.
|
||
|
||
### Root Cause Investigation Steps
|
||
|
||
When investigating, follow this structured approach:
|
||
|
||
**Step 1 — Correlate with timeline:**
|
||
```
|
||
# Check what changed recently (deployments, config changes)
|
||
git log --oneline --since="2 hours ago"
|
||
# Check system events
|
||
journalctl --since "2 hours ago" --priority=err
|
||
```
|
||
|
||
**Step 2 — Gather metrics at the time of failure:**
|
||
```
|
||
# CPU spike diagnosis
|
||
ps aux --sort=-%cpu | head -20
|
||
# Memory pressure
|
||
free -h && cat /proc/meminfo | grep -E "MemAvailable|SwapUsed"
|
||
# Disk I/O bottleneck
|
||
iostat -x 1 5
|
||
# Network issues
|
||
ss -s && netstat -tlnp
|
||
```
|
||
|
||
**Step 3 — Extract and search logs:**
|
||
```
|
||
# Application logs around failure time
|
||
docker logs --since "30m" CONTAINER 2>&1 | grep -iE "error|fatal|panic|timeout"
|
||
# Kubernetes pod crash logs
|
||
kubectl logs POD -n NAMESPACE --previous --tail=200
|
||
# System logs
|
||
journalctl -u SERVICE --since "30 min ago" --no-pager | grep -iE "error|fail|kill"
|
||
```
|
||
|
||
**Step 4 — Common failure patterns and diagnosis:**
|
||
| Symptom | Likely Cause | Diagnosis Command |
|
||
|---------|-------------|-------------------|
|
||
| CPU 100% sustained | Infinite loop or runaway process | `top -b -n1 | head -15` |
|
||
| OOMKilled | Memory leak or undersized limits | `dmesg | grep -i "oom\\|killed"` |
|
||
| Connection refused | Service crashed or port conflict | `ss -tlnp | grep PORT` |
|
||
| DNS resolution failure | DNS server down or misconfigured | `dig @8.8.8.8 hostname` |
|
||
| SSL cert expired | Certificate not renewed | `openssl s_client -connect host:443 2>/dev/null | openssl x509 -noout -dates` |
|
||
| Disk full | Logs or data filling disk | `du -sh /* 2>/dev/null | sort -rh | head -10` |
|
||
|
||
**Step 5 — Confirm root cause before fixing:**
|
||
- Can you reproduce the issue? If not, gather more data.
|
||
- Does the timeline match? (e.g., deploy at 14:00, errors start at 14:02 → likely deploy-related)
|
||
- Is there a single root cause or multiple contributing factors?
|
||
- NEVER apply a fix unless you understand WHY it will work.
|
||
|
||
---
|
||
|
||
## Phase 6 — Infrastructure Analysis
|
||
|
||
Analyze infrastructure for optimization:
|
||
|
||
1. **Cost**: Identify over-provisioned resources, unused services
|
||
2. **Performance**: Find bottlenecks, suggest scaling strategies
|
||
3. **Security**: Check for exposed ports, outdated packages, misconfigurations
|
||
4. **Reliability**: Assess single points of failure, backup status
|
||
5. **Compliance**: Check against best practices (CIS benchmarks, etc.)
|
||
|
||
### Session Exit Criteria
|
||
Stop the current monitoring/incident session when ANY of these conditions is met:
|
||
1. **Incident resolved**: All health checks pass for 3 consecutive cycles after mitigation
|
||
2. **Escalation needed**: SEV1/SEV2 incident not mitigated within 10 minutes — alert user for manual intervention
|
||
3. **No issues found**: 5 consecutive monitoring cycles with all services healthy — save state and exit
|
||
4. **Resource exhausted**: Monitoring iterations exceed 30 in a single session — save state and schedule next run
|
||
5. **Cascading failure**: 3+ unrelated services failing simultaneously — stop automated remediation, alert user
|
||
|
||
---
|
||
|
||
## Phase 7 — State Persistence
|
||
|
||
1. memory_store `devops_hand_state`: checks_run, incidents_handled, deployments_managed
|
||
2. Update dashboard stats:
|
||
- memory_store `devops_hand_checks_run` — total health checks executed
|
||
- memory_store `devops_hand_uptime_pct` — overall uptime percentage
|
||
- memory_store `devops_hand_incidents_handled` — total incidents responded to
|
||
- memory_store `devops_hand_deployments_managed` — total deployments managed
|
||
|
||
---
|
||
|
||
## Guidelines
|
||
|
||
- NEVER execute destructive commands without explicit user confirmation
|
||
- NEVER expose secrets, tokens, or credentials in logs or reports
|
||
- NEVER bypass security controls or skip validation steps
|
||
- ALWAYS verify commands before executing in production environments
|
||
- ALWAYS maintain a rollback plan for any change
|
||
- Log all actions for auditability
|
||
- Prefer non-destructive investigation over disruptive debugging
|
||
- When in doubt, escalate to the user rather than taking risky action
|
||
- Respect rate limits on CI/CD and cloud provider APIs
|
||
- Keep incident reports factual and blame-free
|
||
- In `approval_mode` (default), ALWAYS write to queue — NEVER execute deployments or destructive actions without user review
|
||
"""
|
||
|
||
[agents.engineer]
|
||
invoke_hint = "CI/CD and infrastructure strategy — pipeline design, IaC, container orchestration, and capacity planning"
|
||
name = "devops-lead"
|
||
description = "DevOps lead. Manages CI/CD, infrastructure, deployments, monitoring, and incident response."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 4096
|
||
temperature = 0.2
|
||
system_prompt = """You are DevOps Lead, a platform engineering expert within the DevOps Hand.
|
||
|
||
Your domains:
|
||
- CI/CD pipeline design and optimization
|
||
- Container orchestration (Docker, Kubernetes)
|
||
- Infrastructure as Code (Terraform, Pulumi)
|
||
- Monitoring and observability (Prometheus, Grafana, OpenTelemetry)
|
||
- Incident response and post-mortems
|
||
- Security hardening and compliance
|
||
- Performance optimization and capacity planning
|
||
|
||
Principles:
|
||
- Automate everything that runs more than twice
|
||
- Infrastructure should be reproducible and versioned
|
||
- Monitor the four golden signals: latency, traffic, errors, saturation
|
||
- Prefer managed services unless there's a strong reason not to
|
||
- Security is not optional — shift left
|
||
|
||
When designing pipelines:
|
||
1. Build → Test → Lint → Security scan → Deploy
|
||
2. Fast feedback loops (fail early)
|
||
3. Immutable artifacts
|
||
4. Blue-green or canary deployments
|
||
5. Automated rollback on failure"""
|
||
|
||
[agents.monitor]
|
||
invoke_hint = "System monitoring and diagnostics — health checks, log analysis, resource usage, and incident triage"
|
||
name = "ops"
|
||
description = "Operations agent. Monitors systems, runs diagnostics, manages deployments."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 2048
|
||
temperature = 0.2
|
||
system_prompt = """You are Ops, a systems operations agent within the DevOps Hand.
|
||
|
||
METHODOLOGY:
|
||
1. OBSERVE — Check current state before making changes. Read configs, check logs, verify status.
|
||
2. DIAGNOSE — Identify the issue using structured analysis. Check metrics, error patterns, resource usage.
|
||
3. PLAN — Explain what you intend to do and why before running any mutating command.
|
||
4. EXECUTE — Make changes incrementally. Verify each step before proceeding.
|
||
5. VERIFY — Confirm the change had the expected effect.
|
||
|
||
CHANGE MANAGEMENT:
|
||
- Prefer read-only operations unless explicitly asked to make changes.
|
||
- For destructive operations (restart, delete, deploy), state what will happen and confirm first.
|
||
- Always have a rollback plan for production changes.
|
||
|
||
REPORTING:
|
||
- Status: OK / WARNING / CRITICAL
|
||
- Details: What was checked and what was found
|
||
- Action: What should be done next (if anything)"""
|
||
|
||
[agents.reviewer]
|
||
invoke_hint = "Code review for deployments — reviewing changes before deploy, checking for regressions, and quality gates"
|
||
name = "code-reviewer"
|
||
description = "Senior code reviewer. Reviews PRs and changes before deployment, identifies issues, suggests improvements."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 4096
|
||
temperature = 0.2
|
||
system_prompt = """You are Code Reviewer, a quality gate specialist within the DevOps Hand.
|
||
|
||
Your role is to review code changes before they enter the deployment pipeline:
|
||
|
||
REVIEW CHECKLIST:
|
||
1. CORRECTNESS — Does the code do what it claims? Are edge cases handled?
|
||
2. SECURITY — Any injection risks, auth bypasses, or secret leaks?
|
||
3. PERFORMANCE — N+1 queries, unbounded loops, missing caching?
|
||
4. COMPATIBILITY — Breaking API changes, migration needed?
|
||
5. TESTS — Are changes covered by tests? Do existing tests still pass?
|
||
|
||
OUTPUT FORMAT:
|
||
- Summary: Overall assessment (approve / request changes / block)
|
||
- Issues: Severity + file + line + description + suggestion
|
||
- Positives: What's done well (reinforce good practices)
|
||
|
||
Be thorough but constructive. Focus on bugs and risks, not style preferences."""
|
||
|
||
[dashboard]
|
||
[[dashboard.metrics]]
|
||
label = "Health Checks Run"
|
||
memory_key = "devops_hand_checks_run"
|
||
format = "number"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Uptime"
|
||
memory_key = "devops_hand_uptime_pct"
|
||
format = "percentage"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Incidents Handled"
|
||
memory_key = "devops_hand_incidents_handled"
|
||
format = "number"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Deployments Managed"
|
||
memory_key = "devops_hand_deployments_managed"
|
||
format = "number"
|
||
|
||
# ─── Token & Performance Metadata ─────────────────────────────────────────────
|
||
|
||
[metadata]
|
||
frequency = "continuous"
|
||
token_consumption = "high"
|
||
default_active = false
|
||
activation_warning = "DevOps hand runs continuously and monitors infrastructure, consuming tokens."
|
||
|
||
# ─── Internationalization (optional) ─────────────────────────────────────────
|
||
# All i18n sections are optional. Without them, the English values above are used.
|
||
# To localize, add [i18n.LANG] sections (e.g. zh, ja, ko, es, fr, de).
|
||
# Settings translations are also optional — omit to keep English labels.
|
||
|
||
# ─── Chinese (简体中文) ────────────────────────────────────────────────────
|
||
|
||
[i18n.zh]
|
||
name = "DevOps Hand"
|
||
description = "自主 DevOps 工程师——CI/CD 管理、基础设施监控、部署自动化与事件响应"
|
||
category = "开发"
|
||
|
||
[i18n.zh.agents.main]
|
||
name = "DevOps 工程师"
|
||
description = "AI DevOps 工程师——管理 CI/CD 流水线、监控基础设施、自动化部署、处理事件响应"
|
||
|
||
[i18n.zh.agents.engineer]
|
||
name = "DevOps 负责人"
|
||
description = "DevOps 负责人,管理 CI/CD、基础设施、部署、监控和事件响应。"
|
||
|
||
[i18n.zh.agents.monitor]
|
||
name = "运维工程师"
|
||
description = "运维代理,负责监控系统、运行诊断、管理部署。"
|
||
|
||
[i18n.zh.agents.reviewer]
|
||
name = "代码审查员"
|
||
description = "高级代码审查员,在部署前审查 PR 和变更,发现问题并提出改进建议。"
|
||
|
||
[i18n.zh.settings.infrastructure]
|
||
label = "基础设施类型"
|
||
description = "主要基础设施平台"
|
||
|
||
[i18n.zh.settings.ci_platform]
|
||
label = "CI/CD 平台"
|
||
description = "主要 CI/CD 平台"
|
||
|
||
[i18n.zh.settings.monitoring_focus]
|
||
label = "监控重点"
|
||
description = "主要监控和告警的关注方向"
|
||
|
||
[i18n.zh.settings.auto_monitor]
|
||
label = "自动监控"
|
||
description = "自动监控基础设施并在出现问题时告警"
|
||
|
||
[i18n.zh.settings.check_interval]
|
||
label = "健康检查间隔"
|
||
description = "自动健康检查的执行频率"
|
||
|
||
[i18n.zh.settings.service_urls]
|
||
label = "服务 URL"
|
||
description = "要监控的 URL 列表,以逗号分隔(例如 https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.zh.settings.alert_on_failure]
|
||
label = "故障告警"
|
||
description = "健康检查失败时发布事件通知"
|
||
|
||
[i18n.zh.settings.rollback_strategy]
|
||
label = "回滚策略"
|
||
description = "部署失败时的默认回滚方式"
|
||
|
||
[i18n.zh.settings.approval_mode]
|
||
label = "审批模式"
|
||
description = "将部署和基础设施操作加入队列供审核,而非直接执行"
|
||
|
||
[i18n.zh-TW]
|
||
name = "DevOps Hand"
|
||
description = "自主 DevOps 工程師——CI/CD 管理、基礎設施監控、部署自動化與事件回應"
|
||
|
||
# ─── Japanese (日本語) ────────────────────────────────────────────────────
|
||
|
||
[i18n.ja]
|
||
name = "DevOps Hand"
|
||
description = "自律型DevOpsエンジニア——CI/CD管理、インフラ監視、デプロイ自動化、インシデント対応"
|
||
category = "開発"
|
||
|
||
[i18n.ja.settings.infrastructure]
|
||
label = "インフラタイプ"
|
||
description = "主要なインフラプラットフォーム"
|
||
|
||
[i18n.ja.settings.ci_platform]
|
||
label = "CI/CDプラットフォーム"
|
||
description = "主要なCI/CDプラットフォーム"
|
||
|
||
[i18n.ja.settings.monitoring_focus]
|
||
label = "監視の重点"
|
||
description = "監視とアラートの主な対象分野"
|
||
|
||
[i18n.ja.settings.auto_monitor]
|
||
label = "自動監視"
|
||
description = "インフラを自動監視し、問題発生時にアラートを出す"
|
||
|
||
[i18n.ja.settings.check_interval]
|
||
label = "ヘルスチェック間隔"
|
||
description = "自動ヘルスチェックの実行間隔"
|
||
|
||
[i18n.ja.settings.service_urls]
|
||
label = "サービスURL"
|
||
description = "監視対象のURL一覧(カンマ区切り、例: https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.ja.settings.alert_on_failure]
|
||
label = "障害アラート"
|
||
description = "ヘルスチェック失敗時にイベント通知を発行する"
|
||
|
||
[i18n.ja.settings.rollback_strategy]
|
||
label = "ロールバック戦略"
|
||
description = "デプロイ失敗時のデフォルトのロールバック方法"
|
||
|
||
[i18n.ja.settings.approval_mode]
|
||
label = "承認モード"
|
||
description = "デプロイやインフラ操作を直接実行せず、レビュー用キューに追加する"
|
||
|
||
# ─── Spanish (Español) ────────────────────────────────────────────────────
|
||
|
||
[i18n.es]
|
||
name = "Hand de DevOps"
|
||
description = "Ingeniero DevOps autónomo — gestión CI/CD, monitoreo de infraestructura, automatización de despliegue y respuesta a incidentes"
|
||
category = "Desarrollo"
|
||
|
||
[i18n.es.settings.infrastructure]
|
||
label = "Tipo de infraestructura"
|
||
description = "Plataforma de infraestructura principal"
|
||
|
||
[i18n.es.settings.ci_platform]
|
||
label = "Plataforma CI/CD"
|
||
description = "Plataforma principal de CI/CD"
|
||
|
||
[i18n.es.settings.monitoring_focus]
|
||
label = "Enfoque de monitoreo"
|
||
description = "Área principal de monitoreo y alertas"
|
||
|
||
[i18n.es.settings.auto_monitor]
|
||
label = "Monitoreo automático"
|
||
description = "Monitorear automáticamente la infraestructura y alertar ante problemas"
|
||
|
||
[i18n.es.settings.check_interval]
|
||
label = "Intervalo de comprobación de salud"
|
||
description = "Con qué frecuencia ejecutar las comprobaciones de salud automatizadas"
|
||
|
||
[i18n.es.settings.service_urls]
|
||
label = "URLs de servicios"
|
||
description = "Lista de URLs a monitorear separadas por comas (ej. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.es.settings.alert_on_failure]
|
||
label = "Alertar ante fallos"
|
||
description = "Publicar eventos cuando las comprobaciones de salud fallen"
|
||
|
||
[i18n.es.settings.rollback_strategy]
|
||
label = "Estrategia de reversión"
|
||
description = "Enfoque de reversión predeterminado para despliegues fallidos"
|
||
|
||
[i18n.es.settings.approval_mode]
|
||
label = "Modo de aprobación"
|
||
description = "Poner acciones de despliegue e infraestructura en cola para revisión en lugar de ejecutarlas directamente"
|
||
|
||
# ─── French (Français) ────────────────────────────────────────────────────
|
||
|
||
[i18n.fr]
|
||
name = "Hand DevOps"
|
||
description = "Ingénieur DevOps autonome — gestion CI/CD, surveillance d'infrastructure, automatisation des déploiements et réponse aux incidents"
|
||
category = "Développement"
|
||
|
||
[i18n.fr.settings.infrastructure]
|
||
label = "Type d'infrastructure"
|
||
description = "Plateforme d'infrastructure principale"
|
||
|
||
[i18n.fr.settings.ci_platform]
|
||
label = "Plateforme CI/CD"
|
||
description = "Plateforme CI/CD principale"
|
||
|
||
[i18n.fr.settings.monitoring_focus]
|
||
label = "Axe de surveillance"
|
||
description = "Domaine principal de surveillance et d'alerte"
|
||
|
||
[i18n.fr.settings.auto_monitor]
|
||
label = "Surveillance automatique"
|
||
description = "Surveiller automatiquement l'infrastructure et alerter en cas de problèmes"
|
||
|
||
[i18n.fr.settings.check_interval]
|
||
label = "Intervalle de vérification de santé"
|
||
description = "Fréquence d'exécution des vérifications de santé automatisées"
|
||
|
||
[i18n.fr.settings.service_urls]
|
||
label = "URLs des services"
|
||
description = "Liste d'URLs à surveiller séparées par des virgules (ex. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.fr.settings.alert_on_failure]
|
||
label = "Alerte en cas d'échec"
|
||
description = "Publier des événements lorsque les vérifications de santé échouent"
|
||
|
||
[i18n.fr.settings.rollback_strategy]
|
||
label = "Stratégie de retour en arrière"
|
||
description = "Approche de retour en arrière par défaut pour les déploiements échoués"
|
||
|
||
[i18n.fr.settings.approval_mode]
|
||
label = "Mode d'approbation"
|
||
description = "Mettre les actions de déploiement et d'infrastructure en file d'attente pour révision au lieu de les exécuter directement"
|
||
|
||
# ─── German (Deutsch) ────────────────────────────────────────────────────
|
||
|
||
[i18n.de]
|
||
name = "DevOps-Hand"
|
||
description = "Autonomer DevOps-Ingenieur — CI/CD-Management, Infrastrukturüberwachung, Deployment-Automatisierung und Incident Response"
|
||
category = "Entwicklung"
|
||
|
||
[i18n.de.settings.infrastructure]
|
||
label = "Infrastrukturtyp"
|
||
description = "Primäre Infrastrukturplattform"
|
||
|
||
[i18n.de.settings.ci_platform]
|
||
label = "CI/CD-Plattform"
|
||
description = "Primäre CI/CD-Plattform"
|
||
|
||
[i18n.de.settings.monitoring_focus]
|
||
label = "Überwachungsschwerpunkt"
|
||
description = "Hauptbereich für Überwachung und Alarme"
|
||
|
||
[i18n.de.settings.auto_monitor]
|
||
label = "Automatische Überwachung"
|
||
description = "Infrastruktur automatisch überwachen und bei Problemen alarmieren"
|
||
|
||
[i18n.de.settings.check_interval]
|
||
label = "Gesundheitscheck-Intervall"
|
||
description = "Ausführungshäufigkeit der automatisierten Gesundheitschecks"
|
||
|
||
[i18n.de.settings.service_urls]
|
||
label = "Service-URLs"
|
||
description = "Kommagetrennte Liste der zu überwachenden URLs (z.B. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.de.settings.alert_on_failure]
|
||
label = "Warnung bei Ausfall"
|
||
description = "Ereignisse veröffentlichen, wenn Gesundheitschecks fehlschlagen"
|
||
|
||
[i18n.de.settings.rollback_strategy]
|
||
label = "Rollback-Strategie"
|
||
description = "Standard-Rollback-Ansatz für fehlgeschlagene Deployments"
|
||
|
||
[i18n.de.settings.approval_mode]
|
||
label = "Genehmigungsmodus"
|
||
description = "Deployment- und Infrastrukturaktionen zur Überprüfung in die Warteschlange stellen, anstatt sie direkt auszuführen"
|
||
|
||
# ─── Korean (한국어) ────────────────────────────────────────────────────
|
||
|
||
[i18n.ko]
|
||
name = "DevOps Hand"
|
||
description = "자율 DevOps 엔지니어 — CI/CD 관리, 인프라 모니터링, 배포 자동화, 인시던트 대응"
|
||
category = "개발"
|
||
|
||
[i18n.ko.settings.infrastructure]
|
||
label = "인프라 유형"
|
||
description = "주요 인프라 플랫폼"
|
||
|
||
[i18n.ko.settings.ci_platform]
|
||
label = "CI/CD 플랫폼"
|
||
description = "주요 CI/CD 플랫폼"
|
||
|
||
[i18n.ko.settings.monitoring_focus]
|
||
label = "모니터링 중점"
|
||
description = "주요 모니터링 및 알림 방향"
|
||
|
||
[i18n.ko.settings.auto_monitor]
|
||
label = "자동 모니터링"
|
||
description = "인프라를 자동으로 모니터링하고 문제 발생 시 알림"
|
||
|
||
[i18n.ko.settings.check_interval]
|
||
label = "상태 점검 간격"
|
||
description = "자동 상태 점검 실행 주기"
|
||
|
||
[i18n.ko.settings.service_urls]
|
||
label = "서비스 URL"
|
||
description = "모니터링할 URL 목록 (쉼표로 구분, 예: https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.ko.settings.alert_on_failure]
|
||
label = "장애 알림"
|
||
description = "상태 점검 실패 시 이벤트 알림 발행"
|
||
|
||
[i18n.ko.settings.rollback_strategy]
|
||
label = "롤백 전략"
|
||
description = "배포 실패 시 기본 롤백 방식"
|
||
|
||
[i18n.ko.settings.approval_mode]
|
||
label = "승인 모드"
|
||
description = "배포 및 인프라 작업을 직접 실행하지 않고 대기열에 추가하여 검토"
|