Add [i18n.zh.agents.*] sections to all 15 HAND.toml files, providing Chinese translations for agent names and descriptions. Total: 51 agent translations across 15 hands.
931 lines
32 KiB
TOML
931 lines
32 KiB
TOML
id = "devops"
|
||
version = "1.1.0"
|
||
name = "DevOps Hand"
|
||
description = "Autonomous DevOps engineer — CI/CD management, infrastructure monitoring, deployment automation, and incident response"
|
||
|
||
category = "development"
|
||
icon = "👷"
|
||
|
||
tools = [
|
||
"shell_exec",
|
||
"file_read",
|
||
"file_write",
|
||
"file_list",
|
||
"web_fetch",
|
||
"web_search",
|
||
"memory_store",
|
||
"memory_recall",
|
||
"schedule_create",
|
||
"schedule_list",
|
||
"schedule_delete",
|
||
"knowledge_add_entity",
|
||
"knowledge_add_relation",
|
||
"knowledge_query",
|
||
"event_publish",
|
||
]
|
||
|
||
[[requires]]
|
||
key = "curl"
|
||
label = "curl must be installed"
|
||
requirement_type = "binary"
|
||
check_value = "curl"
|
||
description = "curl is used for HTTP health checks, GitHub API calls, and service endpoint monitoring."
|
||
|
||
[requires.install]
|
||
macos = "brew install curl"
|
||
linux_apt = "sudo apt install curl"
|
||
linux_dnf = "sudo dnf install curl"
|
||
linux_pacman = "sudo pacman -S curl"
|
||
windows = "winget install cURL.cURL"
|
||
estimated_time = "1 min"
|
||
|
||
[[requires]]
|
||
key = "git"
|
||
label = "git must be installed"
|
||
requirement_type = "binary"
|
||
check_value = "git"
|
||
description = "git is used for deployment history, version control operations, and CI/CD pipeline management."
|
||
|
||
[requires.install]
|
||
macos = "brew install git"
|
||
linux_apt = "sudo apt install git"
|
||
linux_dnf = "sudo dnf install git"
|
||
linux_pacman = "sudo pacman -S git"
|
||
windows = "winget install Git.Git"
|
||
estimated_time = "1-2 min"
|
||
|
||
[[requires]]
|
||
key = "docker"
|
||
label = "Docker (optional — needed for container workloads)"
|
||
requirement_type = "binary"
|
||
check_value = "docker"
|
||
optional = true
|
||
description = "Docker is used for container status checks, image management, and service orchestration. Only needed if your infrastructure uses containers."
|
||
|
||
[requires.install]
|
||
macos = "brew install --cask docker"
|
||
linux_apt = "sudo apt install docker.io"
|
||
linux_dnf = "sudo dnf install docker"
|
||
linux_pacman = "sudo pacman -S docker"
|
||
windows = "winget install Docker.DockerDesktop"
|
||
manual_url = "https://docs.docker.com/get-docker/"
|
||
estimated_time = "5-10 min"
|
||
|
||
[[requires]]
|
||
key = "GITHUB_TOKEN"
|
||
label = "GitHub Token (optional — needed for GitHub Actions)"
|
||
requirement_type = "api_key"
|
||
check_value = "GITHUB_TOKEN"
|
||
optional = true
|
||
description = "A GitHub personal access token for accessing GitHub Actions API, checking pipeline status, and triggering workflows."
|
||
|
||
[requires.install]
|
||
signup_url = "https://github.com/settings/tokens"
|
||
docs_url = "https://docs.github.com/en/authentication/keeping-your-account-and-data-secure/managing-your-personal-access-tokens"
|
||
env_example = "GITHUB_TOKEN=ghp_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx"
|
||
estimated_time = "2-5 min"
|
||
steps = [
|
||
"Go to GitHub Settings → Developer settings → Personal access tokens → Fine-grained tokens",
|
||
"Click 'Generate new token'",
|
||
"Select repository access scope and permissions (Actions: read, Contents: read)",
|
||
"Copy the token and set it as GITHUB_TOKEN environment variable",
|
||
]
|
||
|
||
[routing]
|
||
aliases = [
|
||
"ci/cd",
|
||
"pipeline",
|
||
"github actions",
|
||
"infrastructure monitoring",
|
||
"deployment automation",
|
||
"incident response",
|
||
]
|
||
weak_aliases = [
|
||
"deploy",
|
||
"kubernetes",
|
||
"docker",
|
||
"container",
|
||
"terraform",
|
||
"helm",
|
||
]
|
||
|
||
# ─── Configurable settings ───────────────────────────────────────────────────
|
||
|
||
[[settings]]
|
||
key = "infrastructure"
|
||
label = "Infrastructure Type"
|
||
description = "Primary infrastructure platform"
|
||
setting_type = "select"
|
||
default = "cloud"
|
||
|
||
[[settings.options]]
|
||
value = "cloud"
|
||
label = "Cloud (AWS/GCP/Azure)"
|
||
|
||
[[settings.options]]
|
||
value = "kubernetes"
|
||
label = "Kubernetes"
|
||
|
||
[[settings.options]]
|
||
value = "docker"
|
||
label = "Docker / Docker Compose"
|
||
|
||
[[settings.options]]
|
||
value = "bare_metal"
|
||
label = "Bare Metal / VPS"
|
||
|
||
[[settings.options]]
|
||
value = "serverless"
|
||
label = "Serverless"
|
||
|
||
[[settings]]
|
||
key = "ci_platform"
|
||
label = "CI/CD Platform"
|
||
description = "Primary CI/CD platform"
|
||
setting_type = "select"
|
||
default = "github_actions"
|
||
|
||
[[settings.options]]
|
||
value = "github_actions"
|
||
label = "GitHub Actions"
|
||
|
||
[[settings.options]]
|
||
value = "gitlab_ci"
|
||
label = "GitLab CI"
|
||
|
||
[[settings.options]]
|
||
value = "jenkins"
|
||
label = "Jenkins"
|
||
|
||
[[settings.options]]
|
||
value = "circleci"
|
||
label = "CircleCI"
|
||
|
||
[[settings.options]]
|
||
value = "other"
|
||
label = "Other"
|
||
|
||
[[settings]]
|
||
key = "monitoring_focus"
|
||
label = "Monitoring Focus"
|
||
description = "Primary monitoring and alerting focus"
|
||
setting_type = "select"
|
||
default = "balanced"
|
||
|
||
[[settings.options]]
|
||
value = "uptime"
|
||
label = "Uptime & Availability"
|
||
|
||
[[settings.options]]
|
||
value = "performance"
|
||
label = "Performance & Latency"
|
||
|
||
[[settings.options]]
|
||
value = "security"
|
||
label = "Security & Compliance"
|
||
|
||
[[settings.options]]
|
||
value = "cost"
|
||
label = "Cost Optimization"
|
||
|
||
[[settings.options]]
|
||
value = "balanced"
|
||
label = "Balanced (all areas)"
|
||
|
||
[[settings]]
|
||
key = "auto_monitor"
|
||
label = "Auto Monitor"
|
||
description = "Automatically monitor infrastructure and alert on issues"
|
||
setting_type = "toggle"
|
||
default = "false"
|
||
|
||
[[settings]]
|
||
key = "check_interval"
|
||
label = "Health Check Interval"
|
||
description = "How often to run automated health checks"
|
||
setting_type = "select"
|
||
default = "5min"
|
||
|
||
[[settings.options]]
|
||
value = "1min"
|
||
label = "Every minute"
|
||
|
||
[[settings.options]]
|
||
value = "5min"
|
||
label = "Every 5 minutes"
|
||
|
||
[[settings.options]]
|
||
value = "15min"
|
||
label = "Every 15 minutes"
|
||
|
||
[[settings.options]]
|
||
value = "1hour"
|
||
label = "Every hour"
|
||
|
||
[[settings]]
|
||
key = "service_urls"
|
||
label = "Service URLs"
|
||
description = "Comma-separated URLs to monitor (e.g. https://api.example.com/health,https://app.example.com)"
|
||
setting_type = "text"
|
||
default = ""
|
||
|
||
[[settings]]
|
||
key = "alert_on_failure"
|
||
label = "Alert on Failure"
|
||
description = "Publish events when health checks fail"
|
||
setting_type = "toggle"
|
||
default = "true"
|
||
|
||
[[settings]]
|
||
key = "rollback_strategy"
|
||
label = "Rollback Strategy"
|
||
description = "Default rollback approach for failed deployments"
|
||
setting_type = "select"
|
||
default = "manual"
|
||
|
||
[[settings.options]]
|
||
value = "manual"
|
||
label = "Manual (alert and wait for user)"
|
||
|
||
[[settings.options]]
|
||
value = "auto_previous"
|
||
label = "Auto-rollback to previous version"
|
||
|
||
[[settings.options]]
|
||
value = "blue_green"
|
||
label = "Blue-green switch back"
|
||
|
||
[[settings]]
|
||
key = "approval_mode"
|
||
label = "Approval Mode"
|
||
description = "Queue deployment and infrastructure actions for your review instead of executing directly"
|
||
setting_type = "toggle"
|
||
default = "true"
|
||
|
||
# ─── Agent configuration ─────────────────────────────────────────────────────
|
||
|
||
[agents.main]
|
||
coordinator = true
|
||
name = "devops-hand"
|
||
description = "AI DevOps engineer — manages CI/CD pipelines, monitors infrastructure, automates deployments, and handles incident response"
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 16384
|
||
temperature = 0.2
|
||
max_iterations = 60
|
||
system_prompt = """You are DevOps Hand — an autonomous DevOps engineer that manages CI/CD pipelines, monitors infrastructure health, automates deployments, and handles incident response.
|
||
|
||
## Phase 0 — Environment Detection (ALWAYS DO THIS FIRST)
|
||
|
||
Detect the operating system and available tools:
|
||
```
|
||
python -c "import platform; print(platform.system())"
|
||
```
|
||
|
||
Check available DevOps tools:
|
||
```
|
||
docker --version 2>/dev/null
|
||
kubectl version --client 2>/dev/null
|
||
terraform --version 2>/dev/null
|
||
git --version
|
||
curl --version | head -1
|
||
```
|
||
|
||
Load context:
|
||
1. memory_recall `devops_hand_state` — load previous monitoring data and incident history
|
||
2. Read **User Configuration** for infrastructure, ci_platform, service_urls, approval_mode, etc.
|
||
3. file_read `devops_queue.json` if it exists — pending deployment/remediation actions
|
||
4. knowledge_query for known infrastructure topology and previous incidents
|
||
|
||
---
|
||
|
||
## Phase 1 — Infrastructure Health Check
|
||
|
||
Check the health of all configured services:
|
||
|
||
For each URL in `service_urls`:
|
||
```
|
||
curl -s -o /dev/null -w "%{http_code} %{time_total}" --max-time 10 "$URL"
|
||
```
|
||
|
||
Record:
|
||
- HTTP status code
|
||
- Response time
|
||
- SSL certificate expiry (if HTTPS)
|
||
- DNS resolution time
|
||
|
||
For Docker environments:
|
||
```
|
||
docker ps --format "table {{.Names}}\t{{.Status}}\t{{.Ports}}"
|
||
docker stats --no-stream --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}"
|
||
```
|
||
|
||
For Kubernetes environments:
|
||
```
|
||
kubectl get pods --all-namespaces -o wide
|
||
kubectl top pods --all-namespaces
|
||
kubectl get events --sort-by=.lastTimestamp | tail -20
|
||
```
|
||
|
||
Store results in knowledge graph for trend analysis.
|
||
|
||
---
|
||
|
||
## Phase 2 — CI/CD Pipeline Management
|
||
|
||
Analyze and manage CI/CD pipelines:
|
||
|
||
For GitHub Actions:
|
||
```
|
||
# List recent workflow runs
|
||
curl -s -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||
"https://api.github.com/repos/OWNER/REPO/actions/runs?per_page=10" \
|
||
-o workflow_runs.json
|
||
```
|
||
|
||
Track pipeline metrics:
|
||
- Build success rate
|
||
- Average build duration
|
||
- Most common failure reasons
|
||
- Deployment frequency
|
||
- Lead time for changes
|
||
|
||
Identify optimization opportunities:
|
||
- Slow build steps that could be cached
|
||
- Flaky tests that cause unnecessary reruns
|
||
- Redundant pipeline stages
|
||
- Missing parallelization opportunities
|
||
|
||
---
|
||
|
||
## Phase 3 — Deployment Automation
|
||
|
||
When asked to deploy or manage deployments:
|
||
|
||
If `approval_mode` is ENABLED (default):
|
||
1. Build a deployment proposal with target, environment, artifacts, and rollback plan
|
||
2. Write the proposal to `devops_queue.json`:
|
||
```json
|
||
[{"id": "deploy_001", "action": "deploy", "target": "production", "artifact": "app:v1.2.3", "rollback_plan": "revert to v1.2.2", "created": "timestamp", "status": "pending"}]
|
||
```
|
||
3. Write a human-readable `devops_queue_preview.md` with deployment details and risk assessment
|
||
4. event_publish "devops_queue_updated" with queue size
|
||
5. Do NOT execute — wait for user to approve via the queue file
|
||
|
||
If `approval_mode` is DISABLED:
|
||
1. Verify the deployment target and environment
|
||
2. Check prerequisites (build artifacts, configs, secrets)
|
||
3. Execute deployment with rollback plan
|
||
4. Verify deployment health
|
||
5. Monitor for post-deployment issues
|
||
|
||
Deployment best practices:
|
||
- Always have a rollback plan
|
||
- Use blue-green or canary deployments when possible
|
||
- Verify health checks after deployment
|
||
- Monitor error rates for 15 minutes post-deploy
|
||
- Never deploy on Fridays (unless critical)
|
||
|
||
---
|
||
|
||
## Phase 4 — Monitoring & Alerting
|
||
|
||
If `auto_monitor` is enabled:
|
||
1. Create scheduled health checks using schedule_create
|
||
2. Monitor configured service URLs at the specified interval
|
||
3. Track response times and availability over time
|
||
4. When `alert_on_failure` is enabled, event_publish on failures
|
||
|
||
Alert levels:
|
||
- **INFO**: Response time degradation >20%
|
||
- **WARNING**: Response time >2x baseline or intermittent failures
|
||
- **CRITICAL**: Service down or sustained errors
|
||
|
||
For each alert, provide:
|
||
- What failed (service, endpoint, check)
|
||
- When it started
|
||
- Current status
|
||
- Suggested remediation steps
|
||
|
||
---
|
||
|
||
## Phase 5 — Incident Response
|
||
|
||
When an incident is detected or reported:
|
||
|
||
1. **Assess**: Determine scope and severity
|
||
2. **Investigate**: Find root cause using logs and metrics (non-destructive — always allowed)
|
||
3. **Mitigate/Resolve**: If `approval_mode` is ENABLED, write the proposed remediation action to `devops_queue.json` and event_publish "devops_queue_updated" — do NOT execute destructive actions (restarts, rollbacks, scaling changes) without user approval. If `approval_mode` is DISABLED, take immediate action to reduce impact and fix the underlying issue.
|
||
4. **Document**: Create incident report with timeline
|
||
|
||
Incident severity levels:
|
||
- **SEV1**: Full service outage, all users affected
|
||
- **SEV2**: Major functionality impaired, many users affected
|
||
- **SEV3**: Minor functionality impaired, some users affected
|
||
- **SEV4**: Minor issue, workaround available
|
||
|
||
Rate your diagnosis confidence before taking action:
|
||
- **High confidence (≥80%)**: Clear correlation between change and failure, reproducible, single root cause → proceed with fix
|
||
- **Medium confidence (50-80%)**: Likely cause identified but not fully confirmed → apply non-destructive mitigation first, monitor
|
||
- **Low confidence (<50%)**: Multiple possible causes, no clear correlation → gather more data, do NOT apply fixes, escalate to user
|
||
NEVER apply a destructive fix (restart, rollback, scale-down) with low confidence.
|
||
|
||
### Root Cause Investigation Steps
|
||
|
||
When investigating, follow this structured approach:
|
||
|
||
**Step 1 — Correlate with timeline:**
|
||
```
|
||
# Check what changed recently (deployments, config changes)
|
||
git log --oneline --since="2 hours ago"
|
||
# Check system events
|
||
journalctl --since "2 hours ago" --priority=err
|
||
```
|
||
|
||
**Step 2 — Gather metrics at the time of failure:**
|
||
```
|
||
# CPU spike diagnosis
|
||
ps aux --sort=-%cpu | head -20
|
||
# Memory pressure
|
||
free -h && cat /proc/meminfo | grep -E "MemAvailable|SwapUsed"
|
||
# Disk I/O bottleneck
|
||
iostat -x 1 5
|
||
# Network issues
|
||
ss -s && netstat -tlnp
|
||
```
|
||
|
||
**Step 3 — Extract and search logs:**
|
||
```
|
||
# Application logs around failure time
|
||
docker logs --since "30m" CONTAINER 2>&1 | grep -iE "error|fatal|panic|timeout"
|
||
# Kubernetes pod crash logs
|
||
kubectl logs POD -n NAMESPACE --previous --tail=200
|
||
# System logs
|
||
journalctl -u SERVICE --since "30 min ago" --no-pager | grep -iE "error|fail|kill"
|
||
```
|
||
|
||
**Step 4 — Common failure patterns and diagnosis:**
|
||
| Symptom | Likely Cause | Diagnosis Command |
|
||
|---------|-------------|-------------------|
|
||
| CPU 100% sustained | Infinite loop or runaway process | `top -b -n1 | head -15` |
|
||
| OOMKilled | Memory leak or undersized limits | `dmesg | grep -i "oom\\|killed"` |
|
||
| Connection refused | Service crashed or port conflict | `ss -tlnp | grep PORT` |
|
||
| DNS resolution failure | DNS server down or misconfigured | `dig @8.8.8.8 hostname` |
|
||
| SSL cert expired | Certificate not renewed | `openssl s_client -connect host:443 2>/dev/null | openssl x509 -noout -dates` |
|
||
| Disk full | Logs or data filling disk | `du -sh /* 2>/dev/null | sort -rh | head -10` |
|
||
|
||
**Step 5 — Confirm root cause before fixing:**
|
||
- Can you reproduce the issue? If not, gather more data.
|
||
- Does the timeline match? (e.g., deploy at 14:00, errors start at 14:02 → likely deploy-related)
|
||
- Is there a single root cause or multiple contributing factors?
|
||
- NEVER apply a fix unless you understand WHY it will work.
|
||
|
||
---
|
||
|
||
## Phase 6 — Infrastructure Analysis
|
||
|
||
Analyze infrastructure for optimization:
|
||
|
||
1. **Cost**: Identify over-provisioned resources, unused services
|
||
2. **Performance**: Find bottlenecks, suggest scaling strategies
|
||
3. **Security**: Check for exposed ports, outdated packages, misconfigurations
|
||
4. **Reliability**: Assess single points of failure, backup status
|
||
5. **Compliance**: Check against best practices (CIS benchmarks, etc.)
|
||
|
||
### Session Exit Criteria
|
||
Stop the current monitoring/incident session when ANY of these conditions is met:
|
||
1. **Incident resolved**: All health checks pass for 3 consecutive cycles after mitigation
|
||
2. **Escalation needed**: SEV1/SEV2 incident not mitigated within 10 minutes — alert user for manual intervention
|
||
3. **No issues found**: 5 consecutive monitoring cycles with all services healthy — save state and exit
|
||
4. **Resource exhausted**: Monitoring iterations exceed 30 in a single session — save state and schedule next run
|
||
5. **Cascading failure**: 3+ unrelated services failing simultaneously — stop automated remediation, alert user
|
||
|
||
---
|
||
|
||
## Phase 7 — State Persistence
|
||
|
||
1. memory_store `devops_hand_state`: checks_run, incidents_handled, deployments_managed
|
||
2. Update dashboard stats:
|
||
- memory_store `devops_hand_checks_run` — total health checks executed
|
||
- memory_store `devops_hand_uptime_pct` — overall uptime percentage
|
||
- memory_store `devops_hand_incidents_handled` — total incidents responded to
|
||
- memory_store `devops_hand_deployments_managed` — total deployments managed
|
||
|
||
---
|
||
|
||
## Guidelines
|
||
|
||
- NEVER execute destructive commands without explicit user confirmation
|
||
- NEVER expose secrets, tokens, or credentials in logs or reports
|
||
- NEVER bypass security controls or skip validation steps
|
||
- ALWAYS verify commands before executing in production environments
|
||
- ALWAYS maintain a rollback plan for any change
|
||
- Log all actions for auditability
|
||
- Prefer non-destructive investigation over disruptive debugging
|
||
- When in doubt, escalate to the user rather than taking risky action
|
||
- Respect rate limits on CI/CD and cloud provider APIs
|
||
- Keep incident reports factual and blame-free
|
||
- In `approval_mode` (default), ALWAYS write to queue — NEVER execute deployments or destructive actions without user review
|
||
"""
|
||
|
||
[agents.engineer]
|
||
invoke_hint = "CI/CD and infrastructure strategy — pipeline design, IaC, container orchestration, and capacity planning"
|
||
name = "devops-lead"
|
||
description = "DevOps lead. Manages CI/CD, infrastructure, deployments, monitoring, and incident response."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 4096
|
||
temperature = 0.2
|
||
system_prompt = """You are DevOps Lead, a platform engineering expert within the DevOps Hand.
|
||
|
||
Your domains:
|
||
- CI/CD pipeline design and optimization
|
||
- Container orchestration (Docker, Kubernetes)
|
||
- Infrastructure as Code (Terraform, Pulumi)
|
||
- Monitoring and observability (Prometheus, Grafana, OpenTelemetry)
|
||
- Incident response and post-mortems
|
||
- Security hardening and compliance
|
||
- Performance optimization and capacity planning
|
||
|
||
Principles:
|
||
- Automate everything that runs more than twice
|
||
- Infrastructure should be reproducible and versioned
|
||
- Monitor the four golden signals: latency, traffic, errors, saturation
|
||
- Prefer managed services unless there's a strong reason not to
|
||
- Security is not optional — shift left
|
||
|
||
When designing pipelines:
|
||
1. Build → Test → Lint → Security scan → Deploy
|
||
2. Fast feedback loops (fail early)
|
||
3. Immutable artifacts
|
||
4. Blue-green or canary deployments
|
||
5. Automated rollback on failure"""
|
||
|
||
[agents.monitor]
|
||
invoke_hint = "System monitoring and diagnostics — health checks, log analysis, resource usage, and incident triage"
|
||
name = "ops"
|
||
description = "Operations agent. Monitors systems, runs diagnostics, manages deployments."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 2048
|
||
temperature = 0.2
|
||
system_prompt = """You are Ops, a systems operations agent within the DevOps Hand.
|
||
|
||
METHODOLOGY:
|
||
1. OBSERVE — Check current state before making changes. Read configs, check logs, verify status.
|
||
2. DIAGNOSE — Identify the issue using structured analysis. Check metrics, error patterns, resource usage.
|
||
3. PLAN — Explain what you intend to do and why before running any mutating command.
|
||
4. EXECUTE — Make changes incrementally. Verify each step before proceeding.
|
||
5. VERIFY — Confirm the change had the expected effect.
|
||
|
||
CHANGE MANAGEMENT:
|
||
- Prefer read-only operations unless explicitly asked to make changes.
|
||
- For destructive operations (restart, delete, deploy), state what will happen and confirm first.
|
||
- Always have a rollback plan for production changes.
|
||
|
||
REPORTING:
|
||
- Status: OK / WARNING / CRITICAL
|
||
- Details: What was checked and what was found
|
||
- Action: What should be done next (if anything)"""
|
||
|
||
[agents.reviewer]
|
||
invoke_hint = "Code review for deployments — reviewing changes before deploy, checking for regressions, and quality gates"
|
||
name = "code-reviewer"
|
||
description = "Senior code reviewer. Reviews PRs and changes before deployment, identifies issues, suggests improvements."
|
||
module = "builtin:chat"
|
||
provider = "default"
|
||
model = "default"
|
||
max_tokens = 4096
|
||
temperature = 0.2
|
||
system_prompt = """You are Code Reviewer, a quality gate specialist within the DevOps Hand.
|
||
|
||
Your role is to review code changes before they enter the deployment pipeline:
|
||
|
||
REVIEW CHECKLIST:
|
||
1. CORRECTNESS — Does the code do what it claims? Are edge cases handled?
|
||
2. SECURITY — Any injection risks, auth bypasses, or secret leaks?
|
||
3. PERFORMANCE — N+1 queries, unbounded loops, missing caching?
|
||
4. COMPATIBILITY — Breaking API changes, migration needed?
|
||
5. TESTS — Are changes covered by tests? Do existing tests still pass?
|
||
|
||
OUTPUT FORMAT:
|
||
- Summary: Overall assessment (approve / request changes / block)
|
||
- Issues: Severity + file + line + description + suggestion
|
||
- Positives: What's done well (reinforce good practices)
|
||
|
||
Be thorough but constructive. Focus on bugs and risks, not style preferences."""
|
||
|
||
[dashboard]
|
||
[[dashboard.metrics]]
|
||
label = "Health Checks Run"
|
||
memory_key = "devops_hand_checks_run"
|
||
format = "number"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Uptime"
|
||
memory_key = "devops_hand_uptime_pct"
|
||
format = "percentage"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Incidents Handled"
|
||
memory_key = "devops_hand_incidents_handled"
|
||
format = "number"
|
||
|
||
[[dashboard.metrics]]
|
||
label = "Deployments Managed"
|
||
memory_key = "devops_hand_deployments_managed"
|
||
format = "number"
|
||
|
||
# ─── Token & Performance Metadata ─────────────────────────────────────────────
|
||
|
||
[metadata]
|
||
frequency = "continuous"
|
||
token_consumption = "high"
|
||
default_active = false
|
||
activation_warning = "DevOps hand runs continuously and monitors infrastructure, consuming tokens."
|
||
|
||
# ─── Internationalization (optional) ─────────────────────────────────────────
|
||
# All i18n sections are optional. Without them, the English values above are used.
|
||
# To localize, add [i18n.LANG] sections (e.g. zh, ja, ko, es, fr, de).
|
||
# Settings translations are also optional — omit to keep English labels.
|
||
|
||
# ─── Chinese (简体中文) ────────────────────────────────────────────────────
|
||
|
||
[i18n.zh]
|
||
name = "DevOps Hand"
|
||
description = "自主 DevOps 工程师——CI/CD 管理、基础设施监控、部署自动化与事件响应"
|
||
category = "开发"
|
||
|
||
[i18n.zh.agents.main]
|
||
name = "DevOps 工程师"
|
||
description = "AI DevOps 工程师——管理 CI/CD 流水线、监控基础设施、自动化部署、处理事件响应"
|
||
|
||
[i18n.zh.agents.engineer]
|
||
name = "DevOps 负责人"
|
||
description = "DevOps 负责人,管理 CI/CD、基础设施、部署、监控和事件响应。"
|
||
|
||
[i18n.zh.agents.monitor]
|
||
name = "运维工程师"
|
||
description = "运维代理,负责监控系统、运行诊断、管理部署。"
|
||
|
||
[i18n.zh.agents.reviewer]
|
||
name = "代码审查员"
|
||
description = "高级代码审查员,在部署前审查 PR 和变更,发现问题并提出改进建议。"
|
||
|
||
[i18n.zh.settings.infrastructure]
|
||
label = "基础设施类型"
|
||
description = "主要基础设施平台"
|
||
|
||
[i18n.zh.settings.ci_platform]
|
||
label = "CI/CD 平台"
|
||
description = "主要 CI/CD 平台"
|
||
|
||
[i18n.zh.settings.monitoring_focus]
|
||
label = "监控重点"
|
||
description = "主要监控和告警的关注方向"
|
||
|
||
[i18n.zh.settings.auto_monitor]
|
||
label = "自动监控"
|
||
description = "自动监控基础设施并在出现问题时告警"
|
||
|
||
[i18n.zh.settings.check_interval]
|
||
label = "健康检查间隔"
|
||
description = "自动健康检查的执行频率"
|
||
|
||
[i18n.zh.settings.service_urls]
|
||
label = "服务 URL"
|
||
description = "要监控的 URL 列表,以逗号分隔(例如 https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.zh.settings.alert_on_failure]
|
||
label = "故障告警"
|
||
description = "健康检查失败时发布事件通知"
|
||
|
||
[i18n.zh.settings.rollback_strategy]
|
||
label = "回滚策略"
|
||
description = "部署失败时的默认回滚方式"
|
||
|
||
[i18n.zh.settings.approval_mode]
|
||
label = "审批模式"
|
||
description = "将部署和基础设施操作加入队列供审核,而非直接执行"
|
||
|
||
[i18n.zh-TW]
|
||
description = "自主 DevOps 工程師——CI/CD 管理、基礎設施監控、部署自動化與事件回應"
|
||
|
||
# ─── Japanese (日本語) ────────────────────────────────────────────────────
|
||
|
||
[i18n.ja]
|
||
name = "DevOps Hand"
|
||
description = "自律型DevOpsエンジニア——CI/CD管理、インフラ監視、デプロイ自動化、インシデント対応"
|
||
category = "開発"
|
||
|
||
[i18n.ja.settings.infrastructure]
|
||
label = "インフラタイプ"
|
||
description = "主要なインフラプラットフォーム"
|
||
|
||
[i18n.ja.settings.ci_platform]
|
||
label = "CI/CDプラットフォーム"
|
||
description = "主要なCI/CDプラットフォーム"
|
||
|
||
[i18n.ja.settings.monitoring_focus]
|
||
label = "監視の重点"
|
||
description = "監視とアラートの主な対象分野"
|
||
|
||
[i18n.ja.settings.auto_monitor]
|
||
label = "自動監視"
|
||
description = "インフラを自動監視し、問題発生時にアラートを出す"
|
||
|
||
[i18n.ja.settings.check_interval]
|
||
label = "ヘルスチェック間隔"
|
||
description = "自動ヘルスチェックの実行間隔"
|
||
|
||
[i18n.ja.settings.service_urls]
|
||
label = "サービスURL"
|
||
description = "監視対象のURL一覧(カンマ区切り、例: https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.ja.settings.alert_on_failure]
|
||
label = "障害アラート"
|
||
description = "ヘルスチェック失敗時にイベント通知を発行する"
|
||
|
||
[i18n.ja.settings.rollback_strategy]
|
||
label = "ロールバック戦略"
|
||
description = "デプロイ失敗時のデフォルトのロールバック方法"
|
||
|
||
[i18n.ja.settings.approval_mode]
|
||
label = "承認モード"
|
||
description = "デプロイやインフラ操作を直接実行せず、レビュー用キューに追加する"
|
||
|
||
# ─── Spanish (Español) ────────────────────────────────────────────────────
|
||
|
||
[i18n.es]
|
||
name = "Hand de DevOps"
|
||
description = "Ingeniero DevOps autónomo — gestión CI/CD, monitoreo de infraestructura, automatización de despliegue y respuesta a incidentes"
|
||
category = "Desarrollo"
|
||
|
||
[i18n.es.settings.infrastructure]
|
||
label = "Tipo de infraestructura"
|
||
description = "Plataforma de infraestructura principal"
|
||
|
||
[i18n.es.settings.ci_platform]
|
||
label = "Plataforma CI/CD"
|
||
description = "Plataforma principal de CI/CD"
|
||
|
||
[i18n.es.settings.monitoring_focus]
|
||
label = "Enfoque de monitoreo"
|
||
description = "Área principal de monitoreo y alertas"
|
||
|
||
[i18n.es.settings.auto_monitor]
|
||
label = "Monitoreo automático"
|
||
description = "Monitorear automáticamente la infraestructura y alertar ante problemas"
|
||
|
||
[i18n.es.settings.check_interval]
|
||
label = "Intervalo de comprobación de salud"
|
||
description = "Con qué frecuencia ejecutar las comprobaciones de salud automatizadas"
|
||
|
||
[i18n.es.settings.service_urls]
|
||
label = "URLs de servicios"
|
||
description = "Lista de URLs a monitorear separadas por comas (ej. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.es.settings.alert_on_failure]
|
||
label = "Alertar ante fallos"
|
||
description = "Publicar eventos cuando las comprobaciones de salud fallen"
|
||
|
||
[i18n.es.settings.rollback_strategy]
|
||
label = "Estrategia de reversión"
|
||
description = "Enfoque de reversión predeterminado para despliegues fallidos"
|
||
|
||
[i18n.es.settings.approval_mode]
|
||
label = "Modo de aprobación"
|
||
description = "Poner acciones de despliegue e infraestructura en cola para revisión en lugar de ejecutarlas directamente"
|
||
|
||
# ─── French (Français) ────────────────────────────────────────────────────
|
||
|
||
[i18n.fr]
|
||
name = "Hand DevOps"
|
||
description = "Ingénieur DevOps autonome — gestion CI/CD, surveillance d'infrastructure, automatisation des déploiements et réponse aux incidents"
|
||
category = "Développement"
|
||
|
||
[i18n.fr.settings.infrastructure]
|
||
label = "Type d'infrastructure"
|
||
description = "Plateforme d'infrastructure principale"
|
||
|
||
[i18n.fr.settings.ci_platform]
|
||
label = "Plateforme CI/CD"
|
||
description = "Plateforme CI/CD principale"
|
||
|
||
[i18n.fr.settings.monitoring_focus]
|
||
label = "Axe de surveillance"
|
||
description = "Domaine principal de surveillance et d'alerte"
|
||
|
||
[i18n.fr.settings.auto_monitor]
|
||
label = "Surveillance automatique"
|
||
description = "Surveiller automatiquement l'infrastructure et alerter en cas de problèmes"
|
||
|
||
[i18n.fr.settings.check_interval]
|
||
label = "Intervalle de vérification de santé"
|
||
description = "Fréquence d'exécution des vérifications de santé automatisées"
|
||
|
||
[i18n.fr.settings.service_urls]
|
||
label = "URLs des services"
|
||
description = "Liste d'URLs à surveiller séparées par des virgules (ex. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.fr.settings.alert_on_failure]
|
||
label = "Alerte en cas d'échec"
|
||
description = "Publier des événements lorsque les vérifications de santé échouent"
|
||
|
||
[i18n.fr.settings.rollback_strategy]
|
||
label = "Stratégie de retour en arrière"
|
||
description = "Approche de retour en arrière par défaut pour les déploiements échoués"
|
||
|
||
[i18n.fr.settings.approval_mode]
|
||
label = "Mode d'approbation"
|
||
description = "Mettre les actions de déploiement et d'infrastructure en file d'attente pour révision au lieu de les exécuter directement"
|
||
|
||
# ─── German (Deutsch) ────────────────────────────────────────────────────
|
||
|
||
[i18n.de]
|
||
name = "DevOps-Hand"
|
||
description = "Autonomer DevOps-Ingenieur — CI/CD-Management, Infrastrukturüberwachung, Deployment-Automatisierung und Incident Response"
|
||
category = "Entwicklung"
|
||
|
||
[i18n.de.settings.infrastructure]
|
||
label = "Infrastrukturtyp"
|
||
description = "Primäre Infrastrukturplattform"
|
||
|
||
[i18n.de.settings.ci_platform]
|
||
label = "CI/CD-Plattform"
|
||
description = "Primäre CI/CD-Plattform"
|
||
|
||
[i18n.de.settings.monitoring_focus]
|
||
label = "Überwachungsschwerpunkt"
|
||
description = "Hauptbereich für Überwachung und Alarme"
|
||
|
||
[i18n.de.settings.auto_monitor]
|
||
label = "Automatische Überwachung"
|
||
description = "Infrastruktur automatisch überwachen und bei Problemen alarmieren"
|
||
|
||
[i18n.de.settings.check_interval]
|
||
label = "Gesundheitscheck-Intervall"
|
||
description = "Ausführungshäufigkeit der automatisierten Gesundheitschecks"
|
||
|
||
[i18n.de.settings.service_urls]
|
||
label = "Service-URLs"
|
||
description = "Kommagetrennte Liste der zu überwachenden URLs (z.B. https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.de.settings.alert_on_failure]
|
||
label = "Warnung bei Ausfall"
|
||
description = "Ereignisse veröffentlichen, wenn Gesundheitschecks fehlschlagen"
|
||
|
||
[i18n.de.settings.rollback_strategy]
|
||
label = "Rollback-Strategie"
|
||
description = "Standard-Rollback-Ansatz für fehlgeschlagene Deployments"
|
||
|
||
[i18n.de.settings.approval_mode]
|
||
label = "Genehmigungsmodus"
|
||
description = "Deployment- und Infrastrukturaktionen zur Überprüfung in die Warteschlange stellen, anstatt sie direkt auszuführen"
|
||
|
||
# ─── Korean (한국어) ────────────────────────────────────────────────────
|
||
|
||
[i18n.ko]
|
||
name = "DevOps Hand"
|
||
description = "자율 DevOps 엔지니어 — CI/CD 관리, 인프라 모니터링, 배포 자동화, 인시던트 대응"
|
||
category = "개발"
|
||
|
||
[i18n.ko.settings.infrastructure]
|
||
label = "인프라 유형"
|
||
description = "주요 인프라 플랫폼"
|
||
|
||
[i18n.ko.settings.ci_platform]
|
||
label = "CI/CD 플랫폼"
|
||
description = "주요 CI/CD 플랫폼"
|
||
|
||
[i18n.ko.settings.monitoring_focus]
|
||
label = "모니터링 중점"
|
||
description = "주요 모니터링 및 알림 방향"
|
||
|
||
[i18n.ko.settings.auto_monitor]
|
||
label = "자동 모니터링"
|
||
description = "인프라를 자동으로 모니터링하고 문제 발생 시 알림"
|
||
|
||
[i18n.ko.settings.check_interval]
|
||
label = "상태 점검 간격"
|
||
description = "자동 상태 점검 실행 주기"
|
||
|
||
[i18n.ko.settings.service_urls]
|
||
label = "서비스 URL"
|
||
description = "모니터링할 URL 목록 (쉼표로 구분, 예: https://api.example.com/health,https://app.example.com)"
|
||
|
||
[i18n.ko.settings.alert_on_failure]
|
||
label = "장애 알림"
|
||
description = "상태 점검 실패 시 이벤트 알림 발행"
|
||
|
||
[i18n.ko.settings.rollback_strategy]
|
||
label = "롤백 전략"
|
||
description = "배포 실패 시 기본 롤백 방식"
|
||
|
||
[i18n.ko.settings.approval_mode]
|
||
label = "승인 모드"
|
||
description = "배포 및 인프라 작업을 직접 실행하지 않고 대기열에 추가하여 검토"
|