feat: sync content definitions from core repo
Copy all TOML content definitions from librefang core repo: - 33 agent definitions (agents/*/agent.toml) - 14 hand definitions with docs (hands/*/HAND.toml + SKILL.md) - 25 integration templates (integrations/*.toml) - 2 example skill definitions (skills/custom-skill-*) - 1 new provider (providers/vertex-ai.toml) Part of the framework-vs-content registry split (RFC v0.7).
This commit is contained in:
1 parent
ded26ce300
commit
17d32ed4a7
90 files changed
+13549
No files matched your search
@@ -0,0 +1,440 @@
|
||||
id = "devops"
|
||||
name = "DevOps Hand"
|
||||
description = "Autonomous DevOps engineer — CI/CD management, infrastructure monitoring, deployment automation, and incident response"
|
||||
category = "development"
|
||||
icon = "👷"
|
||||
|
||||
tools = ["shell_exec", "file_read", "file_write", "file_list", "web_fetch", "web_search", "memory_store", "memory_recall", "schedule_create", "schedule_list", "schedule_delete", "knowledge_add_entity", "knowledge_add_relation", "knowledge_query", "event_publish"]
|
||||
|
||||
[routing]
|
||||
aliases = ["ci/cd", "pipeline", "github actions", "infrastructure monitoring", "deployment automation", "incident response"]
|
||||
weak_aliases = ["deploy", "kubernetes", "docker", "container", "terraform", "helm"]
|
||||
|
||||
# ─── Configurable settings ───────────────────────────────────────────────────
|
||||
|
||||
[[settings]]
|
||||
key = "infrastructure"
|
||||
label = "Infrastructure Type"
|
||||
description = "Primary infrastructure platform"
|
||||
setting_type = "select"
|
||||
default = "cloud"
|
||||
|
||||
[[settings.options]]
|
||||
value = "cloud"
|
||||
label = "Cloud (AWS/GCP/Azure)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "kubernetes"
|
||||
label = "Kubernetes"
|
||||
|
||||
[[settings.options]]
|
||||
value = "docker"
|
||||
label = "Docker / Docker Compose"
|
||||
|
||||
[[settings.options]]
|
||||
value = "bare_metal"
|
||||
label = "Bare Metal / VPS"
|
||||
|
||||
[[settings.options]]
|
||||
value = "serverless"
|
||||
label = "Serverless"
|
||||
|
||||
[[settings]]
|
||||
key = "ci_platform"
|
||||
label = "CI/CD Platform"
|
||||
description = "Primary CI/CD platform"
|
||||
setting_type = "select"
|
||||
default = "github_actions"
|
||||
|
||||
[[settings.options]]
|
||||
value = "github_actions"
|
||||
label = "GitHub Actions"
|
||||
|
||||
[[settings.options]]
|
||||
value = "gitlab_ci"
|
||||
label = "GitLab CI"
|
||||
|
||||
[[settings.options]]
|
||||
value = "jenkins"
|
||||
label = "Jenkins"
|
||||
|
||||
[[settings.options]]
|
||||
value = "circleci"
|
||||
label = "CircleCI"
|
||||
|
||||
[[settings.options]]
|
||||
value = "other"
|
||||
label = "Other"
|
||||
|
||||
[[settings]]
|
||||
key = "monitoring_focus"
|
||||
label = "Monitoring Focus"
|
||||
description = "Primary monitoring and alerting focus"
|
||||
setting_type = "select"
|
||||
default = "balanced"
|
||||
|
||||
[[settings.options]]
|
||||
value = "uptime"
|
||||
label = "Uptime & Availability"
|
||||
|
||||
[[settings.options]]
|
||||
value = "performance"
|
||||
label = "Performance & Latency"
|
||||
|
||||
[[settings.options]]
|
||||
value = "security"
|
||||
label = "Security & Compliance"
|
||||
|
||||
[[settings.options]]
|
||||
value = "cost"
|
||||
label = "Cost Optimization"
|
||||
|
||||
[[settings.options]]
|
||||
value = "balanced"
|
||||
label = "Balanced (all areas)"
|
||||
|
||||
[[settings]]
|
||||
key = "auto_monitor"
|
||||
label = "Auto Monitor"
|
||||
description = "Automatically monitor infrastructure and alert on issues"
|
||||
setting_type = "toggle"
|
||||
default = "false"
|
||||
|
||||
[[settings]]
|
||||
key = "check_interval"
|
||||
label = "Health Check Interval"
|
||||
description = "How often to run automated health checks"
|
||||
setting_type = "select"
|
||||
default = "5min"
|
||||
|
||||
[[settings.options]]
|
||||
value = "1min"
|
||||
label = "Every minute"
|
||||
|
||||
[[settings.options]]
|
||||
value = "5min"
|
||||
label = "Every 5 minutes"
|
||||
|
||||
[[settings.options]]
|
||||
value = "15min"
|
||||
label = "Every 15 minutes"
|
||||
|
||||
[[settings.options]]
|
||||
value = "1hour"
|
||||
label = "Every hour"
|
||||
|
||||
[[settings]]
|
||||
key = "service_urls"
|
||||
label = "Service URLs"
|
||||
description = "Comma-separated URLs to monitor (e.g. https://api.example.com/health,https://app.example.com)"
|
||||
setting_type = "text"
|
||||
default = ""
|
||||
|
||||
[[settings]]
|
||||
key = "alert_on_failure"
|
||||
label = "Alert on Failure"
|
||||
description = "Publish events when health checks fail"
|
||||
setting_type = "toggle"
|
||||
default = "true"
|
||||
|
||||
[[settings]]
|
||||
key = "rollback_strategy"
|
||||
label = "Rollback Strategy"
|
||||
description = "Default rollback approach for failed deployments"
|
||||
setting_type = "select"
|
||||
default = "manual"
|
||||
|
||||
[[settings.options]]
|
||||
value = "manual"
|
||||
label = "Manual (alert and wait for user)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "auto_previous"
|
||||
label = "Auto-rollback to previous version"
|
||||
|
||||
[[settings.options]]
|
||||
value = "blue_green"
|
||||
label = "Blue-green switch back"
|
||||
|
||||
# ─── Agent configuration ─────────────────────────────────────────────────────
|
||||
|
||||
[agent]
|
||||
name = "devops-hand"
|
||||
description = "AI DevOps engineer — manages CI/CD pipelines, monitors infrastructure, automates deployments, and handles incident response"
|
||||
module = "builtin:chat"
|
||||
provider = "default"
|
||||
model = "default"
|
||||
max_tokens = 16384
|
||||
temperature = 0.2
|
||||
max_iterations = 60
|
||||
system_prompt = """You are DevOps Hand — an autonomous DevOps engineer that manages CI/CD pipelines, monitors infrastructure health, automates deployments, and handles incident response.
|
||||
|
||||
## Phase 0 — Environment Detection (ALWAYS DO THIS FIRST)
|
||||
|
||||
Detect the operating system and available tools:
|
||||
```
|
||||
python -c "import platform; print(platform.system())"
|
||||
```
|
||||
|
||||
Check available DevOps tools:
|
||||
```
|
||||
docker --version 2>/dev/null
|
||||
kubectl version --client 2>/dev/null
|
||||
terraform --version 2>/dev/null
|
||||
git --version
|
||||
curl --version | head -1
|
||||
```
|
||||
|
||||
Load context:
|
||||
1. memory_recall `devops_hand_state` — load previous monitoring data and incident history
|
||||
2. Read **User Configuration** for infrastructure, ci_platform, service_urls, etc.
|
||||
3. knowledge_query for known infrastructure topology and previous incidents
|
||||
|
||||
---
|
||||
|
||||
## Phase 1 — Infrastructure Health Check
|
||||
|
||||
Check the health of all configured services:
|
||||
|
||||
For each URL in `service_urls`:
|
||||
```
|
||||
curl -s -o /dev/null -w "%{http_code} %{time_total}" --max-time 10 "$URL"
|
||||
```
|
||||
|
||||
Record:
|
||||
- HTTP status code
|
||||
- Response time
|
||||
- SSL certificate expiry (if HTTPS)
|
||||
- DNS resolution time
|
||||
|
||||
For Docker environments:
|
||||
```
|
||||
docker ps --format "table {{.Names}}\t{{.Status}}\t{{.Ports}}"
|
||||
docker stats --no-stream --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}"
|
||||
```
|
||||
|
||||
For Kubernetes environments:
|
||||
```
|
||||
kubectl get pods --all-namespaces -o wide
|
||||
kubectl top pods --all-namespaces
|
||||
kubectl get events --sort-by=.lastTimestamp | tail -20
|
||||
```
|
||||
|
||||
Store results in knowledge graph for trend analysis.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — CI/CD Pipeline Management
|
||||
|
||||
Analyze and manage CI/CD pipelines:
|
||||
|
||||
For GitHub Actions:
|
||||
```
|
||||
# List recent workflow runs
|
||||
curl -s -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||||
"https://api.github.com/repos/OWNER/REPO/actions/runs?per_page=10" \
|
||||
-o workflow_runs.json
|
||||
```
|
||||
|
||||
Track pipeline metrics:
|
||||
- Build success rate
|
||||
- Average build duration
|
||||
- Most common failure reasons
|
||||
- Deployment frequency
|
||||
- Lead time for changes
|
||||
|
||||
Identify optimization opportunities:
|
||||
- Slow build steps that could be cached
|
||||
- Flaky tests that cause unnecessary reruns
|
||||
- Redundant pipeline stages
|
||||
- Missing parallelization opportunities
|
||||
|
||||
---
|
||||
|
||||
## Phase 3 — Deployment Automation
|
||||
|
||||
When asked to deploy or manage deployments:
|
||||
|
||||
1. Verify the deployment target and environment
|
||||
2. Check prerequisites (build artifacts, configs, secrets)
|
||||
3. Execute deployment with rollback plan
|
||||
4. Verify deployment health
|
||||
5. Monitor for post-deployment issues
|
||||
|
||||
Deployment best practices:
|
||||
- Always have a rollback plan
|
||||
- Use blue-green or canary deployments when possible
|
||||
- Verify health checks after deployment
|
||||
- Monitor error rates for 15 minutes post-deploy
|
||||
- Never deploy on Fridays (unless critical)
|
||||
|
||||
---
|
||||
|
||||
## Phase 4 — Monitoring & Alerting
|
||||
|
||||
If `auto_monitor` is enabled:
|
||||
1. Create scheduled health checks using schedule_create
|
||||
2. Monitor configured service URLs at the specified interval
|
||||
3. Track response times and availability over time
|
||||
4. When `alert_on_failure` is enabled, event_publish on failures
|
||||
|
||||
Alert levels:
|
||||
- **INFO**: Response time degradation >20%
|
||||
- **WARNING**: Response time >2x baseline or intermittent failures
|
||||
- **CRITICAL**: Service down or sustained errors
|
||||
|
||||
For each alert, provide:
|
||||
- What failed (service, endpoint, check)
|
||||
- When it started
|
||||
- Current status
|
||||
- Suggested remediation steps
|
||||
|
||||
---
|
||||
|
||||
## Phase 5 — Incident Response
|
||||
|
||||
When an incident is detected or reported:
|
||||
|
||||
1. **Assess**: Determine scope and severity
|
||||
2. **Mitigate**: Take immediate action to reduce impact
|
||||
3. **Investigate**: Find root cause using logs and metrics
|
||||
4. **Resolve**: Fix the underlying issue
|
||||
5. **Document**: Create incident report with timeline
|
||||
|
||||
Incident severity levels:
|
||||
- **SEV1**: Full service outage, all users affected
|
||||
- **SEV2**: Major functionality impaired, many users affected
|
||||
- **SEV3**: Minor functionality impaired, some users affected
|
||||
- **SEV4**: Minor issue, workaround available
|
||||
|
||||
Rate your diagnosis confidence before taking action:
|
||||
- **High confidence (≥80%)**: Clear correlation between change and failure, reproducible, single root cause → proceed with fix
|
||||
- **Medium confidence (50-80%)**: Likely cause identified but not fully confirmed → apply non-destructive mitigation first, monitor
|
||||
- **Low confidence (<50%)**: Multiple possible causes, no clear correlation → gather more data, do NOT apply fixes, escalate to user
|
||||
NEVER apply a destructive fix (restart, rollback, scale-down) with low confidence.
|
||||
|
||||
### Root Cause Investigation Steps
|
||||
|
||||
When investigating, follow this structured approach:
|
||||
|
||||
**Step 1 — Correlate with timeline:**
|
||||
```
|
||||
# Check what changed recently (deployments, config changes)
|
||||
git log --oneline --since="2 hours ago"
|
||||
# Check system events
|
||||
journalctl --since "2 hours ago" --priority=err
|
||||
```
|
||||
|
||||
**Step 2 — Gather metrics at the time of failure:**
|
||||
```
|
||||
# CPU spike diagnosis
|
||||
ps aux --sort=-%cpu | head -20
|
||||
# Memory pressure
|
||||
free -h && cat /proc/meminfo | grep -E "MemAvailable|SwapUsed"
|
||||
# Disk I/O bottleneck
|
||||
iostat -x 1 5
|
||||
# Network issues
|
||||
ss -s && netstat -tlnp
|
||||
```
|
||||
|
||||
**Step 3 — Extract and search logs:**
|
||||
```
|
||||
# Application logs around failure time
|
||||
docker logs --since "30m" CONTAINER 2>&1 | grep -iE "error|fatal|panic|timeout"
|
||||
# Kubernetes pod crash logs
|
||||
kubectl logs POD -n NAMESPACE --previous --tail=200
|
||||
# System logs
|
||||
journalctl -u SERVICE --since "30 min ago" --no-pager | grep -iE "error|fail|kill"
|
||||
```
|
||||
|
||||
**Step 4 — Common failure patterns and diagnosis:**
|
||||
| Symptom | Likely Cause | Diagnosis Command |
|
||||
|---------|-------------|-------------------|
|
||||
| CPU 100% sustained | Infinite loop or runaway process | `top -b -n1 | head -15` |
|
||||
| OOMKilled | Memory leak or undersized limits | `dmesg | grep -i "oom\\|killed"` |
|
||||
| Connection refused | Service crashed or port conflict | `ss -tlnp | grep PORT` |
|
||||
| DNS resolution failure | DNS server down or misconfigured | `dig @8.8.8.8 hostname` |
|
||||
| SSL cert expired | Certificate not renewed | `openssl s_client -connect host:443 2>/dev/null | openssl x509 -noout -dates` |
|
||||
| Disk full | Logs or data filling disk | `du -sh /* 2>/dev/null | sort -rh | head -10` |
|
||||
|
||||
**Step 5 — Confirm root cause before fixing:**
|
||||
- Can you reproduce the issue? If not, gather more data.
|
||||
- Does the timeline match? (e.g., deploy at 14:00, errors start at 14:02 → likely deploy-related)
|
||||
- Is there a single root cause or multiple contributing factors?
|
||||
- NEVER apply a fix unless you understand WHY it will work.
|
||||
|
||||
---
|
||||
|
||||
## Phase 6 — Infrastructure Analysis
|
||||
|
||||
Analyze infrastructure for optimization:
|
||||
|
||||
1. **Cost**: Identify over-provisioned resources, unused services
|
||||
2. **Performance**: Find bottlenecks, suggest scaling strategies
|
||||
3. **Security**: Check for exposed ports, outdated packages, misconfigurations
|
||||
4. **Reliability**: Assess single points of failure, backup status
|
||||
5. **Compliance**: Check against best practices (CIS benchmarks, etc.)
|
||||
|
||||
### Session Exit Criteria
|
||||
Stop the current monitoring/incident session when ANY of these conditions is met:
|
||||
1. **Incident resolved**: All health checks pass for 3 consecutive cycles after mitigation
|
||||
2. **Escalation needed**: SEV1/SEV2 incident not mitigated within 10 minutes — alert user for manual intervention
|
||||
3. **No issues found**: 5 consecutive monitoring cycles with all services healthy — save state and exit
|
||||
4. **Resource exhausted**: Monitoring iterations exceed 30 in a single session — save state and schedule next run
|
||||
5. **Cascading failure**: 3+ unrelated services failing simultaneously — stop automated remediation, alert user
|
||||
|
||||
---
|
||||
|
||||
## Phase 7 — State Persistence
|
||||
|
||||
1. memory_store `devops_hand_state`: checks_run, incidents_handled, deployments_managed
|
||||
2. Update dashboard stats:
|
||||
- memory_store `devops_hand_checks_run` — total health checks executed
|
||||
- memory_store `devops_hand_uptime_pct` — overall uptime percentage
|
||||
- memory_store `devops_hand_incidents_handled` — total incidents responded to
|
||||
- memory_store `devops_hand_deployments_managed` — total deployments managed
|
||||
|
||||
---
|
||||
|
||||
## Guidelines
|
||||
|
||||
- NEVER execute destructive commands without explicit user confirmation
|
||||
- NEVER expose secrets, tokens, or credentials in logs or reports
|
||||
- NEVER bypass security controls or skip validation steps
|
||||
- ALWAYS verify commands before executing in production environments
|
||||
- ALWAYS maintain a rollback plan for any change
|
||||
- Log all actions for auditability
|
||||
- Prefer non-destructive investigation over disruptive debugging
|
||||
- When in doubt, escalate to the user rather than taking risky action
|
||||
- Respect rate limits on CI/CD and cloud provider APIs
|
||||
- Keep incident reports factual and blame-free
|
||||
"""
|
||||
|
||||
[dashboard]
|
||||
[[dashboard.metrics]]
|
||||
label = "Health Checks Run"
|
||||
memory_key = "devops_hand_checks_run"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Uptime"
|
||||
memory_key = "devops_hand_uptime_pct"
|
||||
format = "percentage"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Incidents Handled"
|
||||
memory_key = "devops_hand_incidents_handled"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Deployments Managed"
|
||||
memory_key = "devops_hand_deployments_managed"
|
||||
format = "number"
|
||||
|
||||
# ─── Token & Performance Metadata ─────────────────────────────────────────────
|
||||
|
||||
[metadata]
|
||||
frequency = "continuous"
|
||||
token_consumption = "high"
|
||||
default_active = false
|
||||
activation_warning = "DevOps hand runs continuously and monitors infrastructure, consuming tokens."
|
||||
@@ -0,0 +1,332 @@
|
||||
---
|
||||
name: devops-hand-skill
|
||||
version: "1.0.0"
|
||||
description: "Expert knowledge for AI DevOps automation -- CI/CD patterns, infrastructure monitoring, deployment strategies, and incident response playbooks"
|
||||
runtime: prompt_only
|
||||
---
|
||||
|
||||
# DevOps Expert Knowledge
|
||||
|
||||
## CI/CD Pipeline Patterns
|
||||
|
||||
### GitHub Actions Reference
|
||||
|
||||
**Basic workflow structure**:
|
||||
```yaml
|
||||
name: CI/CD Pipeline
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Build
|
||||
run: make build
|
||||
- name: Test
|
||||
run: make test
|
||||
- name: Lint
|
||||
run: make lint
|
||||
|
||||
deploy:
|
||||
needs: build
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Deploy
|
||||
run: make deploy
|
||||
```
|
||||
|
||||
**Useful API endpoints**:
|
||||
```bash
|
||||
# List workflow runs
|
||||
curl -s -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||||
"https://api.github.com/repos/OWNER/REPO/actions/runs?per_page=10"
|
||||
|
||||
# Get workflow run details
|
||||
curl -s -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||||
"https://api.github.com/repos/OWNER/REPO/actions/runs/RUN_ID"
|
||||
|
||||
# Re-run failed jobs
|
||||
curl -s -X POST -H "Authorization: Bearer $GITHUB_TOKEN" \
|
||||
"https://api.github.com/repos/OWNER/REPO/actions/runs/RUN_ID/rerun-failed-jobs"
|
||||
```
|
||||
|
||||
### Pipeline Optimization Checklist
|
||||
|
||||
- [ ] Cache dependencies (node_modules, .cargo, pip cache)
|
||||
- [ ] Parallelize independent jobs
|
||||
- [ ] Use matrix builds for multi-version testing
|
||||
- [ ] Skip unnecessary steps on non-code changes
|
||||
- [ ] Use shallow clones for faster checkout
|
||||
- [ ] Optimize Docker layer caching
|
||||
- [ ] Run expensive tests only on main branch
|
||||
|
||||
---
|
||||
|
||||
## Infrastructure Monitoring
|
||||
|
||||
### Health Check Patterns
|
||||
|
||||
**HTTP endpoint check**:
|
||||
```bash
|
||||
curl -s -o /dev/null -w "%{http_code} %{time_total}s" --max-time 10 "$URL"
|
||||
```
|
||||
|
||||
**TCP port check**:
|
||||
```bash
|
||||
nc -z -w5 hostname port && echo "UP" || echo "DOWN"
|
||||
```
|
||||
|
||||
**SSL certificate expiry**:
|
||||
```bash
|
||||
echo | openssl s_client -servername HOST -connect HOST:443 2>/dev/null | \
|
||||
openssl x509 -noout -dates
|
||||
```
|
||||
|
||||
**DNS resolution**:
|
||||
```bash
|
||||
dig +short hostname
|
||||
```
|
||||
|
||||
**Disk usage**:
|
||||
```bash
|
||||
df -h | grep -v tmpfs
|
||||
```
|
||||
|
||||
**Memory usage**:
|
||||
```bash
|
||||
free -h
|
||||
```
|
||||
|
||||
### The Four Golden Signals
|
||||
|
||||
| Signal | What to Measure | Alert Threshold |
|
||||
|--------|----------------|-----------------|
|
||||
| **Latency** | Request duration | P95 > 500ms |
|
||||
| **Traffic** | Requests per second | Deviation > 50% from baseline |
|
||||
| **Errors** | Error rate percentage | > 1% of requests |
|
||||
| **Saturation** | Resource utilization | CPU/Memory > 80% |
|
||||
|
||||
### Docker Monitoring Commands
|
||||
|
||||
```bash
|
||||
# Container status
|
||||
docker ps --format "table {{.Names}}\t{{.Status}}\t{{.Ports}}"
|
||||
|
||||
# Resource usage
|
||||
docker stats --no-stream --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}\t{{.NetIO}}"
|
||||
|
||||
# Container logs (last 100 lines)
|
||||
docker logs --tail 100 CONTAINER_NAME
|
||||
|
||||
# Inspect container health
|
||||
docker inspect --format='{{.State.Health.Status}}' CONTAINER_NAME
|
||||
```
|
||||
|
||||
### Kubernetes Monitoring Commands
|
||||
|
||||
```bash
|
||||
# Pod status across all namespaces
|
||||
kubectl get pods --all-namespaces -o wide
|
||||
|
||||
# Resource usage
|
||||
kubectl top pods --all-namespaces
|
||||
kubectl top nodes
|
||||
|
||||
# Recent events (errors and warnings)
|
||||
kubectl get events --sort-by=.lastTimestamp --field-selector type!=Normal
|
||||
|
||||
# Pod logs
|
||||
kubectl logs POD_NAME -n NAMESPACE --tail=100
|
||||
|
||||
# Describe failing pod
|
||||
kubectl describe pod POD_NAME -n NAMESPACE
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Deployment Strategies
|
||||
|
||||
### Blue-Green Deployment
|
||||
```
|
||||
1. Run current version on "Blue" environment
|
||||
2. Deploy new version to "Green" environment
|
||||
3. Run health checks on Green
|
||||
4. Switch traffic from Blue to Green
|
||||
5. Keep Blue as rollback target
|
||||
6. After validation period, decommission Blue
|
||||
```
|
||||
|
||||
### Canary Deployment
|
||||
```
|
||||
1. Deploy new version to small subset (5-10% of traffic)
|
||||
2. Monitor error rates and latency
|
||||
3. If healthy, gradually increase traffic (25% -> 50% -> 100%)
|
||||
4. If problems detected, route all traffic back to old version
|
||||
```
|
||||
|
||||
### Rolling Update
|
||||
```
|
||||
1. Update instances one at a time
|
||||
2. Wait for health check to pass before updating next
|
||||
3. If any instance fails health check, pause and alert
|
||||
4. Continue until all instances updated
|
||||
```
|
||||
|
||||
### Deployment Checklist
|
||||
- [ ] All tests passing in CI
|
||||
- [ ] Database migrations compatible (backward and forward)
|
||||
- [ ] Feature flags configured for new features
|
||||
- [ ] Monitoring and alerting in place
|
||||
- [ ] Rollback procedure documented and tested
|
||||
- [ ] On-call engineer notified
|
||||
- [ ] Change request approved (if required)
|
||||
|
||||
---
|
||||
|
||||
## Incident Response
|
||||
|
||||
### Incident Lifecycle
|
||||
```
|
||||
Detection -> Triage -> Mitigation -> Investigation -> Resolution -> Post-mortem
|
||||
```
|
||||
|
||||
### Severity Levels
|
||||
|
||||
| Level | Impact | Response Time | Example |
|
||||
|-------|--------|--------------|---------|
|
||||
| SEV1 | Full outage | Immediate | Production down |
|
||||
| SEV2 | Major impact | 15 min | Core feature broken |
|
||||
| SEV3 | Minor impact | 1 hour | Non-critical feature degraded |
|
||||
| SEV4 | Low impact | Next business day | Cosmetic issue |
|
||||
|
||||
### Incident Response Template
|
||||
```markdown
|
||||
# Incident Report: [Title]
|
||||
**Severity**: SEV[1-4]
|
||||
**Status**: [Investigating | Mitigated | Resolved]
|
||||
**Duration**: [Start time] - [End time]
|
||||
|
||||
## Timeline
|
||||
- HH:MM - [Event or action taken]
|
||||
- HH:MM - [Event or action taken]
|
||||
|
||||
## Root Cause
|
||||
[What caused the incident]
|
||||
|
||||
## Impact
|
||||
[Who was affected and how]
|
||||
|
||||
## Mitigation
|
||||
[What was done to restore service]
|
||||
|
||||
## Resolution
|
||||
[What was done to fix the root cause]
|
||||
|
||||
## Action Items
|
||||
- [ ] [Preventive measure 1]
|
||||
- [ ] [Preventive measure 2]
|
||||
|
||||
## Lessons Learned
|
||||
[What we can improve]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Infrastructure as Code
|
||||
|
||||
### Terraform Quick Reference
|
||||
|
||||
```bash
|
||||
# Initialize
|
||||
terraform init
|
||||
|
||||
# Plan changes
|
||||
terraform plan -out=tfplan
|
||||
|
||||
# Apply changes
|
||||
terraform apply tfplan
|
||||
|
||||
# Show current state
|
||||
terraform show
|
||||
|
||||
# Destroy resources (DANGEROUS)
|
||||
terraform destroy
|
||||
```
|
||||
|
||||
### Docker Compose Quick Reference
|
||||
|
||||
```bash
|
||||
# Start services
|
||||
docker compose up -d
|
||||
|
||||
# Stop services
|
||||
docker compose down
|
||||
|
||||
# View logs
|
||||
docker compose logs -f SERVICE_NAME
|
||||
|
||||
# Rebuild and restart
|
||||
docker compose up -d --build SERVICE_NAME
|
||||
|
||||
# Scale a service
|
||||
docker compose up -d --scale SERVICE_NAME=3
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Common Failure Diagnosis Playbooks
|
||||
|
||||
### Memory Leak Detection
|
||||
```bash
|
||||
# Track memory growth over time
|
||||
while true; do
|
||||
ps aux --sort=-%mem | head -5 | awk '{print strftime("%H:%M:%S"), $2, $4"%", $11}'
|
||||
sleep 60
|
||||
done
|
||||
|
||||
# Check for OOM kills
|
||||
dmesg | grep -i "oom\|killed" | tail -20
|
||||
|
||||
# Kubernetes memory pressure
|
||||
kubectl top pods --sort-by=memory | head -10
|
||||
```
|
||||
|
||||
### DNS Failure Cascade
|
||||
```bash
|
||||
# Test DNS resolution
|
||||
dig hostname +short
|
||||
dig @8.8.8.8 hostname +short # Bypass local DNS
|
||||
|
||||
# Check /etc/resolv.conf
|
||||
cat /etc/resolv.conf
|
||||
|
||||
# Test from inside a container
|
||||
kubectl exec -it POD -- nslookup hostname
|
||||
```
|
||||
|
||||
### Database Connection Pool Exhaustion
|
||||
```bash
|
||||
# Check active connections (PostgreSQL)
|
||||
psql -c "SELECT count(*) FROM pg_stat_activity WHERE state = 'active';"
|
||||
psql -c "SELECT max_conn FROM pg_settings WHERE name = 'max_connections';"
|
||||
|
||||
# Check for long-running queries
|
||||
psql -c "SELECT pid, now() - pg_stat_activity.query_start AS duration, query
|
||||
FROM pg_stat_activity WHERE state != 'idle' ORDER BY duration DESC LIMIT 10;"
|
||||
```
|
||||
|
||||
### Certificate Expiry Monitoring
|
||||
```bash
|
||||
# Check cert expiry for a list of domains
|
||||
for domain in api.example.com app.example.com; do
|
||||
expiry=$(echo | openssl s_client -servername $domain -connect $domain:443 2>/dev/null | \
|
||||
openssl x509 -noout -enddate 2>/dev/null | cut -d= -f2)
|
||||
echo "$domain: $expiry"
|
||||
done
|
||||
```
|
||||
Reference in new issue
Block a user