feat(hands): complete i18n fixes, SKILL.md enhancements, and README overhaul
- Fix French accent characters (é/è/ê/ç/â/ô) across all 14 HAND.toml files - Fix German special characters (ä/ö/ü/ß) across all 14 HAND.toml files - Add category translations to all 6 i18n language blocks in all 14 hands - Enhance SKILL.md content for 9 hands with practical examples and workflows - Trim bloated SKILL.md files (apitester 1400→892, devops 1301→870) - Rewrite root README.md with accurate stats, complete hand/integration tables - Update hands/README.md with full 14-hand listing and i18n documentation
This commit is contained in:
1 parent
315f955ce2
commit
33d279889c
27 files changed
+10001
-78
No files matched your search
@@ -490,6 +490,265 @@ token_consumption = "high"
|
||||
default_active = false
|
||||
activation_warning = "DevOps hand runs continuously and monitors infrastructure, consuming tokens."
|
||||
|
||||
# ─── Internationalization (optional) ─────────────────────────────────────────
|
||||
# All i18n sections are optional. Without them, the English values above are used.
|
||||
# To localize, add [i18n.LANG] sections (e.g. zh, ja, ko, es, fr, de).
|
||||
# Settings translations are also optional — omit to keep English labels.
|
||||
|
||||
# ─── Chinese (简体中文) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.zh]
|
||||
name = "DevOps Hand"
|
||||
description = "自主 DevOps 工程师——CI/CD 管理、基础设施监控、部署自动化和事件响应"
|
||||
category = "开发"
|
||||
|
||||
[i18n.zh.settings.infrastructure]
|
||||
label = "基础设施类型"
|
||||
description = "主要基础设施平台"
|
||||
|
||||
[i18n.zh.settings.ci_platform]
|
||||
label = "CI/CD 平台"
|
||||
description = "主要 CI/CD 平台"
|
||||
|
||||
[i18n.zh.settings.monitoring_focus]
|
||||
label = "监控重点"
|
||||
description = "主要监控和告警的关注方向"
|
||||
|
||||
[i18n.zh.settings.auto_monitor]
|
||||
label = "自动监控"
|
||||
description = "自动监控基础设施并在出现问题时告警"
|
||||
|
||||
[i18n.zh.settings.check_interval]
|
||||
label = "健康检查间隔"
|
||||
description = "自动健康检查的执行频率"
|
||||
|
||||
[i18n.zh.settings.service_urls]
|
||||
label = "服务 URL"
|
||||
description = "要监控的 URL 列表,以逗号分隔(例如 https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.zh.settings.alert_on_failure]
|
||||
label = "故障告警"
|
||||
description = "健康检查失败时发布事件通知"
|
||||
|
||||
[i18n.zh.settings.rollback_strategy]
|
||||
label = "回滚策略"
|
||||
description = "部署失败时的默认回滚方式"
|
||||
|
||||
[i18n.zh.settings.approval_mode]
|
||||
label = "审批模式"
|
||||
description = "将部署和基础设施操作加入队列供审核,而非直接执行"
|
||||
|
||||
# ─── Japanese (日本語) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.ja]
|
||||
name = "DevOps Hand"
|
||||
description = "自律型DevOpsエンジニア——CI/CD管理、インフラ監視、デプロイ自動化、インシデント対応"
|
||||
category = "開発"
|
||||
|
||||
[i18n.ja.settings.infrastructure]
|
||||
label = "インフラタイプ"
|
||||
description = "主要なインフラプラットフォーム"
|
||||
|
||||
[i18n.ja.settings.ci_platform]
|
||||
label = "CI/CDプラットフォーム"
|
||||
description = "主要なCI/CDプラットフォーム"
|
||||
|
||||
[i18n.ja.settings.monitoring_focus]
|
||||
label = "監視の重点"
|
||||
description = "監視とアラートの主な対象分野"
|
||||
|
||||
[i18n.ja.settings.auto_monitor]
|
||||
label = "自動監視"
|
||||
description = "インフラを自動監視し、問題発生時にアラートを出す"
|
||||
|
||||
[i18n.ja.settings.check_interval]
|
||||
label = "ヘルスチェック間隔"
|
||||
description = "自動ヘルスチェックの実行間隔"
|
||||
|
||||
[i18n.ja.settings.service_urls]
|
||||
label = "サービスURL"
|
||||
description = "監視対象のURL一覧(カンマ区切り、例: https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.ja.settings.alert_on_failure]
|
||||
label = "障害アラート"
|
||||
description = "ヘルスチェック失敗時にイベント通知を発行する"
|
||||
|
||||
[i18n.ja.settings.rollback_strategy]
|
||||
label = "ロールバック戦略"
|
||||
description = "デプロイ失敗時のデフォルトのロールバック方法"
|
||||
|
||||
[i18n.ja.settings.approval_mode]
|
||||
label = "承認モード"
|
||||
description = "デプロイやインフラ操作を直接実行せず、レビュー用キューに追加する"
|
||||
|
||||
# ─── Spanish (Español) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.es]
|
||||
name = "Hand de DevOps"
|
||||
description = "Ingeniero DevOps autónomo — gestión de CI/CD, monitoreo de infraestructura, automatización de despliegues y respuesta a incidentes"
|
||||
category = "Desarrollo"
|
||||
|
||||
[i18n.es.settings.infrastructure]
|
||||
label = "Tipo de infraestructura"
|
||||
description = "Plataforma de infraestructura principal"
|
||||
|
||||
[i18n.es.settings.ci_platform]
|
||||
label = "Plataforma CI/CD"
|
||||
description = "Plataforma principal de CI/CD"
|
||||
|
||||
[i18n.es.settings.monitoring_focus]
|
||||
label = "Enfoque de monitoreo"
|
||||
description = "Área principal de monitoreo y alertas"
|
||||
|
||||
[i18n.es.settings.auto_monitor]
|
||||
label = "Monitoreo automático"
|
||||
description = "Monitorear automáticamente la infraestructura y alertar ante problemas"
|
||||
|
||||
[i18n.es.settings.check_interval]
|
||||
label = "Intervalo de comprobación de salud"
|
||||
description = "Con qué frecuencia ejecutar las comprobaciones de salud automatizadas"
|
||||
|
||||
[i18n.es.settings.service_urls]
|
||||
label = "URLs de servicios"
|
||||
description = "Lista de URLs a monitorear separadas por comas (ej. https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.es.settings.alert_on_failure]
|
||||
label = "Alertar ante fallos"
|
||||
description = "Publicar eventos cuando las comprobaciones de salud fallen"
|
||||
|
||||
[i18n.es.settings.rollback_strategy]
|
||||
label = "Estrategia de reversión"
|
||||
description = "Enfoque de reversión predeterminado para despliegues fallidos"
|
||||
|
||||
[i18n.es.settings.approval_mode]
|
||||
label = "Modo de aprobación"
|
||||
description = "Poner acciones de despliegue e infraestructura en cola para revisión en lugar de ejecutarlas directamente"
|
||||
|
||||
# ─── French (Français) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.fr]
|
||||
name = "Hand DevOps"
|
||||
description = "Ingénieur DevOps autonome — gestion CI/CD, surveillance d'infrastructure, automatisation des déploiements et réponse aux incidents"
|
||||
category = "Développement"
|
||||
|
||||
[i18n.fr.settings.infrastructure]
|
||||
label = "Type d'infrastructure"
|
||||
description = "Plateforme d'infrastructure principale"
|
||||
|
||||
[i18n.fr.settings.ci_platform]
|
||||
label = "Plateforme CI/CD"
|
||||
description = "Plateforme CI/CD principale"
|
||||
|
||||
[i18n.fr.settings.monitoring_focus]
|
||||
label = "Axe de surveillance"
|
||||
description = "Domaine principal de surveillance et d'alerte"
|
||||
|
||||
[i18n.fr.settings.auto_monitor]
|
||||
label = "Surveillance automatique"
|
||||
description = "Surveiller automatiquement l'infrastructure et alerter en cas de problèmes"
|
||||
|
||||
[i18n.fr.settings.check_interval]
|
||||
label = "Intervalle de vérification de santé"
|
||||
description = "Fréquence d'exécution des vérifications de santé automatisées"
|
||||
|
||||
[i18n.fr.settings.service_urls]
|
||||
label = "URLs des services"
|
||||
description = "Liste d'URLs à surveiller séparées par des virgules (ex. https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.fr.settings.alert_on_failure]
|
||||
label = "Alerte en cas d'échec"
|
||||
description = "Publier des événements lorsque les vérifications de santé échouent"
|
||||
|
||||
[i18n.fr.settings.rollback_strategy]
|
||||
label = "Stratégie de retour en arrière"
|
||||
description = "Approche de retour en arrière par défaut pour les déploiements échoués"
|
||||
|
||||
[i18n.fr.settings.approval_mode]
|
||||
label = "Mode d'approbation"
|
||||
description = "Mettre les actions de déploiement et d'infrastructure en file d'attente pour révision au lieu de les exécuter directement"
|
||||
|
||||
# ─── German (Deutsch) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.de]
|
||||
name = "DevOps-Hand"
|
||||
description = "Autonomer DevOps-Ingenieur — CI/CD-Verwaltung, Infrastrukturüberwachung, Deployment-Automatisierung und Incident-Response"
|
||||
category = "Entwicklung"
|
||||
|
||||
[i18n.de.settings.infrastructure]
|
||||
label = "Infrastrukturtyp"
|
||||
description = "Primäre Infrastrukturplattform"
|
||||
|
||||
[i18n.de.settings.ci_platform]
|
||||
label = "CI/CD-Plattform"
|
||||
description = "Primäre CI/CD-Plattform"
|
||||
|
||||
[i18n.de.settings.monitoring_focus]
|
||||
label = "Überwachungsschwerpunkt"
|
||||
description = "Hauptbereich für Überwachung und Alarme"
|
||||
|
||||
[i18n.de.settings.auto_monitor]
|
||||
label = "Automatische Überwachung"
|
||||
description = "Infrastruktur automatisch überwachen und bei Problemen alarmieren"
|
||||
|
||||
[i18n.de.settings.check_interval]
|
||||
label = "Gesundheitscheck-Intervall"
|
||||
description = "Ausführungshäufigkeit der automatisierten Gesundheitschecks"
|
||||
|
||||
[i18n.de.settings.service_urls]
|
||||
label = "Service-URLs"
|
||||
description = "Kommagetrennte Liste der zu überwachenden URLs (z.B. https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.de.settings.alert_on_failure]
|
||||
label = "Warnung bei Ausfall"
|
||||
description = "Ereignisse veröffentlichen, wenn Gesundheitschecks fehlschlagen"
|
||||
|
||||
[i18n.de.settings.rollback_strategy]
|
||||
label = "Rollback-Strategie"
|
||||
description = "Standard-Rollback-Ansatz für fehlgeschlagene Deployments"
|
||||
|
||||
[i18n.de.settings.approval_mode]
|
||||
label = "Genehmigungsmodus"
|
||||
description = "Deployment- und Infrastrukturaktionen zur Überprüfung in die Warteschlange stellen, anstatt sie direkt auszuführen"
|
||||
|
||||
# ─── Korean (한국어) ────────────────────────────────────────────────────
|
||||
|
||||
[i18n.ko]
|
||||
name = "DevOps Hand"
|
||||
description = "자율 DevOps 엔지니어 — CI/CD 관리, 인프라 모니터링, 배포 자동화 및 인시던트 대응"
|
||||
category = "개발"
|
||||
|
||||
[i18n.ko.settings.infrastructure]
|
||||
label = "인프라 유형"
|
||||
description = "주요 인프라 플랫폼"
|
||||
|
||||
[i18n.ko.settings.ci_platform]
|
||||
label = "CI/CD 플랫폼"
|
||||
description = "주요 CI/CD 플랫폼"
|
||||
|
||||
[i18n.ko.settings.monitoring_focus]
|
||||
label = "모니터링 중점"
|
||||
description = "주요 모니터링 및 알림 방향"
|
||||
|
||||
[i18n.ko.settings.auto_monitor]
|
||||
label = "자동 모니터링"
|
||||
description = "인프라를 자동으로 모니터링하고 문제 발생 시 알림"
|
||||
|
||||
[i18n.ko.settings.check_interval]
|
||||
label = "상태 점검 간격"
|
||||
description = "자동 상태 점검 실행 주기"
|
||||
|
||||
[i18n.ko.settings.service_urls]
|
||||
label = "서비스 URL"
|
||||
description = "모니터링할 URL 목록 (쉼표로 구분, 예: https://api.example.com/health,https://app.example.com)"
|
||||
|
||||
[i18n.ko.settings.alert_on_failure]
|
||||
label = "장애 알림"
|
||||
description = "상태 점검 실패 시 이벤트 알림 발행"
|
||||
|
||||
[i18n.ko.settings.rollback_strategy]
|
||||
label = "롤백 전략"
|
||||
description = "배포 실패 시 기본 롤백 방식"
|
||||
|
||||
[i18n.ko.settings.approval_mode]
|
||||
label = "승인 모드"
|
||||
description = "배포 및 인프라 작업을 직접 실행하지 않고 대기열에 추가하여 검토"
|
||||
@@ -330,3 +330,541 @@ for domain in api.example.com app.example.com; do
|
||||
echo "$domain: $expiry"
|
||||
done
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Worked Examples
|
||||
|
||||
### Example 1: Zero-Downtime Deployment Pipeline
|
||||
|
||||
Full lifecycle from code merge to production traffic switch with rollback safety.
|
||||
|
||||
**Phases**: CI build and push image -> deploy to Green -> health-gate -> traffic switch -> monitor -> done (or rollback).
|
||||
|
||||
**Traffic switch script** (the critical step):
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
# Record current slot for rollback, then switch
|
||||
CURRENT=$(kubectl get svc myapp-active -n production -o jsonpath='{.spec.selector.slot}')
|
||||
echo "$CURRENT" > /tmp/rollback-slot
|
||||
kubectl patch svc myapp-active -n production -p '{"spec":{"selector":{"slot":"green"}}}'
|
||||
# Verify
|
||||
sleep 5 && curl -sf https://api.example.com/api/health | jq '.version'
|
||||
```
|
||||
|
||||
**Rollback**: read `/tmp/rollback-slot`, patch the service selector back, verify health.
|
||||
|
||||
**Decision flowchart**:
|
||||
```
|
||||
Code merged to main
|
||||
|
|
||||
v
|
||||
CI build + test ----[FAIL]----> Block merge, notify author
|
||||
|
|
||||
[PASS]
|
||||
v
|
||||
Deploy to Green
|
||||
|
|
||||
v
|
||||
Health checks ------[FAIL]----> Alert on-call, keep Blue active
|
||||
|
|
||||
[PASS]
|
||||
v
|
||||
Switch traffic to Green
|
||||
|
|
||||
v
|
||||
Monitor 15 min -----[ERROR SPIKE]----> Rollback to Blue, open incident
|
||||
|
|
||||
[STABLE]
|
||||
v
|
||||
Mark Green as new Blue, done
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Example 2: Production Incident Response
|
||||
|
||||
Walkthrough of a real-world SEV1 incident: API latency spike caused by a database connection pool exhaustion.
|
||||
|
||||
**Timeline**
|
||||
|
||||
| Time (UTC) | Event | Actor |
|
||||
|------------|-------|-------|
|
||||
| 14:02 | PagerDuty alert: P95 latency > 2s on `/api/orders` | Monitoring |
|
||||
| 14:04 | On-call acknowledges, opens incident channel `#inc-2025-0312` | On-call engineer |
|
||||
| 14:06 | Check dashboard: request queue depth spiking, error rate at 12% | On-call engineer |
|
||||
| 14:10 | Identify DB connection pool at 100% utilization | On-call engineer |
|
||||
| 14:12 | Find long-running query from analytics job (started 13:55) | On-call engineer |
|
||||
| 14:14 | Kill the runaway query, pool starts draining | On-call engineer |
|
||||
| 14:18 | Latency returns to normal, error rate drops to 0.2% | Monitoring |
|
||||
| 14:20 | Incident mitigated, continue monitoring | On-call engineer |
|
||||
| 14:45 | Root cause confirmed: analytics cron job without query timeout | Investigation |
|
||||
| 15:00 | Incident resolved, post-mortem scheduled | Incident commander |
|
||||
|
||||
**Detection -- Alerting rules that fired**
|
||||
|
||||
```yaml
|
||||
# Prometheus alerting rule
|
||||
groups:
|
||||
- name: api-latency
|
||||
rules:
|
||||
- alert: HighAPILatency
|
||||
expr: histogram_quantile(0.95, rate(http_request_duration_seconds_bucket{job="api"}[5m])) > 2
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "P95 latency above 2s for 2+ minutes"
|
||||
runbook: "https://wiki.internal/runbooks/high-latency"
|
||||
```
|
||||
|
||||
**Triage -- Quick diagnosis commands**
|
||||
|
||||
```bash
|
||||
# 1. Check if it's a specific endpoint or global
|
||||
curl -s "http://prometheus:9090/api/v1/query?query=topk(5,rate(http_request_duration_seconds_sum[5m])/rate(http_request_duration_seconds_count[5m]))" | jq '.data.result[] | {endpoint: .metric.handler, avg_latency: .value[1]}'
|
||||
|
||||
# 2. Check database connection pool
|
||||
psql -c "SELECT count(*) as total, state FROM pg_stat_activity GROUP BY state;"
|
||||
|
||||
# 3. Find the blocking query
|
||||
psql -c "SELECT pid, now() - query_start AS duration, query
|
||||
FROM pg_stat_activity
|
||||
WHERE state = 'active' AND now() - query_start > interval '1 minute'
|
||||
ORDER BY duration DESC LIMIT 5;"
|
||||
```
|
||||
|
||||
**Mitigation -- Kill the offending query**
|
||||
|
||||
```bash
|
||||
# Kill the long-running query by PID
|
||||
psql -c "SELECT pg_terminate_backend(12345);"
|
||||
|
||||
# Verify pool is recovering
|
||||
watch -n 2 'psql -t -c "SELECT count(*) FROM pg_stat_activity WHERE state = '\''active'\'';"'
|
||||
```
|
||||
|
||||
**Prevention -- Fix applied after incident**
|
||||
|
||||
```sql
|
||||
-- Set statement timeout for analytics role
|
||||
ALTER ROLE analytics_readonly SET statement_timeout = '300s';
|
||||
```
|
||||
|
||||
```yaml
|
||||
# Add connection pool monitoring alert
|
||||
- alert: DBConnectionPoolNearCapacity
|
||||
expr: pg_stat_activity_count / pg_settings_max_connections > 0.8
|
||||
for: 1m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "DB connection pool above 80% capacity"
|
||||
```
|
||||
|
||||
**Post-mortem action items**:
|
||||
- [ ] Add `statement_timeout` to all non-interactive database roles
|
||||
- [ ] Add connection pool utilization alerts (threshold: 80%)
|
||||
- [ ] Move analytics queries to read replica
|
||||
- [ ] Add circuit breaker to API when pool utilization exceeds 90%
|
||||
|
||||
---
|
||||
|
||||
### Example 3: Infrastructure Scaling Event
|
||||
|
||||
Scaling a Kubernetes deployment in response to sustained load increase.
|
||||
|
||||
**Phase 1 -- Alert triggers**
|
||||
|
||||
```
|
||||
Alert: HighCPUUtilization
|
||||
Condition: avg(cpu_usage) > 80% for 10 minutes
|
||||
Current: 87% across 3 pods
|
||||
Namespace: production
|
||||
Deployment: order-service
|
||||
```
|
||||
|
||||
**Phase 2 -- Capacity analysis**
|
||||
|
||||
```bash
|
||||
kubectl top pods -l app=order-service -n production # Per-pod CPU/memory
|
||||
kubectl describe nodes | grep -A 5 "Allocated resources" # Node headroom
|
||||
kubectl get hpa order-service -n production # Current HPA state
|
||||
# Result: 87% CPU across 3 pods, 842 req/s (2x baseline)
|
||||
```
|
||||
|
||||
**Phase 3 -- Scaling decision matrix**
|
||||
|
||||
| Metric | Current | Target | Action |
|
||||
|--------|---------|--------|--------|
|
||||
| CPU usage | 87% | < 70% | Scale out |
|
||||
| Request rate | 842/s | - | 2x normal, sustained |
|
||||
| Memory | 258Mi avg | 512Mi limit | Headroom OK |
|
||||
| Pod count | 3 | 6 (estimated) | Double replicas |
|
||||
| Node capacity | 72% | < 85% | Sufficient for 6 pods |
|
||||
|
||||
**Phase 4 -- Implement scaling**
|
||||
|
||||
```bash
|
||||
# Option A: Manual scale (immediate)
|
||||
kubectl scale deployment order-service -n production --replicas=6
|
||||
|
||||
# Option B: Adjust HPA for sustained load (preferred)
|
||||
kubectl patch hpa order-service -n production \
|
||||
-p '{"spec":{"minReplicas":5,"maxReplicas":15}}'
|
||||
|
||||
# Monitor rollout
|
||||
kubectl rollout status deployment/order-service -n production
|
||||
|
||||
# Watch pods come up
|
||||
kubectl get pods -l app=order-service -n production -w
|
||||
```
|
||||
|
||||
**Phase 5 -- Verify scaling**
|
||||
|
||||
Confirm via `kubectl top pods` (CPU should drop to ~45% per pod), check P95 latency is back below SLO, and verify error rate < 1%.
|
||||
|
||||
**Post-scaling actions**:
|
||||
- [ ] Investigate root cause of traffic increase (marketing event? bot traffic? organic growth?)
|
||||
- [ ] Update capacity planning spreadsheet
|
||||
- [ ] If sustained, adjust resource requests/limits and HPA baselines
|
||||
- [ ] Set calendar reminder to review and potentially scale down in 48h
|
||||
|
||||
---
|
||||
|
||||
## Observability Deep Dive
|
||||
|
||||
### Structured Logging
|
||||
|
||||
Use consistent JSON log format across all services for machine-parseable aggregation.
|
||||
|
||||
**Log format standard**: JSON with required fields: `timestamp`, `level`, `service`, `trace_id`, `span_id`, `request_id`, `message`. Add `error` and `context` (structured key-value) as needed.
|
||||
|
||||
**Log levels -- when to use each**:
|
||||
|
||||
| Level | Purpose | Example | Persisted |
|
||||
|-------|---------|---------|-----------|
|
||||
| `error` | Requires human attention | Payment processing failed | 90 days |
|
||||
| `warn` | Degraded but recoverable | Retry succeeded on 2nd attempt | 30 days |
|
||||
| `info` | Business-significant events | Order placed, user logged in | 14 days |
|
||||
| `debug` | Developer troubleshooting | Cache hit/miss, query timing | 3 days |
|
||||
| `trace` | Fine-grained flow tracking | Function entry/exit, variable state | 1 day (sampled) |
|
||||
|
||||
**Correlation IDs**: generate `X-Request-ID` at API gateway, propagate through all downstream calls. Query across services by filtering on `request_id.keyword` in Elasticsearch/OpenSearch.
|
||||
|
||||
### Distributed Tracing
|
||||
|
||||
**Core concepts**: A Trace is the end-to-end request path. Each service call is a Span with timing. Spans nest to show the call tree (e.g., API Gateway -> Order Service -> DB Query + Payment Service -> Stripe API).
|
||||
|
||||
**OpenTelemetry propagation**: `traceparent: 00-<trace-id>-<span-id>-<flags>`, `tracestate: vendor=value`.
|
||||
|
||||
**Useful trace queries (Jaeger/Tempo)**:
|
||||
```bash
|
||||
curl -s "http://jaeger:16686/api/traces?service=order-service&minDuration=1s&limit=20" # Slow traces
|
||||
curl -s "http://jaeger:16686/api/traces?service=order-service&tags=error%3Dtrue&limit=20" # Error traces
|
||||
```
|
||||
|
||||
### Alerting Best Practices
|
||||
|
||||
**Avoid alert fatigue -- rules of thumb**:
|
||||
- Every alert must have a runbook link
|
||||
- Every alert must be actionable (if no one needs to act, it is a log, not an alert)
|
||||
- Group related alerts to avoid notification storms
|
||||
- Use inhibition rules: if the cluster is down, suppress per-pod alerts
|
||||
|
||||
**SLO-based alerting (burn rate)**:
|
||||
```yaml
|
||||
# SLO: 99.9% availability = 43.2 min/month error budget
|
||||
# Fast burn (exhausts budget in 2h): error_ratio > 14.4 * 0.001 for 2m -> critical
|
||||
# Slow burn (exhausts budget in 3d): error_ratio > 3 * 0.001 for 15m -> warning
|
||||
groups:
|
||||
- name: slo-burn-rate
|
||||
rules:
|
||||
- alert: SLOBurnRateCritical
|
||||
expr: sum(rate(http_requests_total{code=~"5.."}[5m])) / sum(rate(http_requests_total[5m])) > (14.4 * 0.001)
|
||||
for: 2m
|
||||
labels: { severity: critical }
|
||||
- alert: SLOBurnRateWarning
|
||||
expr: sum(rate(http_requests_total{code=~"5.."}[1h])) / sum(rate(http_requests_total[1h])) > (3 * 0.001)
|
||||
for: 15m
|
||||
labels: { severity: warning }
|
||||
```
|
||||
|
||||
**Runbook template**: Each alert runbook should cover: what the alert means (one sentence), impact scope, diagnosis steps (dashboard + commands), mitigation (quick fix vs proper fix), and escalation path (who to contact after 15 min).
|
||||
|
||||
### Metrics Collection Patterns
|
||||
|
||||
**RED Method (request-scoped services)**:
|
||||
|
||||
| Metric | What | PromQL Example |
|
||||
|--------|------|----------------|
|
||||
| **R**ate | Requests per second | `sum(rate(http_requests_total[5m]))` |
|
||||
| **E**rrors | Failed requests per second | `sum(rate(http_requests_total{code=~"5.."}[5m]))` |
|
||||
| **D**uration | Latency distribution | `histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket[5m])) by (le))` |
|
||||
|
||||
**USE Method (infrastructure resources)**:
|
||||
|
||||
| Metric | What | Example Check |
|
||||
|--------|------|---------------|
|
||||
| **U**tilization | % time resource is busy | `avg(rate(node_cpu_seconds_total{mode!="idle"}[5m]))` |
|
||||
| **S**aturation | Queue depth / backlog | `node_load1 / count(node_cpu_seconds_total{mode="idle"})` |
|
||||
| **E**rrors | Error event count | `rate(node_disk_io_time_weighted_seconds_total[5m])` |
|
||||
|
||||
**When to use which**:
|
||||
- RED for services that handle requests (APIs, web servers, message consumers)
|
||||
- USE for infrastructure (CPU, memory, disk, network interfaces, queues)
|
||||
- Combine both for a complete picture
|
||||
|
||||
---
|
||||
|
||||
## Security Operations
|
||||
|
||||
### Secret Management
|
||||
|
||||
**Principles**:
|
||||
- Never store secrets in source code, environment variables (in Dockerfiles), or container images
|
||||
- Use a secrets manager (Vault, AWS Secrets Manager, K8s Secrets with encryption at rest)
|
||||
- Rotate secrets on a schedule and immediately after any suspected compromise
|
||||
- Audit all secret access
|
||||
|
||||
**Vault pattern -- inject secrets at runtime**:
|
||||
```bash
|
||||
# Store a secret
|
||||
vault kv put secret/myapp/db \
|
||||
username="app_user" \
|
||||
password="$(openssl rand -base64 32)"
|
||||
|
||||
# Read a secret (application startup)
|
||||
vault kv get -format=json secret/myapp/db | jq -r '.data.data.password'
|
||||
|
||||
# Enable audit logging
|
||||
vault audit enable file file_path=/var/log/vault-audit.log
|
||||
```
|
||||
|
||||
**Kubernetes secrets -- from Vault using sidecar injector**:
|
||||
|
||||
Annotate the pod template with `vault.hashicorp.com/agent-inject: "true"`, specify the role and secret path. The Vault agent sidecar renders secrets to `/vault/secrets/` and the app sources them at startup. Key annotations: `agent-inject-secret-<name>` for the path, `agent-inject-template-<name>` for the rendering template.
|
||||
|
||||
**Secret rotation checklist**:
|
||||
- [ ] Generate new secret value
|
||||
- [ ] Update secret in secrets manager
|
||||
- [ ] Restart/reload affected services (rolling, not all-at-once)
|
||||
- [ ] Verify services authenticate with new secret
|
||||
- [ ] Revoke the old secret value
|
||||
- [ ] Confirm no services are still using the old secret
|
||||
|
||||
### Container Security Scanning
|
||||
|
||||
```bash
|
||||
# Scan image for vulnerabilities (Trivy) -- fail CI on critical
|
||||
trivy image --exit-code 1 --severity CRITICAL registry.example.com/myapp:$CI_COMMIT_SHA
|
||||
|
||||
# Scan K8s cluster for misconfigurations
|
||||
trivy k8s --report summary cluster
|
||||
```
|
||||
|
||||
**Dockerfile security essentials**: use pinned base image tags (not `:latest`), run as non-root (`USER app`), copy only needed files, never bake secrets into image layers.
|
||||
|
||||
### Network Security Policies
|
||||
|
||||
**Kubernetes NetworkPolicy -- default deny with explicit allow**:
|
||||
```yaml
|
||||
# Default deny all ingress, then allow specific paths
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: allow-gateway-to-orders
|
||||
namespace: production
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels: { app: order-service }
|
||||
ingress:
|
||||
- from:
|
||||
- podSelector:
|
||||
matchLabels: { app: api-gateway }
|
||||
ports:
|
||||
- { protocol: TCP, port: 8080 }
|
||||
```
|
||||
|
||||
Apply a `default-deny-ingress` policy (empty `podSelector`, `policyTypes: [Ingress]`) per namespace first, then layer allow rules on top.
|
||||
|
||||
### Compliance as Code
|
||||
|
||||
**Policy enforcement with OPA/Gatekeeper**: use `K8sRequiredResources` constraints to enforce `limits.cpu`, `limits.memory`, `requests.cpu`, `requests.memory` on all pods in production namespaces.
|
||||
|
||||
**Quick compliance audit commands**:
|
||||
```bash
|
||||
# Find pods without resource limits
|
||||
kubectl get pods -A -o json | jq -r '.items[] | select(.spec.containers[].resources.limits == null) | .metadata.namespace + "/" + .metadata.name'
|
||||
# Find containers running as root
|
||||
kubectl get pods -A -o json | jq -r '.items[] | select(.spec.containers[].securityContext.runAsNonRoot != true) | .metadata.namespace + "/" + .metadata.name'
|
||||
# Find ingress without TLS
|
||||
kubectl get ingress -A -o json | jq -r '.items[] | select(.spec.tls == null) | .metadata.namespace + "/" + .metadata.name'
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Automation Patterns
|
||||
|
||||
### Auto-Remediation
|
||||
|
||||
**Restart on OOM (Kubernetes)**:
|
||||
```yaml
|
||||
# Built-in: set resource limits and let K8s handle OOM restarts
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
spec:
|
||||
template:
|
||||
spec:
|
||||
containers:
|
||||
- name: myapp
|
||||
resources:
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
requests:
|
||||
memory: "256Mi"
|
||||
# Liveness probe: restart if unhealthy
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: 8080
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 10
|
||||
failureThreshold: 3
|
||||
```
|
||||
|
||||
**Scale on load (HPA with custom metrics)**:
|
||||
```yaml
|
||||
apiVersion: autoscaling/v2
|
||||
kind: HorizontalPodAutoscaler
|
||||
metadata:
|
||||
name: order-service
|
||||
spec:
|
||||
scaleTargetRef: { apiVersion: apps/v1, kind: Deployment, name: order-service }
|
||||
minReplicas: 3
|
||||
maxReplicas: 20
|
||||
behavior:
|
||||
scaleUp: { stabilizationWindowSeconds: 60, policies: [{ type: Percent, value: 50, periodSeconds: 60 }] }
|
||||
scaleDown: { stabilizationWindowSeconds: 300, policies: [{ type: Percent, value: 25, periodSeconds: 120 }] }
|
||||
metrics:
|
||||
- type: Resource
|
||||
resource: { name: cpu, target: { type: Utilization, averageUtilization: 70 } }
|
||||
- type: Pods
|
||||
pods: { metric: { name: http_requests_per_second }, target: { type: AverageValue, averageValue: "1000" } }
|
||||
```
|
||||
|
||||
**Rotate secrets on expiry (CronJob)**:
|
||||
|
||||
Use a K8s CronJob (e.g., monthly `"0 2 1 * *"`) with a `secret-rotator` service account that: generates new password -> updates Vault -> alters DB role password -> triggers rolling restart via `kubectl rollout restart`.
|
||||
|
||||
### GitOps Workflow
|
||||
|
||||
**Repository as source of truth**:
|
||||
```
|
||||
infrastructure-repo/
|
||||
|-- apps/
|
||||
| |-- order-service/
|
||||
| | |-- deployment.yaml
|
||||
| | |-- service.yaml
|
||||
| | |-- hpa.yaml
|
||||
| | `-- kustomization.yaml
|
||||
| `-- payment-service/
|
||||
| |-- deployment.yaml
|
||||
| `-- kustomization.yaml
|
||||
|-- base/
|
||||
| |-- namespace.yaml
|
||||
| |-- network-policies.yaml
|
||||
| `-- resource-quotas.yaml
|
||||
`-- overlays/
|
||||
|-- staging/
|
||||
| `-- kustomization.yaml
|
||||
`-- production/
|
||||
`-- kustomization.yaml
|
||||
```
|
||||
|
||||
**Reconciliation loop (ArgoCD application)**:
|
||||
```yaml
|
||||
apiVersion: argoproj.io/v1alpha1
|
||||
kind: Application
|
||||
metadata:
|
||||
name: order-service
|
||||
namespace: argocd
|
||||
spec:
|
||||
project: default
|
||||
source:
|
||||
repoURL: https://github.com/org/infrastructure-repo.git
|
||||
targetRevision: main
|
||||
path: apps/order-service
|
||||
destination: { server: "https://kubernetes.default.svc", namespace: production }
|
||||
syncPolicy:
|
||||
automated: { prune: true, selfHeal: true }
|
||||
syncOptions: [CreateNamespace=true]
|
||||
retry: { limit: 3, backoff: { duration: 5s, factor: 2, maxDuration: 3m } }
|
||||
```
|
||||
|
||||
**GitOps deployment flow**:
|
||||
```
|
||||
Developer pushes image tag update to infrastructure-repo
|
||||
|
|
||||
v
|
||||
ArgoCD detects drift between git state and cluster state
|
||||
|
|
||||
v
|
||||
ArgoCD syncs: applies manifests from git to cluster
|
||||
|
|
||||
v
|
||||
Kubernetes rolls out new pods
|
||||
|
|
||||
v
|
||||
ArgoCD verifies health (readiness probes pass)
|
||||
|
|
||||
[HEALTHY] --> Done
|
||||
[DEGRADED] --> ArgoCD marks sync as failed, alerts on-call
|
||||
```
|
||||
|
||||
### Database Backup and Restore
|
||||
|
||||
**Automated backup (PostgreSQL)** -- run via cron `0 */6 * * *`:
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
TIMESTAMP=$(date +%Y%m%d_%H%M%S)
|
||||
DB_NAME="production"
|
||||
BACKUP_DIR="/backups/postgres"
|
||||
|
||||
pg_dump -Fc -Z 9 "$DB_NAME" > "${BACKUP_DIR}/${DB_NAME}_${TIMESTAMP}.dump"
|
||||
aws s3 cp "${BACKUP_DIR}/${DB_NAME}_${TIMESTAMP}.dump" \
|
||||
"s3://backups-bucket/postgres/" --storage-class STANDARD_IA
|
||||
find "$BACKUP_DIR" -name "*.dump" -mtime +30 -delete
|
||||
```
|
||||
|
||||
**Restore procedure**:
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
BACKUP_FILE=$1 # e.g., "production_20250315_060000.dump"
|
||||
RESTORE_DB="production_restore"
|
||||
|
||||
aws s3 cp "s3://backups-bucket/postgres/$BACKUP_FILE" /tmp/restore.dump
|
||||
psql -c "DROP DATABASE IF EXISTS $RESTORE_DB;" && psql -c "CREATE DATABASE $RESTORE_DB;"
|
||||
pg_restore -d "$RESTORE_DB" -j 4 --no-owner /tmp/restore.dump
|
||||
# Verify: check row counts on key tables, then clean up
|
||||
rm /tmp/restore.dump
|
||||
```
|
||||
|
||||
### Disaster Recovery Runbook Template
|
||||
|
||||
**Recovery Objectives**: Define RTO (e.g., 1 hour) and RPO (e.g., 6 hours) per service.
|
||||
|
||||
**Prerequisites**: backup storage access, Terraform state access, DNS management access, stakeholder comms channel.
|
||||
|
||||
| Scenario | Key Steps |
|
||||
|----------|-----------|
|
||||
| **Single service failure** | Check pod status -> restart deployment -> if fails, `kubectl rollout undo` -> verify health |
|
||||
| **Database failure** | `pg_isready` -> promote replica (or restore from backup) -> update connection strings -> verify data integrity |
|
||||
| **Full region outage** | Confirm via provider status page -> notify stakeholders -> switch DNS to DR region -> verify traffic -> failback when primary recovers |
|
||||
|
||||
**Communication template**: Subject `[INCIDENT] Service -- Status`. Body: what happened, impact, current status, ETA, next update time.
|
||||
|
||||
**Post-recovery checklist**: health checks passing, data integrity verified, monitoring restored, backups resumed, incident report filed, post-mortem scheduled within 48h.
|
||||
Reference in new issue
Block a user