feat(hands): complete i18n fixes, SKILL.md enhancements, and README overhaul

- Fix French accent characters (é/è/ê/ç/â/ô) across all 14 HAND.toml files
- Fix German special characters (ä/ö/ü/ß) across all 14 HAND.toml files
- Add category translations to all 6 i18n language blocks in all 14 hands
- Enhance SKILL.md content for 9 hands with practical examples and workflows
- Trim bloated SKILL.md files (apitester 1400→892, devops 1301→870)
- Rewrite root README.md with accurate stats, complete hand/integration tables
- Update hands/README.md with full 14-hand listing and i18n documentation
This commit is contained in:
Evan Hu committed 2026-03-23 00:18:18 +09:00
1 parent 315f955ce2
commit 33d279889c
27 files changed
+10001 -78

No files matched your search

+259
View File
@@ -490,6 +490,265 @@ token_consumption = "high"
default_active = false
activation_warning = "DevOps hand runs continuously and monitors infrastructure, consuming tokens."
# ─── Internationalization (optional) ─────────────────────────────────────────
# All i18n sections are optional. Without them, the English values above are used.
# To localize, add [i18n.LANG] sections (e.g. zh, ja, ko, es, fr, de).
# Settings translations are also optional — omit to keep English labels.
# ─── Chinese (简体中文) ────────────────────────────────────────────────────
[i18n.zh]
name = "DevOps Hand"
description = "自主 DevOps 工程师——CI/CD 管理、基础设施监控、部署自动化和事件响应"
category = "开发"
[i18n.zh.settings.infrastructure]
label = "基础设施类型"
description = "主要基础设施平台"
[i18n.zh.settings.ci_platform]
label = "CI/CD 平台"
description = "主要 CI/CD 平台"
[i18n.zh.settings.monitoring_focus]
label = "监控重点"
description = "主要监控和告警的关注方向"
[i18n.zh.settings.auto_monitor]
label = "自动监控"
description = "自动监控基础设施并在出现问题时告警"
[i18n.zh.settings.check_interval]
label = "健康检查间隔"
description = "自动健康检查的执行频率"
[i18n.zh.settings.service_urls]
label = "服务 URL"
description = "要监控的 URL 列表,以逗号分隔(例如 https://api.example.com/health,https://app.example.com)"
[i18n.zh.settings.alert_on_failure]
label = "故障告警"
description = "健康检查失败时发布事件通知"
[i18n.zh.settings.rollback_strategy]
label = "回滚策略"
description = "部署失败时的默认回滚方式"
[i18n.zh.settings.approval_mode]
label = "审批模式"
description = "将部署和基础设施操作加入队列供审核,而非直接执行"
# ─── Japanese (日本語) ────────────────────────────────────────────────────
[i18n.ja]
name = "DevOps Hand"
description = "自律型DevOpsエンジニア——CI/CD管理、インフラ監視、デプロイ自動化、インシデント対応"
category = "開発"
[i18n.ja.settings.infrastructure]
label = "インフラタイプ"
description = "主要なインフラプラットフォーム"
[i18n.ja.settings.ci_platform]
label = "CI/CDプラットフォーム"
description = "主要なCI/CDプラットフォーム"
[i18n.ja.settings.monitoring_focus]
label = "監視の重点"
description = "監視とアラートの主な対象分野"
[i18n.ja.settings.auto_monitor]
label = "自動監視"
description = "インフラを自動監視し、問題発生時にアラートを出す"
[i18n.ja.settings.check_interval]
label = "ヘルスチェック間隔"
description = "自動ヘルスチェックの実行間隔"
[i18n.ja.settings.service_urls]
label = "サービスURL"
description = "監視対象のURL一覧(カンマ区切り、例: https://api.example.com/health,https://app.example.com)"
[i18n.ja.settings.alert_on_failure]
label = "障害アラート"
description = "ヘルスチェック失敗時にイベント通知を発行する"
[i18n.ja.settings.rollback_strategy]
label = "ロールバック戦略"
description = "デプロイ失敗時のデフォルトのロールバック方法"
[i18n.ja.settings.approval_mode]
label = "承認モード"
description = "デプロイやインフラ操作を直接実行せず、レビュー用キューに追加する"
# ─── Spanish (Español) ────────────────────────────────────────────────────
[i18n.es]
name = "Hand de DevOps"
description = "Ingeniero DevOps autónomo — gestión de CI/CD, monitoreo de infraestructura, automatización de despliegues y respuesta a incidentes"
category = "Desarrollo"
[i18n.es.settings.infrastructure]
label = "Tipo de infraestructura"
description = "Plataforma de infraestructura principal"
[i18n.es.settings.ci_platform]
label = "Plataforma CI/CD"
description = "Plataforma principal de CI/CD"
[i18n.es.settings.monitoring_focus]
label = "Enfoque de monitoreo"
description = "Área principal de monitoreo y alertas"
[i18n.es.settings.auto_monitor]
label = "Monitoreo automático"
description = "Monitorear automáticamente la infraestructura y alertar ante problemas"
[i18n.es.settings.check_interval]
label = "Intervalo de comprobación de salud"
description = "Con qué frecuencia ejecutar las comprobaciones de salud automatizadas"
[i18n.es.settings.service_urls]
label = "URLs de servicios"
description = "Lista de URLs a monitorear separadas por comas (ej. https://api.example.com/health,https://app.example.com)"
[i18n.es.settings.alert_on_failure]
label = "Alertar ante fallos"
description = "Publicar eventos cuando las comprobaciones de salud fallen"
[i18n.es.settings.rollback_strategy]
label = "Estrategia de reversión"
description = "Enfoque de reversión predeterminado para despliegues fallidos"
[i18n.es.settings.approval_mode]
label = "Modo de aprobación"
description = "Poner acciones de despliegue e infraestructura en cola para revisión en lugar de ejecutarlas directamente"
# ─── French (Français) ────────────────────────────────────────────────────
[i18n.fr]
name = "Hand DevOps"
description = "Ingénieur DevOps autonome — gestion CI/CD, surveillance d'infrastructure, automatisation des déploiements et réponse aux incidents"
category = "Développement"
[i18n.fr.settings.infrastructure]
label = "Type d'infrastructure"
description = "Plateforme d'infrastructure principale"
[i18n.fr.settings.ci_platform]
label = "Plateforme CI/CD"
description = "Plateforme CI/CD principale"
[i18n.fr.settings.monitoring_focus]
label = "Axe de surveillance"
description = "Domaine principal de surveillance et d'alerte"
[i18n.fr.settings.auto_monitor]
label = "Surveillance automatique"
description = "Surveiller automatiquement l'infrastructure et alerter en cas de problèmes"
[i18n.fr.settings.check_interval]
label = "Intervalle de vérification de santé"
description = "Fréquence d'exécution des vérifications de santé automatisées"
[i18n.fr.settings.service_urls]
label = "URLs des services"
description = "Liste d'URLs à surveiller séparées par des virgules (ex. https://api.example.com/health,https://app.example.com)"
[i18n.fr.settings.alert_on_failure]
label = "Alerte en cas d'échec"
description = "Publier des événements lorsque les vérifications de santé échouent"
[i18n.fr.settings.rollback_strategy]
label = "Stratégie de retour en arrière"
description = "Approche de retour en arrière par défaut pour les déploiements échoués"
[i18n.fr.settings.approval_mode]
label = "Mode d'approbation"
description = "Mettre les actions de déploiement et d'infrastructure en file d'attente pour révision au lieu de les exécuter directement"
# ─── German (Deutsch) ────────────────────────────────────────────────────
[i18n.de]
name = "DevOps-Hand"
description = "Autonomer DevOps-Ingenieur — CI/CD-Verwaltung, Infrastrukturüberwachung, Deployment-Automatisierung und Incident-Response"
category = "Entwicklung"
[i18n.de.settings.infrastructure]
label = "Infrastrukturtyp"
description = "Primäre Infrastrukturplattform"
[i18n.de.settings.ci_platform]
label = "CI/CD-Plattform"
description = "Primäre CI/CD-Plattform"
[i18n.de.settings.monitoring_focus]
label = "Überwachungsschwerpunkt"
description = "Hauptbereich für Überwachung und Alarme"
[i18n.de.settings.auto_monitor]
label = "Automatische Überwachung"
description = "Infrastruktur automatisch überwachen und bei Problemen alarmieren"
[i18n.de.settings.check_interval]
label = "Gesundheitscheck-Intervall"
description = "Ausführungshäufigkeit der automatisierten Gesundheitschecks"
[i18n.de.settings.service_urls]
label = "Service-URLs"
description = "Kommagetrennte Liste der zu überwachenden URLs (z.B. https://api.example.com/health,https://app.example.com)"
[i18n.de.settings.alert_on_failure]
label = "Warnung bei Ausfall"
description = "Ereignisse veröffentlichen, wenn Gesundheitschecks fehlschlagen"
[i18n.de.settings.rollback_strategy]
label = "Rollback-Strategie"
description = "Standard-Rollback-Ansatz für fehlgeschlagene Deployments"
[i18n.de.settings.approval_mode]
label = "Genehmigungsmodus"
description = "Deployment- und Infrastrukturaktionen zur Überprüfung in die Warteschlange stellen, anstatt sie direkt auszuführen"
# ─── Korean (한국어) ────────────────────────────────────────────────────
[i18n.ko]
name = "DevOps Hand"
description = "자율 DevOps 엔지니어 — CI/CD 관리, 인프라 모니터링, 배포 자동화 및 인시던트 대응"
category = "개발"
[i18n.ko.settings.infrastructure]
label = "인프라 유형"
description = "주요 인프라 플랫폼"
[i18n.ko.settings.ci_platform]
label = "CI/CD 플랫폼"
description = "주요 CI/CD 플랫폼"
[i18n.ko.settings.monitoring_focus]
label = "모니터링 중점"
description = "주요 모니터링 및 알림 방향"
[i18n.ko.settings.auto_monitor]
label = "자동 모니터링"
description = "인프라를 자동으로 모니터링하고 문제 발생 시 알림"
[i18n.ko.settings.check_interval]
label = "상태 점검 간격"
description = "자동 상태 점검 실행 주기"
[i18n.ko.settings.service_urls]
label = "서비스 URL"
description = "모니터링할 URL 목록 (쉼표로 구분, 예: https://api.example.com/health,https://app.example.com)"
[i18n.ko.settings.alert_on_failure]
label = "장애 알림"
description = "상태 점검 실패 시 이벤트 알림 발행"
[i18n.ko.settings.rollback_strategy]
label = "롤백 전략"
description = "배포 실패 시 기본 롤백 방식"
[i18n.ko.settings.approval_mode]
label = "승인 모드"
description = "배포 및 인프라 작업을 직접 실행하지 않고 대기열에 추가하여 검토"
+538
View File
@@ -330,3 +330,541 @@ for domain in api.example.com app.example.com; do
echo "$domain: $expiry"
done
```
---
## Worked Examples
### Example 1: Zero-Downtime Deployment Pipeline
Full lifecycle from code merge to production traffic switch with rollback safety.
**Phases**: CI build and push image -> deploy to Green -> health-gate -> traffic switch -> monitor -> done (or rollback).
**Traffic switch script** (the critical step):
```bash
#!/bin/bash
set -euo pipefail
# Record current slot for rollback, then switch
CURRENT=$(kubectl get svc myapp-active -n production -o jsonpath='{.spec.selector.slot}')
echo "$CURRENT" > /tmp/rollback-slot
kubectl patch svc myapp-active -n production -p '{"spec":{"selector":{"slot":"green"}}}'
# Verify
sleep 5 && curl -sf https://api.example.com/api/health | jq '.version'
```
**Rollback**: read `/tmp/rollback-slot`, patch the service selector back, verify health.
**Decision flowchart**:
```
Code merged to main
|
v
CI build + test ----[FAIL]----> Block merge, notify author
|
[PASS]
v
Deploy to Green
|
v
Health checks ------[FAIL]----> Alert on-call, keep Blue active
|
[PASS]
v
Switch traffic to Green
|
v
Monitor 15 min -----[ERROR SPIKE]----> Rollback to Blue, open incident
|
[STABLE]
v
Mark Green as new Blue, done
```
---
### Example 2: Production Incident Response
Walkthrough of a real-world SEV1 incident: API latency spike caused by a database connection pool exhaustion.
**Timeline**
| Time (UTC) | Event | Actor |
|------------|-------|-------|
| 14:02 | PagerDuty alert: P95 latency > 2s on `/api/orders` | Monitoring |
| 14:04 | On-call acknowledges, opens incident channel `#inc-2025-0312` | On-call engineer |
| 14:06 | Check dashboard: request queue depth spiking, error rate at 12% | On-call engineer |
| 14:10 | Identify DB connection pool at 100% utilization | On-call engineer |
| 14:12 | Find long-running query from analytics job (started 13:55) | On-call engineer |
| 14:14 | Kill the runaway query, pool starts draining | On-call engineer |
| 14:18 | Latency returns to normal, error rate drops to 0.2% | Monitoring |
| 14:20 | Incident mitigated, continue monitoring | On-call engineer |
| 14:45 | Root cause confirmed: analytics cron job without query timeout | Investigation |
| 15:00 | Incident resolved, post-mortem scheduled | Incident commander |
**Detection -- Alerting rules that fired**
```yaml
# Prometheus alerting rule
groups:
- name: api-latency
rules:
- alert: HighAPILatency
expr: histogram_quantile(0.95, rate(http_request_duration_seconds_bucket{job="api"}[5m])) > 2
for: 2m
labels:
severity: critical
annotations:
summary: "P95 latency above 2s for 2+ minutes"
runbook: "https://wiki.internal/runbooks/high-latency"
```
**Triage -- Quick diagnosis commands**
```bash
# 1. Check if it's a specific endpoint or global
curl -s "http://prometheus:9090/api/v1/query?query=topk(5,rate(http_request_duration_seconds_sum[5m])/rate(http_request_duration_seconds_count[5m]))" | jq '.data.result[] | {endpoint: .metric.handler, avg_latency: .value[1]}'
# 2. Check database connection pool
psql -c "SELECT count(*) as total, state FROM pg_stat_activity GROUP BY state;"
# 3. Find the blocking query
psql -c "SELECT pid, now() - query_start AS duration, query
FROM pg_stat_activity
WHERE state = 'active' AND now() - query_start > interval '1 minute'
ORDER BY duration DESC LIMIT 5;"
```
**Mitigation -- Kill the offending query**
```bash
# Kill the long-running query by PID
psql -c "SELECT pg_terminate_backend(12345);"
# Verify pool is recovering
watch -n 2 'psql -t -c "SELECT count(*) FROM pg_stat_activity WHERE state = '\''active'\'';"'
```
**Prevention -- Fix applied after incident**
```sql
-- Set statement timeout for analytics role
ALTER ROLE analytics_readonly SET statement_timeout = '300s';
```
```yaml
# Add connection pool monitoring alert
- alert: DBConnectionPoolNearCapacity
expr: pg_stat_activity_count / pg_settings_max_connections > 0.8
for: 1m
labels:
severity: warning
annotations:
summary: "DB connection pool above 80% capacity"
```
**Post-mortem action items**:
- [ ] Add `statement_timeout` to all non-interactive database roles
- [ ] Add connection pool utilization alerts (threshold: 80%)
- [ ] Move analytics queries to read replica
- [ ] Add circuit breaker to API when pool utilization exceeds 90%
---
### Example 3: Infrastructure Scaling Event
Scaling a Kubernetes deployment in response to sustained load increase.
**Phase 1 -- Alert triggers**
```
Alert: HighCPUUtilization
Condition: avg(cpu_usage) > 80% for 10 minutes
Current: 87% across 3 pods
Namespace: production
Deployment: order-service
```
**Phase 2 -- Capacity analysis**
```bash
kubectl top pods -l app=order-service -n production # Per-pod CPU/memory
kubectl describe nodes | grep -A 5 "Allocated resources" # Node headroom
kubectl get hpa order-service -n production # Current HPA state
# Result: 87% CPU across 3 pods, 842 req/s (2x baseline)
```
**Phase 3 -- Scaling decision matrix**
| Metric | Current | Target | Action |
|--------|---------|--------|--------|
| CPU usage | 87% | < 70% | Scale out |
| Request rate | 842/s | - | 2x normal, sustained |
| Memory | 258Mi avg | 512Mi limit | Headroom OK |
| Pod count | 3 | 6 (estimated) | Double replicas |
| Node capacity | 72% | < 85% | Sufficient for 6 pods |
**Phase 4 -- Implement scaling**
```bash
# Option A: Manual scale (immediate)
kubectl scale deployment order-service -n production --replicas=6
# Option B: Adjust HPA for sustained load (preferred)
kubectl patch hpa order-service -n production \
-p '{"spec":{"minReplicas":5,"maxReplicas":15}}'
# Monitor rollout
kubectl rollout status deployment/order-service -n production
# Watch pods come up
kubectl get pods -l app=order-service -n production -w
```
**Phase 5 -- Verify scaling**
Confirm via `kubectl top pods` (CPU should drop to ~45% per pod), check P95 latency is back below SLO, and verify error rate < 1%.
**Post-scaling actions**:
- [ ] Investigate root cause of traffic increase (marketing event? bot traffic? organic growth?)
- [ ] Update capacity planning spreadsheet
- [ ] If sustained, adjust resource requests/limits and HPA baselines
- [ ] Set calendar reminder to review and potentially scale down in 48h
---
## Observability Deep Dive
### Structured Logging
Use consistent JSON log format across all services for machine-parseable aggregation.
**Log format standard**: JSON with required fields: `timestamp`, `level`, `service`, `trace_id`, `span_id`, `request_id`, `message`. Add `error` and `context` (structured key-value) as needed.
**Log levels -- when to use each**:
| Level | Purpose | Example | Persisted |
|-------|---------|---------|-----------|
| `error` | Requires human attention | Payment processing failed | 90 days |
| `warn` | Degraded but recoverable | Retry succeeded on 2nd attempt | 30 days |
| `info` | Business-significant events | Order placed, user logged in | 14 days |
| `debug` | Developer troubleshooting | Cache hit/miss, query timing | 3 days |
| `trace` | Fine-grained flow tracking | Function entry/exit, variable state | 1 day (sampled) |
**Correlation IDs**: generate `X-Request-ID` at API gateway, propagate through all downstream calls. Query across services by filtering on `request_id.keyword` in Elasticsearch/OpenSearch.
### Distributed Tracing
**Core concepts**: A Trace is the end-to-end request path. Each service call is a Span with timing. Spans nest to show the call tree (e.g., API Gateway -> Order Service -> DB Query + Payment Service -> Stripe API).
**OpenTelemetry propagation**: `traceparent: 00-<trace-id>-<span-id>-<flags>`, `tracestate: vendor=value`.
**Useful trace queries (Jaeger/Tempo)**:
```bash
curl -s "http://jaeger:16686/api/traces?service=order-service&minDuration=1s&limit=20" # Slow traces
curl -s "http://jaeger:16686/api/traces?service=order-service&tags=error%3Dtrue&limit=20" # Error traces
```
### Alerting Best Practices
**Avoid alert fatigue -- rules of thumb**:
- Every alert must have a runbook link
- Every alert must be actionable (if no one needs to act, it is a log, not an alert)
- Group related alerts to avoid notification storms
- Use inhibition rules: if the cluster is down, suppress per-pod alerts
**SLO-based alerting (burn rate)**:
```yaml
# SLO: 99.9% availability = 43.2 min/month error budget
# Fast burn (exhausts budget in 2h): error_ratio > 14.4 * 0.001 for 2m -> critical
# Slow burn (exhausts budget in 3d): error_ratio > 3 * 0.001 for 15m -> warning
groups:
- name: slo-burn-rate
rules:
- alert: SLOBurnRateCritical
expr: sum(rate(http_requests_total{code=~"5.."}[5m])) / sum(rate(http_requests_total[5m])) > (14.4 * 0.001)
for: 2m
labels: { severity: critical }
- alert: SLOBurnRateWarning
expr: sum(rate(http_requests_total{code=~"5.."}[1h])) / sum(rate(http_requests_total[1h])) > (3 * 0.001)
for: 15m
labels: { severity: warning }
```
**Runbook template**: Each alert runbook should cover: what the alert means (one sentence), impact scope, diagnosis steps (dashboard + commands), mitigation (quick fix vs proper fix), and escalation path (who to contact after 15 min).
### Metrics Collection Patterns
**RED Method (request-scoped services)**:
| Metric | What | PromQL Example |
|--------|------|----------------|
| **R**ate | Requests per second | `sum(rate(http_requests_total[5m]))` |
| **E**rrors | Failed requests per second | `sum(rate(http_requests_total{code=~"5.."}[5m]))` |
| **D**uration | Latency distribution | `histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket[5m])) by (le))` |
**USE Method (infrastructure resources)**:
| Metric | What | Example Check |
|--------|------|---------------|
| **U**tilization | % time resource is busy | `avg(rate(node_cpu_seconds_total{mode!="idle"}[5m]))` |
| **S**aturation | Queue depth / backlog | `node_load1 / count(node_cpu_seconds_total{mode="idle"})` |
| **E**rrors | Error event count | `rate(node_disk_io_time_weighted_seconds_total[5m])` |
**When to use which**:
- RED for services that handle requests (APIs, web servers, message consumers)
- USE for infrastructure (CPU, memory, disk, network interfaces, queues)
- Combine both for a complete picture
---
## Security Operations
### Secret Management
**Principles**:
- Never store secrets in source code, environment variables (in Dockerfiles), or container images
- Use a secrets manager (Vault, AWS Secrets Manager, K8s Secrets with encryption at rest)
- Rotate secrets on a schedule and immediately after any suspected compromise
- Audit all secret access
**Vault pattern -- inject secrets at runtime**:
```bash
# Store a secret
vault kv put secret/myapp/db \
username="app_user" \
password="$(openssl rand -base64 32)"
# Read a secret (application startup)
vault kv get -format=json secret/myapp/db | jq -r '.data.data.password'
# Enable audit logging
vault audit enable file file_path=/var/log/vault-audit.log
```
**Kubernetes secrets -- from Vault using sidecar injector**:
Annotate the pod template with `vault.hashicorp.com/agent-inject: "true"`, specify the role and secret path. The Vault agent sidecar renders secrets to `/vault/secrets/` and the app sources them at startup. Key annotations: `agent-inject-secret-<name>` for the path, `agent-inject-template-<name>` for the rendering template.
**Secret rotation checklist**:
- [ ] Generate new secret value
- [ ] Update secret in secrets manager
- [ ] Restart/reload affected services (rolling, not all-at-once)
- [ ] Verify services authenticate with new secret
- [ ] Revoke the old secret value
- [ ] Confirm no services are still using the old secret
### Container Security Scanning
```bash
# Scan image for vulnerabilities (Trivy) -- fail CI on critical
trivy image --exit-code 1 --severity CRITICAL registry.example.com/myapp:$CI_COMMIT_SHA
# Scan K8s cluster for misconfigurations
trivy k8s --report summary cluster
```
**Dockerfile security essentials**: use pinned base image tags (not `:latest`), run as non-root (`USER app`), copy only needed files, never bake secrets into image layers.
### Network Security Policies
**Kubernetes NetworkPolicy -- default deny with explicit allow**:
```yaml
# Default deny all ingress, then allow specific paths
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: allow-gateway-to-orders
namespace: production
spec:
podSelector:
matchLabels: { app: order-service }
ingress:
- from:
- podSelector:
matchLabels: { app: api-gateway }
ports:
- { protocol: TCP, port: 8080 }
```
Apply a `default-deny-ingress` policy (empty `podSelector`, `policyTypes: [Ingress]`) per namespace first, then layer allow rules on top.
### Compliance as Code
**Policy enforcement with OPA/Gatekeeper**: use `K8sRequiredResources` constraints to enforce `limits.cpu`, `limits.memory`, `requests.cpu`, `requests.memory` on all pods in production namespaces.
**Quick compliance audit commands**:
```bash
# Find pods without resource limits
kubectl get pods -A -o json | jq -r '.items[] | select(.spec.containers[].resources.limits == null) | .metadata.namespace + "/" + .metadata.name'
# Find containers running as root
kubectl get pods -A -o json | jq -r '.items[] | select(.spec.containers[].securityContext.runAsNonRoot != true) | .metadata.namespace + "/" + .metadata.name'
# Find ingress without TLS
kubectl get ingress -A -o json | jq -r '.items[] | select(.spec.tls == null) | .metadata.namespace + "/" + .metadata.name'
```
---
## Automation Patterns
### Auto-Remediation
**Restart on OOM (Kubernetes)**:
```yaml
# Built-in: set resource limits and let K8s handle OOM restarts
apiVersion: apps/v1
kind: Deployment
spec:
template:
spec:
containers:
- name: myapp
resources:
limits:
memory: "512Mi"
requests:
memory: "256Mi"
# Liveness probe: restart if unhealthy
livenessProbe:
httpGet:
path: /health
port: 8080
initialDelaySeconds: 10
periodSeconds: 10
failureThreshold: 3
```
**Scale on load (HPA with custom metrics)**:
```yaml
apiVersion: autoscaling/v2
kind: HorizontalPodAutoscaler
metadata:
name: order-service
spec:
scaleTargetRef: { apiVersion: apps/v1, kind: Deployment, name: order-service }
minReplicas: 3
maxReplicas: 20
behavior:
scaleUp: { stabilizationWindowSeconds: 60, policies: [{ type: Percent, value: 50, periodSeconds: 60 }] }
scaleDown: { stabilizationWindowSeconds: 300, policies: [{ type: Percent, value: 25, periodSeconds: 120 }] }
metrics:
- type: Resource
resource: { name: cpu, target: { type: Utilization, averageUtilization: 70 } }
- type: Pods
pods: { metric: { name: http_requests_per_second }, target: { type: AverageValue, averageValue: "1000" } }
```
**Rotate secrets on expiry (CronJob)**:
Use a K8s CronJob (e.g., monthly `"0 2 1 * *"`) with a `secret-rotator` service account that: generates new password -> updates Vault -> alters DB role password -> triggers rolling restart via `kubectl rollout restart`.
### GitOps Workflow
**Repository as source of truth**:
```
infrastructure-repo/
|-- apps/
| |-- order-service/
| | |-- deployment.yaml
| | |-- service.yaml
| | |-- hpa.yaml
| | `-- kustomization.yaml
| `-- payment-service/
| |-- deployment.yaml
| `-- kustomization.yaml
|-- base/
| |-- namespace.yaml
| |-- network-policies.yaml
| `-- resource-quotas.yaml
`-- overlays/
|-- staging/
| `-- kustomization.yaml
`-- production/
`-- kustomization.yaml
```
**Reconciliation loop (ArgoCD application)**:
```yaml
apiVersion: argoproj.io/v1alpha1
kind: Application
metadata:
name: order-service
namespace: argocd
spec:
project: default
source:
repoURL: https://github.com/org/infrastructure-repo.git
targetRevision: main
path: apps/order-service
destination: { server: "https://kubernetes.default.svc", namespace: production }
syncPolicy:
automated: { prune: true, selfHeal: true }
syncOptions: [CreateNamespace=true]
retry: { limit: 3, backoff: { duration: 5s, factor: 2, maxDuration: 3m } }
```
**GitOps deployment flow**:
```
Developer pushes image tag update to infrastructure-repo
|
v
ArgoCD detects drift between git state and cluster state
|
v
ArgoCD syncs: applies manifests from git to cluster
|
v
Kubernetes rolls out new pods
|
v
ArgoCD verifies health (readiness probes pass)
|
[HEALTHY] --> Done
[DEGRADED] --> ArgoCD marks sync as failed, alerts on-call
```
### Database Backup and Restore
**Automated backup (PostgreSQL)** -- run via cron `0 */6 * * *`:
```bash
#!/bin/bash
set -euo pipefail
TIMESTAMP=$(date +%Y%m%d_%H%M%S)
DB_NAME="production"
BACKUP_DIR="/backups/postgres"
pg_dump -Fc -Z 9 "$DB_NAME" > "${BACKUP_DIR}/${DB_NAME}_${TIMESTAMP}.dump"
aws s3 cp "${BACKUP_DIR}/${DB_NAME}_${TIMESTAMP}.dump" \
"s3://backups-bucket/postgres/" --storage-class STANDARD_IA
find "$BACKUP_DIR" -name "*.dump" -mtime +30 -delete
```
**Restore procedure**:
```bash
#!/bin/bash
set -euo pipefail
BACKUP_FILE=$1 # e.g., "production_20250315_060000.dump"
RESTORE_DB="production_restore"
aws s3 cp "s3://backups-bucket/postgres/$BACKUP_FILE" /tmp/restore.dump
psql -c "DROP DATABASE IF EXISTS $RESTORE_DB;" && psql -c "CREATE DATABASE $RESTORE_DB;"
pg_restore -d "$RESTORE_DB" -j 4 --no-owner /tmp/restore.dump
# Verify: check row counts on key tables, then clean up
rm /tmp/restore.dump
```
### Disaster Recovery Runbook Template
**Recovery Objectives**: Define RTO (e.g., 1 hour) and RPO (e.g., 6 hours) per service.
**Prerequisites**: backup storage access, Terraform state access, DNS management access, stakeholder comms channel.
| Scenario | Key Steps |
|----------|-----------|
| **Single service failure** | Check pod status -> restart deployment -> if fails, `kubectl rollout undo` -> verify health |
| **Database failure** | `pg_isready` -> promote replica (or restore from backup) -> update connection strings -> verify data integrity |
| **Full region outage** | Confirm via provider status page -> notify stakeholders -> switch DNS to DR region -> verify traffic -> failback when primary recovers |
**Communication template**: Subject `[INCIDENT] Service -- Status`. Body: what happened, impact, current status, ETA, next update time.
**Post-recovery checklist**: health checks passing, data integrity verified, monitoring restored, backups resumed, incident report filed, post-mortem scheduled within 48h.