feat(hands): complete i18n fixes, SKILL.md enhancements, and README overhaul

- Fix French accent characters (é/è/ê/ç/â/ô) across all 14 HAND.toml files
- Fix German special characters (ä/ö/ü/ß) across all 14 HAND.toml files
- Add category translations to all 6 i18n language blocks in all 14 hands
- Enhance SKILL.md content for 9 hands with practical examples and workflows
- Trim bloated SKILL.md files (apitester 1400→892, devops 1301→870)
- Rewrite root README.md with accurate stats, complete hand/integration tables
- Update hands/README.md with full 14-hand listing and i18n documentation
This commit is contained in:
Evan Hu committed 2026-03-23 00:18:18 +09:00
1 parent 315f955ce2
commit 33d279889c
27 files changed
+10001 -78

No files matched your search

+235
View File
@@ -425,6 +425,241 @@ token_consumption = "high"
default_active = false
activation_warning = "Predictor hand runs continuously and generates predictions, consuming tokens."
# ─── Internationalization (optional) ─────────────────────────────────────────
# All i18n sections are optional. Without them, the English values above are used.
# To localize, add [i18n.LANG] sections (e.g. zh, ja, ko, es, fr, de).
# Settings translations are also optional — omit to keep English labels.
# ─── Chinese (简体中文) ────────────────────────────────────────────────────
[i18n.zh]
name = "预测 Hand"
description = "自主预测智能体——收集信号、构建推理链、做出校准预测并追踪准确度"
category = "数据"
[i18n.zh.settings.prediction_domain]
label = "预测领域"
description = "预测的主要关注领域"
[i18n.zh.settings.time_horizon]
label = "时间跨度"
description = "预测的前瞻时间范围"
[i18n.zh.settings.data_sources]
label = "数据来源"
description = "监控信号的来源类型"
[i18n.zh.settings.report_frequency]
label = "报告频率"
description = "生成预测报告的频率"
[i18n.zh.settings.predictions_per_report]
label = "每份报告预测数"
description = "每份报告包含的预测条目数量"
[i18n.zh.settings.track_accuracy]
label = "追踪准确度"
description = "在预测时间窗口到期后对历史预测进行评分"
[i18n.zh.settings.confidence_threshold]
label = "置信度阈值"
description = "纳入预测报告的最低置信度"
[i18n.zh.settings.contrarian_mode]
label = "逆向思维模式"
description = "主动寻找并展示与主流共识相反的预测"
# ─── Japanese (日本語) ────────────────────────────────────────────────────
[i18n.ja]
name = "予測 Hand"
description = "自律型予測エージェント——シグナル収集、推論チェーン構築、キャリブレーション済み予測、精度追跡"
category = "データ"
[i18n.ja.settings.prediction_domain]
label = "予測ドメイン"
description = "予測の主な対象分野"
[i18n.ja.settings.time_horizon]
label = "予測期間"
description = "どのくらい先まで予測するか"
[i18n.ja.settings.data_sources]
label = "データソース"
description = "シグナルを監視するソースの種類"
[i18n.ja.settings.report_frequency]
label = "レポート頻度"
description = "予測レポートの生成頻度"
[i18n.ja.settings.predictions_per_report]
label = "レポートあたりの予測数"
description = "各レポートに含める予測項目の数"
[i18n.ja.settings.track_accuracy]
label = "精度追跡"
description = "予測期間が終了した過去の予測にスコアを付ける"
[i18n.ja.settings.confidence_threshold]
label = "信頼度しきい値"
description = "予測をレポートに含めるための最低信頼度"
[i18n.ja.settings.contrarian_mode]
label = "逆張りモード"
description = "コンセンサスに反する予測を積極的に探索・提示する"
# ─── Spanish (Español) ────────────────────────────────────────────────────
[i18n.es]
name = "Hand de Predicciones"
description = "Predictor autónomo del futuro — recopila señales, construye cadenas de razonamiento, genera predicciones calibradas y rastrea la precisión"
category = "Datos"
[i18n.es.settings.prediction_domain]
label = "Dominio de predicción"
description = "Dominio principal para las predicciones"
[i18n.es.settings.time_horizon]
label = "Horizonte temporal"
description = "Qué tan lejos en el futuro predecir"
[i18n.es.settings.data_sources]
label = "Fuentes de datos"
description = "Qué tipos de fuentes monitorear para señales"
[i18n.es.settings.report_frequency]
label = "Frecuencia de informes"
description = "Con qué frecuencia generar informes de predicción"
[i18n.es.settings.predictions_per_report]
label = "Predicciones por informe"
description = "Número de predicciones a incluir por informe"
[i18n.es.settings.track_accuracy]
label = "Rastrear precisión"
description = "Puntuar predicciones pasadas cuando su horizonte temporal expire"
[i18n.es.settings.confidence_threshold]
label = "Umbral de confianza"
description = "Confianza mínima para incluir una predicción"
[i18n.es.settings.contrarian_mode]
label = "Modo contrario"
description = "Buscar y presentar activamente predicciones contrarias al consenso"
# ─── French (Français) ────────────────────────────────────────────────────
[i18n.fr]
name = "Hand de Prédictions"
description = "Prédicteur autonome — collecte de signaux, construction de chaînes de raisonnement, prédictions calibrées et suivi de la précision"
category = "Données"
[i18n.fr.settings.prediction_domain]
label = "Domaine de prédiction"
description = "Domaine principal pour les prédictions"
[i18n.fr.settings.time_horizon]
label = "Horizon temporel"
description = "Jusqu'où prédire dans le futur"
[i18n.fr.settings.data_sources]
label = "Sources de données"
description = "Types de sources à surveiller pour les signaux"
[i18n.fr.settings.report_frequency]
label = "Fréquence des rapports"
description = "Fréquence de génération des rapports de prédiction"
[i18n.fr.settings.predictions_per_report]
label = "Prédictions par rapport"
description = "Nombre de prédictions à inclure par rapport"
[i18n.fr.settings.track_accuracy]
label = "Suivi de la précision"
description = "Évaluer les prédictions passées lorsque leur horizon temporel expire"
[i18n.fr.settings.confidence_threshold]
label = "Seuil de confiance"
description = "Confiance minimale pour inclure une prédiction"
[i18n.fr.settings.contrarian_mode]
label = "Mode contraire"
description = "Rechercher et présenter activement des prédictions contraires au consensus"
# ─── German (Deutsch) ────────────────────────────────────────────────────
[i18n.de]
name = "Vorhersage-Hand"
description = "Autonomer Vorhersage-Agent — Signalerfassung, Aufbau von Argumentationsketten, kalibrierte Vorhersagen und Genauigkeitsverfolgung"
category = "Daten"
[i18n.de.settings.prediction_domain]
label = "Vorhersagedomäne"
description = "Hauptdomäne für Vorhersagen"
[i18n.de.settings.time_horizon]
label = "Zeithorizont"
description = "Wie weit in die Zukunft vorhergesagt werden soll"
[i18n.de.settings.data_sources]
label = "Datenquellen"
description = "Welche Quellentypen auf Signale überwacht werden"
[i18n.de.settings.report_frequency]
label = "Berichtshäufigkeit"
description = "Wie oft Vorhersageberichte generiert werden"
[i18n.de.settings.predictions_per_report]
label = "Vorhersagen pro Bericht"
description = "Anzahl der Vorhersagen pro Bericht"
[i18n.de.settings.track_accuracy]
label = "Genauigkeitsverfolgung"
description = "Vergangene Vorhersagen bewerten, wenn ihr Zeithorizont abläuft"
[i18n.de.settings.confidence_threshold]
label = "Konfidenzschwelle"
description = "Mindestvertrauen für die Aufnahme einer Vorhersage"
[i18n.de.settings.contrarian_mode]
label = "Konträrer Modus"
description = "Aktiv nach Vorhersagen suchen und präsentieren, die dem Konsens widersprechen"
# ─── Korean (한국어) ────────────────────────────────────────────────────
[i18n.ko]
name = "예측 Hand"
description = "자율 미래 예측 에이전트 — 신호 수집, 추론 체인 구축, 보정된 예측 수행 및 정확도 추적"
category = "데이터"
[i18n.ko.settings.prediction_domain]
label = "예측 분야"
description = "예측의 주요 관심 분야"
[i18n.ko.settings.time_horizon]
label = "시간 범위"
description = "예측의 미래 전망 기간"
[i18n.ko.settings.data_sources]
label = "데이터 소스"
description = "신호를 모니터링할 소스 유형"
[i18n.ko.settings.report_frequency]
label = "보고서 빈도"
description = "예측 보고서 생성 주기"
[i18n.ko.settings.predictions_per_report]
label = "보고서당 예측 수"
description = "각 보고서에 포함할 예측 항목 수"
[i18n.ko.settings.track_accuracy]
label = "정확도 추적"
description = "예측 기간 만료 후 과거 예측에 대한 점수 평가"
[i18n.ko.settings.confidence_threshold]
label = "신뢰도 임계값"
description = "예측 보고서에 포함하기 위한 최소 신뢰도"
[i18n.ko.settings.contrarian_mode]
label = "역발상 모드"
description = "주류 컨센서스에 반하는 예측을 적극적으로 탐색하고 제시"
+585
View File
@@ -184,6 +184,591 @@ PREDICTION: [Specific, falsifiable claim]
---
## Worked Examples
### Example 1: Corporate Acquisition
**Question**: "Will Acme Corp be acquired within 12 months?" (asked January 2025)
```
PREDICTION: Acme Corp (mid-cap SaaS, $2B market cap) will be acquired by January 2026
1. REFERENCE CLASS (Outside View)
Base rate: ~5-7% of publicly traded mid-cap SaaS companies receive
acquisition offers in any given 12-month period.
Reference examples:
- Splunk acquired by Cisco (2023) — similar scale, strategic buyer
- Figma attempted acquisition by Adobe (2022) — regulatory block
- Nuance acquired by Microsoft (2021) — vertical SaaS, strategic fit
- Mandiant acquired by Google (2022) — security vertical
- Cvent acquired by Blackstone (2021) — PE buyout at depressed valuation
Starting probability: 6%
2. SPECIFIC EVIDENCE (Inside View)
Signals FOR (+):
a. Board hired Goldman Sachs as advisor (leaked filing)
— strength: STRONG — adjustment: +20%
(Companies that retain M&A advisors complete a transaction ~40% of the time)
b. CEO sold 30% of personal holdings in Q4 (SEC filing)
— strength: MODERATE — adjustment: +5%
c. Two major competitors acquired in past 18 months (market consolidation)
— strength: MODERATE — adjustment: +8%
d. Revenue growth decelerated from 35% to 18% YoY (earnings report)
— strength: MODERATE — adjustment: +5%
(Slower-growth companies more likely to accept acquisition offers)
Signals AGAINST (-):
a. Founder still holds 25% voting control and has said "we're building for
the long term" (recent interview)
— strength: STRONG — adjustment: -12%
b. Stock price at all-time high — acquirer must pay steep premium
— strength: MODERATE — adjustment: -5%
c. Current antitrust environment — FTC blocking more deals
— strength: WEAK — adjustment: -3%
3. SYNTHESIS
Starting probability (base rate): 6%
Signals for: +20% +5% +8% +5% = +38%
Signals against: -12% -5% -3% = -20%
Net adjustment: +18%
Raw probability: 24%
Sanity check: ~1 in 4 feels right given the strong M&A advisor signal
balanced against founder control.
Final probability: 25%
4. KEY ASSUMPTIONS
- Goldman engagement is for M&A (not debt restructuring):
If wrong, probability drops to 8%
- Founder is willing to sell at the right price:
If wrong (founder vetoes any deal), probability drops to 3%
- Regulatory environment doesn't tighten further:
If wrong, probability drops to 18%
5. RESOLUTION
Date: January 31, 2026
Criteria: Definitive merger agreement announced (not just rumors)
Data source: SEC EDGAR (8-K filing), Bloomberg terminal
```
### Example 2: Technology Adoption
**Question**: "Will WebAssembly (Wasm) reach mainstream server-side adoption by 2027?"
```
PREDICTION: >20% of new cloud-deployed services will use Wasm runtimes by
end of 2027
1. REFERENCE CLASS (Outside View)
Technology adoption lifecycle (Rogers curve):
- Innovators (2.5%) → Early Adopters (13.5%) → Early Majority (34%)
- Crossing from Early Adopters to Early Majority typically takes 3-5 years
after first production deployments
- First serious server-side Wasm deployments: ~2022 (Fermyon, Cosmonic)
- Current status (2025): Late Early Adopter stage
Historical analogues for infrastructure tech adoption:
- Containers (Docker 2013 → mainstream 2017-2018): ~4-5 years
- Kubernetes (2014 → mainstream 2018-2019): ~4-5 years
- Serverless (Lambda 2014 → mainstream 2018-2020): ~4-6 years
Base rate for "infrastructure tech reaching 20% adoption within 5 years
of first production use": ~30%
Starting probability: 30%
2. SPECIFIC EVIDENCE (Inside View)
Signals FOR (+):
a. WASI standard maturing — WASI Preview 2 shipped, component model
stabilizing (W3C working group)
— strength: STRONG — adjustment: +8%
b. Major cloud providers offering Wasm runtimes (Fastly, Cloudflare Workers,
Azure, AWS exploring)
— strength: STRONG — adjustment: +10%
c. Docker adding Wasm support natively (announced 2022, shipping)
— strength: MODERATE — adjustment: +5%
Signals AGAINST (-):
a. Ecosystem still fragmented — multiple competing runtimes, toolchain gaps
— strength: STRONG — adjustment: -10%
b. Containers already "good enough" for most workloads — weak forcing
function to switch
— strength: STRONG — adjustment: -8%
c. Wasm language support uneven — great for Rust/C++, mediocre for Python/JS
— strength: MODERATE — adjustment: -5%
Leading indicators to track:
- CNCF survey: % of respondents evaluating/using Wasm
- Job postings mentioning Wasm (Indeed/LinkedIn trend)
- GitHub stars and contributors for top Wasm runtimes (wasmtime, wasmer)
- WASI spec milestone dates vs planned dates
3. SYNTHESIS
Starting probability (base rate): 30%
Signals for: +8% +10% +5% = +23%
Signals against: -10% -8% -5% = -23%
Net adjustment: 0%
Final probability: 30%
Interpretation: The positive and negative signals roughly cancel out.
The base rate from analogous infrastructure technologies holds.
This is genuinely uncertain — the "chasm" crossing is the key risk.
4. KEY ASSUMPTIONS
- "Mainstream" defined as >20% of NEW deployments (not total installed base)
- WASI component model reaches 1.0 stable by mid-2026
If delayed beyond 2026: probability drops to 15%
- No competing paradigm emerges (e.g., eBPF expanding scope):
If strong competitor: probability drops to 20%
5. RESOLUTION
Date: December 31, 2027
Criteria: CNCF annual survey shows >20% respondents using Wasm in production
Data source: CNCF Annual Survey, Datadog Container Report
```
### Example 3: Geopolitical Forecast
**Question**: "Will US-China trade tensions escalate significantly in 2025?"
(Defined as: new tariffs >25% on >$100B of goods, or export controls expanded
to 3+ new technology categories)
```
PREDICTION: Significant escalation of US-China trade tensions in 2025
1. REFERENCE CLASS (Outside View)
Historical trade conflict escalation pattern:
- US-China trade relations since 2018: escalation occurred in 4 of 7 years
- In election year +1 (new/returning administration): escalation rate ~60%
- Trade wars historically escalate in steps, with retaliation cycles
Starting probability: 55%
2. SPECIFIC EVIDENCE (Inside View)
Signals FOR (+):
a. Administration rhetoric on China hawkish across both parties
— strength: STRONG — adjustment: +10%
b. Semiconductor export controls already expanding (ASML, Tokyo Electron)
— strength: STRONG — adjustment: +8%
c. China retaliating with rare earth export restrictions
— strength: MODERATE — adjustment: +5%
Signals AGAINST (-):
a. Business lobbying against further tariffs (Chamber of Commerce, farm lobby)
— strength: MODERATE — adjustment: -5%
b. Inflation concerns create political cost for tariffs
— strength: MODERATE — adjustment: -5%
c. Diplomatic channels active (recent bilateral meetings)
— strength: WEAK — adjustment: -3%
Scenario mapping:
┌─────────────────────────┬─────────────┬────────────────────┐
│ Scenario │ Probability │ Key trigger │
├─────────────────────────┼─────────────┼────────────────────┤
│ Major escalation │ 25% │ Taiwan crisis or │
│ (new tariffs + controls │ │ tech IP theft case │
│ + retaliatory cycle) │ │ │
├─────────────────────────┼─────────────┼────────────────────┤
│ Moderate escalation │ 40% │ Incremental tariff │
│ (meets our threshold) │ │ increases + 1-2 │
│ │ │ new export controls│
├─────────────────────────┼─────────────┼────────────────────┤
│ Status quo / minor │ 30% │ Diplomatic deals, │
│ changes │ │ election distraction│
├─────────────────────────┼─────────────┼────────────────────┤
│ De-escalation │ 5% │ Grand bargain │
│ (reduced tariffs) │ │ (historically rare)│
└─────────────────────────┴─────────────┴────────────────────┘
P(meets our escalation threshold) = 25% + 40% = 65%
3. SYNTHESIS
Starting probability (base rate): 55%
Signals for: +10% +8% +5% = +23%
Signals against: -5% -5% -3% = -13%
Net adjustment: +10%
Raw probability: 65%
Cross-check with scenario mapping: 65% — consistent.
Final probability: 65%
4. KEY ASSUMPTIONS
- No major geopolitical crisis (Taiwan strait) that causes extreme
escalation or extreme restraint: If crisis occurs, split to
80% (escalation) or 20% (restraint/avoidance)
- US economy remains stable: If recession hits, probability drops
to 45% (political cost of tariffs rises)
- China does not make major trade concessions preemptively:
If it does, probability drops to 30%
5. RESOLUTION
Date: December 31, 2025
Criteria: Cumulative new tariffs >25% on >$100B goods OR export controls
expanded to 3+ new technology categories (per USTR/BIS announcements)
Data source: USTR tariff schedule, BIS Entity List updates, Congressional
Research Service reports
```
---
## Fermi Estimation Techniques
Fermi estimation is the art of making reasonable order-of-magnitude guesses
by breaking unknowable questions into smaller, estimable pieces.
### Step-by-Step Process
```
1. DEFINE the quantity you want to estimate
→ Be specific about units, scope, and timeframe
2. DECOMPOSE into factors you can estimate independently
→ Prefer multiplication chains: A × B × C
→ Each factor should be something you can reason about
3. ESTIMATE each factor
→ Use round numbers (powers of 10 when possible)
→ State your confidence range for each factor
4. MULTIPLY and sanity-check
→ Does the result pass the "smell test"?
→ Cross-check with any known anchors
5. STATE your uncertainty
→ Fermi estimates are typically accurate within 1 order of magnitude
→ Give a range: [estimate / 3, estimate × 3] is a reasonable default
```
### Common Reference Anchors
Keep these memorized for quick estimation:
```
POPULATION
World: ~8 billion
US: ~340 million
EU: ~450 million
China: ~1.4 billion
India: ~1.4 billion
ECONOMICS
World GDP: ~$100 trillion
US GDP: ~$28 trillion
US median household: ~$75,000/year
US federal budget: ~$6.5 trillion
S&P 500 total cap: ~$45 trillion
TIME
Seconds in a day: ~86,400 (~10^5)
Seconds in a year: ~31.5 million (~3 × 10^7)
Working hours/year: ~2,000
TECHNOLOGY
Global internet users: ~5.5 billion
Global smartphone users: ~4.5 billion
AWS annual revenue: ~$90 billion
Global IT spending: ~$5 trillion
GitHub developers: ~100 million
INDUSTRY SIZES (annual, global)
Cloud computing: ~$600 billion
Semiconductor: ~$600 billion
Pharmaceutical: ~$1.5 trillion
Automotive: ~$3 trillion
Agriculture: ~$3 trillion
E-commerce: ~$6 trillion
```
### Worked Fermi Examples
**Example A: Estimating the TAM for an AI code review tool**
```
Question: What is the annual TAM for an AI-powered code review SaaS?
Decomposition:
TAM = (Number of professional developers)
× (% who do code reviews regularly)
× (willingness to pay for tooling)
× (average annual price)
Estimates:
Professional developers worldwide: ~30 million
(GitHub has 100M accounts, but ~30% are professional, and
not all professionals use GitHub)
% who do code reviews: ~60%
(Standard in companies > 50 engineers, less common in small shops)
Target market (teams that would buy SaaS): ~40%
(Enterprise and mid-market; small teams use free tools)
Annual price per seat: ~$300/year
(Comparable: GitHub Copilot ~$200, Snyk ~$400, middle ground)
Calculation:
30M × 0.60 × 0.40 × $300 = $2.16 billion
Sanity check:
- GitHub revenue ~$2B (broader product, ~4M paid users)
- Snyk valued at $7B (code security, related space)
- $2B TAM is plausible for a focused code review tool
Result: ~$2 billion TAM (range: $700M to $6B)
```
**Example B: Estimating daily active queries to a search engine**
```
Question: How many search queries does Google process per day?
Decomposition:
Queries/day = (Internet users who use Google)
× (searches per user per day)
Estimates:
Global internet users: ~5.5 billion
Google market share: ~90%
Google users: 5.5B × 0.90 = ~5 billion
But not all use it daily: ~50% daily active rate
Daily active Google searchers: ~2.5 billion
Searches per active user per day: ~3-4
(Some people search 10+ times, many search once or not at all)
Calculation:
2.5 billion × 3.5 = ~8.5 billion queries/day
Sanity check:
Published figure (Google): ~8.5 billion searches/day (2024)
Our estimate nailed it — sometimes Fermi estimation gets lucky.
Result: ~8.5 billion/day (range: 3B to 25B)
```
### Order of Magnitude Sanity Checks
After any estimate, verify it makes sense:
```
CHECK 1: Per-person reasonableness
Divide by relevant population. Is the per-person number realistic?
"$50B market ÷ 340M Americans = $147/person" — plausible?
CHECK 2: Comparison to known quantities
Is your estimate bigger or smaller than things you know?
"Our estimate of X is $3B — that's 5% of AWS revenue. Reasonable?"
CHECK 3: Growth rate implied
If you're estimating a future state, what annual growth rate is implied?
>50% sustained growth for >3 years is extremely rare.
CHECK 4: Upper bound test
What is the theoretical maximum? Is your estimate within it?
"Total possible customers × maximum price = ceiling"
```
---
## Prediction Market Patterns
### Interpreting Market Prices as Probabilities
Prediction market prices map to probabilities, but with important caveats:
```
Market price $0.65 for "Event X occurs"
→ Naive interpretation: 65% probability
→ Adjusted interpretation: depends on market quality
Adjustment factors:
Liquid market (Polymarket, Metaculus with many forecasters):
Price ≈ true probability (±3-5%)
Thin market (<50 traders, <$10K volume):
Price is noisy — treat as ±15% uncertainty
A $0.65 price could represent 50-80% true probability
Binary vs. multi-outcome:
Binary markets are more reliable
Multi-outcome markets often have probabilities summing to >100%
(overround) — normalize before interpreting
```
### Common Prediction Market Biases
| Bias | Description | Impact | Correction |
|------|-------------|--------|------------|
| Favorite-longshot | Favorites underpriced, longshots overpriced | Longshot events appear ~2-3x more likely than they are | If market says 5%, true probability may be 2-3% |
| Recency | Recent events dominate pricing | Probability spikes after news, then slowly reverts | Wait 24-48h after major news before trusting market prices |
| Liquidity premium | Illiquid contracts trade at a discount | Prices biased toward 50% in thin markets | Weight liquid markets more heavily |
| Expiration clustering | Prices converge to 0 or 1 near expiration | Mid-probability contracts vanish near deadline | Most useful signal is months before resolution |
| Hedging distortion | Traders hedging other positions, not expressing beliefs | Prices reflect risk management, not pure probability | Cross-reference with non-market forecasts |
### Aggregation Methods
When combining multiple probability estimates (markets, experts, models):
```
SIMPLE AVERAGE
P = (P1 + P2 + P3) / 3
Use when: Sources are roughly equally credible
Weakness: Susceptible to outliers
MEDIAN
P = middle value of sorted estimates
Use when: One source might be badly miscalibrated
Weakness: Ignores magnitude of disagreement
TRIMMED MEAN
Drop highest and lowest, average the rest
P = average(P2 ... Pn-1) after sorting
Use when: 5+ sources, want outlier robustness
EXTREMIZED AVERAGE
P_avg = simple average
P_extremized = P_avg^a / (P_avg^a + (1-P_avg)^a), where a > 1
Typical a = 1.5 to 2.5 (more extremizing with more independent sources)
Use when: Sources are genuinely independent (not reading each other)
Rationale: If 5 independent sources all say 70%, the true probability
is likely higher than 70% — shared info should push further from 50%
CONFIDENCE-WEIGHTED AVERAGE
P = Σ(wi × Pi) / Σ(wi)
where wi = track record score or source reliability
Use when: Sources have known, differing track records
```
### When Markets Beat Experts (and Vice Versa)
```
MARKETS TEND TO WIN when:
✓ Large, liquid, diverse participant pool
✓ Question is well-defined with clear resolution criteria
✓ Information is widely distributed (no single expert has edge)
✓ Time horizon is 1 month to 2 years
Examples: Election outcomes, product launch dates, economic indicators
EXPERTS TEND TO WIN when:
✓ Question requires deep domain-specific knowledge
✓ Market is thin or participants lack domain context
✓ Very long time horizons (>5 years) — markets discount distant futures
✓ Novel situations with no historical market precedent
Examples: Technical feasibility, scientific breakthroughs, niche regulation
BEST PRACTICE: Use both
Start with the market price, then adjust using expert insight.
Treat the market as the prior and expert analysis as an update.
```
---
## Update Protocol
### Bayesian Updating Worked Example
```
SCENARIO: You predicted 30% chance that Company Z launches Product A in Q1.
New evidence: A leaked internal slide shows a Q1 launch timeline.
STEP 1: State the prior
P(launch in Q1) = 0.30
STEP 2: Assess the evidence
E = leaked slide showing Q1 timeline
How likely is this evidence if the launch IS happening in Q1?
P(E | launch) = 0.85
(Internal slides usually reflect real plans, but plans change)
How likely is this evidence if the launch is NOT in Q1?
P(E | no launch) = 0.15
(Could be outdated slide, aspirational, or decoy)
STEP 3: Calculate the likelihood ratio
LR = P(E | launch) / P(E | no launch) = 0.85 / 0.15 = 5.67
STEP 4: Convert prior to odds, multiply, convert back
Prior odds = 0.30 / 0.70 = 0.429
Posterior odds = 0.429 × 5.67 = 2.43
Posterior probability = 2.43 / (1 + 2.43) = 0.71
STEP 5: State the update
Prior: 30% → Posterior: 71%
Update magnitude: +41 percentage points
This is a LARGE update, appropriate because the evidence (internal
planning document) is strong and directly relevant.
```
### Evidence Strength Classification
How much to update based on different types of evidence:
```
EVIDENCE TIER 1 — Large update (likelihood ratio 5-20x)
→ Official announcement or regulatory filing
→ Confirmed internal document (not rumor)
→ Directly observed outcome of prerequisite event
→ Multiple independent strong sources confirming same fact
Typical update: ±15-30 percentage points
EVIDENCE TIER 2 — Moderate update (likelihood ratio 2-5x)
→ Credible journalist report with named sources
→ Statistical data that changes the base rate
→ Expert with strong track record changing their view
→ Structural/policy change that alters incentives
Typical update: ±5-15 percentage points
EVIDENCE TIER 3 — Small update (likelihood ratio 1.2-2x)
→ Rumor from semi-credible source
→ Anecdotal evidence (single data point)
→ Social media sentiment shift
→ Expert opinion without new information
Typical update: ±2-5 percentage points
EVIDENCE TIER 4 — Negligible update (likelihood ratio ~1x)
→ Repetition of previously known information
→ Pundit opinion with no domain expertise
→ Vague statement open to multiple interpretations
→ Evidence equally consistent with both outcomes
Typical update: ±0-2 percentage points (or skip entirely)
```
### When to Make Large vs. Small Updates
```
MAKE A LARGE UPDATE when:
• Evidence directly addresses your key uncertainty
• The source has a strong track record on this topic
• The evidence would be very surprising if your prediction were correct
(or very unsurprising if it were wrong)
• Multiple independent signals shift in the same direction simultaneously
MAKE A SMALL UPDATE when:
• Evidence is tangentially related to your prediction
• The source's reliability is uncertain
• The evidence is consistent with multiple interpretations
• You've already incorporated similar evidence
RESIST UPDATING when:
• The "evidence" is just someone restating the consensus
• A vivid anecdote feels compelling but carries no statistical weight
• You're reacting emotionally (fear, excitement) rather than analytically
• The evidence source has an obvious incentive to mislead
```
### Common Updating Mistakes
| Mistake | Description | Fix |
|---------|-------------|-----|
| Over-updating on vivid events | A dramatic single event shifts your view by 20+ points when the base rate barely moved | Ask: "Does this event actually change the base rate, or just my emotional state?" |
| Under-updating on base rate changes | New data shows the reference class frequency shifted, but you keep your old anchor | Periodically re-derive the base rate from scratch instead of only adjusting incrementally |
| Asymmetric updating | Updating strongly on confirming evidence, weakly on disconfirming evidence | Force yourself to calculate the likelihood ratio for disconfirming evidence explicitly |
| Double-counting | Updating on a news article, then updating again on a tweet quoting the same article | Track the original source — if two signals share the same root cause, count once |
| Failure to update | Knowing the evidence should change your view but not bothering because your current number "feels right" | Set calendar reminders to review active predictions monthly with fresh evidence |
| Stampede updating | A prediction market spikes, causing you to rush your update to match | Market moves are data, not commands — assess independently, then compare |
---
## Prediction Tracking & Scoring
### Prediction Ledger Format