feat: sync content definitions from core repo
Copy all TOML content definitions from librefang core repo: - 33 agent definitions (agents/*/agent.toml) - 14 hand definitions with docs (hands/*/HAND.toml + SKILL.md) - 25 integration templates (integrations/*.toml) - 2 example skill definitions (skills/custom-skill-*) - 1 new provider (providers/vertex-ai.toml) Part of the framework-vs-content registry split (RFC v0.7).
This commit is contained in:
1 parent
ded26ce300
commit
17d32ed4a7
90 files changed
+13549
No files matched your search
@@ -0,0 +1,447 @@
|
||||
id = "analytics"
|
||||
name = "Analytics Hand"
|
||||
description = "Autonomous data analytics agent — data collection, analysis, visualization, dashboards, and automated reporting"
|
||||
category = "data"
|
||||
icon = "📈"
|
||||
tools = ["shell_exec", "file_read", "file_write", "file_list", "web_fetch", "web_search", "memory_store", "memory_recall", "schedule_create", "schedule_list", "schedule_delete", "knowledge_add_entity", "knowledge_add_relation", "knowledge_query", "event_publish"]
|
||||
|
||||
[routing]
|
||||
aliases = ["data analysis", "data visualization", "dashboard", "automated report", "statistical analysis"]
|
||||
weak_aliases = ["visualization", "chart", "histogram", "csv analysis", "excel analysis", "data pipeline", "etl"]
|
||||
|
||||
[[requires]]
|
||||
key = "python3"
|
||||
label = "Python 3"
|
||||
requirement_type = "binary"
|
||||
check_value = "python3"
|
||||
description = "Python 3 interpreter. Required for data analysis with pandas, matplotlib, and seaborn."
|
||||
|
||||
[requires.install]
|
||||
macos = "brew install python3"
|
||||
windows = "winget install Python.Python.3.12"
|
||||
linux = "sudo apt install python3 python3-pip"
|
||||
pip = "python3 --version"
|
||||
|
||||
# ─── Configurable settings ───────────────────────────────────────────────────
|
||||
|
||||
[[settings]]
|
||||
key = "data_source"
|
||||
label = "Data Source"
|
||||
description = "Primary data source type"
|
||||
setting_type = "select"
|
||||
default = "csv"
|
||||
|
||||
[[settings.options]]
|
||||
value = "csv"
|
||||
label = "CSV / Excel files"
|
||||
|
||||
[[settings.options]]
|
||||
value = "json"
|
||||
label = "JSON files / API responses"
|
||||
|
||||
[[settings.options]]
|
||||
value = "database"
|
||||
label = "Database (SQL)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "api"
|
||||
label = "REST API"
|
||||
|
||||
[[settings.options]]
|
||||
value = "web"
|
||||
label = "Web scraping"
|
||||
|
||||
[[settings]]
|
||||
key = "analysis_type"
|
||||
label = "Analysis Type"
|
||||
description = "Default analysis approach"
|
||||
setting_type = "select"
|
||||
default = "descriptive"
|
||||
|
||||
[[settings.options]]
|
||||
value = "descriptive"
|
||||
label = "Descriptive (what happened)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "diagnostic"
|
||||
label = "Diagnostic (why it happened)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "predictive"
|
||||
label = "Predictive (what will happen)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "prescriptive"
|
||||
label = "Prescriptive (what to do about it)"
|
||||
|
||||
[[settings]]
|
||||
key = "output_format"
|
||||
label = "Output Format"
|
||||
description = "How to present analysis results"
|
||||
setting_type = "select"
|
||||
default = "report"
|
||||
|
||||
[[settings.options]]
|
||||
value = "report"
|
||||
label = "Markdown Report"
|
||||
|
||||
[[settings.options]]
|
||||
value = "dashboard"
|
||||
label = "Dashboard (HTML)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "slides"
|
||||
label = "Slide Deck Outline"
|
||||
|
||||
[[settings.options]]
|
||||
value = "executive"
|
||||
label = "Executive Summary"
|
||||
|
||||
[[settings]]
|
||||
key = "visualization"
|
||||
label = "Visualization"
|
||||
description = "Generate charts and visualizations"
|
||||
setting_type = "toggle"
|
||||
default = "true"
|
||||
|
||||
[[settings]]
|
||||
key = "auto_schedule"
|
||||
label = "Scheduled Reports"
|
||||
description = "Automatically generate reports on a schedule"
|
||||
setting_type = "toggle"
|
||||
default = "false"
|
||||
|
||||
[[settings]]
|
||||
key = "report_frequency"
|
||||
label = "Report Frequency"
|
||||
description = "How often to generate scheduled reports"
|
||||
setting_type = "select"
|
||||
default = "weekly"
|
||||
|
||||
[[settings.options]]
|
||||
value = "daily"
|
||||
label = "Daily"
|
||||
|
||||
[[settings.options]]
|
||||
value = "weekly"
|
||||
label = "Weekly"
|
||||
|
||||
[[settings.options]]
|
||||
value = "monthly"
|
||||
label = "Monthly"
|
||||
|
||||
[[settings]]
|
||||
key = "confidence_threshold"
|
||||
label = "Confidence Threshold"
|
||||
description = "Minimum confidence level for including findings in reports"
|
||||
setting_type = "select"
|
||||
default = "medium"
|
||||
|
||||
[[settings.options]]
|
||||
value = "low"
|
||||
label = "Low (include exploratory findings)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "medium"
|
||||
label = "Medium (include likely findings)"
|
||||
|
||||
[[settings.options]]
|
||||
value = "high"
|
||||
label = "High (only statistically significant)"
|
||||
|
||||
# ─── Agent configuration ─────────────────────────────────────────────────────
|
||||
|
||||
[agent]
|
||||
name = "analytics-hand"
|
||||
description = "AI data analyst — collects data, performs statistical analysis, creates visualizations, and generates automated reports with actionable insights"
|
||||
module = "builtin:chat"
|
||||
provider = "default"
|
||||
model = "default"
|
||||
max_tokens = 16384
|
||||
temperature = 0.3
|
||||
max_iterations = 60
|
||||
system_prompt = """You are Analytics Hand — an autonomous data analytics agent that collects data, performs statistical analysis, creates visualizations, and produces automated reports with actionable insights.
|
||||
|
||||
## Phase 0 — Environment Setup (ALWAYS DO THIS FIRST)
|
||||
|
||||
Detect the operating system and available tools:
|
||||
```
|
||||
python -c "import platform; print(platform.system())"
|
||||
python -c "import pandas; print('pandas', pandas.__version__)" 2>/dev/null || echo "pandas not installed"
|
||||
python -c "import matplotlib; print('matplotlib', matplotlib.__version__)" 2>/dev/null || echo "matplotlib not installed"
|
||||
```
|
||||
|
||||
If pandas/matplotlib are missing, install them:
|
||||
```
|
||||
pip install pandas matplotlib seaborn
|
||||
```
|
||||
|
||||
Load context:
|
||||
1. memory_recall `analytics_hand_state` — load previous analysis results and report history
|
||||
2. Read **User Configuration** for data_source, analysis_type, output_format, etc.
|
||||
3. knowledge_query for previously discovered data patterns and insights
|
||||
|
||||
---
|
||||
|
||||
## Phase 1 — Data Ingestion
|
||||
|
||||
Based on the configured `data_source`:
|
||||
|
||||
**CSV/Excel files**:
|
||||
```python
|
||||
import pandas as pd
|
||||
df = pd.read_csv('data.csv')
|
||||
print(df.shape)
|
||||
print(df.dtypes)
|
||||
print(df.describe())
|
||||
```
|
||||
|
||||
**JSON files**:
|
||||
```python
|
||||
import pandas as pd
|
||||
df = pd.read_json('data.json')
|
||||
```
|
||||
|
||||
**REST API**:
|
||||
```
|
||||
curl -s -H "Authorization: Bearer $TOKEN" "$API_URL" -o data.json
|
||||
```
|
||||
Then parse with pandas.
|
||||
|
||||
**Web scraping**:
|
||||
Use web_fetch to retrieve pages, then parse structured data.
|
||||
|
||||
For all sources:
|
||||
1. Load and inspect the data shape (rows, columns, types)
|
||||
2. Check for missing values, duplicates, and outliers
|
||||
3. Document data quality issues
|
||||
4. Store data profile in knowledge graph
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — Data Exploration
|
||||
|
||||
Perform exploratory data analysis (EDA):
|
||||
|
||||
```python
|
||||
import pandas as pd
|
||||
import json
|
||||
|
||||
df = pd.read_csv('data.csv')
|
||||
|
||||
# Basic statistics
|
||||
stats = {
|
||||
'shape': list(df.shape),
|
||||
'columns': list(df.columns),
|
||||
'dtypes': {str(k): str(v) for k, v in df.dtypes.items()},
|
||||
'missing': df.isnull().sum().to_dict(),
|
||||
'describe': df.describe().to_dict()
|
||||
}
|
||||
|
||||
with open('eda_results.json', 'w') as f:
|
||||
json.dump(stats, f, indent=2, default=str)
|
||||
print(json.dumps(stats, indent=2, default=str))
|
||||
```
|
||||
|
||||
Key explorations:
|
||||
1. Distribution of key variables
|
||||
2. Correlations between variables
|
||||
3. Time-series patterns (if temporal data)
|
||||
4. Outlier detection
|
||||
5. Segment analysis (group by categories)
|
||||
|
||||
---
|
||||
|
||||
## Phase 3 — Statistical Analysis
|
||||
|
||||
Based on `analysis_type`:
|
||||
|
||||
**Descriptive**: Summary statistics, frequency distributions, central tendency, variability.
|
||||
|
||||
**Diagnostic**: Correlation analysis, regression, hypothesis testing, root cause analysis.
|
||||
|
||||
**Predictive**: Trend analysis, forecasting, classification patterns.
|
||||
|
||||
**Prescriptive**: Optimization recommendations, scenario analysis, decision support.
|
||||
|
||||
For each analysis:
|
||||
1. State the question being answered
|
||||
2. Check data normality: `scipy.stats.shapiro(data)` — if p > 0.05, data is normal
|
||||
3. Select the appropriate test based on data type and distribution (see SKILL.md decision guide)
|
||||
4. Run the test and report: p-value, effect size (Cohen's d), and sample size
|
||||
5. Apply the `confidence_threshold` setting to filter findings:
|
||||
- **High**: Only include findings with p < 0.01, effect size ≥ 0.5, and n ≥ 100
|
||||
- **Medium**: Include findings with p < 0.05, effect size ≥ 0.3, and n ≥ 30
|
||||
- **Low**: Include all findings with p < 0.10 (exploratory)
|
||||
6. Present results with confidence levels
|
||||
7. Note limitations and caveats
|
||||
|
||||
### Result Validation
|
||||
Before reporting any finding, cross-check:
|
||||
1. **Sanity check**: Does the result make intuitive sense? If not, verify the data and methodology
|
||||
2. **Simpson's paradox**: Could the trend reverse when data is split by a confounding variable?
|
||||
3. **Multiple comparisons**: If you ran 20+ tests, apply Bonferroni correction (divide α by number of tests)
|
||||
4. **Survivorship bias**: Is the dataset missing failed/dropped/churned cases?
|
||||
If any validation fails, downgrade the finding's confidence level by one tier.
|
||||
|
||||
---
|
||||
|
||||
## Phase 4 — Visualization
|
||||
|
||||
If `visualization` is enabled, create charts using Python:
|
||||
|
||||
```python
|
||||
import matplotlib
|
||||
matplotlib.use('Agg')
|
||||
import matplotlib.pyplot as plt
|
||||
import pandas as pd
|
||||
|
||||
df = pd.read_csv('data.csv')
|
||||
|
||||
# Example: bar chart
|
||||
fig, ax = plt.subplots(figsize=(10, 6))
|
||||
df['category'].value_counts().plot(kind='bar', ax=ax)
|
||||
ax.set_title('Distribution by Category')
|
||||
ax.set_xlabel('Category')
|
||||
ax.set_ylabel('Count')
|
||||
plt.tight_layout()
|
||||
plt.savefig('chart_distribution.png', dpi=150)
|
||||
plt.close()
|
||||
print('Chart saved: chart_distribution.png')
|
||||
```
|
||||
|
||||
Chart types to use:
|
||||
- **Bar chart**: Comparisons between categories
|
||||
- **Line chart**: Trends over time
|
||||
- **Scatter plot**: Relationships between variables
|
||||
- **Histogram**: Distribution of a variable
|
||||
- **Heatmap**: Correlation matrix
|
||||
- **Pie chart**: Proportions (use sparingly)
|
||||
- **Box plot**: Distribution and outliers
|
||||
|
||||
Save all charts as PNG files with descriptive names.
|
||||
|
||||
---
|
||||
|
||||
## Phase 5 — Report Generation
|
||||
|
||||
Generate report based on `output_format`:
|
||||
|
||||
**Markdown Report**:
|
||||
```markdown
|
||||
# Analytics Report: [Topic]
|
||||
**Date**: YYYY-MM-DD
|
||||
**Data Source**: [Source description]
|
||||
**Records Analyzed**: N
|
||||
|
||||
## Executive Summary
|
||||
[2-3 key takeaways]
|
||||
|
||||
## Data Overview
|
||||
[Data quality, shape, key characteristics]
|
||||
|
||||
## Key Findings
|
||||
### Finding 1: [Title]
|
||||
[Description with supporting data]
|
||||

|
||||
|
||||
### Finding 2: [Title]
|
||||
[Description with supporting data]
|
||||
|
||||
## Recommendations
|
||||
1. [Actionable recommendation with expected impact]
|
||||
2. [Actionable recommendation with expected impact]
|
||||
|
||||
## Methodology
|
||||
[Analysis approach and tools used]
|
||||
|
||||
## Caveats & Limitations
|
||||
[Data quality issues, confidence levels, assumptions]
|
||||
```
|
||||
|
||||
**Executive Summary**: 1-page brief with key metrics and recommendations.
|
||||
**Dashboard**: HTML file with embedded charts and interactive elements.
|
||||
**Slide Deck Outline**: Key points per slide with chart references.
|
||||
|
||||
Save report to: `analytics_report_YYYY-MM-DD.md`
|
||||
|
||||
### Analysis Exit Criteria
|
||||
Stop the current analysis when ANY of these conditions is met:
|
||||
1. **Data quality too low**: >50% missing values or >30% outliers — report data quality issues, do NOT draw conclusions
|
||||
2. **Sample too small**: n < 10 for any key analysis — flag as "insufficient data" and recommend data collection
|
||||
3. **No significant findings**: All tests return p > 0.10 — report "no statistically significant patterns found" (this IS a valid result)
|
||||
4. **Iteration cap**: 10+ analysis iterations on the same dataset — summarize current findings and stop
|
||||
5. **Compute timeout**: Any single Python script runs >5 minutes — kill it, simplify the analysis approach
|
||||
|
||||
---
|
||||
|
||||
## Phase 6 — Scheduled Reporting
|
||||
|
||||
If `auto_schedule` is enabled:
|
||||
1. Create schedules using schedule_create based on `report_frequency`
|
||||
2. On each scheduled run:
|
||||
- Re-ingest data from configured source
|
||||
- Compare with previous period
|
||||
- Highlight changes and trends
|
||||
- Generate and save updated report
|
||||
3. event_publish "analytics_report_ready" with report path
|
||||
|
||||
---
|
||||
|
||||
## Phase 7 — State Persistence
|
||||
|
||||
1. memory_store `analytics_hand_state`: analyses_run, reports_generated, data_sources_profiled
|
||||
2. Update dashboard stats:
|
||||
- memory_store `analytics_hand_analyses_run` — total analyses executed
|
||||
- memory_store `analytics_hand_reports_generated` — total reports created
|
||||
- memory_store `analytics_hand_data_points_processed` — total data points analyzed
|
||||
- memory_store `analytics_hand_active_schedules` — active scheduled reports
|
||||
|
||||
---
|
||||
|
||||
## Guidelines
|
||||
|
||||
- ALWAYS verify data quality before drawing conclusions
|
||||
- NEVER fabricate data, statistics, or analysis results
|
||||
- NEVER present correlation as causation without additional evidence
|
||||
- Clearly state confidence levels for all findings
|
||||
- Flag sample size limitations and selection bias
|
||||
- Use appropriate statistical tests for the data type
|
||||
- Preserve raw data — never modify source files
|
||||
- Document all data transformations and assumptions
|
||||
- When results are inconclusive, say so clearly
|
||||
- Respect data privacy — redact PII in reports
|
||||
"""
|
||||
|
||||
[dashboard]
|
||||
[[dashboard.metrics]]
|
||||
label = "Analyses Run"
|
||||
memory_key = "analytics_hand_analyses_run"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Reports Generated"
|
||||
memory_key = "analytics_hand_reports_generated"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Data Points Processed"
|
||||
memory_key = "analytics_hand_data_points_processed"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Active Schedules"
|
||||
memory_key = "analytics_hand_active_schedules"
|
||||
format = "number"
|
||||
|
||||
[[dashboard.metrics]]
|
||||
label = "Findings Reported"
|
||||
memory_key = "analytics_hand_findings_reported"
|
||||
format = "number"
|
||||
|
||||
# ─── Token & Performance Metadata ─────────────────────────────────────────────
|
||||
[metadata]
|
||||
frequency = "continuous"
|
||||
token_consumption = "high"
|
||||
default_active = true
|
||||
# Note: High consumption when actively analyzing data, lower when idle
|
||||
@@ -0,0 +1,339 @@
|
||||
---
|
||||
name: analytics-hand-skill
|
||||
version: "1.0.0"
|
||||
description: "Expert knowledge for AI data analytics -- statistical methods, visualization best practices, pandas reference, and reporting patterns"
|
||||
runtime: prompt_only
|
||||
---
|
||||
|
||||
# Data Analytics Expert Knowledge
|
||||
|
||||
## pandas Quick Reference
|
||||
|
||||
### Data Loading
|
||||
```python
|
||||
import pandas as pd
|
||||
|
||||
# CSV
|
||||
df = pd.read_csv('data.csv')
|
||||
df = pd.read_csv('data.csv', parse_dates=['date_col'], index_col='id')
|
||||
|
||||
# JSON
|
||||
df = pd.read_json('data.json')
|
||||
df = pd.read_json('data.json', orient='records')
|
||||
|
||||
# Excel
|
||||
df = pd.read_excel('data.xlsx', sheet_name='Sheet1')
|
||||
|
||||
# From dict
|
||||
df = pd.DataFrame({'col1': [1, 2, 3], 'col2': ['a', 'b', 'c']})
|
||||
```
|
||||
|
||||
### Data Inspection
|
||||
```python
|
||||
df.shape # (rows, columns)
|
||||
df.dtypes # Column types
|
||||
df.info() # Summary including memory usage
|
||||
df.describe() # Statistical summary
|
||||
df.head(10) # First 10 rows
|
||||
df.isnull().sum() # Missing values per column
|
||||
df.duplicated().sum() # Number of duplicate rows
|
||||
df.nunique() # Unique values per column
|
||||
```
|
||||
|
||||
### Data Cleaning
|
||||
```python
|
||||
# Handle missing values
|
||||
df.dropna() # Drop rows with any NaN
|
||||
df.fillna(0) # Fill NaN with 0
|
||||
df.fillna(df.mean()) # Fill with column means
|
||||
df['col'].interpolate() # Interpolate missing values
|
||||
|
||||
# Remove duplicates
|
||||
df.drop_duplicates()
|
||||
df.drop_duplicates(subset=['col1', 'col2'])
|
||||
|
||||
# Type conversion
|
||||
df['col'] = df['col'].astype(int)
|
||||
df['date'] = pd.to_datetime(df['date'])
|
||||
df['cat'] = df['cat'].astype('category')
|
||||
|
||||
# Outlier removal (IQR method)
|
||||
Q1 = df['col'].quantile(0.25)
|
||||
Q3 = df['col'].quantile(0.75)
|
||||
IQR = Q3 - Q1
|
||||
df = df[(df['col'] >= Q1 - 1.5*IQR) & (df['col'] <= Q3 + 1.5*IQR)]
|
||||
```
|
||||
|
||||
### Aggregation & Grouping
|
||||
```python
|
||||
# Group by
|
||||
df.groupby('category').agg({'value': ['mean', 'sum', 'count']})
|
||||
|
||||
# Pivot table
|
||||
pd.pivot_table(df, values='value', index='row_cat', columns='col_cat', aggfunc='mean')
|
||||
|
||||
# Cross tabulation
|
||||
pd.crosstab(df['cat1'], df['cat2'])
|
||||
|
||||
# Rolling statistics
|
||||
df['rolling_mean'] = df['value'].rolling(window=7).mean()
|
||||
|
||||
# Percentage change
|
||||
df['pct_change'] = df['value'].pct_change()
|
||||
```
|
||||
|
||||
### Time Series
|
||||
```python
|
||||
# Set datetime index
|
||||
df.set_index('date', inplace=True)
|
||||
|
||||
# Resample
|
||||
df.resample('W').mean() # Weekly average
|
||||
df.resample('M').sum() # Monthly sum
|
||||
df.resample('Q').count() # Quarterly count
|
||||
|
||||
# Date range
|
||||
pd.date_range(start='2025-01-01', periods=30, freq='D')
|
||||
|
||||
# Shift/Lag
|
||||
df['prev_value'] = df['value'].shift(1)
|
||||
df['next_value'] = df['value'].shift(-1)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Visualization Best Practices
|
||||
|
||||
### matplotlib + seaborn Reference
|
||||
|
||||
```python
|
||||
import matplotlib
|
||||
matplotlib.use('Agg') # Non-interactive backend
|
||||
import matplotlib.pyplot as plt
|
||||
import seaborn as sns
|
||||
|
||||
# Set style
|
||||
sns.set_theme(style='whitegrid')
|
||||
plt.rcParams['figure.figsize'] = (10, 6)
|
||||
```
|
||||
|
||||
### Chart Selection Guide
|
||||
|
||||
| Data Type | Question | Chart Type |
|
||||
|-----------|----------|------------|
|
||||
| Categorical | Comparison | Bar chart |
|
||||
| Categorical | Proportion | Pie chart (if <6 categories) |
|
||||
| Numerical | Distribution | Histogram / Box plot |
|
||||
| Two numerical | Relationship | Scatter plot |
|
||||
| Time series | Trend | Line chart |
|
||||
| Matrix | Correlation | Heatmap |
|
||||
| Categories + values | Comparison | Grouped bar / Stacked bar |
|
||||
| Geographical | Location | Map / Choropleth |
|
||||
|
||||
### Chart Templates
|
||||
|
||||
**Bar Chart**:
|
||||
```python
|
||||
fig, ax = plt.subplots(figsize=(10, 6))
|
||||
data = df['category'].value_counts()
|
||||
data.plot(kind='bar', ax=ax, color='steelblue')
|
||||
ax.set_title('Distribution by Category', fontsize=14, fontweight='bold')
|
||||
ax.set_xlabel('Category')
|
||||
ax.set_ylabel('Count')
|
||||
plt.xticks(rotation=45, ha='right')
|
||||
plt.tight_layout()
|
||||
plt.savefig('bar_chart.png', dpi=150, bbox_inches='tight')
|
||||
plt.close()
|
||||
```
|
||||
|
||||
**Line Chart (Time Series)**:
|
||||
```python
|
||||
fig, ax = plt.subplots(figsize=(12, 6))
|
||||
ax.plot(df.index, df['value'], linewidth=2, color='steelblue')
|
||||
ax.fill_between(df.index, df['value'], alpha=0.1, color='steelblue')
|
||||
ax.set_title('Trend Over Time', fontsize=14, fontweight='bold')
|
||||
ax.set_xlabel('Date')
|
||||
ax.set_ylabel('Value')
|
||||
plt.tight_layout()
|
||||
plt.savefig('line_chart.png', dpi=150, bbox_inches='tight')
|
||||
plt.close()
|
||||
```
|
||||
|
||||
**Correlation Heatmap**:
|
||||
```python
|
||||
fig, ax = plt.subplots(figsize=(10, 8))
|
||||
corr = df.select_dtypes(include='number').corr()
|
||||
sns.heatmap(corr, annot=True, fmt='.2f', cmap='RdBu_r', center=0, ax=ax)
|
||||
ax.set_title('Correlation Matrix', fontsize=14, fontweight='bold')
|
||||
plt.tight_layout()
|
||||
plt.savefig('heatmap.png', dpi=150, bbox_inches='tight')
|
||||
plt.close()
|
||||
```
|
||||
|
||||
**Scatter Plot**:
|
||||
```python
|
||||
fig, ax = plt.subplots(figsize=(10, 6))
|
||||
ax.scatter(df['x'], df['y'], alpha=0.6, edgecolors='black', linewidth=0.5)
|
||||
ax.set_title('X vs Y', fontsize=14, fontweight='bold')
|
||||
ax.set_xlabel('X Variable')
|
||||
ax.set_ylabel('Y Variable')
|
||||
plt.tight_layout()
|
||||
plt.savefig('scatter.png', dpi=150, bbox_inches='tight')
|
||||
plt.close()
|
||||
```
|
||||
|
||||
### Visualization Do's and Don'ts
|
||||
|
||||
**Do**:
|
||||
- Start y-axis at 0 for bar charts
|
||||
- Use consistent colors across related charts
|
||||
- Label axes clearly with units
|
||||
- Add titles that describe the insight, not just the data
|
||||
- Use appropriate scales (log scale for exponential data)
|
||||
|
||||
**Don't**:
|
||||
- Use 3D charts (distorts perception)
|
||||
- Use more than 6-7 colors in one chart
|
||||
- Truncate axes to exaggerate differences
|
||||
- Use pie charts for more than 5 categories
|
||||
- Add unnecessary chart junk (borders, backgrounds, grids)
|
||||
|
||||
---
|
||||
|
||||
## Statistical Methods
|
||||
|
||||
### Descriptive Statistics
|
||||
| Measure | pandas | Purpose |
|
||||
|---------|--------|---------|
|
||||
| Mean | `df['col'].mean()` | Central tendency |
|
||||
| Median | `df['col'].median()` | Robust central tendency |
|
||||
| Std Dev | `df['col'].std()` | Variability |
|
||||
| Skewness | `df['col'].skew()` | Distribution symmetry |
|
||||
| Kurtosis | `df['col'].kurtosis()` | Distribution tails |
|
||||
| Percentiles | `df['col'].quantile([0.25, 0.5, 0.75])` | Distribution spread |
|
||||
|
||||
### Correlation Analysis
|
||||
```python
|
||||
# Pearson correlation (linear)
|
||||
df['col1'].corr(df['col2'])
|
||||
|
||||
# Spearman correlation (monotonic)
|
||||
df['col1'].corr(df['col2'], method='spearman')
|
||||
|
||||
# Full correlation matrix
|
||||
df.select_dtypes(include='number').corr()
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
- |r| > 0.7: Strong correlation
|
||||
- 0.4 < |r| < 0.7: Moderate correlation
|
||||
- |r| < 0.4: Weak correlation
|
||||
- Correlation != Causation
|
||||
|
||||
### Hypothesis Testing (scipy)
|
||||
```python
|
||||
from scipy import stats
|
||||
|
||||
# T-test (compare two group means)
|
||||
t_stat, p_value = stats.ttest_ind(group1, group2)
|
||||
|
||||
# Chi-squared test (categorical independence)
|
||||
chi2, p_value, dof, expected = stats.chi2_contingency(contingency_table)
|
||||
|
||||
# Significance: p < 0.05 is commonly used threshold
|
||||
|
||||
# Mann-Whitney U test (non-parametric alternative to t-test)
|
||||
u_stat, p_value = stats.mannwhitneyu(group1, group2, alternative='two-sided')
|
||||
|
||||
# One-way ANOVA (compare 3+ group means)
|
||||
f_stat, p_value = stats.f_oneway(group1, group2, group3)
|
||||
|
||||
# Normality check (determines which test to use)
|
||||
shapiro_stat, p_value = stats.shapiro(data) # p > 0.05 means normal
|
||||
```
|
||||
|
||||
### Statistical Significance Decision Guide
|
||||
|
||||
**Test selection flowchart:**
|
||||
| Data Situation | Normal Distribution? | Test to Use |
|
||||
|---------------|---------------------|-------------|
|
||||
| Compare 2 group means | Yes | Independent t-test (`ttest_ind`) |
|
||||
| Compare 2 group means | No | Mann-Whitney U (`mannwhitneyu`) |
|
||||
| Compare 3+ group means | Yes | One-way ANOVA (`f_oneway`) |
|
||||
| Compare 3+ group means | No | Kruskal-Wallis (`kruskal`) |
|
||||
| Compare paired samples | Yes | Paired t-test (`ttest_rel`) |
|
||||
| Compare paired samples | No | Wilcoxon signed-rank (`wilcoxon`) |
|
||||
| Test categorical independence | N/A | Chi-squared (`chi2_contingency`) |
|
||||
| Test correlation | Yes | Pearson (`pearsonr`) |
|
||||
| Test correlation | No | Spearman (`spearmanr`) |
|
||||
|
||||
**P-value interpretation:**
|
||||
| p-value | Interpretation | Action |
|
||||
|---------|---------------|--------|
|
||||
| p < 0.01 | Strong evidence against null hypothesis | Report as statistically significant |
|
||||
| 0.01 ≤ p < 0.05 | Moderate evidence | Report as significant with caveat |
|
||||
| 0.05 ≤ p < 0.10 | Weak evidence | Report as marginally significant |
|
||||
| p ≥ 0.10 | Insufficient evidence | Do not claim significance |
|
||||
|
||||
**Practical significance — always report effect size:**
|
||||
```python
|
||||
# Cohen's d for comparing two means
|
||||
def cohens_d(group1, group2):
|
||||
n1, n2 = len(group1), len(group2)
|
||||
var1, var2 = group1.var(), group2.var()
|
||||
pooled_std = ((n1 - 1) * var1 + (n2 - 1) * var2) / (n1 + n2 - 2)
|
||||
return (group1.mean() - group2.mean()) / (pooled_std ** 0.5)
|
||||
|
||||
# Interpretation: |d| < 0.2 = negligible, 0.2-0.5 = small, 0.5-0.8 = medium, > 0.8 = large
|
||||
```
|
||||
|
||||
**Sample size awareness:**
|
||||
- n < 30: Use non-parametric tests; results are exploratory
|
||||
- 30 ≤ n < 100: Parametric tests OK if normality holds; moderate confidence
|
||||
- n ≥ 100: Central Limit Theorem applies; high confidence in parametric tests
|
||||
- Always report sample size alongside p-values
|
||||
|
||||
**Confidence threshold mapping:**
|
||||
| Setting | p-value threshold | Minimum effect size | Minimum sample size |
|
||||
|---------|------------------|--------------------|--------------------|
|
||||
| High | p < 0.01 | Cohen's d ≥ 0.5 | n ≥ 100 |
|
||||
| Medium | p < 0.05 | Cohen's d ≥ 0.3 | n ≥ 30 |
|
||||
| Low | p < 0.10 | Any | Any |
|
||||
|
||||
---
|
||||
|
||||
## Report Structure Best Practices
|
||||
|
||||
### CRISP-DM Framework
|
||||
1. **Business Understanding**: What question are we answering?
|
||||
2. **Data Understanding**: What data do we have? Quality?
|
||||
3. **Data Preparation**: Cleaning, transformation, feature engineering
|
||||
4. **Modeling**: Statistical analysis, ML models
|
||||
5. **Evaluation**: Are results valid and useful?
|
||||
6. **Deployment**: Reports, dashboards, recommendations
|
||||
|
||||
### Insight Hierarchy
|
||||
```
|
||||
Level 1: What happened (descriptive)
|
||||
"Revenue increased 15% last quarter"
|
||||
|
||||
Level 2: Why it happened (diagnostic)
|
||||
"Revenue increase driven by 30% growth in enterprise segment"
|
||||
|
||||
Level 3: What will happen (predictive)
|
||||
"Based on current trends, Q2 revenue projected at $X"
|
||||
|
||||
Level 4: What to do (prescriptive)
|
||||
"Invest in enterprise sales team to capitalize on growth trajectory"
|
||||
```
|
||||
|
||||
### Data Quality Assessment Template
|
||||
```
|
||||
| Dimension | Score | Details |
|
||||
|-----------|-------|---------|
|
||||
| Completeness | 85% | 15% missing values in 'email' column |
|
||||
| Accuracy | High | Validated against source system |
|
||||
| Consistency | Medium | Date formats vary across sources |
|
||||
| Timeliness | Current | Data refreshed daily |
|
||||
| Uniqueness | 99% | 1% duplicate records found |
|
||||
```
|
||||
Reference in new issue
Block a user