id = "data-pipeline" name = "Data Analysis & Transform" description = "Clean, transform, analyse, and summarise raw data. Handles messy inputs: CSV, JSON, tables, or free-form text." category = "engineering" tags = ["data", "analysis", "transform", "etl"] [i18n.zh] name = "数据分析与转换" description = "清洗、转换、分析并摘要原始数据,支持 CSV、JSON、表格或自由文本等杂乱输入。" [[parameters]] name = "raw_data" description = "The raw data to process (paste CSV, JSON, table, or any structured text)" param_type = "string" required = true [[parameters]] name = "output_format" description = "Desired output format for the transformed data (json, csv, markdown-table)" param_type = "string" required = false default = "json" [[parameters]] name = "analysis_goal" description = "What you want to understand or extract from the data" param_type = "string" required = false default = "general summary and key insights" [[steps]] name = "profile" prompt_template = """ You are a data engineer. Profile the following raw data: 1. Detected format and structure 2. Number of rows and columns/fields 3. Data types per field 4. Missing value count per field 5. Obvious data quality issues (duplicates, inconsistent formatting, out-of-range values) 6. Sample of first 5 rows for reference Raw data: {{raw_data}} """ [[steps]] name = "clean_transform" prompt_template = """ Clean and transform the raw data based on the profile below. Apply: - Trim whitespace from string fields - Normalise dates to ISO-8601 - Remove exact duplicate rows - Convert empty strings to null - Standardise inconsistent categorical values (e.g. 'Y'/'Yes'/'yes' → true) - Flag but preserve rows with suspicious values (do not silently drop them) Output the cleaned data in {{output_format}} format, followed by a change log listing every transformation applied. Data profile: {{profile}} Raw data: {{raw_data}} """ depends_on = ["profile"] [[steps]] name = "analyse" prompt_template = """ Analyse the cleaned data to address the following goal: {{analysis_goal}} Provide: 1. Key statistics (counts, totals, averages, distributions as appropriate) 2. Notable patterns or trends 3. Outliers or anomalies worth investigating 4. Correlations between fields (if applicable) 5. Top 3 actionable insights from the data Cleaned data: {{clean_transform}} """ depends_on = ["clean_transform"] [i18n.zh-TW] name = "資料分析與轉換" description = "清洗、轉換、分析並摘要原始資料,支援 CSV、JSON、表格或自由文字等雜亂輸入。" [i18n.ja] name = "データ分析と変換" description = "生データをクレンジング・変換・分析・要約。CSV、JSON、表、自由記述など乱雑な入力に対応。" [i18n.ko] name = "데이터 분석 및 변환" description = "원본 데이터를 정제·변환·분석·요약합니다. CSV, JSON, 표, 자유 텍스트 등 지저분한 입력을 처리." [i18n.de] name = "Datenanalyse & Transformation" description = "Rohdaten bereinigen, transformieren, analysieren und zusammenfassen. Verarbeitet unsaubere Eingaben wie CSV, JSON, Tabellen oder Freitext." [i18n.es] name = "Análisis y transformación de datos" description = "Limpia, transforma, analiza y resume datos crudos. Admite entradas desordenadas: CSV, JSON, tablas o texto libre." [i18n.fr] name = "Analyse et transformation de données" description = "Nettoie, transforme, analyse et résume les données brutes. Gère les entrées désordonnées : CSV, JSON, tableaux ou texte libre."