id = "data-pipeline" name = "Data Analysis & Transform" description = "Clean, transform, analyse, and summarise raw data. Handles messy inputs: CSV, JSON, tables, or free-form text." category = "engineering" tags = ["data", "analysis", "transform", "etl"] [i18n.zh] name = "数据分析与转换" description = "清洗、转换、分析并摘要原始数据,支持 CSV、JSON、表格或自由文本等杂乱输入。" [[parameters]] name = "raw_data" description = "The raw data to process (paste CSV, JSON, table, or any structured text)" param_type = "string" required = true [[parameters]] name = "output_format" description = "Desired output format for the transformed data (json, csv, markdown-table)" param_type = "string" required = false default = "json" [[parameters]] name = "analysis_goal" description = "What you want to understand or extract from the data" param_type = "string" required = false default = "general summary and key insights" [[steps]] name = "profile" prompt_template = """ You are a data engineer. Profile the following raw data: 1. Detected format and structure 2. Number of rows and columns/fields 3. Data types per field 4. Missing value count per field 5. Obvious data quality issues (duplicates, inconsistent formatting, out-of-range values) 6. Sample of first 5 rows for reference Raw data: {{raw_data}} """ [[steps]] name = "clean_transform" prompt_template = """ Clean and transform the raw data based on the profile below. Apply: - Trim whitespace from string fields - Normalise dates to ISO-8601 - Remove exact duplicate rows - Convert empty strings to null - Standardise inconsistent categorical values (e.g. 'Y'/'Yes'/'yes' → true) - Flag but preserve rows with suspicious values (do not silently drop them) Output the cleaned data in {{output_format}} format, followed by a change log listing every transformation applied. Data profile: {{profile}} Raw data: {{raw_data}} """ depends_on = ["profile"] [[steps]] name = "analyse" prompt_template = """ Analyse the cleaned data to address the following goal: {{analysis_goal}} Provide: 1. Key statistics (counts, totals, averages, distributions as appropriate) 2. Notable patterns or trends 3. Outliers or anomalies worth investigating 4. Correlations between fields (if applicable) 5. Top 3 actionable insights from the data Cleaned data: {{clean_transform}} """ depends_on = ["clean_transform"]