#!/usr/bin/env python3 """Validate all TOML files in the LibreFang registry. Usage: python scripts/validate.py python scripts/validate.py --strict python scripts/validate.py --type providers python scripts/validate.py --strict --type agents Validates: - Provider TOML files (required fields, valid tiers, non-negative costs, no duplicates) - Agent TOML files (required fields: name, description, module) - Hand TOML files (required fields: id, name, description, category) - Integration TOML files (required fields: id, name, [transport]) - Skill TOML files (required fields: [skill].name, [runtime].type) - aliases.toml parsing - Cross-reference: hand [[requires]] integration references - Routing alias collision detection (WARNING, ERROR with --strict) Exit code 0 on success, 1 on any validation error. """ import argparse import sys from pathlib import Path try: import tomllib # Python 3.11+ except ImportError: try: import tomli as tomllib # pip install tomli except ImportError: print("ERROR: Python 3.11+ required, or install 'tomli': pip install tomli") sys.exit(1) VALID_TIERS = {"frontier", "smart", "balanced", "fast", "local"} VALID_MODALITIES = {"text", "image", "audio", "video", "music"} VALID_HAND_CATEGORIES = { "communication", "content", "data", "development", "devops", "finance", "productivity", "research", "social", } VALID_SKILL_RUNTIMES = {"promptonly", "python", "node", "shell"} REQUIRED_PROVIDER_FIELDS = {"id", "display_name", "api_key_env", "base_url", "key_required"} # Required for all modalities. REQUIRED_MODEL_FIELDS_COMMON = { "id", "display_name", "tier", "input_cost_per_m", "output_cost_per_m", } # Additionally required when modality='text' (the default). REQUIRED_MODEL_FIELDS_TEXT_ONLY = {"context_window", "max_output_tokens"} VALID_CONTENT_TYPES = {"providers", "agents", "hands", "mcp", "skills", "plugins", "aliases"} def load_toml(filepath: Path) -> tuple[dict | None, str | None]: """Load a TOML file, return (data, error).""" try: with open(filepath, "rb") as f: return tomllib.load(f), None except Exception as e: return None, f"{filepath}: Failed to parse TOML: {e}" def validate_provider_file(filepath: Path) -> list[str]: """Validate a single provider TOML file.""" data, err = load_toml(filepath) if err: return [err] errors = [] provider = data.get("provider") if provider is None: errors.append(f"{filepath.name}: Missing [provider] section") else: for field in REQUIRED_PROVIDER_FIELDS: if field not in provider: errors.append(f"{filepath.name}: Missing provider field '{field}'") models = data.get("models", []) seen_ids = set() for i, model in enumerate(models): label = model.get("id", f"models[{i}]") modality = model.get("modality", "text") if modality not in VALID_MODALITIES: errors.append( f"{filepath.name}: Model '{label}' invalid modality '{modality}' " f"(valid: {', '.join(sorted(VALID_MODALITIES))})" ) required_fields = set(REQUIRED_MODEL_FIELDS_COMMON) if modality == "text": required_fields |= REQUIRED_MODEL_FIELDS_TEXT_ONLY for field in required_fields: if field not in model: errors.append(f"{filepath.name}: Model '{label}' missing field '{field}'") tier = model.get("tier") if tier is not None and tier not in VALID_TIERS: errors.append(f"{filepath.name}: Model '{label}' invalid tier '{tier}'") for cost_field in ( "input_cost_per_m", "output_cost_per_m", "image_input_cost_per_m", "image_output_cost_per_m", ): val = model.get(cost_field) if val is not None and (not isinstance(val, (int, float)) or val < 0): errors.append(f"{filepath.name}: Model '{label}' invalid {cost_field}: {val}") for int_field in ("context_window", "max_output_tokens"): val = model.get(int_field) if val is not None and (not isinstance(val, int) or val <= 0): errors.append(f"{filepath.name}: Model '{label}' invalid {int_field}: {val}") model_id = model.get("id") if model_id: if model_id in seen_ids: errors.append(f"{filepath.name}: Duplicate model ID '{model_id}'") seen_ids.add(model_id) return errors def validate_agent_file(filepath: Path) -> list[str]: """Validate an agent.toml file.""" data, err = load_toml(filepath) if err: return [err] errors = [] dir_name = filepath.parent.name rel = filepath.relative_to(filepath.parent.parent) for field in ("name", "description", "module"): if field not in data: errors.append(f"{rel}: Missing required field '{field}'") # name should match directory name name = data.get("name") if name and name != dir_name: errors.append(f"{rel}: name '{name}' does not match directory '{dir_name}'") if "model" in data: model = data["model"] if "system_prompt" not in model and "provider" not in model: errors.append(f"{rel}: [model] section should have 'provider' or 'system_prompt'") return errors def validate_hand_file(filepath: Path, integration_ids: set[str]) -> list[str]: """Validate a HAND.toml file. Args: filepath: Path to the HAND.toml file. integration_ids: Set of known MCP server IDs (filenames without .toml from the mcp/ directory) for cross-reference validation. """ data, err = load_toml(filepath) if err: return [err] errors = [] dir_name = filepath.parent.name rel = filepath.relative_to(filepath.parent.parent.parent) for field in ("id", "name", "description"): if field not in data: errors.append(f"{rel}: Missing required field '{field}'") # id should match directory name hand_id = data.get("id") if hand_id and hand_id != dir_name: errors.append(f"{rel}: id '{hand_id}' does not match directory '{dir_name}'") category = data.get("category") if category and category not in VALID_HAND_CATEGORIES: errors.append(f"{rel}: Invalid category '{category}' (valid: {', '.join(sorted(VALID_HAND_CATEGORIES))})") if "agents" not in data: errors.append(f"{rel}: Missing [agents] section") # Cross-reference: check that [[requires]] with requirement_type = "integration" # reference existing integration TOML files requires_list = data.get("requires", []) if isinstance(requires_list, list): for req in requires_list: if isinstance(req, dict) and req.get("requirement_type") == "integration": req_key = req.get("key", "") if req_key and req_key not in integration_ids: errors.append( f"{rel}: [[requires]] references integration '{req_key}' " f"but no mcp/{req_key}.toml exists" ) return errors def validate_integration_file(filepath: Path) -> list[str]: """Validate an integration TOML file.""" data, err = load_toml(filepath) if err: return [err] errors = [] expected_id = filepath.stem # filename without .toml for field in ("id", "name"): if field not in data: errors.append(f"{filepath.name}: Missing required field '{field}'") # id should match filename int_id = data.get("id") if int_id and int_id != expected_id: errors.append(f"{filepath.name}: id '{int_id}' does not match filename '{expected_id}'") if "transport" not in data: errors.append(f"{filepath.name}: Missing [transport] section") else: transport = data["transport"] if "type" not in transport: errors.append(f"{filepath.name}: Missing transport.type") if "command" not in transport: errors.append(f"{filepath.name}: Missing transport.command") return errors def validate_skill_file(filepath: Path) -> list[str]: """Validate a skill.toml file.""" data, err = load_toml(filepath) if err: return [err] errors = [] rel = filepath.relative_to(filepath.parent.parent) skill = data.get("skill") if skill is None: errors.append(f"{rel}: Missing [skill] section") else: if "name" not in skill: errors.append(f"{rel}: Missing skill.name") runtime = data.get("runtime") if runtime is None: errors.append(f"{rel}: Missing [runtime] section") else: rt = runtime.get("type") if rt and rt not in VALID_SKILL_RUNTIMES: errors.append(f"{rel}: Invalid runtime type '{rt}' (valid: {', '.join(sorted(VALID_SKILL_RUNTIMES))})") return errors def parse_skill_md_frontmatter(filepath: Path) -> tuple[dict[str, str], list[str]]: """Parse YAML frontmatter from a SKILL.md file. Returns (metadata, errors). Metadata may be empty if parsing fails. """ errors: list[str] = [] rel = filepath.relative_to(filepath.parent.parent) try: text = filepath.read_text(encoding="utf-8") except OSError as e: return {}, [f"{rel}: Failed to read file: {e}"] if not text.startswith("---"): return {}, [f"{rel}: Missing YAML frontmatter (must start with '---')"] end = text.find("\n---", 3) if end == -1: return {}, [f"{rel}: Unterminated YAML frontmatter (missing closing '---')"] fm = text[3:end].strip() meta: dict[str, str] = {} for line in fm.splitlines(): line = line.strip() if not line or line.startswith("#"): continue if ":" not in line: continue k, _, v = line.partition(":") meta[k.strip()] = v.strip().strip('"').strip("'") return meta, errors def validate_skill_md_file(filepath: Path) -> tuple[dict[str, str], list[str]]: """Validate a Claude Code-style SKILL.md file with YAML frontmatter. Returns (metadata, errors) so callers can cross-check against skill.toml. """ meta, errors = parse_skill_md_frontmatter(filepath) rel = filepath.relative_to(filepath.parent.parent) if errors: return meta, errors if not meta.get("name"): errors.append(f"{rel}: Missing frontmatter field 'name'") if not meta.get("description"): errors.append(f"{rel}: Missing frontmatter field 'description'") return meta, errors def validate_aliases_file(filepath: Path) -> list[str]: """Validate aliases.toml.""" if not filepath.exists(): return [f"aliases.toml not found at {filepath}"] data, err = load_toml(filepath) if err: return [err] errors = [] aliases = data.get("aliases", {}) if not isinstance(aliases, dict): errors.append("aliases.toml: [aliases] must be a table of string -> string mappings") else: for alias, target in aliases.items(): if not isinstance(target, str): errors.append(f"aliases.toml: Alias '{alias}' must map to a string, got {type(target).__name__}") return errors def parse_args() -> argparse.Namespace: """Parse command-line arguments.""" parser = argparse.ArgumentParser( description="Validate all TOML files in the LibreFang registry." ) parser.add_argument( "--strict", action="store_true", help="Treat all warnings as errors.", ) parser.add_argument( "--type", choices=sorted(VALID_CONTENT_TYPES), dest="content_type", default=None, help="Validate only the specified content type (e.g. providers, agents, hands).", ) return parser.parse_args() def should_validate(content_type: str, filter_type: str | None) -> bool: """Check whether a content type should be validated given the --type filter.""" if filter_type is None: return True return content_type == filter_type def main(): args = parse_args() script_dir = Path(__file__).resolve().parent root = script_dir.parent all_errors = [] stats = {} # --- Collect MCP server IDs for cross-reference validation --- mcp_dir = root / "mcp" integration_ids: set[str] = set() if mcp_dir.is_dir(): for fp in mcp_dir.glob("*.toml"): integration_ids.add(fp.stem) # --- Providers --- providers_dir = root / "providers" if providers_dir.is_dir() and should_validate("providers", args.content_type): toml_files = sorted(providers_dir.glob("*.toml")) total_models = 0 global_model_ids = {} for fp in toml_files: all_errors.extend(validate_provider_file(fp)) data, _ = load_toml(fp) if data: models = data.get("models", []) total_models += len(models) provider_id = data.get("provider", {}).get("id", fp.stem) for m in models: mid = m.get("id") if mid: key = (mid, provider_id) if key in global_model_ids: prev = global_model_ids[key] if prev != fp.name: all_errors.append( f"Cross-file duplicate: model '{mid}' (provider '{provider_id}') " f"in both {prev} and {fp.name}" ) else: global_model_ids[key] = fp.name stats["providers"] = len(toml_files) stats["models"] = total_models # --- Agents --- agents_dir = root / "agents" if agents_dir.is_dir() and should_validate("agents", args.content_type): agent_dirs = sorted([d for d in agents_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) for d in agent_dirs: agent_toml = d / "agent.toml" if agent_toml.exists(): all_errors.extend(validate_agent_file(agent_toml)) else: all_errors.append(f"agents/{d.name}: Missing agent.toml") stats["agents"] = len(agent_dirs) # --- Hands --- hands_dir = root / "hands" if hands_dir.is_dir() and should_validate("hands", args.content_type): hand_dirs = sorted([d for d in hands_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) for d in hand_dirs: hand_toml = d / "HAND.toml" if hand_toml.exists(): all_errors.extend(validate_hand_file(hand_toml, integration_ids)) else: all_errors.append(f"hands/{d.name}: Missing HAND.toml") stats["hands"] = len(hand_dirs) # --- MCP Servers --- if mcp_dir.is_dir() and should_validate("mcp", args.content_type): int_files = sorted(mcp_dir.glob("*.toml")) for fp in int_files: all_errors.extend(validate_integration_file(fp)) stats["mcp"] = len(int_files) # --- Skills --- skills_dir = root / "skills" if skills_dir.is_dir() and should_validate("skills", args.content_type): skill_dirs = sorted([d for d in skills_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) for d in skill_dirs: skill_toml = d / "skill.toml" skill_md = d / "SKILL.md" if not skill_md.exists(): all_errors.append(f"skills/{d.name}: Missing SKILL.md (required entry point)") continue md_meta, md_errors = validate_skill_md_file(skill_md) all_errors.extend(md_errors) if skill_toml.exists(): all_errors.extend(validate_skill_file(skill_toml)) toml_data, _ = load_toml(skill_toml) if toml_data: toml_skill = toml_data.get("skill", {}) or {} for field in ("name", "description"): md_val = md_meta.get(field) toml_val = toml_skill.get(field) if md_val and toml_val and md_val != toml_val: all_errors.append( f"skills/{d.name}: {field} mismatch between SKILL.md " f"('{md_val}') and skill.toml ('{toml_val}')" ) stats["skills"] = len(skill_dirs) # --- Plugins --- plugins_dir = root / "plugins" if plugins_dir.is_dir() and should_validate("plugins", args.content_type): plugin_dirs = sorted([d for d in plugins_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) for d in plugin_dirs: plugin_toml = d / "plugin.toml" if plugin_toml.exists(): data, err = load_toml(plugin_toml) if err: all_errors.append(err) elif data: if "name" not in data: all_errors.append(f"plugins/{d.name}: Missing required field 'name'") elif data["name"] != d.name: all_errors.append(f"plugins/{d.name}: name '{data['name']}' does not match directory '{d.name}'") if "hooks" not in data: all_errors.append(f"plugins/{d.name}: Missing [hooks] section") else: for hook_name, hook_path in data["hooks"].items(): full = d / hook_path if not full.exists(): all_errors.append(f"plugins/{d.name}: Hook file '{hook_path}' not found") else: all_errors.append(f"plugins/{d.name}: Missing plugin.toml") stats["plugins"] = len(plugin_dirs) # --- Aliases --- if should_validate("aliases", args.content_type): all_errors.extend(validate_aliases_file(root / "aliases.toml")) # --- Cross-type: duplicate routing aliases (now treated as ERRORS) --- # Only run this cross-type check when no --type filter is active, # or when filtering to agents or hands specifically. all_aliases = {} # alias -> (type, name) alias_collisions = [] run_alias_check = args.content_type is None or args.content_type in ("agents", "hands") # Collect agent routing aliases if run_alias_check and agents_dir.is_dir(): for d in sorted([d for d in agents_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]): data, _ = load_toml(d / "agent.toml") if data: routing = data.get("metadata", {}).get("routing", {}) for alias in routing.get("aliases", []) + routing.get("weak_aliases", []): key = alias.lower() if key in all_aliases: prev_type, prev_name = all_aliases[key] alias_collisions.append(f"Routing alias '{alias}' used by both {prev_type}/{prev_name} and agent/{d.name}") else: all_aliases[key] = ("agent", d.name) # Collect hand routing aliases if run_alias_check and hands_dir.is_dir(): for d in sorted([d for d in hands_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]): data, _ = load_toml(d / "HAND.toml") if data: routing = data.get("routing", {}) for alias in routing.get("aliases", []) + routing.get("weak_aliases", []): key = alias.lower() if key in all_aliases: prev_type, prev_name = all_aliases[key] alias_collisions.append(f"Routing alias '{alias}' used by both {prev_type}/{prev_name} and hand/{d.name}") else: all_aliases[key] = ("hand", d.name) # --- Collect warnings --- warnings = list(alias_collisions) # --- In --strict mode, promote warnings to errors --- if args.strict and warnings: all_errors.extend(warnings) warnings = [] # --- Report --- print("=" * 60) print("LibreFang Registry Validation") if args.content_type: print(f" Filter: --type {args.content_type}") if args.strict: print(" Mode: --strict (warnings are errors)") print("=" * 60) print() if all_errors: print(f"ERRORS ({len(all_errors)}):") for error in all_errors: print(f" - {error}") print() print("Content summary:") for key in ("providers", "models", "agents", "hands", "mcp", "skills", "plugins"): if key in stats: print(f" {key:20s} {stats[key]:>4}") print() if warnings: print(f"WARNINGS ({len(warnings)}):") for w in warnings: print(f" - {w}") print() if all_errors: print(f"VALIDATION FAILED with {len(all_errors)} error(s)") sys.exit(1) else: print("VALIDATION PASSED") sys.exit(0) if __name__ == "__main__": main()