From 1f3ef406ee22d4437e045ee06fa74ee1901a1798 Mon Sep 17 00:00:00 2001 From: Evan Hu Date: Sat, 21 Mar 2026 02:10:12 +0900 Subject: [PATCH] docs: rewrite README and CONTRIBUTING for full registry scope - README now covers all 5 content types (agents, hands, integrations, skills, providers) - CONTRIBUTING has per-type instructions and checklists - validate.py expanded to validate agents, hands, integrations, and skills - PR template covers all content types - Added new-content.yml issue template for non-model contributions - CI workflow updated to Python 3.12 --- .github/ISSUE_TEMPLATE/new-content.yml | 49 ++++ .github/pull_request_template.md | 56 ++++- .github/workflows/validate.yml | 9 +- CONTRIBUTING.md | 246 +++++++++++++----- README.md | 218 ++++++++++------ scripts/validate.py | 329 ++++++++++++++++--------- 6 files changed, 639 insertions(+), 268 deletions(-) create mode 100644 .github/ISSUE_TEMPLATE/new-content.yml diff --git a/.github/ISSUE_TEMPLATE/new-content.yml b/.github/ISSUE_TEMPLATE/new-content.yml new file mode 100644 index 0000000..e3b09fe --- /dev/null +++ b/.github/ISSUE_TEMPLATE/new-content.yml @@ -0,0 +1,49 @@ +name: New Content Request +description: Request addition of an agent, hand, integration, or skill +labels: ["new-content"] +body: + - type: dropdown + id: content_type + attributes: + label: Content Type + description: What type of content are you requesting? + options: + - Agent + - Hand + - Integration (MCP server) + - Skill + validations: + required: true + + - type: input + id: name + attributes: + label: Name + description: The name/ID for this content. + placeholder: e.g. my-agent, browser, github + validations: + required: true + + - type: textarea + id: description + attributes: + label: Description + description: What does this content do? Why is it useful? + placeholder: | + Describe the purpose and key capabilities. + validations: + required: true + + - type: textarea + id: details + attributes: + label: Additional Details + description: | + Any extra context: required tools, external dependencies, + MCP server package, reference links, etc. + placeholder: | + - MCP server: @some/mcp-server + - Requires: API key from https://... + - Tools needed: web_search, file_read + validations: + required: false diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 8f90ad8..61c7713 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -1,14 +1,52 @@ -## Model Changes +## Content Type -### Added/Updated Models -- [ ] Model ID: -- [ ] Provider: -- [ ] Pricing verified from official source: + -### Checklist +- [ ] Agent (`agents/`) +- [ ] Hand (`hands/`) +- [ ] Integration (`integrations/`) +- [ ] Skill (`skills/`) +- [ ] Provider / Model (`providers/`) +- [ ] Other + +## Changes + +### Added/Updated +- + +### Details + + +## Checklist + +### General +- [ ] Content placed in the correct directory +- [ ] TOML files parse without errors +- [ ] Description is clear and concise + +### Providers / Models (if applicable) - [ ] `python scripts/validate.py` passes - [ ] No duplicate model IDs -- [ ] Pricing is in USD per million tokens +- [ ] Pricing is in USD per million tokens and verified from official source - [ ] Tier is one of: `frontier`, `smart`, `balanced`, `fast`, `local` -- [ ] `context_window` and `max_output_tokens` are positive integers -- [ ] Boolean fields (`supports_tools`, `supports_vision`, `supports_streaming`) are correct + +### Agents (if applicable) +- [ ] `name` matches directory name +- [ ] System prompt provides clear instructions +- [ ] Tools list only includes required tools + +### Hands (if applicable) +- [ ] `id` matches directory name +- [ ] `[agent]` section has complete system prompt +- [ ] `[[requires]]` lists external dependencies +- [ ] `[[settings]]` provides user-configurable options + +### Integrations (if applicable) +- [ ] `[transport]` config tested locally +- [ ] `[[required_env]]` lists all needed variables +- [ ] `setup_instructions` are clear for first-time users + +### Skills (if applicable) +- [ ] `[runtime].type` is correct +- [ ] `[input]` documents all parameters +- [ ] Prompt templates use correct `{{param}}` syntax diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 86d5d04..98a277d 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -1,4 +1,4 @@ -name: Validate Catalog +name: Validate Registry on: push: @@ -14,10 +14,7 @@ jobs: - uses: actions/setup-python@v6 with: - python-version: "3.11" + python-version: "3.12" - - name: Install dependencies - run: pip install toml - - - name: Validate catalog + - name: Validate registry run: python scripts/validate.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 79103c1..00d347f 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,59 +1,202 @@ -# Contributing to LibreFang Model Catalog +# Contributing to LibreFang Registry -Thank you for helping keep the model catalog up to date! This guide explains how to add or update model entries. +Thank you for helping grow the LibreFang ecosystem! This guide explains how to add or update content for each type. -## How to Add a New Model +## General Workflow -### 1. Fork & Clone +1. Fork & clone the repository +2. Create a branch: `git checkout -b feat/add-my-content` +3. Add or edit files in the appropriate directory +4. Run validation: `python scripts/validate.py` +5. Submit a Pull Request -```bash -git clone https://github.com//model-catalog.git -cd model-catalog -``` +## Adding an Agent -### 2. Find the Right Provider File - -Each provider has its own file in `providers/`. For example: -- OpenAI models go in `providers/openai.toml` -- Anthropic models go in `providers/anthropic.toml` - -If the provider doesn't exist yet, create a new file (e.g. `providers/newprovider.toml`) with the `[provider]` section and your `[[models]]` entries. - -### 3. Add Your Model Entry - -Append a `[[models]]` block to the provider file: +Create a directory `agents//` with an `agent.toml` file: ```toml +name = "my-agent" +version = "0.1.0" +description = "What this agent does" +author = "your-name" +module = "builtin:chat" + +[model] +provider = "default" +model = "default" +max_tokens = 4096 +temperature = 0.7 +system_prompt = """Your system prompt here.""" + +[capabilities] +tools = ["web_search", "file_read"] +``` + +### Agent Checklist + +- [ ] `name` matches the directory name +- [ ] `description` is clear and concise (one sentence) +- [ ] `system_prompt` provides clear behavioral instructions +- [ ] `tools` only lists tools the agent actually needs +- [ ] Routing aliases (if any) are relevant and don't conflict with existing agents + +## Adding a Hand + +Create a directory `hands//` with a `HAND.toml` file and optionally a `SKILL.md`: + +```toml +id = "my-hand" +name = "My Hand" +description = "What this hand does" +category = "productivity" # communication | content | data | development | devops | finance | productivity | research | social +icon = "🔧" + +tools = ["tool1", "tool2"] + +[routing] +aliases = ["activate my hand", "do the thing"] + +[agent] +name = "my-hand-agent" +module = "builtin:chat" +system_prompt = """Your agent prompt here.""" + +[[settings]] +key = "some_setting" +label = "Setting Label" +setting_type = "toggle" +default = "true" +``` + +### Hand Checklist + +- [ ] `id` matches the directory name +- [ ] `category` is valid (`communication`, `content`, `data`, `development`, `devops`, `finance`, `productivity`, `research`, `social`) +- [ ] `tools` lists all required tools +- [ ] `[agent]` section has a complete system prompt +- [ ] `[[requires]]` sections list any external dependencies (binaries, services) +- [ ] `[[settings]]` sections provide user-configurable options where appropriate + +## Adding an Integration + +Create a file `integrations/.toml`: + +```toml +id = "my-service" +name = "My Service" +description = "What this integration provides" +category = "devtools" # devtools | communication | storage | monitoring | data +icon = "🔌" +tags = ["relevant", "tags"] + +[transport] +type = "stdio" +command = "npx" +args = ["-y", "@some/mcp-server"] + +[[required_env]] +name = "MY_SERVICE_API_KEY" +label = "API Key" +help = "Get your key from https://..." +is_secret = true +get_url = "https://my-service.com/settings/api-keys" + +setup_instructions = """ +1. Get an API key from ... +2. Paste it into the field above. +""" +``` + +### Integration Checklist + +- [ ] `id` matches the filename (without `.toml`) +- [ ] `[transport]` section is correct (test the MCP server command locally) +- [ ] `[[required_env]]` lists all needed environment variables +- [ ] `setup_instructions` are clear enough for first-time users +- [ ] `is_secret = true` for any sensitive values (API keys, tokens) + +## Adding a Skill + +Create a directory `skills//` with a `skill.toml` and optionally implementation files: + +### Prompt-only Skill + +```toml +[skill] +name = "my-skill" +version = "0.1.0" +description = "What this skill does" +author = "your-name" +tags = ["relevant", "tags"] + +[runtime] +type = "promptonly" + +[input] +param1 = { type = "string", description = "Description", required = true } + +[prompt] +template = """Your prompt template using {{param1}}.""" +``` + +### Python Skill + +```toml +[skill] +name = "my-skill" +version = "0.1.0" +description = "What this skill does" + +[runtime] +type = "python" +entry = "main.py" +``` + +Plus a `main.py` with your implementation. + +### Skill Checklist + +- [ ] `name` matches the directory name +- [ ] `[runtime].type` is `promptonly` or `python` +- [ ] `[input]` section documents all parameters +- [ ] Prompt-only skills have a `[prompt].template` with correct `{{param}}` placeholders +- [ ] Python skills include all required files + +## Adding or Updating a Provider / Model + +Edit the appropriate provider file in `providers/`. If the provider doesn't exist, create a new file. + +```toml +[provider] +id = "my-provider" +display_name = "My Provider" +api_key_env = "MY_PROVIDER_API_KEY" +base_url = "https://api.my-provider.com" +key_required = true + [[models]] -id = "new-model-id" # The exact API model ID -display_name = "New Model Name" # Human-readable name -tier = "smart" # frontier | smart | balanced | fast | local +id = "model-id" +display_name = "Model Name" +tier = "smart" # frontier | smart | balanced | fast | local context_window = 128000 max_output_tokens = 16384 -input_cost_per_m = 2.50 # USD per million input tokens -output_cost_per_m = 10.0 # USD per million output tokens +input_cost_per_m = 2.50 # USD per million input tokens +output_cost_per_m = 10.0 # USD per million output tokens supports_tools = true supports_vision = false supports_streaming = true -aliases = [] # Optional short names +aliases = ["short-name"] ``` -### 4. Validate +### Provider Checklist -```bash -python scripts/validate.py -``` - -This checks: -- All TOML files parse correctly -- Required fields are present -- Tier values are valid -- Costs are non-negative -- No duplicate model IDs - -### 5. Submit a Pull Request - -Push your branch and open a PR. The PR template will guide you through the checklist. +- [ ] `python scripts/validate.py` passes +- [ ] No duplicate model IDs +- [ ] Pricing is in USD per million tokens +- [ ] Tier is one of: `frontier`, `smart`, `balanced`, `fast`, `local` +- [ ] `context_window` and `max_output_tokens` are positive integers +- [ ] Boolean capability fields are correct +- [ ] Pricing verified from official source ## Where to Find Model Information @@ -64,26 +207,11 @@ Push your branch and open a PR. The PR template will guide you through the check - **Mistral**: https://mistral.ai/technology/#pricing - **Groq**: https://wow.groq.com/ - **xAI**: https://docs.x.ai/docs -- **Cohere**: https://cohere.com/pricing -- **Together**: https://www.together.ai/pricing -- **Fireworks**: https://fireworks.ai/pricing -- **Perplexity**: https://docs.perplexity.ai/guides/pricing - -## Tier Definitions - -| Tier | Description | Examples | -|------|-------------|----------| -| `frontier` | Most capable, cutting-edge | Claude Opus, GPT-4.1, Gemini 2.5 Pro | -| `smart` | Smart and cost-effective | Claude Sonnet, GPT-4o, Gemini 2.5 Flash | -| `balanced` | Balanced speed and cost | GPT-4.1 Mini, Llama 3.3 70B | -| `fast` | Fastest, cheapest | GPT-4o Mini, Claude Haiku, Gemma 2 9B | -| `local` | Local models, zero cost | Ollama, vLLM, LM Studio | ## Guidelines -- **Pricing must be in USD per million tokens** -- convert from other units if needed -- **Use the exact model ID** that the provider's API expects - **Don't guess** -- only add data you can verify from official sources -- **One provider per file** -- don't mix providers in a single TOML file -- **Keep aliases short** -- 1-3 word abbreviations that users would naturally type - +- **Keep descriptions concise** -- one sentence that explains the purpose +- **Test locally** -- try your content with LibreFang before submitting +- **One PR per content type** -- don't mix agent additions with provider updates +- **Keep aliases short** -- 1-3 word abbreviations users would naturally type diff --git a/README.md b/README.md index e5a6e48..af8a157 100644 --- a/README.md +++ b/README.md @@ -1,101 +1,148 @@ -# LibreFang Model Catalog +# LibreFang Registry -Community-maintained model metadata catalog for [LibreFang](https://github.com/librefang/librefang) -- the open-source Agent Operating System. +Community-maintained content registry for [LibreFang](https://github.com/librefang/librefang) -- the open-source Agent Operating System. -This repository is the source of truth for model metadata (pricing, context windows, capabilities). When new models are released (e.g. GPT-5.5, Claude 5), anyone can submit a PR here without touching the LibreFang binary. +This repository is the single source of truth for all installable content definitions. Anyone can submit a PR to add new agents, hands, integrations, skills, or provider models -- no changes to the LibreFang binary required. ## Structure ``` -model-catalog/ -├── providers/ # One TOML file per provider +librefang-registry/ +├── agents/ # Agent definitions (TOML manifests) +│ ├── hello-world/agent.toml +│ ├── researcher/agent.toml +│ └── ... (33 agents) +├── hands/ # Hand definitions (TOML + docs) +│ ├── browser/HAND.toml +│ ├── trader/HAND.toml +│ └── ... (14 hands) +├── integrations/ # MCP server integration templates +│ ├── github.toml +│ ├── slack.toml +│ └── ... (25 integrations) +├── skills/ # Reusable skill definitions +│ ├── custom-skill-prompt/skill.toml +│ └── custom-skill-python/ +├── providers/ # LLM provider & model metadata │ ├── anthropic.toml │ ├── openai.toml -│ ├── gemini.toml -│ └── ... -├── aliases.toml # Global alias mappings (e.g. "sonnet" -> "claude-sonnet-4-6") -├── schema.toml # Reference schema documenting all fields +│ └── ... (46 providers, 190+ models) +├── plugins/ # Plugin packages +├── aliases.toml # Global model alias mappings +├── schema.toml # Provider/model schema reference ├── scripts/ │ └── validate.py # Validation script -├── CONTRIBUTING.md # How to add a new model +├── CONTRIBUTING.md └── LICENSE # MIT ``` -## How LibreFang Uses This Catalog +## Content Types -LibreFang ships with a built-in model catalog compiled into the binary. This repository serves as the upstream source. To update your local catalog: +### Agents -```bash -librefang catalog update -``` - -This fetches the latest TOML files from this repository and merges them into your local catalog. - -### Custom Local Models - -You can also add custom models locally without submitting a PR: - -```bash -# Add to your personal config -# ~/.librefang/model_catalog.toml - -[[models]] -id = "my-custom-model" -display_name = "My Custom Model" -provider = "ollama" -tier = "local" -context_window = 32768 -max_output_tokens = 4096 -input_cost_per_m = 0.0 -output_cost_per_m = 0.0 -supports_tools = true -supports_vision = false -supports_streaming = true -``` - -## Schema Reference - -Each provider file contains a `[provider]` section and one or more `[[models]]` entries: +Agent definitions in `agents//agent.toml` describe autonomous agents with their model config, tools, capabilities, and routing aliases. ```toml -[provider] -id = "provider-id" # Unique provider identifier -display_name = "Provider Name" # Human-readable name -api_key_env = "PROVIDER_API_KEY" # Environment variable for API key -base_url = "https://api.example.com" # Default API endpoint -key_required = true # Whether an API key is needed +name = "hello-world" +description = "A friendly greeting agent" +module = "builtin:chat" -[[models]] -id = "model-id" # Unique model identifier (API model ID) -display_name = "Human Name" # Human-readable display name -tier = "smart" # frontier | smart | balanced | fast | local -context_window = 128000 # Maximum input tokens -max_output_tokens = 16384 # Maximum output tokens -input_cost_per_m = 2.50 # USD per million input tokens -output_cost_per_m = 10.0 # USD per million output tokens -supports_tools = true # Tool/function calling support -supports_vision = true # Vision/image input support -supports_streaming = true # Streaming response support -aliases = ["alias1", "alias2"] # Short names for this model +[model] +provider = "default" +model = "default" +system_prompt = "You are a helpful assistant." + +[capabilities] +tools = ["web_search", "file_read"] ``` -### Tier Definitions +### Hands -| Tier | Description | Examples | -|------|-------------|----------| -| `frontier` | Most capable, cutting-edge models | Claude Opus, GPT-4.1, Gemini 2.5 Pro | -| `smart` | Smart, cost-effective models | Claude Sonnet, GPT-4o, Gemini 2.5 Flash | -| `balanced` | Balanced speed/cost | GPT-4.1 Mini, Llama 3.3 70B | -| `fast` | Fastest, cheapest | GPT-4o Mini, Claude Haiku | -| `local` | Local models (zero cost) | Ollama, vLLM, LM Studio | +Hands in `hands//HAND.toml` are higher-level application bundles -- the user-facing "apps" in LibreFang. Each hand bundles an agent config, tools, settings, dashboard metrics, and dependency requirements. -## How to Add a New Model +```toml +id = "browser" +name = "Browser Hand" +category = "productivity" +tools = ["browser_navigate", "browser_click", "browser_type"] -1. Edit the appropriate provider file in `providers/` -2. Run validation: `python scripts/validate.py` -3. Submit a Pull Request +[agent] +name = "browser-hand" +module = "builtin:chat" +system_prompt = "You are an autonomous web browser agent..." -See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed instructions. +[[settings]] +key = "headless" +setting_type = "toggle" +default = "true" +``` + +### Integrations + +Integration templates in `integrations/.toml` define MCP server connections (GitHub, Slack, databases, etc.) with transport config, required env vars, and setup instructions. + +```toml +id = "github" +name = "GitHub" +category = "devtools" + +[transport] +type = "stdio" +command = "npx" +args = ["-y", "@modelcontextprotocol/server-github"] + +[[required_env]] +name = "GITHUB_PERSONAL_ACCESS_TOKEN" +is_secret = true +``` + +### Skills + +Skills in `skills//skill.toml` are reusable prompt templates or Python scripts that agents can invoke. + +```toml +[skill] +name = "meeting-agenda" +description = "Generate a structured meeting agenda" + +[runtime] +type = "promptonly" + +[prompt] +template = "Create a meeting agenda for: {{topic}}" +``` + +### Providers + +Provider files in `providers/.toml` define LLM providers and their models with pricing, context windows, and capability flags. See [schema.toml](schema.toml) for the full field reference. + +## How LibreFang Uses This Registry + +LibreFang ships with built-in content compiled into the binary. This repository serves as the upstream source for updates and community contributions. + +```bash +# Update all registry content +librefang catalog update + +# Install a specific hand +librefang hand install browser + +# Install a specific integration +librefang integration install github +``` + +### Custom Local Content + +You can also create custom content locally without submitting a PR: + +```bash +# Create a custom agent +mkdir -p ~/.librefang/agents/my-agent +# Edit ~/.librefang/agents/my-agent/agent.toml + +# Add custom models to your config +# ~/.librefang/model_catalog.toml +``` ## Validation @@ -103,13 +150,28 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed instructions. python scripts/validate.py ``` -This checks all TOML files for correctness: required fields, valid tiers, non-negative costs, no duplicate IDs. +This validates all provider TOML files for correctness: required fields, valid tiers, non-negative costs, no duplicate IDs. + +## How to Contribute + +1. Fork this repository +2. Add or edit content in the appropriate directory +3. Run validation: `python scripts/validate.py` +4. Submit a Pull Request + +See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed instructions for each content type. ## Current Stats -- **30+ providers** including Anthropic, OpenAI, Google, DeepSeek, Groq, Mistral, xAI, and more -- **190+ models** with pricing, context windows, and capability flags -- **80+ aliases** for quick model selection +| Type | Count | +|------|-------| +| Agents | 33 | +| Hands | 14 | +| Integrations | 25 | +| Skills | 2 | +| Providers | 46 | +| Models | 190+ | +| Aliases | 80+ | ## License diff --git a/scripts/validate.py b/scripts/validate.py index d835ce8..f1d6f07 100755 --- a/scripts/validate.py +++ b/scripts/validate.py @@ -1,19 +1,16 @@ #!/usr/bin/env python3 -"""Validate all TOML files in the model catalog against the schema. +"""Validate all TOML files in the LibreFang registry. Usage: python scripts/validate.py -Checks: - - All provider TOML files parse correctly - - Required provider fields exist - - Required model fields exist for each [[models]] entry - - Tier values are one of: frontier, smart, balanced, fast, local - - Cost values are non-negative - - context_window and max_output_tokens are positive integers - - No duplicate model IDs within the same provider file - - No duplicate model IDs across ALL provider files (same provider) - - aliases.toml parses correctly +Validates: + - Provider TOML files (required fields, valid tiers, non-negative costs, no duplicates) + - Agent TOML files (required fields: name, description, module) + - Hand TOML files (required fields: id, name, description, category) + - Integration TOML files (required fields: id, name, [transport]) + - Skill TOML files (required fields: [skill].name, [runtime].type) + - aliases.toml parsing Exit code 0 on success, 1 on any validation error. """ @@ -32,31 +29,36 @@ except ImportError: sys.exit(1) VALID_TIERS = {"frontier", "smart", "balanced", "fast", "local"} +VALID_HAND_CATEGORIES = { + "communication", "content", "data", "development", + "devops", "finance", "productivity", "research", "social", +} +VALID_SKILL_RUNTIMES = {"promptonly", "python", "node", "shell"} REQUIRED_PROVIDER_FIELDS = {"id", "display_name", "api_key_env", "base_url", "key_required"} - REQUIRED_MODEL_FIELDS = { - "id", - "display_name", - "tier", - "context_window", - "max_output_tokens", - "input_cost_per_m", - "output_cost_per_m", + "id", "display_name", "tier", + "context_window", "max_output_tokens", + "input_cost_per_m", "output_cost_per_m", } -def validate_provider_file(filepath: Path) -> list[str]: - """Validate a single provider TOML file. Returns list of error messages.""" - errors = [] - +def load_toml(filepath: Path) -> tuple[dict | None, str | None]: + """Load a TOML file, return (data, error).""" try: with open(filepath, "rb") as f: - data = tomllib.load(f) + return tomllib.load(f), None except Exception as e: - return [f"{filepath.name}: Failed to parse TOML: {e}"] + return None, f"{filepath}: Failed to parse TOML: {e}" - # Validate [provider] section + +def validate_provider_file(filepath: Path) -> list[str]: + """Validate a single provider TOML file.""" + data, err = load_toml(filepath) + if err: + return [err] + + errors = [] provider = data.get("provider") if provider is None: errors.append(f"{filepath.name}: Missing [provider] section") @@ -65,66 +67,141 @@ def validate_provider_file(filepath: Path) -> list[str]: if field not in provider: errors.append(f"{filepath.name}: Missing provider field '{field}'") - # Validate [[models]] entries models = data.get("models", []) seen_ids = set() for i, model in enumerate(models): - model_label = model.get("id", f"models[{i}]") - - # Check required fields + label = model.get("id", f"models[{i}]") for field in REQUIRED_MODEL_FIELDS: if field not in model: - errors.append(f"{filepath.name}: Model '{model_label}' missing field '{field}'") + errors.append(f"{filepath.name}: Model '{label}' missing field '{field}'") - # Validate tier tier = model.get("tier") if tier is not None and tier not in VALID_TIERS: - errors.append( - f"{filepath.name}: Model '{model_label}' has invalid tier '{tier}' " - f"(must be one of: {', '.join(sorted(VALID_TIERS))})" - ) + errors.append(f"{filepath.name}: Model '{label}' invalid tier '{tier}'") - # Validate costs for cost_field in ("input_cost_per_m", "output_cost_per_m"): val = model.get(cost_field) if val is not None and (not isinstance(val, (int, float)) or val < 0): - errors.append( - f"{filepath.name}: Model '{model_label}' has invalid {cost_field}: {val} " - f"(must be >= 0)" - ) + errors.append(f"{filepath.name}: Model '{label}' invalid {cost_field}: {val}") - # Validate context_window and max_output_tokens for int_field in ("context_window", "max_output_tokens"): val = model.get(int_field) if val is not None and (not isinstance(val, int) or val <= 0): - errors.append( - f"{filepath.name}: Model '{model_label}' has invalid {int_field}: {val} " - f"(must be a positive integer)" - ) + errors.append(f"{filepath.name}: Model '{label}' invalid {int_field}: {val}") - # Check for duplicate IDs within the file model_id = model.get("id") if model_id: if model_id in seen_ids: - errors.append( - f"{filepath.name}: Duplicate model ID '{model_id}' within file" - ) + errors.append(f"{filepath.name}: Duplicate model ID '{model_id}'") seen_ids.add(model_id) return errors -def validate_aliases_file(filepath: Path) -> list[str]: - """Validate aliases.toml. Returns list of error messages.""" - if not filepath.exists(): - return [f"aliases.toml: File not found at {filepath}"] +def validate_agent_file(filepath: Path) -> list[str]: + """Validate an agent.toml file.""" + data, err = load_toml(filepath) + if err: + return [err] - try: - with open(filepath, "rb") as f: - data = tomllib.load(f) - except Exception as e: - return [f"aliases.toml: Failed to parse TOML: {e}"] + errors = [] + rel = filepath.relative_to(filepath.parent.parent) + + for field in ("name", "description", "module"): + if field not in data: + errors.append(f"{rel}: Missing required field '{field}'") + + if "model" in data: + model = data["model"] + if "system_prompt" not in model and "provider" not in model: + errors.append(f"{rel}: [model] section should have 'provider' or 'system_prompt'") + + return errors + + +def validate_hand_file(filepath: Path) -> list[str]: + """Validate a HAND.toml file.""" + data, err = load_toml(filepath) + if err: + return [err] + + errors = [] + rel = filepath.relative_to(filepath.parent.parent.parent) + + for field in ("id", "name", "description"): + if field not in data: + errors.append(f"{rel}: Missing required field '{field}'") + + category = data.get("category") + if category and category not in VALID_HAND_CATEGORIES: + errors.append(f"{rel}: Invalid category '{category}' (valid: {', '.join(sorted(VALID_HAND_CATEGORIES))})") + + if "agent" not in data: + errors.append(f"{rel}: Missing [agent] section") + + return errors + + +def validate_integration_file(filepath: Path) -> list[str]: + """Validate an integration TOML file.""" + data, err = load_toml(filepath) + if err: + return [err] + + errors = [] + + for field in ("id", "name"): + if field not in data: + errors.append(f"{filepath.name}: Missing required field '{field}'") + + if "transport" not in data: + errors.append(f"{filepath.name}: Missing [transport] section") + else: + transport = data["transport"] + if "type" not in transport: + errors.append(f"{filepath.name}: Missing transport.type") + if "command" not in transport: + errors.append(f"{filepath.name}: Missing transport.command") + + return errors + + +def validate_skill_file(filepath: Path) -> list[str]: + """Validate a skill.toml file.""" + data, err = load_toml(filepath) + if err: + return [err] + + errors = [] + rel = filepath.relative_to(filepath.parent.parent) + + skill = data.get("skill") + if skill is None: + errors.append(f"{rel}: Missing [skill] section") + else: + if "name" not in skill: + errors.append(f"{rel}: Missing skill.name") + + runtime = data.get("runtime") + if runtime is None: + errors.append(f"{rel}: Missing [runtime] section") + else: + rt = runtime.get("type") + if rt and rt not in VALID_SKILL_RUNTIMES: + errors.append(f"{rel}: Invalid runtime type '{rt}' (valid: {', '.join(sorted(VALID_SKILL_RUNTIMES))})") + + return errors + + +def validate_aliases_file(filepath: Path) -> list[str]: + """Validate aliases.toml.""" + if not filepath.exists(): + return [f"aliases.toml not found at {filepath}"] + + data, err = load_toml(filepath) + if err: + return [err] errors = [] aliases = data.get("aliases", {}) @@ -139,67 +216,93 @@ def validate_aliases_file(filepath: Path) -> list[str]: def main(): - # Find the catalog root script_dir = Path(__file__).resolve().parent - catalog_root = script_dir.parent - providers_dir = catalog_root / "providers" - - if not providers_dir.is_dir(): - print(f"ERROR: providers/ directory not found at {providers_dir}") - sys.exit(1) + root = script_dir.parent all_errors = [] - total_models = 0 - provider_counts = {} - global_model_ids = {} # model_id -> (provider_id, filename) + stats = {} - # Validate each provider file - toml_files = sorted(providers_dir.glob("*.toml")) - if not toml_files: - print("ERROR: No .toml files found in providers/") - sys.exit(1) + # --- Providers --- + providers_dir = root / "providers" + if providers_dir.is_dir(): + toml_files = sorted(providers_dir.glob("*.toml")) + total_models = 0 + global_model_ids = {} - for filepath in toml_files: - errors = validate_provider_file(filepath) - all_errors.extend(errors) - - # Count models if file parsed successfully - if not any("Failed to parse" in e for e in errors): - try: - with open(filepath, "rb") as f: - data = tomllib.load(f) - provider_id = data.get("provider", {}).get("id", filepath.stem) + for fp in toml_files: + all_errors.extend(validate_provider_file(fp)) + data, _ = load_toml(fp) + if data: models = data.get("models", []) - count = len(models) - total_models += count - provider_counts[provider_id] = count - - # Track global model IDs for cross-file duplicate detection - for model in models: - mid = model.get("id") - mprov = model.get("provider", provider_id) + total_models += len(models) + provider_id = data.get("provider", {}).get("id", fp.stem) + for m in models: + mid = m.get("id") if mid: - key = (mid, mprov) + key = (mid, provider_id) if key in global_model_ids: - prev_file = global_model_ids[key] - # Only flag if same provider in different files - if prev_file != filepath.name: + prev = global_model_ids[key] + if prev != fp.name: all_errors.append( - f"Cross-file duplicate: model '{mid}' (provider '{mprov}') " - f"found in both {prev_file} and {filepath.name}" + f"Cross-file duplicate: model '{mid}' (provider '{provider_id}') " + f"in both {prev} and {fp.name}" ) else: - global_model_ids[key] = filepath.name - except Exception: - pass + global_model_ids[key] = fp.name - # Validate aliases.toml - aliases_file = catalog_root / "aliases.toml" - all_errors.extend(validate_aliases_file(aliases_file)) + stats["providers"] = len(toml_files) + stats["models"] = total_models - # Print results + # --- Agents --- + agents_dir = root / "agents" + if agents_dir.is_dir(): + agent_dirs = sorted([d for d in agents_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) + for d in agent_dirs: + agent_toml = d / "agent.toml" + if agent_toml.exists(): + all_errors.extend(validate_agent_file(agent_toml)) + else: + all_errors.append(f"agents/{d.name}: Missing agent.toml") + stats["agents"] = len(agent_dirs) + + # --- Hands --- + hands_dir = root / "hands" + if hands_dir.is_dir(): + hand_dirs = sorted([d for d in hands_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) + for d in hand_dirs: + hand_toml = d / "HAND.toml" + if hand_toml.exists(): + all_errors.extend(validate_hand_file(hand_toml)) + else: + all_errors.append(f"hands/{d.name}: Missing HAND.toml") + stats["hands"] = len(hand_dirs) + + # --- Integrations --- + integrations_dir = root / "integrations" + if integrations_dir.is_dir(): + int_files = sorted(integrations_dir.glob("*.toml")) + for fp in int_files: + all_errors.extend(validate_integration_file(fp)) + stats["integrations"] = len(int_files) + + # --- Skills --- + skills_dir = root / "skills" + if skills_dir.is_dir(): + skill_dirs = sorted([d for d in skills_dir.iterdir() if d.is_dir() and not d.name.startswith(".")]) + for d in skill_dirs: + skill_toml = d / "skill.toml" + if skill_toml.exists(): + all_errors.extend(validate_skill_file(skill_toml)) + else: + all_errors.append(f"skills/{d.name}: Missing skill.toml") + stats["skills"] = len(skill_dirs) + + # --- Aliases --- + all_errors.extend(validate_aliases_file(root / "aliases.toml")) + + # --- Report --- print("=" * 60) - print("LibreFang Model Catalog Validation") + print("LibreFang Registry Validation") print("=" * 60) print() @@ -209,16 +312,10 @@ def main(): print(f" - {error}") print() - print(f"Provider files: {len(toml_files)}") - print(f"Total models: {total_models}") - print() - - print("Per-provider model counts:") - for provider_id in sorted(provider_counts.keys()): - count = provider_counts[provider_id] - if count > 0: - print(f" {provider_id:30s} {count:>3}") - + print("Content summary:") + for key in ("providers", "models", "agents", "hands", "integrations", "skills"): + if key in stats: + print(f" {key:20s} {stats[key]:>4}") print() if all_errors: