Files
librefang-registry/providers/meta-llama.toml
T
Evan c192f493b4 fix: route providers through correct APIs (#39)
Providers with known public APIs use their official endpoints:
- meta-llama → api.llama.com/v1
- microsoft → models.inference.ai.azure.com (GitHub Models)
- ibm-granite → us-south.ml.cloud.ibm.com/ml/v1 (watsonx)
- tencent → api.hunyuan.cloud.tencent.com/v1
- morph → api.morphllm.com/v1

16 remaining providers without known public APIs route through
OpenRouter (base_url = openrouter.ai/api/v1, OPENROUTER_API_KEY).

sync-pricing.py updated with PROVIDER_API mapping.
2026-04-02 23:55:42 +08:00

149 lines
3.2 KiB
TOML

# meta-llama — auto-generated from OpenRouter API
[provider]
id = "meta-llama"
display_name = "Meta Llama"
api_key_env = "LLAMA_API_KEY"
base_url = "https://api.llama.com/v1"
key_required = true
[[models]]
id = "llama-3-70b-instruct"
display_name = "Meta: Llama 3 70B Instruct"
tier = "smart"
context_window = 8192
max_output_tokens = 8000
input_cost_per_m = 0.51
output_cost_per_m = 0.74
supports_streaming = true
[[models]]
id = "llama-3-8b-instruct"
display_name = "Meta: Llama 3 8B Instruct"
tier = "fast"
context_window = 8192
max_output_tokens = 16384
input_cost_per_m = 0.03
output_cost_per_m = 0.04
supports_streaming = true
[[models]]
id = "llama-3.1-70b-instruct"
display_name = "Meta: Llama 3.1 70B Instruct"
tier = "fast"
context_window = 131072
max_output_tokens = 16384
input_cost_per_m = 0.4
output_cost_per_m = 0.4
supports_streaming = true
[[models]]
id = "llama-3.1-8b-instruct"
display_name = "Meta: Llama 3.1 8B Instruct"
tier = "fast"
context_window = 16384
max_output_tokens = 16384
input_cost_per_m = 0.02
output_cost_per_m = 0.05
supports_streaming = true
[[models]]
id = "llama-3.2-11b-vision-instruct"
display_name = "Meta: Llama 3.2 11B Vision Instruct"
tier = "fast"
context_window = 131072
max_output_tokens = 16384
input_cost_per_m = 0.049
output_cost_per_m = 0.049
supports_streaming = true
[[models]]
id = "llama-3.2-1b-instruct"
display_name = "Meta: Llama 3.2 1B Instruct"
tier = "fast"
context_window = 60000
max_output_tokens = 15000
input_cost_per_m = 0.027
output_cost_per_m = 0.2
supports_streaming = true
[[models]]
id = "llama-3.2-3b-instruct"
display_name = "Meta: Llama 3.2 3B Instruct"
tier = "fast"
context_window = 80000
max_output_tokens = 16384
input_cost_per_m = 0.051
output_cost_per_m = 0.34
supports_streaming = true
[[models]]
id = "llama-3.2-3b-instruct:free"
display_name = "Meta: Llama 3.2 3B Instruct (free)"
tier = "fast"
context_window = 131072
max_output_tokens = 16384
input_cost_per_m = 0.0
output_cost_per_m = 0.0
supports_streaming = true
[[models]]
id = "llama-3.3-70b-instruct"
display_name = "Meta: Llama 3.3 70B Instruct"
tier = "fast"
context_window = 131072
max_output_tokens = 16384
input_cost_per_m = 0.1
output_cost_per_m = 0.32
supports_streaming = true
[[models]]
id = "llama-3.3-70b-instruct:free"
display_name = "Meta: Llama 3.3 70B Instruct (free)"
tier = "fast"
context_window = 65536
max_output_tokens = 16384
input_cost_per_m = 0.0
output_cost_per_m = 0.0
supports_streaming = true
[[models]]
id = "llama-4-maverick"
display_name = "Meta: Llama 4 Maverick"
tier = "fast"
context_window = 1048576
max_output_tokens = 16384
input_cost_per_m = 0.15
output_cost_per_m = 0.6
supports_streaming = true
[[models]]
id = "llama-4-scout"
display_name = "Meta: Llama 4 Scout"
tier = "fast"
context_window = 327680
max_output_tokens = 16384
input_cost_per_m = 0.08
output_cost_per_m = 0.3
supports_streaming = true
[[models]]
id = "llama-guard-3-8b"
display_name = "Llama Guard 3 8B"
tier = "fast"
context_window = 131072
max_output_tokens = 16384
input_cost_per_m = 0.02
output_cost_per_m = 0.06
supports_streaming = true
[[models]]
id = "llama-guard-4-12b"
display_name = "Meta: Llama Guard 4 12B"
tier = "fast"
context_window = 163840
max_output_tokens = 16384
input_cost_per_m = 0.18
output_cost_per_m = 0.18
supports_streaming = true