* feat: add 4 context engine plugins
- topic-memory: keyword clustering for topic-aware memory recall
- episodic-memory: conversation segmentation and cross-session recall
- user-profile: persistent user profiling from conversation patterns
- context-decay: time-based memory decay with reinforcement dynamics
All plugins use the ingest/after_turn hook protocol with stdin/stdout JSON.
* chore: add plugin scaffolding, update docs and templates
- Add plugin.toml template with {{NAME}} placeholder
- Add new-plugin Makefile target with hooks/ scaffolding
- Update plugins/README.md with all 10 plugins
- Update README.md stats (10 plugins, 220+ models)
- Add Plugin checkbox and checklist to PR template
- Add Plugin to issue template content type dropdown
- Fix CONTRIBUTING.md: last_verified is recommended, not required
* fix: correct model pricing and remove deprecated entries
- openrouter/gemma-2-9b-it: fix pricing from 0.0 to 0.03/0.09 per M tokens
(free variant correctly stays at 0.0)
- github-copilot: remove deprecated copilot/gpt-4 model entry
(GPT-4 retired in favor of GPT-4o for Copilot)
* docs: annotate kimi-coding as membership-gated
Kimi Code CLI uses quota-based membership model (not per-token billing).
Free tier has limited weekly requests; underlying model is K2.5.
Pricing kept at 0.0 consistent with other subscription providers
(chatgpt, github-copilot) but with explanatory comments.
* style: fix trailing newline in github-copilot.toml
* fix: correct Moonshot/Kimi model pricing from official sources
All 5 models had incorrect pricing:
- moonshot-v1-8k: 0.10/0.10 → 0.20/2.00
- moonshot-v1-32k: 0.30/0.30 → 1.00/3.00
- moonshot-v1-128k: 0.80/0.80 → 2.00/5.00
- kimi-k2: 2.00/8.00 → 0.60/2.50
- kimi-k2.5: 2.00/8.00 → 0.45/2.20
Sources: platform.moonshot.ai/docs/pricing/chat, costgoat.com, getmaxim.ai
* feat: add MiniMax M2.7 and M2.7-highspeed models
Released 2026-03-18, MiniMax's latest flagship text model.
10B activated params, 200K context, 128K output, tool use, streaming.
Pricing: $0.30/$1.20 per M tokens (input/output).
Added to both international (minimax.io) and China (minimaxi.com) providers.
233 lines
5.2 KiB
TOML
233 lines
5.2 KiB
TOML
# OpenRouter — https://openrouter.ai
|
|
# Models: 17 (10 paid + 7 free)
|
|
|
|
[provider]
|
|
id = "openrouter"
|
|
display_name = "OpenRouter"
|
|
api_key_env = "OPENROUTER_API_KEY"
|
|
base_url = "https://openrouter.ai/api/v1"
|
|
key_required = true
|
|
|
|
[[models]]
|
|
id = "openrouter/google/gemini-2.5-flash"
|
|
display_name = "Gemini 2.5 Flash (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 1048576
|
|
max_output_tokens = 65536
|
|
input_cost_per_m = 0.15
|
|
output_cost_per_m = 0.60
|
|
supports_tools = true
|
|
supports_vision = true
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/anthropic/claude-sonnet-4"
|
|
display_name = "Claude Sonnet 4 (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 200000
|
|
max_output_tokens = 64000
|
|
input_cost_per_m = 3.0
|
|
output_cost_per_m = 15.0
|
|
supports_tools = true
|
|
supports_vision = true
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/openai/gpt-4o"
|
|
display_name = "GPT-4o (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 128000
|
|
max_output_tokens = 16384
|
|
input_cost_per_m = 2.5
|
|
output_cost_per_m = 10.0
|
|
supports_tools = true
|
|
supports_vision = true
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/deepseek/deepseek-chat"
|
|
display_name = "DeepSeek V3 (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 128000
|
|
max_output_tokens = 32768
|
|
input_cost_per_m = 0.14
|
|
output_cost_per_m = 0.28
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/meta-llama/llama-3.3-70b-instruct"
|
|
display_name = "Llama 3.3 70B (OpenRouter)"
|
|
tier = "balanced"
|
|
context_window = 128000
|
|
max_output_tokens = 32768
|
|
input_cost_per_m = 0.39
|
|
output_cost_per_m = 0.39
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/qwen/qwen-2.5-72b-instruct"
|
|
display_name = "Qwen 2.5 72B (OpenRouter)"
|
|
tier = "balanced"
|
|
context_window = 128000
|
|
max_output_tokens = 32768
|
|
input_cost_per_m = 0.36
|
|
output_cost_per_m = 0.36
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/google/gemini-2.5-pro"
|
|
display_name = "Gemini 2.5 Pro (OpenRouter)"
|
|
tier = "frontier"
|
|
context_window = 1048576
|
|
max_output_tokens = 65536
|
|
input_cost_per_m = 1.25
|
|
output_cost_per_m = 10.0
|
|
supports_tools = true
|
|
supports_vision = true
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/mistralai/mistral-large-latest"
|
|
display_name = "Mistral Large (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 128000
|
|
max_output_tokens = 8192
|
|
input_cost_per_m = 2.0
|
|
output_cost_per_m = 6.0
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/google/gemma-2-9b-it"
|
|
display_name = "Gemma 2 9B (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 8192
|
|
max_output_tokens = 4096
|
|
input_cost_per_m = 0.03
|
|
output_cost_per_m = 0.09
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/deepseek/deepseek-r1"
|
|
display_name = "DeepSeek R1 (OpenRouter)"
|
|
tier = "frontier"
|
|
context_window = 128000
|
|
max_output_tokens = 32768
|
|
input_cost_per_m = 0.55
|
|
output_cost_per_m = 2.19
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
# Free models
|
|
|
|
[[models]]
|
|
id = "openrouter/stepfun/step-3.5-flash:free"
|
|
display_name = "Step 3.5 Flash Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 128000
|
|
max_output_tokens = 8192
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/google/gemma-2-9b-it:free"
|
|
display_name = "Gemma 2 9B Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 8192
|
|
max_output_tokens = 4096
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/meta-llama/llama-3.1-8b-instruct:free"
|
|
display_name = "Llama 3.1 8B Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 131072
|
|
max_output_tokens = 4096
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/qwen/qwen-2.5-7b-instruct:free"
|
|
display_name = "Qwen 2.5 7B Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 32768
|
|
max_output_tokens = 4096
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = true
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/mistralai/mistral-7b-instruct:free"
|
|
display_name = "Mistral 7B Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 32768
|
|
max_output_tokens = 4096
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/huggingfaceh4/zephyr-7b-beta:free"
|
|
display_name = "Zephyr 7B Free (OpenRouter)"
|
|
tier = "fast"
|
|
context_window = 4096
|
|
max_output_tokens = 2048
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|
|
|
|
[[models]]
|
|
id = "openrouter/deepseek/deepseek-r1:free"
|
|
display_name = "DeepSeek R1 Free (OpenRouter)"
|
|
tier = "smart"
|
|
context_window = 128000
|
|
max_output_tokens = 32768
|
|
input_cost_per_m = 0.0
|
|
output_cost_per_m = 0.0
|
|
supports_tools = false
|
|
supports_vision = false
|
|
supports_streaming = true
|
|
aliases = []
|