Files
Evan 6785807633 feat(providers): add reasoning_echo_policy field for OpenAI-compat reasoning_content handling (#90)
Refs librefang/librefang#4842 — long-term replacement for the substring
match that the OpenAI driver currently uses to decide how to handle
`reasoning_content` on historical assistant turns.

Three provider-specific behaviours that the driver must distinguish at
wire time, now expressed as catalog metadata:

* `strip`  — DeepSeek R1 / deepseek-reasoner. The API rejects requests
  that carry reasoning_content on previous assistant messages.
* `echo`   — DeepSeek V4 Flash. Thinking mode is on by default and the
  API rejects multi-turn requests when assistant turns containing
  tool_calls don't echo back the original reasoning text. This is the
  bug surfaced in librefang/librefang#4842.
* `empty_string` — Moonshot / Kimi K2 family. The field must be present
  (empty string) on tool_calls turns, with thinking disabled wire-side
  for multi-turn compatibility.
* `none` (default) — most providers; field is omitted entirely.

V4 Pro is intentionally NOT marked `echo` — librefang#4842 reports it
working out-of-the-box; flip when there's an empirical reproducer.

Marks affected models:

  providers/deepseek.toml
    deepseek-v4-flash → echo
    deepseek-reasoner → strip
  providers/moonshot.toml
    kimi-k2.6, kimi-k2.5, kimi-k2 → empty_string
  providers/kimi-coding.toml
    kimi-for-coding → empty_string
  providers/byteplus-coding.toml
    kimi-k2.5 → empty_string
  providers/novita.toml
    moonshotai/kimi-k2-thinking → empty_string

Tooling:

* schema.toml registers the field with the four enum options and a
  `none` default so existing TOML files keep parsing unchanged.
* scripts/validate.py rejects unknown enum values; verified with a
  hand-crafted negative case (`reasoning_echo_policy = "bogus"` →
  validation fails with the expected message).
* `python3 scripts/validate.py` passes (267 models).

The librefang side that consumes this field will land in a follow-up
PR — until then, registry consumers ignore the field via
`#[serde(default)]` and the existing substring fallback continues to
work, so this commit is safe to ship independently.
2026-05-11 00:41:24 +09:00

72 lines
1.8 KiB
TOML

# DeepSeek — https://deepseek.com
# Models: 4
[provider]
id = "deepseek"
display_name = "DeepSeek"
api_key_env = "DEEPSEEK_API_KEY"
base_url = "https://api.deepseek.com/v1"
key_required = true
[[models]]
id = "deepseek-v4-pro"
display_name = "DeepSeek V4 Pro"
tier = "frontier"
context_window = 1000000
max_output_tokens = 384000
input_cost_per_m = 0.435
output_cost_per_m = 0.87
supports_tools = true
supports_vision = false
supports_streaming = true
supports_thinking = true
aliases = ["deepseek-v4", "deepseek-pro"]
[[models]]
id = "deepseek-v4-flash"
display_name = "DeepSeek V4 Flash"
tier = "smart"
context_window = 1000000
max_output_tokens = 384000
input_cost_per_m = 0.14
output_cost_per_m = 0.28
supports_tools = true
supports_vision = false
supports_streaming = true
supports_thinking = true
# Thinking mode is on by default and the API rejects multi-turn requests
# where assistant turns containing tool_calls don't echo back the original
# reasoning_content. See librefang/librefang#4842.
reasoning_echo_policy = "echo"
aliases = ["deepseek-flash"]
[[models]]
id = "deepseek-chat"
display_name = "DeepSeek V3"
tier = "smart"
context_window = 64000
max_output_tokens = 8192
input_cost_per_m = 0.32
output_cost_per_m = 0.89
supports_tools = true
supports_vision = false
supports_streaming = true
aliases = ["deepseek", "deepseek-v3"]
[[models]]
id = "deepseek-reasoner"
display_name = "DeepSeek R1"
tier = "frontier"
context_window = 64000
max_output_tokens = 8192
input_cost_per_m = 0.55
output_cost_per_m = 2.19
supports_tools = false
supports_vision = false
supports_streaming = true
supports_thinking = true
# R1 returns reasoning_content in responses but the API rejects multi-turn
# requests that carry it on previous assistant messages — drivers must strip.
reasoning_echo_policy = "strip"
aliases = ["deepseek-r1"]