feat(reasoning): expose DeepSeek effort controls

This commit is contained in:
Release Repro 2026-07-31 13:14:11 -04:00
parent f613821bfb
commit 76dae568f5
No known key found for this signature in database
9 changed files with 194 additions and 9 deletions

View file

@ -70,6 +70,34 @@ digraph Example {
}
```
## Thinking and reasoning effort
DeepSeek enables thinking by default at `high` effort. Fabro advertises only effort values that produce a distinct model behavior on each route:
| Route | V4 Flash | V4 Pro |
|---|---|---|
| Direct DeepSeek | `low`, `high`, `max` | `high`, `max` |
| Fireworks AI | `high`, `max` | `high`, `max` |
| OpenRouter | `low`, `high`, `max` | `high`, `xhigh` |
DeepSeek currently maps a V4 Pro request for `low` to `high`; its documentation says this mapping will change in early August 2026. Fireworks maps `low` and `medium` to `high`, and maps `xhigh` to `max`. OpenRouter names the Pro maximum tier `xhigh`.
Fabro omits `temperature` and `top_p` for these models because DeepSeek ignores sampling parameters while thinking is enabled.
To disable thinking on the direct provider, omit typed `reasoning_effort` and pass DeepSeek's native toggle through provider options:
```json
{
"provider_options": {
"deepseek": {
"thinking": { "type": "disabled" }
}
}
}
```
DeepSeek requires `reasoning_content` from an assistant tool call to appear in every later request in that tool-use turn. Fabro captures this content and replays it with the assistant message. This keeps multi-step tool calls valid and prevents DeepSeek's HTTP 400 response for missing reasoning history.
## Prompt caching and pricing
DeepSeek applies prefix caching automatically. Fabro reads DeepSeek's `prompt_cache_hit_tokens` usage field and reports cache-read tokens separately from uncached input tokens.
@ -99,11 +127,14 @@ See the [Fireworks AI integration](/integrations/fireworks) and [OpenRouter inte
## Further reading
<Columns cols={2}>
<Columns cols={3}>
<Card title="DeepSeek API" icon="code" href="https://api-docs.deepseek.com/">
Official authentication, endpoints, and API reference.
</Card>
<Card title="Models and pricing" icon="money-bill" href="https://api-docs.deepseek.com/quick_start/pricing/">
Official limits, features, and token prices.
</Card>
<Card title="Thinking mode" icon="brain" href="https://api-docs.deepseek.com/guides/thinking_mode/">
Official thinking toggles, effort mappings, and tool-call replay rules.
</Card>
</Columns>

View file

@ -237,7 +237,9 @@ pub(super) fn translate_response_format(format: &ResponseFormat) -> serde_json::
#[cfg(test)]
mod tests {
use super::*;
use crate::types::{AudioData, ContentPart, DocumentData, Message, Role, ToolCall};
use crate::types::{
AudioData, ContentPart, DocumentData, Message, Role, ThinkingData, ToolCall,
};
#[test]
fn translate_assistant_message_with_tool_calls_only() {
@ -291,6 +293,37 @@ mod tests {
assert_eq!(tool_calls[0].function.name, "get_weather");
}
#[test]
fn translate_assistant_tool_call_replays_reasoning_content() {
let msg = Message {
role: Role::Assistant,
content: vec![
ContentPart::Thinking(ThinkingData {
text: "I need the weather tool.".to_string(),
signature: None,
redacted: false,
}),
ContentPart::ToolCall(ToolCall::new(
"call_2",
"get_weather",
serde_json::json!({"city": "NYC"}),
)),
],
name: None,
tool_call_id: None,
};
let translated = translate_messages(&[msg]);
assert_eq!(
translated[0].reasoning_content.as_deref(),
Some("I need the weather tool.")
);
assert_eq!(translated[0].tool_calls.as_ref().unwrap().len(), 1);
let json = serde_json::to_value(&translated[0]).unwrap();
assert_eq!(json["reasoning_content"], "I need the weather tool.");
}
#[test]
fn translate_assistant_message_with_raw_arguments() {
let mut tc = ToolCall::new("call_3", "search", serde_json::json!({"q": "rust"}));

View file

@ -44,7 +44,7 @@ pub(super) struct ChatMessage {
#[serde(skip_serializing_if = "Option::is_none")]
pub content: Option<ChatContent>,
/// Reasoning/thinking content echoed back for providers that require it
/// (Kimi).
/// during tool-call continuations (including Kimi and DeepSeek).
#[serde(skip_serializing_if = "Option::is_none")]
pub reasoning_content: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]

View file

@ -621,6 +621,17 @@ async fn deepseek_complete() {
assert_eq!(response.provider, "deepseek");
}
#[fabro_macros::e2e_test(live("DEEPSEEK_API_KEY"))]
async fn deepseek_v4_flash_deep_tool_round_trip() {
let api_key = std::env::var(EnvVars::DEEPSEEK_API_KEY).expect("DEEPSEEK_API_KEY must be set");
let provider = ProviderId::new("deepseek");
let catalog = enabled_provider_catalog(&provider, None);
let credential = ApiCredential::from_api_key(provider.clone(), api_key, &catalog)
.expect("DeepSeek credential should resolve from the catalog");
assert_deep_tool_round_trip(&catalog, &provider, "deepseek-v4-flash", credential).await;
}
#[fabro_macros::e2e_test(live("FIREWORKS_API_KEY"))]
async fn fireworks_kimi_k2_7_code_deep_tool_round_trip() {
let api_key = std::env::var(EnvVars::FIREWORKS_API_KEY).expect("FIREWORKS_API_KEY must be set");

View file

@ -185,8 +185,8 @@ async fn encode_tool_round_trip() {
fabro_test::fabro_json_snapshot!(capture.body);
}
/// Assistant thinking parts echo back as `reasoning_content` (Kimi-motivated,
/// applies to every compat assistant message).
/// Assistant thinking parts echo back as `reasoning_content` (required by
/// Kimi and DeepSeek during tool-call continuations).
#[tokio::test]
async fn encode_thinking_round_trip_as_reasoning_content() {
let capture = encode_capture(&corpus_thinking_round_trip(MODEL)).await;

View file

@ -3099,6 +3099,72 @@ enabled = true
}
}
#[test]
fn builtin_deepseek_reasoning_controls_match_provider_dialects() {
let catalog = Catalog::from_builtin_with_overrides(&minimal_settings(
r"
[providers.fireworks]
enabled = true
[providers.openrouter]
enabled = true
",
))
.expect("DeepSeek gateway providers should build when enabled");
let expected = [
(ProviderId::new("deepseek"), "deepseek-v4-flash", vec![
ReasoningEffort::Low,
ReasoningEffort::High,
ReasoningEffort::Max,
]),
(ProviderId::new("deepseek"), "deepseek-v4-pro", vec![
ReasoningEffort::High,
ReasoningEffort::Max,
]),
(ProviderId::new("fireworks"), "deepseek-v4-flash", vec![
ReasoningEffort::High,
ReasoningEffort::Max,
]),
(ProviderId::new("fireworks"), "deepseek-v4-pro", vec![
ReasoningEffort::High,
ReasoningEffort::Max,
]),
(ProviderId::new("openrouter"), "deepseek-v4-flash", vec![
ReasoningEffort::Low,
ReasoningEffort::High,
ReasoningEffort::Max,
]),
(ProviderId::new("openrouter"), "deepseek-v4-pro", vec![
ReasoningEffort::High,
ReasoningEffort::XHigh,
]),
];
for (provider, id, efforts) in expected {
let model = catalog
.get_on_provider(&provider, id)
.unwrap_or_else(|| panic!("{provider}/{id} should be present"));
assert!(model.features.reasoning, "{provider}/{id}");
assert_eq!(
model.features.reasoning_effort,
ReasoningEffortFeature::Levels,
"{provider}/{id}"
);
assert_eq!(model.controls.reasoning_effort, efforts, "{provider}/{id}");
assert!(!model.features.sampling_params, "{provider}/{id}");
let settings = catalog
.model_settings_on_provider(&provider, id)
.unwrap_or_else(|| panic!("{provider}/{id} settings should be present"));
assert!(settings.reasoning_by_default, "{provider}/{id}");
assert_eq!(
settings.controls.reasoning_effort, efforts,
"{provider}/{id}"
);
}
}
#[test]
fn builtin_openrouter_provider_is_opt_in() {
let openrouter = ProviderId::new("openrouter");
@ -3154,6 +3220,12 @@ enabled = true
catalog.settings_for(deepseek).unwrap().api_id,
"deepseek/deepseek-v4-flash-0731"
);
let deepseek_pro = catalog
.get_on_provider(&openrouter, "deepseek-v4-pro")
.expect("DeepSeek V4 Pro should be present on OpenRouter");
assert_eq!(deepseek_pro.limits.max_output, Some(384_000));
assert!(deepseek_pro.features.prompt_cache);
assert_eq!(deepseek_pro.costs.cache_input_cost_per_mtok, Some(0.003625));
assert_eq!(
catalog
.default_for_provider(&openrouter)
@ -3962,7 +4034,7 @@ enabled = true
1_048_576,
384_000,
false,
false,
true,
0.14,
0.28,
0.028,

View file

@ -30,10 +30,14 @@ max_output = 384000
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.deepseek.models."deepseek-v4-flash".controls]
reasoning_effort = ["low", "high", "max"]
[providers.deepseek.models."deepseek-v4-flash".costs]
input_cost_per_mtok = 0.14
output_cost_per_mtok = 0.28
@ -51,10 +55,16 @@ max_output = 384000
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.deepseek.models."deepseek-v4-pro".controls]
# V4 Pro currently maps low to high. Keep only its distinct effort levels;
# DeepSeek says it plans to change Pro's mapping in early August 2026.
reasoning_effort = ["high", "max"]
[providers.deepseek.models."deepseek-v4-pro".costs]
input_cost_per_mtok = 0.435
output_cost_per_mtok = 0.87

View file

@ -81,7 +81,14 @@ max_output = 16384
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.fireworks.models."deepseek-v4-pro".controls]
# Fireworks promotes low/medium to high and xhigh to max for DeepSeek V4.
reasoning_effort = ["high", "max"]
[providers.fireworks.models."deepseek-v4-pro".costs]
input_cost_per_mtok = 1.74
@ -101,8 +108,14 @@ max_output = 384000
[providers.fireworks.models."deepseek-v4-flash".features]
tools = true
vision = false
reasoning = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.fireworks.models."deepseek-v4-flash".controls]
reasoning_effort = ["high", "max"]
[providers.fireworks.models."deepseek-v4-flash".costs]
input_cost_per_mtok = 0.14

View file

@ -419,16 +419,25 @@ family = "deepseek-v4"
[providers.openrouter.models."deepseek-v4-pro".limits]
context_window = 1050000
max_output = 16384
max_output = 384000
[providers.openrouter.models."deepseek-v4-pro".features]
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.openrouter.models."deepseek-v4-pro".controls]
# OpenRouter names DeepSeek's max tier xhigh on this route.
reasoning_effort = ["high", "xhigh"]
[providers.openrouter.models."deepseek-v4-pro".costs]
input_cost_per_mtok = 0.435
output_cost_per_mtok = 0.87
cache_input_cost_per_mtok = 0.003625
[providers.openrouter.models."deepseek-v4-flash"]
api_id = "deepseek/deepseek-v4-flash-0731"
@ -443,8 +452,14 @@ max_output = 384000
[providers.openrouter.models."deepseek-v4-flash".features]
tools = true
vision = false
reasoning = false
reasoning = true
reasoning_effort = "levels"
reasoning_by_default = true
prompt_cache = true
sampling_params = false
[providers.openrouter.models."deepseek-v4-flash".controls]
reasoning_effort = ["low", "high", "max"]
[providers.openrouter.models."deepseek-v4-flash".costs]
input_cost_per_mtok = 0.14