From 76dae568f5e29eeafc3edfe8c6a293b2555cbe53 Mon Sep 17 00:00:00 2001 From: Release Repro Date: Fri, 31 Jul 2026 13:14:11 -0400 Subject: [PATCH] feat(reasoning): expose DeepSeek effort controls --- docs/public/integrations/deepseek.mdx | 33 ++++++++- .../src/codec/openai_compatible/translate.rs | 35 ++++++++- .../src/codec/openai_compatible/wire.rs | 2 +- lib/components/fabro-llm/tests/integration.rs | 11 +++ .../tests/it/wire/openai_compatible.rs | 4 +- lib/foundation/fabro-model/src/catalog.rs | 74 ++++++++++++++++++- .../src/catalog/providers/deepseek.toml | 10 +++ .../src/catalog/providers/fireworks.toml | 15 +++- .../src/catalog/providers/openrouter.toml | 19 ++++- 9 files changed, 194 insertions(+), 9 deletions(-) diff --git a/docs/public/integrations/deepseek.mdx b/docs/public/integrations/deepseek.mdx index 9180a0461..69d0474cc 100644 --- a/docs/public/integrations/deepseek.mdx +++ b/docs/public/integrations/deepseek.mdx @@ -70,6 +70,34 @@ digraph Example { } ``` +## Thinking and reasoning effort + +DeepSeek enables thinking by default at `high` effort. Fabro advertises only effort values that produce a distinct model behavior on each route: + +| Route | V4 Flash | V4 Pro | +|---|---|---| +| Direct DeepSeek | `low`, `high`, `max` | `high`, `max` | +| Fireworks AI | `high`, `max` | `high`, `max` | +| OpenRouter | `low`, `high`, `max` | `high`, `xhigh` | + +DeepSeek currently maps a V4 Pro request for `low` to `high`; its documentation says this mapping will change in early August 2026. Fireworks maps `low` and `medium` to `high`, and maps `xhigh` to `max`. OpenRouter names the Pro maximum tier `xhigh`. + +Fabro omits `temperature` and `top_p` for these models because DeepSeek ignores sampling parameters while thinking is enabled. + +To disable thinking on the direct provider, omit typed `reasoning_effort` and pass DeepSeek's native toggle through provider options: + +```json +{ + "provider_options": { + "deepseek": { + "thinking": { "type": "disabled" } + } + } +} +``` + +DeepSeek requires `reasoning_content` from an assistant tool call to appear in every later request in that tool-use turn. Fabro captures this content and replays it with the assistant message. This keeps multi-step tool calls valid and prevents DeepSeek's HTTP 400 response for missing reasoning history. + ## Prompt caching and pricing DeepSeek applies prefix caching automatically. Fabro reads DeepSeek's `prompt_cache_hit_tokens` usage field and reports cache-read tokens separately from uncached input tokens. @@ -99,11 +127,14 @@ See the [Fireworks AI integration](/integrations/fireworks) and [OpenRouter inte ## Further reading - + Official authentication, endpoints, and API reference. Official limits, features, and token prices. + + Official thinking toggles, effort mappings, and tool-call replay rules. + diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/translate.rs b/lib/components/fabro-llm/src/codec/openai_compatible/translate.rs index dbef4831f..b32fac567 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/translate.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/translate.rs @@ -237,7 +237,9 @@ pub(super) fn translate_response_format(format: &ResponseFormat) -> serde_json:: #[cfg(test)] mod tests { use super::*; - use crate::types::{AudioData, ContentPart, DocumentData, Message, Role, ToolCall}; + use crate::types::{ + AudioData, ContentPart, DocumentData, Message, Role, ThinkingData, ToolCall, + }; #[test] fn translate_assistant_message_with_tool_calls_only() { @@ -291,6 +293,37 @@ mod tests { assert_eq!(tool_calls[0].function.name, "get_weather"); } + #[test] + fn translate_assistant_tool_call_replays_reasoning_content() { + let msg = Message { + role: Role::Assistant, + content: vec![ + ContentPart::Thinking(ThinkingData { + text: "I need the weather tool.".to_string(), + signature: None, + redacted: false, + }), + ContentPart::ToolCall(ToolCall::new( + "call_2", + "get_weather", + serde_json::json!({"city": "NYC"}), + )), + ], + name: None, + tool_call_id: None, + }; + + let translated = translate_messages(&[msg]); + + assert_eq!( + translated[0].reasoning_content.as_deref(), + Some("I need the weather tool.") + ); + assert_eq!(translated[0].tool_calls.as_ref().unwrap().len(), 1); + let json = serde_json::to_value(&translated[0]).unwrap(); + assert_eq!(json["reasoning_content"], "I need the weather tool."); + } + #[test] fn translate_assistant_message_with_raw_arguments() { let mut tc = ToolCall::new("call_3", "search", serde_json::json!({"q": "rust"})); diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs index bc3883aa6..1b7ff3b33 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs @@ -44,7 +44,7 @@ pub(super) struct ChatMessage { #[serde(skip_serializing_if = "Option::is_none")] pub content: Option, /// Reasoning/thinking content echoed back for providers that require it - /// (Kimi). + /// during tool-call continuations (including Kimi and DeepSeek). #[serde(skip_serializing_if = "Option::is_none")] pub reasoning_content: Option, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/lib/components/fabro-llm/tests/integration.rs b/lib/components/fabro-llm/tests/integration.rs index 96b6adad8..b9790b5ad 100644 --- a/lib/components/fabro-llm/tests/integration.rs +++ b/lib/components/fabro-llm/tests/integration.rs @@ -621,6 +621,17 @@ async fn deepseek_complete() { assert_eq!(response.provider, "deepseek"); } +#[fabro_macros::e2e_test(live("DEEPSEEK_API_KEY"))] +async fn deepseek_v4_flash_deep_tool_round_trip() { + let api_key = std::env::var(EnvVars::DEEPSEEK_API_KEY).expect("DEEPSEEK_API_KEY must be set"); + let provider = ProviderId::new("deepseek"); + let catalog = enabled_provider_catalog(&provider, None); + let credential = ApiCredential::from_api_key(provider.clone(), api_key, &catalog) + .expect("DeepSeek credential should resolve from the catalog"); + + assert_deep_tool_round_trip(&catalog, &provider, "deepseek-v4-flash", credential).await; +} + #[fabro_macros::e2e_test(live("FIREWORKS_API_KEY"))] async fn fireworks_kimi_k2_7_code_deep_tool_round_trip() { let api_key = std::env::var(EnvVars::FIREWORKS_API_KEY).expect("FIREWORKS_API_KEY must be set"); diff --git a/lib/components/fabro-llm/tests/it/wire/openai_compatible.rs b/lib/components/fabro-llm/tests/it/wire/openai_compatible.rs index 1294e2e8a..1074cc222 100644 --- a/lib/components/fabro-llm/tests/it/wire/openai_compatible.rs +++ b/lib/components/fabro-llm/tests/it/wire/openai_compatible.rs @@ -185,8 +185,8 @@ async fn encode_tool_round_trip() { fabro_test::fabro_json_snapshot!(capture.body); } -/// Assistant thinking parts echo back as `reasoning_content` (Kimi-motivated, -/// applies to every compat assistant message). +/// Assistant thinking parts echo back as `reasoning_content` (required by +/// Kimi and DeepSeek during tool-call continuations). #[tokio::test] async fn encode_thinking_round_trip_as_reasoning_content() { let capture = encode_capture(&corpus_thinking_round_trip(MODEL)).await; diff --git a/lib/foundation/fabro-model/src/catalog.rs b/lib/foundation/fabro-model/src/catalog.rs index 3eb647f0c..3484b130b 100644 --- a/lib/foundation/fabro-model/src/catalog.rs +++ b/lib/foundation/fabro-model/src/catalog.rs @@ -3099,6 +3099,72 @@ enabled = true } } + #[test] + fn builtin_deepseek_reasoning_controls_match_provider_dialects() { + let catalog = Catalog::from_builtin_with_overrides(&minimal_settings( + r" +[providers.fireworks] +enabled = true + +[providers.openrouter] +enabled = true +", + )) + .expect("DeepSeek gateway providers should build when enabled"); + + let expected = [ + (ProviderId::new("deepseek"), "deepseek-v4-flash", vec![ + ReasoningEffort::Low, + ReasoningEffort::High, + ReasoningEffort::Max, + ]), + (ProviderId::new("deepseek"), "deepseek-v4-pro", vec![ + ReasoningEffort::High, + ReasoningEffort::Max, + ]), + (ProviderId::new("fireworks"), "deepseek-v4-flash", vec![ + ReasoningEffort::High, + ReasoningEffort::Max, + ]), + (ProviderId::new("fireworks"), "deepseek-v4-pro", vec![ + ReasoningEffort::High, + ReasoningEffort::Max, + ]), + (ProviderId::new("openrouter"), "deepseek-v4-flash", vec![ + ReasoningEffort::Low, + ReasoningEffort::High, + ReasoningEffort::Max, + ]), + (ProviderId::new("openrouter"), "deepseek-v4-pro", vec![ + ReasoningEffort::High, + ReasoningEffort::XHigh, + ]), + ]; + + for (provider, id, efforts) in expected { + let model = catalog + .get_on_provider(&provider, id) + .unwrap_or_else(|| panic!("{provider}/{id} should be present")); + assert!(model.features.reasoning, "{provider}/{id}"); + assert_eq!( + model.features.reasoning_effort, + ReasoningEffortFeature::Levels, + "{provider}/{id}" + ); + assert_eq!(model.controls.reasoning_effort, efforts, "{provider}/{id}"); + assert!(!model.features.sampling_params, "{provider}/{id}"); + + let settings = catalog + .model_settings_on_provider(&provider, id) + .unwrap_or_else(|| panic!("{provider}/{id} settings should be present")); + assert!(settings.reasoning_by_default, "{provider}/{id}"); + assert_eq!( + settings.controls.reasoning_effort, efforts, + "{provider}/{id}" + ); + } + } + #[test] fn builtin_openrouter_provider_is_opt_in() { let openrouter = ProviderId::new("openrouter"); @@ -3154,6 +3220,12 @@ enabled = true catalog.settings_for(deepseek).unwrap().api_id, "deepseek/deepseek-v4-flash-0731" ); + let deepseek_pro = catalog + .get_on_provider(&openrouter, "deepseek-v4-pro") + .expect("DeepSeek V4 Pro should be present on OpenRouter"); + assert_eq!(deepseek_pro.limits.max_output, Some(384_000)); + assert!(deepseek_pro.features.prompt_cache); + assert_eq!(deepseek_pro.costs.cache_input_cost_per_mtok, Some(0.003625)); assert_eq!( catalog .default_for_provider(&openrouter) @@ -3962,7 +4034,7 @@ enabled = true 1_048_576, 384_000, false, - false, + true, 0.14, 0.28, 0.028, diff --git a/lib/foundation/fabro-model/src/catalog/providers/deepseek.toml b/lib/foundation/fabro-model/src/catalog/providers/deepseek.toml index f7b073d72..17d06efa8 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/deepseek.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/deepseek.toml @@ -30,10 +30,14 @@ max_output = 384000 tools = true vision = false reasoning = true +reasoning_effort = "levels" reasoning_by_default = true prompt_cache = true sampling_params = false +[providers.deepseek.models."deepseek-v4-flash".controls] +reasoning_effort = ["low", "high", "max"] + [providers.deepseek.models."deepseek-v4-flash".costs] input_cost_per_mtok = 0.14 output_cost_per_mtok = 0.28 @@ -51,10 +55,16 @@ max_output = 384000 tools = true vision = false reasoning = true +reasoning_effort = "levels" reasoning_by_default = true prompt_cache = true sampling_params = false +[providers.deepseek.models."deepseek-v4-pro".controls] +# V4 Pro currently maps low to high. Keep only its distinct effort levels; +# DeepSeek says it plans to change Pro's mapping in early August 2026. +reasoning_effort = ["high", "max"] + [providers.deepseek.models."deepseek-v4-pro".costs] input_cost_per_mtok = 0.435 output_cost_per_mtok = 0.87 diff --git a/lib/foundation/fabro-model/src/catalog/providers/fireworks.toml b/lib/foundation/fabro-model/src/catalog/providers/fireworks.toml index 6701af919..d6f1ae412 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/fireworks.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/fireworks.toml @@ -81,7 +81,14 @@ max_output = 16384 tools = true vision = false reasoning = true +reasoning_effort = "levels" +reasoning_by_default = true prompt_cache = true +sampling_params = false + +[providers.fireworks.models."deepseek-v4-pro".controls] +# Fireworks promotes low/medium to high and xhigh to max for DeepSeek V4. +reasoning_effort = ["high", "max"] [providers.fireworks.models."deepseek-v4-pro".costs] input_cost_per_mtok = 1.74 @@ -101,8 +108,14 @@ max_output = 384000 [providers.fireworks.models."deepseek-v4-flash".features] tools = true vision = false -reasoning = false +reasoning = true +reasoning_effort = "levels" +reasoning_by_default = true prompt_cache = true +sampling_params = false + +[providers.fireworks.models."deepseek-v4-flash".controls] +reasoning_effort = ["high", "max"] [providers.fireworks.models."deepseek-v4-flash".costs] input_cost_per_mtok = 0.14 diff --git a/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml b/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml index daf00a77e..9d8fa87dc 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml @@ -419,16 +419,25 @@ family = "deepseek-v4" [providers.openrouter.models."deepseek-v4-pro".limits] context_window = 1050000 -max_output = 16384 +max_output = 384000 [providers.openrouter.models."deepseek-v4-pro".features] tools = true vision = false reasoning = true +reasoning_effort = "levels" +reasoning_by_default = true +prompt_cache = true +sampling_params = false + +[providers.openrouter.models."deepseek-v4-pro".controls] +# OpenRouter names DeepSeek's max tier xhigh on this route. +reasoning_effort = ["high", "xhigh"] [providers.openrouter.models."deepseek-v4-pro".costs] input_cost_per_mtok = 0.435 output_cost_per_mtok = 0.87 +cache_input_cost_per_mtok = 0.003625 [providers.openrouter.models."deepseek-v4-flash"] api_id = "deepseek/deepseek-v4-flash-0731" @@ -443,8 +452,14 @@ max_output = 384000 [providers.openrouter.models."deepseek-v4-flash".features] tools = true vision = false -reasoning = false +reasoning = true +reasoning_effort = "levels" +reasoning_by_default = true prompt_cache = true +sampling_params = false + +[providers.openrouter.models."deepseek-v4-flash".controls] +reasoning_effort = ["low", "high", "max"] [providers.openrouter.models."deepseek-v4-flash".costs] input_cost_per_mtok = 0.14