From f098fb9104651d1689eb21124c867a6182fb31e3 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Thu, 30 Jul 2026 17:28:24 -0400 Subject: [PATCH 1/2] fix(llm): decode Modal reasoning token usage --- .../src/codec/openai_compatible/wire.rs | 75 ++++++++++++++++++- 1 file changed, 73 insertions(+), 2 deletions(-) diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs index 3e1de949d..88fb02672 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs @@ -294,14 +294,18 @@ pub(super) struct ApiFunction { pub(super) struct ApiUsage { pub prompt_tokens: i64, pub completion_tokens: i64, - /// Tolerant superset: aggregator dialects (OpenRouter) report in-band - /// USD cost and cache/reasoning token detail. Absent on plain providers. + /// Tolerant superset: compatible provider dialects report in-band USD cost + /// and cache/reasoning token detail. Absent on plain providers. #[serde(default)] pub cost: Option, #[serde(default)] pub prompt_tokens_details: Option, #[serde(default)] pub completion_tokens_details: Option, + /// Modal reports reasoning tokens directly on `usage` instead of nesting + /// them under `completion_tokens_details`. + #[serde(default)] + pub reasoning_tokens: Option, } #[derive(serde::Deserialize)] @@ -339,6 +343,7 @@ impl ApiUsage { .completion_tokens_details .as_ref() .and_then(|d| d.reasoning_tokens) + .or(self.reasoning_tokens) .unwrap_or(0); let (uncached_input, cached) = split_inclusive_token_total(self.prompt_tokens, cached_detail); @@ -538,6 +543,72 @@ mod tests { }); } + #[test] + fn non_streaming_usage_normalizes_modal_reasoning_tokens() { + let response: ApiResponse = serde_json::from_value(serde_json::json!({ + "id": "chatcmpl-modal", + "model": "moonshotai/Kimi-K3", + "choices": [{ + "message": { + "content": "1275", + "reasoning_content": "The arithmetic series sums to 1275." + }, + "finish_reason": "stop" + }], + "usage": { + "prompt_tokens": 116, + "completion_tokens": 66, + "prompt_tokens_details": { + "cached_tokens": 64 + }, + "reasoning_tokens": 54, + "total_tokens": 182 + } + })) + .unwrap(); + let usage = response + .usage + .expect("Modal response should include token usage"); + + assert_eq!(usage.token_counts(), TokenCounts { + input_tokens: 52, + output_tokens: 12, + reasoning_tokens: 54, + cache_read_tokens: 64, + ..TokenCounts::default() + }); + } + + #[test] + fn streaming_usage_normalizes_modal_reasoning_tokens() { + let chunk: StreamChunk = serde_json::from_value(serde_json::json!({ + "id": "chatcmpl-modal", + "model": "moonshotai/Kimi-K3", + "choices": [], + "usage": { + "prompt_tokens": 116, + "completion_tokens": 66, + "prompt_tokens_details": { + "cached_tokens": 64 + }, + "reasoning_tokens": 54, + "total_tokens": 182 + } + })) + .unwrap(); + let usage = chunk + .usage + .expect("Modal stream should include token usage"); + + assert_eq!(usage.token_counts(), TokenCounts { + input_tokens: 52, + output_tokens: 12, + reasoning_tokens: 54, + cache_read_tokens: 64, + ..TokenCounts::default() + }); + } + #[test] fn reasoning_accepts_provider_and_openrouter_spellings() { let provider_response: ApiResponse = serde_json::from_value(serde_json::json!({ From 641539dd4b66b62615dd5efe1f4b735356ecf12f Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 1 Aug 2026 09:44:13 -0400 Subject: [PATCH 2/2] refactor(llm): tighten Modal reasoning token tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the two envelope-level tests with focused `ApiUsage` tests that match the file's existing `token_counts_*` convention. The streaming and non-streaming tests were the same test paid for twice: `ApiResponse::usage` and `StreamChunk::usage` are both `Option`, so the envelope cannot change the result. Envelope-level usage decoding is already covered by `stream_chunk_usage_parsing`. Also pin the precedence rule this change introduces — nested detail wins over the flat spelling, and an empty `completion_tokens_details` still falls back — and document it on `token_counts`. Revert the unrelated `cost` doc edit that dropped the OpenRouter reference. Co-Authored-By: Claude Opus 5 (1M context) --- .../src/codec/openai_compatible/wire.rs | 111 +++++++----------- 1 file changed, 43 insertions(+), 68 deletions(-) diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs index 23cc0e260..62c6ede6b 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs @@ -294,8 +294,8 @@ pub(super) struct ApiFunction { pub(super) struct ApiUsage { pub prompt_tokens: i64, pub completion_tokens: i64, - /// Tolerant superset: compatible provider dialects report in-band USD cost - /// and cache/reasoning token detail. Absent on plain providers. + /// Tolerant superset: aggregator dialects (OpenRouter) report in-band + /// USD cost and cache/reasoning token detail. Absent on plain providers. #[serde(default)] pub cost: Option, #[serde(default)] @@ -332,6 +332,9 @@ impl ApiUsage { /// cache-write detail tokens are subtracted out of `input_tokens`, and /// reasoning tokens out of `output_tokens`, mirroring the /// `openai_responses` convention. + /// + /// Nested detail fields win over the flat `prompt_cache_hit_tokens` and + /// `reasoning_tokens` spellings that some providers send instead. pub(super) fn token_counts(&self) -> TokenCounts { let cached_detail = self .prompt_tokens_details @@ -548,72 +551,6 @@ mod tests { }); } - #[test] - fn non_streaming_usage_normalizes_modal_reasoning_tokens() { - let response: ApiResponse = serde_json::from_value(serde_json::json!({ - "id": "chatcmpl-modal", - "model": "moonshotai/Kimi-K3", - "choices": [{ - "message": { - "content": "1275", - "reasoning_content": "The arithmetic series sums to 1275." - }, - "finish_reason": "stop" - }], - "usage": { - "prompt_tokens": 116, - "completion_tokens": 66, - "prompt_tokens_details": { - "cached_tokens": 64 - }, - "reasoning_tokens": 54, - "total_tokens": 182 - } - })) - .unwrap(); - let usage = response - .usage - .expect("Modal response should include token usage"); - - assert_eq!(usage.token_counts(), TokenCounts { - input_tokens: 52, - output_tokens: 12, - reasoning_tokens: 54, - cache_read_tokens: 64, - ..TokenCounts::default() - }); - } - - #[test] - fn streaming_usage_normalizes_modal_reasoning_tokens() { - let chunk: StreamChunk = serde_json::from_value(serde_json::json!({ - "id": "chatcmpl-modal", - "model": "moonshotai/Kimi-K3", - "choices": [], - "usage": { - "prompt_tokens": 116, - "completion_tokens": 66, - "prompt_tokens_details": { - "cached_tokens": 64 - }, - "reasoning_tokens": 54, - "total_tokens": 182 - } - })) - .unwrap(); - let usage = chunk - .usage - .expect("Modal stream should include token usage"); - - assert_eq!(usage.token_counts(), TokenCounts { - input_tokens: 52, - output_tokens: 12, - reasoning_tokens: 54, - cache_read_tokens: 64, - ..TokenCounts::default() - }); - } - #[test] fn token_counts_accept_deepseek_cache_hit_field() { let usage: ApiUsage = serde_json::from_value(serde_json::json!({ @@ -631,6 +568,44 @@ mod tests { }); } + #[test] + fn token_counts_accept_modal_reasoning_tokens_field() { + let usage: ApiUsage = serde_json::from_value(serde_json::json!({ + "prompt_tokens": 116, + "completion_tokens": 66, + "reasoning_tokens": 54 + })) + .unwrap(); + + assert_eq!(usage.token_counts(), TokenCounts { + input_tokens: 116, + output_tokens: 12, + reasoning_tokens: 54, + ..TokenCounts::default() + }); + } + + #[test] + fn token_counts_prefer_nested_reasoning_detail_over_top_level() { + let both_spellings: ApiUsage = serde_json::from_value(serde_json::json!({ + "prompt_tokens": 10, + "completion_tokens": 66, + "completion_tokens_details": {"reasoning_tokens": 20}, + "reasoning_tokens": 54 + })) + .unwrap(); + assert_eq!(both_spellings.token_counts().reasoning_tokens, 20); + + let empty_detail: ApiUsage = serde_json::from_value(serde_json::json!({ + "prompt_tokens": 10, + "completion_tokens": 66, + "completion_tokens_details": {}, + "reasoning_tokens": 54 + })) + .unwrap(); + assert_eq!(empty_detail.token_counts().reasoning_tokens, 54); + } + #[test] fn reasoning_accepts_provider_and_openrouter_spellings() { let provider_response: ApiResponse = serde_json::from_value(serde_json::json!({