diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/request.rs b/lib/components/fabro-llm/src/codec/openai_compatible/request.rs index 12be9e84b..1ec6dc8fb 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/request.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/request.rs @@ -1,7 +1,7 @@ //! Request encoding: canonical `Request` → Chat Completions body. use super::translate; -use super::wire::{ApiRequest, ChatMessage}; +use super::wire::{ApiRequest, ChatMessage, StreamOptions}; use crate::codec::{CodecCtx, EncodedRequest, cache, merge_named_provider_options}; use crate::error::Error; @@ -10,8 +10,10 @@ use crate::error::Error; const KNOWN_OPTION_KEYS: &[&str] = &["auto_cache"]; /// Build the Chat Completions request for `ctx.request`. `stream` toggles the -/// `stream` body field. The body is assembled as a `serde_json::Value` so -/// `provider_options.` fields can be merged in before sending. +/// `stream` body field and the `stream_options.include_usage` opt-in that makes +/// providers emit the trailing usage chunk. The body is assembled as a +/// `serde_json::Value` so `provider_options.` fields can be +/// merged in before sending. /// /// Returns an error when the request contains a custom tool definition, which /// the Chat Completions tool envelope cannot represent. @@ -55,6 +57,9 @@ pub(super) fn encode(ctx: &CodecCtx<'_>, stream: bool) -> Result` merge runs after the body is built, so a + /// caller pointed at a gateway that rejects the field can still turn it + /// off. + #[test] + fn provider_options_can_override_stream_options() { + let mut request = minimal_request(); + request.provider_options = Some(serde_json::json!({ + "kimi": { "stream_options": serde_json::Value::Null } + })); + + let body = encode_body(&request, "kimi", true); + + assert_eq!(body["stream_options"], serde_json::Value::Null); } #[test] diff --git a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs index 5bc0a3b93..31019a230 100644 --- a/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs +++ b/lib/components/fabro-llm/src/codec/openai_compatible/wire.rs @@ -26,6 +26,16 @@ pub(super) struct ApiRequest { pub response_format: Option, #[serde(skip_serializing_if = "Option::is_none")] pub stream: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub stream_options: Option, +} + +/// Streaming options. Chat Completions only emits the trailing usage chunk +/// when the request opts in, so without this a streamed response reports zero +/// tokens and costs are estimated at $0. +#[derive(serde::Serialize)] +pub(super) struct StreamOptions { + pub include_usage: bool, } #[derive(serde::Serialize)] diff --git a/lib/components/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_text_happy_path_request.snap b/lib/components/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_text_happy_path_request.snap index fa711fab9..297b156fa 100644 --- a/lib/components/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_text_happy_path_request.snap +++ b/lib/components/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_text_happy_path_request.snap @@ -11,5 +11,8 @@ expression: rendered } ], "max_tokens": 128, - "stream": true + "stream": true, + "stream_options": { + "include_usage": true + } }