fix(llm): request streaming usage on openai_compatible providers

Chat Completions only emits the trailing usage chunk when the request sets
`stream_options: {"include_usage": true}`. The openai_compatible codec never
sent it, so providers that follow the spec strictly returned no usage at all
on streamed responses. Every message came back with zero tokens, and the
catalog cost estimate multiplied those zeros into $0.

Kimi is the visible case: a run's kimi-k3 stages report 0 tokens and no
dollars, while an openrouter stage in the same run bills normally because
OpenRouter volunteers usage (and an in-band cost) without being asked.

Send the opt-in whenever we stream. Providers that already volunteer usage
accept the field and are unaffected.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Bryan Helmkamp 2026-07-24 16:21:00 -04:00
parent 280afbb649
commit 0169725b4e
No known key found for this signature in database
3 changed files with 65 additions and 4 deletions

View file

@ -1,7 +1,7 @@
//! Request encoding: canonical `Request` → Chat Completions body.
use super::translate;
use super::wire::{ApiRequest, ChatMessage};
use super::wire::{ApiRequest, ChatMessage, StreamOptions};
use crate::codec::{CodecCtx, EncodedRequest, cache, merge_named_provider_options};
use crate::error::Error;
@ -10,8 +10,10 @@ use crate::error::Error;
const KNOWN_OPTION_KEYS: &[&str] = &["auto_cache"];
/// Build the Chat Completions request for `ctx.request`. `stream` toggles the
/// `stream` body field. The body is assembled as a `serde_json::Value` so
/// `provider_options.<provider_name>` fields can be merged in before sending.
/// `stream` body field and the `stream_options.include_usage` opt-in that makes
/// providers emit the trailing usage chunk. The body is assembled as a
/// `serde_json::Value` so `provider_options.<provider_name>` fields can be
/// merged in before sending.
///
/// Returns an error when the request contains a custom tool definition, which
/// the Chat Completions tool envelope cannot represent.
@ -55,6 +57,9 @@ pub(super) fn encode(ctx: &CodecCtx<'_>, stream: bool) -> Result<EncodedRequest,
tool_choice,
response_format,
stream: stream.then_some(true),
stream_options: stream.then_some(StreamOptions {
include_usage: true,
}),
};
let mut body = serde_json::to_value(&api_request).unwrap_or_default();
@ -172,9 +177,13 @@ mod tests {
tool_choice: None,
response_format: None,
stream: Some(true),
stream_options: Some(StreamOptions {
include_usage: true,
}),
};
let json = serde_json::to_value(&req).unwrap();
assert_eq!(json["stream"], true);
assert_eq!(json["stream_options"]["include_usage"], true);
let req_no_stream = ApiRequest {
model: "test".into(),
@ -188,9 +197,48 @@ mod tests {
tool_choice: None,
response_format: None,
stream: None,
stream_options: None,
};
let json_no_stream = serde_json::to_value(&req_no_stream).unwrap();
assert!(json_no_stream.get("stream").is_none());
assert!(json_no_stream.get("stream_options").is_none());
}
/// Chat Completions only emits the trailing usage chunk when the request
/// opts in; without it streamed responses report zero tokens and cost
/// estimation silently produces $0.
#[test]
fn encode_opts_into_streaming_usage_when_streaming() {
let request = minimal_request();
let body = encode_body(&request, "kimi", true);
assert_eq!(body["stream"], true);
assert_eq!(body["stream_options"]["include_usage"], true);
}
#[test]
fn encode_omits_stream_options_when_not_streaming() {
let request = minimal_request();
let body = encode_body(&request, "kimi", false);
assert!(body.get("stream_options").is_none());
}
/// The `provider_options.<name>` merge runs after the body is built, so a
/// caller pointed at a gateway that rejects the field can still turn it
/// off.
#[test]
fn provider_options_can_override_stream_options() {
let mut request = minimal_request();
request.provider_options = Some(serde_json::json!({
"kimi": { "stream_options": serde_json::Value::Null }
}));
let body = encode_body(&request, "kimi", true);
assert_eq!(body["stream_options"], serde_json::Value::Null);
}
#[test]

View file

@ -26,6 +26,16 @@ pub(super) struct ApiRequest {
pub response_format: Option<serde_json::Value>,
#[serde(skip_serializing_if = "Option::is_none")]
pub stream: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub stream_options: Option<StreamOptions>,
}
/// Streaming options. Chat Completions only emits the trailing usage chunk
/// when the request opts in, so without this a streamed response reports zero
/// tokens and costs are estimated at $0.
#[derive(serde::Serialize)]
pub(super) struct StreamOptions {
pub include_usage: bool,
}
#[derive(serde::Serialize)]

View file

@ -11,5 +11,8 @@ expression: rendered
}
],
"max_tokens": 128,
"stream": true
"stream": true,
"stream_options": {
"include_usage": true
}
}