test(llm): cover non-streaming cached-token parsing in openai-compatible

Mirror the existing stream_chunk_usage_parses_cached_tokens test for
the non-streaming ApiResponse path. The non-streaming complete()
flow uses a different deserialization target (ApiResponse vs
StreamChunk) and a different TokenCounts construction site, so a
regression in either is now caught by its own test rather than only
by the live e2e test.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Scott Werner 2026-05-27 16:03:35 -04:00
parent 8bc8eae3f1
commit 5983c52434

View file

@ -1072,6 +1072,25 @@ mod tests {
assert!(usage.cache_write_tokens.is_none());
}
#[test]
fn api_response_usage_parses_cached_tokens() {
// Non-streaming response uses ApiResponse, which is a distinct
// code path from the streaming StreamChunk parse. Verify the
// same cached-token fields land on TokenCounts via that path.
let json = r#"{"id":"chatcmpl-1","model":"anthropic/claude-sonnet-4.6","choices":[{"message":{"role":"assistant","content":"hi"},"finish_reason":"stop"}],"usage":{"prompt_tokens":100,"completion_tokens":50,"prompt_tokens_details":{"cached_tokens":80},"cache_write_tokens":12}}"#;
let resp: ApiResponse = serde_json::from_str(json).unwrap();
let u = resp.usage.unwrap();
assert_eq!(u.prompt_tokens, 100);
assert_eq!(u.completion_tokens, 50);
assert_eq!(
u.prompt_tokens_details
.as_ref()
.and_then(|d| d.cached_tokens),
Some(80)
);
assert_eq!(u.cache_write_tokens, Some(12));
}
#[test]
fn stream_chunk_usage_parses_cached_tokens() {
let json = r#"{"id":"chatcmpl-1","model":"gpt-4","choices":[],"usage":{"prompt_tokens":100,"completion_tokens":50,"prompt_tokens_details":{"cached_tokens":80},"cache_write_tokens":12}}"#;