test(main): pin what a streamed response costs, end to end (#37812)

Rebuilding a streamed response and pricing it is the path a spend row comes
from, and nothing asserted it end to end. Reversing either half of the usage
the provider reported left the file green.

Three cases: the rebuilt response bills the usage the last chunk carried,
streaming and not streaming bill the same usage the same, and a stream that
reported no usage is still billed rather than dropped.

The cost is asserted against the catalog prices the run itself reads, with a
non-zero guard in front of it so an all-zeros lookup cannot satisfy it
vacuously. Pinning the dollar figure as a literal would have made a routine
gpt-4o price update fail a test about usage reconstruction.
This commit is contained in:
yuneng-jiang 2026-08-21 20:15:27 -07:00 • committed by GitHub
parent 35fcc9f7b8
commit b416bdadd3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -2754,3 +2754,109 @@ def test_completion_default_api_base_sends_prompt_cache_breakpoint_for_gpt_5_6()
{"type": "text", "text": "sys", "prompt_cache_breakpoint": {"mode": "explicit"}}
]
assert request_body["extra_body"]["prompt_cache_options"] == {"mode": "explicit"}
STREAM_COST_MODEL = "gpt-4o"
STREAMED_USAGE = {"prompt_tokens": 137, "completion_tokens": 42, "total_tokens": 179}
def _text_chunk(content, finish_reason=None, usage=None):
chunk = {
"id": "chatcmpl-stream-cost",
"object": "chat.completion.chunk",
"created": 1700000000,
"model": STREAM_COST_MODEL,
"choices": [
{
"index": 0,
"delta": {"role": "assistant", "content": content},
"finish_reason": finish_reason,
}
],
}
if usage is not None:
chunk["usage"] = usage
return chunk
def _priced_at(prompt_tokens, completion_tokens):
prices = litellm.model_cost[STREAM_COST_MODEL]
return (
prompt_tokens * prices["input_cost_per_token"]
+ completion_tokens * prices["output_cost_per_token"]
)
@pytest.fixture
def local_cost_map(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_map):
rebuilt = litellm.stream_chunk_builder(
chunks=[
_text_chunk("Hello"),
_text_chunk(" there"),
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
],
messages=[{"role": "user", "content": "hi"}],
)
assert rebuilt.choices[0].message.content == "Hello there"
assert rebuilt.usage.prompt_tokens == STREAMED_USAGE["prompt_tokens"]
assert rebuilt.usage.completion_tokens == STREAMED_USAGE["completion_tokens"]
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
assert cost == pytest.approx(_priced_at(137, 42))
assert cost == pytest.approx(0.0007625)
def test_streaming_and_not_streaming_bill_the_same_usage_the_same(local_cost_map):
rebuilt = litellm.stream_chunk_builder(
chunks=[
_text_chunk("Hello"),
_text_chunk(" there"),
_text_chunk(None, finish_reason="stop", usage=STREAMED_USAGE),
],
messages=[{"role": "user", "content": "hi"}],
)
whole = litellm.ModelResponse(
id="chatcmpl-stream-cost",
model=STREAM_COST_MODEL,
object="chat.completion",
created=1700000000,
choices=[
{
"index": 0,
"message": {"role": "assistant", "content": "Hello there"},
"finish_reason": "stop",
}
],
usage=STREAMED_USAGE,
)
assert litellm.completion_cost(
completion_response=rebuilt, model=STREAM_COST_MODEL
) == pytest.approx(litellm.completion_cost(completion_response=whole, model=STREAM_COST_MODEL))
def test_a_stream_that_reported_no_usage_is_still_billed(local_cost_map):
rebuilt = litellm.stream_chunk_builder(
chunks=[
_text_chunk("Hello"),
_text_chunk(" there"),
_text_chunk(None, finish_reason="stop"),
],
messages=[{"role": "user", "content": "hi"}],
)
assert rebuilt.usage.prompt_tokens > 0
assert rebuilt.usage.completion_tokens > 0
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
assert cost > 0
assert cost == pytest.approx(
_priced_at(rebuilt.usage.prompt_tokens, rebuilt.usage.completion_tokens)
)