mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(streaming): ignore non-positive usage.cost on stream assemble
Vertex Anthropic streams never hit OpenRouter cost-header propagation, so usage.cost=0 was stamped as response_cost in stream_chunk_builder and logged as spend=0. Ignore non-positive assembled costs there (and when stamping usage.cost) so token-based pricing runs. Addresses greptile P1 on #40950.
This commit is contained in:
parent
faf60c14e8
commit
def3efa72b
2 changed files with 25 additions and 4 deletions
|
|
@ -8700,7 +8700,13 @@ def _reported_cost_is_priced_by_calculator(logging_obj: Optional["Logging"]) ->
|
|||
|
||||
def _stream_builder_response_cost(response: ModelResponse, logging_obj: Optional["Logging"]) -> float | None:
|
||||
usage_cost: Final = getattr(getattr(response, "usage", None), "cost", None)
|
||||
if isinstance(usage_cost, (int, float)) and not _reported_cost_is_priced_by_calculator(logging_obj):
|
||||
# Non-positive leftovers (common on Vertex Anthropic stream assembly) are not
|
||||
# real provider totals; leave them absent so token-based pricing runs.
|
||||
if (
|
||||
isinstance(usage_cost, (int, float))
|
||||
and usage_cost > 0
|
||||
and not _reported_cost_is_priced_by_calculator(logging_obj)
|
||||
):
|
||||
return float(usage_cost)
|
||||
if logging_obj is not None:
|
||||
return None
|
||||
|
|
@ -8742,7 +8748,9 @@ def _set_stream_builder_response_cost(response: ModelResponse, logging_obj: Opti
|
|||
def _stamp_streaming_usage_cost(usage: Usage, response: ModelResponse, logging_obj: Optional["Logging"]) -> None:
|
||||
if logging_obj is None:
|
||||
return
|
||||
if isinstance(getattr(usage, "cost", None), (int, float)):
|
||||
existing_cost: Final = getattr(usage, "cost", None)
|
||||
# Treat non-positive assembled costs as absent so token pricing can stamp a real total.
|
||||
if isinstance(existing_cost, (int, float)) and existing_cost > 0:
|
||||
return
|
||||
computed_cost: Final = logging_obj._response_cost_calculator(result=response)
|
||||
if isinstance(computed_cost, (int, float)) and computed_cost > 0:
|
||||
|
|
|
|||
|
|
@ -2023,10 +2023,23 @@ def test_stream_spend_prices_vertex_anthropic_cache_read_tokens():
|
|||
assert token_total > 0
|
||||
|
||||
# A leftover usage.cost=0 must not override token-based stream spend.
|
||||
# Production path: stream_chunk_builder stamps hidden response_cost from usage.cost.
|
||||
usage.cost = 0
|
||||
CustomStreamWrapper._propagate_usage_cost_to_hidden_params(
|
||||
complete_response, "vertex_ai"
|
||||
logging_for_builder = Logging(
|
||||
model="claude-opus-5",
|
||||
messages=[{"role": "user", "content": "count to five"}],
|
||||
stream=True,
|
||||
call_type="completion",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="stream-spend-cache-read-builder",
|
||||
function_id="1245",
|
||||
)
|
||||
logging_for_builder.model_call_details["custom_llm_provider"] = "vertex_ai"
|
||||
logging_for_builder.optional_params = {}
|
||||
from litellm.main import _set_stream_builder_response_cost
|
||||
|
||||
_set_stream_builder_response_cost(complete_response, logging_for_builder)
|
||||
assert complete_response._hidden_params.get("response_cost") is None
|
||||
assert get_response_cost_from_hidden_params(complete_response._hidden_params) is None
|
||||
|
||||
stream_spend = _stream_spend_via_logging(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue