fix(streaming): ignore non-positive usage.cost on stream assemble

Vertex Anthropic streams never hit OpenRouter cost-header propagation, so
usage.cost=0 was stamped as response_cost in stream_chunk_builder and
logged as spend=0. Ignore non-positive assembled costs there (and when
stamping usage.cost) so token-based pricing runs.

Addresses greptile P1 on #40950.
This commit is contained in:
leilei3167 2026-09-16 15:08:52 +00:00
parent faf60c14e8
commit def3efa72b
2 changed files with 25 additions and 4 deletions

View file

@ -8700,7 +8700,13 @@ def _reported_cost_is_priced_by_calculator(logging_obj: Optional["Logging"]) ->
def _stream_builder_response_cost(response: ModelResponse, logging_obj: Optional["Logging"]) -> float | None:
usage_cost: Final = getattr(getattr(response, "usage", None), "cost", None)
if isinstance(usage_cost, (int, float)) and not _reported_cost_is_priced_by_calculator(logging_obj):
# Non-positive leftovers (common on Vertex Anthropic stream assembly) are not
# real provider totals; leave them absent so token-based pricing runs.
if (
isinstance(usage_cost, (int, float))
and usage_cost > 0
and not _reported_cost_is_priced_by_calculator(logging_obj)
):
return float(usage_cost)
if logging_obj is not None:
return None
@ -8742,7 +8748,9 @@ def _set_stream_builder_response_cost(response: ModelResponse, logging_obj: Opti
def _stamp_streaming_usage_cost(usage: Usage, response: ModelResponse, logging_obj: Optional["Logging"]) -> None:
if logging_obj is None:
return
if isinstance(getattr(usage, "cost", None), (int, float)):
existing_cost: Final = getattr(usage, "cost", None)
# Treat non-positive assembled costs as absent so token pricing can stamp a real total.
if isinstance(existing_cost, (int, float)) and existing_cost > 0:
return
computed_cost: Final = logging_obj._response_cost_calculator(result=response)
if isinstance(computed_cost, (int, float)) and computed_cost > 0:

View file

@ -2023,10 +2023,23 @@ def test_stream_spend_prices_vertex_anthropic_cache_read_tokens():
assert token_total > 0
# A leftover usage.cost=0 must not override token-based stream spend.
# Production path: stream_chunk_builder stamps hidden response_cost from usage.cost.
usage.cost = 0
CustomStreamWrapper._propagate_usage_cost_to_hidden_params(
complete_response, "vertex_ai"
logging_for_builder = Logging(
model="claude-opus-5",
messages=[{"role": "user", "content": "count to five"}],
stream=True,
call_type="completion",
start_time=time.time(),
litellm_call_id="stream-spend-cache-read-builder",
function_id="1245",
)
logging_for_builder.model_call_details["custom_llm_provider"] = "vertex_ai"
logging_for_builder.optional_params = {}
from litellm.main import _set_stream_builder_response_cost
_set_stream_builder_response_cost(complete_response, logging_for_builder)
assert complete_response._hidden_params.get("response_cost") is None
assert get_response_cost_from_hidden_params(complete_response._hidden_params) is None
stream_spend = _stream_spend_via_logging(