diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index bdbaee00c19..d90e85799a6 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -968,8 +968,9 @@ def generic_cost_per_token( usage.completion_tokens - reasoning_tokens - audio_tokens - image_tokens - video_tokens, ) else: - # No breakdown at all, all tokens are text tokens - text_tokens = usage.completion_tokens + # No breakdown at all, all tokens are text tokens. Clamped like + # the branch above, so a negative count cannot credit the budget. + text_tokens = max(0, usage.completion_tokens) is_text_tokens_total = True ## TEXT COST completion_cost = float(text_tokens) * completion_base_cost diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index ce90719789a..0f76bec2a0c 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -3841,3 +3841,70 @@ def test_generic_cost_per_token_grok_46_long_context(_local_model_cost_map): ) assert prompt_cost == pytest.approx(200_000 * 4e-06 + 50_000 * 1e-06) assert completion_cost == pytest.approx(1_000 * 1.2e-05) + + +def test_negative_completion_tokens_do_not_produce_negative_cost(): + """A negative completion count must not yield a negative cost. + + generic_cost_per_token guards three of its four token paths with max(0, ...). + The fourth - completion tokens with no modality breakdown, which is the + common case - multiplied the raw value straight into the cost, so a negative + count produced a negative cost and a negative total. + + That matters because these numbers feed spend tracking: a negative cost does + not merely report wrongly, it credits the budget. + """ + usage = Usage(prompt_tokens=0, completion_tokens=-100, total_tokens=-100) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gpt-4o", usage=usage, custom_llm_provider="openai" + ) + + assert completion_cost == 0.0 + assert prompt_cost + completion_cost >= 0.0 + + +def test_negative_completion_tokens_with_positive_prompt_keeps_total_positive(): + """The mixed case: a real prompt with a bad completion count. + + Before the fix this returned a negative total, so the request reduced + recorded spend instead of increasing it. + """ + usage = Usage(prompt_tokens=100, completion_tokens=-100, total_tokens=0) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gpt-4o", usage=usage, custom_llm_provider="openai" + ) + + assert prompt_cost > 0.0 + assert completion_cost == 0.0 + assert prompt_cost + completion_cost > 0.0 + + +def test_negative_completion_tokens_with_breakdown_already_clamped(): + """The branch that already clamped must keep behaving as it did.""" + usage = Usage( + prompt_tokens=0, + completion_tokens=-100, + total_tokens=-100, + completion_tokens_details=CompletionTokensDetailsWrapper(reasoning_tokens=5), + ) + + _, completion_cost = generic_cost_per_token( + model="gpt-4o", usage=usage, custom_llm_provider="openai" + ) + + assert completion_cost >= 0.0 + + +def test_positive_token_costs_are_unchanged(): + """The clamp must not alter ordinary pricing.""" + usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gpt-4o", usage=usage, custom_llm_provider="openai" + ) + + model_info = litellm.get_model_info("gpt-4o") + assert prompt_cost == pytest.approx(1000 * model_info["input_cost_per_token"]) + assert completion_cost == pytest.approx(500 * model_info["output_cost_per_token"])