From ad9f6ce23b4d4d39f493b3997d3b9f095cc1f2b7 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 13:16:07 +0300 Subject: [PATCH] fix(logging): track token usage for rerank calls in spend logs RerankResponse has no top-level `usage` field, only `meta.billed_units`/ `meta.tokens`, so get_usage_from_response_obj() always returned zeroed Usage for every rerank provider, causing /ui/usage to show 0 tokens for all rerank calls regardless of provider. --- litellm/litellm_core_utils/litellm_logging.py | 14 +++++++ .../test_standard_logging_payload.py | 37 +++++++++++++++++++ 2 files changed, 51 insertions(+) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 99721c3ffa2..c26c69c3d48 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4719,6 +4719,20 @@ class StandardLoggingPayloadSetup: ) usage: Final = response_obj.get("usage", None) or {} + if not usage: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units = meta.get("billed_units") or {} + tokens = meta.get("tokens") or {} + total_tokens = billed_units.get("total_tokens", 0) or 0 + prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + return Usage( + prompt_tokens=prompt_tokens, + completion_tokens=0, + total_tokens=total_tokens, + ) if usage is None or (not isinstance(usage, dict) and not isinstance(usage, Usage)): return Usage( prompt_tokens=0, diff --git a/tests/logging_callback_tests/test_standard_logging_payload.py b/tests/logging_callback_tests/test_standard_logging_payload.py index f29b245b3be..79771ada9a4 100644 --- a/tests/logging_callback_tests/test_standard_logging_payload.py +++ b/tests/logging_callback_tests/test_standard_logging_payload.py @@ -127,6 +127,43 @@ def test_get_usage_from_image_generation_response(): assert usage.completion_tokens_details.text_tokens == 100 +def test_get_usage_from_rerank_response(): + """ + RerankResponse has no top-level `usage` field - token usage lives under + `meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what + hosted_vllm/other rerank servers return. Without this, /ui/usage always + showed 0 tokens for every rerank call. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 42 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 42 + + +def test_get_usage_from_rerank_response_no_meta(): + """ + A rerank response with no usable `meta` (and no `usage`) should still + fall back to all-zero usage instead of raising. + """ + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 0 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 0 + + def test_get_additional_headers(): additional_headers = { "x-ratelimit-limit-requests": "2000",