From 5ae21a66d189b652f5e0d83ea430e9cfc16c2f53 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 14:39:43 +0300 Subject: [PATCH] fix(logging): fix rerank token usage in get_usage_as_dict hot path too get_standard_logging_object_payload() actually calls get_usage_as_dict(), not get_usage_from_response_obj() - a newer, separate hot-path duplicate added after the branch's original base. It had the same `usage`-only gap, so /ui/usage still showed 0 tokens for rerank calls even with the other fix applied. Same meta.billed_units/meta.tokens fallback applied here. --- litellm/litellm_core_utils/litellm_logging.py | 15 +++++- .../test_standard_logging_payload.py | 50 +++++++++++++++++++ 2 files changed, 64 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index c26c69c3d48..3d05e6de35a 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4765,7 +4765,20 @@ class StandardLoggingPayloadSetup: if not response_obj: return _empty _raw: Final = response_obj.get("usage", None) - if _raw is None: + if not _raw: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units = meta.get("billed_units") or {} + tokens = meta.get("tokens") or {} + total_tokens = billed_units.get("total_tokens", 0) or 0 + prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + return { + "prompt_tokens": prompt_tokens, + "completion_tokens": 0, + "total_tokens": total_tokens, + } return _empty if isinstance(_raw, ResponseAPIUsage): return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump() diff --git a/tests/logging_callback_tests/test_standard_logging_payload.py b/tests/logging_callback_tests/test_standard_logging_payload.py index 79771ada9a4..c5c2899eae8 100644 --- a/tests/logging_callback_tests/test_standard_logging_payload.py +++ b/tests/logging_callback_tests/test_standard_logging_payload.py @@ -164,6 +164,56 @@ def test_get_usage_from_rerank_response_no_meta(): assert usage.total_tokens == 0 +def test_get_usage_as_dict_from_rerank_response(): + """ + get_usage_as_dict() is the function actually called by + get_standard_logging_object_payload() (the hot path used to build spend + logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens` + shape the same way get_usage_from_response_obj() does, or /ui/usage keeps + showing 0 tokens for rerank calls even after fixing the other function. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 42 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 42 + + +def test_get_usage_as_dict_from_rerank_response_no_meta(): + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 0 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 0 + + +def test_get_usage_as_dict_from_chat_response(): + response_obj = { + "usage": { + "prompt_tokens": 10, + "completion_tokens": 20, + "total_tokens": 30, + } + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 10 + assert usage_dict["completion_tokens"] == 20 + assert usage_dict["total_tokens"] == 30 + + def test_get_additional_headers(): additional_headers = { "x-ratelimit-limit-requests": "2000",