From ad9f6ce23b4d4d39f493b3997d3b9f095cc1f2b7 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 13:16:07 +0300 Subject: [PATCH 1/6] fix(logging): track token usage for rerank calls in spend logs RerankResponse has no top-level `usage` field, only `meta.billed_units`/ `meta.tokens`, so get_usage_from_response_obj() always returned zeroed Usage for every rerank provider, causing /ui/usage to show 0 tokens for all rerank calls regardless of provider. --- litellm/litellm_core_utils/litellm_logging.py | 14 +++++++ .../test_standard_logging_payload.py | 37 +++++++++++++++++++ 2 files changed, 51 insertions(+) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 99721c3ffa2..c26c69c3d48 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4719,6 +4719,20 @@ class StandardLoggingPayloadSetup: ) usage: Final = response_obj.get("usage", None) or {} + if not usage: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units = meta.get("billed_units") or {} + tokens = meta.get("tokens") or {} + total_tokens = billed_units.get("total_tokens", 0) or 0 + prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + return Usage( + prompt_tokens=prompt_tokens, + completion_tokens=0, + total_tokens=total_tokens, + ) if usage is None or (not isinstance(usage, dict) and not isinstance(usage, Usage)): return Usage( prompt_tokens=0, diff --git a/tests/logging_callback_tests/test_standard_logging_payload.py b/tests/logging_callback_tests/test_standard_logging_payload.py index f29b245b3be..79771ada9a4 100644 --- a/tests/logging_callback_tests/test_standard_logging_payload.py +++ b/tests/logging_callback_tests/test_standard_logging_payload.py @@ -127,6 +127,43 @@ def test_get_usage_from_image_generation_response(): assert usage.completion_tokens_details.text_tokens == 100 +def test_get_usage_from_rerank_response(): + """ + RerankResponse has no top-level `usage` field - token usage lives under + `meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what + hosted_vllm/other rerank servers return. Without this, /ui/usage always + showed 0 tokens for every rerank call. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 42 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 42 + + +def test_get_usage_from_rerank_response_no_meta(): + """ + A rerank response with no usable `meta` (and no `usage`) should still + fall back to all-zero usage instead of raising. + """ + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 0 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 0 + + def test_get_additional_headers(): additional_headers = { "x-ratelimit-limit-requests": "2000", From beb462c85cb916a21bc623a740ab9c82b3e14b05 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 13:17:49 +0300 Subject: [PATCH 2/6] fix(hosted_vllm): read rerank usage from both `meta` and `usage` HostedVLLMRerankConfig._transform_response() only read token counts from a top-level `usage` key. vLLM's /rerank endpoint (Cohere-compatible API) actually returns them under `meta.billed_units`/`meta.tokens`, so usage was silently zeroed for hosted_vllm rerank models. Now checks both shapes, keeping backward compatibility with servers that use `usage`. --- .../llms/hosted_vllm/rerank/transformation.py | 13 ++++++++++--- .../test_hosted_vllm_rerank_transformation.py | 19 +++++++++++++++++++ 2 files changed, 29 insertions(+), 3 deletions(-) diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 74c13b450f5..b467bcc5317 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -172,10 +172,17 @@ class HostedVLLMRerankConfig(BaseRerankConfig): return HostedVLLMRerankError(message=error_message, status_code=status_code, headers=headers) def _transform_response(self, response: dict) -> RerankResponse: - # Extract usage information + # Extract usage information - some servers (vLLM/Cohere-style) return a + # top-level `meta` object; others (OpenAI/TEI-style) return `usage`. + # Check both so either shape is picked up. + raw_meta: Final = response.get("meta") or {} usage_data: Final = response.get("usage", {}) - _billed_units: Final = RerankBilledUnits(total_tokens=usage_data.get("total_tokens", 0)) - _tokens: Final = RerankTokens(input_tokens=usage_data.get("total_tokens", 0)) + total_tokens: Final = raw_meta.get("billed_units", {}).get("total_tokens") or usage_data.get( + "total_tokens", 0 + ) + input_tokens: Final = raw_meta.get("tokens", {}).get("input_tokens") or usage_data.get("total_tokens", 0) + _billed_units: Final = RerankBilledUnits(total_tokens=total_tokens) + _tokens: Final = RerankTokens(input_tokens=input_tokens) rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens) # Extract results diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index 6425e815db0..989efe4c105 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -131,6 +131,25 @@ class TestHostedVLLMRerankTransform: assert result.meta["billed_units"]["total_tokens"] == 42 assert result.meta["tokens"]["input_tokens"] == 42 + def test_transform_response_with_meta(self): + """Some vLLM rerank servers (Cohere-compatible API) return usage under + `meta.billed_units`/`meta.tokens` instead of a top-level `usage` key.""" + response_dict = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [ + {"index": 0, "relevance_score": 0.9, "document": {"text": "doc1 text"}}, + {"index": 1, "relevance_score": 0.7, "document": {"text": "doc2 text"}}, + ], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + result = self.config._transform_response(response_dict) + assert result.id == "rerank-d618748e0f5543e8ba09ee7dd131ac59" + assert result.meta["billed_units"]["total_tokens"] == 42 + assert result.meta["tokens"]["input_tokens"] == 42 + def test_transform_response_missing_results(self): response_dict = {"id": "abc123", "usage": {"total_tokens": 10}} with pytest.raises(ValueError, match="No results found in the response="): From 5ae21a66d189b652f5e0d83ea430e9cfc16c2f53 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 14:39:43 +0300 Subject: [PATCH 3/6] fix(logging): fix rerank token usage in get_usage_as_dict hot path too get_standard_logging_object_payload() actually calls get_usage_as_dict(), not get_usage_from_response_obj() - a newer, separate hot-path duplicate added after the branch's original base. It had the same `usage`-only gap, so /ui/usage still showed 0 tokens for rerank calls even with the other fix applied. Same meta.billed_units/meta.tokens fallback applied here. --- litellm/litellm_core_utils/litellm_logging.py | 15 +++++- .../test_standard_logging_payload.py | 50 +++++++++++++++++++ 2 files changed, 64 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index c26c69c3d48..3d05e6de35a 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4765,7 +4765,20 @@ class StandardLoggingPayloadSetup: if not response_obj: return _empty _raw: Final = response_obj.get("usage", None) - if _raw is None: + if not _raw: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units = meta.get("billed_units") or {} + tokens = meta.get("tokens") or {} + total_tokens = billed_units.get("total_tokens", 0) or 0 + prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + return { + "prompt_tokens": prompt_tokens, + "completion_tokens": 0, + "total_tokens": total_tokens, + } return _empty if isinstance(_raw, ResponseAPIUsage): return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump() diff --git a/tests/logging_callback_tests/test_standard_logging_payload.py b/tests/logging_callback_tests/test_standard_logging_payload.py index 79771ada9a4..c5c2899eae8 100644 --- a/tests/logging_callback_tests/test_standard_logging_payload.py +++ b/tests/logging_callback_tests/test_standard_logging_payload.py @@ -164,6 +164,56 @@ def test_get_usage_from_rerank_response_no_meta(): assert usage.total_tokens == 0 +def test_get_usage_as_dict_from_rerank_response(): + """ + get_usage_as_dict() is the function actually called by + get_standard_logging_object_payload() (the hot path used to build spend + logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens` + shape the same way get_usage_from_response_obj() does, or /ui/usage keeps + showing 0 tokens for rerank calls even after fixing the other function. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 42 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 42 + + +def test_get_usage_as_dict_from_rerank_response_no_meta(): + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 0 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 0 + + +def test_get_usage_as_dict_from_chat_response(): + response_obj = { + "usage": { + "prompt_tokens": 10, + "completion_tokens": 20, + "total_tokens": 30, + } + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 10 + assert usage_dict["completion_tokens"] == 20 + assert usage_dict["total_tokens"] == 30 + + def test_get_additional_headers(): additional_headers = { "x-ratelimit-limit-requests": "2000", From 63cd9e14677c98191911e9bf39a208f7c1876ec2 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 16:17:17 +0300 Subject: [PATCH 4/6] style: apply ruff format to hosted_vllm rerank transformation --- litellm/llms/hosted_vllm/rerank/transformation.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index b467bcc5317..55605d82c8d 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -177,9 +177,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): # Check both so either shape is picked up. raw_meta: Final = response.get("meta") or {} usage_data: Final = response.get("usage", {}) - total_tokens: Final = raw_meta.get("billed_units", {}).get("total_tokens") or usage_data.get( - "total_tokens", 0 - ) + total_tokens: Final = raw_meta.get("billed_units", {}).get("total_tokens") or usage_data.get("total_tokens", 0) input_tokens: Final = raw_meta.get("tokens", {}).get("input_tokens") or usage_data.get("total_tokens", 0) _billed_units: Final = RerankBilledUnits(total_tokens=total_tokens) _tokens: Final = RerankTokens(input_tokens=input_tokens) From de365aa6c7cac77723ec0ebdfdd2e5afb3fe45b6 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 16:34:50 +0300 Subject: [PATCH 5/6] fix: satisfy type-discipline gate (Final annotations, no dict literals) The repo's LIT002/LIT010 budget gate flags mutable dict-literal construction and non-Final local assignments. Rewrote both new usage fallback blocks to read via .get()/isinstance guards instead of `or {}` defaults, and annotated every new local as Final, matching the codebase's existing style. The one unavoidable dict literal (get_usage_as_dict's hot-path return, mirroring the pre-existing _empty dict in the same function) is marked `# mutable-ok` with a reason. --- litellm/litellm_core_utils/litellm_logging.py | 30 ++++++++++++------- .../llms/hosted_vllm/rerank/transformation.py | 15 +++++++--- 2 files changed, 30 insertions(+), 15 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 3d05e6de35a..c3727c777ea 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4722,12 +4722,16 @@ class StandardLoggingPayloadSetup: if not usage: # RerankResponse has no top-level `usage` field - token usage lives # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). - meta = response_obj.get("meta") + meta: Final = response_obj.get("meta") if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): - billed_units = meta.get("billed_units") or {} - tokens = meta.get("tokens") or {} - total_tokens = billed_units.get("total_tokens", 0) or 0 - prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens return Usage( prompt_tokens=prompt_tokens, completion_tokens=0, @@ -4768,13 +4772,17 @@ class StandardLoggingPayloadSetup: if not _raw: # RerankResponse has no top-level `usage` field - token usage lives # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). - meta = response_obj.get("meta") + meta: Final = response_obj.get("meta") if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): - billed_units = meta.get("billed_units") or {} - tokens = meta.get("tokens") or {} - total_tokens = billed_units.get("total_tokens", 0) or 0 - prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens - return { + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens + return { # mutable-ok: hot-path return type is a plain dict by design, matching `_empty` above "prompt_tokens": prompt_tokens, "completion_tokens": 0, "total_tokens": total_tokens, diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 55605d82c8d..04d918a54e8 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -175,10 +175,17 @@ class HostedVLLMRerankConfig(BaseRerankConfig): # Extract usage information - some servers (vLLM/Cohere-style) return a # top-level `meta` object; others (OpenAI/TEI-style) return `usage`. # Check both so either shape is picked up. - raw_meta: Final = response.get("meta") or {} - usage_data: Final = response.get("usage", {}) - total_tokens: Final = raw_meta.get("billed_units", {}).get("total_tokens") or usage_data.get("total_tokens", 0) - input_tokens: Final = raw_meta.get("tokens", {}).get("input_tokens") or usage_data.get("total_tokens", 0) + raw_meta: Final = response.get("meta") + usage_data: Final = response.get("usage") + meta_billed_units: Final = raw_meta.get("billed_units") if isinstance(raw_meta, dict) else None + meta_tokens: Final = raw_meta.get("tokens") if isinstance(raw_meta, dict) else None + usage_total_tokens: Final = usage_data.get("total_tokens", 0) if isinstance(usage_data, dict) else 0 + total_tokens: Final = ( + meta_billed_units.get("total_tokens") if isinstance(meta_billed_units, dict) else None + ) or usage_total_tokens + input_tokens: Final = ( + meta_tokens.get("input_tokens") if isinstance(meta_tokens, dict) else None + ) or usage_total_tokens _billed_units: Final = RerankBilledUnits(total_tokens=total_tokens) _tokens: Final = RerankTokens(input_tokens=input_tokens) rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens) From c6f2323ed9327a062074b7741e659e24e09e19c7 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 17:01:11 +0300 Subject: [PATCH 6/6] test: cover rerank usage fallback in the GitHub Actions test suite tests/logging_callback_tests/test_standard_logging_payload.py runs via CircleCI, which doesn't trigger on this fork's PR, so its coverage of get_usage_from_response_obj/get_usage_as_dict's new rerank fallback never reached codecov/patch here. Mirrored the same assertions into tests/test_litellm/litellm_core_utils/test_litellm_logging.py, which the "core-utils" GitHub Actions job does run and upload coverage for. --- .../test_litellm_logging.py | 64 +++++++++++++++++++ 1 file changed, 64 insertions(+) diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 23e0975cd08..a5981e3a98d 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -4230,3 +4230,67 @@ def test_pre_call_does_not_pin_request_in_module_state(logging_obj): logging_obj.post_call(original_response='{"ok": true}', input=big_input, api_key="sk-test") assert litellm.error_logs == {} + + +def test_get_usage_from_response_obj_rerank_meta(): + """ + RerankResponse has no top-level `usage` field - token usage lives under + `meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what + hosted_vllm/other rerank servers return. Without this, /ui/usage always + showed 0 tokens for every rerank call. + """ + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 42 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 42 + + +def test_get_usage_as_dict_rerank_meta(): + """ + get_usage_as_dict() is the function actually called by + get_standard_logging_object_payload() (the hot path used to build spend + logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens` + shape the same way get_usage_from_response_obj() does, or /ui/usage keeps + showing 0 tokens for rerank calls even after fixing the other function. + """ + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 42 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 42 + + +def test_get_usage_as_dict_rerank_meta_no_meta(): + """A rerank response with no usable `meta` (and no `usage`) falls back to zero.""" + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 0 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 0