diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 3018f0c4d24..32ea6c25c48 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -5254,6 +5254,24 @@ class StandardLoggingPayloadSetup: ) usage: Final = response_obj.get("usage", None) or {} + if not usage: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta: Final = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens + return Usage( + prompt_tokens=prompt_tokens, + completion_tokens=0, + total_tokens=total_tokens, + ) if usage is None or (not isinstance(usage, dict) and not isinstance(usage, Usage)): return Usage( prompt_tokens=0, @@ -5288,7 +5306,24 @@ class StandardLoggingPayloadSetup: if not response_obj: return _empty _raw: Final = response_obj.get("usage", None) - if _raw is None: + if not _raw: + # RerankResponse has no top-level `usage` field - token usage lives + # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). + meta: Final = response_obj.get("meta") + if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens + return { # mutable-ok: hot-path return type is a plain dict by design, matching `_empty` above + "prompt_tokens": prompt_tokens, + "completion_tokens": 0, + "total_tokens": total_tokens, + } return _empty if isinstance(_raw, ResponseAPIUsage): return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump() diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 0e8fa294f5d..b96d6325fd7 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -174,10 +174,22 @@ class HostedVLLMRerankConfig(BaseRerankConfig): return HostedVLLMRerankError(message=error_message, status_code=status_code, headers=headers) def _transform_response(self, response: dict) -> RerankResponse: - # Extract usage information - usage_data: Final = response.get("usage", {}) - _billed_units: Final = RerankBilledUnits(total_tokens=usage_data.get("total_tokens", 0)) - _tokens: Final = RerankTokens(input_tokens=usage_data.get("total_tokens", 0)) + # Extract usage information - some servers (vLLM/Cohere-style) return a + # top-level `meta` object; others (OpenAI/TEI-style) return `usage`. + # Check both so either shape is picked up. + raw_meta: Final = response.get("meta") + usage_data: Final = response.get("usage") + meta_billed_units: Final = raw_meta.get("billed_units") if isinstance(raw_meta, dict) else None + meta_tokens: Final = raw_meta.get("tokens") if isinstance(raw_meta, dict) else None + usage_total_tokens: Final = usage_data.get("total_tokens", 0) if isinstance(usage_data, dict) else 0 + total_tokens: Final = ( + meta_billed_units.get("total_tokens") if isinstance(meta_billed_units, dict) else None + ) or usage_total_tokens + input_tokens: Final = ( + meta_tokens.get("input_tokens") if isinstance(meta_tokens, dict) else None + ) or usage_total_tokens + _billed_units: Final = RerankBilledUnits(total_tokens=total_tokens) + _tokens: Final = RerankTokens(input_tokens=input_tokens) rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens) # Extract results diff --git a/tests/logging_callback_tests/test_standard_logging_payload.py b/tests/logging_callback_tests/test_standard_logging_payload.py index da1fbbaa04f..2037d5261e1 100644 --- a/tests/logging_callback_tests/test_standard_logging_payload.py +++ b/tests/logging_callback_tests/test_standard_logging_payload.py @@ -122,6 +122,93 @@ def test_get_usage_from_image_generation_response(): assert usage.completion_tokens_details.text_tokens == 100 +def test_get_usage_from_rerank_response(): + """ + RerankResponse has no top-level `usage` field - token usage lives under + `meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what + hosted_vllm/other rerank servers return. Without this, /ui/usage always + showed 0 tokens for every rerank call. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 42 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 42 + + +def test_get_usage_from_rerank_response_no_meta(): + """ + A rerank response with no usable `meta` (and no `usage`) should still + fall back to all-zero usage instead of raising. + """ + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 0 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 0 + + +def test_get_usage_as_dict_from_rerank_response(): + """ + get_usage_as_dict() is the function actually called by + get_standard_logging_object_payload() (the hot path used to build spend + logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens` + shape the same way get_usage_from_response_obj() does, or /ui/usage keeps + showing 0 tokens for rerank calls even after fixing the other function. + """ + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 42 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 42 + + +def test_get_usage_as_dict_from_rerank_response_no_meta(): + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 0 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 0 + + +def test_get_usage_as_dict_from_chat_response(): + response_obj = { + "usage": { + "prompt_tokens": 10, + "completion_tokens": 20, + "total_tokens": 30, + } + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 10 + assert usage_dict["completion_tokens"] == 20 + assert usage_dict["total_tokens"] == 30 + + def test_get_additional_headers(): additional_headers = { "x-ratelimit-limit-requests": "2000", diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 0222e756ba1..8a3819dc133 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -4962,6 +4962,70 @@ def test_pre_call_does_not_pin_request_in_module_state(logging_obj): assert litellm.error_logs == {} +def test_get_usage_from_response_obj_rerank_meta(): + """ + RerankResponse has no top-level `usage` field - token usage lives under + `meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what + hosted_vllm/other rerank servers return. Without this, /ui/usage always + showed 0 tokens for every rerank call. + """ + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj) + + assert usage.prompt_tokens == 42 + assert usage.completion_tokens == 0 + assert usage.total_tokens == 42 + + +def test_get_usage_as_dict_rerank_meta(): + """ + get_usage_as_dict() is the function actually called by + get_standard_logging_object_payload() (the hot path used to build spend + logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens` + shape the same way get_usage_from_response_obj() does, or /ui/usage keeps + showing 0 tokens for rerank calls even after fixing the other function. + """ + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 42 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 42 + + +def test_get_usage_as_dict_rerank_meta_no_meta(): + """A rerank response with no usable `meta` (and no `usage`) falls back to zero.""" + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + response_obj = {"id": "rerank-abc", "results": [], "meta": None} + + usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj) + + assert usage_dict["prompt_tokens"] == 0 + assert usage_dict["completion_tokens"] == 0 + assert usage_dict["total_tokens"] == 0 + + def test_handle_anthropic_messages_response_logging_preserves_fast_mode_speed(): """/v1/messages non-streaming rebuilds usage by re-transforming the raw Anthropic response. Anthropic's fast-mode multiplier is applied off ``usage.speed``, which the diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index e6e6aa946d5..846fbba98ff 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -131,6 +131,25 @@ class TestHostedVLLMRerankTransform: assert result.meta["billed_units"]["total_tokens"] == 42 assert result.meta["tokens"]["input_tokens"] == 42 + def test_transform_response_with_meta(self): + """Some vLLM rerank servers (Cohere-compatible API) return usage under + `meta.billed_units`/`meta.tokens` instead of a top-level `usage` key.""" + response_dict = { + "id": "rerank-d618748e0f5543e8ba09ee7dd131ac59", + "results": [ + {"index": 0, "relevance_score": 0.9, "document": {"text": "doc1 text"}}, + {"index": 1, "relevance_score": 0.7, "document": {"text": "doc2 text"}}, + ], + "meta": { + "billed_units": {"total_tokens": 42}, + "tokens": {"input_tokens": 42}, + }, + } + result = self.config._transform_response(response_dict) + assert result.id == "rerank-d618748e0f5543e8ba09ee7dd131ac59" + assert result.meta["billed_units"]["total_tokens"] == 42 + assert result.meta["tokens"]["input_tokens"] == 42 + def test_transform_response_missing_results(self): response_dict = {"id": "abc123", "usage": {"total_tokens": 10}} with pytest.raises(ValueError, match="No results found in the response="):