This commit is contained in:
Oleg 2026-08-27 19:30:53 -05:00 • committed by GitHub
commit eb0e54950e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 222 additions and 5 deletions

View file

@ -5254,6 +5254,24 @@ class StandardLoggingPayloadSetup:
)
usage: Final = response_obj.get("usage", None) or {}
if not usage:
# RerankResponse has no top-level `usage` field - token usage lives
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
meta: Final = response_obj.get("meta")
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
billed_units: Final = meta.get("billed_units")
tokens: Final = meta.get("tokens")
total_tokens: Final = (
billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0
) or 0
prompt_tokens: Final = (
tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens
) or total_tokens
return Usage(
prompt_tokens=prompt_tokens,
completion_tokens=0,
total_tokens=total_tokens,
)
if usage is None or (not isinstance(usage, dict) and not isinstance(usage, Usage)):
return Usage(
prompt_tokens=0,
@ -5288,7 +5306,24 @@ class StandardLoggingPayloadSetup:
if not response_obj:
return _empty
_raw: Final = response_obj.get("usage", None)
if _raw is None:
if not _raw:
# RerankResponse has no top-level `usage` field - token usage lives
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
meta: Final = response_obj.get("meta")
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
billed_units: Final = meta.get("billed_units")
tokens: Final = meta.get("tokens")
total_tokens: Final = (
billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0
) or 0
prompt_tokens: Final = (
tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens
) or total_tokens
return { # mutable-ok: hot-path return type is a plain dict by design, matching `_empty` above
"prompt_tokens": prompt_tokens,
"completion_tokens": 0,
"total_tokens": total_tokens,
}
return _empty
if isinstance(_raw, ResponseAPIUsage):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()

View file

@ -174,10 +174,22 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
return HostedVLLMRerankError(message=error_message, status_code=status_code, headers=headers)
def _transform_response(self, response: dict) -> RerankResponse:
# Extract usage information
usage_data: Final = response.get("usage", {})
_billed_units: Final = RerankBilledUnits(total_tokens=usage_data.get("total_tokens", 0))
_tokens: Final = RerankTokens(input_tokens=usage_data.get("total_tokens", 0))
# Extract usage information - some servers (vLLM/Cohere-style) return a
# top-level `meta` object; others (OpenAI/TEI-style) return `usage`.
# Check both so either shape is picked up.
raw_meta: Final = response.get("meta")
usage_data: Final = response.get("usage")
meta_billed_units: Final = raw_meta.get("billed_units") if isinstance(raw_meta, dict) else None
meta_tokens: Final = raw_meta.get("tokens") if isinstance(raw_meta, dict) else None
usage_total_tokens: Final = usage_data.get("total_tokens", 0) if isinstance(usage_data, dict) else 0
total_tokens: Final = (
meta_billed_units.get("total_tokens") if isinstance(meta_billed_units, dict) else None
) or usage_total_tokens
input_tokens: Final = (
meta_tokens.get("input_tokens") if isinstance(meta_tokens, dict) else None
) or usage_total_tokens
_billed_units: Final = RerankBilledUnits(total_tokens=total_tokens)
_tokens: Final = RerankTokens(input_tokens=input_tokens)
rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens)
# Extract results

View file

@ -122,6 +122,93 @@ def test_get_usage_from_image_generation_response():
assert usage.completion_tokens_details.text_tokens == 100
def test_get_usage_from_rerank_response():
"""
RerankResponse has no top-level `usage` field - token usage lives under
`meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what
hosted_vllm/other rerank servers return. Without this, /ui/usage always
showed 0 tokens for every rerank call.
"""
response_obj = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
assert usage.prompt_tokens == 42
assert usage.completion_tokens == 0
assert usage.total_tokens == 42
def test_get_usage_from_rerank_response_no_meta():
"""
A rerank response with no usable `meta` (and no `usage`) should still
fall back to all-zero usage instead of raising.
"""
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
assert usage.prompt_tokens == 0
assert usage.completion_tokens == 0
assert usage.total_tokens == 0
def test_get_usage_as_dict_from_rerank_response():
"""
get_usage_as_dict() is the function actually called by
get_standard_logging_object_payload() (the hot path used to build spend
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
showing 0 tokens for rerank calls even after fixing the other function.
"""
response_obj = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 42
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 42
def test_get_usage_as_dict_from_rerank_response_no_meta():
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 0
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 0
def test_get_usage_as_dict_from_chat_response():
response_obj = {
"usage": {
"prompt_tokens": 10,
"completion_tokens": 20,
"total_tokens": 30,
}
}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 10
assert usage_dict["completion_tokens"] == 20
assert usage_dict["total_tokens"] == 30
def test_get_additional_headers():
additional_headers = {
"x-ratelimit-limit-requests": "2000",

View file

@ -4962,6 +4962,70 @@ def test_pre_call_does_not_pin_request_in_module_state(logging_obj):
assert litellm.error_logs == {}
def test_get_usage_from_response_obj_rerank_meta():
"""
RerankResponse has no top-level `usage` field - token usage lives under
`meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what
hosted_vllm/other rerank servers return. Without this, /ui/usage always
showed 0 tokens for every rerank call.
"""
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
response_obj = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
assert usage.prompt_tokens == 42
assert usage.completion_tokens == 0
assert usage.total_tokens == 42
def test_get_usage_as_dict_rerank_meta():
"""
get_usage_as_dict() is the function actually called by
get_standard_logging_object_payload() (the hot path used to build spend
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
showing 0 tokens for rerank calls even after fixing the other function.
"""
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
response_obj = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 42
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 42
def test_get_usage_as_dict_rerank_meta_no_meta():
"""A rerank response with no usable `meta` (and no `usage`) falls back to zero."""
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 0
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 0
def test_handle_anthropic_messages_response_logging_preserves_fast_mode_speed():
"""/v1/messages non-streaming rebuilds usage by re-transforming the raw Anthropic
response. Anthropic's fast-mode multiplier is applied off ``usage.speed``, which the

View file

@ -131,6 +131,25 @@ class TestHostedVLLMRerankTransform:
assert result.meta["billed_units"]["total_tokens"] == 42
assert result.meta["tokens"]["input_tokens"] == 42
def test_transform_response_with_meta(self):
"""Some vLLM rerank servers (Cohere-compatible API) return usage under
`meta.billed_units`/`meta.tokens` instead of a top-level `usage` key."""
response_dict = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [
{"index": 0, "relevance_score": 0.9, "document": {"text": "doc1 text"}},
{"index": 1, "relevance_score": 0.7, "document": {"text": "doc2 text"}},
],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
result = self.config._transform_response(response_dict)
assert result.id == "rerank-d618748e0f5543e8ba09ee7dd131ac59"
assert result.meta["billed_units"]["total_tokens"] == 42
assert result.meta["tokens"]["input_tokens"] == 42
def test_transform_response_missing_results(self):
response_dict = {"id": "abc123", "usage": {"total_tokens": 10}}
with pytest.raises(ValueError, match="No results found in the response="):