mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Merge a51ec7df57 into e6c4580a31
This commit is contained in:
commit
eb0e54950e
5 changed files with 222 additions and 5 deletions
|
|
@ -5254,6 +5254,24 @@ class StandardLoggingPayloadSetup:
|
|||
)
|
||||
|
||||
usage: Final = response_obj.get("usage", None) or {}
|
||||
if not usage:
|
||||
# RerankResponse has no top-level `usage` field - token usage lives
|
||||
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
|
||||
meta: Final = response_obj.get("meta")
|
||||
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
|
||||
billed_units: Final = meta.get("billed_units")
|
||||
tokens: Final = meta.get("tokens")
|
||||
total_tokens: Final = (
|
||||
billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0
|
||||
) or 0
|
||||
prompt_tokens: Final = (
|
||||
tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens
|
||||
) or total_tokens
|
||||
return Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=0,
|
||||
total_tokens=total_tokens,
|
||||
)
|
||||
if usage is None or (not isinstance(usage, dict) and not isinstance(usage, Usage)):
|
||||
return Usage(
|
||||
prompt_tokens=0,
|
||||
|
|
@ -5288,7 +5306,24 @@ class StandardLoggingPayloadSetup:
|
|||
if not response_obj:
|
||||
return _empty
|
||||
_raw: Final = response_obj.get("usage", None)
|
||||
if _raw is None:
|
||||
if not _raw:
|
||||
# RerankResponse has no top-level `usage` field - token usage lives
|
||||
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
|
||||
meta: Final = response_obj.get("meta")
|
||||
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
|
||||
billed_units: Final = meta.get("billed_units")
|
||||
tokens: Final = meta.get("tokens")
|
||||
total_tokens: Final = (
|
||||
billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0
|
||||
) or 0
|
||||
prompt_tokens: Final = (
|
||||
tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens
|
||||
) or total_tokens
|
||||
return { # mutable-ok: hot-path return type is a plain dict by design, matching `_empty` above
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": 0,
|
||||
"total_tokens": total_tokens,
|
||||
}
|
||||
return _empty
|
||||
if isinstance(_raw, ResponseAPIUsage):
|
||||
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()
|
||||
|
|
|
|||
|
|
@ -174,10 +174,22 @@ class HostedVLLMRerankConfig(BaseRerankConfig):
|
|||
return HostedVLLMRerankError(message=error_message, status_code=status_code, headers=headers)
|
||||
|
||||
def _transform_response(self, response: dict) -> RerankResponse:
|
||||
# Extract usage information
|
||||
usage_data: Final = response.get("usage", {})
|
||||
_billed_units: Final = RerankBilledUnits(total_tokens=usage_data.get("total_tokens", 0))
|
||||
_tokens: Final = RerankTokens(input_tokens=usage_data.get("total_tokens", 0))
|
||||
# Extract usage information - some servers (vLLM/Cohere-style) return a
|
||||
# top-level `meta` object; others (OpenAI/TEI-style) return `usage`.
|
||||
# Check both so either shape is picked up.
|
||||
raw_meta: Final = response.get("meta")
|
||||
usage_data: Final = response.get("usage")
|
||||
meta_billed_units: Final = raw_meta.get("billed_units") if isinstance(raw_meta, dict) else None
|
||||
meta_tokens: Final = raw_meta.get("tokens") if isinstance(raw_meta, dict) else None
|
||||
usage_total_tokens: Final = usage_data.get("total_tokens", 0) if isinstance(usage_data, dict) else 0
|
||||
total_tokens: Final = (
|
||||
meta_billed_units.get("total_tokens") if isinstance(meta_billed_units, dict) else None
|
||||
) or usage_total_tokens
|
||||
input_tokens: Final = (
|
||||
meta_tokens.get("input_tokens") if isinstance(meta_tokens, dict) else None
|
||||
) or usage_total_tokens
|
||||
_billed_units: Final = RerankBilledUnits(total_tokens=total_tokens)
|
||||
_tokens: Final = RerankTokens(input_tokens=input_tokens)
|
||||
rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens)
|
||||
|
||||
# Extract results
|
||||
|
|
|
|||
|
|
@ -122,6 +122,93 @@ def test_get_usage_from_image_generation_response():
|
|||
assert usage.completion_tokens_details.text_tokens == 100
|
||||
|
||||
|
||||
def test_get_usage_from_rerank_response():
|
||||
"""
|
||||
RerankResponse has no top-level `usage` field - token usage lives under
|
||||
`meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what
|
||||
hosted_vllm/other rerank servers return. Without this, /ui/usage always
|
||||
showed 0 tokens for every rerank call.
|
||||
"""
|
||||
response_obj = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
|
||||
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
|
||||
|
||||
assert usage.prompt_tokens == 42
|
||||
assert usage.completion_tokens == 0
|
||||
assert usage.total_tokens == 42
|
||||
|
||||
|
||||
def test_get_usage_from_rerank_response_no_meta():
|
||||
"""
|
||||
A rerank response with no usable `meta` (and no `usage`) should still
|
||||
fall back to all-zero usage instead of raising.
|
||||
"""
|
||||
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
|
||||
|
||||
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
|
||||
|
||||
assert usage.prompt_tokens == 0
|
||||
assert usage.completion_tokens == 0
|
||||
assert usage.total_tokens == 0
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_rerank_response():
|
||||
"""
|
||||
get_usage_as_dict() is the function actually called by
|
||||
get_standard_logging_object_payload() (the hot path used to build spend
|
||||
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
|
||||
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
|
||||
showing 0 tokens for rerank calls even after fixing the other function.
|
||||
"""
|
||||
response_obj = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 42
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 42
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_rerank_response_no_meta():
|
||||
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 0
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 0
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_chat_response():
|
||||
response_obj = {
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"total_tokens": 30,
|
||||
}
|
||||
}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 10
|
||||
assert usage_dict["completion_tokens"] == 20
|
||||
assert usage_dict["total_tokens"] == 30
|
||||
|
||||
|
||||
def test_get_additional_headers():
|
||||
additional_headers = {
|
||||
"x-ratelimit-limit-requests": "2000",
|
||||
|
|
|
|||
|
|
@ -4962,6 +4962,70 @@ def test_pre_call_does_not_pin_request_in_module_state(logging_obj):
|
|||
assert litellm.error_logs == {}
|
||||
|
||||
|
||||
def test_get_usage_from_response_obj_rerank_meta():
|
||||
"""
|
||||
RerankResponse has no top-level `usage` field - token usage lives under
|
||||
`meta.billed_units`/`meta.tokens` (Cohere-style API), which is also what
|
||||
hosted_vllm/other rerank servers return. Without this, /ui/usage always
|
||||
showed 0 tokens for every rerank call.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
|
||||
|
||||
response_obj = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
|
||||
usage = StandardLoggingPayloadSetup.get_usage_from_response_obj(response_obj)
|
||||
|
||||
assert usage.prompt_tokens == 42
|
||||
assert usage.completion_tokens == 0
|
||||
assert usage.total_tokens == 42
|
||||
|
||||
|
||||
def test_get_usage_as_dict_rerank_meta():
|
||||
"""
|
||||
get_usage_as_dict() is the function actually called by
|
||||
get_standard_logging_object_payload() (the hot path used to build spend
|
||||
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
|
||||
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
|
||||
showing 0 tokens for rerank calls even after fixing the other function.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
|
||||
|
||||
response_obj = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 42
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 42
|
||||
|
||||
|
||||
def test_get_usage_as_dict_rerank_meta_no_meta():
|
||||
"""A rerank response with no usable `meta` (and no `usage`) falls back to zero."""
|
||||
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
|
||||
|
||||
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 0
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 0
|
||||
|
||||
|
||||
def test_handle_anthropic_messages_response_logging_preserves_fast_mode_speed():
|
||||
"""/v1/messages non-streaming rebuilds usage by re-transforming the raw Anthropic
|
||||
response. Anthropic's fast-mode multiplier is applied off ``usage.speed``, which the
|
||||
|
|
|
|||
|
|
@ -131,6 +131,25 @@ class TestHostedVLLMRerankTransform:
|
|||
assert result.meta["billed_units"]["total_tokens"] == 42
|
||||
assert result.meta["tokens"]["input_tokens"] == 42
|
||||
|
||||
def test_transform_response_with_meta(self):
|
||||
"""Some vLLM rerank servers (Cohere-compatible API) return usage under
|
||||
`meta.billed_units`/`meta.tokens` instead of a top-level `usage` key."""
|
||||
response_dict = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [
|
||||
{"index": 0, "relevance_score": 0.9, "document": {"text": "doc1 text"}},
|
||||
{"index": 1, "relevance_score": 0.7, "document": {"text": "doc2 text"}},
|
||||
],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
result = self.config._transform_response(response_dict)
|
||||
assert result.id == "rerank-d618748e0f5543e8ba09ee7dd131ac59"
|
||||
assert result.meta["billed_units"]["total_tokens"] == 42
|
||||
assert result.meta["tokens"]["input_tokens"] == 42
|
||||
|
||||
def test_transform_response_missing_results(self):
|
||||
response_dict = {"id": "abc123", "usage": {"total_tokens": 10}}
|
||||
with pytest.raises(ValueError, match="No results found in the response="):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue