mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(logging): fix rerank token usage in get_usage_as_dict hot path too
get_standard_logging_object_payload() actually calls get_usage_as_dict(), not get_usage_from_response_obj() - a newer, separate hot-path duplicate added after the branch's original base. It had the same `usage`-only gap, so /ui/usage still showed 0 tokens for rerank calls even with the other fix applied. Same meta.billed_units/meta.tokens fallback applied here.
This commit is contained in:
parent
beb462c85c
commit
5ae21a66d1
2 changed files with 64 additions and 1 deletions
|
|
@ -4765,7 +4765,20 @@ class StandardLoggingPayloadSetup:
|
|||
if not response_obj:
|
||||
return _empty
|
||||
_raw: Final = response_obj.get("usage", None)
|
||||
if _raw is None:
|
||||
if not _raw:
|
||||
# RerankResponse has no top-level `usage` field - token usage lives
|
||||
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
|
||||
meta = response_obj.get("meta")
|
||||
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
|
||||
billed_units = meta.get("billed_units") or {}
|
||||
tokens = meta.get("tokens") or {}
|
||||
total_tokens = billed_units.get("total_tokens", 0) or 0
|
||||
prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens
|
||||
return {
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": 0,
|
||||
"total_tokens": total_tokens,
|
||||
}
|
||||
return _empty
|
||||
if isinstance(_raw, ResponseAPIUsage):
|
||||
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()
|
||||
|
|
|
|||
|
|
@ -164,6 +164,56 @@ def test_get_usage_from_rerank_response_no_meta():
|
|||
assert usage.total_tokens == 0
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_rerank_response():
|
||||
"""
|
||||
get_usage_as_dict() is the function actually called by
|
||||
get_standard_logging_object_payload() (the hot path used to build spend
|
||||
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
|
||||
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
|
||||
showing 0 tokens for rerank calls even after fixing the other function.
|
||||
"""
|
||||
response_obj = {
|
||||
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
|
||||
"results": [],
|
||||
"meta": {
|
||||
"billed_units": {"total_tokens": 42},
|
||||
"tokens": {"input_tokens": 42},
|
||||
},
|
||||
}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 42
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 42
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_rerank_response_no_meta():
|
||||
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 0
|
||||
assert usage_dict["completion_tokens"] == 0
|
||||
assert usage_dict["total_tokens"] == 0
|
||||
|
||||
|
||||
def test_get_usage_as_dict_from_chat_response():
|
||||
response_obj = {
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"total_tokens": 30,
|
||||
}
|
||||
}
|
||||
|
||||
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
|
||||
|
||||
assert usage_dict["prompt_tokens"] == 10
|
||||
assert usage_dict["completion_tokens"] == 20
|
||||
assert usage_dict["total_tokens"] == 30
|
||||
|
||||
|
||||
def test_get_additional_headers():
|
||||
additional_headers = {
|
||||
"x-ratelimit-limit-requests": "2000",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue