fix(logging): fix rerank token usage in get_usage_as_dict hot path too

get_standard_logging_object_payload() actually calls get_usage_as_dict(),
not get_usage_from_response_obj() - a newer, separate hot-path duplicate
added after the branch's original base. It had the same `usage`-only gap,
so /ui/usage still showed 0 tokens for rerank calls even with the other
fix applied. Same meta.billed_units/meta.tokens fallback applied here.
This commit is contained in:
spions 2026-08-07 14:39:43 +03:00
parent beb462c85c
commit 5ae21a66d1
2 changed files with 64 additions and 1 deletions

View file

@ -4765,7 +4765,20 @@ class StandardLoggingPayloadSetup:
if not response_obj:
return _empty
_raw: Final = response_obj.get("usage", None)
if _raw is None:
if not _raw:
# RerankResponse has no top-level `usage` field - token usage lives
# under `meta.billed_units`/`meta.tokens` instead (Cohere-style API).
meta = response_obj.get("meta")
if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta):
billed_units = meta.get("billed_units") or {}
tokens = meta.get("tokens") or {}
total_tokens = billed_units.get("total_tokens", 0) or 0
prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens
return {
"prompt_tokens": prompt_tokens,
"completion_tokens": 0,
"total_tokens": total_tokens,
}
return _empty
if isinstance(_raw, ResponseAPIUsage):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_raw).model_dump()

View file

@ -164,6 +164,56 @@ def test_get_usage_from_rerank_response_no_meta():
assert usage.total_tokens == 0
def test_get_usage_as_dict_from_rerank_response():
"""
get_usage_as_dict() is the function actually called by
get_standard_logging_object_payload() (the hot path used to build spend
logs / /ui/usage). It must handle the rerank `meta.billed_units`/`meta.tokens`
shape the same way get_usage_from_response_obj() does, or /ui/usage keeps
showing 0 tokens for rerank calls even after fixing the other function.
"""
response_obj = {
"id": "rerank-d618748e0f5543e8ba09ee7dd131ac59",
"results": [],
"meta": {
"billed_units": {"total_tokens": 42},
"tokens": {"input_tokens": 42},
},
}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 42
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 42
def test_get_usage_as_dict_from_rerank_response_no_meta():
response_obj = {"id": "rerank-abc", "results": [], "meta": None}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 0
assert usage_dict["completion_tokens"] == 0
assert usage_dict["total_tokens"] == 0
def test_get_usage_as_dict_from_chat_response():
response_obj = {
"usage": {
"prompt_tokens": 10,
"completion_tokens": 20,
"total_tokens": 30,
}
}
usage_dict = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj)
assert usage_dict["prompt_tokens"] == 10
assert usage_dict["completion_tokens"] == 20
assert usage_dict["total_tokens"] == 30
def test_get_additional_headers():
additional_headers = {
"x-ratelimit-limit-requests": "2000",