From de365aa6c7cac77723ec0ebdfdd2e5afb3fe45b6 Mon Sep 17 00:00:00 2001 From: spions Date: Fri, 7 Aug 2026 16:34:50 +0300 Subject: [PATCH] fix: satisfy type-discipline gate (Final annotations, no dict literals) The repo's LIT002/LIT010 budget gate flags mutable dict-literal construction and non-Final local assignments. Rewrote both new usage fallback blocks to read via .get()/isinstance guards instead of `or {}` defaults, and annotated every new local as Final, matching the codebase's existing style. The one unavoidable dict literal (get_usage_as_dict's hot-path return, mirroring the pre-existing _empty dict in the same function) is marked `# mutable-ok` with a reason. --- litellm/litellm_core_utils/litellm_logging.py | 30 ++++++++++++------- .../llms/hosted_vllm/rerank/transformation.py | 15 +++++++--- 2 files changed, 30 insertions(+), 15 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 3d05e6de35a..c3727c777ea 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4722,12 +4722,16 @@ class StandardLoggingPayloadSetup: if not usage: # RerankResponse has no top-level `usage` field - token usage lives # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). - meta = response_obj.get("meta") + meta: Final = response_obj.get("meta") if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): - billed_units = meta.get("billed_units") or {} - tokens = meta.get("tokens") or {} - total_tokens = billed_units.get("total_tokens", 0) or 0 - prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens return Usage( prompt_tokens=prompt_tokens, completion_tokens=0, @@ -4768,13 +4772,17 @@ class StandardLoggingPayloadSetup: if not _raw: # RerankResponse has no top-level `usage` field - token usage lives # under `meta.billed_units`/`meta.tokens` instead (Cohere-style API). - meta = response_obj.get("meta") + meta: Final = response_obj.get("meta") if isinstance(meta, dict) and ("billed_units" in meta or "tokens" in meta): - billed_units = meta.get("billed_units") or {} - tokens = meta.get("tokens") or {} - total_tokens = billed_units.get("total_tokens", 0) or 0 - prompt_tokens = tokens.get("input_tokens", total_tokens) or total_tokens - return { + billed_units: Final = meta.get("billed_units") + tokens: Final = meta.get("tokens") + total_tokens: Final = ( + billed_units.get("total_tokens", 0) if isinstance(billed_units, dict) else 0 + ) or 0 + prompt_tokens: Final = ( + tokens.get("input_tokens", total_tokens) if isinstance(tokens, dict) else total_tokens + ) or total_tokens + return { # mutable-ok: hot-path return type is a plain dict by design, matching `_empty` above "prompt_tokens": prompt_tokens, "completion_tokens": 0, "total_tokens": total_tokens, diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 55605d82c8d..04d918a54e8 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -175,10 +175,17 @@ class HostedVLLMRerankConfig(BaseRerankConfig): # Extract usage information - some servers (vLLM/Cohere-style) return a # top-level `meta` object; others (OpenAI/TEI-style) return `usage`. # Check both so either shape is picked up. - raw_meta: Final = response.get("meta") or {} - usage_data: Final = response.get("usage", {}) - total_tokens: Final = raw_meta.get("billed_units", {}).get("total_tokens") or usage_data.get("total_tokens", 0) - input_tokens: Final = raw_meta.get("tokens", {}).get("input_tokens") or usage_data.get("total_tokens", 0) + raw_meta: Final = response.get("meta") + usage_data: Final = response.get("usage") + meta_billed_units: Final = raw_meta.get("billed_units") if isinstance(raw_meta, dict) else None + meta_tokens: Final = raw_meta.get("tokens") if isinstance(raw_meta, dict) else None + usage_total_tokens: Final = usage_data.get("total_tokens", 0) if isinstance(usage_data, dict) else 0 + total_tokens: Final = ( + meta_billed_units.get("total_tokens") if isinstance(meta_billed_units, dict) else None + ) or usage_total_tokens + input_tokens: Final = ( + meta_tokens.get("input_tokens") if isinstance(meta_tokens, dict) else None + ) or usage_total_tokens _billed_units: Final = RerankBilledUnits(total_tokens=total_tokens) _tokens: Final = RerankTokens(input_tokens=input_tokens) rerank_meta: Final = RerankResponseMeta(billed_units=_billed_units, tokens=_tokens)