From d89de671d84b1b4cc395a9a2c08d3f4fa0e67c8b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 18 May 2026 14:07:39 +0000 Subject: [PATCH] fix: address bugs in async pass-through, anthropic cache token detection, rerank tests - async_get_available_deployment_for_pass_through: enforce blocked check on specific deployments - cost_calculator: detect anthropic-style usage by attribute presence (not truthiness) to avoid mixing OpenAI cached_tokens into anthropic normalization when read=0 - dashscope rerank tests: pass request to httpx.Response constructions for consistency Co-authored-by: Yassin Kortam --- litellm/cost_calculator.py | 6 +++--- litellm/router.py | 6 ++++++ .../dashscope/test_dashscope_rerank_transformation.py | 10 ++++++++-- 3 files changed, 17 insertions(+), 5 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index b05cd01af2c..2257861aff6 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -375,11 +375,11 @@ def cost_per_token( # noqa: PLR0915 _anthropic_read = getattr(usage_object, "cache_read_input_tokens", None) _anthropic_create = getattr(usage_object, "cache_creation_input_tokens", None) - if _anthropic_read or _anthropic_create: + if _anthropic_read is not None or _anthropic_create is not None: _is_anthropic_style = True - if _anthropic_read: + if _anthropic_read is not None: _cache_read_tokens = float(_anthropic_read) - if _anthropic_create: + if _anthropic_create is not None: _cache_creation_tokens = float(_anthropic_create) if not _cache_read_tokens and cache_read_input_tokens: diff --git a/litellm/router.py b/litellm/router.py index 52e4e9bab85..420c9b8a816 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -10426,6 +10426,12 @@ class Router: # 3. If specific deployment returned, verify if it supports pass-through if isinstance(healthy_deployments, dict): + if (healthy_deployments.get("model_info") or {}).get("blocked") is True: + raise litellm.ServiceUnavailableError( + message=f"Model '{model}' is administratively paused. Contact your proxy admin to unblock it.", + model=model, + llm_provider="", + ) litellm_params = healthy_deployments.get("litellm_params", {}) if litellm_params.get("use_in_pass_through"): return healthy_deployments diff --git a/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py b/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py index 26e3881f83c..0e8d58b6530 100644 --- a/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py +++ b/tests/test_litellm/llms/dashscope/test_dashscope_rerank_transformation.py @@ -166,7 +166,9 @@ class TestDashScopeRerankResponse: def _resp(self, body, status_code=200): return httpx.Response( - status_code=status_code, content=json.dumps(body).encode() + status_code=status_code, + content=json.dumps(body).encode(), + request=httpx.Request("POST", "https://example.com"), ) def test_success_response(self): @@ -291,7 +293,11 @@ class TestDashScopeRerankResponse: assert "Invalid API-key provided." in str(exc_info.value) def test_non_json_response_raises(self): - bad = httpx.Response(status_code=500, content=b"bad gateway") + bad = httpx.Response( + status_code=500, + content=b"bad gateway", + request=httpx.Request("POST", "https://example.com"), + ) with pytest.raises(DashScopeError): self.config.transform_rerank_response( model="qwen3-rerank",