mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
fix: address bugs in async pass-through, anthropic cache token detection, rerank tests
- async_get_available_deployment_for_pass_through: enforce blocked check on specific deployments - cost_calculator: detect anthropic-style usage by attribute presence (not truthiness) to avoid mixing OpenAI cached_tokens into anthropic normalization when read=0 - dashscope rerank tests: pass request to httpx.Response constructions for consistency Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
9de0fcea93
commit
d89de671d8
3 changed files with 17 additions and 5 deletions
|
|
@ -375,11 +375,11 @@ def cost_per_token( # noqa: PLR0915
|
|||
|
||||
_anthropic_read = getattr(usage_object, "cache_read_input_tokens", None)
|
||||
_anthropic_create = getattr(usage_object, "cache_creation_input_tokens", None)
|
||||
if _anthropic_read or _anthropic_create:
|
||||
if _anthropic_read is not None or _anthropic_create is not None:
|
||||
_is_anthropic_style = True
|
||||
if _anthropic_read:
|
||||
if _anthropic_read is not None:
|
||||
_cache_read_tokens = float(_anthropic_read)
|
||||
if _anthropic_create:
|
||||
if _anthropic_create is not None:
|
||||
_cache_creation_tokens = float(_anthropic_create)
|
||||
|
||||
if not _cache_read_tokens and cache_read_input_tokens:
|
||||
|
|
|
|||
|
|
@ -10426,6 +10426,12 @@ class Router:
|
|||
|
||||
# 3. If specific deployment returned, verify if it supports pass-through
|
||||
if isinstance(healthy_deployments, dict):
|
||||
if (healthy_deployments.get("model_info") or {}).get("blocked") is True:
|
||||
raise litellm.ServiceUnavailableError(
|
||||
message=f"Model '{model}' is administratively paused. Contact your proxy admin to unblock it.",
|
||||
model=model,
|
||||
llm_provider="",
|
||||
)
|
||||
litellm_params = healthy_deployments.get("litellm_params", {})
|
||||
if litellm_params.get("use_in_pass_through"):
|
||||
return healthy_deployments
|
||||
|
|
|
|||
|
|
@ -166,7 +166,9 @@ class TestDashScopeRerankResponse:
|
|||
|
||||
def _resp(self, body, status_code=200):
|
||||
return httpx.Response(
|
||||
status_code=status_code, content=json.dumps(body).encode()
|
||||
status_code=status_code,
|
||||
content=json.dumps(body).encode(),
|
||||
request=httpx.Request("POST", "https://example.com"),
|
||||
)
|
||||
|
||||
def test_success_response(self):
|
||||
|
|
@ -291,7 +293,11 @@ class TestDashScopeRerankResponse:
|
|||
assert "Invalid API-key provided." in str(exc_info.value)
|
||||
|
||||
def test_non_json_response_raises(self):
|
||||
bad = httpx.Response(status_code=500, content=b"<html>bad gateway</html>")
|
||||
bad = httpx.Response(
|
||||
status_code=500,
|
||||
content=b"<html>bad gateway</html>",
|
||||
request=httpx.Request("POST", "https://example.com"),
|
||||
)
|
||||
with pytest.raises(DashScopeError):
|
||||
self.config.transform_rerank_response(
|
||||
model="qwen3-rerank",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue