mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test(anthropic): drop low-priority Claude Code error and count_tokens tests
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
6325975f32
commit
4356a00b5f
4 changed files with 0 additions and 302 deletions
|
|
@ -1,50 +0,0 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from integration._support.client import Gateway, eventually
|
||||
from integration._support.database import read_rows
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from integration.messages_endpoint import _claude_code as cc
|
||||
|
||||
|
||||
def test_count_tokens_forwards_to_anthropic_and_bills_nothing(gateway: Gateway) -> None:
|
||||
pytest.skip(
|
||||
"BUG: /v1/messages/count_tokens on an anthropic deployment runs the internal token_counter "
|
||||
"and never forwards to the provider"
|
||||
)
|
||||
request_body: Final = {
|
||||
key: value
|
||||
for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items()
|
||||
if key not in ("stream", "max_tokens", "thinking", "output_config")
|
||||
}
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
assert request.method == "POST"
|
||||
assert request.target == "/v1/messages/count_tokens", request.target
|
||||
body = cc.JSON_OBJECT.validate_json(request.body)
|
||||
assert body == {**request_body, "model": cc.FABLE}, body
|
||||
return Reply(body=b'{"input_tokens": 37}')
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages/count_tokens",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
assert cc.JSON_OBJECT.validate_json(response.content) == {"input_tokens": 37}
|
||||
assert len(wire.drain()) == 1
|
||||
call_id: Final = response.headers.get("x-litellm-call-id", "")
|
||||
assert call_id, dict(response.headers)
|
||||
leftover: Final = eventually(
|
||||
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=20,
|
||||
return_last_on_timeout=True,
|
||||
)
|
||||
assert all(float(row["spend"]) == 0 for row in leftover), leftover
|
||||
|
|
@ -1,136 +0,0 @@
|
|||
import json
|
||||
import uuid
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from integration._support.client import Gateway, eventually
|
||||
from integration._support.database import read_rows
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from integration.messages_endpoint import _claude_code as cc
|
||||
|
||||
|
||||
def _error_body(error_type: str, message: str) -> bytes:
|
||||
return json.dumps({"type": "error", "error": {"type": error_type, "message": message}}).encode()
|
||||
|
||||
|
||||
def _assert_upstream_error_status_passthrough(gateway: Gateway, status: int, error_type: str) -> None:
|
||||
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
assert request.method == "POST"
|
||||
assert request.target == "/v1/messages", request.target
|
||||
body = cc.JSON_OBJECT.validate_json(request.body)
|
||||
assert body["stream"] is False
|
||||
return Reply(status=status, body=_error_body(error_type, "Upstream rejected"))
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(
|
||||
model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY, num_retries=0
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == status, response.text
|
||||
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
|
||||
assert payload["type"] == "error", payload
|
||||
assert payload["error"]["type"] == error_type, payload
|
||||
assert len(wire.drain()) == 1
|
||||
call_id: Final = response.headers.get("x-litellm-call-id", "")
|
||||
assert call_id, dict(response.headers)
|
||||
leftover: Final = eventually(
|
||||
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=20,
|
||||
return_last_on_timeout=True,
|
||||
)
|
||||
assert all(float(row["spend"]) == 0 for row in leftover), leftover
|
||||
|
||||
|
||||
def test_anthropic_529_overloaded_error_passes_through_with_client_status(gateway: Gateway) -> None:
|
||||
pytest.skip(
|
||||
"BUG: upstream 529 overloaded_error is re-raised through exception_type as InternalServerError "
|
||||
"and reaches the client as 500 api_error"
|
||||
)
|
||||
_assert_upstream_error_status_passthrough(gateway, 529, "overloaded_error")
|
||||
|
||||
|
||||
def test_anthropic_429_rate_limit_error_passes_through_with_client_status(gateway: Gateway) -> None:
|
||||
_assert_upstream_error_status_passthrough(gateway, 429, "rate_limit_error")
|
||||
|
||||
|
||||
def test_anthropic_stream_stop_reason_max_tokens_and_refusal_reach_client(gateway: Gateway) -> None:
|
||||
for stop_reason in ("max_tokens", "refusal"):
|
||||
_assert_stream_stop_reason_reaches_client(gateway, stop_reason)
|
||||
|
||||
|
||||
def _assert_stream_stop_reason_reaches_client(gateway: Gateway, stop_reason: str) -> None:
|
||||
identity: Final = f"msg_stop_{uuid.uuid4().hex}"
|
||||
request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
return Reply(
|
||||
content_type="text/event-stream",
|
||||
chunks=(
|
||||
cc.sse_frame(
|
||||
"message_start",
|
||||
{
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"id": identity,
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": cc.SONNET,
|
||||
"content": [],
|
||||
"stop_reason": None,
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 12, "output_tokens": 1},
|
||||
},
|
||||
},
|
||||
),
|
||||
cc.sse_frame(
|
||||
"content_block_start",
|
||||
{"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}},
|
||||
),
|
||||
cc.sse_frame(
|
||||
"content_block_delta",
|
||||
{"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PAR"}},
|
||||
),
|
||||
cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}),
|
||||
cc.sse_frame(
|
||||
"message_delta",
|
||||
{
|
||||
"type": "message_delta",
|
||||
"delta": {"stop_reason": stop_reason, "stop_sequence": None},
|
||||
"usage": {"output_tokens": 32000},
|
||||
},
|
||||
),
|
||||
cc.sse_frame("message_stop", {"type": "message_stop"}),
|
||||
),
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
events: Final = cc.sse_events(response.text)
|
||||
assert events[2][1]["delta"]["text"] == "PAR"
|
||||
assert events[4][1]["delta"]["stop_reason"] == stop_reason
|
||||
assert len(wire.drain()) == 1
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=20,
|
||||
return_last_on_timeout=True,
|
||||
)
|
||||
assert isinstance(rows, list)
|
||||
|
|
@ -1,39 +0,0 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from integration.messages_endpoint import _claude_code as cc
|
||||
|
||||
|
||||
def test_count_tokens_on_openai_deployment_returns_token_count(gateway: Gateway) -> None:
|
||||
request_body: Final = {
|
||||
key: value
|
||||
for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items()
|
||||
if key not in ("stream", "max_tokens", "thinking", "output_config")
|
||||
}
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
assert request.method == "POST"
|
||||
assert request.target == "/responses/input_tokens", request.target
|
||||
body: Final = cc.JSON_OBJECT.validate_json(request.body)
|
||||
assert body["model"] == cc.OPENAI_BACKEND, body
|
||||
assert body["instructions"] == str(request_body["system"]), body["instructions"]
|
||||
assert len(body["input"]) == 1 and body["input"][0]["role"] == "user", body["input"]
|
||||
assert "cache-bust-" in body["input"][0]["content"], body["input"]
|
||||
assert body["tools"] == request_body["tools"], body["tools"]
|
||||
return Reply(body=b'{"object": "response.input_tokens", "input_tokens": 37}')
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages/count_tokens",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
|
||||
assert payload == {"input_tokens": 37}, payload
|
||||
assert len(wire.drain()) == 1
|
||||
|
|
@ -1,77 +0,0 @@
|
|||
import json
|
||||
import uuid
|
||||
from typing import Final
|
||||
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from integration.messages_endpoint import _claude_code as cc
|
||||
|
||||
|
||||
def test_openai_429_error_comes_back_in_anthropic_shape(gateway: Gateway) -> None:
|
||||
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
assert request.method == "POST"
|
||||
assert request.target == "/responses", request.target
|
||||
return Reply(
|
||||
status=429,
|
||||
body=json.dumps(
|
||||
{"error": {"type": "rate_limit_error", "message": "Rate limit reached", "code": "rate_limit_exceeded"}}
|
||||
).encode(),
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(
|
||||
model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY, num_retries=0
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == 429, response.text
|
||||
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
|
||||
assert payload["type"] == "error", payload
|
||||
assert payload["error"]["type"] == "rate_limit_error", payload
|
||||
assert len(wire.drain()) == 1
|
||||
|
||||
|
||||
def test_incomplete_responses_completion_maps_to_max_tokens(gateway: Gateway) -> None:
|
||||
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
assert request.target == "/responses", request.target
|
||||
return Reply(
|
||||
body=cc.responses_completed(
|
||||
"inc",
|
||||
cc.OPENAI_BACKEND,
|
||||
(
|
||||
{
|
||||
"type": "message",
|
||||
"id": "msg_1",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "PAR", "annotations": []}],
|
||||
},
|
||||
),
|
||||
{"input_tokens": 41, "output_tokens": 5, "total_tokens": 46},
|
||||
status="incomplete",
|
||||
incomplete_details={"reason": "max_output_tokens"},
|
||||
)
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages",
|
||||
{**request_body, "model": model},
|
||||
params={"beta": "true"},
|
||||
headers=cc.cli_headers(gateway.key),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
|
||||
assert payload["stop_reason"] == "max_tokens", payload
|
||||
assert len(wire.drain()) == 1
|
||||
Loading…
Add table
Reference in a new issue