test(anthropic): drop low-priority Claude Code error and count_tokens tests

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-28 23:32:03 +00:00
parent 6325975f32
commit 4356a00b5f
4 changed files with 0 additions and 302 deletions

View file

@ -1,50 +0,0 @@
import uuid
from typing import Final
import pytest
from integration._support.client import Gateway, eventually
from integration._support.database import read_rows
from integration._support.wire import Reply, Request, wire_server
from integration.messages_endpoint import _claude_code as cc
def test_count_tokens_forwards_to_anthropic_and_bills_nothing(gateway: Gateway) -> None:
pytest.skip(
"BUG: /v1/messages/count_tokens on an anthropic deployment runs the internal token_counter "
"and never forwards to the provider"
)
request_body: Final = {
key: value
for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items()
if key not in ("stream", "max_tokens", "thinking", "output_config")
}
def respond(request: Request) -> Reply:
assert request.method == "POST"
assert request.target == "/v1/messages/count_tokens", request.target
body = cc.JSON_OBJECT.validate_json(request.body)
assert body == {**request_body, "model": cc.FABLE}, body
return Reply(body=b'{"input_tokens": 37}')
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
response: Final = gateway.request(
"POST",
"/v1/messages/count_tokens",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == 200, response.text
assert cc.JSON_OBJECT.validate_json(response.content) == {"input_tokens": 37}
assert len(wire.drain()) == 1
call_id: Final = response.headers.get("x-litellm-call-id", "")
assert call_id, dict(response.headers)
leftover: Final = eventually(
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)),
lambda values: len(values) == 1,
seconds=20,
return_last_on_timeout=True,
)
assert all(float(row["spend"]) == 0 for row in leftover), leftover

View file

@ -1,136 +0,0 @@
import json
import uuid
from typing import Final
import pytest
from integration._support.client import Gateway, eventually
from integration._support.database import read_rows
from integration._support.wire import Reply, Request, wire_server
from integration.messages_endpoint import _claude_code as cc
def _error_body(error_type: str, message: str) -> bytes:
return json.dumps({"type": "error", "error": {"type": error_type, "message": message}}).encode()
def _assert_upstream_error_status_passthrough(gateway: Gateway, status: int, error_type: str) -> None:
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
def respond(request: Request) -> Reply:
assert request.method == "POST"
assert request.target == "/v1/messages", request.target
body = cc.JSON_OBJECT.validate_json(request.body)
assert body["stream"] is False
return Reply(status=status, body=_error_body(error_type, "Upstream rejected"))
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY, num_retries=0
)
response: Final = gateway.request(
"POST",
"/v1/messages",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == status, response.text
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
assert payload["type"] == "error", payload
assert payload["error"]["type"] == error_type, payload
assert len(wire.drain()) == 1
call_id: Final = response.headers.get("x-litellm-call-id", "")
assert call_id, dict(response.headers)
leftover: Final = eventually(
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)),
lambda values: len(values) == 1,
seconds=20,
return_last_on_timeout=True,
)
assert all(float(row["spend"]) == 0 for row in leftover), leftover
def test_anthropic_529_overloaded_error_passes_through_with_client_status(gateway: Gateway) -> None:
pytest.skip(
"BUG: upstream 529 overloaded_error is re-raised through exception_type as InternalServerError "
"and reaches the client as 500 api_error"
)
_assert_upstream_error_status_passthrough(gateway, 529, "overloaded_error")
def test_anthropic_429_rate_limit_error_passes_through_with_client_status(gateway: Gateway) -> None:
_assert_upstream_error_status_passthrough(gateway, 429, "rate_limit_error")
def test_anthropic_stream_stop_reason_max_tokens_and_refusal_reach_client(gateway: Gateway) -> None:
for stop_reason in ("max_tokens", "refusal"):
_assert_stream_stop_reason_reaches_client(gateway, stop_reason)
def _assert_stream_stop_reason_reaches_client(gateway: Gateway, stop_reason: str) -> None:
identity: Final = f"msg_stop_{uuid.uuid4().hex}"
request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")
def respond(request: Request) -> Reply:
return Reply(
content_type="text/event-stream",
chunks=(
cc.sse_frame(
"message_start",
{
"type": "message_start",
"message": {
"id": identity,
"type": "message",
"role": "assistant",
"model": cc.SONNET,
"content": [],
"stop_reason": None,
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 1},
},
},
),
cc.sse_frame(
"content_block_start",
{"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}},
),
cc.sse_frame(
"content_block_delta",
{"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PAR"}},
),
cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}),
cc.sse_frame(
"message_delta",
{
"type": "message_delta",
"delta": {"stop_reason": stop_reason, "stop_sequence": None},
"usage": {"output_tokens": 32000},
},
),
cc.sse_frame("message_stop", {"type": "message_stop"}),
),
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
response: Final = gateway.request(
"POST",
"/v1/messages",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == 200, response.text
events: Final = cc.sse_events(response.text)
assert events[2][1]["delta"]["text"] == "PAR"
assert events[4][1]["delta"]["stop_reason"] == stop_reason
assert len(wire.drain()) == 1
rows: Final = eventually(
lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)),
lambda values: len(values) == 1,
seconds=20,
return_last_on_timeout=True,
)
assert isinstance(rows, list)

View file

@ -1,39 +0,0 @@
import uuid
from typing import Final
from integration._support.client import Gateway
from integration._support.wire import Reply, Request, wire_server
from integration.messages_endpoint import _claude_code as cc
def test_count_tokens_on_openai_deployment_returns_token_count(gateway: Gateway) -> None:
request_body: Final = {
key: value
for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items()
if key not in ("stream", "max_tokens", "thinking", "output_config")
}
def respond(request: Request) -> Reply:
assert request.method == "POST"
assert request.target == "/responses/input_tokens", request.target
body: Final = cc.JSON_OBJECT.validate_json(request.body)
assert body["model"] == cc.OPENAI_BACKEND, body
assert body["instructions"] == str(request_body["system"]), body["instructions"]
assert len(body["input"]) == 1 and body["input"][0]["role"] == "user", body["input"]
assert "cache-bust-" in body["input"][0]["content"], body["input"]
assert body["tools"] == request_body["tools"], body["tools"]
return Reply(body=b'{"object": "response.input_tokens", "input_tokens": 37}')
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY)
response: Final = gateway.request(
"POST",
"/v1/messages/count_tokens",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == 200, response.text
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
assert payload == {"input_tokens": 37}, payload
assert len(wire.drain()) == 1

View file

@ -1,77 +0,0 @@
import json
import uuid
from typing import Final
from integration._support.client import Gateway
from integration._support.wire import Reply, Request, wire_server
from integration.messages_endpoint import _claude_code as cc
def test_openai_429_error_comes_back_in_anthropic_shape(gateway: Gateway) -> None:
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
def respond(request: Request) -> Reply:
assert request.method == "POST"
assert request.target == "/responses", request.target
return Reply(
status=429,
body=json.dumps(
{"error": {"type": "rate_limit_error", "message": "Rate limit reached", "code": "rate_limit_exceeded"}}
).encode(),
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY, num_retries=0
)
response: Final = gateway.request(
"POST",
"/v1/messages",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == 429, response.text
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
assert payload["type"] == "error", payload
assert payload["error"]["type"] == "rate_limit_error", payload
assert len(wire.drain()) == 1
def test_incomplete_responses_completion_maps_to_max_tokens(gateway: Gateway) -> None:
request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False}
def respond(request: Request) -> Reply:
assert request.target == "/responses", request.target
return Reply(
body=cc.responses_completed(
"inc",
cc.OPENAI_BACKEND,
(
{
"type": "message",
"id": "msg_1",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "PAR", "annotations": []}],
},
),
{"input_tokens": 41, "output_tokens": 5, "total_tokens": 46},
status="incomplete",
incomplete_details={"reason": "max_output_tokens"},
)
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY)
response: Final = gateway.request(
"POST",
"/v1/messages",
{**request_body, "model": model},
params={"beta": "true"},
headers=cc.cli_headers(gateway.key),
)
assert response.status_code == 200, response.text
payload: Final = cc.JSON_OBJECT.validate_json(response.content)
assert payload["stop_reason"] == "max_tokens", payload
assert len(wire.drain()) == 1