diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py deleted file mode 100644 index 3a4bd61fc8e..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py +++ /dev/null @@ -1,50 +0,0 @@ -import uuid -from typing import Final - -import pytest - -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_count_tokens_forwards_to_anthropic_and_bills_nothing(gateway: Gateway) -> None: - pytest.skip( - "BUG: /v1/messages/count_tokens on an anthropic deployment runs the internal token_counter " - "and never forwards to the provider" - ) - request_body: Final = { - key: value - for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() - if key not in ("stream", "max_tokens", "thinking", "output_config") - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages/count_tokens", request.target - body = cc.JSON_OBJECT.validate_json(request.body) - assert body == {**request_body, "model": cc.FABLE}, body - return Reply(body=b'{"input_tokens": 37}') - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages/count_tokens", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - assert cc.JSON_OBJECT.validate_json(response.content) == {"input_tokens": 37} - assert len(wire.drain()) == 1 - call_id: Final = response.headers.get("x-litellm-call-id", "") - assert call_id, dict(response.headers) - leftover: Final = eventually( - lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert all(float(row["spend"]) == 0 for row in leftover), leftover diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py deleted file mode 100644 index 17cfaee9dd2..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py +++ /dev/null @@ -1,136 +0,0 @@ -import json -import uuid -from typing import Final - -import pytest - -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def _error_body(error_type: str, message: str) -> bytes: - return json.dumps({"type": "error", "error": {"type": error_type, "message": message}}).encode() - - -def _assert_upstream_error_status_passthrough(gateway: Gateway, status: int, error_type: str) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body = cc.JSON_OBJECT.validate_json(request.body) - assert body["stream"] is False - return Reply(status=status, body=_error_body(error_type, "Upstream rejected")) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY, num_retries=0 - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == status, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["type"] == "error", payload - assert payload["error"]["type"] == error_type, payload - assert len(wire.drain()) == 1 - call_id: Final = response.headers.get("x-litellm-call-id", "") - assert call_id, dict(response.headers) - leftover: Final = eventually( - lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert all(float(row["spend"]) == 0 for row in leftover), leftover - - -def test_anthropic_529_overloaded_error_passes_through_with_client_status(gateway: Gateway) -> None: - pytest.skip( - "BUG: upstream 529 overloaded_error is re-raised through exception_type as InternalServerError " - "and reaches the client as 500 api_error" - ) - _assert_upstream_error_status_passthrough(gateway, 529, "overloaded_error") - - -def test_anthropic_429_rate_limit_error_passes_through_with_client_status(gateway: Gateway) -> None: - _assert_upstream_error_status_passthrough(gateway, 429, "rate_limit_error") - - -def test_anthropic_stream_stop_reason_max_tokens_and_refusal_reach_client(gateway: Gateway) -> None: - for stop_reason in ("max_tokens", "refusal"): - _assert_stream_stop_reason_reaches_client(gateway, stop_reason) - - -def _assert_stream_stop_reason_reaches_client(gateway: Gateway, stop_reason: str) -> None: - identity: Final = f"msg_stop_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - - def respond(request: Request) -> Reply: - return Reply( - content_type="text/event-stream", - chunks=( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.SONNET, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PAR"}}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": stop_reason, "stop_sequence": None}, - "usage": {"output_tokens": 32000}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - assert events[2][1]["delta"]["text"] == "PAR" - assert events[4][1]["delta"]["stop_reason"] == stop_reason - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert isinstance(rows, list) diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py deleted file mode 100644 index b03e06ff867..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py +++ /dev/null @@ -1,39 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_count_tokens_on_openai_deployment_returns_token_count(gateway: Gateway) -> None: - request_body: Final = { - key: value - for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() - if key not in ("stream", "max_tokens", "thinking", "output_config") - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses/input_tokens", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - assert body["model"] == cc.OPENAI_BACKEND, body - assert body["instructions"] == str(request_body["system"]), body["instructions"] - assert len(body["input"]) == 1 and body["input"][0]["role"] == "user", body["input"] - assert "cache-bust-" in body["input"][0]["content"], body["input"] - assert body["tools"] == request_body["tools"], body["tools"] - return Reply(body=b'{"object": "response.input_tokens", "input_tokens": 37}') - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages/count_tokens", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload == {"input_tokens": 37}, payload - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py deleted file mode 100644 index 32a7ce3bd6e..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py +++ /dev/null @@ -1,77 +0,0 @@ -import json -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_openai_429_error_comes_back_in_anthropic_shape(gateway: Gateway) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - return Reply( - status=429, - body=json.dumps( - {"error": {"type": "rate_limit_error", "message": "Rate limit reached", "code": "rate_limit_exceeded"}} - ).encode(), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY, num_retries=0 - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 429, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["type"] == "error", payload - assert payload["error"]["type"] == "rate_limit_error", payload - assert len(wire.drain()) == 1 - - -def test_incomplete_responses_completion_maps_to_max_tokens(gateway: Gateway) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.target == "/responses", request.target - return Reply( - body=cc.responses_completed( - "inc", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PAR", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, - status="incomplete", - incomplete_details={"reason": "max_output_tokens"}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["stop_reason"] == "max_tokens", payload - assert len(wire.drain()) == 1