mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
* fix(anthropic): skip non-dict content items in beta-header and file-id helpers so malformed content lists return 400 instead of 500 Fixes #42094 Supersedes #42101 Co-authored-by: Pawan-Shahane <shahanepawan511@gmail.com> Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): spawn the DB-less regression proxy with -P so the cwd cannot shadow the pinned checkout Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): launch the DB-less proxy via -I -c with an explicit sys.path so python 3.10 works, drop DIRECT_URL, remove restating docstrings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): gate the self-booted DB-less proxy behind the owned_gateway opt-in the Buildkite container cannot satisfy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): move the bare string content item repro to tests/integration Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(tests): wrap the anthropic bare string wire test to the 120 column limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(tests): wrap anthropic common_utils test literals to the 120 column limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style: fix ruff findings in touched test files Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Pawan-Shahane <shahanepawan511@gmail.com> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
83 lines
5.6 KiB
Python
83 lines
5.6 KiB
Python
import json
|
|
import uuid
|
|
from typing import Final
|
|
|
|
import pytest
|
|
from integration._support.client import Gateway, eventually, object_value
|
|
from integration._support.database import read_rows
|
|
from integration._support.wire import Reply, Request, wire_server
|
|
|
|
|
|
@pytest.mark.covers("other.provider_wire.anthropic.tool_history_system_cache_and_internal_fields", "quota_management.spend_tracking.cache_tokens.disjoint_classes_use_explicit_rates")
|
|
def test_anthropic_tool_history_and_cache_tokens_keep_wire_and_accounting_contracts(gateway: Gateway) -> None:
|
|
identity: Final = "anthropic-wire-" + uuid.uuid4().hex
|
|
tool_schema: Final = {"type": "object", "properties": {"x": {"type": "integer"}, "y": {"type": "integer"}}, "required": ["x", "y"]}
|
|
|
|
def respond(request: Request) -> Reply:
|
|
assert request.method == "POST" and request.target == "/v1/messages"
|
|
assert request.headers["x-api-key"] == "synthetic-anthropic-key"
|
|
body: Final = json.loads(request.body)
|
|
assert body["model"] == "claude-sonnet-4-5-20250929"
|
|
assert body["system"] == [{"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}}]
|
|
assert body["tools"][0]["name"] == "add" and body["tools"][0]["input_schema"] == tool_schema
|
|
assert body["max_tokens"] == 16
|
|
assert not {"timeout", "stream_chunk_size", "litellm_params", "litellm_metadata", "rpm", "tpm"}.intersection(body)
|
|
messages: Final = body["messages"]
|
|
assert [message["role"] for message in messages] == ["user", "assistant", "user"]
|
|
assert messages[0]["content"] == [{"type": "text", "text": "first"}]
|
|
assert messages[1]["content"] == [{"type": "tool_use", "id": "history-call", "name": "add", "input": {"x": 1, "y": 2}}]
|
|
assert messages[2]["content"] == [{"type": "tool_result", "tool_use_id": "history-call", "content": "3"}, {"type": "text", "text": "next"}]
|
|
return Reply(body=json.dumps({"id": identity, "type": "message", "role": "assistant", "model": "claude-sonnet-4-5-20250929", "content": [{"type": "tool_use", "id": "next-call", "name": "add", "input": {"x": 3, "y": 4}}], "stop_reason": "tool_use", "stop_sequence": None, "usage": {"input_tokens": 10, "output_tokens": 4, "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7}}).encode())
|
|
|
|
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
|
model: Final = scenario.model(model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key", input_cost_per_token=0.001, output_cost_per_token=0.002, cache_read_input_token_cost=0.0001, cache_creation_input_token_cost=0.002)
|
|
response: Final = gateway.request("POST", "/v1/chat/completions", {
|
|
"model": model, "max_tokens": 16, "timeout": 5,
|
|
"messages": [
|
|
{"role": "system", "content": [{"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}}]},
|
|
{"role": "user", "content": "first"},
|
|
{"role": "assistant", "tool_calls": [{"id": "history-call", "type": "function", "function": {"name": "add", "arguments": '{"x":1,"y":2}'}}]},
|
|
{"role": "tool", "tool_call_id": "history-call", "content": "3"},
|
|
{"role": "user", "content": "next"},
|
|
],
|
|
"tools": [{"type": "function", "function": {"name": "add", "parameters": tool_schema}}],
|
|
})
|
|
assert response.status_code == 200, response.text
|
|
body: Final = response.json()
|
|
assert body["id"].startswith("chatcmpl-")
|
|
assert body["choices"][0]["finish_reason"] == "tool_calls"
|
|
tool: Final = body["choices"][0]["message"]["tool_calls"][0]
|
|
assert tool["id"] == "next-call" and tool["function"]["name"] == "add"
|
|
assert json.loads(tool["function"]["arguments"]) == {"x": 3, "y": 4}
|
|
assert body["usage"]["prompt_tokens"] == 22 and body["usage"]["completion_tokens"] == 4
|
|
assert len(wire.drain()) == 1
|
|
rows: Final = eventually(lambda: read_rows('SELECT spend, prompt_tokens, completion_tokens, metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (body["id"],)), lambda values: len(values) == 1, seconds=70)
|
|
assert float(rows[0]["spend"]) == pytest.approx(10 * 0.001 + 5 * 0.0001 + 7 * 0.002 + 4 * 0.002)
|
|
assert rows[0]["prompt_tokens"] == 22 and rows[0]["completion_tokens"] == 4
|
|
metadata: Final = rows[0]["metadata"]
|
|
parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata)
|
|
assert parsed["cost_breakdown"]["input_cost"] == pytest.approx(0.0245)
|
|
assert parsed["cost_breakdown"]["output_cost"] == pytest.approx(0.008)
|
|
|
|
|
|
@pytest.mark.covers("other.provider_wire.anthropic.bare_string_content_item_is_client_error")
|
|
@pytest.mark.parametrize(
|
|
"text", [pytest.param("what type of file is this?", id="type_word"), pytest.param("hello", id="plain")]
|
|
)
|
|
def test_anthropic_bare_string_content_item_is_rejected_as_client_error_before_the_wire(
|
|
gateway: Gateway, text: str
|
|
) -> None:
|
|
def respond(request: Request) -> Reply:
|
|
raise AssertionError(f"upstream must not be reached: {request.target}")
|
|
|
|
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
|
model: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key"
|
|
)
|
|
response: Final = gateway.request(
|
|
"POST",
|
|
"/v1/chat/completions",
|
|
{"model": model, "max_tokens": 16, "timeout": 5, "messages": [{"role": "system", "content": [text]}]},
|
|
)
|
|
assert response.status_code == 400, response.text
|
|
assert wire.drain() == ()
|