mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
test(anthropic): native /v1/messages reasoning integration tests built on a captured Claude Code request (#43361)
* test(integration): group /v1/messages contracts under tests/integration/messages Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): make ci coverage census collect nested test dirs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): nest /v1/messages contracts under messages_endpoint/providers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): replay a real Claude Code /v1/messages request through the native wire Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): drop legacy covers marker from claude code wire test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): inline the Claude Code request instead of a json fixture Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): drop the legacy covers marker from the new contract Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): add Claude Code /v1/messages customer-journey matrix (native + responses bridge) (#43386) * test(anthropic): add Claude Code /v1/messages customer-journey matrix (native + responses bridge) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): strengthen bot-flagged assertions in the Claude Code matrix Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): type the usage mapping parameter in the shared builders Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): drop low-priority Claude Code error and count_tokens tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): send the full 24-tool Claude Code request and pin upstream headers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): drop responses bridge Claude Code tests to keep this PR Anthropic direct only Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): fix duplicate WebSearch tool, drop mutation in stream builders, ignore pings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): assert upstream request order in multi-turn Claude Code tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): name Claude Code wire tests by behavior and move provider-agnostic ones to routing and streaming Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): sort imports after moving the Claude Code fixture into _support Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): group anthropic messages tests into feature subfolders Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): drop the pre-move anthropic test paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): cover native reasoning translation, response and pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): narrow PR to new reasoning tests, restore moved files and drop non-reasoning tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): scope reasoning tests to reasoning and cover betas, thinking usage and streamed pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): write reasoning cases as literal sent and received fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): require stopped stream blocks and check upstream model on switch Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
3a11192f68
commit
4b06d04334
6 changed files with 1780 additions and 0 deletions
1009
tests/integration/_support/claude_code.py
Normal file
1009
tests/integration/_support/claude_code.py
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -0,0 +1,81 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
from integration._support import claude_code as cc
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
|
||||
_READ_A: Final = {"type": "tool_use", "id": "toolu_a", "name": "Read", "input": {"file_path": "/tmp/cc_probe/a.txt"}}
|
||||
_READ_B: Final = {"type": "tool_use", "id": "toolu_b", "name": "Read", "input": {"file_path": "/tmp/cc_probe/b.txt"}}
|
||||
|
||||
|
||||
def test_interleaved_thinking_history_reaches_anthropic_and_interleaved_blocks_stream_back(gateway: Gateway) -> None:
|
||||
turn1: Final = cc.frontier_request(
|
||||
f"cache-bust-{uuid.uuid4().hex}",
|
||||
"high",
|
||||
64000,
|
||||
prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words",
|
||||
)
|
||||
turn2: Final = cc.tool_loop_turn2(
|
||||
turn1, ({"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A), (("toolu_a", "ALPHA"),)
|
||||
)
|
||||
turn3: Final = cc.tool_loop_turn2(
|
||||
turn2,
|
||||
({"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B),
|
||||
(("toolu_b", "BRAVO"),),
|
||||
)
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
if b"toolu_b" not in request.body:
|
||||
return Reply(
|
||||
content_type="text/event-stream",
|
||||
chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}),
|
||||
)
|
||||
return Reply(
|
||||
content_type="text/event-stream",
|
||||
chunks=cc.message_stream(
|
||||
f"msg_il_{uuid.uuid4().hex}",
|
||||
cc.FABLE,
|
||||
(
|
||||
{"type": "thinking", "thinking": "got B", "signature": "sig3"},
|
||||
{"type": "text", "text": "got B"},
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": "toolu_c",
|
||||
"name": "Read",
|
||||
"input": {"file_path": "/tmp/cc_probe/c.txt"},
|
||||
},
|
||||
),
|
||||
{"input_tokens": 20, "output_tokens": 12},
|
||||
),
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA)
|
||||
response2: Final = gateway.request(
|
||||
"POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers
|
||||
)
|
||||
assert response2.status_code == 200, response2.text
|
||||
response3: Final = gateway.request(
|
||||
"POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers
|
||||
)
|
||||
assert response3.status_code == 200, response3.text
|
||||
received: Final = wire.drain()
|
||||
assert len(received) == 2, received
|
||||
second: Final = cc.forwarded(turn2, received[0])
|
||||
third: Final = cc.forwarded(turn3, received[1])
|
||||
assert second.assistant_history == ([{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A],)
|
||||
assert third.assistant_history == (
|
||||
[{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A],
|
||||
[{"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B],
|
||||
), third.assistant_history
|
||||
adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}
|
||||
assert (second.reasoning, third.reasoning) == (adaptive_high, adaptive_high)
|
||||
assert (second.other_changes, third.other_changes) == ({}, {})
|
||||
assert (second.reasoning_betas, third.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2
|
||||
assert cc.streamed_content(response3.text) == [
|
||||
{"type": "thinking", "thinking": "got B", "signature": "sig3"},
|
||||
{"type": "text", "text": "got B"},
|
||||
{"type": "tool_use", "id": "toolu_c", "name": "Read", "input": {"file_path": "/tmp/cc_probe/c.txt"}},
|
||||
], response3.text
|
||||
|
|
@ -0,0 +1,81 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
from integration._support import claude_code as cc
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
|
||||
|
||||
def test_mid_loop_model_switch_replays_thinking_history_and_reasoning_unchanged(gateway: Gateway) -> None:
|
||||
turn1: Final = cc.frontier_request(
|
||||
f"cache-bust-{uuid.uuid4().hex}",
|
||||
"high",
|
||||
64000,
|
||||
prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word",
|
||||
)
|
||||
turn2: Final = cc.tool_loop_turn2(
|
||||
turn1,
|
||||
(
|
||||
{"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"},
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": "toolu_read_1",
|
||||
"name": "Read",
|
||||
"input": {"file_path": "/tmp/cc_probe/hello.txt"},
|
||||
},
|
||||
),
|
||||
(("toolu_read_1", "1\tPROBE\n2\t"),),
|
||||
)
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
if b"tool_result" not in request.body:
|
||||
return Reply(
|
||||
content_type="text/event-stream",
|
||||
chunks=cc.tool_use_stream(
|
||||
f"msg_{uuid.uuid4().hex}",
|
||||
cc.FABLE,
|
||||
"need to read the file",
|
||||
"sig_anthropic_1",
|
||||
(("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),),
|
||||
{"input_tokens": 20, "output_tokens": 10},
|
||||
),
|
||||
)
|
||||
return Reply(
|
||||
content_type="text/event-stream",
|
||||
chunks=cc.text_stream(
|
||||
f"msg_{uuid.uuid4().hex}", cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3}
|
||||
),
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA)
|
||||
response1: Final = gateway.request(
|
||||
"POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers
|
||||
)
|
||||
assert response1.status_code == 200, response1.text
|
||||
response2: Final = gateway.request(
|
||||
"POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers
|
||||
)
|
||||
assert response2.status_code == 200, response2.text
|
||||
received: Final = wire.drain()
|
||||
assert len(received) == 2, received
|
||||
to_fable: Final = cc.forwarded(turn1, received[0])
|
||||
to_opus: Final = cc.forwarded(turn2, received[1])
|
||||
assert (to_fable.model, to_opus.model) == (cc.FABLE, cc.OPUS), received
|
||||
assert to_opus.assistant_history == (
|
||||
[
|
||||
{"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"},
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": "toolu_read_1",
|
||||
"name": "Read",
|
||||
"input": {"file_path": "/tmp/cc_probe/hello.txt"},
|
||||
},
|
||||
],
|
||||
), to_opus.assistant_history
|
||||
adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}
|
||||
assert (to_fable.reasoning, to_opus.reasoning) == (adaptive_high, adaptive_high)
|
||||
assert (to_fable.other_changes, to_opus.other_changes) == ({}, {})
|
||||
assert (to_fable.reasoning_betas, to_opus.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2
|
||||
|
|
@ -0,0 +1,482 @@
|
|||
import uuid
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from integration._support import claude_code as cc
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from pydantic import JsonValue
|
||||
|
||||
|
||||
def _claude_code_turn(sent: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
|
||||
default_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")
|
||||
without_reasoning: Final = {key: value for key, value in default_turn.items() if key != "thinking"}
|
||||
return {**without_reasoning, "stream": False, **sent}
|
||||
|
||||
|
||||
def _forward(gateway: Gateway, upstream_model: str, sent: Mapping[str, JsonValue]) -> cc.Forwarded:
|
||||
client_body: Final = _claude_code_turn(sent)
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
return Reply(
|
||||
body=cc.message_reply(
|
||||
f"msg_{uuid.uuid4().hex}",
|
||||
upstream_model,
|
||||
({"type": "text", "text": "PONG"},),
|
||||
{"input_tokens": 12, "output_tokens": 4},
|
||||
)
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(
|
||||
model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
"/v1/messages",
|
||||
{**client_body, "model": model},
|
||||
headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA),
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
received: Final = wire.drain()
|
||||
assert len(received) == 1, received
|
||||
return cc.forwarded(client_body, received[0])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "sent", "received"),
|
||||
(
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "low"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 1024}},
|
||||
id="haiku-4.5-low",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "medium"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
id="haiku-4.5-medium",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 4096}},
|
||||
id="haiku-4.5-high",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 8192}},
|
||||
id="haiku-4.5-xhigh",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "max"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 16384}},
|
||||
id="haiku-4.5-max",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "max"},
|
||||
"max_tokens": 4000,
|
||||
},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 3999}},
|
||||
id="haiku-4.5-budget-capped-below-max-tokens",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
"max_tokens": 1024,
|
||||
},
|
||||
{},
|
||||
id="haiku-4.5-max-tokens-below-minimum-budget",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
{"output_config": {"effort": "high"}},
|
||||
id="opus-4.5-keeps-effort-drops-adaptive",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 8192}},
|
||||
id="opus-4.5-xhigh-falls-back-to-budget",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-6",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
id="opus-4.6-unchanged",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
id="opus-4.7-unchanged",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-fable-5-1",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
|
||||
id="fable-5.1-unchanged",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-5-5",
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
{"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
|
||||
id="opus-5.5-unchanged",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_adaptive_thinking_and_effort_are_rewritten_only_for_models_without_adaptive_thinking(
|
||||
gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, upstream_model, sent)
|
||||
assert forwarded.reasoning == received, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "sent", "received"),
|
||||
(
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 1024}},
|
||||
{"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
|
||||
id="opus-4.7-1024-is-low",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
{"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
|
||||
id="opus-4.7-2048-is-medium",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 4096}},
|
||||
{"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
|
||||
id="opus-4.7-4096-is-high",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 8192}},
|
||||
{"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
|
||||
id="opus-4.7-8192-is-xhigh",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 8192}, "output_config": {"effort": "medium"}},
|
||||
{"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
|
||||
id="opus-4.7-keeps-the-callers-effort",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
id="haiku-4.5-unchanged",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_legacy_thinking_budget_becomes_adaptive_effort_only_on_models_that_reject_budgets(
|
||||
gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, upstream_model, sent)
|
||||
assert forwarded.reasoning == received, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "sent", "received"),
|
||||
(
|
||||
pytest.param(
|
||||
"claude-fable-5-1",
|
||||
{"thinking": {"type": "disabled"}},
|
||||
{},
|
||||
id="fable-5.1-always-thinks-so-disabled-is-dropped",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"thinking": {"type": "disabled"}},
|
||||
{"thinking": {"type": "disabled"}},
|
||||
id="opus-4.7-keeps-disabled",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_disabled_thinking_is_dropped_only_for_always_on_thinking_models(
|
||||
gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, upstream_model, sent)
|
||||
assert forwarded.reasoning == received, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "sent", "received"),
|
||||
(
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "minimal"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
|
||||
id="opus-4.7-minimal",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "low"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
|
||||
id="opus-4.7-low",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "medium"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "medium"}},
|
||||
id="opus-4.7-medium",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "high"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}},
|
||||
id="opus-4.7-high",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "xhigh"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "xhigh"}},
|
||||
id="opus-4.7-xhigh",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "max"},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "max"}},
|
||||
id="opus-4.7-max",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "minimal"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 1024}},
|
||||
id="haiku-4.5-minimal",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "low"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 1024}},
|
||||
id="haiku-4.5-low",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "medium"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
id="haiku-4.5-medium",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "high"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 4096}},
|
||||
id="haiku-4.5-high",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "xhigh"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 8192}},
|
||||
id="haiku-4.5-xhigh",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "max"},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 16384}},
|
||||
id="haiku-4.5-max",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "max", "max_tokens": 4000},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 3999}},
|
||||
id="haiku-4.5-budget-capped-below-max-tokens",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "high", "max_tokens": 1024},
|
||||
{},
|
||||
id="haiku-4.5-max-tokens-below-minimum-budget",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{
|
||||
"reasoning_effort": "none",
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
{},
|
||||
id="none-clears-thinking-and-effort",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"reasoning_effort": "high", "thinking": {"type": "enabled", "budget_tokens": 2000}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2000}},
|
||||
id="callers-thinking-wins",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-7",
|
||||
{"reasoning_effort": "high", "output_config": {"effort": "low"}},
|
||||
{"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
|
||||
id="callers-effort-wins",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_reasoning_effort_becomes_the_thinking_shape_each_model_accepts(
|
||||
gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, upstream_model, sent)
|
||||
assert forwarded.reasoning == received, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "sent", "received"),
|
||||
(
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{
|
||||
"temperature": 0,
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 4096}},
|
||||
id="haiku-4.5-drops-temperature-0-with-effort",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-5",
|
||||
{
|
||||
"temperature": 0,
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
{"output_config": {"effort": "high"}},
|
||||
id="opus-4.5-drops-temperature-0-with-effort",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"temperature": 0, "thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
{"thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
id="haiku-4.5-drops-temperature-0-with-budget",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
{"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}},
|
||||
id="haiku-4.5-keeps-temperature-1",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-haiku-4-5",
|
||||
{"temperature": 0},
|
||||
{"temperature": 0},
|
||||
id="haiku-4.5-keeps-temperature-without-thinking",
|
||||
),
|
||||
pytest.param(
|
||||
"claude-opus-4-6",
|
||||
{
|
||||
"temperature": 0,
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
{
|
||||
"temperature": 0,
|
||||
"thinking": {"type": "adaptive", "display": "omitted"},
|
||||
"output_config": {"effort": "high"},
|
||||
},
|
||||
id="opus-4.6-adaptive-keeps-temperature",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_temperature_is_dropped_only_when_a_non_adaptive_model_thinks(
|
||||
gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, upstream_model, sent)
|
||||
assert forwarded.reasoning == received, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
def _tool_loop(assistant_content: list[JsonValue]) -> dict[str, JsonValue]:
|
||||
first_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")["messages"]
|
||||
assert isinstance(first_turn, list)
|
||||
tool_result: Final = {"type": "tool_result", "tool_use_id": "toolu_01", "content": "ok"}
|
||||
return {
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048},
|
||||
"messages": [
|
||||
*first_turn,
|
||||
{"role": "assistant", "content": assistant_content},
|
||||
{"role": "user", "content": [tool_result]},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("sent_history", "received_history"),
|
||||
(
|
||||
pytest.param(
|
||||
[
|
||||
{"type": "thinking", "thinking": "bridge reasoning", "signature": "litellm_encrypted_reasoning:gAAAAB"},
|
||||
{"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:gAAAAC"},
|
||||
{"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"},
|
||||
{"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
|
||||
],
|
||||
[
|
||||
{"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"},
|
||||
{"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
|
||||
],
|
||||
id="encrypted-reasoning-from-another-provider-stripped-anthropic-signed-kept",
|
||||
),
|
||||
pytest.param(
|
||||
[
|
||||
{"type": "thinking", "thinking": "", "signature": "EqQBCkgIBRABGAIiQM"},
|
||||
{"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"},
|
||||
{"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
|
||||
],
|
||||
[
|
||||
{"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"},
|
||||
{"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
|
||||
],
|
||||
id="empty-thinking-stripped-redacted-thinking-kept",
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_thinking_history_keeps_only_blocks_anthropic_can_verify(
|
||||
gateway: Gateway, sent_history: list[JsonValue], received_history: list[JsonValue]
|
||||
) -> None:
|
||||
forwarded: Final = _forward(gateway, "claude-haiku-4-5", _tool_loop(sent_history))
|
||||
assert forwarded.assistant_history == (received_history,), forwarded.assistant_history
|
||||
assert forwarded.reasoning == {"thinking": {"type": "enabled", "budget_tokens": 2048}}, forwarded
|
||||
assert forwarded.other_changes == {}, forwarded.other_changes
|
||||
assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("upstream_model", "reasoning_effort"),
|
||||
(
|
||||
pytest.param("claude-haiku-4-5", "turbo", id="unknown-value"),
|
||||
pytest.param("claude-opus-4-6", "xhigh", id="level-the-model-lacks"),
|
||||
),
|
||||
)
|
||||
def test_unsupported_reasoning_effort_is_rejected_before_reaching_anthropic(
|
||||
gateway: Gateway, upstream_model: str, reasoning_effort: str
|
||||
) -> None:
|
||||
with wire_server(lambda request: Reply()) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(
|
||||
model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY
|
||||
)
|
||||
response: Final = gateway.request(
|
||||
"POST", "/v1/messages", {**_claude_code_turn({"reasoning_effort": reasoning_effort}), "model": model}
|
||||
)
|
||||
assert response.status_code == 400, response.text
|
||||
assert response.json()["error"]["type"] == "invalid_request_error", response.text
|
||||
assert wire.drain() == ()
|
||||
|
|
@ -0,0 +1,63 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
from integration._support import claude_code as cc
|
||||
from integration._support.client import Gateway
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
from pydantic import JsonValue
|
||||
|
||||
_MODEL: Final = "claude-haiku-4-5"
|
||||
_ANTHROPIC_CONTENT: Final = (
|
||||
{"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
|
||||
{"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
|
||||
{"type": "text", "text": "PONG"},
|
||||
)
|
||||
_ANTHROPIC_USAGE: Final = {"input_tokens": 12, "output_tokens": 30, "output_tokens_details": {"thinking_tokens": 20}}
|
||||
|
||||
|
||||
def _client_body(stream: bool) -> dict[str, JsonValue]:
|
||||
return {
|
||||
**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"),
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048},
|
||||
"stream": stream,
|
||||
}
|
||||
|
||||
|
||||
def test_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None:
|
||||
def respond(request: Request) -> Reply:
|
||||
return Reply(
|
||||
chunks=cc.message_stream(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE),
|
||||
content_type="text/event-stream",
|
||||
)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=True), "model": model})
|
||||
assert response.status_code == 200, response.text
|
||||
assert len(wire.drain()) == 1
|
||||
assert cc.streamed_content(response.text) == [
|
||||
{"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
|
||||
{"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
|
||||
{"type": "text", "text": "PONG"},
|
||||
], response.text
|
||||
assert cc.streamed_usage(response.text) == {
|
||||
"output_tokens": 30,
|
||||
"output_tokens_details": {"thinking_tokens": 20},
|
||||
}, response.text
|
||||
|
||||
|
||||
def test_non_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None:
|
||||
def respond(request: Request) -> Reply:
|
||||
return Reply(body=cc.message_reply(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE))
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
|
||||
response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=False), "model": model})
|
||||
assert response.status_code == 200, response.text
|
||||
assert len(wire.drain()) == 1
|
||||
assert response.json()["content"] == [
|
||||
{"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
|
||||
{"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
|
||||
{"type": "text", "text": "PONG"},
|
||||
], response.text
|
||||
assert response.json()["usage"]["output_tokens_details"] == {"thinking_tokens": 20}, response.text
|
||||
|
|
@ -0,0 +1,64 @@
|
|||
import uuid
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from integration._support import claude_code as cc
|
||||
from integration._support.client import Gateway, eventually
|
||||
from integration._support.database import read_rows
|
||||
from integration._support.wire import Reply, Request, wire_server
|
||||
|
||||
_MODEL: Final = "claude-haiku-4-5"
|
||||
_INPUT_RATE: Final = 1e-6
|
||||
_OUTPUT_RATE: Final = 2e-6
|
||||
_REASONING_RATE: Final = 7e-6
|
||||
_CONTENT: Final = (
|
||||
{"type": "thinking", "thinking": "count the words", "signature": "EqQBCkgIBRABGAIiQLz"},
|
||||
{"type": "text", "text": "PONG"},
|
||||
)
|
||||
_USAGE: Final = {"input_tokens": 100, "output_tokens": 50, "output_tokens_details": {"thinking_tokens": 30}}
|
||||
|
||||
|
||||
def _reply(identity: str, stream: bool) -> Reply:
|
||||
if stream:
|
||||
return Reply(chunks=cc.message_stream(identity, _MODEL, _CONTENT, _USAGE), content_type="text/event-stream")
|
||||
return Reply(body=cc.message_reply(identity, _MODEL, _CONTENT, _USAGE))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("stream", (pytest.param(False, id="non-streamed"), pytest.param(True, id="streamed")))
|
||||
def test_reported_thinking_tokens_are_billed_at_the_reasoning_rate_and_the_rest_at_the_output_rate(
|
||||
gateway: Gateway, stream: bool
|
||||
) -> None:
|
||||
identity: Final = f"msg_{uuid.uuid4().hex}"
|
||||
|
||||
def respond(request: Request) -> Reply:
|
||||
return _reply(identity, stream)
|
||||
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = scenario.model(
|
||||
model=f"anthropic/{_MODEL}",
|
||||
api_base=wire.url,
|
||||
api_key=cc.ANTHROPIC_API_KEY,
|
||||
input_cost_per_token=_INPUT_RATE,
|
||||
output_cost_per_token=_OUTPUT_RATE,
|
||||
output_cost_per_reasoning_token=_REASONING_RATE,
|
||||
)
|
||||
body: Final = {
|
||||
**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"),
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048},
|
||||
"stream": stream,
|
||||
"model": model,
|
||||
}
|
||||
response: Final = gateway.request("POST", "/v1/messages", body)
|
||||
assert response.status_code == 200, response.text
|
||||
assert len(wire.drain()) == 1
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
input_tokens, output_tokens, thinking_tokens = 100, 50, 30
|
||||
assert float(rows[0]["spend"]) == pytest.approx(
|
||||
input_tokens * _INPUT_RATE
|
||||
+ (output_tokens - thinking_tokens) * _OUTPUT_RATE
|
||||
+ thinking_tokens * _REASONING_RATE
|
||||
), rows
|
||||
Loading…
Add table
Reference in a new issue