From 9260932ddf5d03e28c3d2284e19849583843645f Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 21:59:47 +0000 Subject: [PATCH 01/19] test(integration): group /v1/messages contracts under tests/integration/messages Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/README.md | 2 ++ tests/integration/_support/manifest.py | 1 + .../anthropic}/test_anthropic_advisor_wire.py | 0 .../anthropic}/test_anthropic_legacy_thinking_budget_wire.py | 0 .../anthropic}/test_anthropic_messages_timeout_wire.py | 0 .../test_anthropic_thinking_signature_retry_wire.py | 0 .../{providers => messages/anthropic}/test_anthropic_wire.py | 0 .../anthropic}/test_websearch_interception_wire.py | 0 .../bedrock}/test_bedrock_invoke_tool_search_wire.py | 0 .../bedrock}/test_bedrock_messages_web_search_replay_wire.py | 0 .../test_anthropic_messages_fireworks_stop_wire.py | 0 .../gemini}/test_gemini_messages_cache_control_wire.py | 0 .../test_anthropic_messages_claude_code_cache_key_wire.py | 0 .../test_anthropic_messages_openai_bridge_wire.py | 0 .../test_anthropic_messages_openai_tools_wire.py | 0 .../responses_bridge}/test_responses_bridge_stream_options.py | 0 tests/integration/run.py | 4 ++-- 17 files changed, 5 insertions(+), 2 deletions(-) rename tests/integration/{providers => messages/anthropic}/test_anthropic_advisor_wire.py (100%) rename tests/integration/{providers => messages/anthropic}/test_anthropic_legacy_thinking_budget_wire.py (100%) rename tests/integration/{providers => messages/anthropic}/test_anthropic_messages_timeout_wire.py (100%) rename tests/integration/{providers => messages/anthropic}/test_anthropic_thinking_signature_retry_wire.py (100%) rename tests/integration/{providers => messages/anthropic}/test_anthropic_wire.py (100%) rename tests/integration/{providers => messages/anthropic}/test_websearch_interception_wire.py (100%) rename tests/integration/{providers => messages/bedrock}/test_bedrock_invoke_tool_search_wire.py (100%) rename tests/integration/{providers => messages/bedrock}/test_bedrock_messages_web_search_replay_wire.py (100%) rename tests/integration/{providers => messages/chat_bridge}/test_anthropic_messages_fireworks_stop_wire.py (100%) rename tests/integration/{providers => messages/gemini}/test_gemini_messages_cache_control_wire.py (100%) rename tests/integration/{providers => messages/responses_bridge}/test_anthropic_messages_claude_code_cache_key_wire.py (100%) rename tests/integration/{providers => messages/responses_bridge}/test_anthropic_messages_openai_bridge_wire.py (100%) rename tests/integration/{providers => messages/responses_bridge}/test_anthropic_messages_openai_tools_wire.py (100%) rename tests/integration/{providers => messages/responses_bridge}/test_responses_bridge_stream_options.py (100%) diff --git a/tests/integration/README.md b/tests/integration/README.md index ac9b01786b9..1601123038b 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -28,6 +28,8 @@ Provider contracts exercise actual TCP requests with synthetic credentials and l Streaming checks send real HTTP transfer chunks, including one-byte partitions, fragmented tools, incomplete transfers and a cancellation barrier. They assert meaningful text, tool arguments, final usage and persisted cost. The Redis recovery case owns a separate database and Redis process, uses the supported one-second circuit-breaker recovery setting, waits for the real subscriber and verifies response data in Redis after restart. CircleCI reuses its existing Redis image for that extra process; it never pulls an image during tests +The `messages/` directory holds `/v1/messages` endpoint contracts grouped by backend subfolder (`anthropic`, `bedrock`, `gemini`, `responses_bridge`, `chat_bridge`) and runs in the providers shard; `run.py` selects test files recursively under each scheduled directory + The sdk shard exercises the SDK's own HTTP clients against local protocol peers with no gateway in the path, so a case here fails only when the client library or its wire behavior changes. The HTTP/2 case runs a hypercorn TLS peer offering h2 and http/1.1 over ALPN, drives the sync and async httpx handlers at it with `LITELLM_HTTP2` off and on, and asserts the version both the client and the peer observed on the wire. Put a test here only when it needs no proxy, database or Redis; a case that reaches the gateway belongs in one of the other shards The extensions shard uses the built-in generic callback and guardrail transports. It checks callback correlation and credential exclusion, guardrail rewriting and denial, retained OpenAI consumers and A2A wire versions diff --git a/tests/integration/_support/manifest.py b/tests/integration/_support/manifest.py index aa0b27eceda..0a0d1ca52fd 100644 --- a/tests/integration/_support/manifest.py +++ b/tests/integration/_support/manifest.py @@ -10,6 +10,7 @@ OWNED_DIRECTORIES: Final = frozenset( "routing", "providers", "streaming", + "messages", "configuration", "mcp", "observability", diff --git a/tests/integration/providers/test_anthropic_advisor_wire.py b/tests/integration/messages/anthropic/test_anthropic_advisor_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_advisor_wire.py rename to tests/integration/messages/anthropic/test_anthropic_advisor_wire.py diff --git a/tests/integration/providers/test_anthropic_legacy_thinking_budget_wire.py b/tests/integration/messages/anthropic/test_anthropic_legacy_thinking_budget_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_legacy_thinking_budget_wire.py rename to tests/integration/messages/anthropic/test_anthropic_legacy_thinking_budget_wire.py diff --git a/tests/integration/providers/test_anthropic_messages_timeout_wire.py b/tests/integration/messages/anthropic/test_anthropic_messages_timeout_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_messages_timeout_wire.py rename to tests/integration/messages/anthropic/test_anthropic_messages_timeout_wire.py diff --git a/tests/integration/providers/test_anthropic_thinking_signature_retry_wire.py b/tests/integration/messages/anthropic/test_anthropic_thinking_signature_retry_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_thinking_signature_retry_wire.py rename to tests/integration/messages/anthropic/test_anthropic_thinking_signature_retry_wire.py diff --git a/tests/integration/providers/test_anthropic_wire.py b/tests/integration/messages/anthropic/test_anthropic_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_wire.py rename to tests/integration/messages/anthropic/test_anthropic_wire.py diff --git a/tests/integration/providers/test_websearch_interception_wire.py b/tests/integration/messages/anthropic/test_websearch_interception_wire.py similarity index 100% rename from tests/integration/providers/test_websearch_interception_wire.py rename to tests/integration/messages/anthropic/test_websearch_interception_wire.py diff --git a/tests/integration/providers/test_bedrock_invoke_tool_search_wire.py b/tests/integration/messages/bedrock/test_bedrock_invoke_tool_search_wire.py similarity index 100% rename from tests/integration/providers/test_bedrock_invoke_tool_search_wire.py rename to tests/integration/messages/bedrock/test_bedrock_invoke_tool_search_wire.py diff --git a/tests/integration/providers/test_bedrock_messages_web_search_replay_wire.py b/tests/integration/messages/bedrock/test_bedrock_messages_web_search_replay_wire.py similarity index 100% rename from tests/integration/providers/test_bedrock_messages_web_search_replay_wire.py rename to tests/integration/messages/bedrock/test_bedrock_messages_web_search_replay_wire.py diff --git a/tests/integration/providers/test_anthropic_messages_fireworks_stop_wire.py b/tests/integration/messages/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_messages_fireworks_stop_wire.py rename to tests/integration/messages/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py diff --git a/tests/integration/providers/test_gemini_messages_cache_control_wire.py b/tests/integration/messages/gemini/test_gemini_messages_cache_control_wire.py similarity index 100% rename from tests/integration/providers/test_gemini_messages_cache_control_wire.py rename to tests/integration/messages/gemini/test_gemini_messages_cache_control_wire.py diff --git a/tests/integration/providers/test_anthropic_messages_claude_code_cache_key_wire.py b/tests/integration/messages/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_messages_claude_code_cache_key_wire.py rename to tests/integration/messages/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py diff --git a/tests/integration/providers/test_anthropic_messages_openai_bridge_wire.py b/tests/integration/messages/responses_bridge/test_anthropic_messages_openai_bridge_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_messages_openai_bridge_wire.py rename to tests/integration/messages/responses_bridge/test_anthropic_messages_openai_bridge_wire.py diff --git a/tests/integration/providers/test_anthropic_messages_openai_tools_wire.py b/tests/integration/messages/responses_bridge/test_anthropic_messages_openai_tools_wire.py similarity index 100% rename from tests/integration/providers/test_anthropic_messages_openai_tools_wire.py rename to tests/integration/messages/responses_bridge/test_anthropic_messages_openai_tools_wire.py diff --git a/tests/integration/providers/test_responses_bridge_stream_options.py b/tests/integration/messages/responses_bridge/test_responses_bridge_stream_options.py similarity index 100% rename from tests/integration/providers/test_responses_bridge_stream_options.py rename to tests/integration/messages/responses_bridge/test_responses_bridge_stream_options.py diff --git a/tests/integration/run.py b/tests/integration/run.py index 9ef585def3d..65b5524506f 100644 --- a/tests/integration/run.py +++ b/tests/integration/run.py @@ -14,7 +14,7 @@ GROUPS: Final = MappingProxyType( "management": ("management", "authorization", "configuration"), "accounting": ("pricing", "spend"), "database": ("database",), - "providers": ("providers", "routing", "streaming"), + "providers": ("providers", "routing", "streaming", "messages"), "extensions": ("observability", "compatibility"), "mcp": ("mcp",), "sdk": ("sdk",), @@ -35,7 +35,7 @@ def main() -> int: selected: Final = tuple( str(path.relative_to(root)) for folder in GROUPS[options.group] - for path in sorted((root / "tests/integration" / folder).glob("test_*.py")) + for path in sorted((root / "tests/integration" / folder).rglob("test_*.py")) ) if not selected: parser.error(f"No integration test files selected for {options.group}") From 2122f45a98c74b20c877fb422ccef267ddceb696 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:01:51 +0000 Subject: [PATCH 02/19] test(integration): make ci coverage census collect nested test dirs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .github/scripts/assert_ci_coverage.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/scripts/assert_ci_coverage.py b/.github/scripts/assert_ci_coverage.py index 01a01b1034b..a483dcec9d7 100644 --- a/.github/scripts/assert_ci_coverage.py +++ b/.github/scripts/assert_ci_coverage.py @@ -516,7 +516,7 @@ def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozens str(path.relative_to(repo_root)) for folders in groups.values() for folder in folders - for path in (integration_root / folder).glob("test_*.py") + for path in (integration_root / folder).rglob("test_*.py") ) browser_manifest: Final = repo_root / "tests/e2e/ui/tests/integrationCritical/expected.json" browser_nodes: Final = json.loads(browser_manifest.read_text()) if browser_manifest.exists() else () From 62e786c7161426dfc72991d298311fa4f8149310 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:03:33 +0000 Subject: [PATCH 03/19] test(integration): nest /v1/messages contracts under messages_endpoint/providers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/README.md | 2 +- tests/integration/_support/manifest.py | 2 +- .../chat_bridge/test_anthropic_messages_fireworks_stop_wire.py | 0 .../providers}/anthropic/test_anthropic_advisor_wire.py | 0 .../anthropic/test_anthropic_legacy_thinking_budget_wire.py | 0 .../anthropic/test_anthropic_messages_timeout_wire.py | 0 .../anthropic/test_anthropic_thinking_signature_retry_wire.py | 0 .../providers}/anthropic/test_anthropic_wire.py | 0 .../providers}/anthropic/test_websearch_interception_wire.py | 0 .../providers}/bedrock/test_bedrock_invoke_tool_search_wire.py | 0 .../bedrock/test_bedrock_messages_web_search_replay_wire.py | 0 .../gemini/test_gemini_messages_cache_control_wire.py | 0 .../test_anthropic_messages_claude_code_cache_key_wire.py | 0 .../test_anthropic_messages_openai_bridge_wire.py | 0 .../test_anthropic_messages_openai_tools_wire.py | 0 .../responses_bridge/test_responses_bridge_stream_options.py | 0 tests/integration/run.py | 2 +- 17 files changed, 3 insertions(+), 3 deletions(-) rename tests/integration/{messages => messages_endpoint}/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_anthropic_advisor_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_anthropic_legacy_thinking_budget_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_anthropic_messages_timeout_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_anthropic_thinking_signature_retry_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_anthropic_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/anthropic/test_websearch_interception_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/bedrock/test_bedrock_invoke_tool_search_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/bedrock/test_bedrock_messages_web_search_replay_wire.py (100%) rename tests/integration/{messages => messages_endpoint/providers}/gemini/test_gemini_messages_cache_control_wire.py (100%) rename tests/integration/{messages => messages_endpoint}/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py (100%) rename tests/integration/{messages => messages_endpoint}/responses_bridge/test_anthropic_messages_openai_bridge_wire.py (100%) rename tests/integration/{messages => messages_endpoint}/responses_bridge/test_anthropic_messages_openai_tools_wire.py (100%) rename tests/integration/{messages => messages_endpoint}/responses_bridge/test_responses_bridge_stream_options.py (100%) diff --git a/tests/integration/README.md b/tests/integration/README.md index 1601123038b..1865366123a 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -28,7 +28,7 @@ Provider contracts exercise actual TCP requests with synthetic credentials and l Streaming checks send real HTTP transfer chunks, including one-byte partitions, fragmented tools, incomplete transfers and a cancellation barrier. They assert meaningful text, tool arguments, final usage and persisted cost. The Redis recovery case owns a separate database and Redis process, uses the supported one-second circuit-breaker recovery setting, waits for the real subscriber and verifies response data in Redis after restart. CircleCI reuses its existing Redis image for that extra process; it never pulls an image during tests -The `messages/` directory holds `/v1/messages` endpoint contracts grouped by backend subfolder (`anthropic`, `bedrock`, `gemini`, `responses_bridge`, `chat_bridge`) and runs in the providers shard; `run.py` selects test files recursively under each scheduled directory +The `messages_endpoint/` directory holds `/v1/messages` endpoint contracts: native-provider backends under `providers/` (`anthropic`, `bedrock`, `gemini`) and the translation bridges (`responses_bridge`, `chat_bridge`) at the top level. It runs in the providers shard; `run.py` selects test files recursively under each scheduled directory The sdk shard exercises the SDK's own HTTP clients against local protocol peers with no gateway in the path, so a case here fails only when the client library or its wire behavior changes. The HTTP/2 case runs a hypercorn TLS peer offering h2 and http/1.1 over ALPN, drives the sync and async httpx handlers at it with `LITELLM_HTTP2` off and on, and asserts the version both the client and the peer observed on the wire. Put a test here only when it needs no proxy, database or Redis; a case that reaches the gateway belongs in one of the other shards diff --git a/tests/integration/_support/manifest.py b/tests/integration/_support/manifest.py index 0a0d1ca52fd..376c2a515b7 100644 --- a/tests/integration/_support/manifest.py +++ b/tests/integration/_support/manifest.py @@ -10,7 +10,7 @@ OWNED_DIRECTORIES: Final = frozenset( "routing", "providers", "streaming", - "messages", + "messages_endpoint", "configuration", "mcp", "observability", diff --git a/tests/integration/messages/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py b/tests/integration/messages_endpoint/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py similarity index 100% rename from tests/integration/messages/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py rename to tests/integration/messages_endpoint/chat_bridge/test_anthropic_messages_fireworks_stop_wire.py diff --git a/tests/integration/messages/anthropic/test_anthropic_advisor_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_anthropic_advisor_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py diff --git a/tests/integration/messages/anthropic/test_anthropic_legacy_thinking_budget_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_anthropic_legacy_thinking_budget_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py diff --git a/tests/integration/messages/anthropic/test_anthropic_messages_timeout_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_anthropic_messages_timeout_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py diff --git a/tests/integration/messages/anthropic/test_anthropic_thinking_signature_retry_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_anthropic_thinking_signature_retry_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py diff --git a/tests/integration/messages/anthropic/test_anthropic_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_anthropic_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py diff --git a/tests/integration/messages/anthropic/test_websearch_interception_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py similarity index 100% rename from tests/integration/messages/anthropic/test_websearch_interception_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py diff --git a/tests/integration/messages/bedrock/test_bedrock_invoke_tool_search_wire.py b/tests/integration/messages_endpoint/providers/bedrock/test_bedrock_invoke_tool_search_wire.py similarity index 100% rename from tests/integration/messages/bedrock/test_bedrock_invoke_tool_search_wire.py rename to tests/integration/messages_endpoint/providers/bedrock/test_bedrock_invoke_tool_search_wire.py diff --git a/tests/integration/messages/bedrock/test_bedrock_messages_web_search_replay_wire.py b/tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_web_search_replay_wire.py similarity index 100% rename from tests/integration/messages/bedrock/test_bedrock_messages_web_search_replay_wire.py rename to tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_web_search_replay_wire.py diff --git a/tests/integration/messages/gemini/test_gemini_messages_cache_control_wire.py b/tests/integration/messages_endpoint/providers/gemini/test_gemini_messages_cache_control_wire.py similarity index 100% rename from tests/integration/messages/gemini/test_gemini_messages_cache_control_wire.py rename to tests/integration/messages_endpoint/providers/gemini/test_gemini_messages_cache_control_wire.py diff --git a/tests/integration/messages/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py similarity index 100% rename from tests/integration/messages/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py rename to tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_claude_code_cache_key_wire.py diff --git a/tests/integration/messages/responses_bridge/test_anthropic_messages_openai_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_openai_bridge_wire.py similarity index 100% rename from tests/integration/messages/responses_bridge/test_anthropic_messages_openai_bridge_wire.py rename to tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_openai_bridge_wire.py diff --git a/tests/integration/messages/responses_bridge/test_anthropic_messages_openai_tools_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_openai_tools_wire.py similarity index 100% rename from tests/integration/messages/responses_bridge/test_anthropic_messages_openai_tools_wire.py rename to tests/integration/messages_endpoint/responses_bridge/test_anthropic_messages_openai_tools_wire.py diff --git a/tests/integration/messages/responses_bridge/test_responses_bridge_stream_options.py b/tests/integration/messages_endpoint/responses_bridge/test_responses_bridge_stream_options.py similarity index 100% rename from tests/integration/messages/responses_bridge/test_responses_bridge_stream_options.py rename to tests/integration/messages_endpoint/responses_bridge/test_responses_bridge_stream_options.py diff --git a/tests/integration/run.py b/tests/integration/run.py index 65b5524506f..b416e4869dd 100644 --- a/tests/integration/run.py +++ b/tests/integration/run.py @@ -14,7 +14,7 @@ GROUPS: Final = MappingProxyType( "management": ("management", "authorization", "configuration"), "accounting": ("pricing", "spend"), "database": ("database",), - "providers": ("providers", "routing", "streaming", "messages"), + "providers": ("providers", "routing", "streaming", "messages_endpoint"), "extensions": ("observability", "compatibility"), "mcp": ("mcp",), "sdk": ("sdk",), From bac71195725553e775222d0ef52ac7fbad9dfb62 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:34:19 +0000 Subject: [PATCH 04/19] test(anthropic): replay a real Claude Code /v1/messages request through the native wire Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../anthropic/claude_code_request.json | 859 ++++++++++++++++++ .../anthropic/test_claude_code_native_wire.py | 151 +++ 2 files changed, 1010 insertions(+) create mode 100644 tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json b/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json new file mode 100644 index 00000000000..b7b03c9ed72 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json @@ -0,0 +1,859 @@ +{ + "model": "claude-sonnet-4-5", + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "\nSynthetic environment reminder.\n" + }, + { + "type": "text", + "text": "\nSynthetic model identity reminder.\n" + }, + { + "type": "text", + "text": "\nSynthetic agent types reminder.\n" + }, + { + "type": "text", + "text": "\nSynthetic skills reminder.\n" + }, + { + "type": "text", + "text": "\n15000000 tokens left\n" + }, + { + "type": "text", + "text": "\nSynthetic date reminder.\n" + }, + { + "type": "text", + "text": "\nSynthetic attribution reminder.\n" + }, + { + "type": "text", + "text": "Reply with exactly the word PONG", + "cache_control": { + "type": "ephemeral" + } + } + ] + } + ], + "system": [ + { + "type": "text", + "text": "Synthetic billing header block from a Claude Code request." + }, + { + "type": "text", + "text": "Synthetic agent identity system prompt.", + "cache_control": { + "type": "ephemeral" + } + }, + { + "type": "text", + "text": "Synthetic interactive agent instructions.", + "cache_control": { + "type": "ephemeral" + } + } + ], + "tools": [ + { + "name": "Agent", + "description": "Launch a new agent to handle complex, multi-step tasks.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "description": { + "description": "A short (3-5 word) description of the task", + "type": "string" + }, + "prompt": { + "description": "The task for the agent to perform", + "type": "string" + }, + "subagent_type": { + "description": "The type of specialized agent to use for this task", + "type": "string" + }, + "model": { + "description": "Optional model override for this agent.", + "type": "string", + "enum": [ + "sonnet", + "opus", + "haiku", + "fable" + ] + }, + "run_in_background": { + "description": "Agents run in the background by default; you will be notified when one completes.", + "type": "boolean" + }, + "isolation": { + "description": "Isolation mode.", + "type": "string", + "enum": [ + "worktree", + "remote" + ] + } + }, + "required": [ + "description", + "prompt" + ], + "additionalProperties": false + } + }, + { + "name": "Bash", + "description": "Executes a given bash command and returns its output.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "command": { + "description": "The command to execute", + "type": "string" + }, + "timeout": { + "description": "Optional timeout in milliseconds (max 600000)", + "type": "number" + }, + "description": { + "description": "Clear, concise description of what this command does in active voice.", + "type": "string" + }, + "run_in_background": { + "description": "Set to true to run this command in the background.", + "type": "boolean" + }, + "dangerouslyDisableSandbox": { + "description": "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", + "type": "boolean" + } + }, + "required": [ + "command" + ], + "additionalProperties": false + } + }, + { + "name": "CronCreate", + "description": "Schedule a prompt to be enqueued at a future time.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "cron": { + "description": "Standard 5-field cron expression in local time: \"M H DoM Mon DoW\" (e.g.", + "type": "string" + }, + "prompt": { + "description": "The prompt to enqueue at each fire time.", + "type": "string" + }, + "recurring": { + "description": "true (default) = fire on every cron match until deleted or auto-expired after 7 days.", + "type": "boolean" + }, + "durable": { + "description": "true = persist to .claude/scheduled_tasks.json and survive restarts.", + "type": "boolean" + } + }, + "required": [ + "cron", + "prompt" + ], + "additionalProperties": false + } + }, + { + "name": "CronDelete", + "description": "Cancel a cron job previously scheduled with CronCreate.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "id": { + "description": "Job ID returned by CronCreate.", + "type": "string" + } + }, + "required": [ + "id" + ], + "additionalProperties": false + } + }, + { + "name": "CronList", + "description": "List all cron jobs scheduled via CronCreate, both durable (.claude/scheduled_tasks.json) and session-only.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": false + } + }, + { + "name": "Edit", + "description": "Performs exact string replacements in files.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "file_path": { + "description": "The absolute path to the file to modify", + "type": "string" + }, + "old_string": { + "description": "The text to replace", + "type": "string" + }, + "new_string": { + "description": "The text to replace it with (must be different from old_string)", + "type": "string" + }, + "replace_all": { + "description": "Replace all occurrences of old_string (default false)", + "default": false, + "type": "boolean" + } + }, + "required": [ + "file_path", + "old_string", + "new_string" + ], + "additionalProperties": false + } + }, + { + "name": "EnterWorktree", + "description": "Use this tool ONLY when explicitly instructed to work in a worktree \u2014 either by the user directly, or by project instruc", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "name": { + "description": "Optional name for a new worktree.", + "type": "string" + }, + "path": { + "description": "Path to an existing worktree to switch into instead of creating a new one.", + "type": "string" + } + }, + "additionalProperties": false + } + }, + { + "name": "ExitWorktree", + "description": "Exit a worktree session created by EnterWorktree and return the session to the original working directory.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "action": { + "description": "\"keep\" leaves the worktree and branch on disk; \"remove\" deletes both.", + "type": "string", + "enum": [ + "keep", + "remove" + ] + }, + "discard_changes": { + "description": "Required true when action is \"remove\" and the worktree has uncommitted files or unmerged commits.", + "type": "boolean" + } + }, + "required": [ + "action" + ], + "additionalProperties": false + } + }, + { + "name": "ListAgents", + "description": "Lists agents you can SendMessage to \u2014 in-process subagents you spawned, the teammates on your team, other local Claude s", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "channel": { + "description": "Not available in this build; leave unset.", + "type": "string", + "maxLength": 256 + }, + "q": { + "description": "Not available in this build; leave unset.", + "type": "string", + "maxLength": 256 + } + }, + "additionalProperties": false + } + }, + { + "name": "NotebookEdit", + "description": "Replaces, inserts, or deletes a single cell in a Jupyter notebook (.ipynb file).", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "notebook_path": { + "description": "The absolute path to the Jupyter notebook file to edit (must be absolute, not relative)", + "type": "string" + }, + "cell_id": { + "description": "The ID of the cell to edit.", + "type": "string" + }, + "new_source": { + "description": "The new source for the cell", + "type": "string" + }, + "cell_type": { + "description": "The type of the cell (code or markdown).", + "type": "string", + "enum": [ + "code", + "markdown" + ] + }, + "edit_mode": { + "description": "The type of edit to make (replace, insert, delete).", + "type": "string", + "enum": [ + "replace", + "insert", + "delete" + ] + } + }, + "required": [ + "notebook_path", + "new_source" + ], + "additionalProperties": false + } + }, + { + "name": "Read", + "description": "Reads a file from the local filesystem.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "file_path": { + "description": "The absolute path to the file to read", + "type": "string" + }, + "offset": { + "description": "The line number to start reading from.", + "type": "integer", + "minimum": 0, + "maximum": 9007199254740991 + }, + "limit": { + "description": "The number of lines to read.", + "type": "integer", + "exclusiveMinimum": 0, + "maximum": 9007199254740991 + }, + "pages": { + "description": "Page range for PDF files (e.g., \"1-5\", \"3\", \"10-20\").", + "type": "string" + } + }, + "required": [ + "file_path" + ], + "additionalProperties": false + } + }, + { + "name": "ReportFindings", + "description": "Report code-review findings as a typed list so the host UI can render them.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "level": { + "description": "Effort level the review ran at", + "type": "string", + "enum": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + }, + "findings": { + "description": "Verified findings, most-severe first; empty if none survived", + "maxItems": 32, + "type": "array", + "items": { + "type": "object", + "properties": { + "file": { + "description": "Repo-relative path of the file the finding is in", + "type": "string" + }, + "line": { + "description": "1-indexed line the finding anchors to", + "type": "integer", + "minimum": -9007199254740991, + "maximum": 9007199254740991 + }, + "summary": { + "description": "One-sentence statement of the defect", + "type": "string" + }, + "short_summary": { + "description": "Compressed label for compact UI (\u226460 chars): the claim alone, no rationale or consequence clause", + "type": "string", + "maxLength": 60 + }, + "failure_scenario": { + "description": "Concrete inputs/state \u2192 wrong output/crash", + "type": "string" + }, + "category": { + "description": "Short kebab-case slug of the finding type, e.g.", + "type": "string", + "maxLength": 40 + }, + "verdict": { + "description": "Set when a verify pass ran; absent on inline-only reviews", + "type": "string", + "enum": [ + "CONFIRMED", + "PLAUSIBLE" + ] + }, + "outcome": { + "description": "Set ONLY when re-reporting after applying fixes: what happened to this finding", + "type": "string", + "enum": [ + "fixed", + "skipped", + "no_change_needed" + ] + } + }, + "required": [ + "file", + "summary", + "failure_scenario" + ], + "additionalProperties": false + } + } + }, + "required": [ + "findings" + ], + "additionalProperties": false + } + }, + { + "name": "ScheduleWakeup", + "description": "Schedule when to resume work in /loop dynamic mode \u2014 the user invoked /loop without an interval, asking you to self-pace", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "delaySeconds": { + "description": "Seconds from now to wake up.", + "type": "number" + }, + "reason": { + "description": "One short sentence explaining the chosen delay.", + "type": "string" + }, + "prompt": { + "description": "The /loop input to fire on wake-up.", + "type": "string" + }, + "stop": { + "description": "Set to true to end the dynamic loop immediately instead of scheduling another wakeup.", + "type": "boolean" + }, + "noop": { + "description": "true = nothing changed (you checked and there is nothing to report).", + "type": "boolean" + } + }, + "additionalProperties": false + } + }, + { + "name": "SendMessage", + "description": "# SendMessage\n\nSend a message to another agent.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "to": { + "description": "Recipient: a name from ListAgents (append its \" [ref]\" only when a listing or an error shows one), a teammate name, \"mai", + "type": "string", + "allOf": [ + { + "pattern": "^[^\\n\\r]*$" + }, + { + "pattern": "^[\\s\\S]{0,300}$" + } + ] + }, + "summary": { + "description": "A 5-10 word label for your own transcript row (not transmitted \u2014 the recipient previews the first line of `message`).", + "type": "string", + "maxLength": 200 + }, + "message": { + "default": "", + "description": "Plain text message content.", + "type": "string" + }, + "notify_when_idle": { + "description": "Ask a session ON THIS MACHINE to send you ONE notice when it next goes idle (finishes its turn with nothing queued) or e", + "type": "boolean" + } + }, + "required": [ + "to", + "message" + ], + "additionalProperties": false + } + }, + { + "name": "Skill", + "description": "Invoke a skill.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "skill": { + "description": "The name of a skill from the available-skills list.", + "type": "string" + }, + "args": { + "description": "Optional arguments for the skill", + "type": "string" + } + }, + "required": [ + "skill" + ], + "additionalProperties": false + } + }, + { + "name": "TaskCreate", + "description": "Use this tool to create a structured task list for your current coding session.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "subject": { + "description": "A brief title for the task", + "type": "string" + }, + "description": { + "description": "What needs to be done", + "type": "string" + }, + "activeForm": { + "description": "Present continuous form shown in spinner when in_progress (e.g., \"Running tests\")", + "type": "string" + }, + "metadata": { + "description": "Arbitrary metadata to attach to the task", + "type": "object", + "propertyNames": { + "type": "string" + }, + "additionalProperties": {} + } + }, + "required": [ + "subject", + "description" + ], + "additionalProperties": false + } + }, + { + "name": "TaskGet", + "description": "Use this tool to retrieve a task by its ID from the task list.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "taskId": { + "description": "The ID of the task to retrieve", + "type": "string" + } + }, + "required": [ + "taskId" + ], + "additionalProperties": false + } + }, + { + "name": "TaskList", + "description": "Use this tool to list all tasks in the task list.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": false + } + }, + { + "name": "TaskStop", + "description": "- Stops a running background task by its ID\n- Takes a task_id parameter identifying the task to stop\n- To stop an agent-", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "task_id": { + "description": "The ID of the background task to stop.", + "type": "string" + }, + "shell_id": { + "description": "Deprecated: use task_id instead", + "type": "string" + } + }, + "additionalProperties": false + } + }, + { + "name": "TaskUpdate", + "description": "Use this tool to update a task in the task list.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "taskId": { + "description": "The ID of the task to update", + "type": "string" + }, + "subject": { + "description": "New subject for the task", + "type": "string" + }, + "description": { + "description": "New description for the task", + "type": "string" + }, + "activeForm": { + "description": "Present continuous form shown in spinner when in_progress (e.g., \"Running tests\")", + "type": "string" + }, + "status": { + "description": "New status for the task", + "anyOf": [ + { + "type": "string", + "enum": [ + "pending", + "in_progress", + "completed" + ] + }, + { + "type": "string", + "const": "deleted" + } + ] + }, + "addBlocks": { + "description": "Task IDs that this task blocks", + "type": "array", + "items": { + "type": "string" + } + }, + "addBlockedBy": { + "description": "Task IDs that block this task", + "type": "array", + "items": { + "type": "string" + } + }, + "owner": { + "description": "New owner for the task", + "type": "string" + }, + "metadata": { + "description": "Metadata keys to merge into the task.", + "type": "object", + "propertyNames": { + "type": "string" + }, + "additionalProperties": {} + } + }, + "required": [ + "taskId" + ], + "additionalProperties": false + } + }, + { + "name": "WebFetch", + "description": "IMPORTANT: WebFetch WILL FAIL for authenticated or private URLs.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "url": { + "description": "The URL to fetch content from", + "type": "string", + "format": "uri" + }, + "prompt": { + "description": "The prompt to run on the fetched content", + "type": "string" + } + }, + "required": [ + "url", + "prompt" + ], + "additionalProperties": false + } + }, + { + "name": "WebSearch", + "description": "- Allows Claude to search the web and use the results to inform responses\n- Provides up-to-date information for current ", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "query": { + "description": "The search query to use", + "type": "string", + "minLength": 2 + }, + "allowed_domains": { + "description": "Only include search results from these domains", + "type": "array", + "items": { + "type": "string" + } + }, + "blocked_domains": { + "description": "Never include search results from these domains", + "type": "array", + "items": { + "type": "string" + } + } + }, + "required": [ + "query" + ], + "additionalProperties": false + } + }, + { + "name": "Workflow", + "description": "Execute a workflow script that orchestrates multiple subagents deterministically.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "script": { + "description": "Self-contained workflow script.", + "type": "string", + "maxLength": 524288 + }, + "name": { + "description": "Name of a predefined workflow (built-in or from .claude/workflows/).", + "type": "string" + }, + "description": { + "description": "Ignored \u2014 set the workflow description in the script's `meta` block.", + "type": "string" + }, + "title": { + "description": "Ignored \u2014 set the workflow title in the script's `meta` block.", + "type": "string" + }, + "args": { + "description": "Optional input value exposed to the script as the global `args`, verbatim." + }, + "scriptPath": { + "description": "Path to a workflow script file on disk.", + "type": "string" + }, + "resumeFromRunId": { + "description": "Run ID of a prior Workflow invocation to resume from.", + "type": "string", + "pattern": "^wf_[a-z0-9-]{6,}$" + } + }, + "additionalProperties": false + } + }, + { + "name": "Write", + "description": "Writes a file to the local filesystem.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "file_path": { + "description": "The absolute path to the file to write (must be absolute, not relative)", + "type": "string" + }, + "content": { + "description": "The content to write to the file", + "type": "string" + } + }, + "required": [ + "file_path", + "content" + ], + "additionalProperties": false + } + } + ], + "metadata": { + "user_id": "{\"device_id\": \"0000000000000000000000000000000000000000000000000000000000000000\", \"account_uuid\": \"\", \"session_id\": \"00000000-0000-4000-8000-000000000000\"}" + }, + "max_tokens": 32000, + "thinking": { + "budget_tokens": 31999, + "type": "enabled", + "display": "omitted" + }, + "context_management": { + "edits": [ + { + "type": "clear_thinking_20251015", + "keep": "all" + } + ] + }, + "stream": true +} diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py new file mode 100644 index 00000000000..5378c904b90 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -0,0 +1,151 @@ +import json +import uuid +from pathlib import Path +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue, TypeAdapter + +_API_KEY: Final = "synthetic-anthropic-key" +_MODEL: Final = "claude-sonnet-4-5" +_FIXTURE: Final = Path(__file__).with_name("claude_code_request.json") +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_CLI_BETA: Final = ( + "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13," + "context-management-2025-06-27,prompt-caching-scope-2026-01-05" +) + + +def _sse_frame(event: str, data: JsonValue) -> bytes: + return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode() + + +def _message_stream(identity: str) -> tuple[bytes, ...]: + return ( + _sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 1}, + }, + }, + ), + _sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + _sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PONG"}}, + ), + _sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + _sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 4}, + }, + ), + _sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: + frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) + return tuple( + ( + next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), + json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), + ) + for frame in frames + ) + + +@pytest.mark.covers("other.provider_wire.anthropic.claude_code_native_request_survives_and_streams_back") +def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: + identity: Final = f"msg_cc_{uuid.uuid4().hex}" + fixture: Final = _JSON_OBJECT.validate_json(_FIXTURE.read_bytes()) + cli_beta: Final = frozenset(_CLI_BETA.split(",")) + content: Final = list(object_value(fixture["messages"][0])["content"]) + content[0] = {**content[0], "text": f"cache-bust-{uuid.uuid4().hex}"} + request_body: Final = {**fixture, "messages": [{"role": "user", "content": content}]} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == _API_KEY + assert request.headers["anthropic-version"] == "2023-06-01" + upstream_beta: Final = frozenset(request.headers.get("anthropic-beta", "").split(",")) + assert cli_beta <= upstream_beta, request.headers.get("anthropic-beta") + body: Final = _JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": _MODEL} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply(content_type="text/event-stream", chunks=_message_stream(identity)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers={ + "accept": "application/json", + "content-type": "application/json", + "user-agent": "claude-cli/2.1.283 (external, sdk-cli)", + "x-claude-code-session-id": "00000000-0000-4000-8000-000000000000", + "x-stainless-arch": "x64", + "x-stainless-lang": "js", + "x-stainless-os": "Linux", + "x-stainless-package-version": "0.112.1", + "x-stainless-retry-count": "0", + "x-stainless-runtime": "node", + "x-stainless-runtime-version": "v26.3.0", + "x-stainless-timeout": "600", + "anthropic-beta": _CLI_BETA, + "anthropic-dangerous-direct-browser-access": "true", + "anthropic-version": "2023-06-01", + "x-app": "cli", + "x-api-key": gateway.key, + }, + ) + assert response.status_code == 200, response.text + assert response.headers["content-type"].startswith("text/event-stream"), dict(response.headers) + events: Final = sse_events(response.text) + assert [event for event, _ in events] == [ + "message_start", + "content_block_start", + "content_block_delta", + "content_block_stop", + "message_delta", + "message_stop", + ] + assert events[2][1]["delta"] == {"type": "text_delta", "text": "PONG"} + assert events[4][1]["delta"]["stop_reason"] == "end_turn" + assert events[4][1]["usage"]["output_tokens"] == 4 + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 From edfb89eea72d7c8eadbdf25f0d90577ba0f8da99 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:36:29 +0000 Subject: [PATCH 05/19] test(anthropic): drop legacy covers marker from claude code wire test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../providers/anthropic/test_claude_code_native_wire.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py index 5378c904b90..f4a9bccb104 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -3,7 +3,6 @@ import uuid from pathlib import Path from typing import Final -import pytest from integration._support.client import Gateway, eventually, object_value from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server @@ -73,7 +72,6 @@ def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: ) -@pytest.mark.covers("other.provider_wire.anthropic.claude_code_native_request_survives_and_streams_back") def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: identity: Final = f"msg_cc_{uuid.uuid4().hex}" fixture: Final = _JSON_OBJECT.validate_json(_FIXTURE.read_bytes()) From f235d198957050caa10ae1a5622138f0cbe63903 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:42:03 +0000 Subject: [PATCH 06/19] test(anthropic): inline the Claude Code request instead of a json fixture Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../anthropic/claude_code_request.json | 859 ------------------ .../anthropic/test_claude_code_native_wire.py | 150 ++- 2 files changed, 143 insertions(+), 866 deletions(-) delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json diff --git a/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json b/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json deleted file mode 100644 index b7b03c9ed72..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/claude_code_request.json +++ /dev/null @@ -1,859 +0,0 @@ -{ - "model": "claude-sonnet-4-5", - "messages": [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "\nSynthetic environment reminder.\n" - }, - { - "type": "text", - "text": "\nSynthetic model identity reminder.\n" - }, - { - "type": "text", - "text": "\nSynthetic agent types reminder.\n" - }, - { - "type": "text", - "text": "\nSynthetic skills reminder.\n" - }, - { - "type": "text", - "text": "\n15000000 tokens left\n" - }, - { - "type": "text", - "text": "\nSynthetic date reminder.\n" - }, - { - "type": "text", - "text": "\nSynthetic attribution reminder.\n" - }, - { - "type": "text", - "text": "Reply with exactly the word PONG", - "cache_control": { - "type": "ephemeral" - } - } - ] - } - ], - "system": [ - { - "type": "text", - "text": "Synthetic billing header block from a Claude Code request." - }, - { - "type": "text", - "text": "Synthetic agent identity system prompt.", - "cache_control": { - "type": "ephemeral" - } - }, - { - "type": "text", - "text": "Synthetic interactive agent instructions.", - "cache_control": { - "type": "ephemeral" - } - } - ], - "tools": [ - { - "name": "Agent", - "description": "Launch a new agent to handle complex, multi-step tasks.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "description": { - "description": "A short (3-5 word) description of the task", - "type": "string" - }, - "prompt": { - "description": "The task for the agent to perform", - "type": "string" - }, - "subagent_type": { - "description": "The type of specialized agent to use for this task", - "type": "string" - }, - "model": { - "description": "Optional model override for this agent.", - "type": "string", - "enum": [ - "sonnet", - "opus", - "haiku", - "fable" - ] - }, - "run_in_background": { - "description": "Agents run in the background by default; you will be notified when one completes.", - "type": "boolean" - }, - "isolation": { - "description": "Isolation mode.", - "type": "string", - "enum": [ - "worktree", - "remote" - ] - } - }, - "required": [ - "description", - "prompt" - ], - "additionalProperties": false - } - }, - { - "name": "Bash", - "description": "Executes a given bash command and returns its output.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "command": { - "description": "The command to execute", - "type": "string" - }, - "timeout": { - "description": "Optional timeout in milliseconds (max 600000)", - "type": "number" - }, - "description": { - "description": "Clear, concise description of what this command does in active voice.", - "type": "string" - }, - "run_in_background": { - "description": "Set to true to run this command in the background.", - "type": "boolean" - }, - "dangerouslyDisableSandbox": { - "description": "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", - "type": "boolean" - } - }, - "required": [ - "command" - ], - "additionalProperties": false - } - }, - { - "name": "CronCreate", - "description": "Schedule a prompt to be enqueued at a future time.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "cron": { - "description": "Standard 5-field cron expression in local time: \"M H DoM Mon DoW\" (e.g.", - "type": "string" - }, - "prompt": { - "description": "The prompt to enqueue at each fire time.", - "type": "string" - }, - "recurring": { - "description": "true (default) = fire on every cron match until deleted or auto-expired after 7 days.", - "type": "boolean" - }, - "durable": { - "description": "true = persist to .claude/scheduled_tasks.json and survive restarts.", - "type": "boolean" - } - }, - "required": [ - "cron", - "prompt" - ], - "additionalProperties": false - } - }, - { - "name": "CronDelete", - "description": "Cancel a cron job previously scheduled with CronCreate.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "id": { - "description": "Job ID returned by CronCreate.", - "type": "string" - } - }, - "required": [ - "id" - ], - "additionalProperties": false - } - }, - { - "name": "CronList", - "description": "List all cron jobs scheduled via CronCreate, both durable (.claude/scheduled_tasks.json) and session-only.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": {}, - "additionalProperties": false - } - }, - { - "name": "Edit", - "description": "Performs exact string replacements in files.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "file_path": { - "description": "The absolute path to the file to modify", - "type": "string" - }, - "old_string": { - "description": "The text to replace", - "type": "string" - }, - "new_string": { - "description": "The text to replace it with (must be different from old_string)", - "type": "string" - }, - "replace_all": { - "description": "Replace all occurrences of old_string (default false)", - "default": false, - "type": "boolean" - } - }, - "required": [ - "file_path", - "old_string", - "new_string" - ], - "additionalProperties": false - } - }, - { - "name": "EnterWorktree", - "description": "Use this tool ONLY when explicitly instructed to work in a worktree \u2014 either by the user directly, or by project instruc", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "name": { - "description": "Optional name for a new worktree.", - "type": "string" - }, - "path": { - "description": "Path to an existing worktree to switch into instead of creating a new one.", - "type": "string" - } - }, - "additionalProperties": false - } - }, - { - "name": "ExitWorktree", - "description": "Exit a worktree session created by EnterWorktree and return the session to the original working directory.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "action": { - "description": "\"keep\" leaves the worktree and branch on disk; \"remove\" deletes both.", - "type": "string", - "enum": [ - "keep", - "remove" - ] - }, - "discard_changes": { - "description": "Required true when action is \"remove\" and the worktree has uncommitted files or unmerged commits.", - "type": "boolean" - } - }, - "required": [ - "action" - ], - "additionalProperties": false - } - }, - { - "name": "ListAgents", - "description": "Lists agents you can SendMessage to \u2014 in-process subagents you spawned, the teammates on your team, other local Claude s", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "channel": { - "description": "Not available in this build; leave unset.", - "type": "string", - "maxLength": 256 - }, - "q": { - "description": "Not available in this build; leave unset.", - "type": "string", - "maxLength": 256 - } - }, - "additionalProperties": false - } - }, - { - "name": "NotebookEdit", - "description": "Replaces, inserts, or deletes a single cell in a Jupyter notebook (.ipynb file).", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "notebook_path": { - "description": "The absolute path to the Jupyter notebook file to edit (must be absolute, not relative)", - "type": "string" - }, - "cell_id": { - "description": "The ID of the cell to edit.", - "type": "string" - }, - "new_source": { - "description": "The new source for the cell", - "type": "string" - }, - "cell_type": { - "description": "The type of the cell (code or markdown).", - "type": "string", - "enum": [ - "code", - "markdown" - ] - }, - "edit_mode": { - "description": "The type of edit to make (replace, insert, delete).", - "type": "string", - "enum": [ - "replace", - "insert", - "delete" - ] - } - }, - "required": [ - "notebook_path", - "new_source" - ], - "additionalProperties": false - } - }, - { - "name": "Read", - "description": "Reads a file from the local filesystem.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "file_path": { - "description": "The absolute path to the file to read", - "type": "string" - }, - "offset": { - "description": "The line number to start reading from.", - "type": "integer", - "minimum": 0, - "maximum": 9007199254740991 - }, - "limit": { - "description": "The number of lines to read.", - "type": "integer", - "exclusiveMinimum": 0, - "maximum": 9007199254740991 - }, - "pages": { - "description": "Page range for PDF files (e.g., \"1-5\", \"3\", \"10-20\").", - "type": "string" - } - }, - "required": [ - "file_path" - ], - "additionalProperties": false - } - }, - { - "name": "ReportFindings", - "description": "Report code-review findings as a typed list so the host UI can render them.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "level": { - "description": "Effort level the review ran at", - "type": "string", - "enum": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - }, - "findings": { - "description": "Verified findings, most-severe first; empty if none survived", - "maxItems": 32, - "type": "array", - "items": { - "type": "object", - "properties": { - "file": { - "description": "Repo-relative path of the file the finding is in", - "type": "string" - }, - "line": { - "description": "1-indexed line the finding anchors to", - "type": "integer", - "minimum": -9007199254740991, - "maximum": 9007199254740991 - }, - "summary": { - "description": "One-sentence statement of the defect", - "type": "string" - }, - "short_summary": { - "description": "Compressed label for compact UI (\u226460 chars): the claim alone, no rationale or consequence clause", - "type": "string", - "maxLength": 60 - }, - "failure_scenario": { - "description": "Concrete inputs/state \u2192 wrong output/crash", - "type": "string" - }, - "category": { - "description": "Short kebab-case slug of the finding type, e.g.", - "type": "string", - "maxLength": 40 - }, - "verdict": { - "description": "Set when a verify pass ran; absent on inline-only reviews", - "type": "string", - "enum": [ - "CONFIRMED", - "PLAUSIBLE" - ] - }, - "outcome": { - "description": "Set ONLY when re-reporting after applying fixes: what happened to this finding", - "type": "string", - "enum": [ - "fixed", - "skipped", - "no_change_needed" - ] - } - }, - "required": [ - "file", - "summary", - "failure_scenario" - ], - "additionalProperties": false - } - } - }, - "required": [ - "findings" - ], - "additionalProperties": false - } - }, - { - "name": "ScheduleWakeup", - "description": "Schedule when to resume work in /loop dynamic mode \u2014 the user invoked /loop without an interval, asking you to self-pace", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "delaySeconds": { - "description": "Seconds from now to wake up.", - "type": "number" - }, - "reason": { - "description": "One short sentence explaining the chosen delay.", - "type": "string" - }, - "prompt": { - "description": "The /loop input to fire on wake-up.", - "type": "string" - }, - "stop": { - "description": "Set to true to end the dynamic loop immediately instead of scheduling another wakeup.", - "type": "boolean" - }, - "noop": { - "description": "true = nothing changed (you checked and there is nothing to report).", - "type": "boolean" - } - }, - "additionalProperties": false - } - }, - { - "name": "SendMessage", - "description": "# SendMessage\n\nSend a message to another agent.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "to": { - "description": "Recipient: a name from ListAgents (append its \" [ref]\" only when a listing or an error shows one), a teammate name, \"mai", - "type": "string", - "allOf": [ - { - "pattern": "^[^\\n\\r]*$" - }, - { - "pattern": "^[\\s\\S]{0,300}$" - } - ] - }, - "summary": { - "description": "A 5-10 word label for your own transcript row (not transmitted \u2014 the recipient previews the first line of `message`).", - "type": "string", - "maxLength": 200 - }, - "message": { - "default": "", - "description": "Plain text message content.", - "type": "string" - }, - "notify_when_idle": { - "description": "Ask a session ON THIS MACHINE to send you ONE notice when it next goes idle (finishes its turn with nothing queued) or e", - "type": "boolean" - } - }, - "required": [ - "to", - "message" - ], - "additionalProperties": false - } - }, - { - "name": "Skill", - "description": "Invoke a skill.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "skill": { - "description": "The name of a skill from the available-skills list.", - "type": "string" - }, - "args": { - "description": "Optional arguments for the skill", - "type": "string" - } - }, - "required": [ - "skill" - ], - "additionalProperties": false - } - }, - { - "name": "TaskCreate", - "description": "Use this tool to create a structured task list for your current coding session.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "subject": { - "description": "A brief title for the task", - "type": "string" - }, - "description": { - "description": "What needs to be done", - "type": "string" - }, - "activeForm": { - "description": "Present continuous form shown in spinner when in_progress (e.g., \"Running tests\")", - "type": "string" - }, - "metadata": { - "description": "Arbitrary metadata to attach to the task", - "type": "object", - "propertyNames": { - "type": "string" - }, - "additionalProperties": {} - } - }, - "required": [ - "subject", - "description" - ], - "additionalProperties": false - } - }, - { - "name": "TaskGet", - "description": "Use this tool to retrieve a task by its ID from the task list.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "taskId": { - "description": "The ID of the task to retrieve", - "type": "string" - } - }, - "required": [ - "taskId" - ], - "additionalProperties": false - } - }, - { - "name": "TaskList", - "description": "Use this tool to list all tasks in the task list.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": {}, - "additionalProperties": false - } - }, - { - "name": "TaskStop", - "description": "- Stops a running background task by its ID\n- Takes a task_id parameter identifying the task to stop\n- To stop an agent-", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "task_id": { - "description": "The ID of the background task to stop.", - "type": "string" - }, - "shell_id": { - "description": "Deprecated: use task_id instead", - "type": "string" - } - }, - "additionalProperties": false - } - }, - { - "name": "TaskUpdate", - "description": "Use this tool to update a task in the task list.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "taskId": { - "description": "The ID of the task to update", - "type": "string" - }, - "subject": { - "description": "New subject for the task", - "type": "string" - }, - "description": { - "description": "New description for the task", - "type": "string" - }, - "activeForm": { - "description": "Present continuous form shown in spinner when in_progress (e.g., \"Running tests\")", - "type": "string" - }, - "status": { - "description": "New status for the task", - "anyOf": [ - { - "type": "string", - "enum": [ - "pending", - "in_progress", - "completed" - ] - }, - { - "type": "string", - "const": "deleted" - } - ] - }, - "addBlocks": { - "description": "Task IDs that this task blocks", - "type": "array", - "items": { - "type": "string" - } - }, - "addBlockedBy": { - "description": "Task IDs that block this task", - "type": "array", - "items": { - "type": "string" - } - }, - "owner": { - "description": "New owner for the task", - "type": "string" - }, - "metadata": { - "description": "Metadata keys to merge into the task.", - "type": "object", - "propertyNames": { - "type": "string" - }, - "additionalProperties": {} - } - }, - "required": [ - "taskId" - ], - "additionalProperties": false - } - }, - { - "name": "WebFetch", - "description": "IMPORTANT: WebFetch WILL FAIL for authenticated or private URLs.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "url": { - "description": "The URL to fetch content from", - "type": "string", - "format": "uri" - }, - "prompt": { - "description": "The prompt to run on the fetched content", - "type": "string" - } - }, - "required": [ - "url", - "prompt" - ], - "additionalProperties": false - } - }, - { - "name": "WebSearch", - "description": "- Allows Claude to search the web and use the results to inform responses\n- Provides up-to-date information for current ", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "query": { - "description": "The search query to use", - "type": "string", - "minLength": 2 - }, - "allowed_domains": { - "description": "Only include search results from these domains", - "type": "array", - "items": { - "type": "string" - } - }, - "blocked_domains": { - "description": "Never include search results from these domains", - "type": "array", - "items": { - "type": "string" - } - } - }, - "required": [ - "query" - ], - "additionalProperties": false - } - }, - { - "name": "Workflow", - "description": "Execute a workflow script that orchestrates multiple subagents deterministically.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "script": { - "description": "Self-contained workflow script.", - "type": "string", - "maxLength": 524288 - }, - "name": { - "description": "Name of a predefined workflow (built-in or from .claude/workflows/).", - "type": "string" - }, - "description": { - "description": "Ignored \u2014 set the workflow description in the script's `meta` block.", - "type": "string" - }, - "title": { - "description": "Ignored \u2014 set the workflow title in the script's `meta` block.", - "type": "string" - }, - "args": { - "description": "Optional input value exposed to the script as the global `args`, verbatim." - }, - "scriptPath": { - "description": "Path to a workflow script file on disk.", - "type": "string" - }, - "resumeFromRunId": { - "description": "Run ID of a prior Workflow invocation to resume from.", - "type": "string", - "pattern": "^wf_[a-z0-9-]{6,}$" - } - }, - "additionalProperties": false - } - }, - { - "name": "Write", - "description": "Writes a file to the local filesystem.", - "input_schema": { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": { - "file_path": { - "description": "The absolute path to the file to write (must be absolute, not relative)", - "type": "string" - }, - "content": { - "description": "The content to write to the file", - "type": "string" - } - }, - "required": [ - "file_path", - "content" - ], - "additionalProperties": false - } - } - ], - "metadata": { - "user_id": "{\"device_id\": \"0000000000000000000000000000000000000000000000000000000000000000\", \"account_uuid\": \"\", \"session_id\": \"00000000-0000-4000-8000-000000000000\"}" - }, - "max_tokens": 32000, - "thinking": { - "budget_tokens": 31999, - "type": "enabled", - "display": "omitted" - }, - "context_management": { - "edits": [ - { - "type": "clear_thinking_20251015", - "keep": "all" - } - ] - }, - "stream": true -} diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py index f4a9bccb104..5384c0bc16e 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -1,21 +1,159 @@ import json import uuid -from pathlib import Path from typing import Final -from integration._support.client import Gateway, eventually, object_value +import pytest +from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server from pydantic import JsonValue, TypeAdapter _API_KEY: Final = "synthetic-anthropic-key" _MODEL: Final = "claude-sonnet-4-5" -_FIXTURE: Final = Path(__file__).with_name("claude_code_request.json") _JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) _CLI_BETA: Final = ( "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13," "context-management-2025-06-27,prompt-caching-scope-2026-01-05" ) +_CACHE: Final = {"type": "ephemeral"} + + +def _schema(properties: JsonValue, required: tuple[str, ...]) -> dict[str, JsonValue]: + return { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": properties, + "required": list(required), + "additionalProperties": False, + } + + +def _field(description: str, **extra: JsonValue) -> dict[str, JsonValue]: + return {"description": description, **extra} + + +def _tools() -> tuple[dict[str, JsonValue], ...]: + _MAX: Final = 9007199254740991 + return ( + { + "name": "Bash", + "description": "Executes a given bash command and returns its output.", + "input_schema": _schema( + { + "command": _field("The command to execute", type="string"), + "timeout": _field("Optional timeout in milliseconds (max 600000)", type="number"), + "description": _field( + "Clear, concise description of what this command does in active voice.", type="string" + ), + "run_in_background": _field("Set to true to run this command in the background.", type="boolean"), + "dangerouslyDisableSandbox": _field( + "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", + type="boolean", + ), + }, + ("command",), + ), + }, + { + "name": "Read", + "description": "Reads a file from the local filesystem.", + "input_schema": _schema( + { + "file_path": _field("The absolute path to the file to read", type="string"), + "offset": _field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), + "limit": _field("The number of lines to read.", type="integer", exclusiveMinimum=0, maximum=_MAX), + "pages": _field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), + }, + ("file_path",), + ), + }, + { + "name": "Edit", + "description": "Performs exact string replacements in files.", + "input_schema": _schema( + { + "file_path": _field("The absolute path to the file to modify", type="string"), + "old_string": _field("The text to replace", type="string"), + "new_string": _field( + "The text to replace it with (must be different from old_string)", type="string" + ), + "replace_all": _field( + "Replace all occurrences of old_string (default false)", default=False, type="boolean" + ), + }, + ("file_path", "old_string", "new_string"), + ), + }, + { + "name": "Agent", + "description": "Launch a new agent to handle complex, multi-step tasks.", + "input_schema": _schema( + { + "description": _field("A short (3-5 word) description of the task", type="string"), + "prompt": _field("The task for the agent to perform", type="string"), + "subagent_type": _field("The type of specialized agent to use for this task", type="string"), + "model": _field( + "Optional model override for this agent.", + type="string", + enum=["sonnet", "opus", "haiku", "fable"], + ), + "run_in_background": _field( + "Agents run in the background by default; you will be notified when one completes.", + type="boolean", + ), + "isolation": _field("Isolation mode.", type="string", enum=["worktree", "remote"]), + }, + ("description", "prompt"), + ), + }, + ) + + +def _claude_code_request(cache_bust: str) -> dict[str, JsonValue]: + reminders: Final = ( + f"\n{cache_bust}\n", + "\nSynthetic model identity reminder.\n", + "\nSynthetic agent types reminder.\n", + "\nSynthetic skills reminder.\n", + "\n15000000 tokens left\n", + "\nSynthetic date reminder.\n", + "\nSynthetic attribution reminder.\n", + ) + return { + "model": "", + "system": [ + {"type": "text", "text": "Synthetic billing header block from a Claude Code request."}, + {"type": "text", "text": "Synthetic agent identity system prompt.", "cache_control": _CACHE}, + {"type": "text", "text": "Synthetic interactive agent instructions.", "cache_control": _CACHE}, + ], + "messages": [ + { + "role": "user", + "content": [ + *[{"type": "text", "text": reminder} for reminder in reminders], + { + "type": "text", + "text": "Reply with exactly the word PONG", + "cache_control": _CACHE, + }, + ], + } + ], + "tools": list(_tools()), + "metadata": { + "user_id": json.dumps( + { + "device_id": "0" * 64, + "account_uuid": "", + "session_id": "00000000-0000-4000-8000-000000000000", + } + ) + }, + "max_tokens": 32000, + "thinking": {"budget_tokens": 31999, "type": "enabled", "display": "omitted"}, + "context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]}, + "stream": True, + } def _sse_frame(event: str, data: JsonValue) -> bytes: @@ -72,13 +210,11 @@ def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: ) +@pytest.mark.covers("other.provider_wire.anthropic.claude_code_native_request_survives_and_streams_back") def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: identity: Final = f"msg_cc_{uuid.uuid4().hex}" - fixture: Final = _JSON_OBJECT.validate_json(_FIXTURE.read_bytes()) + request_body: Final = _claude_code_request(f"cache-bust-{uuid.uuid4().hex}") cli_beta: Final = frozenset(_CLI_BETA.split(",")) - content: Final = list(object_value(fixture["messages"][0])["content"]) - content[0] = {**content[0], "text": f"cache-bust-{uuid.uuid4().hex}"} - request_body: Final = {**fixture, "messages": [{"role": "user", "content": content}]} def respond(request: Request) -> Reply: assert request.method == "POST" From de0fbc39c9083e8772fbbf98f791ceb35869e365 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 22:58:23 +0000 Subject: [PATCH 07/19] test(anthropic): drop the legacy covers marker from the new contract Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../providers/anthropic/test_claude_code_native_wire.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py index 5384c0bc16e..6d7fd461fa6 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -2,7 +2,6 @@ import json import uuid from typing import Final -import pytest from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server @@ -210,7 +209,6 @@ def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: ) -@pytest.mark.covers("other.provider_wire.anthropic.claude_code_native_request_survives_and_streams_back") def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: identity: Final = f"msg_cc_{uuid.uuid4().hex}" request_body: Final = _claude_code_request(f"cache-bust-{uuid.uuid4().hex}") From 6325975f32dd39c2a1ff371820832b3beaa2cc13 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:27:58 -0700 Subject: [PATCH 08/19] test(anthropic): add Claude Code /v1/messages customer-journey matrix (native + responses bridge) (#43386) * test(anthropic): add Claude Code /v1/messages customer-journey matrix (native + responses bridge) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): strengthen bot-flagged assertions in the Claude Code matrix Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(anthropic): type the usage mapping parameter in the shared builders Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../messages_endpoint/_claude_code.py | 511 ++++++++++++++++++ ...test_claude_code_client_disconnect_wire.py | 85 +++ .../test_claude_code_compaction_wire.py | 108 ++++ .../test_claude_code_count_tokens_wire.py | 50 ++ .../test_claude_code_document_input_wire.py | 132 +++++ .../test_claude_code_fallback_wire.py | 92 ++++ .../test_claude_code_frontier_wire.py | 111 ++++ .../test_claude_code_image_input_wire.py | 69 +++ ...t_claude_code_interleaved_thinking_wire.py | 167 ++++++ ...test_claude_code_long_context_beta_wire.py | 53 ++ .../test_claude_code_model_switch_wire.py | 86 +++ .../anthropic/test_claude_code_native_wire.py | 241 +-------- .../test_claude_code_prompt_cache_wire.py | 86 +++ .../test_claude_code_tool_loop_wire.py | 165 ++++++ .../test_claude_code_upstream_errors_wire.py | 136 +++++ .../test_claude_code_web_search_wire.py | 184 +++++++ ...test_claude_code_compaction_bridge_wire.py | 107 ++++ ...st_claude_code_count_tokens_bridge_wire.py | 39 ++ .../test_claude_code_document_bridge_wire.py | 73 +++ .../test_claude_code_errors_bridge_wire.py | 77 +++ .../test_claude_code_frontier_bridge_wire.py | 149 +++++ .../test_claude_code_image_bridge_wire.py | 112 ++++ ...est_claude_code_interleaved_bridge_wire.py | 128 +++++ ...st_claude_code_model_switch_bridge_wire.py | 95 ++++ .../test_claude_code_tool_loop_bridge_wire.py | 223 ++++++++ ...test_claude_code_web_search_bridge_wire.py | 95 ++++ 26 files changed, 3146 insertions(+), 228 deletions(-) create mode 100644 tests/integration/messages_endpoint/_claude_code.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py create mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py diff --git a/tests/integration/messages_endpoint/_claude_code.py b/tests/integration/messages_endpoint/_claude_code.py new file mode 100644 index 00000000000..5688a7e9776 --- /dev/null +++ b/tests/integration/messages_endpoint/_claude_code.py @@ -0,0 +1,511 @@ +"""Shared Claude Code-shaped request builders and upstream stream fixtures for the /v1/messages contracts.""" + +import json +from collections.abc import Mapping +from typing import Final + +from pydantic import JsonValue, TypeAdapter + +JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +ANTHROPIC_API_KEY: Final = "synthetic-anthropic-key" +OPENAI_API_KEY: Final = "synthetic-openai-key" +OPENAI_BACKEND: Final = "gpt-5.4-mini" +SONNET: Final = "claude-sonnet-4-5" +FABLE: Final = "claude-fable-5-1" +OPUS: Final = "claude-opus-5-5" +CLI_BETA: Final = ( + "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13," + "context-management-2025-06-27,prompt-caching-scope-2026-01-05" +) +FRONTIER_CLI_BETA: Final = ( + f"{CLI_BETA},mid-conversation-system-2026-04-07,per-turn-control-2026-07-01," + "mid-conversation-tool-changes-2026-07-01,effort-2025-11-24" +) +CACHE: Final = {"type": "ephemeral"} +CONTEXT_MANAGEMENT: Final = {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]} +THINKING_BUDGET: Final = {"budget_tokens": 31999, "type": "enabled", "display": "omitted"} +THINKING_ADAPTIVE: Final = {"type": "adaptive", "display": "omitted"} +METADATA_USER_ID: Final = json.dumps( + { + "device_id": "0" * 64, + "account_uuid": "", + "session_id": "00000000-0000-4000-8000-000000000000", + } +) + + +def schema(properties: JsonValue, required: tuple[str, ...]) -> dict[str, JsonValue]: + return { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": properties, + "required": list(required), + "additionalProperties": False, + } + + +def field(description: str, **extra: JsonValue) -> dict[str, JsonValue]: + return {"description": description, **extra} + + +def tools() -> tuple[dict[str, JsonValue], ...]: + _MAX: Final = 9007199254740991 + return ( + { + "name": "Bash", + "description": "Executes a given bash command and returns its output.", + "input_schema": schema( + { + "command": field("The command to execute", type="string"), + "timeout": field("Optional timeout in milliseconds (max 600000)", type="number"), + "description": field( + "Clear, concise description of what this command does in active voice.", type="string" + ), + "run_in_background": field("Set to true to run this command in the background.", type="boolean"), + "dangerouslyDisableSandbox": field( + "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", + type="boolean", + ), + }, + ("command",), + ), + }, + { + "name": "Read", + "description": "Reads a file from the local filesystem.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to read", type="string"), + "offset": field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), + "limit": field("The number of lines to read.", type="integer", exclusiveMinimum=0, maximum=_MAX), + "pages": field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), + }, + ("file_path",), + ), + }, + { + "name": "Edit", + "description": "Performs exact string replacements in files.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to modify", type="string"), + "old_string": field("The text to replace", type="string"), + "new_string": field( + "The text to replace it with (must be different from old_string)", type="string" + ), + "replace_all": field( + "Replace all occurrences of old_string (default false)", default=False, type="boolean" + ), + }, + ("file_path", "old_string", "new_string"), + ), + }, + { + "name": "Agent", + "description": "Launch a new agent to handle complex, multi-step tasks.", + "input_schema": schema( + { + "description": field("A short (3-5 word) description of the task", type="string"), + "prompt": field("The task for the agent to perform", type="string"), + "subagent_type": field("The type of specialized agent to use for this task", type="string"), + "model": field( + "Optional model override for this agent.", + type="string", + enum=["sonnet", "opus", "haiku", "fable"], + ), + "run_in_background": field( + "Agents run in the background by default; you will be notified when one completes.", + type="boolean", + ), + "isolation": field("Isolation mode.", type="string", enum=["worktree", "remote"]), + }, + ("description", "prompt"), + ), + }, + ) + + +def system_blocks() -> tuple[dict[str, JsonValue], ...]: + return ( + {"type": "text", "text": "x-anthropic-billing-header: cc_version=2.1.283.00; cc_entrypoint=sdk-cli;"}, + {"type": "text", "text": "Synthetic agent identity system prompt.", "cache_control": CACHE}, + {"type": "text", "text": "Synthetic interactive agent instructions.", "cache_control": CACHE}, + ) + + +def claude_code_request(cache_bust: str) -> dict[str, JsonValue]: + reminders: Final = ( + f"\n{cache_bust}\n", + "\nSynthetic model identity reminder.\n", + "\nSynthetic agent types reminder.\n", + "\nSynthetic skills reminder.\n", + "\n15000000 tokens left\n", + "\nSynthetic date reminder.\n", + "\nSynthetic attribution reminder.\n", + ) + return { + "model": "", + "system": list(system_blocks()), + "messages": [ + { + "role": "user", + "content": [ + *[{"type": "text", "text": reminder} for reminder in reminders], + {"type": "text", "text": "Reply with exactly the word PONG", "cache_control": CACHE}, + ], + } + ], + "tools": list(tools()), + "metadata": {"user_id": METADATA_USER_ID}, + "max_tokens": 32000, + "thinking": dict(THINKING_BUDGET), + "context_management": dict(CONTEXT_MANAGEMENT), + "stream": True, + } + + +def frontier_request( + cache_bust: str, + effort: str, + max_tokens: int, + prompt_text: str = "Reply with exactly the word PONG", + stream: bool = True, +) -> dict[str, JsonValue]: + return { + "model": "", + "system": list(system_blocks()), + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": f"\n{cache_bust}\n"}, + {"type": "text", "text": prompt_text}, + ], + }, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "# Environment\nSynthetic environment block.", + "cache_control": CACHE, + } + ], + }, + ], + "tools": list(tools()), + "metadata": {"user_id": METADATA_USER_ID}, + "max_tokens": max_tokens, + "thinking": dict(THINKING_ADAPTIVE), + "context_management": dict(CONTEXT_MANAGEMENT), + "output_config": {"effort": effort}, + "stream": stream, + } + + +def tool_loop_turn2( + base: dict[str, JsonValue], + assistant_content: tuple[dict[str, JsonValue], ...], + tool_results: tuple[tuple[str, JsonValue], ...], +) -> dict[str, JsonValue]: + return { + **base, + "messages": [ + *base["messages"], + {"role": "assistant", "content": list(assistant_content)}, + { + "role": "user", + "content": [ + {"tool_use_id": tool_use_id, "type": "tool_result", "content": content} + for tool_use_id, content in tool_results + ], + }, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "14999970 tokens left", + "cache_control": CACHE, + }, + { + "type": "text", + "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", + }, + ], + }, + ], + } + + +def cli_headers(key: str, beta: str = CLI_BETA) -> dict[str, str]: + return { + "accept": "application/json", + "content-type": "application/json", + "user-agent": "claude-cli/2.1.283 (external, sdk-cli)", + "x-claude-code-session-id": "00000000-0000-4000-8000-000000000000", + "x-stainless-arch": "x64", + "x-stainless-lang": "js", + "x-stainless-os": "Linux", + "x-stainless-package-version": "0.112.1", + "x-stainless-retry-count": "0", + "x-stainless-runtime": "node", + "x-stainless-runtime-version": "v26.3.0", + "x-stainless-timeout": "600", + "anthropic-beta": beta, + "anthropic-dangerous-direct-browser-access": "true", + "anthropic-version": "2023-06-01", + "x-app": "cli", + "x-api-key": key, + } + + +def sse_frame(event: str, data: JsonValue) -> bytes: + return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode() + + +def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: + frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) + return tuple( + ( + next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), + json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), + ) + for frame in frames + ) + + +def _start_usage(usage: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return {key: value for key, value in usage.items() if key != "output_tokens"} + + +def text_stream(identity: str, model: str, text: str, usage: dict[str, int]) -> tuple[bytes, ...]: + return ( + sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": _start_usage(usage), + }, + }, + ), + sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}}, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def tool_use_stream( + identity: str, + model: str, + thinking: str, + signature: str, + tool_calls: tuple[tuple[str, str, JsonValue], ...], + usage: dict[str, int], +) -> tuple[bytes, ...]: + frames: list[bytes] = [ + sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": _start_usage(usage), + }, + }, + ), + sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": thinking}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": signature}}, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + ] + for index, (tool_id, name, tool_input) in enumerate(tool_calls, start=1): + arguments: Final = json.dumps(tool_input) + frames += [ + sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": index, + "content_block": {"type": "tool_use", "id": tool_id, "name": name, "input": {}}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[: len(arguments) // 2]}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[len(arguments) // 2 :]}, + }, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}), + ] + frames += [ + sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "tool_use", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + sse_frame("message_stop", {"type": "message_stop"}), + ] + return tuple(frames) + + +def responses_completed( + identity: str, + model: str, + output_items: tuple[dict[str, JsonValue], ...], + usage: dict[str, int], + status: str = "completed", + incomplete_details: JsonValue = None, +) -> bytes: + return json.dumps( + { + "id": f"resp_{identity}", + "object": "response", + "created_at": 1789788253, + "status": status, + "incomplete_details": incomplete_details, + "model": model, + "output": list(output_items), + "usage": usage, + } + ).encode() + + +def responses_stream(identity: str, model: str, output_items: tuple[dict[str, JsonValue], ...]) -> tuple[bytes, ...]: + frames: list[bytes] = [ + sse_frame( + "response.created", + { + "type": "response.created", + "response": { + "id": f"resp_{identity}", + "object": "response", + "status": "in_progress", + "model": model, + "output": [], + }, + }, + ) + ] + for index, item in enumerate(output_items): + item_id: Final = str(item.get("id", f"item_{index}")) + frames.append( + sse_frame( + "response.output_item.added", + { + "type": "response.output_item.added", + "output_index": index, + "item": {**item, "content": []} if item.get("type") == "message" else item, + }, + ) + ) + if item.get("type") == "message": + text: Final = "".join(part.get("text", "") for part in item.get("content", ()) if isinstance(part, dict)) + frames.append( + sse_frame( + "response.output_text.delta", + {"type": "response.output_text.delta", "output_index": index, "item_id": item_id, "delta": text}, + ) + ) + if item.get("type") == "reasoning": + summary_text: Final = "".join( + str(part.get("text", "")) for part in item.get("summary", ()) if isinstance(part, dict) + ) + if summary_text: + frames.append( + sse_frame( + "response.reasoning_summary_text.delta", + { + "type": "response.reasoning_summary_text.delta", + "output_index": index, + "item_id": item_id, + "delta": summary_text, + }, + ) + ) + if item.get("type") == "function_call": + frames.append( + sse_frame( + "response.function_call_arguments.delta", + { + "type": "response.function_call_arguments.delta", + "output_index": index, + "item_id": item_id, + "delta": item.get("arguments", ""), + }, + ) + ) + frames.append( + sse_frame( + "response.output_item.done", + {"type": "response.output_item.done", "output_index": index, "item": item}, + ) + ) + frames.append( + sse_frame( + "response.completed", + { + "type": "response.completed", + "response": { + "id": f"resp_{identity}", + "object": "response", + "status": "completed", + "model": model, + "output": list(output_items), + "usage": {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, + }, + }, + ) + ) + return tuple(frames) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py new file mode 100644 index 00000000000..0e7f00943c2 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py @@ -0,0 +1,85 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def test_client_disconnect_mid_stream_still_bills_the_message(gateway: Gateway) -> None: + identity: Final = f"msg_dc_{uuid.uuid4().hex}" + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + + def respond(request: Request) -> Reply: + return Reply( + content_type="text/event-stream", + chunks=( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.SONNET, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PONG"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 4}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ), + pause_between_chunks=3.0, + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + with gateway.client.stream( + "POST", + "/v1/messages", + params={"beta": "true"}, + json={**request_body, "model": model}, + headers={ + **cc.cli_headers(gateway.key), + "authorization": f"Bearer {gateway.key}", + }, + ) as response: + assert response.status_code == 200, response.status_code + first: Final = next(response.iter_text()) + assert "message_start" in first, first + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + return_last_on_timeout=True, + ) + assert rows and rows[0]["prompt_tokens"] == 12, rows + assert wire.disconnected.empty(), ( + "closing the client stream must not abort the upstream call before it finishes; " + f"wire recorded a disconnect on {wire.disconnected.get_nowait()}" + ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py new file mode 100644 index 00000000000..a1e903ab5c6 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py @@ -0,0 +1,108 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_CONTEXT_MANAGEMENT: Final = { + "edits": [ + {"type": "clear_thinking_20251015", "keep": "all"}, + {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, + ] +} +_COMPACTION_BLOCK: Final = {"type": "compaction", "content": ""} + + +def _compaction_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": dict(_COMPACTION_BLOCK)}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 5}, + "context_management": { + "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] + }, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gateway: Gateway) -> None: + identity: Final = f"msg_cm_{uuid.uuid4().hex}" + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "context_management": _CONTEXT_MANAGEMENT, + } + turn3: Final = cc.tool_loop_turn2( + request_body, + ( + dict(_COMPACTION_BLOCK), + {"type": "text", "text": "continuing after compaction"}, + ), + (), + ) + seen: list[dict[str, JsonValue]] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + expected: Final = {**(turn3 if len(seen) == 2 else request_body), "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + assert body["context_management"] == _CONTEXT_MANAGEMENT + if len(seen) == 1: + return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream("msg_cm_next", cc.FABLE, "OK", {"input_tokens": 20, "output_tokens": 2}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**request_body, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + events: Final = cc.sse_events(response1.text) + assert events[1][1]["content_block"] == _COMPACTION_BLOCK, events[1] + deltas: Final = [data for event, data in events if event == "message_delta"] + assert len(deltas) == 1 and deltas[0].get("context_management") == { + "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] + }, events + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + assert len(wire.drain()) == 2 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py new file mode 100644 index 00000000000..3a4bd61fc8e --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py @@ -0,0 +1,50 @@ +import uuid +from typing import Final + +import pytest + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def test_count_tokens_forwards_to_anthropic_and_bills_nothing(gateway: Gateway) -> None: + pytest.skip( + "BUG: /v1/messages/count_tokens on an anthropic deployment runs the internal token_counter " + "and never forwards to the provider" + ) + request_body: Final = { + key: value + for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() + if key not in ("stream", "max_tokens", "thinking", "output_config") + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages/count_tokens", request.target + body = cc.JSON_OBJECT.validate_json(request.body) + assert body == {**request_body, "model": cc.FABLE}, body + return Reply(body=b'{"input_tokens": 37}') + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages/count_tokens", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + assert cc.JSON_OBJECT.validate_json(response.content) == {"input_tokens": 37} + assert len(wire.drain()) == 1 + call_id: Final = response.headers.get("x-litellm-call-id", "") + assert call_id, dict(response.headers) + leftover: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=20, + return_last_on_timeout=True, + ) + assert all(float(row["spend"]) == 0 for row in leftover), leftover diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py new file mode 100644 index 00000000000..7fa67549372 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py @@ -0,0 +1,132 @@ +import base64 +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_PDF_BYTES: Final = ( + b"%PDF-1.1\n" + b"1 0 obj<>endobj\n" + b"2 0 obj<>endobj\n" + b"3 0 obj<>endobj\n" + b"trailer<>\n%%EOF" +) +_DOC_BLOCK: Final = { + "type": "document", + "source": {"type": "base64", "data": base64.b64encode(_PDF_BYTES).decode(), "media_type": "application/pdf"}, + "citations": {"enabled": True}, +} + + +def _cited_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "A page."}}, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 0, + "delta": { + "type": "citations_delta", + "citation": { + "type": "page_location", + "document_index": 0, + "document_title": "dot.pdf", + "start_page_number": 1, + "end_page_number": 1, + "cited_text": "Page", + }, + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 6}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_base64_pdf_document_with_citations_reaches_anthropic_identical(gateway: Gateway) -> None: + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + request_body["messages"] = [ + { + "role": "user", + "content": [ + dict(_DOC_BLOCK), + {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, + ], + } + ] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply(content_type="text/event-stream", chunks=_cited_stream(f"msg_doc_{uuid.uuid4().hex}")) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + citations: Final = [ + data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" + ] + assert citations == [ + { + "type": "citations_delta", + "citation": { + "type": "page_location", + "document_index": 0, + "document_title": "dot.pdf", + "start_page_number": 1, + "end_page_number": 1, + "cited_text": "Page", + }, + } + ], citations + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py new file mode 100644 index 00000000000..0075acb6da8 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py @@ -0,0 +1,92 @@ +import json +import uuid +from pathlib import Path +from typing import Final + +import pytest +import yaml +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.process import owned_proxy +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def _error_529() -> Reply: + return Reply( + status=529, + body=json.dumps({"type": "error", "error": {"type": "overloaded_error", "message": "Overloaded"}}).encode(), + ) + + +def test_anthropic_overloaded_primary_falls_back_to_second_deployment(gateway: Gateway, tmp_path: Path) -> None: + identity: Final = f"msg_fb_{uuid.uuid4().hex}" + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + + def respond_fallback(request: Request) -> Reply: + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.SONNET, "PONG", {"input_tokens": 12, "output_tokens": 4}), + ) + + with ( + wire_server(lambda request: _error_529()) as primary, + wire_server(respond_fallback) as fallback, + ): + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["model_list"] = [ + { + "model_name": "cc-primary", + "litellm_params": { + "model": f"anthropic/{cc.SONNET}", + "api_key": cc.ANTHROPIC_API_KEY, + "api_base": primary.url, + "model_info": {"id": "primary-cc"}, + }, + }, + { + "model_name": "cc-fallback-group", + "litellm_params": { + "model": f"anthropic/{cc.SONNET}", + "api_key": cc.ANTHROPIC_API_KEY, + "api_base": fallback.url, + "model_info": {"id": "fallback-cc"}, + }, + }, + ] + config["router_settings"] = { + "num_retries": 0, + "disable_cooldowns": True, + "fallbacks": [{"cc-primary": ["cc-fallback-group"]}], + } + path: Final = tmp_path / "fallbacks.yaml" + path.write_text(yaml.safe_dump(config)) + with owned_proxy( + gateway, tmp_path, {"REDIS_HOST": "127.0.0.1", "REDIS_PORT": "6379"}, config=path + ) as candidate: + with candidate.client.stream( + "POST", + "/v1/messages", + params={"beta": "true"}, + json={**request_body, "model": "cc-primary"}, + headers={**cc.cli_headers(candidate.key), "authorization": f"Bearer {candidate.key}"}, + ) as response: + assert response.status_code == 200, response.status_code + body: Final = "".join(response.iter_text()) + assert "message_stop" in body, body + assert "PONG" in body, body + deployments: Final = candidate.get("/model/info")["data"] + fallback_id: Final = next( + entry["model_info"]["id"] + for entry in deployments + if entry["litellm_params"]["api_base"] == fallback.url + ) + assert response.headers.get("x-litellm-model-id") == fallback_id, dict(response.headers) + assert len(primary.drain()) == 1 + assert len(fallback.drain()) == 1 + rows: Final = eventually( + lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=70, + ) + assert len(rows) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py new file mode 100644 index 00000000000..a9a41b96751 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py @@ -0,0 +1,111 @@ +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def test_claude_code_adaptive_thinking_and_effort_reach_anthropic_intact(gateway: Gateway) -> None: + identity: Final = f"msg_fable_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY + assert request.headers["anthropic-version"] == "2023-06-01" + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta + assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 12, "output_tokens": 4}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + assert [event for event, _ in events][-1] == "message_stop" + assert events[4][1]["delta"]["stop_reason"] == "end_turn" + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 + + +def test_claude_code_xhigh_effort_reaches_anthropic_and_charges_by_usage(gateway: Gateway) -> None: + identity: Final = f"msg_opus_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "xhigh", 128000) + cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta + assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.OPUS} + assert body == expected, _diff(expected, body) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.OPUS, "PONG", {"input_tokens": 10, "output_tokens": 5}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.OPUS}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 5 * 5e-6) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py new file mode 100644 index 00000000000..0deed7db56a --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py @@ -0,0 +1,69 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" +_IMAGE_BLOCK: Final = { + "type": "image", + "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, +} + + +def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", + ) + with_image_result: Final = cc.tool_loop_turn2( + turn1, + ({"type": "tool_use", "id": "toolu_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), + (("toolu_img", [dict(_IMAGE_BLOCK)]),), + ) + pasted: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + pasted["messages"] = [ + { + "role": "user", + "content": [ + dict(_IMAGE_BLOCK), + {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, + ], + } + ] + seen: list[dict[str, JsonValue]] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + expected: Final = {**(pasted if len(seen) == 2 else with_image_result), "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream( + f"msg_img_{uuid.uuid4().hex}", cc.FABLE, "RED", {"input_tokens": 20, "output_tokens": 2} + ), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**with_image_result, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**pasted, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + assert len(wire.drain()) == 2 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py new file mode 100644 index 00000000000..c6a46ca2806 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py @@ -0,0 +1,167 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_TURN1_TOOL: Final = ("toolu_a", "Read", {"file_path": "/tmp/cc_probe/a.txt"}) +_TURN2_TOOL: Final = ("toolu_b", "Read", {"file_path": "/tmp/cc_probe/b.txt"}) + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def _turn2(base: dict[str, JsonValue]) -> dict[str, JsonValue]: + return cc.tool_loop_turn2( + base, + ( + {"type": "thinking", "thinking": "plan", "signature": "sig1"}, + {"type": "tool_use", "id": _TURN1_TOOL[0], "name": _TURN1_TOOL[1], "input": _TURN1_TOOL[2]}, + ), + ((_TURN1_TOOL[0], "ALPHA"),), + ) + + +def _turn3(turn2: dict[str, JsonValue]) -> dict[str, JsonValue]: + return cc.tool_loop_turn2( + turn2, + ( + {"type": "thinking", "thinking": "got A", "signature": "sig2"}, + {"type": "text", "text": "got A"}, + {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": _TURN2_TOOL[2]}, + ), + ((_TURN2_TOOL[0], "BRAVO"),), + ) + + +def _interleaved_stream(identity: str) -> tuple[bytes, ...]: + usage: Final = {"input_tokens": 20, "output_tokens": 12} + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": usage["input_tokens"], "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "got A"}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": "sig2"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 1, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 1, "delta": {"type": "text_delta", "text": "got A"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 2, + "content_block": {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": {}}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": {"type": "input_json_delta", "partial_json": '{"file_path": "/tmp/cc_pr'}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": {"type": "input_json_delta", "partial_json": 'obe/b.txt"}'}, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "tool_use", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identical(gateway: Gateway) -> None: + identity: Final = f"msg_il_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words", + ) + turn2: Final = _turn2(turn1) + turn3: Final = _turn3(turn2) + seen: list[dict[str, JsonValue]] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert upstream_beta.split(",").count("interleaved-thinking-2025-05-14") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + expected: Final = {**turn3, "model": cc.FABLE} if len(seen) == 2 else {**turn2, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + if len(seen) == 1: + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}), + ) + return Reply(content_type="text/event-stream", chunks=_interleaved_stream(identity)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + response3: Final = gateway.request( + "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response3.status_code == 200, response3.text + events: Final = cc.sse_events(response3.text) + started: Final = [ + (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" + ] + assert started == [(0, "thinking"), (1, "text"), (2, "tool_use")], started + assert events[-1][0] == "message_stop" + assert len(wire.drain()) == 2 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py new file mode 100644 index 00000000000..a198ec8500a --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py @@ -0,0 +1,53 @@ +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + +_BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" + + +def test_1m_context_beta_forwarded_and_tiered_prompt_priced_above_200k(gateway: Gateway) -> None: + identity: Final = f"msg_1m_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert upstream_beta.split(",").count("context-1m-2025-08-07") == 1, upstream_beta + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 250000, "output_tokens": 100}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + input_cost_per_token_above_200k_tokens=2e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, _BETA_1M), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(250000 * 2e-6 + 100 * 5e-6), dict(rows[0]) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py new file mode 100644 index 00000000000..ccdd4b5bd06 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py @@ -0,0 +1,86 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + + +def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: + identity1: Final = f"msg_sw1_{uuid.uuid4().hex}" + identity2: Final = f"msg_sw2_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + seen: list[dict[str, JsonValue]] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + if len(seen) == 1: + assert body == {**turn1, "model": cc.FABLE}, body.get("model") + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream( + identity1, + cc.FABLE, + "need to read the file", + "sig_anthropic_1", + (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), + {"input_tokens": 20, "output_tokens": 10}, + ), + ) + expected: Final = {**turn2, "model": cc.OPUS} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity2, cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + assert len(wire.drain()) == 2 + assert seen[1]["model"] == cc.OPUS + rows: Final = eventually( + lambda: read_rows( + 'SELECT model FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity2,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["model"] == f"anthropic/{cc.OPUS}" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py index 6d7fd461fa6..6391912d781 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -1,265 +1,50 @@ -import json import uuid from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from pydantic import JsonValue, TypeAdapter +from integration.messages_endpoint import _claude_code as cc -_API_KEY: Final = "synthetic-anthropic-key" -_MODEL: Final = "claude-sonnet-4-5" -_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) -_CLI_BETA: Final = ( - "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13," - "context-management-2025-06-27,prompt-caching-scope-2026-01-05" -) -_CACHE: Final = {"type": "ephemeral"} - - -def _schema(properties: JsonValue, required: tuple[str, ...]) -> dict[str, JsonValue]: - return { - "$schema": "https://json-schema.org/draft/2020-12/schema", - "type": "object", - "properties": properties, - "required": list(required), - "additionalProperties": False, - } - - -def _field(description: str, **extra: JsonValue) -> dict[str, JsonValue]: - return {"description": description, **extra} - - -def _tools() -> tuple[dict[str, JsonValue], ...]: - _MAX: Final = 9007199254740991 - return ( - { - "name": "Bash", - "description": "Executes a given bash command and returns its output.", - "input_schema": _schema( - { - "command": _field("The command to execute", type="string"), - "timeout": _field("Optional timeout in milliseconds (max 600000)", type="number"), - "description": _field( - "Clear, concise description of what this command does in active voice.", type="string" - ), - "run_in_background": _field("Set to true to run this command in the background.", type="boolean"), - "dangerouslyDisableSandbox": _field( - "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", - type="boolean", - ), - }, - ("command",), - ), - }, - { - "name": "Read", - "description": "Reads a file from the local filesystem.", - "input_schema": _schema( - { - "file_path": _field("The absolute path to the file to read", type="string"), - "offset": _field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), - "limit": _field("The number of lines to read.", type="integer", exclusiveMinimum=0, maximum=_MAX), - "pages": _field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), - }, - ("file_path",), - ), - }, - { - "name": "Edit", - "description": "Performs exact string replacements in files.", - "input_schema": _schema( - { - "file_path": _field("The absolute path to the file to modify", type="string"), - "old_string": _field("The text to replace", type="string"), - "new_string": _field( - "The text to replace it with (must be different from old_string)", type="string" - ), - "replace_all": _field( - "Replace all occurrences of old_string (default false)", default=False, type="boolean" - ), - }, - ("file_path", "old_string", "new_string"), - ), - }, - { - "name": "Agent", - "description": "Launch a new agent to handle complex, multi-step tasks.", - "input_schema": _schema( - { - "description": _field("A short (3-5 word) description of the task", type="string"), - "prompt": _field("The task for the agent to perform", type="string"), - "subagent_type": _field("The type of specialized agent to use for this task", type="string"), - "model": _field( - "Optional model override for this agent.", - type="string", - enum=["sonnet", "opus", "haiku", "fable"], - ), - "run_in_background": _field( - "Agents run in the background by default; you will be notified when one completes.", - type="boolean", - ), - "isolation": _field("Isolation mode.", type="string", enum=["worktree", "remote"]), - }, - ("description", "prompt"), - ), - }, - ) - - -def _claude_code_request(cache_bust: str) -> dict[str, JsonValue]: - reminders: Final = ( - f"\n{cache_bust}\n", - "\nSynthetic model identity reminder.\n", - "\nSynthetic agent types reminder.\n", - "\nSynthetic skills reminder.\n", - "\n15000000 tokens left\n", - "\nSynthetic date reminder.\n", - "\nSynthetic attribution reminder.\n", - ) - return { - "model": "", - "system": [ - {"type": "text", "text": "Synthetic billing header block from a Claude Code request."}, - {"type": "text", "text": "Synthetic agent identity system prompt.", "cache_control": _CACHE}, - {"type": "text", "text": "Synthetic interactive agent instructions.", "cache_control": _CACHE}, - ], - "messages": [ - { - "role": "user", - "content": [ - *[{"type": "text", "text": reminder} for reminder in reminders], - { - "type": "text", - "text": "Reply with exactly the word PONG", - "cache_control": _CACHE, - }, - ], - } - ], - "tools": list(_tools()), - "metadata": { - "user_id": json.dumps( - { - "device_id": "0" * 64, - "account_uuid": "", - "session_id": "00000000-0000-4000-8000-000000000000", - } - ) - }, - "max_tokens": 32000, - "thinking": {"budget_tokens": 31999, "type": "enabled", "display": "omitted"}, - "context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]}, - "stream": True, - } - - -def _sse_frame(event: str, data: JsonValue) -> bytes: - return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode() - - -def _message_stream(identity: str) -> tuple[bytes, ...]: - return ( - _sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": _MODEL, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 1}, - }, - }, - ), - _sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - _sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PONG"}}, - ), - _sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - _sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 4}, - }, - ), - _sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: - frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) - return tuple( - ( - next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), - json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), - ) - for frame in frames - ) +_MODEL: Final = cc.SONNET def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: identity: Final = f"msg_cc_{uuid.uuid4().hex}" - request_body: Final = _claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - cli_beta: Final = frozenset(_CLI_BETA.split(",")) + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + cli_beta: Final = frozenset(cc.CLI_BETA.split(",")) def respond(request: Request) -> Reply: assert request.method == "POST" assert request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == _API_KEY + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY assert request.headers["anthropic-version"] == "2023-06-01" upstream_beta: Final = frozenset(request.headers.get("anthropic-beta", "").split(",")) assert cli_beta <= upstream_beta, request.headers.get("anthropic-beta") - body: Final = _JSON_OBJECT.validate_json(request.body) + body: Final = cc.JSON_OBJECT.validate_json(request.body) expected: Final = {**request_body, "model": _MODEL} assert body == expected, { key: (expected.get(key), body.get(key)) for key in expected.keys() | body.keys() if expected.get(key) != body.get(key) } - return Reply(content_type="text/event-stream", chunks=_message_stream(identity)) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, _MODEL, "PONG", {"input_tokens": 12, "output_tokens": 4}), + ) with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_API_KEY) + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) response: Final = gateway.request( "POST", "/v1/messages", {**request_body, "model": model}, params={"beta": "true"}, - headers={ - "accept": "application/json", - "content-type": "application/json", - "user-agent": "claude-cli/2.1.283 (external, sdk-cli)", - "x-claude-code-session-id": "00000000-0000-4000-8000-000000000000", - "x-stainless-arch": "x64", - "x-stainless-lang": "js", - "x-stainless-os": "Linux", - "x-stainless-package-version": "0.112.1", - "x-stainless-retry-count": "0", - "x-stainless-runtime": "node", - "x-stainless-runtime-version": "v26.3.0", - "x-stainless-timeout": "600", - "anthropic-beta": _CLI_BETA, - "anthropic-dangerous-direct-browser-access": "true", - "anthropic-version": "2023-06-01", - "x-app": "cli", - "x-api-key": gateway.key, - }, + headers=cc.cli_headers(gateway.key), ) assert response.status_code == 200, response.text assert response.headers["content-type"].startswith("text/event-stream"), dict(response.headers) - events: Final = sse_events(response.text) + events: Final = cc.sse_events(response.text) assert [event for event, _ in events] == [ "message_start", "content_block_start", diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py new file mode 100644 index 00000000000..54bbe2a0704 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py @@ -0,0 +1,86 @@ +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_USAGE: Final = { + "input_tokens": 10, + "cache_read_input_tokens": 3000, + "cache_creation_input_tokens": 200, + "output_tokens": 5, +} + + +def test_claude_code_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: + identity: Final = f"msg_pc_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**turn2, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply(content_type="text/event-stream", chunks=cc.text_stream(identity, cc.FABLE, "PROBE", _USAGE)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + cache_read_input_token_cost=1e-7, + cache_creation_input_token_cost=1.25e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + usage: Final = events[0][1]["message"]["usage"] + assert usage["cache_read_input_tokens"] == 3000, usage + assert usage["cache_creation_input_tokens"] == 200, usage + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 3000 * 1e-7 + 200 * 1.25e-6 + 5 * 5e-6), dict( + rows[0] + ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py new file mode 100644 index 00000000000..f386a206451 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py @@ -0,0 +1,165 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_THINKING: Final = "need to read the file" +_SIGNATURE: Final = "sig_probe_1" +_USAGE: Final = {"input_tokens": 20, "output_tokens": 10} + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def test_claude_code_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: + identity1: Final = f"msg_tl1_{uuid.uuid4().hex}" + identity2: Final = f"msg_tl2_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + calls: Final = (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + seen: list[dict[str, JsonValue]] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + expected: Final = {**turn2, "model": cc.FABLE} if len(seen) == 2 else {**turn1, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + if len(seen) == 1: + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream(identity1, cc.FABLE, _THINKING, _SIGNATURE, calls, _USAGE), + ) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity2, cc.FABLE, "PROBE", {"input_tokens": 30, "output_tokens": 3}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + events: Final = cc.sse_events(response1.text) + assert [ + ( + event, + data.get("delta", {}).get( + "type", data.get("content_block", {}).get("type", data.get("delta", {}).get("stop_reason")) + ), + ) + for event, data in events + ] == [ + ("message_start", None), + ("content_block_start", "thinking"), + ("content_block_delta", "thinking_delta"), + ("content_block_delta", "signature_delta"), + ("content_block_stop", None), + ("content_block_start", "tool_use"), + ("content_block_delta", "input_json_delta"), + ("content_block_delta", "input_json_delta"), + ("content_block_stop", None), + ("message_delta", "tool_use"), + ("message_stop", None), + ] + assert events[5][1]["content_block"]["id"] == "toolu_read_1" + assert events[5][1]["content_block"]["name"] == "Read" + partial: Final = events[6][1]["delta"]["partial_json"] + events[7][1]["delta"]["partial_json"] + assert partial == '{"file_path": "/tmp/cc_probe/hello.txt"}' + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + events2: Final = cc.sse_events(response2.text) + assert events2[2][1]["delta"] == {"type": "text_delta", "text": "PROBE"} + assert events2[4][1]["delta"]["stop_reason"] == "end_turn" + assert len(wire.drain()) == 2 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity2,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 30 + + +def test_claude_code_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and /tmp/cc_probe/world.txt and reply with both words", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + { + "type": "tool_use", + "id": "toolu_read_2", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/world.txt"}, + }, + ), + (("toolu_read_2", "1\tPROBE2\n2\t"), ("toolu_read_1", "1\tPROBE\n2\t")), + ) + + def respond(request: Request) -> Reply: + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**turn2, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + results: Final = [block for block in body["messages"][3]["content"] if block["type"] == "tool_result"] + assert [block["tool_use_id"] for block in results] == ["toolu_read_2", "toolu_read_1"] + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(f"msg_mt_{uuid.uuid4().hex}", cc.FABLE, "PROBE PROBE2", _USAGE), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py new file mode 100644 index 00000000000..17cfaee9dd2 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py @@ -0,0 +1,136 @@ +import json +import uuid +from typing import Final + +import pytest + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def _error_body(error_type: str, message: str) -> bytes: + return json.dumps({"type": "error", "error": {"type": error_type, "message": message}}).encode() + + +def _assert_upstream_error_status_passthrough(gateway: Gateway, status: int, error_type: str) -> None: + request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body = cc.JSON_OBJECT.validate_json(request.body) + assert body["stream"] is False + return Reply(status=status, body=_error_body(error_type, "Upstream rejected")) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY, num_retries=0 + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == status, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload["type"] == "error", payload + assert payload["error"]["type"] == error_type, payload + assert len(wire.drain()) == 1 + call_id: Final = response.headers.get("x-litellm-call-id", "") + assert call_id, dict(response.headers) + leftover: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=20, + return_last_on_timeout=True, + ) + assert all(float(row["spend"]) == 0 for row in leftover), leftover + + +def test_anthropic_529_overloaded_error_passes_through_with_client_status(gateway: Gateway) -> None: + pytest.skip( + "BUG: upstream 529 overloaded_error is re-raised through exception_type as InternalServerError " + "and reaches the client as 500 api_error" + ) + _assert_upstream_error_status_passthrough(gateway, 529, "overloaded_error") + + +def test_anthropic_429_rate_limit_error_passes_through_with_client_status(gateway: Gateway) -> None: + _assert_upstream_error_status_passthrough(gateway, 429, "rate_limit_error") + + +def test_anthropic_stream_stop_reason_max_tokens_and_refusal_reach_client(gateway: Gateway) -> None: + for stop_reason in ("max_tokens", "refusal"): + _assert_stream_stop_reason_reaches_client(gateway, stop_reason) + + +def _assert_stream_stop_reason_reaches_client(gateway: Gateway, stop_reason: str) -> None: + identity: Final = f"msg_stop_{uuid.uuid4().hex}" + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + + def respond(request: Request) -> Reply: + return Reply( + content_type="text/event-stream", + chunks=( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.SONNET, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PAR"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": stop_reason, "stop_sequence": None}, + "usage": {"output_tokens": 32000}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + assert events[2][1]["delta"]["text"] == "PAR" + assert events[4][1]["delta"]["stop_reason"] == stop_reason + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=20, + return_last_on_timeout=True, + ) + assert isinstance(rows, list) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py new file mode 100644 index 00000000000..a275b2e3305 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py @@ -0,0 +1,184 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +WEB_SEARCH_TOOL: Final = { + "name": "WebSearch", + "description": "Search the web. Returns result blocks with titles and URLs.", + "input_schema": cc.schema( + { + "query": cc.field("The search query to use", type="string", minLength=2), + "allowed_domains": cc.field( + "Only include search results from these domains", type="array", items={"type": "string"} + ), + "blocked_domains": cc.field( + "Never include search results from these domains", type="array", items={"type": "string"} + ), + }, + ("query",), + ), +} + + +def _web_search_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1, "server_tool_use": {"web_search_requests": 1}}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {}}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "input_json_delta", "partial_json": '{"query": "current LiteLLM version"}'}, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 1, + "content_block": { + "type": "web_search_tool_result", + "tool_use_id": "srvtoolu_1", + "content": [ + { + "type": "web_search_result", + "title": "litellm releases", + "url": "https://example.com/litellm", + "page_age": None, + "encrypted_content": "enc_ws_1", + } + ], + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 2, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 2, "delta": {"type": "text_delta", "text": "1.104.0"}}, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": { + "type": "citations_delta", + "citation": { + "type": "web_search_result_location", + "url": "https://example.com/litellm", + "title": "litellm releases", + "cited_text": "version 1.104.0", + "encrypted_index": "eidx_1", + }, + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 15, "server_tool_use": {"web_search_requests": 1}}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_claude_code_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: + identity: Final = f"msg_ws_{uuid.uuid4().hex}" + request_body: Final = { + **cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Use web search to find the current LiteLLM version and answer in one word", + ), + } + request_body["tools"] = [*request_body["tools"], WEB_SEARCH_TOOL] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + assert body["tools"][-1] == WEB_SEARCH_TOOL + return Reply(content_type="text/event-stream", chunks=_web_search_stream(identity)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + started: Final = [ + (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" + ] + assert started == [(0, "server_tool_use"), (1, "web_search_tool_result"), (2, "text")], started + citations: Final = [ + data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" + ] + assert len(citations) == 1 and citations[0]["citation"]["url"] == "https://example.com/litellm", citations + start_usage: Final = events[0][1]["message"]["usage"] + assert start_usage["server_tool_use"]["web_search_requests"] == 1, start_usage + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + token_cost: Final = 20 * 1e-6 + 15 * 5e-6 + assert float(rows[0]["spend"]) >= token_cost, dict(rows[0]) diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py new file mode 100644 index 00000000000..2e1d4d431ff --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py @@ -0,0 +1,107 @@ +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def test_compact_edit_maps_to_responses_context_management(gateway: Gateway) -> None: + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "context_management": { + "edits": [ + {"type": "clear_thinking_20251015", "keep": "all"}, + {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, + ] + }, + "stream": False, + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + assert body.get("context_management") == [{"type": "compaction", "compact_threshold": 150000}], body.get( + "context_management" + ) + return Reply( + body=cc.responses_completed( + "cm", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "OK", "annotations": []}], + }, + ), + {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload["content"] == [{"type": "text", "text": "OK"}], payload["content"] + assert len(wire.drain()) == 1 + + +def test_compaction_output_item_reaches_client_as_compaction_block(gateway: Gateway) -> None: + pytest.skip( + "BUG: the responses bridge drops compaction output items in translate_response " + "(transformation.py handles only message/reasoning/function_call), so the client loses " + "the compaction block entirely" + ) + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "context_management": { + "edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}] + }, + "stream": False, + } + + def respond(request: Request) -> Reply: + return Reply( + body=cc.responses_completed( + "cm", + cc.OPENAI_BACKEND, + ( + {"type": "compaction", "content": ""}, + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "OK", "annotations": []}], + }, + ), + {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + compaction_blocks: Final = [block for block in payload["content"] if block.get("type") == "compaction"] + assert compaction_blocks == [{"type": "compaction", "content": ""}], payload["content"] + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py new file mode 100644 index 00000000000..b03e06ff867 --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py @@ -0,0 +1,39 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def test_count_tokens_on_openai_deployment_returns_token_count(gateway: Gateway) -> None: + request_body: Final = { + key: value + for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() + if key not in ("stream", "max_tokens", "thinking", "output_config") + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses/input_tokens", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + assert body["model"] == cc.OPENAI_BACKEND, body + assert body["instructions"] == str(request_body["system"]), body["instructions"] + assert len(body["input"]) == 1 and body["input"][0]["role"] == "user", body["input"] + assert "cache-bust-" in body["input"][0]["content"], body["input"] + assert body["tools"] == request_body["tools"], body["tools"] + return Reply(body=b'{"object": "response.input_tokens", "input_tokens": 37}') + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages/count_tokens", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload == {"input_tokens": 37}, payload + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py new file mode 100644 index 00000000000..121d007858e --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py @@ -0,0 +1,73 @@ +import base64 +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + +_PDF_BYTES: Final = ( + b"%PDF-1.1\n" + b"1 0 obj<>endobj\n" + b"2 0 obj<>endobj\n" + b"3 0 obj<>endobj\n" + b"trailer<>\n%%EOF" +) +_PDF_B64: Final = base64.b64encode(_PDF_BYTES).decode() + + +def test_pdf_document_block_maps_to_input_file_on_bridge(gateway: Gateway) -> None: + request_body: Final = {**cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), "stream": False} + doc_text: Final = f"What is on page one? {uuid.uuid4().hex}" + request_body["messages"] = [ + { + "role": "user", + "content": [ + { + "type": "document", + "source": {"type": "base64", "data": _PDF_B64, "media_type": "application/pdf"}, + "title": "dot.pdf", + }, + {"type": "text", "text": doc_text}, + ], + } + ] + + def respond(request: Request) -> Reply: + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + user_msg: Final = body["input"][0] + assert user_msg["content"][0] == { + "type": "input_file", + "filename": "dot.pdf", + "file_data": f"data:application/pdf;base64,{_PDF_B64}", + }, user_msg + assert user_msg["content"][1] == {"type": "input_text", "text": doc_text}, user_msg + return Reply( + body=cc.responses_completed( + "doc", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "Page one.", "annotations": []}], + }, + ), + {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py new file mode 100644 index 00000000000..32a7ce3bd6e --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py @@ -0,0 +1,77 @@ +import json +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc + + +def test_openai_429_error_comes_back_in_anthropic_shape(gateway: Gateway) -> None: + request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + return Reply( + status=429, + body=json.dumps( + {"error": {"type": "rate_limit_error", "message": "Rate limit reached", "code": "rate_limit_exceeded"}} + ).encode(), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY, num_retries=0 + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 429, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload["type"] == "error", payload + assert payload["error"]["type"] == "rate_limit_error", payload + assert len(wire.drain()) == 1 + + +def test_incomplete_responses_completion_maps_to_max_tokens(gateway: Gateway) -> None: + request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} + + def respond(request: Request) -> Reply: + assert request.target == "/responses", request.target + return Reply( + body=cc.responses_completed( + "inc", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "PAR", "annotations": []}], + }, + ), + {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, + status="incomplete", + incomplete_details={"reason": "max_output_tokens"}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload["stop_reason"] == "max_tokens", payload + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py new file mode 100644 index 00000000000..355004c1f16 --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py @@ -0,0 +1,149 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_INSTRUCTIONS: Final = "\n".join(block["text"] for block in cc.system_blocks()) +_OUTPUT_ITEMS: Final = ( + { + "type": "reasoning", + "id": "rs_1", + "summary": [{"type": "summary_text", "text": "short plan"}], + "encrypted_content": "enc_1", + }, + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "PONG", "annotations": []}], + }, +) + + +def _expected_responses_body(request_body: dict[str, JsonValue], effort: str) -> dict[str, JsonValue]: + user_blocks: Final = request_body["messages"][0]["content"] + expected_tools: Final = tuple( + { + "type": "function", + "name": tool["name"], + "strict": False, + "description": tool["description"], + "parameters": tool["input_schema"], + } + for tool in request_body["tools"] + ) + input_items: Final = [ + { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": block["text"]} for block in user_blocks], + } + ] + for message in request_body["messages"][1:]: + input_items.append( + { + "type": "message", + "role": "system", + "content": [ + {"type": "input_text", "text": block["text"]} + for block in message["content"] + if block.get("type") == "text" + ], + } + ) + return { + "model": cc.OPENAI_BACKEND, + "input": input_items, + "include": ["reasoning.encrypted_content"], + "instructions": _INSTRUCTIONS, + "max_output_tokens": request_body["max_tokens"], + "tools": list(expected_tools), + "reasoning": {"effort": effort}, + "stream": True, + "user": cc.METADATA_USER_ID[:64], + "prompt_cache_key": "00000000-0000-4000-8000-000000000000", + } + + +def _assert_client_events(text: str) -> None: + events: Final = cc.sse_events(text) + assert [event for event, _ in events] == [ + "message_start", + "content_block_start", + "content_block_delta", + "content_block_delta", + "content_block_stop", + "content_block_start", + "content_block_delta", + "content_block_stop", + "message_delta", + "message_stop", + ], [event for event, _ in events] + assert events[1][1]["content_block"]["type"] == "thinking" + assert events[2][1]["delta"] == {"type": "thinking_delta", "thinking": "short plan"} + assert events[3][1]["delta"]["type"] == "signature_delta" + assert events[6][1]["delta"] == {"type": "text_delta", "text": "PONG"} + assert events[8][1]["delta"]["stop_reason"] == "end_turn" + + +def test_claude_code_frontier_body_becomes_reasoning_effort_on_responses_bridge(gateway: Gateway) -> None: + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + assert request.headers["authorization"] == f"Bearer {cc.OPENAI_API_KEY}" + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = _expected_responses_body(request_body, "high") + assert body == expected, { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply( + content_type="text/event-stream", chunks=cc.responses_stream("bridge1", cc.OPENAI_BACKEND, _OUTPUT_ITEMS) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + _assert_client_events(response.text) + assert len(wire.drain()) == 1 + + +def test_claude_code_legacy_thinking_budget_maps_to_reasoning_effort_on_bridge(gateway: Gateway) -> None: + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + assert body["reasoning"] == {"effort": "high"}, body.get("reasoning") + assert body["model"] == cc.OPENAI_BACKEND + return Reply( + content_type="text/event-stream", chunks=cc.responses_stream("bridge2", cc.OPENAI_BACKEND, _OUTPUT_ITEMS) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + _assert_client_events(response.text) + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py new file mode 100644 index 00000000000..8abdde6f2cd --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py @@ -0,0 +1,112 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" +_IMAGE_BLOCK: Final = { + "type": "image", + "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, +} +_DATA_URL: Final = f"data:image/png;base64,{_PNG_B64}" + + +def _respond_ok(request: Request) -> Reply: + return Reply( + body=cc.responses_completed( + "img", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "RED", "annotations": []}], + }, + ), + {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, + ) + ) + + +def test_tool_result_image_maps_to_input_image_on_bridge(gateway: Gateway) -> None: + turn1: Final = { + **cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", + ), + "stream": False, + } + turn2: Final = cc.tool_loop_turn2( + turn1, + ({"type": "tool_use", "id": "call_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), + (("call_img", [dict(_IMAGE_BLOCK)]),), + ) + + def respond(request: Request) -> Reply: + body: Final = cc.JSON_OBJECT.validate_json(request.body) + outputs: Final = [ + item for item in body["input"] if isinstance(item, dict) and item.get("type") == "function_call_output" + ] + assert len(outputs) == 1 and outputs[0]["call_id"] == "call_img", outputs + image_messages: Final = [ + item + for item in body["input"] + if isinstance(item, dict) + and item.get("type") == "message" + and item.get("role") == "user" + and any(isinstance(part, dict) and part.get("type") == "input_image" for part in item.get("content", ())) + ] + assert image_messages, body["input"] + image_parts: Final = [ + part + for part in image_messages[0]["content"] + if isinstance(part, dict) and part.get("type") == "input_image" + ] + assert image_parts[0]["image_url"] == _DATA_URL, image_parts + return _respond_ok(request) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + + +def test_pasted_image_maps_to_input_image_on_bridge(gateway: Gateway) -> None: + request_body: Final = {**cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), "stream": False} + pasted_text: Final = f"What colour is this? {uuid.uuid4().hex}" + request_body["messages"] = [ + {"role": "user", "content": [dict(_IMAGE_BLOCK), {"type": "text", "text": pasted_text}]} + ] + + def respond(request: Request) -> Reply: + body: Final = cc.JSON_OBJECT.validate_json(request.body) + user_msg: Final = body["input"][0] + assert user_msg["content"][0] == {"type": "input_image", "image_url": _DATA_URL}, user_msg + assert user_msg["content"][1] == {"type": "input_text", "text": pasted_text}, user_msg + return _respond_ok(request) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py new file mode 100644 index 00000000000..4944c399a09 --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py @@ -0,0 +1,128 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + + +def _expected_input(turn1: dict[str, JsonValue]) -> list[JsonValue]: + return [ + { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][0]["content"]], + }, + { + "type": "message", + "role": "system", + "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][1]["content"]], + }, + {"type": "reasoning", "summary": [{"type": "summary_text", "text": "plan"}]}, + { + "type": "function_call", + "call_id": "toolu_a", + "name": "Read", + "arguments": '{"file_path": "/tmp/cc_probe/a.txt"}', + }, + {"type": "function_call_output", "call_id": "toolu_a", "output": "ALPHA"}, + { + "type": "message", + "role": "system", + "content": [ + {"type": "input_text", "text": "14999970 tokens left"}, + { + "type": "input_text", + "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", + }, + ], + }, + {"type": "reasoning", "summary": [{"type": "summary_text", "text": "got A"}]}, + { + "type": "function_call", + "call_id": "toolu_b", + "name": "Read", + "arguments": '{"file_path": "/tmp/cc_probe/b.txt"}', + }, + { + "type": "message", + "role": "assistant", + "content": [{"type": "output_text", "text": "got A"}], + }, + {"type": "function_call_output", "call_id": "toolu_b", "output": "BRAVO"}, + { + "type": "message", + "role": "system", + "content": [ + {"type": "input_text", "text": "14999970 tokens left"}, + { + "type": "input_text", + "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", + }, + ], + }, + ] + + +def test_bridge_replays_interleaved_history_in_order(gateway: Gateway) -> None: + turn1: Final = { + **cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time", + ), + "stream": False, + } + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "plan", "signature": "sig_anthropic_1"}, + {"type": "tool_use", "id": "toolu_a", "name": "Read", "input": {"file_path": "/tmp/cc_probe/a.txt"}}, + ), + (("toolu_a", "ALPHA"),), + ) + turn3: Final = cc.tool_loop_turn2( + turn2, + ( + {"type": "thinking", "thinking": "got A", "signature": "sig_anthropic_2"}, + {"type": "text", "text": "got A"}, + {"type": "tool_use", "id": "toolu_b", "name": "Read", "input": {"file_path": "/tmp/cc_probe/b.txt"}}, + ), + (("toolu_b", "BRAVO"),), + ) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + assert body["input"] == _expected_input(turn1), body["input"] + return Reply( + body=cc.responses_completed( + "il", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "ALPHA BRAVO", "annotations": []}], + }, + ), + {"input_tokens": 50, "output_tokens": 4, "total_tokens": 54}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn3, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py new file mode 100644 index 00000000000..67f9a41300d --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py @@ -0,0 +1,95 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + + +def test_anthropic_signed_thinking_in_history_crosses_to_responses_bridge(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + bridge_input_box: list[JsonValue] = [] + + def respond_anthropic(request: Request) -> Reply: + assert request.target == "/v1/messages", request.target + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream( + f"msg_sw_{uuid.uuid4().hex}", + cc.FABLE, + "need to read the file", + "sig_anthropic_1", + (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), + {"input_tokens": 20, "output_tokens": 10}, + ), + ) + + def respond_openai(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + bridge_input_box.append(body.get("input")) + reasoning_items: Final = [ + item for item in body["input"] if isinstance(item, dict) and item.get("type") == "reasoning" + ] + assert reasoning_items == [ + {"type": "reasoning", "summary": [{"type": "summary_text", "text": "need to read the file"}]} + ], reasoning_items + return Reply( + content_type="text/event-stream", + chunks=cc.responses_stream( + "sw", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "PROBE", "annotations": []}], + }, + ), + ), + ) + + with ( + wire_server(respond_anthropic) as wire_a, + wire_server(respond_openai) as wire_b, + gateway.scenario() as scenario, + ): + fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire_a.url, api_key=cc.ANTHROPIC_API_KEY) + openai_alias: Final = scenario.model( + model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire_b.url, api_key=cc.OPENAI_API_KEY + ) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": openai_alias}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + events: Final = cc.sse_events(response2.text) + assert events[-1][0] == "message_stop" + assert len(wire_a.drain()) == 1 + assert len(wire_b.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py new file mode 100644 index 00000000000..fa33c98048a --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py @@ -0,0 +1,223 @@ +import json +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_INSTRUCTIONS: Final = "\n".join(block["text"] for block in cc.system_blocks()) + + +def _turn1() -> dict[str, JsonValue]: + return { + **cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ), + "stream": False, + } + + +def _upstream_items(calls: tuple[tuple[str, str, JsonValue], ...]) -> tuple[dict[str, JsonValue], ...]: + return ( + { + "type": "reasoning", + "id": "rs_1", + "summary": [{"type": "summary_text", "text": "short plan"}], + "encrypted_content": "enc_1", + }, + *( + { + "type": "function_call", + "id": f"fc_{i}", + "call_id": call_id, + "name": name, + "arguments": json.dumps(tool_input), + "status": "completed", + } + for i, (call_id, name, tool_input) in enumerate(calls, start=1) + ), + ) + + +def _expected_turn2_input( + turn1: dict[str, JsonValue], + assistant_content: tuple[dict[str, JsonValue], ...], + tool_results: tuple[tuple[str, JsonValue], ...], +) -> list[JsonValue]: + items: Final = [ + { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][0]["content"]], + }, + { + "type": "message", + "role": "system", + "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][1]["content"]], + }, + { + "type": "reasoning", + "summary": [{"type": "summary_text", "text": "short plan"}], + "encrypted_content": "enc_1", + }, + ] + items += [ + { + "type": "function_call", + "call_id": block["id"], + "name": block["name"], + "arguments": json.dumps(block["input"]), + } + for block in assistant_content + if block.get("type") == "tool_use" + ] + items += [ + {"type": "function_call_output", "call_id": tool_use_id, "output": content} + for tool_use_id, content in tool_results + ] + items.append( + { + "type": "message", + "role": "system", + "content": [ + {"type": "input_text", "text": "14999970 tokens left"}, + { + "type": "input_text", + "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", + }, + ], + } + ) + return items + + +def test_responses_bridge_replays_reasoning_and_tool_call_on_turn_two(gateway: Gateway) -> None: + turn1: Final = _turn1() + calls: Final = (("call_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) + tool_results: Final = (("call_1", "1\tPROBE\n2\t"),) + seen: list[dict[str, JsonValue]] = [] + signature_box: list[str] = [] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen.append(body) + if len(seen) == 1: + return Reply( + body=cc.responses_completed( + "tl1", + cc.OPENAI_BACKEND, + _upstream_items(calls), + {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, + ) + ) + assistant_content: Final = ( + {"type": "thinking", "thinking": "short plan", "signature": signature_box[0]}, + {"type": "tool_use", "id": "call_1", "name": "Read", "input": {"file_path": "/tmp/cc_probe/hello.txt"}}, + ) + assert body["input"] == _expected_turn2_input(turn1, assistant_content, tool_results), body["input"] + return Reply( + body=cc.responses_completed( + "tl2", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "PROBE", "annotations": []}], + }, + ), + {"input_tokens": 50, "output_tokens": 3, "total_tokens": 53}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + payload: Final = cc.JSON_OBJECT.validate_json(response1.content) + assert payload["stop_reason"] == "tool_use", payload + assert payload["content"][0]["type"] == "thinking", payload["content"] + signature: Final = payload["content"][0].get("signature") + assert signature, payload["content"][0] + signature_box.append(signature) + assert payload["content"][1] == { + "type": "tool_use", + "id": "call_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, payload["content"] + assistant_content: Final = ( + {"type": "thinking", "thinking": "short plan", "signature": signature}, + {"type": "tool_use", "id": "call_1", "name": "Read", "input": {"file_path": "/tmp/cc_probe/hello.txt"}}, + ) + turn2: Final = cc.tool_loop_turn2(turn1, assistant_content, tool_results) + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + payload2: Final = cc.JSON_OBJECT.validate_json(response2.content) + assert payload2["stop_reason"] == "end_turn", payload2 + assert len(wire.drain()) == 2 + + +def test_responses_bridge_replays_parallel_tool_calls_in_order(gateway: Gateway) -> None: + turn1: Final = _turn1() + calls: Final = ( + ("call_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}), + ("call_2", "Read", {"file_path": "/tmp/cc_probe/world.txt"}), + ) + + def respond(request: Request) -> Reply: + body: Final = cc.JSON_OBJECT.validate_json(request.body) + seen_items: Final = body["input"] + outputs: Final = [ + item for item in seen_items if isinstance(item, dict) and item.get("type") == "function_call_output" + ] + assert [item["call_id"] for item in outputs] == ["call_1", "call_2"], outputs + return Reply( + body=cc.responses_completed( + "mt", + cc.OPENAI_BACKEND, + ( + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "PROBE PROBE2", "annotations": []}], + }, + ), + {"input_tokens": 50, "output_tokens": 4, "total_tokens": 54}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + assistant_content: Final = tuple( + {"type": "tool_use", "id": call_id, "name": name, "input": tool_input} + for call_id, name, tool_input in calls + ) + turn2: Final = cc.tool_loop_turn2( + turn1, assistant_content, (("call_1", "1\tPROBE\n2\t"), ("call_2", "1\tPROBE2\n2\t")) + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py new file mode 100644 index 00000000000..e06e1b8d403 --- /dev/null +++ b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py @@ -0,0 +1,95 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from integration.messages_endpoint import _claude_code as cc +from pydantic import JsonValue + +_WEB_SEARCH_TOOL: Final = { + "name": "WebSearch", + "description": "Search the web. Returns result blocks with titles and URLs.", + "input_schema": cc.schema( + { + "query": cc.field("The search query to use", type="string", minLength=2), + "allowed_domains": cc.field( + "Only include search results from these domains", type="array", items={"type": "string"} + ), + "blocked_domains": cc.field( + "Never include search results from these domains", type="array", items={"type": "string"} + ), + }, + ("query",), + ), +} + + +def test_web_search_tool_and_cited_output_on_responses_bridge(gateway: Gateway) -> None: + request_body: Final = { + **cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Use web search to find the current LiteLLM version and answer in one word", + ), + "stream": False, + } + request_body["tools"] = [*request_body["tools"], _WEB_SEARCH_TOOL] + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/responses", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + tools: Final = body["tools"] + assert tools[-1] == { + "type": "function", + "name": "WebSearch", + "strict": False, + "description": _WEB_SEARCH_TOOL["description"], + "parameters": _WEB_SEARCH_TOOL["input_schema"], + }, tools[-1] + return Reply( + body=cc.responses_completed( + "ws", + cc.OPENAI_BACKEND, + ( + {"type": "web_search_call", "id": "ws_1", "status": "completed"}, + { + "type": "message", + "id": "msg_1", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "1.104.0", + "annotations": [ + { + "type": "url_citation", + "url": "https://example.com/litellm", + "title": "litellm releases", + "start_index": 0, + "end_index": 7, + } + ], + } + ], + }, + ), + {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + payload: Final = cc.JSON_OBJECT.validate_json(response.content) + assert payload["content"] == [{"type": "text", "text": "1.104.0"}], payload["content"] + assert len(wire.drain()) == 1 From 4356a00b5fe67fc810ca9d1707d5de1cc66f9545 Mon Sep 17 00:00:00 2001 From: kerry Date: Mon, 28 Sep 2026 23:32:03 +0000 Subject: [PATCH 09/19] test(anthropic): drop low-priority Claude Code error and count_tokens tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_claude_code_count_tokens_wire.py | 50 ------- .../test_claude_code_upstream_errors_wire.py | 136 ------------------ ...st_claude_code_count_tokens_bridge_wire.py | 39 ----- .../test_claude_code_errors_bridge_wire.py | 77 ---------- 4 files changed, 302 deletions(-) delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py deleted file mode 100644 index 3a4bd61fc8e..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_count_tokens_wire.py +++ /dev/null @@ -1,50 +0,0 @@ -import uuid -from typing import Final - -import pytest - -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_count_tokens_forwards_to_anthropic_and_bills_nothing(gateway: Gateway) -> None: - pytest.skip( - "BUG: /v1/messages/count_tokens on an anthropic deployment runs the internal token_counter " - "and never forwards to the provider" - ) - request_body: Final = { - key: value - for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() - if key not in ("stream", "max_tokens", "thinking", "output_config") - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages/count_tokens", request.target - body = cc.JSON_OBJECT.validate_json(request.body) - assert body == {**request_body, "model": cc.FABLE}, body - return Reply(body=b'{"input_tokens": 37}') - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages/count_tokens", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - assert cc.JSON_OBJECT.validate_json(response.content) == {"input_tokens": 37} - assert len(wire.drain()) == 1 - call_id: Final = response.headers.get("x-litellm-call-id", "") - assert call_id, dict(response.headers) - leftover: Final = eventually( - lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert all(float(row["spend"]) == 0 for row in leftover), leftover diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py deleted file mode 100644 index 17cfaee9dd2..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_upstream_errors_wire.py +++ /dev/null @@ -1,136 +0,0 @@ -import json -import uuid -from typing import Final - -import pytest - -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def _error_body(error_type: str, message: str) -> bytes: - return json.dumps({"type": "error", "error": {"type": error_type, "message": message}}).encode() - - -def _assert_upstream_error_status_passthrough(gateway: Gateway, status: int, error_type: str) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body = cc.JSON_OBJECT.validate_json(request.body) - assert body["stream"] is False - return Reply(status=status, body=_error_body(error_type, "Upstream rejected")) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY, num_retries=0 - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == status, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["type"] == "error", payload - assert payload["error"]["type"] == error_type, payload - assert len(wire.drain()) == 1 - call_id: Final = response.headers.get("x-litellm-call-id", "") - assert call_id, dict(response.headers) - leftover: Final = eventually( - lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert all(float(row["spend"]) == 0 for row in leftover), leftover - - -def test_anthropic_529_overloaded_error_passes_through_with_client_status(gateway: Gateway) -> None: - pytest.skip( - "BUG: upstream 529 overloaded_error is re-raised through exception_type as InternalServerError " - "and reaches the client as 500 api_error" - ) - _assert_upstream_error_status_passthrough(gateway, 529, "overloaded_error") - - -def test_anthropic_429_rate_limit_error_passes_through_with_client_status(gateway: Gateway) -> None: - _assert_upstream_error_status_passthrough(gateway, 429, "rate_limit_error") - - -def test_anthropic_stream_stop_reason_max_tokens_and_refusal_reach_client(gateway: Gateway) -> None: - for stop_reason in ("max_tokens", "refusal"): - _assert_stream_stop_reason_reaches_client(gateway, stop_reason) - - -def _assert_stream_stop_reason_reaches_client(gateway: Gateway, stop_reason: str) -> None: - identity: Final = f"msg_stop_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - - def respond(request: Request) -> Reply: - return Reply( - content_type="text/event-stream", - chunks=( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.SONNET, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PAR"}}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": stop_reason, "stop_sequence": None}, - "usage": {"output_tokens": 32000}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - assert events[2][1]["delta"]["text"] == "PAR" - assert events[4][1]["delta"]["stop_reason"] == stop_reason - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), - lambda values: len(values) == 1, - seconds=20, - return_last_on_timeout=True, - ) - assert isinstance(rows, list) diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py deleted file mode 100644 index b03e06ff867..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_count_tokens_bridge_wire.py +++ /dev/null @@ -1,39 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_count_tokens_on_openai_deployment_returns_token_count(gateway: Gateway) -> None: - request_body: Final = { - key: value - for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() - if key not in ("stream", "max_tokens", "thinking", "output_config") - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses/input_tokens", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - assert body["model"] == cc.OPENAI_BACKEND, body - assert body["instructions"] == str(request_body["system"]), body["instructions"] - assert len(body["input"]) == 1 and body["input"][0]["role"] == "user", body["input"] - assert "cache-bust-" in body["input"][0]["content"], body["input"] - assert body["tools"] == request_body["tools"], body["tools"] - return Reply(body=b'{"object": "response.input_tokens", "input_tokens": 37}') - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages/count_tokens", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload == {"input_tokens": 37}, payload - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py deleted file mode 100644 index 32a7ce3bd6e..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_errors_bridge_wire.py +++ /dev/null @@ -1,77 +0,0 @@ -import json -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_openai_429_error_comes_back_in_anthropic_shape(gateway: Gateway) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - return Reply( - status=429, - body=json.dumps( - {"error": {"type": "rate_limit_error", "message": "Rate limit reached", "code": "rate_limit_exceeded"}} - ).encode(), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY, num_retries=0 - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 429, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["type"] == "error", payload - assert payload["error"]["type"] == "rate_limit_error", payload - assert len(wire.drain()) == 1 - - -def test_incomplete_responses_completion_maps_to_max_tokens(gateway: Gateway) -> None: - request_body: Final = {**cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), "stream": False} - - def respond(request: Request) -> Reply: - assert request.target == "/responses", request.target - return Reply( - body=cc.responses_completed( - "inc", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PAR", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, - status="incomplete", - incomplete_details={"reason": "max_output_tokens"}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["stop_reason"] == "max_tokens", payload - assert len(wire.drain()) == 1 From 90c8f22174c055e5a747d3dc29a2a7af6649cc50 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 00:13:31 +0000 Subject: [PATCH 10/19] test(anthropic): send the full 24-tool Claude Code request and pin upstream headers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../messages_endpoint/_claude_code.py | 472 ++++++++++++++++-- .../anthropic/test_claude_code_native_wire.py | 7 +- 2 files changed, 428 insertions(+), 51 deletions(-) diff --git a/tests/integration/messages_endpoint/_claude_code.py b/tests/integration/messages_endpoint/_claude_code.py index 5688a7e9776..aea893780ab 100644 --- a/tests/integration/messages_endpoint/_claude_code.py +++ b/tests/integration/messages_endpoint/_claude_code.py @@ -51,55 +51,6 @@ def field(description: str, **extra: JsonValue) -> dict[str, JsonValue]: def tools() -> tuple[dict[str, JsonValue], ...]: _MAX: Final = 9007199254740991 return ( - { - "name": "Bash", - "description": "Executes a given bash command and returns its output.", - "input_schema": schema( - { - "command": field("The command to execute", type="string"), - "timeout": field("Optional timeout in milliseconds (max 600000)", type="number"), - "description": field( - "Clear, concise description of what this command does in active voice.", type="string" - ), - "run_in_background": field("Set to true to run this command in the background.", type="boolean"), - "dangerouslyDisableSandbox": field( - "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", - type="boolean", - ), - }, - ("command",), - ), - }, - { - "name": "Read", - "description": "Reads a file from the local filesystem.", - "input_schema": schema( - { - "file_path": field("The absolute path to the file to read", type="string"), - "offset": field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), - "limit": field("The number of lines to read.", type="integer", exclusiveMinimum=0, maximum=_MAX), - "pages": field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), - }, - ("file_path",), - ), - }, - { - "name": "Edit", - "description": "Performs exact string replacements in files.", - "input_schema": schema( - { - "file_path": field("The absolute path to the file to modify", type="string"), - "old_string": field("The text to replace", type="string"), - "new_string": field( - "The text to replace it with (must be different from old_string)", type="string" - ), - "replace_all": field( - "Replace all occurrences of old_string (default false)", default=False, type="boolean" - ), - }, - ("file_path", "old_string", "new_string"), - ), - }, { "name": "Agent", "description": "Launch a new agent to handle complex, multi-step tasks.", @@ -122,6 +73,429 @@ def tools() -> tuple[dict[str, JsonValue], ...]: ("description", "prompt"), ), }, + { + "name": "Bash", + "description": "Executes a given bash command and returns its output.", + "input_schema": schema( + { + "command": field("The command to execute", type="string"), + "timeout": field("Optional timeout in milliseconds (max 600000)", type="number"), + "description": field( + "Clear, concise description of what this command does in active voice.", + type="string", + ), + "run_in_background": field("Set to true to run this command in the background.", type="boolean"), + "dangerouslyDisableSandbox": field( + "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", + type="boolean", + ), + }, + ("command",), + ), + }, + { + "name": "CronCreate", + "description": "Schedule a prompt to be enqueued at a future time.", + "input_schema": schema( + { + "cron": field( + 'Standard 5-field cron expression in local time: "M H DoM Mon DoW" (e.g.', + type="string", + ), + "prompt": field("The prompt to enqueue at each fire time.", type="string"), + "recurring": field( + "true (default) = fire on every cron match until deleted or auto-expired after 7 days.", + type="boolean", + ), + "durable": field( + "true = persist to .claude/scheduled_tasks.json and survive restarts.", + type="boolean", + ), + }, + ("cron", "prompt"), + ), + }, + { + "name": "CronDelete", + "description": "Cancel a cron job previously scheduled with CronCreate.", + "input_schema": schema( + { + "id": field("Job ID returned by CronCreate.", type="string"), + }, + ("id",), + ), + }, + { + "name": "CronList", + "description": "List all cron jobs scheduled via CronCreate, both durable (.claude/scheduled_tasks.json) and session-only.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + }, + { + "name": "Edit", + "description": "Performs exact string replacements in files.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to modify", type="string"), + "old_string": field("The text to replace", type="string"), + "new_string": field("The text to replace it with (must be different from old_string)", type="string"), + "replace_all": field( + "Replace all occurrences of old_string (default false)", + default=False, + type="boolean", + ), + }, + ("file_path", "old_string", "new_string"), + ), + }, + { + "name": "EnterWorktree", + "description": "Use this tool ONLY when explicitly instructed to work in a worktree — either by the user directly, or by project instruc", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "name": field("Optional name for a new worktree.", type="string"), + "path": field( + "Path to an existing worktree to switch into instead of creating a new one.", + type="string", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "ExitWorktree", + "description": "Exit a worktree session created by EnterWorktree and return the session to the original working directory.", + "input_schema": schema( + { + "action": field( + '"keep" leaves the worktree and branch on disk; "remove" deletes both.', + type="string", + enum=["keep", "remove"], + ), + "discard_changes": field( + 'Required true when action is "remove" and the worktree has uncommitted files or unmerged commits.', + type="boolean", + ), + }, + ("action",), + ), + }, + { + "name": "ListAgents", + "description": "Lists agents you can SendMessage to — in-process subagents you spawned, the teammates on your team, other local Claude s", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "channel": field("Not available in this build; leave unset.", type="string", maxLength=256), + "q": field("Not available in this build; leave unset.", type="string", maxLength=256), + }, + "additionalProperties": False, + }, + }, + { + "name": "NotebookEdit", + "description": "Replaces, inserts, or deletes a single cell in a Jupyter notebook (.ipynb file).", + "input_schema": schema( + { + "notebook_path": field( + "The absolute path to the Jupyter notebook file to edit (must be absolute, not relative)", + type="string", + ), + "cell_id": field("The ID of the cell to edit.", type="string"), + "new_source": field("The new source for the cell", type="string"), + "cell_type": field( + "The type of the cell (code or markdown).", + type="string", + enum=["code", "markdown"], + ), + "edit_mode": field( + "The type of edit to make (replace, insert, delete).", + type="string", + enum=["replace", "insert", "delete"], + ), + }, + ("notebook_path", "new_source"), + ), + }, + { + "name": "Read", + "description": "Reads a file from the local filesystem.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to read", type="string"), + "offset": field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), + "limit": field( + "The number of lines to read.", + type="integer", + exclusiveMinimum=0, + maximum=_MAX, + ), + "pages": field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), + }, + ("file_path",), + ), + }, + { + "name": "ReportFindings", + "description": "Report code-review findings as a typed list so the host UI can render them.", + "input_schema": schema( + { + "level": field( + "Effort level the review ran at", + type="string", + enum=["low", "medium", "high", "xhigh", "max"], + ), + "findings": field( + "Verified findings, most-severe first; empty if none survived", + maxItems=32, + type="array", + items={ + "type": "object", + "properties": { + "file": field("Repo-relative path of the file the finding is in", type="string"), + "line": field( + "1-indexed line the finding anchors to", + type="integer", + minimum=-_MAX, + maximum=_MAX, + ), + "summary": field("One-sentence statement of the defect", type="string"), + "short_summary": field( + "Compressed label for compact UI (≤60 chars): the claim alone, no rationale or consequence clause", + type="string", + maxLength=60, + ), + "failure_scenario": field("Concrete inputs/state → wrong output/crash", type="string"), + "category": field( + "Short kebab-case slug of the finding type, e.g.", + type="string", + maxLength=40, + ), + "verdict": field( + "Set when a verify pass ran; absent on inline-only reviews", + type="string", + enum=["CONFIRMED", "PLAUSIBLE"], + ), + "outcome": field( + "Set ONLY when re-reporting after applying fixes: what happened to this finding", + type="string", + enum=["fixed", "skipped", "no_change_needed"], + ), + }, + "required": ["file", "summary", "failure_scenario"], + "additionalProperties": False, + }, + ), + }, + ("findings",), + ), + }, + { + "name": "ScheduleWakeup", + "description": "Schedule when to resume work in /loop dynamic mode — the user invoked /loop without an interval, asking you to self-pace", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "delaySeconds": field("Seconds from now to wake up.", type="number"), + "reason": field("One short sentence explaining the chosen delay.", type="string"), + "prompt": field("The /loop input to fire on wake-up.", type="string"), + "stop": field( + "Set to true to end the dynamic loop immediately instead of scheduling another wakeup.", + type="boolean", + ), + "noop": field( + "true = nothing changed (you checked and there is nothing to report).", + type="boolean", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "SendMessage", + "description": "# SendMessage\n\nSend a message to another agent.", + "input_schema": schema( + { + "to": field( + 'Recipient: a name from ListAgents (append its " [ref]" only when a listing or an error shows one), a teammate name, "mai', + type="string", + allOf=[{"pattern": "^[^\\n\\r]*$"}, {"pattern": "^[\\s\\S]{0,300}$"}], + ), + "summary": field( + "A 5-10 word label for your own transcript row (not transmitted — the recipient previews the first line of `message`).", + type="string", + maxLength=200, + ), + "message": field("Plain text message content.", default="", type="string"), + "notify_when_idle": field( + "Ask a session ON THIS MACHINE to send you ONE notice when it next goes idle (finishes its turn with nothing queued) or e", + type="boolean", + ), + }, + ("to", "message"), + ), + }, + { + "name": "Skill", + "description": "Invoke a skill.", + "input_schema": schema( + { + "skill": field("The name of a skill from the available-skills list.", type="string"), + "args": field("Optional arguments for the skill", type="string"), + }, + ("skill",), + ), + }, + { + "name": "TaskCreate", + "description": "Use this tool to create a structured task list for your current coding session.", + "input_schema": schema( + { + "subject": field("A brief title for the task", type="string"), + "description": field("What needs to be done", type="string"), + "activeForm": field( + 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")', + type="string", + ), + "metadata": field( + "Arbitrary metadata to attach to the task", + type="object", + propertyNames={"type": "string"}, + additionalProperties={}, + ), + }, + ("subject", "description"), + ), + }, + { + "name": "TaskGet", + "description": "Use this tool to retrieve a task by its ID from the task list.", + "input_schema": schema( + { + "taskId": field("The ID of the task to retrieve", type="string"), + }, + ("taskId",), + ), + }, + { + "name": "TaskList", + "description": "Use this tool to list all tasks in the task list.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + }, + { + "name": "TaskStop", + "description": "- Stops a running background task by its ID\n- Takes a task_id parameter identifying the task to stop\n- To stop an agent-", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "task_id": field("The ID of the background task to stop.", type="string"), + "shell_id": field("Deprecated: use task_id instead", type="string"), + }, + "additionalProperties": False, + }, + }, + { + "name": "TaskUpdate", + "description": "Use this tool to update a task in the task list.", + "input_schema": schema( + { + "taskId": field("The ID of the task to update", type="string"), + "subject": field("New subject for the task", type="string"), + "description": field("New description for the task", type="string"), + "activeForm": field( + 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")', + type="string", + ), + "status": field( + "New status for the task", + anyOf=[ + {"type": "string", "enum": ["pending", "in_progress", "completed"]}, + {"type": "string", "const": "deleted"}, + ], + ), + "addBlocks": field("Task IDs that this task blocks", type="array", items={"type": "string"}), + "addBlockedBy": field("Task IDs that block this task", type="array", items={"type": "string"}), + "owner": field("New owner for the task", type="string"), + "metadata": field( + "Metadata keys to merge into the task.", + type="object", + propertyNames={"type": "string"}, + additionalProperties={}, + ), + }, + ("taskId",), + ), + }, + { + "name": "WebFetch", + "description": "IMPORTANT: WebFetch WILL FAIL for authenticated or private URLs.", + "input_schema": schema( + { + "url": field("The URL to fetch content from", type="string", format="uri"), + "prompt": field("The prompt to run on the fetched content", type="string"), + }, + ("url", "prompt"), + ), + }, + { + "name": "WebSearch", + "description": "- Allows Claude to search the web and use the results to inform responses\n- Provides up-to-date information for current ", + "input_schema": schema( + { + "query": field("The search query to use", type="string", minLength=2), + "allowed_domains": field("Only include search results from these domains", type="array", items={"type": "string"}), + "blocked_domains": field("Never include search results from these domains", type="array", items={"type": "string"}), + }, + ("query",), + ), + }, + { + "name": "Workflow", + "description": "Execute a workflow script that orchestrates multiple subagents deterministically.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "script": field("Self-contained workflow script.", type="string", maxLength=524288), + "name": field("Name of a predefined workflow (built-in or from .claude/workflows/).", type="string"), + "description": field("Ignored — set the workflow description in the script's `meta` block.", type="string"), + "title": field("Ignored — set the workflow title in the script's `meta` block.", type="string"), + "args": field("Optional input value exposed to the script as the global `args`, verbatim."), + "scriptPath": field("Path to a workflow script file on disk.", type="string"), + "resumeFromRunId": field( + "Run ID of a prior Workflow invocation to resume from.", + type="string", + pattern="^wf_[a-z0-9-]{6,}$", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "Write", + "description": "Writes a file to the local filesystem.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to write (must be absolute, not relative)", type="string"), + "content": field("The content to write to the file", type="string"), + }, + ("file_path", "content"), + ), + }, ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py index 6391912d781..5eaf96cd696 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py @@ -19,8 +19,11 @@ def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back assert request.target == "/v1/messages", request.target assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY assert request.headers["anthropic-version"] == "2023-06-01" - upstream_beta: Final = frozenset(request.headers.get("anthropic-beta", "").split(",")) - assert cli_beta <= upstream_beta, request.headers.get("anthropic-beta") + assert frozenset(request.headers.get("anthropic-beta", "").split(",")) == cli_beta, request.headers.get( + "anthropic-beta" + ) + assert "authorization" not in request.headers, dict(request.headers) + assert all(gateway.key not in value for value in request.headers.values()), dict(request.headers) body: Final = cc.JSON_OBJECT.validate_json(request.body) expected: Final = {**request_body, "model": _MODEL} assert body == expected, { From b93c26a78959c21a4ce2b7e4eb623f560f1e9d37 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 00:20:20 +0000 Subject: [PATCH 11/19] test(anthropic): drop responses bridge Claude Code tests to keep this PR Anthropic direct only Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../messages_endpoint/_claude_code.py | 113 --------- ...test_claude_code_compaction_bridge_wire.py | 107 --------- .../test_claude_code_document_bridge_wire.py | 73 ------ .../test_claude_code_frontier_bridge_wire.py | 149 ------------ .../test_claude_code_image_bridge_wire.py | 112 --------- ...est_claude_code_interleaved_bridge_wire.py | 128 ---------- ...st_claude_code_model_switch_bridge_wire.py | 95 -------- .../test_claude_code_tool_loop_bridge_wire.py | 223 ------------------ ...test_claude_code_web_search_bridge_wire.py | 95 -------- 9 files changed, 1095 deletions(-) delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py delete mode 100644 tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py diff --git a/tests/integration/messages_endpoint/_claude_code.py b/tests/integration/messages_endpoint/_claude_code.py index aea893780ab..323312186c1 100644 --- a/tests/integration/messages_endpoint/_claude_code.py +++ b/tests/integration/messages_endpoint/_claude_code.py @@ -8,8 +8,6 @@ from pydantic import JsonValue, TypeAdapter JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) ANTHROPIC_API_KEY: Final = "synthetic-anthropic-key" -OPENAI_API_KEY: Final = "synthetic-openai-key" -OPENAI_BACKEND: Final = "gpt-5.4-mini" SONNET: Final = "claude-sonnet-4-5" FABLE: Final = "claude-fable-5-1" OPUS: Final = "claude-opus-5-5" @@ -772,114 +770,3 @@ def tool_use_stream( sse_frame("message_stop", {"type": "message_stop"}), ] return tuple(frames) - - -def responses_completed( - identity: str, - model: str, - output_items: tuple[dict[str, JsonValue], ...], - usage: dict[str, int], - status: str = "completed", - incomplete_details: JsonValue = None, -) -> bytes: - return json.dumps( - { - "id": f"resp_{identity}", - "object": "response", - "created_at": 1789788253, - "status": status, - "incomplete_details": incomplete_details, - "model": model, - "output": list(output_items), - "usage": usage, - } - ).encode() - - -def responses_stream(identity: str, model: str, output_items: tuple[dict[str, JsonValue], ...]) -> tuple[bytes, ...]: - frames: list[bytes] = [ - sse_frame( - "response.created", - { - "type": "response.created", - "response": { - "id": f"resp_{identity}", - "object": "response", - "status": "in_progress", - "model": model, - "output": [], - }, - }, - ) - ] - for index, item in enumerate(output_items): - item_id: Final = str(item.get("id", f"item_{index}")) - frames.append( - sse_frame( - "response.output_item.added", - { - "type": "response.output_item.added", - "output_index": index, - "item": {**item, "content": []} if item.get("type") == "message" else item, - }, - ) - ) - if item.get("type") == "message": - text: Final = "".join(part.get("text", "") for part in item.get("content", ()) if isinstance(part, dict)) - frames.append( - sse_frame( - "response.output_text.delta", - {"type": "response.output_text.delta", "output_index": index, "item_id": item_id, "delta": text}, - ) - ) - if item.get("type") == "reasoning": - summary_text: Final = "".join( - str(part.get("text", "")) for part in item.get("summary", ()) if isinstance(part, dict) - ) - if summary_text: - frames.append( - sse_frame( - "response.reasoning_summary_text.delta", - { - "type": "response.reasoning_summary_text.delta", - "output_index": index, - "item_id": item_id, - "delta": summary_text, - }, - ) - ) - if item.get("type") == "function_call": - frames.append( - sse_frame( - "response.function_call_arguments.delta", - { - "type": "response.function_call_arguments.delta", - "output_index": index, - "item_id": item_id, - "delta": item.get("arguments", ""), - }, - ) - ) - frames.append( - sse_frame( - "response.output_item.done", - {"type": "response.output_item.done", "output_index": index, "item": item}, - ) - ) - frames.append( - sse_frame( - "response.completed", - { - "type": "response.completed", - "response": { - "id": f"resp_{identity}", - "object": "response", - "status": "completed", - "model": model, - "output": list(output_items), - "usage": {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, - }, - }, - ) - ) - return tuple(frames) diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py deleted file mode 100644 index 2e1d4d431ff..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_compaction_bridge_wire.py +++ /dev/null @@ -1,107 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - - -def test_compact_edit_maps_to_responses_context_management(gateway: Gateway) -> None: - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "context_management": { - "edits": [ - {"type": "clear_thinking_20251015", "keep": "all"}, - {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, - ] - }, - "stream": False, - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - assert body.get("context_management") == [{"type": "compaction", "compact_threshold": 150000}], body.get( - "context_management" - ) - return Reply( - body=cc.responses_completed( - "cm", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "OK", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["content"] == [{"type": "text", "text": "OK"}], payload["content"] - assert len(wire.drain()) == 1 - - -def test_compaction_output_item_reaches_client_as_compaction_block(gateway: Gateway) -> None: - pytest.skip( - "BUG: the responses bridge drops compaction output items in translate_response " - "(transformation.py handles only message/reasoning/function_call), so the client loses " - "the compaction block entirely" - ) - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "context_management": { - "edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}] - }, - "stream": False, - } - - def respond(request: Request) -> Reply: - return Reply( - body=cc.responses_completed( - "cm", - cc.OPENAI_BACKEND, - ( - {"type": "compaction", "content": ""}, - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "OK", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - compaction_blocks: Final = [block for block in payload["content"] if block.get("type") == "compaction"] - assert compaction_blocks == [{"type": "compaction", "content": ""}], payload["content"] - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py deleted file mode 100644 index 121d007858e..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_document_bridge_wire.py +++ /dev/null @@ -1,73 +0,0 @@ -import base64 -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc - -_PDF_BYTES: Final = ( - b"%PDF-1.1\n" - b"1 0 obj<>endobj\n" - b"2 0 obj<>endobj\n" - b"3 0 obj<>endobj\n" - b"trailer<>\n%%EOF" -) -_PDF_B64: Final = base64.b64encode(_PDF_BYTES).decode() - - -def test_pdf_document_block_maps_to_input_file_on_bridge(gateway: Gateway) -> None: - request_body: Final = {**cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), "stream": False} - doc_text: Final = f"What is on page one? {uuid.uuid4().hex}" - request_body["messages"] = [ - { - "role": "user", - "content": [ - { - "type": "document", - "source": {"type": "base64", "data": _PDF_B64, "media_type": "application/pdf"}, - "title": "dot.pdf", - }, - {"type": "text", "text": doc_text}, - ], - } - ] - - def respond(request: Request) -> Reply: - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - user_msg: Final = body["input"][0] - assert user_msg["content"][0] == { - "type": "input_file", - "filename": "dot.pdf", - "file_data": f"data:application/pdf;base64,{_PDF_B64}", - }, user_msg - assert user_msg["content"][1] == {"type": "input_text", "text": doc_text}, user_msg - return Reply( - body=cc.responses_completed( - "doc", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "Page one.", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py deleted file mode 100644 index 355004c1f16..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_frontier_bridge_wire.py +++ /dev/null @@ -1,149 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - -_INSTRUCTIONS: Final = "\n".join(block["text"] for block in cc.system_blocks()) -_OUTPUT_ITEMS: Final = ( - { - "type": "reasoning", - "id": "rs_1", - "summary": [{"type": "summary_text", "text": "short plan"}], - "encrypted_content": "enc_1", - }, - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PONG", "annotations": []}], - }, -) - - -def _expected_responses_body(request_body: dict[str, JsonValue], effort: str) -> dict[str, JsonValue]: - user_blocks: Final = request_body["messages"][0]["content"] - expected_tools: Final = tuple( - { - "type": "function", - "name": tool["name"], - "strict": False, - "description": tool["description"], - "parameters": tool["input_schema"], - } - for tool in request_body["tools"] - ) - input_items: Final = [ - { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": block["text"]} for block in user_blocks], - } - ] - for message in request_body["messages"][1:]: - input_items.append( - { - "type": "message", - "role": "system", - "content": [ - {"type": "input_text", "text": block["text"]} - for block in message["content"] - if block.get("type") == "text" - ], - } - ) - return { - "model": cc.OPENAI_BACKEND, - "input": input_items, - "include": ["reasoning.encrypted_content"], - "instructions": _INSTRUCTIONS, - "max_output_tokens": request_body["max_tokens"], - "tools": list(expected_tools), - "reasoning": {"effort": effort}, - "stream": True, - "user": cc.METADATA_USER_ID[:64], - "prompt_cache_key": "00000000-0000-4000-8000-000000000000", - } - - -def _assert_client_events(text: str) -> None: - events: Final = cc.sse_events(text) - assert [event for event, _ in events] == [ - "message_start", - "content_block_start", - "content_block_delta", - "content_block_delta", - "content_block_stop", - "content_block_start", - "content_block_delta", - "content_block_stop", - "message_delta", - "message_stop", - ], [event for event, _ in events] - assert events[1][1]["content_block"]["type"] == "thinking" - assert events[2][1]["delta"] == {"type": "thinking_delta", "thinking": "short plan"} - assert events[3][1]["delta"]["type"] == "signature_delta" - assert events[6][1]["delta"] == {"type": "text_delta", "text": "PONG"} - assert events[8][1]["delta"]["stop_reason"] == "end_turn" - - -def test_claude_code_frontier_body_becomes_reasoning_effort_on_responses_bridge(gateway: Gateway) -> None: - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - assert request.headers["authorization"] == f"Bearer {cc.OPENAI_API_KEY}" - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = _expected_responses_body(request_body, "high") - assert body == expected, { - key: {"expected": expected.get(key), "upstream": body.get(key)} - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply( - content_type="text/event-stream", chunks=cc.responses_stream("bridge1", cc.OPENAI_BACKEND, _OUTPUT_ITEMS) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - _assert_client_events(response.text) - assert len(wire.drain()) == 1 - - -def test_claude_code_legacy_thinking_budget_maps_to_reasoning_effort_on_bridge(gateway: Gateway) -> None: - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - assert body["reasoning"] == {"effort": "high"}, body.get("reasoning") - assert body["model"] == cc.OPENAI_BACKEND - return Reply( - content_type="text/event-stream", chunks=cc.responses_stream("bridge2", cc.OPENAI_BACKEND, _OUTPUT_ITEMS) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - _assert_client_events(response.text) - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py deleted file mode 100644 index 8abdde6f2cd..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_image_bridge_wire.py +++ /dev/null @@ -1,112 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - -_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" -_IMAGE_BLOCK: Final = { - "type": "image", - "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, -} -_DATA_URL: Final = f"data:image/png;base64,{_PNG_B64}" - - -def _respond_ok(request: Request) -> Reply: - return Reply( - body=cc.responses_completed( - "img", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "RED", "annotations": []}], - }, - ), - {"input_tokens": 41, "output_tokens": 3, "total_tokens": 44}, - ) - ) - - -def test_tool_result_image_maps_to_input_image_on_bridge(gateway: Gateway) -> None: - turn1: Final = { - **cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", - ), - "stream": False, - } - turn2: Final = cc.tool_loop_turn2( - turn1, - ({"type": "tool_use", "id": "call_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), - (("call_img", [dict(_IMAGE_BLOCK)]),), - ) - - def respond(request: Request) -> Reply: - body: Final = cc.JSON_OBJECT.validate_json(request.body) - outputs: Final = [ - item for item in body["input"] if isinstance(item, dict) and item.get("type") == "function_call_output" - ] - assert len(outputs) == 1 and outputs[0]["call_id"] == "call_img", outputs - image_messages: Final = [ - item - for item in body["input"] - if isinstance(item, dict) - and item.get("type") == "message" - and item.get("role") == "user" - and any(isinstance(part, dict) and part.get("type") == "input_image" for part in item.get("content", ())) - ] - assert image_messages, body["input"] - image_parts: Final = [ - part - for part in image_messages[0]["content"] - if isinstance(part, dict) and part.get("type") == "input_image" - ] - assert image_parts[0]["image_url"] == _DATA_URL, image_parts - return _respond_ok(request) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 - - -def test_pasted_image_maps_to_input_image_on_bridge(gateway: Gateway) -> None: - request_body: Final = {**cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), "stream": False} - pasted_text: Final = f"What colour is this? {uuid.uuid4().hex}" - request_body["messages"] = [ - {"role": "user", "content": [dict(_IMAGE_BLOCK), {"type": "text", "text": pasted_text}]} - ] - - def respond(request: Request) -> Reply: - body: Final = cc.JSON_OBJECT.validate_json(request.body) - user_msg: Final = body["input"][0] - assert user_msg["content"][0] == {"type": "input_image", "image_url": _DATA_URL}, user_msg - assert user_msg["content"][1] == {"type": "input_text", "text": pasted_text}, user_msg - return _respond_ok(request) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py deleted file mode 100644 index 4944c399a09..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_interleaved_bridge_wire.py +++ /dev/null @@ -1,128 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - - -def _expected_input(turn1: dict[str, JsonValue]) -> list[JsonValue]: - return [ - { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][0]["content"]], - }, - { - "type": "message", - "role": "system", - "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][1]["content"]], - }, - {"type": "reasoning", "summary": [{"type": "summary_text", "text": "plan"}]}, - { - "type": "function_call", - "call_id": "toolu_a", - "name": "Read", - "arguments": '{"file_path": "/tmp/cc_probe/a.txt"}', - }, - {"type": "function_call_output", "call_id": "toolu_a", "output": "ALPHA"}, - { - "type": "message", - "role": "system", - "content": [ - {"type": "input_text", "text": "14999970 tokens left"}, - { - "type": "input_text", - "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", - }, - ], - }, - {"type": "reasoning", "summary": [{"type": "summary_text", "text": "got A"}]}, - { - "type": "function_call", - "call_id": "toolu_b", - "name": "Read", - "arguments": '{"file_path": "/tmp/cc_probe/b.txt"}', - }, - { - "type": "message", - "role": "assistant", - "content": [{"type": "output_text", "text": "got A"}], - }, - {"type": "function_call_output", "call_id": "toolu_b", "output": "BRAVO"}, - { - "type": "message", - "role": "system", - "content": [ - {"type": "input_text", "text": "14999970 tokens left"}, - { - "type": "input_text", - "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", - }, - ], - }, - ] - - -def test_bridge_replays_interleaved_history_in_order(gateway: Gateway) -> None: - turn1: Final = { - **cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time", - ), - "stream": False, - } - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": "plan", "signature": "sig_anthropic_1"}, - {"type": "tool_use", "id": "toolu_a", "name": "Read", "input": {"file_path": "/tmp/cc_probe/a.txt"}}, - ), - (("toolu_a", "ALPHA"),), - ) - turn3: Final = cc.tool_loop_turn2( - turn2, - ( - {"type": "thinking", "thinking": "got A", "signature": "sig_anthropic_2"}, - {"type": "text", "text": "got A"}, - {"type": "tool_use", "id": "toolu_b", "name": "Read", "input": {"file_path": "/tmp/cc_probe/b.txt"}}, - ), - (("toolu_b", "BRAVO"),), - ) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - assert body["input"] == _expected_input(turn1), body["input"] - return Reply( - body=cc.responses_completed( - "il", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "ALPHA BRAVO", "annotations": []}], - }, - ), - {"input_tokens": 50, "output_tokens": 4, "total_tokens": 54}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn3, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py deleted file mode 100644 index 67f9a41300d..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_model_switch_bridge_wire.py +++ /dev/null @@ -1,95 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - - -def test_anthropic_signed_thinking_in_history_crosses_to_responses_bridge(gateway: Gateway) -> None: - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - bridge_input_box: list[JsonValue] = [] - - def respond_anthropic(request: Request) -> Reply: - assert request.target == "/v1/messages", request.target - return Reply( - content_type="text/event-stream", - chunks=cc.tool_use_stream( - f"msg_sw_{uuid.uuid4().hex}", - cc.FABLE, - "need to read the file", - "sig_anthropic_1", - (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), - {"input_tokens": 20, "output_tokens": 10}, - ), - ) - - def respond_openai(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - bridge_input_box.append(body.get("input")) - reasoning_items: Final = [ - item for item in body["input"] if isinstance(item, dict) and item.get("type") == "reasoning" - ] - assert reasoning_items == [ - {"type": "reasoning", "summary": [{"type": "summary_text", "text": "need to read the file"}]} - ], reasoning_items - return Reply( - content_type="text/event-stream", - chunks=cc.responses_stream( - "sw", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PROBE", "annotations": []}], - }, - ), - ), - ) - - with ( - wire_server(respond_anthropic) as wire_a, - wire_server(respond_openai) as wire_b, - gateway.scenario() as scenario, - ): - fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire_a.url, api_key=cc.ANTHROPIC_API_KEY) - openai_alias: Final = scenario.model( - model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire_b.url, api_key=cc.OPENAI_API_KEY - ) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": openai_alias}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - events: Final = cc.sse_events(response2.text) - assert events[-1][0] == "message_stop" - assert len(wire_a.drain()) == 1 - assert len(wire_b.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py deleted file mode 100644 index fa33c98048a..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_tool_loop_bridge_wire.py +++ /dev/null @@ -1,223 +0,0 @@ -import json -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - -_INSTRUCTIONS: Final = "\n".join(block["text"] for block in cc.system_blocks()) - - -def _turn1() -> dict[str, JsonValue]: - return { - **cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ), - "stream": False, - } - - -def _upstream_items(calls: tuple[tuple[str, str, JsonValue], ...]) -> tuple[dict[str, JsonValue], ...]: - return ( - { - "type": "reasoning", - "id": "rs_1", - "summary": [{"type": "summary_text", "text": "short plan"}], - "encrypted_content": "enc_1", - }, - *( - { - "type": "function_call", - "id": f"fc_{i}", - "call_id": call_id, - "name": name, - "arguments": json.dumps(tool_input), - "status": "completed", - } - for i, (call_id, name, tool_input) in enumerate(calls, start=1) - ), - ) - - -def _expected_turn2_input( - turn1: dict[str, JsonValue], - assistant_content: tuple[dict[str, JsonValue], ...], - tool_results: tuple[tuple[str, JsonValue], ...], -) -> list[JsonValue]: - items: Final = [ - { - "type": "message", - "role": "user", - "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][0]["content"]], - }, - { - "type": "message", - "role": "system", - "content": [{"type": "input_text", "text": block["text"]} for block in turn1["messages"][1]["content"]], - }, - { - "type": "reasoning", - "summary": [{"type": "summary_text", "text": "short plan"}], - "encrypted_content": "enc_1", - }, - ] - items += [ - { - "type": "function_call", - "call_id": block["id"], - "name": block["name"], - "arguments": json.dumps(block["input"]), - } - for block in assistant_content - if block.get("type") == "tool_use" - ] - items += [ - {"type": "function_call_output", "call_id": tool_use_id, "output": content} - for tool_use_id, content in tool_results - ] - items.append( - { - "type": "message", - "role": "system", - "content": [ - {"type": "input_text", "text": "14999970 tokens left"}, - { - "type": "input_text", - "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", - }, - ], - } - ) - return items - - -def test_responses_bridge_replays_reasoning_and_tool_call_on_turn_two(gateway: Gateway) -> None: - turn1: Final = _turn1() - calls: Final = (("call_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) - tool_results: Final = (("call_1", "1\tPROBE\n2\t"),) - seen: list[dict[str, JsonValue]] = [] - signature_box: list[str] = [] - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - if len(seen) == 1: - return Reply( - body=cc.responses_completed( - "tl1", - cc.OPENAI_BACKEND, - _upstream_items(calls), - {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, - ) - ) - assistant_content: Final = ( - {"type": "thinking", "thinking": "short plan", "signature": signature_box[0]}, - {"type": "tool_use", "id": "call_1", "name": "Read", "input": {"file_path": "/tmp/cc_probe/hello.txt"}}, - ) - assert body["input"] == _expected_turn2_input(turn1, assistant_content, tool_results), body["input"] - return Reply( - body=cc.responses_completed( - "tl2", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PROBE", "annotations": []}], - }, - ), - {"input_tokens": 50, "output_tokens": 3, "total_tokens": 53}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - payload: Final = cc.JSON_OBJECT.validate_json(response1.content) - assert payload["stop_reason"] == "tool_use", payload - assert payload["content"][0]["type"] == "thinking", payload["content"] - signature: Final = payload["content"][0].get("signature") - assert signature, payload["content"][0] - signature_box.append(signature) - assert payload["content"][1] == { - "type": "tool_use", - "id": "call_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, payload["content"] - assistant_content: Final = ( - {"type": "thinking", "thinking": "short plan", "signature": signature}, - {"type": "tool_use", "id": "call_1", "name": "Read", "input": {"file_path": "/tmp/cc_probe/hello.txt"}}, - ) - turn2: Final = cc.tool_loop_turn2(turn1, assistant_content, tool_results) - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - payload2: Final = cc.JSON_OBJECT.validate_json(response2.content) - assert payload2["stop_reason"] == "end_turn", payload2 - assert len(wire.drain()) == 2 - - -def test_responses_bridge_replays_parallel_tool_calls_in_order(gateway: Gateway) -> None: - turn1: Final = _turn1() - calls: Final = ( - ("call_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}), - ("call_2", "Read", {"file_path": "/tmp/cc_probe/world.txt"}), - ) - - def respond(request: Request) -> Reply: - body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen_items: Final = body["input"] - outputs: Final = [ - item for item in seen_items if isinstance(item, dict) and item.get("type") == "function_call_output" - ] - assert [item["call_id"] for item in outputs] == ["call_1", "call_2"], outputs - return Reply( - body=cc.responses_completed( - "mt", - cc.OPENAI_BACKEND, - ( - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "PROBE PROBE2", "annotations": []}], - }, - ), - {"input_tokens": 50, "output_tokens": 4, "total_tokens": 54}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - assistant_content: Final = tuple( - {"type": "tool_use", "id": call_id, "name": name, "input": tool_input} - for call_id, name, tool_input in calls - ) - turn2: Final = cc.tool_loop_turn2( - turn1, assistant_content, (("call_1", "1\tPROBE\n2\t"), ("call_2", "1\tPROBE2\n2\t")) - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py b/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py deleted file mode 100644 index e06e1b8d403..00000000000 --- a/tests/integration/messages_endpoint/responses_bridge/test_claude_code_web_search_bridge_wire.py +++ /dev/null @@ -1,95 +0,0 @@ -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue - -_WEB_SEARCH_TOOL: Final = { - "name": "WebSearch", - "description": "Search the web. Returns result blocks with titles and URLs.", - "input_schema": cc.schema( - { - "query": cc.field("The search query to use", type="string", minLength=2), - "allowed_domains": cc.field( - "Only include search results from these domains", type="array", items={"type": "string"} - ), - "blocked_domains": cc.field( - "Never include search results from these domains", type="array", items={"type": "string"} - ), - }, - ("query",), - ), -} - - -def test_web_search_tool_and_cited_output_on_responses_bridge(gateway: Gateway) -> None: - request_body: Final = { - **cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Use web search to find the current LiteLLM version and answer in one word", - ), - "stream": False, - } - request_body["tools"] = [*request_body["tools"], _WEB_SEARCH_TOOL] - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/responses", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - tools: Final = body["tools"] - assert tools[-1] == { - "type": "function", - "name": "WebSearch", - "strict": False, - "description": _WEB_SEARCH_TOOL["description"], - "parameters": _WEB_SEARCH_TOOL["input_schema"], - }, tools[-1] - return Reply( - body=cc.responses_completed( - "ws", - cc.OPENAI_BACKEND, - ( - {"type": "web_search_call", "id": "ws_1", "status": "completed"}, - { - "type": "message", - "id": "msg_1", - "status": "completed", - "role": "assistant", - "content": [ - { - "type": "output_text", - "text": "1.104.0", - "annotations": [ - { - "type": "url_citation", - "url": "https://example.com/litellm", - "title": "litellm releases", - "start_index": 0, - "end_index": 7, - } - ], - } - ], - }, - ), - {"input_tokens": 41, "output_tokens": 5, "total_tokens": 46}, - ) - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"openai/{cc.OPENAI_BACKEND}", api_base=wire.url, api_key=cc.OPENAI_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - payload: Final = cc.JSON_OBJECT.validate_json(response.content) - assert payload["content"] == [{"type": "text", "text": "1.104.0"}], payload["content"] - assert len(wire.drain()) == 1 From 337d5cc482121e25a00e1cd6317fc5cde57015f1 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 00:41:06 +0000 Subject: [PATCH 12/19] test(anthropic): fix duplicate WebSearch tool, drop mutation in stream builders, ignore pings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../messages_endpoint/_claude_code.py | 83 +++++++++++-------- .../test_claude_code_compaction_wire.py | 18 ++-- .../test_claude_code_document_input_wire.py | 22 ++--- .../test_claude_code_fallback_wire.py | 46 +++++----- .../test_claude_code_image_input_wire.py | 34 ++++---- ...t_claude_code_interleaved_thinking_wire.py | 9 +- .../test_claude_code_model_switch_wire.py | 17 ++-- .../test_claude_code_tool_loop_wire.py | 9 +- .../test_claude_code_web_search_wire.py | 34 +++----- 9 files changed, 134 insertions(+), 138 deletions(-) diff --git a/tests/integration/messages_endpoint/_claude_code.py b/tests/integration/messages_endpoint/_claude_code.py index 323312186c1..53c439fbc82 100644 --- a/tests/integration/messages_endpoint/_claude_code.py +++ b/tests/integration/messages_endpoint/_claude_code.py @@ -2,6 +2,7 @@ import json from collections.abc import Mapping +from itertools import chain from typing import Final from pydantic import JsonValue, TypeAdapter @@ -640,10 +641,12 @@ def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) return tuple( ( - next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), + event, json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), ) for frame in frames + if (event := next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: "))) + != "ping" ) @@ -690,6 +693,37 @@ def text_stream(identity: str, model: str, text: str, usage: dict[str, int]) -> ) +def _tool_use_frames(index: int, tool_id: str, name: str, tool_input: JsonValue) -> tuple[bytes, ...]: + arguments: Final = json.dumps(tool_input) + return ( + sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": index, + "content_block": {"type": "tool_use", "id": tool_id, "name": name, "input": {}}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[: len(arguments) // 2]}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[len(arguments) // 2 :]}, + }, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}), + ) + + def tool_use_stream( identity: str, model: str, @@ -698,7 +732,7 @@ def tool_use_stream( tool_calls: tuple[tuple[str, str, JsonValue], ...], usage: dict[str, int], ) -> tuple[bytes, ...]: - frames: list[bytes] = [ + head: Final = ( sse_frame( "message_start", { @@ -728,37 +762,8 @@ def tool_use_stream( {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": signature}}, ), sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - ] - for index, (tool_id, name, tool_input) in enumerate(tool_calls, start=1): - arguments: Final = json.dumps(tool_input) - frames += [ - sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": index, - "content_block": {"type": "tool_use", "id": tool_id, "name": name, "input": {}}, - }, - ), - sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": index, - "delta": {"type": "input_json_delta", "partial_json": arguments[: len(arguments) // 2]}, - }, - ), - sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": index, - "delta": {"type": "input_json_delta", "partial_json": arguments[len(arguments) // 2 :]}, - }, - ), - sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}), - ] - frames += [ + ) + tail: Final = ( sse_frame( "message_delta", { @@ -768,5 +773,13 @@ def tool_use_stream( }, ), sse_frame("message_stop", {"type": "message_stop"}), - ] - return tuple(frames) + ) + frames: Final = ( + *head, + *chain.from_iterable( + _tool_use_frames(index, tool_id, name, tool_input) + for index, (tool_id, name, tool_input) in enumerate(tool_calls, start=1) + ), + *tail, + ) + return frames diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py index a1e903ab5c6..0132cf7a1e6 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py @@ -67,22 +67,22 @@ def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gat ), (), ) - seen: list[dict[str, JsonValue]] = [] + first_expected: Final = {**request_body, "model": cc.FABLE} + second_expected: Final = {**turn3, "model": cc.FABLE} def respond(request: Request) -> Reply: assert request.method == "POST" assert request.target == "/v1/messages", request.target body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - expected: Final = {**(turn3 if len(seen) == 2 else request_body), "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) + if body == first_expected: + assert body["context_management"] == _CONTEXT_MANAGEMENT + return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) + assert body == second_expected, { + key: (second_expected.get(key), body.get(key)) + for key in second_expected.keys() | body.keys() + if second_expected.get(key) != body.get(key) } assert body["context_management"] == _CONTEXT_MANAGEMENT - if len(seen) == 1: - return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) return Reply( content_type="text/event-stream", chunks=cc.text_stream("msg_cm_next", cc.FABLE, "OK", {"input_tokens": 20, "output_tokens": 2}), diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py index 7fa67549372..d4a827d7ece 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py @@ -79,16 +79,18 @@ def _cited_stream(identity: str) -> tuple[bytes, ...]: def test_base64_pdf_document_with_citations_reaches_anthropic_identical(gateway: Gateway) -> None: - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - request_body["messages"] = [ - { - "role": "user", - "content": [ - dict(_DOC_BLOCK), - {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, - ], - } - ] + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "messages": [ + { + "role": "user", + "content": [ + dict(_DOC_BLOCK), + {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, + ], + } + ], + } def respond(request: Request) -> Reply: assert request.method == "POST" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py index 0075acb6da8..830f84c858c 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py @@ -33,31 +33,33 @@ def test_anthropic_overloaded_primary_falls_back_to_second_deployment(gateway: G wire_server(lambda request: _error_529()) as primary, wire_server(respond_fallback) as fallback, ): - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["model_list"] = [ - { - "model_name": "cc-primary", - "litellm_params": { - "model": f"anthropic/{cc.SONNET}", - "api_key": cc.ANTHROPIC_API_KEY, - "api_base": primary.url, - "model_info": {"id": "primary-cc"}, + config: Final = { + **yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()), + "model_list": [ + { + "model_name": "cc-primary", + "litellm_params": { + "model": f"anthropic/{cc.SONNET}", + "api_key": cc.ANTHROPIC_API_KEY, + "api_base": primary.url, + "model_info": {"id": "primary-cc"}, + }, }, - }, - { - "model_name": "cc-fallback-group", - "litellm_params": { - "model": f"anthropic/{cc.SONNET}", - "api_key": cc.ANTHROPIC_API_KEY, - "api_base": fallback.url, - "model_info": {"id": "fallback-cc"}, + { + "model_name": "cc-fallback-group", + "litellm_params": { + "model": f"anthropic/{cc.SONNET}", + "api_key": cc.ANTHROPIC_API_KEY, + "api_base": fallback.url, + "model_info": {"id": "fallback-cc"}, + }, }, + ], + "router_settings": { + "num_retries": 0, + "disable_cooldowns": True, + "fallbacks": [{"cc-primary": ["cc-fallback-group"]}], }, - ] - config["router_settings"] = { - "num_retries": 0, - "disable_cooldowns": True, - "fallbacks": [{"cc-primary": ["cc-fallback-group"]}], } path: Final = tmp_path / "fallbacks.yaml" path.write_text(yaml.safe_dump(config)) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py index 0deed7db56a..6d2f4d06dec 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py @@ -25,29 +25,27 @@ def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gate ({"type": "tool_use", "id": "toolu_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), (("toolu_img", [dict(_IMAGE_BLOCK)]),), ) - pasted: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - pasted["messages"] = [ - { - "role": "user", - "content": [ - dict(_IMAGE_BLOCK), - {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, - ], - } - ] - seen: list[dict[str, JsonValue]] = [] + pasted: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "messages": [ + { + "role": "user", + "content": [ + dict(_IMAGE_BLOCK), + {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, + ], + } + ], + } + first_expected: Final = {**with_image_result, "model": cc.FABLE} + second_expected: Final = {**pasted, "model": cc.FABLE} def respond(request: Request) -> Reply: assert request.method == "POST" assert request.target == "/v1/messages", request.target body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - expected: Final = {**(pasted if len(seen) == 2 else with_image_result), "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } + if body != first_expected: + assert body == second_expected, body return Reply( content_type="text/event-stream", chunks=cc.text_stream( diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py index c6a46ca2806..f5da9b070f5 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py @@ -129,7 +129,8 @@ def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identi ) turn2: Final = _turn2(turn1) turn3: Final = _turn3(turn2) - seen: list[dict[str, JsonValue]] = [] + turn2_expected: Final = {**turn2, "model": cc.FABLE} + turn3_expected: Final = {**turn3, "model": cc.FABLE} def respond(request: Request) -> Reply: assert request.method == "POST" @@ -137,14 +138,12 @@ def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identi upstream_beta: Final = request.headers.get("anthropic-beta", "") assert upstream_beta.split(",").count("interleaved-thinking-2025-05-14") == 1, upstream_beta body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - expected: Final = {**turn3, "model": cc.FABLE} if len(seen) == 2 else {**turn2, "model": cc.FABLE} - assert body == expected, _diff(expected, body) - if len(seen) == 1: + if body == turn2_expected: return Reply( content_type="text/event-stream", chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}), ) + assert body == turn3_expected, _diff(turn3_expected, body) return Reply(content_type="text/event-stream", chunks=_interleaved_stream(identity)) with wire_server(respond) as wire, gateway.scenario() as scenario: diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py index ccdd4b5bd06..c04b974da01 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py @@ -30,15 +30,14 @@ def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gatewa ), (("toolu_read_1", "1\tPROBE\n2\t"),), ) - seen: list[dict[str, JsonValue]] = [] + first_expected: Final = {**turn1, "model": cc.FABLE} + second_expected: Final = {**turn2, "model": cc.OPUS} def respond(request: Request) -> Reply: assert request.method == "POST" assert request.target == "/v1/messages", request.target body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - if len(seen) == 1: - assert body == {**turn1, "model": cc.FABLE}, body.get("model") + if body == first_expected: return Reply( content_type="text/event-stream", chunks=cc.tool_use_stream( @@ -50,11 +49,10 @@ def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gatewa {"input_tokens": 20, "output_tokens": 10}, ), ) - expected: Final = {**turn2, "model": cc.OPUS} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) + assert body == second_expected, { + key: (second_expected.get(key), body.get(key)) + for key in second_expected.keys() | body.keys() + if second_expected.get(key) != body.get(key) } return Reply( content_type="text/event-stream", @@ -74,7 +72,6 @@ def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gatewa ) assert response2.status_code == 200, response2.text assert len(wire.drain()) == 2 - assert seen[1]["model"] == cc.OPUS rows: Final = eventually( lambda: read_rows( 'SELECT model FROM "LiteLLM_SpendLogs" WHERE request_id=%s', diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py index f386a206451..0fd9828186d 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py @@ -43,20 +43,19 @@ def test_claude_code_tool_loop_round_trips_thinking_tool_use_and_tool_result(gat ), (("toolu_read_1", "1\tPROBE\n2\t"),), ) - seen: list[dict[str, JsonValue]] = [] + first_expected: Final = {**turn1, "model": cc.FABLE} + second_expected: Final = {**turn2, "model": cc.FABLE} def respond(request: Request) -> Reply: assert request.method == "POST" assert request.target == "/v1/messages", request.target body: Final = cc.JSON_OBJECT.validate_json(request.body) - seen.append(body) - expected: Final = {**turn2, "model": cc.FABLE} if len(seen) == 2 else {**turn1, "model": cc.FABLE} - assert body == expected, _diff(expected, body) - if len(seen) == 1: + if body == first_expected: return Reply( content_type="text/event-stream", chunks=cc.tool_use_stream(identity1, cc.FABLE, _THINKING, _SIGNATURE, calls, _USAGE), ) + assert body == second_expected, _diff(second_expected, body) return Reply( content_type="text/event-stream", chunks=cc.text_stream(identity2, cc.FABLE, "PROBE", {"input_tokens": 30, "output_tokens": 3}), diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py index a275b2e3305..aed0a5abd40 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py @@ -5,24 +5,8 @@ from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server from integration.messages_endpoint import _claude_code as cc -from pydantic import JsonValue -WEB_SEARCH_TOOL: Final = { - "name": "WebSearch", - "description": "Search the web. Returns result blocks with titles and URLs.", - "input_schema": cc.schema( - { - "query": cc.field("The search query to use", type="string", minLength=2), - "allowed_domains": cc.field( - "Only include search results from these domains", type="array", items={"type": "string"} - ), - "blocked_domains": cc.field( - "Never include search results from these domains", type="array", items={"type": "string"} - ), - }, - ("query",), - ), -} +WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} def _web_search_stream(identity: str) -> tuple[bytes, ...]: @@ -121,15 +105,16 @@ def _web_search_stream(identity: str) -> tuple[bytes, ...]: def test_claude_code_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: identity: Final = f"msg_ws_{uuid.uuid4().hex}" + base: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Use web search to find the current LiteLLM version and answer in one word", + ) request_body: Final = { - **cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Use web search to find the current LiteLLM version and answer in one word", - ), + **base, + "tools": [*base["tools"], WEB_SEARCH_TOOL], } - request_body["tools"] = [*request_body["tools"], WEB_SEARCH_TOOL] def respond(request: Request) -> Reply: assert request.method == "POST" @@ -142,6 +127,7 @@ def test_claude_code_web_search_tool_passthrough_and_cited_response(gateway: Gat if expected.get(key) != body.get(key) } assert body["tools"][-1] == WEB_SEARCH_TOOL + assert len({tool["name"] for tool in body["tools"]}) == len(body["tools"]), body["tools"] return Reply(content_type="text/event-stream", chunks=_web_search_stream(identity)) with wire_server(respond) as wire, gateway.scenario() as scenario: From 0ea21ef9e055459605d9a47b39c07d59c09737a2 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 00:56:21 +0000 Subject: [PATCH 13/19] test(anthropic): assert upstream request order in multi-turn Claude Code tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../providers/anthropic/test_claude_code_compaction_wire.py | 3 ++- .../providers/anthropic/test_claude_code_image_input_wire.py | 3 ++- .../anthropic/test_claude_code_interleaved_thinking_wire.py | 3 ++- .../providers/anthropic/test_claude_code_model_switch_wire.py | 3 ++- .../providers/anthropic/test_claude_code_tool_loop_wire.py | 3 ++- 5 files changed, 10 insertions(+), 5 deletions(-) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py index 0132cf7a1e6..4be3a2b9977 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py @@ -105,4 +105,5 @@ def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gat "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers ) assert response2.status_code == 200, response2.text - assert len(wire.drain()) == 2 + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py index 6d2f4d06dec..b77f328403b 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py @@ -64,4 +64,5 @@ def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gate "POST", "/v1/messages", {**pasted, "model": model}, params={"beta": "true"}, headers=headers ) assert response2.status_code == 200, response2.text - assert len(wire.drain()) == 2 + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py index f5da9b070f5..ff8ccf0df4f 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py @@ -163,4 +163,5 @@ def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identi ] assert started == [(0, "thinking"), (1, "text"), (2, "tool_use")], started assert events[-1][0] == "message_stop" - assert len(wire.drain()) == 2 + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (turn2_expected, turn3_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py index c04b974da01..05b936c93f3 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py @@ -71,7 +71,8 @@ def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gatewa "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers ) assert response2.status_code == 200, response2.text - assert len(wire.drain()) == 2 + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies rows: Final = eventually( lambda: read_rows( 'SELECT model FROM "LiteLLM_SpendLogs" WHERE request_id=%s', diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py index 0fd9828186d..39bda48c2c4 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py @@ -101,7 +101,8 @@ def test_claude_code_tool_loop_round_trips_thinking_tool_use_and_tool_result(gat events2: Final = cc.sse_events(response2.text) assert events2[2][1]["delta"] == {"type": "text_delta", "text": "PROBE"} assert events2[4][1]["delta"]["stop_reason"] == "end_turn" - assert len(wire.drain()) == 2 + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies rows: Final = eventually( lambda: read_rows( 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', From 77e3b8552aa41093ad0e7470ae39f939f38932e1 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 22:10:31 +0000 Subject: [PATCH 14/19] test(integration): name Claude Code wire tests by behavior and move provider-agnostic ones to routing and streaming Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/README.md | 2 ++ .../_claude_code.py => _support/claude_code.py} | 2 +- ...e.py => test_anthropic_adaptive_thinking_effort_wire.py} | 6 +++--- ...compaction_wire.py => test_anthropic_compaction_wire.py} | 2 +- ..._input_wire.py => test_anthropic_document_input_wire.py} | 2 +- ...age_input_wire.py => test_anthropic_image_input_wire.py} | 2 +- ... => test_anthropic_interleaved_thinking_history_wire.py} | 2 +- ...eta_wire.py => test_anthropic_long_context_beta_wire.py} | 2 +- ..._wire.py => test_anthropic_model_switch_history_wire.py} | 4 ++-- ..._wire.py => test_anthropic_prompt_cache_billing_wire.py} | 4 ++-- ...tive_wire.py => test_anthropic_request_fidelity_wire.py} | 4 ++-- ...e_tool_loop_wire.py => test_anthropic_tool_loop_wire.py} | 6 +++--- ..._wire.py => test_anthropic_web_search_citations_wire.py} | 4 ++-- .../test_overloaded_deployment_fallback.py} | 4 ++-- .../test_client_disconnect_still_bills.py} | 2 +- 15 files changed, 25 insertions(+), 23 deletions(-) rename tests/integration/{messages_endpoint/_claude_code.py => _support/claude_code.py} (99%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_frontier_wire.py => test_anthropic_adaptive_thinking_effort_wire.py} (94%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_compaction_wire.py => test_anthropic_compaction_wire.py} (98%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_document_input_wire.py => test_anthropic_document_input_wire.py} (98%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_image_input_wire.py => test_anthropic_image_input_wire.py} (97%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_interleaved_thinking_wire.py => test_anthropic_interleaved_thinking_history_wire.py} (99%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_long_context_beta_wire.py => test_anthropic_long_context_beta_wire.py} (97%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_model_switch_wire.py => test_anthropic_model_switch_history_wire.py} (95%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_prompt_cache_wire.py => test_anthropic_prompt_cache_billing_wire.py} (95%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_native_wire.py => test_anthropic_request_fidelity_wire.py} (94%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_tool_loop_wire.py => test_anthropic_tool_loop_wire.py} (96%) rename tests/integration/messages_endpoint/providers/anthropic/{test_claude_code_web_search_wire.py => test_anthropic_web_search_citations_wire.py} (97%) rename tests/integration/{messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py => routing/test_overloaded_deployment_fallback.py} (95%) rename tests/integration/{messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py => streaming/test_client_disconnect_still_bills.py} (98%) diff --git a/tests/integration/README.md b/tests/integration/README.md index 1865366123a..cf4e3be3f02 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -30,6 +30,8 @@ Streaming checks send real HTTP transfer chunks, including one-byte partitions, The `messages_endpoint/` directory holds `/v1/messages` endpoint contracts: native-provider backends under `providers/` (`anthropic`, `bedrock`, `gemini`) and the translation bridges (`responses_bridge`, `chat_bridge`) at the top level. It runs in the providers shard; `run.py` selects test files recursively under each scheduled directory +A provider folder holds only what depends on that provider's wire format: request and response fidelity, multi-turn history, request parameters, content types, and provider-specific pricing. Behavior every provider shares, such as fallback or billing after a client disconnect, lives in the feature directory it exercises (`routing/`, `streaming/`, `spend/`). `_support/claude_code.py` holds a captured Claude Code request and stream builders that any directory can use as a realistic agent payload + The sdk shard exercises the SDK's own HTTP clients against local protocol peers with no gateway in the path, so a case here fails only when the client library or its wire behavior changes. The HTTP/2 case runs a hypercorn TLS peer offering h2 and http/1.1 over ALPN, drives the sync and async httpx handlers at it with `LITELLM_HTTP2` off and on, and asserts the version both the client and the peer observed on the wire. Put a test here only when it needs no proxy, database or Redis; a case that reaches the gateway belongs in one of the other shards The extensions shard uses the built-in generic callback and guardrail transports. It checks callback correlation and credential exclusion, guardrail rewriting and denial, retained OpenAI consumers and A2A wire versions diff --git a/tests/integration/messages_endpoint/_claude_code.py b/tests/integration/_support/claude_code.py similarity index 99% rename from tests/integration/messages_endpoint/_claude_code.py rename to tests/integration/_support/claude_code.py index 53c439fbc82..8f2502952be 100644 --- a/tests/integration/messages_endpoint/_claude_code.py +++ b/tests/integration/_support/claude_code.py @@ -1,4 +1,4 @@ -"""Shared Claude Code-shaped request builders and upstream stream fixtures for the /v1/messages contracts.""" +"""Shared Claude Code-shaped request builders and upstream stream fixtures for integration contracts.""" import json from collections.abc import Mapping diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py similarity index 94% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py index a9a41b96751..d327181a491 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_frontier_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py @@ -5,7 +5,7 @@ import pytest from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue @@ -17,7 +17,7 @@ def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[st } -def test_claude_code_adaptive_thinking_and_effort_reach_anthropic_intact(gateway: Gateway) -> None: +def test_adaptive_thinking_and_effort_reach_anthropic_intact(gateway: Gateway) -> None: identity: Final = f"msg_fable_{uuid.uuid4().hex}" request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) @@ -63,7 +63,7 @@ def test_claude_code_adaptive_thinking_and_effort_reach_anthropic_intact(gateway assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 -def test_claude_code_xhigh_effort_reaches_anthropic_and_charges_by_usage(gateway: Gateway) -> None: +def test_xhigh_effort_reaches_anthropic_and_charges_by_usage(gateway: Gateway) -> None: identity: Final = f"msg_opus_{uuid.uuid4().hex}" request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "xhigh", 128000) cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py similarity index 98% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py index 4be3a2b9977..ce87a76a5f2 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_compaction_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py @@ -3,7 +3,7 @@ from typing import Final from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _CONTEXT_MANAGEMENT: Final = { diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py similarity index 98% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py index d4a827d7ece..29ab1a70449 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_document_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py @@ -4,7 +4,7 @@ from typing import Final from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _PDF_BYTES: Final = ( diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py similarity index 97% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py index b77f328403b..271f064becc 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_image_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py @@ -3,7 +3,7 @@ from typing import Final from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py similarity index 99% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py index ff8ccf0df4f..a1d1a9bcb03 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_interleaved_thinking_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py @@ -3,7 +3,7 @@ from typing import Final from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _TURN1_TOOL: Final = ("toolu_a", "Read", {"file_path": "/tmp/cc_probe/a.txt"}) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py similarity index 97% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py index a198ec8500a..4c4b075fcc1 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_long_context_beta_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py @@ -5,7 +5,7 @@ import pytest from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc _BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py similarity index 95% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py index 05b936c93f3..d70f689ce07 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_model_switch_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py @@ -4,11 +4,11 @@ from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue -def test_claude_code_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: +def test_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: identity1: Final = f"msg_sw1_{uuid.uuid4().hex}" identity2: Final = f"msg_sw2_{uuid.uuid4().hex}" turn1: Final = cc.frontier_request( diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py similarity index 95% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py index 54bbe2a0704..8d4338f4910 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_prompt_cache_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py @@ -5,7 +5,7 @@ import pytest from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _USAGE: Final = { @@ -16,7 +16,7 @@ _USAGE: Final = { } -def test_claude_code_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: +def test_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: identity: Final = f"msg_pc_{uuid.uuid4().hex}" turn1: Final = cc.frontier_request( f"cache-bust-{uuid.uuid4().hex}", diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py similarity index 94% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py index 5eaf96cd696..288787f5765 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_native_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py @@ -4,12 +4,12 @@ from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc _MODEL: Final = cc.SONNET -def test_claude_code_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: +def test_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: identity: Final = f"msg_cc_{uuid.uuid4().hex}" request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") cli_beta: Final = frozenset(cc.CLI_BETA.split(",")) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py similarity index 96% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py index 39bda48c2c4..7d421345ed6 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_tool_loop_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py @@ -4,7 +4,7 @@ from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc from pydantic import JsonValue _THINKING: Final = "need to read the file" @@ -20,7 +20,7 @@ def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[st } -def test_claude_code_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: +def test_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: identity1: Final = f"msg_tl1_{uuid.uuid4().hex}" identity2: Final = f"msg_tl2_{uuid.uuid4().hex}" turn1: Final = cc.frontier_request( @@ -114,7 +114,7 @@ def test_claude_code_tool_loop_round_trips_thinking_tool_use_and_tool_result(gat assert rows[0]["prompt_tokens"] == 30 -def test_claude_code_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: +def test_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: turn1: Final = cc.frontier_request( f"cache-bust-{uuid.uuid4().hex}", "high", diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py similarity index 97% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py index aed0a5abd40..f63126116be 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_web_search_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py @@ -4,7 +4,7 @@ from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} @@ -103,7 +103,7 @@ def _web_search_stream(identity: str) -> tuple[bytes, ...]: ) -def test_claude_code_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: +def test_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: identity: Final = f"msg_ws_{uuid.uuid4().hex}" base: Final = cc.frontier_request( f"cache-bust-{uuid.uuid4().hex}", diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py b/tests/integration/routing/test_overloaded_deployment_fallback.py similarity index 95% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py rename to tests/integration/routing/test_overloaded_deployment_fallback.py index 830f84c858c..d5ca213d3d9 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_fallback_wire.py +++ b/tests/integration/routing/test_overloaded_deployment_fallback.py @@ -9,7 +9,7 @@ from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.process import owned_proxy from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc def _error_529() -> Reply: @@ -19,7 +19,7 @@ def _error_529() -> Reply: ) -def test_anthropic_overloaded_primary_falls_back_to_second_deployment(gateway: Gateway, tmp_path: Path) -> None: +def test_overloaded_primary_falls_back_to_second_deployment(gateway: Gateway, tmp_path: Path) -> None: identity: Final = f"msg_fb_{uuid.uuid4().hex}" request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py b/tests/integration/streaming/test_client_disconnect_still_bills.py similarity index 98% rename from tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py rename to tests/integration/streaming/test_client_disconnect_still_bills.py index 0e7f00943c2..5a80e13cf44 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_claude_code_client_disconnect_wire.py +++ b/tests/integration/streaming/test_client_disconnect_still_bills.py @@ -4,7 +4,7 @@ from typing import Final from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration.messages_endpoint import _claude_code as cc +from integration._support import claude_code as cc def test_client_disconnect_mid_stream_still_bills_the_message(gateway: Gateway) -> None: From ad6ec3faa629701299ef67cd72578d93588cb471 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 22:10:39 +0000 Subject: [PATCH 15/19] test(integration): sort imports after moving the Claude Code fixture into _support Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/_support/claude_code.py | 24 ++++++++++++++----- ...anthropic_adaptive_thinking_effort_wire.py | 2 +- .../test_anthropic_compaction_wire.py | 3 +-- .../test_anthropic_document_input_wire.py | 3 +-- .../test_anthropic_image_input_wire.py | 3 +-- ...ropic_interleaved_thinking_history_wire.py | 2 +- .../test_anthropic_long_context_beta_wire.py | 2 +- ...est_anthropic_model_switch_history_wire.py | 3 +-- ...est_anthropic_prompt_cache_billing_wire.py | 3 +-- .../test_anthropic_request_fidelity_wire.py | 2 +- .../test_anthropic_tool_loop_wire.py | 2 +- ...est_anthropic_web_search_citations_wire.py | 2 +- .../test_overloaded_deployment_fallback.py | 3 +-- .../test_client_disconnect_still_bills.py | 2 +- 14 files changed, 31 insertions(+), 25 deletions(-) diff --git a/tests/integration/_support/claude_code.py b/tests/integration/_support/claude_code.py index 8f2502952be..909728ae5b0 100644 --- a/tests/integration/_support/claude_code.py +++ b/tests/integration/_support/claude_code.py @@ -141,7 +141,9 @@ def tools() -> tuple[dict[str, JsonValue], ...]: { "file_path": field("The absolute path to the file to modify", type="string"), "old_string": field("The text to replace", type="string"), - "new_string": field("The text to replace it with (must be different from old_string)", type="string"), + "new_string": field( + "The text to replace it with (must be different from old_string)", type="string" + ), "replace_all": field( "Replace all occurrences of old_string (default false)", default=False, @@ -456,8 +458,12 @@ def tools() -> tuple[dict[str, JsonValue], ...]: "input_schema": schema( { "query": field("The search query to use", type="string", minLength=2), - "allowed_domains": field("Only include search results from these domains", type="array", items={"type": "string"}), - "blocked_domains": field("Never include search results from these domains", type="array", items={"type": "string"}), + "allowed_domains": field( + "Only include search results from these domains", type="array", items={"type": "string"} + ), + "blocked_domains": field( + "Never include search results from these domains", type="array", items={"type": "string"} + ), }, ("query",), ), @@ -470,8 +476,12 @@ def tools() -> tuple[dict[str, JsonValue], ...]: "type": "object", "properties": { "script": field("Self-contained workflow script.", type="string", maxLength=524288), - "name": field("Name of a predefined workflow (built-in or from .claude/workflows/).", type="string"), - "description": field("Ignored — set the workflow description in the script's `meta` block.", type="string"), + "name": field( + "Name of a predefined workflow (built-in or from .claude/workflows/).", type="string" + ), + "description": field( + "Ignored — set the workflow description in the script's `meta` block.", type="string" + ), "title": field("Ignored — set the workflow title in the script's `meta` block.", type="string"), "args": field("Optional input value exposed to the script as the global `args`, verbatim."), "scriptPath": field("Path to a workflow script file on disk.", type="string"), @@ -489,7 +499,9 @@ def tools() -> tuple[dict[str, JsonValue], ...]: "description": "Writes a file to the local filesystem.", "input_schema": schema( { - "file_path": field("The absolute path to the file to write (must be absolute, not relative)", type="string"), + "file_path": field( + "The absolute path to the file to write (must be absolute, not relative)", type="string" + ), "content": field("The content to write to the file", type="string"), }, ("file_path", "content"), diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py index d327181a491..9f929d69a05 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py @@ -2,10 +2,10 @@ import uuid from typing import Final import pytest +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc from pydantic import JsonValue diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py index ce87a76a5f2..c6ac00c20e8 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py @@ -1,10 +1,9 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc -from pydantic import JsonValue _CONTEXT_MANAGEMENT: Final = { "edits": [ diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py index 29ab1a70449..48b914a7374 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py @@ -2,10 +2,9 @@ import base64 import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc -from pydantic import JsonValue _PDF_BYTES: Final = ( b"%PDF-1.1\n" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py index 271f064becc..8a929ee2a06 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py @@ -1,10 +1,9 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc -from pydantic import JsonValue _PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" _IMAGE_BLOCK: Final = { diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py index a1d1a9bcb03..6990f33e51c 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py @@ -1,9 +1,9 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc from pydantic import JsonValue _TURN1_TOOL: Final = ("toolu_a", "Read", {"file_path": "/tmp/cc_probe/a.txt"}) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py index 4c4b075fcc1..7fe405ca439 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py @@ -2,10 +2,10 @@ import uuid from typing import Final import pytest +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc _BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py index d70f689ce07..5753d395bc4 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py @@ -1,11 +1,10 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc -from pydantic import JsonValue def test_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py index 8d4338f4910..06318b52aee 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py @@ -2,11 +2,10 @@ import uuid from typing import Final import pytest +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc -from pydantic import JsonValue _USAGE: Final = { "input_tokens": 10, diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py index 288787f5765..a5a57737fc1 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py @@ -1,10 +1,10 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc _MODEL: Final = cc.SONNET diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py index 7d421345ed6..f3fb9786e02 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py @@ -1,10 +1,10 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc from pydantic import JsonValue _THINKING: Final = "need to read the file" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py index f63126116be..66852e6e21b 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py @@ -1,10 +1,10 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} diff --git a/tests/integration/routing/test_overloaded_deployment_fallback.py b/tests/integration/routing/test_overloaded_deployment_fallback.py index d5ca213d3d9..7f5e669472d 100644 --- a/tests/integration/routing/test_overloaded_deployment_fallback.py +++ b/tests/integration/routing/test_overloaded_deployment_fallback.py @@ -3,13 +3,12 @@ import uuid from pathlib import Path from typing import Final -import pytest import yaml +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.process import owned_proxy from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc def _error_529() -> Reply: diff --git a/tests/integration/streaming/test_client_disconnect_still_bills.py b/tests/integration/streaming/test_client_disconnect_still_bills.py index 5a80e13cf44..19a52f274a1 100644 --- a/tests/integration/streaming/test_client_disconnect_still_bills.py +++ b/tests/integration/streaming/test_client_disconnect_still_bills.py @@ -1,10 +1,10 @@ import uuid from typing import Final +from integration._support import claude_code as cc from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server -from integration._support import claude_code as cc def test_client_disconnect_mid_stream_still_bills_the_message(gateway: Gateway) -> None: From 533fab43abb7a067f88b49b2302ab9bbd15e6c1e Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 22:36:35 +0000 Subject: [PATCH 16/19] test(integration): group anthropic messages tests into feature subfolders Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/README.md | 2 +- ...est_anthropic_prompt_cache_pricing_wire.py | 85 ++++ ...nthropic_tool_history_cache_tokens_wire.py | 128 ++++++ .../context/test_anthropic_compaction_wire.py | 108 +++++ ...ropic_bare_string_content_rejected_wire.py | 32 ++ .../test_anthropic_messages_timeout_wire.py | 58 +++ ...est_anthropic_slow_upstream_cutoff_wire.py | 64 +++ .../test_anthropic_request_fidelity_wire.py | 71 +++ .../test_anthropic_document_input_wire.py | 133 ++++++ .../test_anthropic_image_input_wire.py | 67 +++ ...anthropic_adaptive_thinking_effort_wire.py | 111 +++++ ...ropic_interleaved_thinking_history_wire.py | 167 +++++++ ...t_anthropic_legacy_thinking_budget_wire.py | 77 ++++ ...est_anthropic_model_switch_history_wire.py | 83 ++++ ...anthropic_thinking_signature_retry_wire.py | 95 ++++ ..._anthropic_messages_live_lifecycle_wire.py | 82 ++++ .../tools/test_anthropic_advisor_wire.py | 201 +++++++++ .../tools/test_anthropic_tool_loop_wire.py | 165 +++++++ ...est_anthropic_web_search_citations_wire.py | 170 ++++++++ .../tools/test_websearch_interception_wire.py | 409 ++++++++++++++++++ .../test_anthropic_long_context_beta_wire.py | 53 +++ 21 files changed, 2360 insertions(+), 1 deletion(-) create mode 100644 tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_adaptive_thinking_effort_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py diff --git a/tests/integration/README.md b/tests/integration/README.md index 7c5297cc5e1..1af3004b40e 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -30,7 +30,7 @@ Streaming checks send real HTTP transfer chunks, including one-byte partitions, The `messages_endpoint/` directory holds `/v1/messages` endpoint contracts: native-provider backends under `providers/` (`anthropic`, `bedrock`, `gemini`) and the translation bridges (`responses_bridge`, `chat_bridge`) at the top level. It runs in the providers shard; `run.py` selects test files recursively under each scheduled directory -A provider folder holds only what depends on that provider's wire format: request and response fidelity, multi-turn history, request parameters, content types, and provider-specific pricing. Behavior every provider shares, such as fallback or billing after a client disconnect, lives in the feature directory it exercises (`routing/`, `streaming/`, `spend/`). `_support/claude_code.py` holds a captured Claude Code request and stream builders that any directory can use as a realistic agent payload +A provider folder holds only what depends on that provider's wire format, and each subfolder is one feature that provider implements its own way: `headers/`, `streaming/`, `reasoning/`, `tools/`, `caching/`, `usage/` (reading the provider's token counts and pricing them), `multimodal/`, `context/` and `errors/`. A test goes in the folder of the feature it varies; one that fits no single folder tests two things and gets split. Behavior every provider shares, such as fallback or billing after a client disconnect, lives in the feature directory it exercises (`routing/`, `streaming/`, `spend/`). `_support/claude_code.py` holds a captured Claude Code request and stream builders that any directory can use as a realistic agent payload The sdk shard exercises the SDK's own HTTP clients against local protocol peers with no gateway in the path, so a case here fails only when the client library or its wire behavior changes. The HTTP/2 case runs a hypercorn TLS peer offering h2 and http/1.1 over ALPN, drives the sync and async httpx handlers at it with `LITELLM_HTTP2` off and on, and asserts the version both the client and the peer observed on the wire. Put a test here only when it needs no proxy, database or Redis; a case that reaches the gateway belongs in one of the other shards diff --git a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py new file mode 100644 index 00000000000..06318b52aee --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py @@ -0,0 +1,85 @@ +import uuid +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +_USAGE: Final = { + "input_tokens": 10, + "cache_read_input_tokens": 3000, + "cache_creation_input_tokens": 200, + "output_tokens": 5, +} + + +def test_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: + identity: Final = f"msg_pc_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**turn2, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply(content_type="text/event-stream", chunks=cc.text_stream(identity, cc.FABLE, "PROBE", _USAGE)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + cache_read_input_token_cost=1e-7, + cache_creation_input_token_cost=1.25e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + usage: Final = events[0][1]["message"]["usage"] + assert usage["cache_read_input_tokens"] == 3000, usage + assert usage["cache_creation_input_tokens"] == 200, usage + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 3000 * 1e-7 + 200 * 1.25e-6 + 5 * 5e-6), dict( + rows[0] + ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py b/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py new file mode 100644 index 00000000000..eff5fd565b5 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py @@ -0,0 +1,128 @@ +import json +import time +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + + +@pytest.mark.covers( + "other.provider_wire.anthropic.tool_history_system_cache_and_internal_fields", + "quota_management.spend_tracking.cache_tokens.disjoint_classes_use_explicit_rates", +) +def test_anthropic_tool_history_and_cache_tokens_keep_wire_and_accounting_contracts(gateway: Gateway) -> None: + identity: Final = "anthropic-wire-" + uuid.uuid4().hex + tool_schema: Final = { + "type": "object", + "properties": {"x": {"type": "integer"}, "y": {"type": "integer"}}, + "required": ["x", "y"], + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == "synthetic-anthropic-key" + body: Final = json.loads(request.body) + assert body["model"] == "claude-sonnet-4-5-20250929" + assert body["system"] == [{"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}}] + assert body["tools"][0]["name"] == "add" and body["tools"][0]["input_schema"] == tool_schema + assert body["max_tokens"] == 16 + assert not {"timeout", "stream_chunk_size", "litellm_params", "litellm_metadata", "rpm", "tpm"}.intersection( + body + ) + messages: Final = body["messages"] + assert [message["role"] for message in messages] == ["user", "assistant", "user"] + assert messages[0]["content"] == [{"type": "text", "text": "first"}] + assert messages[1]["content"] == [ + {"type": "tool_use", "id": "history-call", "name": "add", "input": {"x": 1, "y": 2}} + ] + assert messages[2]["content"] == [ + {"type": "tool_result", "tool_use_id": "history-call", "content": "3"}, + {"type": "text", "text": "next"}, + ] + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "tool_use", "id": "next-call", "name": "add", "input": {"x": 3, "y": 4}}], + "stop_reason": "tool_use", + "stop_sequence": None, + "usage": { + "input_tokens": 10, + "output_tokens": 4, + "cache_read_input_tokens": 5, + "cache_creation_input_tokens": 7, + }, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=wire.url, + api_key="synthetic-anthropic-key", + input_cost_per_token=0.001, + output_cost_per_token=0.002, + cache_read_input_token_cost=0.0001, + cache_creation_input_token_cost=0.002, + ) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "max_tokens": 16, + "timeout": 5, + "messages": [ + { + "role": "system", + "content": [ + {"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}} + ], + }, + {"role": "user", "content": "first"}, + { + "role": "assistant", + "tool_calls": [ + { + "id": "history-call", + "type": "function", + "function": {"name": "add", "arguments": '{"x":1,"y":2}'}, + } + ], + }, + {"role": "tool", "tool_call_id": "history-call", "content": "3"}, + {"role": "user", "content": "next"}, + ], + "tools": [{"type": "function", "function": {"name": "add", "parameters": tool_schema}}], + }, + ) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["id"].startswith("chatcmpl-") + assert body["choices"][0]["finish_reason"] == "tool_calls" + tool: Final = body["choices"][0]["message"]["tool_calls"][0] + assert tool["id"] == "next-call" and tool["function"]["name"] == "add" + assert json.loads(tool["function"]["arguments"]) == {"x": 3, "y": 4} + assert body["usage"]["prompt_tokens"] == 22 and body["usage"]["completion_tokens"] == 4 + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens, metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (body["id"],), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(10 * 0.001 + 5 * 0.0001 + 7 * 0.002 + 4 * 0.002) + assert rows[0]["prompt_tokens"] == 22 and rows[0]["completion_tokens"] == 4 + metadata: Final = rows[0]["metadata"] + parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) + assert parsed["cost_breakdown"]["input_cost"] == pytest.approx(0.0245) + assert parsed["cost_breakdown"]["output_cost"] == pytest.approx(0.008) diff --git a/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py new file mode 100644 index 00000000000..c6ac00c20e8 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py @@ -0,0 +1,108 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_CONTEXT_MANAGEMENT: Final = { + "edits": [ + {"type": "clear_thinking_20251015", "keep": "all"}, + {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, + ] +} +_COMPACTION_BLOCK: Final = {"type": "compaction", "content": ""} + + +def _compaction_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": dict(_COMPACTION_BLOCK)}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 5}, + "context_management": { + "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] + }, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gateway: Gateway) -> None: + identity: Final = f"msg_cm_{uuid.uuid4().hex}" + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "context_management": _CONTEXT_MANAGEMENT, + } + turn3: Final = cc.tool_loop_turn2( + request_body, + ( + dict(_COMPACTION_BLOCK), + {"type": "text", "text": "continuing after compaction"}, + ), + (), + ) + first_expected: Final = {**request_body, "model": cc.FABLE} + second_expected: Final = {**turn3, "model": cc.FABLE} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + if body == first_expected: + assert body["context_management"] == _CONTEXT_MANAGEMENT + return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) + assert body == second_expected, { + key: (second_expected.get(key), body.get(key)) + for key in second_expected.keys() | body.keys() + if second_expected.get(key) != body.get(key) + } + assert body["context_management"] == _CONTEXT_MANAGEMENT + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream("msg_cm_next", cc.FABLE, "OK", {"input_tokens": 20, "output_tokens": 2}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**request_body, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + events: Final = cc.sse_events(response1.text) + assert events[1][1]["content_block"] == _COMPACTION_BLOCK, events[1] + deltas: Final = [data for event, data in events if event == "message_delta"] + assert len(deltas) == 1 and deltas[0].get("context_management") == { + "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] + }, events + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py new file mode 100644 index 00000000000..ffea1830e54 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py @@ -0,0 +1,32 @@ +import json +import time +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + + +@pytest.mark.covers("other.provider_wire.anthropic.bare_string_content_item_is_client_error") +@pytest.mark.parametrize( + "text", [pytest.param("what type of file is this?", id="type_word"), pytest.param("hello", id="plain")] +) +def test_anthropic_bare_string_content_item_is_rejected_as_client_error_before_the_wire( + gateway: Gateway, text: str +) -> None: + def respond(request: Request) -> Reply: + raise AssertionError(f"upstream must not be reached: {request.target}") + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "max_tokens": 16, "timeout": 5, "messages": [{"role": "system", "content": [text]}]}, + ) + assert response.status_code == 400, response.text + assert wire.drain() == () diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py new file mode 100644 index 00000000000..76c29ad7763 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py @@ -0,0 +1,58 @@ +import json +import time +import uuid +from typing import Final + +import pytest +from integration._support.client import JSON_OBJECT, Gateway +from integration._support.wire import Reply, Request, wire_server + +_UPSTREAM_STALL_SECONDS: Final = 4.0 +_CONFIGURED_TIMEOUT_SECONDS: Final = 1.0 + + +@pytest.mark.covers("providers.anthropic_messages.configured_timeout_aborts_stalled_upstream") +def test_messages_endpoint_honors_configured_timeout_against_stalled_upstream(gateway: Gateway) -> None: + prompt: Final = "stall-" + uuid.uuid4().hex + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == "synthetic-anthropic-key" + body: Final = JSON_OBJECT.validate_json(request.body) + assert body["model"] == "claude-sonnet-4-5-20250929" + assert body["messages"] == [{"role": "user", "content": prompt}] + assert body["max_tokens"] == 16 + assert "timeout" not in body + time.sleep(_UPSTREAM_STALL_SECONDS) + return Reply( + body=json.dumps( + { + "id": "msg_stalled", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "too late"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 2}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=wire.url, + api_key="synthetic-anthropic-key", + timeout=_CONFIGURED_TIMEOUT_SECONDS, + ) + started: Final = time.monotonic() + response: Final = gateway.request( + "POST", + "/v1/messages", + {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, + ) + elapsed: Final = time.monotonic() - started + assert response.status_code == 408, response.text + assert elapsed < _UPSTREAM_STALL_SECONDS, f"timed out only after {elapsed:.2f}s: {response.text}" + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py new file mode 100644 index 00000000000..7532afe718d --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py @@ -0,0 +1,64 @@ +import json +import time +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + + +@pytest.mark.covers("other.provider_wire.anthropic.messages_request_timeout_reaches_transport") +def test_anthropic_messages_slow_upstream_is_cut_off_at_the_deployment_request_timeout(gateway: Gateway) -> None: + identity: Final = "anthropic-timeout-" + uuid.uuid4().hex + prompt: Final = f"slow answer {identity}" + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == "synthetic-anthropic-key" + body: Final = json.loads(request.body) + assert body["model"] == "claude-sonnet-4-5-20250929" + assert body["max_tokens"] == 16 + assert body["messages"] == [{"role": "user", "content": prompt}] + assert not { + "timeout", + "request_timeout", + "stream_chunk_size", + "litellm_params", + "litellm_metadata", + "rpm", + "tpm", + }.intersection(body) + time.sleep(1.5) + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "late"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 3, "output_tokens": 1}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=wire.url, + api_key="synthetic-anthropic-key", + request_timeout=0.3, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, + headers={"anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 408, response.text + assert "Timeout" in response.json()["error"]["message"], response.text + assert eventually(wire.drain, lambda requests: len(requests) == 1, seconds=5, return_last_on_timeout=True) diff --git a/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py b/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py new file mode 100644 index 00000000000..a5a57737fc1 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py @@ -0,0 +1,71 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = cc.SONNET + + +def test_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: + identity: Final = f"msg_cc_{uuid.uuid4().hex}" + request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + cli_beta: Final = frozenset(cc.CLI_BETA.split(",")) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY + assert request.headers["anthropic-version"] == "2023-06-01" + assert frozenset(request.headers.get("anthropic-beta", "").split(",")) == cli_beta, request.headers.get( + "anthropic-beta" + ) + assert "authorization" not in request.headers, dict(request.headers) + assert all(gateway.key not in value for value in request.headers.values()), dict(request.headers) + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": _MODEL} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, _MODEL, "PONG", {"input_tokens": 12, "output_tokens": 4}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key), + ) + assert response.status_code == 200, response.text + assert response.headers["content-type"].startswith("text/event-stream"), dict(response.headers) + events: Final = cc.sse_events(response.text) + assert [event for event, _ in events] == [ + "message_start", + "content_block_start", + "content_block_delta", + "content_block_stop", + "message_delta", + "message_stop", + ] + assert events[2][1]["delta"] == {"type": "text_delta", "text": "PONG"} + assert events[4][1]["delta"]["stop_reason"] == "end_turn" + assert events[4][1]["usage"]["output_tokens"] == 4 + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 diff --git a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py new file mode 100644 index 00000000000..48b914a7374 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py @@ -0,0 +1,133 @@ +import base64 +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_PDF_BYTES: Final = ( + b"%PDF-1.1\n" + b"1 0 obj<>endobj\n" + b"2 0 obj<>endobj\n" + b"3 0 obj<>endobj\n" + b"trailer<>\n%%EOF" +) +_DOC_BLOCK: Final = { + "type": "document", + "source": {"type": "base64", "data": base64.b64encode(_PDF_BYTES).decode(), "media_type": "application/pdf"}, + "citations": {"enabled": True}, +} + + +def _cited_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "A page."}}, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 0, + "delta": { + "type": "citations_delta", + "citation": { + "type": "page_location", + "document_index": 0, + "document_title": "dot.pdf", + "start_page_number": 1, + "end_page_number": 1, + "cited_text": "Page", + }, + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 6}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_base64_pdf_document_with_citations_reaches_anthropic_identical(gateway: Gateway) -> None: + request_body: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "messages": [ + { + "role": "user", + "content": [ + dict(_DOC_BLOCK), + {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, + ], + } + ], + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + return Reply(content_type="text/event-stream", chunks=_cited_stream(f"msg_doc_{uuid.uuid4().hex}")) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + citations: Final = [ + data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" + ] + assert citations == [ + { + "type": "citations_delta", + "citation": { + "type": "page_location", + "document_index": 0, + "document_title": "dot.pdf", + "start_page_number": 1, + "end_page_number": 1, + "cited_text": "Page", + }, + } + ], citations + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py new file mode 100644 index 00000000000..8a929ee2a06 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py @@ -0,0 +1,67 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" +_IMAGE_BLOCK: Final = { + "type": "image", + "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, +} + + +def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", + ) + with_image_result: Final = cc.tool_loop_turn2( + turn1, + ({"type": "tool_use", "id": "toolu_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), + (("toolu_img", [dict(_IMAGE_BLOCK)]),), + ) + pasted: Final = { + **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), + "messages": [ + { + "role": "user", + "content": [ + dict(_IMAGE_BLOCK), + {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, + ], + } + ], + } + first_expected: Final = {**with_image_result, "model": cc.FABLE} + second_expected: Final = {**pasted, "model": cc.FABLE} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + if body != first_expected: + assert body == second_expected, body + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream( + f"msg_img_{uuid.uuid4().hex}", cc.FABLE, "RED", {"input_tokens": 20, "output_tokens": 2} + ), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**with_image_result, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**pasted, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_adaptive_thinking_effort_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_adaptive_thinking_effort_wire.py new file mode 100644 index 00000000000..9f929d69a05 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_adaptive_thinking_effort_wire.py @@ -0,0 +1,111 @@ +import uuid +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def test_adaptive_thinking_and_effort_reach_anthropic_intact(gateway: Gateway) -> None: + identity: Final = f"msg_fable_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY + assert request.headers["anthropic-version"] == "2023-06-01" + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta + assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 12, "output_tokens": 4}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + assert [event for event, _ in events][-1] == "message_stop" + assert events[4][1]["delta"]["stop_reason"] == "end_turn" + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 + + +def test_xhigh_effort_reaches_anthropic_and_charges_by_usage(gateway: Gateway) -> None: + identity: Final = f"msg_opus_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "xhigh", 128000) + cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta + assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.OPUS} + assert body == expected, _diff(expected, body) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.OPUS, "PONG", {"input_tokens": 10, "output_tokens": 5}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.OPUS}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 5 * 5e-6) diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py new file mode 100644 index 00000000000..6990f33e51c --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py @@ -0,0 +1,167 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_TURN1_TOOL: Final = ("toolu_a", "Read", {"file_path": "/tmp/cc_probe/a.txt"}) +_TURN2_TOOL: Final = ("toolu_b", "Read", {"file_path": "/tmp/cc_probe/b.txt"}) + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def _turn2(base: dict[str, JsonValue]) -> dict[str, JsonValue]: + return cc.tool_loop_turn2( + base, + ( + {"type": "thinking", "thinking": "plan", "signature": "sig1"}, + {"type": "tool_use", "id": _TURN1_TOOL[0], "name": _TURN1_TOOL[1], "input": _TURN1_TOOL[2]}, + ), + ((_TURN1_TOOL[0], "ALPHA"),), + ) + + +def _turn3(turn2: dict[str, JsonValue]) -> dict[str, JsonValue]: + return cc.tool_loop_turn2( + turn2, + ( + {"type": "thinking", "thinking": "got A", "signature": "sig2"}, + {"type": "text", "text": "got A"}, + {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": _TURN2_TOOL[2]}, + ), + ((_TURN2_TOOL[0], "BRAVO"),), + ) + + +def _interleaved_stream(identity: str) -> tuple[bytes, ...]: + usage: Final = {"input_tokens": 20, "output_tokens": 12} + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": usage["input_tokens"], "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "got A"}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": "sig2"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 1, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 1, "delta": {"type": "text_delta", "text": "got A"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 2, + "content_block": {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": {}}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": {"type": "input_json_delta", "partial_json": '{"file_path": "/tmp/cc_pr'}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": {"type": "input_json_delta", "partial_json": 'obe/b.txt"}'}, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "tool_use", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identical(gateway: Gateway) -> None: + identity: Final = f"msg_il_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words", + ) + turn2: Final = _turn2(turn1) + turn3: Final = _turn3(turn2) + turn2_expected: Final = {**turn2, "model": cc.FABLE} + turn3_expected: Final = {**turn3, "model": cc.FABLE} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert upstream_beta.split(",").count("interleaved-thinking-2025-05-14") == 1, upstream_beta + body: Final = cc.JSON_OBJECT.validate_json(request.body) + if body == turn2_expected: + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}), + ) + assert body == turn3_expected, _diff(turn3_expected, body) + return Reply(content_type="text/event-stream", chunks=_interleaved_stream(identity)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + response3: Final = gateway.request( + "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response3.status_code == 200, response3.text + events: Final = cc.sse_events(response3.text) + started: Final = [ + (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" + ] + assert started == [(0, "thinking"), (1, "text"), (2, "tool_use")], started + assert events[-1][0] == "message_stop" + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (turn2_expected, turn3_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py new file mode 100644 index 00000000000..242e5c7ec5a --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py @@ -0,0 +1,77 @@ +import json +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = "claude-sonnet-4-6" +_KEY: Final = "synthetic-anthropic-key" +_THINKING: Final = {"type": "enabled", "budget_tokens": 8000} +_TOOL: Final = { + "name": "read_file", + "description": "read a file", + "input_schema": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}, +} +_NEXT_CALL: Final = {"type": "tool_use", "id": "call-2", "name": "read_file", "input": {"path": "schema.prisma"}} + + +def _tool_loop_history(identity: str) -> tuple[dict[str, object], ...]: + return ( + {"role": "user", "content": f"open the config for {identity}"}, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "call-1", "name": "read_file", "input": {"path": "config.yaml"}}], + }, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "call-1", "content": "model_list: []"}]}, + ) + + +def _tool_use_reply(identity: str) -> Reply: + return Reply( + body=json.dumps( + { + "id": f"msg-{identity}", + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [_NEXT_CALL], + "stop_reason": "tool_use", + "stop_sequence": None, + "usage": {"input_tokens": 40, "output_tokens": 12}, + } + ).encode() + ) + + +@pytest.mark.covers("providers.anthropic_messages.claude_4_6_legacy_thinking_budget_reaches_the_wire_unchanged") +def test_claude_4_6_thinking_budget_tokens_on_messages_is_forwarded_instead_of_rewritten_to_adaptive( + gateway: Gateway, +) -> None: + identity: Final = "legacy-thinking-" + uuid.uuid4().hex + history: Final = _tool_loop_history(identity) + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages", request.target + assert request.headers["x-api-key"] == _KEY + body: Final = json.loads(request.body) + assert body["thinking"] == _THINKING, body + assert "output_config" not in body, body + assert body["max_tokens"] == 32768, body + assert body["messages"] == list(history), body + assert body["tools"] == [_TOOL], body + return _tool_use_reply(identity) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {"model": model, "max_tokens": 32768, "thinking": _THINKING, "messages": history, "tools": [_TOOL]}, + ) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["content"] == [_NEXT_CALL], response.text + assert body["stop_reason"] == "tool_use", response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py new file mode 100644 index 00000000000..5753d395bc4 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py @@ -0,0 +1,83 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + + +def test_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: + identity1: Final = f"msg_sw1_{uuid.uuid4().hex}" + identity2: Final = f"msg_sw2_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + first_expected: Final = {**turn1, "model": cc.FABLE} + second_expected: Final = {**turn2, "model": cc.OPUS} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + if body == first_expected: + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream( + identity1, + cc.FABLE, + "need to read the file", + "sig_anthropic_1", + (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), + {"input_tokens": 20, "output_tokens": 10}, + ), + ) + assert body == second_expected, { + key: (second_expected.get(key), body.get(key)) + for key in second_expected.keys() | body.keys() + if second_expected.get(key) != body.get(key) + } + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity2, cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies + rows: Final = eventually( + lambda: read_rows( + 'SELECT model FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity2,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["model"] == f"anthropic/{cc.OPUS}" diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py new file mode 100644 index 00000000000..e414d8f0d11 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py @@ -0,0 +1,95 @@ +import json +import uuid +from typing import Final + +import pytest +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +MODEL: Final = "claude-sonnet-4-5-20250929" +KEY: Final = "synthetic-anthropic-key" +SIGNATURE_ERROR: Final = json.dumps( + { + "type": "error", + "error": { + "type": "invalid_request_error", + "message": "messages.2.content.0.thinking.signature.str: Input should be a valid string", + }, + } +).encode() +TOOLS: Final = ({"name": "lookup", "input_schema": {"type": "object", "properties": {"key": {"type": "string"}}}},) + + +def _history_with_unsigned_thinking(identity: str) -> tuple[dict[str, object], ...]: + return ( + {"role": "user", "content": [{"type": "text", "text": f"first question {identity}"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "first answer"}]}, + {"role": "user", "content": [{"type": "text", "text": "second question"}]}, + { + "role": "assistant", + "content": [ + {"type": "thinking", "thinking": "replayed from another provider", "signature": None}, + {"type": "tool_use", "id": "call-1", "name": "lookup", "input": {"key": "value"}}, + ], + }, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "call-1", "content": "found"}]}, + ) + + +@pytest.mark.covers("providers.anthropic_messages.missing_thinking_signature_400_retries_without_thinking_blocks") +def test_missing_thinking_signature_400_retries_once_without_thinking_blocks_and_returns_200( + gateway: Gateway, +) -> None: + identity: Final = "thinking-signature-" + uuid.uuid4().hex + history: Final = _history_with_unsigned_thinking(identity) + tool_use_only_turn: Final = { + "role": "assistant", + "content": [{"type": "tool_use", "id": "call-1", "name": "lookup", "input": {"key": "value"}}], + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == KEY + body: Final = json.loads(request.body) + assert body["model"] == MODEL + assert body["tools"] == list(TOOLS), body + if body["messages"][3]["content"][0]["type"] == "thinking": + assert body["messages"] == list(history), body + assert body["thinking"] == {"type": "enabled", "budget_tokens": 1024}, body + return Reply(status=400, body=SIGNATURE_ERROR) + assert body["messages"] == [*history[:3], tool_use_only_turn, history[4]], body + assert "thinking" not in body, body + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": MODEL, + "content": [{"type": "text", "text": "recovered without thinking history"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 30, "output_tokens": 6}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{MODEL}", api_base=wire.url, api_key=KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 64, + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "tools": list(TOOLS), + "messages": list(history), + }, + ) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["id"] == identity, response.text + assert body["content"] == [{"type": "text", "text": "recovered without thinking history"}], response.text + assert body["stop_reason"] == "end_turn", response.text + assert [request.target for request in wire.drain()] == ["/v1/messages", "/v1/messages"] diff --git a/tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py b/tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py new file mode 100644 index 00000000000..cb7043c0362 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py @@ -0,0 +1,82 @@ +import json +import threading +import uuid +from typing import Final + +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = "claude-sonnet-4-5-20250929" +_API_KEY: Final = "synthetic-anthropic-key" + + +def _sse(event: str, payload: dict[str, object]) -> bytes: + return f"event: {event}\ndata: {json.dumps(payload)}\n\n".encode() + + +def test_messages_stream_message_start_reaches_client_before_content_without_fallback( + gateway: Gateway, +) -> None: + """With no fallback able to take over, the proxy must not hold lifecycle + frames back for a retry that cannot happen: message_start reaches the + client while the upstream is still thinking.""" + gate: Final = threading.Event() + head: Final = _sse("message_start", {"type": "message_start", "message": {"id": "msg_live_1"}}) + _sse( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ) + tail: Final = ( + _sse( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "Hello"}}, + ) + + _sse("content_block_stop", {"type": "content_block_stop", "index": 0}) + + _sse( + "message_delta", + {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 3}}, + ) + + _sse("message_stop", {"type": "message_stop"}) + ) + prompt: Final = "live-lifecycle-" + uuid.uuid4().hex + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == _API_KEY + body: Final = json.loads(request.body) + assert body["model"] == _MODEL + assert body["stream"] is True + assert body["messages"] == [{"role": "user", "content": prompt}] + return Reply(content_type="text/event-stream", chunks=(head, tail), gate_after_first=gate) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_API_KEY) + with gateway.client.stream( + "POST", + "/v1/messages", + json={ + "model": model, + "max_tokens": 16, + "stream": True, + "messages": [{"role": "user", "content": prompt}], + }, + headers={"Authorization": f"Bearer {gateway.key}"}, + ) as response: + assert response.status_code == 200, response.read().decode() + lines = response.iter_lines() + first_event: Final = next( + json.loads(line.removeprefix("data: ")) for line in lines if line.startswith("data: ") + ) + assert first_event["type"] == "message_start" + gate.set() + events: Final = (first_event,) + tuple( + json.loads(line.removeprefix("data: ")) for line in lines if line.startswith("data: ") + ) + assert tuple(event["type"] for event in events) == ( + "message_start", + "content_block_start", + "content_block_delta", + "content_block_stop", + "message_delta", + "message_stop", + ), f"observed events: {events!r}" + assert [request.target for request in wire.drain()] == ["/v1/messages"] diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py new file mode 100644 index 00000000000..b2f44d8f155 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py @@ -0,0 +1,201 @@ +import json +import uuid +from pathlib import Path +from typing import Final + +import pytest +import yaml +from integration._support.client import Gateway +from integration._support.process import owned_proxy +from integration._support.wire import Reply, Request, wire_server + +_ADVISOR_KEY: Final = "synthetic-advisor-key" +_PROXY_ANTHROPIC_KEY: Final = "sk-proxy-owned-anthropic-secret" +_QUESTION: Final = "which index should this query use" +_ADVICE: Final = "use the composite index on (tenant_id, created_at)" +_FINAL_ANSWER: Final = "done, the composite index is the right one" + + +def _advisor_call_message(question: str) -> dict[str, object]: + return { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "advisor-call", + "type": "function", + "function": {"name": "advisor", "arguments": json.dumps({"question": question})}, + } + ], + } + + +_FINAL_MESSAGE: Final = {"role": "assistant", "content": _FINAL_ANSWER} + + +def _chat_completion(identity: str, message: dict[str, object], finish_reason: str) -> Reply: + return Reply( + body=json.dumps( + { + "id": f"chatcmpl-{identity}", + "object": "chat.completion", + "created": 1, + "model": "llama-3.3-70b-versatile", + "choices": [{"index": 0, "message": message, "finish_reason": finish_reason}], + "usage": {"prompt_tokens": 10, "completion_tokens": 4, "total_tokens": 14}, + } + ).encode() + ) + + +def _executor_reply(body: dict[str, object], identity: str, question: str) -> Reply: + messages: Final = body["messages"] + assert isinstance(messages, list) + if any(message.get("role") == "tool" for message in messages): + assert messages[-1]["content"] == _ADVICE + return _chat_completion(identity, _FINAL_MESSAGE, "stop") + tools: Final = body["tools"] + assert isinstance(tools, list) + assert tools[0]["function"]["name"] == "advisor" + return _chat_completion(identity, _advisor_call_message(question), "tool_calls") + + +@pytest.mark.covers("providers.anthropic_messages_advisor.sub_call_uses_the_configured_advisor_deployment") +def test_advisor_sub_call_reaches_the_router_deployment_with_its_key_instead_of_anthropic_unauthenticated( + gateway: Gateway, +) -> None: + identity: Final = "advisor-wire-" + uuid.uuid4().hex + migration: Final = "please plan the migration " + identity + question: Final = _QUESTION + " " + identity + + def respond(request: Request) -> Reply: + body: Final = json.loads(request.body) + if request.target == "/v1/chat/completions": + assert request.headers["authorization"] == "Bearer integration-provider-key" + return _executor_reply(body, identity, question) + assert request.target == "/v1/messages" + assert request.headers["x-api-key"] == _ADVISOR_KEY + assert body["model"] == "claude-opus-4-1-20250805" + assert body["messages"] == [ + {"role": "user", "content": migration}, + {"role": "user", "content": question}, + ] + assert "tools" not in body + return Reply( + body=json.dumps( + { + "id": f"msg-{identity}", + "type": "message", + "role": "assistant", + "model": "claude-opus-4-1-20250805", + "content": [{"type": "text", "text": _ADVICE}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 6}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + executor: Final = scenario.model(model="hosted_vllm/gpt-4o-mini", api_base=wire.url + "/v1") + advisor: Final = scenario.model( + model="anthropic/claude-opus-4-1-20250805", api_base=wire.url, api_key=_ADVISOR_KEY + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + { + "model": executor, + "max_tokens": 64, + "messages": [{"role": "user", "content": migration}], + "tools": [{"type": "advisor_20260301", "name": "advisor", "model": advisor}], + }, + ) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["content"] == [{"type": "text", "text": _FINAL_ANSWER}], response.text + assert body["stop_reason"] == "end_turn", response.text + assert [request.target for request in wire.drain()] == [ + "/v1/chat/completions", + "/v1/messages", + "/v1/chat/completions", + ] + + +def _advice_reply(identity: str) -> Reply: + return Reply( + body=json.dumps( + { + "id": f"msg-{identity}", + "type": "message", + "role": "assistant", + "model": "claude-opus-4-1-20250805", + "content": [{"type": "text", "text": _ADVICE}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 6}, + } + ).encode() + ) + + +@pytest.mark.covers("providers.anthropic_messages_advisor.caller_api_base_without_api_key_never_receives_the_proxy_key") +def test_advisor_api_base_without_api_key_is_rejected_before_the_proxy_anthropic_key_reaches_the_caller_host( + gateway: Gateway, tmp_path: Path +) -> None: + identity: Final = "advisor-leak-" + uuid.uuid4().hex + question: Final = _QUESTION + " " + identity + + def executor(request: Request) -> Reply: + assert request.target == "/v1/chat/completions", request.target + return _executor_reply(json.loads(request.body), identity, question) + + def caller_host(request: Request) -> Reply: + return _advice_reply(identity) + + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["general_settings"]["allow_client_side_credentials"] = True + path: Final = tmp_path / "client-side-credentials.yaml" + path.write_text(yaml.safe_dump(config)) + with ( + wire_server(executor) as executor_wire, + wire_server(caller_host) as caller_wire, + owned_proxy(gateway, tmp_path, {"ANTHROPIC_API_KEY": _PROXY_ANTHROPIC_KEY}, config=path) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model="hosted_vllm/gpt-4o-mini", api_base=executor_wire.url + "/v1") + response: Final = candidate.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 64, + "messages": [{"role": "user", "content": "please plan the migration"}], + "tools": [ + { + "type": "advisor_20260301", + "name": "advisor", + "model": "anthropic/claude-opus-4-1-20250805", + "api_base": caller_wire.url, + } + ], + }, + ) + received: Final = caller_wire.drain() + assert [ + (request.target, request.headers.get("x-api-key"), json.loads(request.body)["messages"]) + for request in received + ] == [], response.text + assert response.is_error, response.text + assert response.json() == { + "type": "error", + "error": { + "type": "api_error", + "message": ( + "advisor tool definition sets 'api_base' without 'api_key'. A caller-supplied api_base is only " + "honored alongside a caller-supplied api_key, so the proxy's own credentials are never sent to a " + "caller-chosen destination." + ), + }, + }, response.text + assert executor_wire.drain() == (), response.text diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py new file mode 100644 index 00000000000..f3fb9786e02 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py @@ -0,0 +1,165 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_THINKING: Final = "need to read the file" +_SIGNATURE: Final = "sig_probe_1" +_USAGE: Final = {"input_tokens": 20, "output_tokens": 10} + + +def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def test_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: + identity1: Final = f"msg_tl1_{uuid.uuid4().hex}" + identity2: Final = f"msg_tl2_{uuid.uuid4().hex}" + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + calls: Final = (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + first_expected: Final = {**turn1, "model": cc.FABLE} + second_expected: Final = {**turn2, "model": cc.FABLE} + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + if body == first_expected: + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream(identity1, cc.FABLE, _THINKING, _SIGNATURE, calls, _USAGE), + ) + assert body == second_expected, _diff(second_expected, body) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity2, cc.FABLE, "PROBE", {"input_tokens": 30, "output_tokens": 3}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + events: Final = cc.sse_events(response1.text) + assert [ + ( + event, + data.get("delta", {}).get( + "type", data.get("content_block", {}).get("type", data.get("delta", {}).get("stop_reason")) + ), + ) + for event, data in events + ] == [ + ("message_start", None), + ("content_block_start", "thinking"), + ("content_block_delta", "thinking_delta"), + ("content_block_delta", "signature_delta"), + ("content_block_stop", None), + ("content_block_start", "tool_use"), + ("content_block_delta", "input_json_delta"), + ("content_block_delta", "input_json_delta"), + ("content_block_stop", None), + ("message_delta", "tool_use"), + ("message_stop", None), + ] + assert events[5][1]["content_block"]["id"] == "toolu_read_1" + assert events[5][1]["content_block"]["name"] == "Read" + partial: Final = events[6][1]["delta"]["partial_json"] + events[7][1]["delta"]["partial_json"] + assert partial == '{"file_path": "/tmp/cc_probe/hello.txt"}' + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + events2: Final = cc.sse_events(response2.text) + assert events2[2][1]["delta"] == {"type": "text_delta", "text": "PROBE"} + assert events2[4][1]["delta"]["stop_reason"] == "end_turn" + bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) + assert bodies == (first_expected, second_expected), bodies + rows: Final = eventually( + lambda: read_rows( + 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity2,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert rows[0]["prompt_tokens"] == 30 + + +def test_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and /tmp/cc_probe/world.txt and reply with both words", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + { + "type": "tool_use", + "id": "toolu_read_2", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/world.txt"}, + }, + ), + (("toolu_read_2", "1\tPROBE2\n2\t"), ("toolu_read_1", "1\tPROBE\n2\t")), + ) + + def respond(request: Request) -> Reply: + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**turn2, "model": cc.FABLE} + assert body == expected, _diff(expected, body) + results: Final = [block for block in body["messages"][3]["content"] if block["type"] == "tool_result"] + assert [block["tool_use_id"] for block in results] == ["toolu_read_2", "toolu_read_1"] + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(f"msg_mt_{uuid.uuid4().hex}", cc.FABLE, "PROBE PROBE2", _USAGE), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**turn2, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py new file mode 100644 index 00000000000..66852e6e21b --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py @@ -0,0 +1,170 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} + + +def _web_search_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": cc.FABLE, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 20, "output_tokens": 1, "server_tool_use": {"web_search_requests": 1}}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {}}, + }, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "input_json_delta", "partial_json": '{"query": "current LiteLLM version"}'}, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 1, + "content_block": { + "type": "web_search_tool_result", + "tool_use_id": "srvtoolu_1", + "content": [ + { + "type": "web_search_result", + "title": "litellm releases", + "url": "https://example.com/litellm", + "page_age": None, + "encrypted_content": "enc_ws_1", + } + ], + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 2, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 2, "delta": {"type": "text_delta", "text": "1.104.0"}}, + ), + cc.sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": 2, + "delta": { + "type": "citations_delta", + "citation": { + "type": "web_search_result_location", + "url": "https://example.com/litellm", + "title": "litellm releases", + "cited_text": "version 1.104.0", + "encrypted_index": "eidx_1", + }, + }, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 15, "server_tool_use": {"web_search_requests": 1}}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def test_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: + identity: Final = f"msg_ws_{uuid.uuid4().hex}" + base: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Use web search to find the current LiteLLM version and answer in one word", + ) + request_body: Final = { + **base, + "tools": [*base["tools"], WEB_SEARCH_TOOL], + } + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + body: Final = cc.JSON_OBJECT.validate_json(request.body) + expected: Final = {**request_body, "model": cc.FABLE} + assert body == expected, { + key: (expected.get(key), body.get(key)) + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + assert body["tools"][-1] == WEB_SEARCH_TOOL + assert len({tool["name"] for tool in body["tools"]}) == len(body["tools"]), body["tools"] + return Reply(content_type="text/event-stream", chunks=_web_search_stream(identity)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + events: Final = cc.sse_events(response.text) + started: Final = [ + (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" + ] + assert started == [(0, "server_tool_use"), (1, "web_search_tool_result"), (2, "text")], started + citations: Final = [ + data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" + ] + assert len(citations) == 1 and citations[0]["citation"]["url"] == "https://example.com/litellm", citations + start_usage: Final = events[0][1]["message"]["usage"] + assert start_usage["server_tool_use"]["web_search_requests"] == 1, start_usage + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + token_cost: Final = 20 * 1e-6 + 15 * 5e-6 + assert float(rows[0]["spend"]) >= token_cost, dict(rows[0]) diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py new file mode 100644 index 00000000000..a6f098cf64f --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py @@ -0,0 +1,409 @@ +import json +from pathlib import Path +from typing import Final + +import pytest +import yaml +from integration._support.client import Gateway +from integration._support.process import owned_proxy +from integration._support.wire import Reply, Request, wire_server + +BEDROCK_MODEL: Final = "us.anthropic.claude-haiku-4-5-20251001-v1:0" +INVOKE_TARGET: Final = f"/model/{BEDROCK_MODEL}/invoke" +SEARCH_TARGET: Final = "/tavily/search" +SEARCH_RESULT: Final = { + "title": "Synthetic result", + "url": "https://example.test/result", + "content": "the snippet text", +} + + +def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: + frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) + return tuple( + ( + next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), + json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), + ) + for frame in frames + ) + + +@pytest.mark.covers("other.provider_wire.bedrock.websearch_interception_streamed_capped_turn_ends_with_native_results") +def test_streamed_web_search_turn_capped_by_max_agentic_loops_ends_turn_with_snippets_and_ordered_blocks( + gateway: Gateway, tmp_path: Path +) -> None: + def respond(request: Request) -> Reply: + assert request.method == "POST", request.target + body: Final = json.loads(request.body) + if request.target == SEARCH_TARGET: + assert request.headers["authorization"] == "Bearer synthetic-tavily-key" + assert body["query"] == "query-0", body + return Reply(body=json.dumps({"query": "query-0", "results": [SEARCH_RESULT]}).encode()) + assert request.target == INVOKE_TARGET + assert request.headers["authorization"] == "Bearer synthetic-bedrock-token" + assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] + assert "stream" not in body, body + depth: Final = sum( + 1 + for message in body["messages"] + if isinstance(message["content"], list) + for block in message["content"] + if block["type"] == "tool_result" + ) + if depth == 1: + assert body["messages"][2]["content"] == [ + { + "type": "tool_result", + "tool_use_id": "toolu_0", + "content": "Title: Synthetic result\nURL: https://example.test/result\nSnippet: the snippet text", + } + ], body["messages"] + return Reply( + body=json.dumps( + { + "id": f"msg_{depth}", + "type": "message", + "role": "assistant", + "model": BEDROCK_MODEL, + "content": [ + {"type": "text", "text": f"turn-{depth}"}, + { + "type": "tool_use", + "id": f"toolu_{depth}", + "name": "litellm_web_search", + "input": {"query": f"query-{depth}"}, + }, + ], + "stop_reason": "tool_use", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 4}, + } + ).encode() + ) + + with wire_server(respond) as wire: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["search_tools"] = [ + { + "search_tool_name": "integration-search", + "litellm_params": { + "search_provider": "tavily", + "api_key": "synthetic-tavily-key", + "api_base": wire.url + "/tavily", + }, + } + ] + config["litellm_settings"].update( + { + "callbacks": ["websearch_interception"], + "websearch_interception_params": { + "enabled_providers": ["bedrock"], + "search_tool_name": "integration-search", + "max_agentic_loops": 1, + }, + } + ) + path: Final = tmp_path / "websearch.yaml" + path.write_text(yaml.safe_dump(config)) + with owned_proxy(gateway, tmp_path, {}, config=path) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model( + model=f"bedrock/{BEDROCK_MODEL}", + api_key="synthetic-bedrock-token", + api_base=wire.url, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=wire.url, + ) + response: Final = candidate.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 64, + "stream": True, + "messages": [{"role": "user", "content": "search control"}], + "tools": [{"type": "web_search_20250305", "name": "web_search"}], + }, + ) + assert response.status_code == 200, response.text + events: Final = sse_events(response.text) + assert [name for name, _ in events][:1] == ["message_start"], response.text + assert [name for name, _ in events][-2:] == ["message_delta", "message_stop"], response.text + for position, (name, event) in enumerate(events): + if name == "content_block_stop": + assert event["index"] in { + earlier_event["index"] + for earlier, earlier_event in events[:position] + if earlier == "content_block_start" + }, response.text + started: Final = tuple(event["content_block"] for name, event in events if name == "content_block_start") + search_ids: Final = tuple(block["id"] for block in started if block["type"] == "server_tool_use") + assert search_ids and all(search_id.startswith("srvtoolu_") for search_id in search_ids), response.text + assert started[-1] == {"type": "text", "text": ""}, response.text + assert started[:-1] == tuple( + block + for search_id in search_ids + for block in ( + {"type": "server_tool_use", "id": search_id, "name": "web_search", "input": {"query": "query-0"}}, + { + "type": "web_search_tool_result", + "tool_use_id": search_id, + "content": [ + { + "type": "web_search_result", + "url": "https://example.test/result", + "title": "Synthetic result", + "page_age": None, + "encrypted_content": "", + "snippet": "the snippet text", + } + ], + }, + ) + ), response.text + assert ( + "".join(event["delta"]["text"] for name, event in events if name == "content_block_delta") == "turn-1" + ), response.text + assert [event["delta"]["stop_reason"] for name, event in events if name == "message_delta"] == [ + "end_turn" + ], response.text + assert "litellm_web_search" not in response.text, response.text + assert [request.target for request in wire.drain()] == [INVOKE_TARGET, SEARCH_TARGET, INVOKE_TARGET] + + +import threading +import uuid +from typing import Final +from urllib.parse import parse_qs, urlsplit + +import httpx +import pytest +from integration._support.client import Gateway, eventually + +_QUERY: Final = "integration capped search" +_TEXT_BLOCK: Final = {"type": "text", "text": "searching once more"} +_NOT_INTERCEPTED: Final = "native tool reached the provider" +_FINAL_BLOCK: Final = {"type": "text", "text": "answered from the stored backend"} +_OWNED_RESULT_TEXT: Final = "Title: Owned result\nURL: https://owned.invalid/a\nSnippet: owned snippet" +_SEARCH_RESULT_BLOCK: Final = { + "type": "web_search_result", + "url": "https://owned.invalid/a", + "title": "Owned result", + "page_age": None, + "encrypted_content": "", + "snippet": "owned snippet", +} + + +def _search_tool_use(identity: str) -> dict[str, object]: + return {"type": "tool_use", "id": identity, "name": "litellm_web_search", "input": {"query": _QUERY}} + + +def _anthropic_reply(identity: str, content: list[dict[str, object]], stop_reason: str) -> Reply: + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": content, + "stop_reason": stop_reason, + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 4}, + } + ).encode() + ) + + +@pytest.mark.covers( + "other.provider_wire.anthropic.websearch_interception_capped_loop_ends_turn_without_internal_tool_use" +) +def test_capped_websearch_interception_loop_ends_turn_instead_of_exposing_internal_tool_use( + gateway: Gateway, tmp_path: Path +) -> None: + identity: Final = "websearch-wire-" + uuid.uuid4().hex + searched: Final = threading.Event() + + def respond(request: Request) -> Reply: + parts: Final = urlsplit(request.target) + if request.method == "GET" and parts.path == "/search": + assert parse_qs(parts.query)["q"] == [_QUERY], request.target + searched.set() + return Reply( + body=json.dumps( + { + "results": [ + {"title": "Owned result", "url": "https://owned.invalid/a", "content": "owned snippet"} + ] + } + ).encode() + ) + assert request.method == "POST" and parts.path == "/v1/messages", request.target + body: Final = json.loads(request.body) + if any(tool.get("type") == "web_search_20250305" for tool in body["tools"]): + return _anthropic_reply(identity, [{"type": "text", "text": _NOT_INTERCEPTED}], "end_turn") + assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] + return _anthropic_reply(identity, [_TEXT_BLOCK, _search_tool_use(identity)], "tool_use") + + def send(candidate: Gateway, model: str) -> httpx.Response: + return candidate.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 64, + "messages": [{"role": "user", "content": identity + " attempt " + uuid.uuid4().hex}], + "tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 3}], + }, + ) + + def searched_through_proxy(response: httpx.Response) -> bool: + return searched.is_set() and _NOT_INTERCEPTED not in response.text + + with wire_server(respond) as wire: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["search_tools"] = [ + { + "search_tool_name": "integration-searxng", + "litellm_params": {"search_provider": "searxng", "api_base": wire.url}, + } + ] + config["litellm_settings"].update( + { + "callbacks": ["websearch_interception"], + "websearch_interception_params": { + "enabled": True, + "enabled_providers": ["anthropic"], + "search_tool_name": "integration-searxng", + }, + } + ) + path: Final = tmp_path / "websearch.yaml" + path.write_text(yaml.safe_dump(config)) + with owned_proxy(gateway, tmp_path, {}, config=path) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + response: Final = eventually(lambda: send(candidate, model), searched_through_proxy, seconds=40) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["stop_reason"] == "end_turn", response.text + content: Final = body["content"] + assert [block["type"] for block in content] == ["server_tool_use", "web_search_tool_result", "text"], ( + response.text + ) + assert content[0]["name"] == "web_search" and content[0]["input"] == {"query": _QUERY}, response.text + assert content[1]["tool_use_id"] == content[0]["id"], response.text + assert content[1]["content"] == [_SEARCH_RESULT_BLOCK], response.text + assert content[2] == _TEXT_BLOCK, response.text + targets: Final = tuple((request.method, urlsplit(request.target).path) for request in wire.drain()) + assert targets[-3:] == (("POST", "/v1/messages"), ("GET", "/search"), ("POST", "/v1/messages")), targets + + +@pytest.mark.covers("other.provider_wire.anthropic.websearch_interception_uses_database_search_tool_backend") +def test_database_created_search_tool_backend_receives_the_intercepted_query_over_a_same_named_config_tool( + gateway: Gateway, tmp_path: Path +) -> None: + identity: Final = "websearch-db-" + uuid.uuid4().hex + tool_name: Final = "integration-db-searxng-" + uuid.uuid4().hex + searched: Final = threading.Event() + + def respond(request: Request) -> Reply: + parts: Final = urlsplit(request.target) + if request.method == "GET" and parts.path == "/database/search": + assert parse_qs(parts.query)["q"] == [_QUERY], request.target + searched.set() + return Reply( + body=json.dumps( + { + "results": [ + {"title": "Owned result", "url": "https://owned.invalid/a", "content": "owned snippet"} + ] + } + ).encode() + ) + assert request.method == "POST" and parts.path == "/v1/messages", request.target + body: Final = json.loads(request.body) + if any(tool.get("type") == "web_search_20250305" for tool in body["tools"]): + return _anthropic_reply(identity, [{"type": "text", "text": _NOT_INTERCEPTED}], "end_turn") + assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] + results: Final = [ + block + for message in body["messages"] + if isinstance(message["content"], list) + for block in message["content"] + if block["type"] == "tool_result" + ] + if not results: + return _anthropic_reply(identity, [_TEXT_BLOCK, _search_tool_use(identity)], "tool_use") + assert results == [{"type": "tool_result", "tool_use_id": identity, "content": _OWNED_RESULT_TEXT}], results + return _anthropic_reply(identity, [_FINAL_BLOCK], "end_turn") + + def send(candidate: Gateway, model: str) -> httpx.Response: + return candidate.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 64, + "messages": [{"role": "user", "content": identity + " attempt " + uuid.uuid4().hex}], + "tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 3}], + }, + ) + + def searched_through_proxy(response: httpx.Response) -> bool: + return searched.is_set() and _NOT_INTERCEPTED not in response.text + + with wire_server(respond) as wire, gateway.scenario() as scenario: + created: Final = gateway.post( + "/search_tools", + { + "search_tool": { + "search_tool_name": tool_name, + "litellm_params": {"search_provider": "searxng", "api_base": wire.url + "/database"}, + } + }, + ) + scenario.cleanups.callback(gateway.request, "DELETE", f"/search_tools/{created['search_tool_id']}") + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["search_tools"] = [ + { + "search_tool_name": tool_name, + "litellm_params": {"search_provider": "searxng", "api_base": wire.url + "/config"}, + } + ] + config["litellm_settings"].update( + { + "callbacks": ["websearch_interception"], + "websearch_interception_params": { + "enabled": True, + "enabled_providers": ["anthropic"], + "search_tool_name": tool_name, + }, + } + ) + path: Final = tmp_path / "websearch-db.yaml" + path.write_text(yaml.safe_dump(config)) + environment: Final = {"ANTHROPIC_API_BASE": wire.url} + with owned_proxy(gateway, tmp_path, environment, config=path) as candidate, candidate.scenario() as models: + model: Final = models.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + response: Final = eventually(lambda: send(candidate, model), searched_through_proxy, seconds=40) + assert response.status_code == 200, response.text + body: Final = response.json() + assert body["stop_reason"] == "end_turn", response.text + assert body["content"][-1] == _FINAL_BLOCK, response.text + found: Final = [ + (result["url"], result["title"]) + for block in body["content"] + if block["type"] == "web_search_tool_result" + for result in block["content"] + ] + assert found == [("https://owned.invalid/a", "Owned result")], response.text + assert "litellm_web_search" not in response.text, response.text + targets: Final = tuple((request.method, urlsplit(request.target).path) for request in wire.drain()) + assert targets[-3:] == (("POST", "/v1/messages"), ("GET", "/database/search"), ("POST", "/v1/messages")), ( + targets + ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py new file mode 100644 index 00000000000..7fe405ca439 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py @@ -0,0 +1,53 @@ +import uuid +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +_BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" + + +def test_1m_context_beta_forwarded_and_tiered_prompt_priced_above_200k(gateway: Gateway) -> None: + identity: Final = f"msg_1m_{uuid.uuid4().hex}" + request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) + + def respond(request: Request) -> Reply: + assert request.method == "POST" + assert request.target == "/v1/messages", request.target + upstream_beta: Final = request.headers.get("anthropic-beta", "") + assert upstream_beta.split(",").count("context-1m-2025-08-07") == 1, upstream_beta + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 250000, "output_tokens": 100}), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{cc.FABLE}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=1e-6, + input_cost_per_token_above_200k_tokens=2e-6, + output_cost_per_token=5e-6, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**request_body, "model": model}, + params={"beta": "true"}, + headers=cc.cli_headers(gateway.key, _BETA_1M), + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows( + 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(250000 * 2e-6 + 100 * 5e-6), dict(rows[0]) From 3380361e7164da9f2e7cd7c3975cfa08a91499b1 Mon Sep 17 00:00:00 2001 From: kerry Date: Tue, 29 Sep 2026 22:36:46 +0000 Subject: [PATCH 17/19] test(integration): drop the pre-move anthropic test paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...anthropic_adaptive_thinking_effort_wire.py | 111 ----- .../anthropic/test_anthropic_advisor_wire.py | 201 --------- .../test_anthropic_compaction_wire.py | 108 ----- .../test_anthropic_document_input_wire.py | 133 ------ .../test_anthropic_image_input_wire.py | 67 --- ...ropic_interleaved_thinking_history_wire.py | 167 ------- ...t_anthropic_legacy_thinking_budget_wire.py | 77 ---- .../test_anthropic_long_context_beta_wire.py | 53 --- ..._anthropic_messages_live_lifecycle_wire.py | 82 ---- .../test_anthropic_messages_timeout_wire.py | 58 --- ...est_anthropic_model_switch_history_wire.py | 83 ---- ...est_anthropic_prompt_cache_billing_wire.py | 85 ---- .../test_anthropic_request_fidelity_wire.py | 71 --- ...anthropic_thinking_signature_retry_wire.py | 95 ---- .../test_anthropic_tool_loop_wire.py | 165 ------- ...est_anthropic_web_search_citations_wire.py | 170 -------- .../anthropic/test_anthropic_wire.py | 206 --------- .../test_websearch_interception_wire.py | 409 ------------------ 18 files changed, 2341 deletions(-) delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py deleted file mode 100644 index 9f929d69a05..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_adaptive_thinking_effort_wire.py +++ /dev/null @@ -1,111 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from pydantic import JsonValue - - -def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: - return { - key: {"expected": expected.get(key), "upstream": body.get(key)} - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - - -def test_adaptive_thinking_and_effort_reach_anthropic_intact(gateway: Gateway) -> None: - identity: Final = f"msg_fable_{uuid.uuid4().hex}" - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY - assert request.headers["anthropic-version"] == "2023-06-01" - upstream_beta: Final = request.headers.get("anthropic-beta", "") - assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta - assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.FABLE} - assert body == expected, _diff(expected, body) - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 12, "output_tokens": 4}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - assert [event for event, _ in events][-1] == "message_stop" - assert events[4][1]["delta"]["stop_reason"] == "end_turn" - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 - - -def test_xhigh_effort_reaches_anthropic_and_charges_by_usage(gateway: Gateway) -> None: - identity: Final = f"msg_opus_{uuid.uuid4().hex}" - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "xhigh", 128000) - cli_beta: Final = frozenset(cc.FRONTIER_CLI_BETA.split(",")) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY - upstream_beta: Final = request.headers.get("anthropic-beta", "") - assert cli_beta <= frozenset(upstream_beta.split(",")), upstream_beta - assert upstream_beta.split(",").count("effort-2025-11-24") == 1, upstream_beta - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.OPUS} - assert body == expected, _diff(expected, body) - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, cc.OPUS, "PONG", {"input_tokens": 10, "output_tokens": 5}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.OPUS}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - output_cost_per_token=5e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 5 * 5e-6) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py deleted file mode 100644 index b2f44d8f155..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py +++ /dev/null @@ -1,201 +0,0 @@ -import json -import uuid -from pathlib import Path -from typing import Final - -import pytest -import yaml -from integration._support.client import Gateway -from integration._support.process import owned_proxy -from integration._support.wire import Reply, Request, wire_server - -_ADVISOR_KEY: Final = "synthetic-advisor-key" -_PROXY_ANTHROPIC_KEY: Final = "sk-proxy-owned-anthropic-secret" -_QUESTION: Final = "which index should this query use" -_ADVICE: Final = "use the composite index on (tenant_id, created_at)" -_FINAL_ANSWER: Final = "done, the composite index is the right one" - - -def _advisor_call_message(question: str) -> dict[str, object]: - return { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": "advisor-call", - "type": "function", - "function": {"name": "advisor", "arguments": json.dumps({"question": question})}, - } - ], - } - - -_FINAL_MESSAGE: Final = {"role": "assistant", "content": _FINAL_ANSWER} - - -def _chat_completion(identity: str, message: dict[str, object], finish_reason: str) -> Reply: - return Reply( - body=json.dumps( - { - "id": f"chatcmpl-{identity}", - "object": "chat.completion", - "created": 1, - "model": "llama-3.3-70b-versatile", - "choices": [{"index": 0, "message": message, "finish_reason": finish_reason}], - "usage": {"prompt_tokens": 10, "completion_tokens": 4, "total_tokens": 14}, - } - ).encode() - ) - - -def _executor_reply(body: dict[str, object], identity: str, question: str) -> Reply: - messages: Final = body["messages"] - assert isinstance(messages, list) - if any(message.get("role") == "tool" for message in messages): - assert messages[-1]["content"] == _ADVICE - return _chat_completion(identity, _FINAL_MESSAGE, "stop") - tools: Final = body["tools"] - assert isinstance(tools, list) - assert tools[0]["function"]["name"] == "advisor" - return _chat_completion(identity, _advisor_call_message(question), "tool_calls") - - -@pytest.mark.covers("providers.anthropic_messages_advisor.sub_call_uses_the_configured_advisor_deployment") -def test_advisor_sub_call_reaches_the_router_deployment_with_its_key_instead_of_anthropic_unauthenticated( - gateway: Gateway, -) -> None: - identity: Final = "advisor-wire-" + uuid.uuid4().hex - migration: Final = "please plan the migration " + identity - question: Final = _QUESTION + " " + identity - - def respond(request: Request) -> Reply: - body: Final = json.loads(request.body) - if request.target == "/v1/chat/completions": - assert request.headers["authorization"] == "Bearer integration-provider-key" - return _executor_reply(body, identity, question) - assert request.target == "/v1/messages" - assert request.headers["x-api-key"] == _ADVISOR_KEY - assert body["model"] == "claude-opus-4-1-20250805" - assert body["messages"] == [ - {"role": "user", "content": migration}, - {"role": "user", "content": question}, - ] - assert "tools" not in body - return Reply( - body=json.dumps( - { - "id": f"msg-{identity}", - "type": "message", - "role": "assistant", - "model": "claude-opus-4-1-20250805", - "content": [{"type": "text", "text": _ADVICE}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - executor: Final = scenario.model(model="hosted_vllm/gpt-4o-mini", api_base=wire.url + "/v1") - advisor: Final = scenario.model( - model="anthropic/claude-opus-4-1-20250805", api_base=wire.url, api_key=_ADVISOR_KEY - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - { - "model": executor, - "max_tokens": 64, - "messages": [{"role": "user", "content": migration}], - "tools": [{"type": "advisor_20260301", "name": "advisor", "model": advisor}], - }, - ) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["content"] == [{"type": "text", "text": _FINAL_ANSWER}], response.text - assert body["stop_reason"] == "end_turn", response.text - assert [request.target for request in wire.drain()] == [ - "/v1/chat/completions", - "/v1/messages", - "/v1/chat/completions", - ] - - -def _advice_reply(identity: str) -> Reply: - return Reply( - body=json.dumps( - { - "id": f"msg-{identity}", - "type": "message", - "role": "assistant", - "model": "claude-opus-4-1-20250805", - "content": [{"type": "text", "text": _ADVICE}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - ).encode() - ) - - -@pytest.mark.covers("providers.anthropic_messages_advisor.caller_api_base_without_api_key_never_receives_the_proxy_key") -def test_advisor_api_base_without_api_key_is_rejected_before_the_proxy_anthropic_key_reaches_the_caller_host( - gateway: Gateway, tmp_path: Path -) -> None: - identity: Final = "advisor-leak-" + uuid.uuid4().hex - question: Final = _QUESTION + " " + identity - - def executor(request: Request) -> Reply: - assert request.target == "/v1/chat/completions", request.target - return _executor_reply(json.loads(request.body), identity, question) - - def caller_host(request: Request) -> Reply: - return _advice_reply(identity) - - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["general_settings"]["allow_client_side_credentials"] = True - path: Final = tmp_path / "client-side-credentials.yaml" - path.write_text(yaml.safe_dump(config)) - with ( - wire_server(executor) as executor_wire, - wire_server(caller_host) as caller_wire, - owned_proxy(gateway, tmp_path, {"ANTHROPIC_API_KEY": _PROXY_ANTHROPIC_KEY}, config=path) as candidate, - candidate.scenario() as scenario, - ): - model: Final = scenario.model(model="hosted_vllm/gpt-4o-mini", api_base=executor_wire.url + "/v1") - response: Final = candidate.request( - "POST", - "/v1/messages", - { - "model": model, - "max_tokens": 64, - "messages": [{"role": "user", "content": "please plan the migration"}], - "tools": [ - { - "type": "advisor_20260301", - "name": "advisor", - "model": "anthropic/claude-opus-4-1-20250805", - "api_base": caller_wire.url, - } - ], - }, - ) - received: Final = caller_wire.drain() - assert [ - (request.target, request.headers.get("x-api-key"), json.loads(request.body)["messages"]) - for request in received - ] == [], response.text - assert response.is_error, response.text - assert response.json() == { - "type": "error", - "error": { - "type": "api_error", - "message": ( - "advisor tool definition sets 'api_base' without 'api_key'. A caller-supplied api_base is only " - "honored alongside a caller-supplied api_key, so the proxy's own credentials are never sent to a " - "caller-chosen destination." - ), - }, - }, response.text - assert executor_wire.drain() == (), response.text diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py deleted file mode 100644 index c6ac00c20e8..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_compaction_wire.py +++ /dev/null @@ -1,108 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_CONTEXT_MANAGEMENT: Final = { - "edits": [ - {"type": "clear_thinking_20251015", "keep": "all"}, - {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, - ] -} -_COMPACTION_BLOCK: Final = {"type": "compaction", "content": ""} - - -def _compaction_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": dict(_COMPACTION_BLOCK)}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 5}, - "context_management": { - "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] - }, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gateway: Gateway) -> None: - identity: Final = f"msg_cm_{uuid.uuid4().hex}" - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "context_management": _CONTEXT_MANAGEMENT, - } - turn3: Final = cc.tool_loop_turn2( - request_body, - ( - dict(_COMPACTION_BLOCK), - {"type": "text", "text": "continuing after compaction"}, - ), - (), - ) - first_expected: Final = {**request_body, "model": cc.FABLE} - second_expected: Final = {**turn3, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == first_expected: - assert body["context_management"] == _CONTEXT_MANAGEMENT - return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) - assert body == second_expected, { - key: (second_expected.get(key), body.get(key)) - for key in second_expected.keys() | body.keys() - if second_expected.get(key) != body.get(key) - } - assert body["context_management"] == _CONTEXT_MANAGEMENT - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream("msg_cm_next", cc.FABLE, "OK", {"input_tokens": 20, "output_tokens": 2}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**request_body, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - events: Final = cc.sse_events(response1.text) - assert events[1][1]["content_block"] == _COMPACTION_BLOCK, events[1] - deltas: Final = [data for event, data in events if event == "message_delta"] - assert len(deltas) == 1 and deltas[0].get("context_management") == { - "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] - }, events - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py deleted file mode 100644 index 48b914a7374..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_document_input_wire.py +++ /dev/null @@ -1,133 +0,0 @@ -import base64 -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_PDF_BYTES: Final = ( - b"%PDF-1.1\n" - b"1 0 obj<>endobj\n" - b"2 0 obj<>endobj\n" - b"3 0 obj<>endobj\n" - b"trailer<>\n%%EOF" -) -_DOC_BLOCK: Final = { - "type": "document", - "source": {"type": "base64", "data": base64.b64encode(_PDF_BYTES).decode(), "media_type": "application/pdf"}, - "citations": {"enabled": True}, -} - - -def _cited_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "A page."}}, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 0, - "delta": { - "type": "citations_delta", - "citation": { - "type": "page_location", - "document_index": 0, - "document_title": "dot.pdf", - "start_page_number": 1, - "end_page_number": 1, - "cited_text": "Page", - }, - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 6}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_base64_pdf_document_with_citations_reaches_anthropic_identical(gateway: Gateway) -> None: - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "messages": [ - { - "role": "user", - "content": [ - dict(_DOC_BLOCK), - {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, - ], - } - ], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply(content_type="text/event-stream", chunks=_cited_stream(f"msg_doc_{uuid.uuid4().hex}")) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - citations: Final = [ - data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" - ] - assert citations == [ - { - "type": "citations_delta", - "citation": { - "type": "page_location", - "document_index": 0, - "document_title": "dot.pdf", - "start_page_number": 1, - "end_page_number": 1, - "cited_text": "Page", - }, - } - ], citations - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py deleted file mode 100644 index 8a929ee2a06..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_image_input_wire.py +++ /dev/null @@ -1,67 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" -_IMAGE_BLOCK: Final = { - "type": "image", - "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, -} - - -def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gateway: Gateway) -> None: - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", - ) - with_image_result: Final = cc.tool_loop_turn2( - turn1, - ({"type": "tool_use", "id": "toolu_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), - (("toolu_img", [dict(_IMAGE_BLOCK)]),), - ) - pasted: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "messages": [ - { - "role": "user", - "content": [ - dict(_IMAGE_BLOCK), - {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, - ], - } - ], - } - first_expected: Final = {**with_image_result, "model": cc.FABLE} - second_expected: Final = {**pasted, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body != first_expected: - assert body == second_expected, body - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream( - f"msg_img_{uuid.uuid4().hex}", cc.FABLE, "RED", {"input_tokens": 20, "output_tokens": 2} - ), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**with_image_result, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - response2: Final = gateway.request( - "POST", "/v1/messages", {**pasted, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py deleted file mode 100644 index 6990f33e51c..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_interleaved_thinking_history_wire.py +++ /dev/null @@ -1,167 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server -from pydantic import JsonValue - -_TURN1_TOOL: Final = ("toolu_a", "Read", {"file_path": "/tmp/cc_probe/a.txt"}) -_TURN2_TOOL: Final = ("toolu_b", "Read", {"file_path": "/tmp/cc_probe/b.txt"}) - - -def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: - return { - key: {"expected": expected.get(key), "upstream": body.get(key)} - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - - -def _turn2(base: dict[str, JsonValue]) -> dict[str, JsonValue]: - return cc.tool_loop_turn2( - base, - ( - {"type": "thinking", "thinking": "plan", "signature": "sig1"}, - {"type": "tool_use", "id": _TURN1_TOOL[0], "name": _TURN1_TOOL[1], "input": _TURN1_TOOL[2]}, - ), - ((_TURN1_TOOL[0], "ALPHA"),), - ) - - -def _turn3(turn2: dict[str, JsonValue]) -> dict[str, JsonValue]: - return cc.tool_loop_turn2( - turn2, - ( - {"type": "thinking", "thinking": "got A", "signature": "sig2"}, - {"type": "text", "text": "got A"}, - {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": _TURN2_TOOL[2]}, - ), - ((_TURN2_TOOL[0], "BRAVO"),), - ) - - -def _interleaved_stream(identity: str) -> tuple[bytes, ...]: - usage: Final = {"input_tokens": 20, "output_tokens": 12} - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": usage["input_tokens"], "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "got A"}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": "sig2"}}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 1, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 1, "delta": {"type": "text_delta", "text": "got A"}}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), - cc.sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": 2, - "content_block": {"type": "tool_use", "id": _TURN2_TOOL[0], "name": _TURN2_TOOL[1], "input": {}}, - }, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 2, - "delta": {"type": "input_json_delta", "partial_json": '{"file_path": "/tmp/cc_pr'}, - }, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 2, - "delta": {"type": "input_json_delta", "partial_json": 'obe/b.txt"}'}, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "tool_use", "stop_sequence": None}, - "usage": {"output_tokens": usage["output_tokens"]}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_interleaved_thinking_text_and_tool_use_history_reaches_anthropic_identical(gateway: Gateway) -> None: - identity: Final = f"msg_il_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words", - ) - turn2: Final = _turn2(turn1) - turn3: Final = _turn3(turn2) - turn2_expected: Final = {**turn2, "model": cc.FABLE} - turn3_expected: Final = {**turn3, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - upstream_beta: Final = request.headers.get("anthropic-beta", "") - assert upstream_beta.split(",").count("interleaved-thinking-2025-05-14") == 1, upstream_beta - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == turn2_expected: - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}), - ) - assert body == turn3_expected, _diff(turn3_expected, body) - return Reply(content_type="text/event-stream", chunks=_interleaved_stream(identity)) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - response3: Final = gateway.request( - "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response3.status_code == 200, response3.text - events: Final = cc.sse_events(response3.text) - started: Final = [ - (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" - ] - assert started == [(0, "thinking"), (1, "text"), (2, "tool_use")], started - assert events[-1][0] == "message_stop" - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (turn2_expected, turn3_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py deleted file mode 100644 index 242e5c7ec5a..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py +++ /dev/null @@ -1,77 +0,0 @@ -import json -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_MODEL: Final = "claude-sonnet-4-6" -_KEY: Final = "synthetic-anthropic-key" -_THINKING: Final = {"type": "enabled", "budget_tokens": 8000} -_TOOL: Final = { - "name": "read_file", - "description": "read a file", - "input_schema": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}, -} -_NEXT_CALL: Final = {"type": "tool_use", "id": "call-2", "name": "read_file", "input": {"path": "schema.prisma"}} - - -def _tool_loop_history(identity: str) -> tuple[dict[str, object], ...]: - return ( - {"role": "user", "content": f"open the config for {identity}"}, - { - "role": "assistant", - "content": [{"type": "tool_use", "id": "call-1", "name": "read_file", "input": {"path": "config.yaml"}}], - }, - {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "call-1", "content": "model_list: []"}]}, - ) - - -def _tool_use_reply(identity: str) -> Reply: - return Reply( - body=json.dumps( - { - "id": f"msg-{identity}", - "type": "message", - "role": "assistant", - "model": _MODEL, - "content": [_NEXT_CALL], - "stop_reason": "tool_use", - "stop_sequence": None, - "usage": {"input_tokens": 40, "output_tokens": 12}, - } - ).encode() - ) - - -@pytest.mark.covers("providers.anthropic_messages.claude_4_6_legacy_thinking_budget_reaches_the_wire_unchanged") -def test_claude_4_6_thinking_budget_tokens_on_messages_is_forwarded_instead_of_rewritten_to_adaptive( - gateway: Gateway, -) -> None: - identity: Final = "legacy-thinking-" + uuid.uuid4().hex - history: Final = _tool_loop_history(identity) - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == _KEY - body: Final = json.loads(request.body) - assert body["thinking"] == _THINKING, body - assert "output_config" not in body, body - assert body["max_tokens"] == 32768, body - assert body["messages"] == list(history), body - assert body["tools"] == [_TOOL], body - return _tool_use_reply(identity) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {"model": model, "max_tokens": 32768, "thinking": _THINKING, "messages": history, "tools": [_TOOL]}, - ) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["content"] == [_NEXT_CALL], response.text - assert body["stop_reason"] == "tool_use", response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py deleted file mode 100644 index 7fe405ca439..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_long_context_beta_wire.py +++ /dev/null @@ -1,53 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" - - -def test_1m_context_beta_forwarded_and_tiered_prompt_priced_above_200k(gateway: Gateway) -> None: - identity: Final = f"msg_1m_{uuid.uuid4().hex}" - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - upstream_beta: Final = request.headers.get("anthropic-beta", "") - assert upstream_beta.split(",").count("context-1m-2025-08-07") == 1, upstream_beta - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 250000, "output_tokens": 100}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - input_cost_per_token_above_200k_tokens=2e-6, - output_cost_per_token=5e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, _BETA_1M), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(250000 * 2e-6 + 100 * 5e-6), dict(rows[0]) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py deleted file mode 100644 index cb7043c0362..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py +++ /dev/null @@ -1,82 +0,0 @@ -import json -import threading -import uuid -from typing import Final - -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_MODEL: Final = "claude-sonnet-4-5-20250929" -_API_KEY: Final = "synthetic-anthropic-key" - - -def _sse(event: str, payload: dict[str, object]) -> bytes: - return f"event: {event}\ndata: {json.dumps(payload)}\n\n".encode() - - -def test_messages_stream_message_start_reaches_client_before_content_without_fallback( - gateway: Gateway, -) -> None: - """With no fallback able to take over, the proxy must not hold lifecycle - frames back for a retry that cannot happen: message_start reaches the - client while the upstream is still thinking.""" - gate: Final = threading.Event() - head: Final = _sse("message_start", {"type": "message_start", "message": {"id": "msg_live_1"}}) + _sse( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ) - tail: Final = ( - _sse( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "Hello"}}, - ) - + _sse("content_block_stop", {"type": "content_block_stop", "index": 0}) - + _sse( - "message_delta", - {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 3}}, - ) - + _sse("message_stop", {"type": "message_stop"}) - ) - prompt: Final = "live-lifecycle-" + uuid.uuid4().hex - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == _API_KEY - body: Final = json.loads(request.body) - assert body["model"] == _MODEL - assert body["stream"] is True - assert body["messages"] == [{"role": "user", "content": prompt}] - return Reply(content_type="text/event-stream", chunks=(head, tail), gate_after_first=gate) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_API_KEY) - with gateway.client.stream( - "POST", - "/v1/messages", - json={ - "model": model, - "max_tokens": 16, - "stream": True, - "messages": [{"role": "user", "content": prompt}], - }, - headers={"Authorization": f"Bearer {gateway.key}"}, - ) as response: - assert response.status_code == 200, response.read().decode() - lines = response.iter_lines() - first_event: Final = next( - json.loads(line.removeprefix("data: ")) for line in lines if line.startswith("data: ") - ) - assert first_event["type"] == "message_start" - gate.set() - events: Final = (first_event,) + tuple( - json.loads(line.removeprefix("data: ")) for line in lines if line.startswith("data: ") - ) - assert tuple(event["type"] for event in events) == ( - "message_start", - "content_block_start", - "content_block_delta", - "content_block_stop", - "message_delta", - "message_stop", - ), f"observed events: {events!r}" - assert [request.target for request in wire.drain()] == ["/v1/messages"] diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py deleted file mode 100644 index 76c29ad7763..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py +++ /dev/null @@ -1,58 +0,0 @@ -import json -import time -import uuid -from typing import Final - -import pytest -from integration._support.client import JSON_OBJECT, Gateway -from integration._support.wire import Reply, Request, wire_server - -_UPSTREAM_STALL_SECONDS: Final = 4.0 -_CONFIGURED_TIMEOUT_SECONDS: Final = 1.0 - - -@pytest.mark.covers("providers.anthropic_messages.configured_timeout_aborts_stalled_upstream") -def test_messages_endpoint_honors_configured_timeout_against_stalled_upstream(gateway: Gateway) -> None: - prompt: Final = "stall-" + uuid.uuid4().hex - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == "synthetic-anthropic-key" - body: Final = JSON_OBJECT.validate_json(request.body) - assert body["model"] == "claude-sonnet-4-5-20250929" - assert body["messages"] == [{"role": "user", "content": prompt}] - assert body["max_tokens"] == 16 - assert "timeout" not in body - time.sleep(_UPSTREAM_STALL_SECONDS) - return Reply( - body=json.dumps( - { - "id": "msg_stalled", - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": [{"type": "text", "text": "too late"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 1, "output_tokens": 2}, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", - api_base=wire.url, - api_key="synthetic-anthropic-key", - timeout=_CONFIGURED_TIMEOUT_SECONDS, - ) - started: Final = time.monotonic() - response: Final = gateway.request( - "POST", - "/v1/messages", - {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, - ) - elapsed: Final = time.monotonic() - started - assert response.status_code == 408, response.text - assert elapsed < _UPSTREAM_STALL_SECONDS, f"timed out only after {elapsed:.2f}s: {response.text}" - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py deleted file mode 100644 index 5753d395bc4..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_model_switch_history_wire.py +++ /dev/null @@ -1,83 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - - -def test_mid_loop_model_switch_replays_history_byte_identical(gateway: Gateway) -> None: - identity1: Final = f"msg_sw1_{uuid.uuid4().hex}" - identity2: Final = f"msg_sw2_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - first_expected: Final = {**turn1, "model": cc.FABLE} - second_expected: Final = {**turn2, "model": cc.OPUS} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == first_expected: - return Reply( - content_type="text/event-stream", - chunks=cc.tool_use_stream( - identity1, - cc.FABLE, - "need to read the file", - "sig_anthropic_1", - (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), - {"input_tokens": 20, "output_tokens": 10}, - ), - ) - assert body == second_expected, { - key: (second_expected.get(key), body.get(key)) - for key in second_expected.keys() | body.keys() - if second_expected.get(key) != body.get(key) - } - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity2, cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies - rows: Final = eventually( - lambda: read_rows( - 'SELECT model FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity2,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["model"] == f"anthropic/{cc.OPUS}" diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py deleted file mode 100644 index 06318b52aee..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_prompt_cache_billing_wire.py +++ /dev/null @@ -1,85 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_USAGE: Final = { - "input_tokens": 10, - "cache_read_input_tokens": 3000, - "cache_creation_input_tokens": 200, - "output_tokens": 5, -} - - -def test_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: - identity: Final = f"msg_pc_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**turn2, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply(content_type="text/event-stream", chunks=cc.text_stream(identity, cc.FABLE, "PROBE", _USAGE)) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - output_cost_per_token=5e-6, - cache_read_input_token_cost=1e-7, - cache_creation_input_token_cost=1.25e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - usage: Final = events[0][1]["message"]["usage"] - assert usage["cache_read_input_tokens"] == 3000, usage - assert usage["cache_creation_input_tokens"] == 200, usage - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 3000 * 1e-7 + 200 * 1.25e-6 + 5 * 5e-6), dict( - rows[0] - ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py deleted file mode 100644 index a5a57737fc1..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_request_fidelity_wire.py +++ /dev/null @@ -1,71 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_MODEL: Final = cc.SONNET - - -def test_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: - identity: Final = f"msg_cc_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - cli_beta: Final = frozenset(cc.CLI_BETA.split(",")) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY - assert request.headers["anthropic-version"] == "2023-06-01" - assert frozenset(request.headers.get("anthropic-beta", "").split(",")) == cli_beta, request.headers.get( - "anthropic-beta" - ) - assert "authorization" not in request.headers, dict(request.headers) - assert all(gateway.key not in value for value in request.headers.values()), dict(request.headers) - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": _MODEL} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, _MODEL, "PONG", {"input_tokens": 12, "output_tokens": 4}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - assert response.headers["content-type"].startswith("text/event-stream"), dict(response.headers) - events: Final = cc.sse_events(response.text) - assert [event for event, _ in events] == [ - "message_start", - "content_block_start", - "content_block_delta", - "content_block_stop", - "message_delta", - "message_stop", - ] - assert events[2][1]["delta"] == {"type": "text_delta", "text": "PONG"} - assert events[4][1]["delta"]["stop_reason"] == "end_turn" - assert events[4][1]["usage"]["output_tokens"] == 4 - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py deleted file mode 100644 index e414d8f0d11..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py +++ /dev/null @@ -1,95 +0,0 @@ -import json -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -MODEL: Final = "claude-sonnet-4-5-20250929" -KEY: Final = "synthetic-anthropic-key" -SIGNATURE_ERROR: Final = json.dumps( - { - "type": "error", - "error": { - "type": "invalid_request_error", - "message": "messages.2.content.0.thinking.signature.str: Input should be a valid string", - }, - } -).encode() -TOOLS: Final = ({"name": "lookup", "input_schema": {"type": "object", "properties": {"key": {"type": "string"}}}},) - - -def _history_with_unsigned_thinking(identity: str) -> tuple[dict[str, object], ...]: - return ( - {"role": "user", "content": [{"type": "text", "text": f"first question {identity}"}]}, - {"role": "assistant", "content": [{"type": "text", "text": "first answer"}]}, - {"role": "user", "content": [{"type": "text", "text": "second question"}]}, - { - "role": "assistant", - "content": [ - {"type": "thinking", "thinking": "replayed from another provider", "signature": None}, - {"type": "tool_use", "id": "call-1", "name": "lookup", "input": {"key": "value"}}, - ], - }, - {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "call-1", "content": "found"}]}, - ) - - -@pytest.mark.covers("providers.anthropic_messages.missing_thinking_signature_400_retries_without_thinking_blocks") -def test_missing_thinking_signature_400_retries_once_without_thinking_blocks_and_returns_200( - gateway: Gateway, -) -> None: - identity: Final = "thinking-signature-" + uuid.uuid4().hex - history: Final = _history_with_unsigned_thinking(identity) - tool_use_only_turn: Final = { - "role": "assistant", - "content": [{"type": "tool_use", "id": "call-1", "name": "lookup", "input": {"key": "value"}}], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == KEY - body: Final = json.loads(request.body) - assert body["model"] == MODEL - assert body["tools"] == list(TOOLS), body - if body["messages"][3]["content"][0]["type"] == "thinking": - assert body["messages"] == list(history), body - assert body["thinking"] == {"type": "enabled", "budget_tokens": 1024}, body - return Reply(status=400, body=SIGNATURE_ERROR) - assert body["messages"] == [*history[:3], tool_use_only_turn, history[4]], body - assert "thinking" not in body, body - return Reply( - body=json.dumps( - { - "id": identity, - "type": "message", - "role": "assistant", - "model": MODEL, - "content": [{"type": "text", "text": "recovered without thinking history"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 30, "output_tokens": 6}, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{MODEL}", api_base=wire.url, api_key=KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - { - "model": model, - "max_tokens": 64, - "thinking": {"type": "enabled", "budget_tokens": 1024}, - "tools": list(TOOLS), - "messages": list(history), - }, - ) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["id"] == identity, response.text - assert body["content"] == [{"type": "text", "text": "recovered without thinking history"}], response.text - assert body["stop_reason"] == "end_turn", response.text - assert [request.target for request in wire.drain()] == ["/v1/messages", "/v1/messages"] diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py deleted file mode 100644 index f3fb9786e02..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_tool_loop_wire.py +++ /dev/null @@ -1,165 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from pydantic import JsonValue - -_THINKING: Final = "need to read the file" -_SIGNATURE: Final = "sig_probe_1" -_USAGE: Final = {"input_tokens": 20, "output_tokens": 10} - - -def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: - return { - key: {"expected": expected.get(key), "upstream": body.get(key)} - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - - -def test_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: - identity1: Final = f"msg_tl1_{uuid.uuid4().hex}" - identity2: Final = f"msg_tl2_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - calls: Final = (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - first_expected: Final = {**turn1, "model": cc.FABLE} - second_expected: Final = {**turn2, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == first_expected: - return Reply( - content_type="text/event-stream", - chunks=cc.tool_use_stream(identity1, cc.FABLE, _THINKING, _SIGNATURE, calls, _USAGE), - ) - assert body == second_expected, _diff(second_expected, body) - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity2, cc.FABLE, "PROBE", {"input_tokens": 30, "output_tokens": 3}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - events: Final = cc.sse_events(response1.text) - assert [ - ( - event, - data.get("delta", {}).get( - "type", data.get("content_block", {}).get("type", data.get("delta", {}).get("stop_reason")) - ), - ) - for event, data in events - ] == [ - ("message_start", None), - ("content_block_start", "thinking"), - ("content_block_delta", "thinking_delta"), - ("content_block_delta", "signature_delta"), - ("content_block_stop", None), - ("content_block_start", "tool_use"), - ("content_block_delta", "input_json_delta"), - ("content_block_delta", "input_json_delta"), - ("content_block_stop", None), - ("message_delta", "tool_use"), - ("message_stop", None), - ] - assert events[5][1]["content_block"]["id"] == "toolu_read_1" - assert events[5][1]["content_block"]["name"] == "Read" - partial: Final = events[6][1]["delta"]["partial_json"] + events[7][1]["delta"]["partial_json"] - assert partial == '{"file_path": "/tmp/cc_probe/hello.txt"}' - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - events2: Final = cc.sse_events(response2.text) - assert events2[2][1]["delta"] == {"type": "text_delta", "text": "PROBE"} - assert events2[4][1]["delta"]["stop_reason"] == "end_turn" - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity2,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["prompt_tokens"] == 30 - - -def test_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and /tmp/cc_probe/world.txt and reply with both words", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - { - "type": "tool_use", - "id": "toolu_read_2", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/world.txt"}, - }, - ), - (("toolu_read_2", "1\tPROBE2\n2\t"), ("toolu_read_1", "1\tPROBE\n2\t")), - ) - - def respond(request: Request) -> Reply: - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**turn2, "model": cc.FABLE} - assert body == expected, _diff(expected, body) - results: Final = [block for block in body["messages"][3]["content"] if block["type"] == "tool_result"] - assert [block["tool_use_id"] for block in results] == ["toolu_read_2", "toolu_read_1"] - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(f"msg_mt_{uuid.uuid4().hex}", cc.FABLE, "PROBE PROBE2", _USAGE), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py deleted file mode 100644 index 66852e6e21b..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_web_search_citations_wire.py +++ /dev/null @@ -1,170 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} - - -def _web_search_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1, "server_tool_use": {"web_search_requests": 1}}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": 0, - "content_block": {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {}}, - }, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 0, - "delta": {"type": "input_json_delta", "partial_json": '{"query": "current LiteLLM version"}'}, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": 1, - "content_block": { - "type": "web_search_tool_result", - "tool_use_id": "srvtoolu_1", - "content": [ - { - "type": "web_search_result", - "title": "litellm releases", - "url": "https://example.com/litellm", - "page_age": None, - "encrypted_content": "enc_ws_1", - } - ], - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 2, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 2, "delta": {"type": "text_delta", "text": "1.104.0"}}, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 2, - "delta": { - "type": "citations_delta", - "citation": { - "type": "web_search_result_location", - "url": "https://example.com/litellm", - "title": "litellm releases", - "cited_text": "version 1.104.0", - "encrypted_index": "eidx_1", - }, - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 15, "server_tool_use": {"web_search_requests": 1}}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: - identity: Final = f"msg_ws_{uuid.uuid4().hex}" - base: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Use web search to find the current LiteLLM version and answer in one word", - ) - request_body: Final = { - **base, - "tools": [*base["tools"], WEB_SEARCH_TOOL], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - assert body["tools"][-1] == WEB_SEARCH_TOOL - assert len({tool["name"] for tool in body["tools"]}) == len(body["tools"]), body["tools"] - return Reply(content_type="text/event-stream", chunks=_web_search_stream(identity)) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - output_cost_per_token=5e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - started: Final = [ - (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" - ] - assert started == [(0, "server_tool_use"), (1, "web_search_tool_result"), (2, "text")], started - citations: Final = [ - data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" - ] - assert len(citations) == 1 and citations[0]["citation"]["url"] == "https://example.com/litellm", citations - start_usage: Final = events[0][1]["message"]["usage"] - assert start_usage["server_tool_use"]["web_search_requests"] == 1, start_usage - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - token_cost: Final = 20 * 1e-6 + 15 * 5e-6 - assert float(rows[0]["spend"]) >= token_cost, dict(rows[0]) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py deleted file mode 100644 index 7e9c5be227a..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py +++ /dev/null @@ -1,206 +0,0 @@ -import json -import time -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway, eventually, object_value -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - - -@pytest.mark.covers( - "other.provider_wire.anthropic.tool_history_system_cache_and_internal_fields", - "quota_management.spend_tracking.cache_tokens.disjoint_classes_use_explicit_rates", -) -def test_anthropic_tool_history_and_cache_tokens_keep_wire_and_accounting_contracts(gateway: Gateway) -> None: - identity: Final = "anthropic-wire-" + uuid.uuid4().hex - tool_schema: Final = { - "type": "object", - "properties": {"x": {"type": "integer"}, "y": {"type": "integer"}}, - "required": ["x", "y"], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == "synthetic-anthropic-key" - body: Final = json.loads(request.body) - assert body["model"] == "claude-sonnet-4-5-20250929" - assert body["system"] == [{"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}}] - assert body["tools"][0]["name"] == "add" and body["tools"][0]["input_schema"] == tool_schema - assert body["max_tokens"] == 16 - assert not {"timeout", "stream_chunk_size", "litellm_params", "litellm_metadata", "rpm", "tpm"}.intersection( - body - ) - messages: Final = body["messages"] - assert [message["role"] for message in messages] == ["user", "assistant", "user"] - assert messages[0]["content"] == [{"type": "text", "text": "first"}] - assert messages[1]["content"] == [ - {"type": "tool_use", "id": "history-call", "name": "add", "input": {"x": 1, "y": 2}} - ] - assert messages[2]["content"] == [ - {"type": "tool_result", "tool_use_id": "history-call", "content": "3"}, - {"type": "text", "text": "next"}, - ] - return Reply( - body=json.dumps( - { - "id": identity, - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": [{"type": "tool_use", "id": "next-call", "name": "add", "input": {"x": 3, "y": 4}}], - "stop_reason": "tool_use", - "stop_sequence": None, - "usage": { - "input_tokens": 10, - "output_tokens": 4, - "cache_read_input_tokens": 5, - "cache_creation_input_tokens": 7, - }, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", - api_base=wire.url, - api_key="synthetic-anthropic-key", - input_cost_per_token=0.001, - output_cost_per_token=0.002, - cache_read_input_token_cost=0.0001, - cache_creation_input_token_cost=0.002, - ) - response: Final = gateway.request( - "POST", - "/v1/chat/completions", - { - "model": model, - "max_tokens": 16, - "timeout": 5, - "messages": [ - { - "role": "system", - "content": [ - {"type": "text", "text": "synthetic policy", "cache_control": {"type": "ephemeral"}} - ], - }, - {"role": "user", "content": "first"}, - { - "role": "assistant", - "tool_calls": [ - { - "id": "history-call", - "type": "function", - "function": {"name": "add", "arguments": '{"x":1,"y":2}'}, - } - ], - }, - {"role": "tool", "tool_call_id": "history-call", "content": "3"}, - {"role": "user", "content": "next"}, - ], - "tools": [{"type": "function", "function": {"name": "add", "parameters": tool_schema}}], - }, - ) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["id"].startswith("chatcmpl-") - assert body["choices"][0]["finish_reason"] == "tool_calls" - tool: Final = body["choices"][0]["message"]["tool_calls"][0] - assert tool["id"] == "next-call" and tool["function"]["name"] == "add" - assert json.loads(tool["function"]["arguments"]) == {"x": 3, "y": 4} - assert body["usage"]["prompt_tokens"] == 22 and body["usage"]["completion_tokens"] == 4 - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens, metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (body["id"],), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(10 * 0.001 + 5 * 0.0001 + 7 * 0.002 + 4 * 0.002) - assert rows[0]["prompt_tokens"] == 22 and rows[0]["completion_tokens"] == 4 - metadata: Final = rows[0]["metadata"] - parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) - assert parsed["cost_breakdown"]["input_cost"] == pytest.approx(0.0245) - assert parsed["cost_breakdown"]["output_cost"] == pytest.approx(0.008) - - -@pytest.mark.covers("other.provider_wire.anthropic.bare_string_content_item_is_client_error") -@pytest.mark.parametrize( - "text", [pytest.param("what type of file is this?", id="type_word"), pytest.param("hello", id="plain")] -) -def test_anthropic_bare_string_content_item_is_rejected_as_client_error_before_the_wire( - gateway: Gateway, text: str -) -> None: - def respond(request: Request) -> Reply: - raise AssertionError(f"upstream must not be reached: {request.target}") - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" - ) - response: Final = gateway.request( - "POST", - "/v1/chat/completions", - {"model": model, "max_tokens": 16, "timeout": 5, "messages": [{"role": "system", "content": [text]}]}, - ) - assert response.status_code == 400, response.text - assert wire.drain() == () - - -@pytest.mark.covers("other.provider_wire.anthropic.messages_request_timeout_reaches_transport") -def test_anthropic_messages_slow_upstream_is_cut_off_at_the_deployment_request_timeout(gateway: Gateway) -> None: - identity: Final = "anthropic-timeout-" + uuid.uuid4().hex - prompt: Final = f"slow answer {identity}" - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == "synthetic-anthropic-key" - body: Final = json.loads(request.body) - assert body["model"] == "claude-sonnet-4-5-20250929" - assert body["max_tokens"] == 16 - assert body["messages"] == [{"role": "user", "content": prompt}] - assert not { - "timeout", - "request_timeout", - "stream_chunk_size", - "litellm_params", - "litellm_metadata", - "rpm", - "tpm", - }.intersection(body) - time.sleep(1.5) - return Reply( - body=json.dumps( - { - "id": identity, - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": [{"type": "text", "text": "late"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 3, "output_tokens": 1}, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", - api_base=wire.url, - api_key="synthetic-anthropic-key", - request_timeout=0.3, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, - headers={"anthropic-version": "2023-06-01"}, - ) - assert response.status_code == 408, response.text - assert "Timeout" in response.json()["error"]["message"], response.text - assert eventually(wire.drain, lambda requests: len(requests) == 1, seconds=5, return_last_on_timeout=True) diff --git a/tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py deleted file mode 100644 index a6f098cf64f..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py +++ /dev/null @@ -1,409 +0,0 @@ -import json -from pathlib import Path -from typing import Final - -import pytest -import yaml -from integration._support.client import Gateway -from integration._support.process import owned_proxy -from integration._support.wire import Reply, Request, wire_server - -BEDROCK_MODEL: Final = "us.anthropic.claude-haiku-4-5-20251001-v1:0" -INVOKE_TARGET: Final = f"/model/{BEDROCK_MODEL}/invoke" -SEARCH_TARGET: Final = "/tavily/search" -SEARCH_RESULT: Final = { - "title": "Synthetic result", - "url": "https://example.test/result", - "content": "the snippet text", -} - - -def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: - frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) - return tuple( - ( - next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")), - json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), - ) - for frame in frames - ) - - -@pytest.mark.covers("other.provider_wire.bedrock.websearch_interception_streamed_capped_turn_ends_with_native_results") -def test_streamed_web_search_turn_capped_by_max_agentic_loops_ends_turn_with_snippets_and_ordered_blocks( - gateway: Gateway, tmp_path: Path -) -> None: - def respond(request: Request) -> Reply: - assert request.method == "POST", request.target - body: Final = json.loads(request.body) - if request.target == SEARCH_TARGET: - assert request.headers["authorization"] == "Bearer synthetic-tavily-key" - assert body["query"] == "query-0", body - return Reply(body=json.dumps({"query": "query-0", "results": [SEARCH_RESULT]}).encode()) - assert request.target == INVOKE_TARGET - assert request.headers["authorization"] == "Bearer synthetic-bedrock-token" - assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] - assert "stream" not in body, body - depth: Final = sum( - 1 - for message in body["messages"] - if isinstance(message["content"], list) - for block in message["content"] - if block["type"] == "tool_result" - ) - if depth == 1: - assert body["messages"][2]["content"] == [ - { - "type": "tool_result", - "tool_use_id": "toolu_0", - "content": "Title: Synthetic result\nURL: https://example.test/result\nSnippet: the snippet text", - } - ], body["messages"] - return Reply( - body=json.dumps( - { - "id": f"msg_{depth}", - "type": "message", - "role": "assistant", - "model": BEDROCK_MODEL, - "content": [ - {"type": "text", "text": f"turn-{depth}"}, - { - "type": "tool_use", - "id": f"toolu_{depth}", - "name": "litellm_web_search", - "input": {"query": f"query-{depth}"}, - }, - ], - "stop_reason": "tool_use", - "stop_sequence": None, - "usage": {"input_tokens": 10, "output_tokens": 4}, - } - ).encode() - ) - - with wire_server(respond) as wire: - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["search_tools"] = [ - { - "search_tool_name": "integration-search", - "litellm_params": { - "search_provider": "tavily", - "api_key": "synthetic-tavily-key", - "api_base": wire.url + "/tavily", - }, - } - ] - config["litellm_settings"].update( - { - "callbacks": ["websearch_interception"], - "websearch_interception_params": { - "enabled_providers": ["bedrock"], - "search_tool_name": "integration-search", - "max_agentic_loops": 1, - }, - } - ) - path: Final = tmp_path / "websearch.yaml" - path.write_text(yaml.safe_dump(config)) - with owned_proxy(gateway, tmp_path, {}, config=path) as candidate, candidate.scenario() as scenario: - model: Final = scenario.model( - model=f"bedrock/{BEDROCK_MODEL}", - api_key="synthetic-bedrock-token", - api_base=wire.url, - aws_region_name="us-east-1", - aws_bedrock_runtime_endpoint=wire.url, - ) - response: Final = candidate.request( - "POST", - "/v1/messages", - { - "model": model, - "max_tokens": 64, - "stream": True, - "messages": [{"role": "user", "content": "search control"}], - "tools": [{"type": "web_search_20250305", "name": "web_search"}], - }, - ) - assert response.status_code == 200, response.text - events: Final = sse_events(response.text) - assert [name for name, _ in events][:1] == ["message_start"], response.text - assert [name for name, _ in events][-2:] == ["message_delta", "message_stop"], response.text - for position, (name, event) in enumerate(events): - if name == "content_block_stop": - assert event["index"] in { - earlier_event["index"] - for earlier, earlier_event in events[:position] - if earlier == "content_block_start" - }, response.text - started: Final = tuple(event["content_block"] for name, event in events if name == "content_block_start") - search_ids: Final = tuple(block["id"] for block in started if block["type"] == "server_tool_use") - assert search_ids and all(search_id.startswith("srvtoolu_") for search_id in search_ids), response.text - assert started[-1] == {"type": "text", "text": ""}, response.text - assert started[:-1] == tuple( - block - for search_id in search_ids - for block in ( - {"type": "server_tool_use", "id": search_id, "name": "web_search", "input": {"query": "query-0"}}, - { - "type": "web_search_tool_result", - "tool_use_id": search_id, - "content": [ - { - "type": "web_search_result", - "url": "https://example.test/result", - "title": "Synthetic result", - "page_age": None, - "encrypted_content": "", - "snippet": "the snippet text", - } - ], - }, - ) - ), response.text - assert ( - "".join(event["delta"]["text"] for name, event in events if name == "content_block_delta") == "turn-1" - ), response.text - assert [event["delta"]["stop_reason"] for name, event in events if name == "message_delta"] == [ - "end_turn" - ], response.text - assert "litellm_web_search" not in response.text, response.text - assert [request.target for request in wire.drain()] == [INVOKE_TARGET, SEARCH_TARGET, INVOKE_TARGET] - - -import threading -import uuid -from typing import Final -from urllib.parse import parse_qs, urlsplit - -import httpx -import pytest -from integration._support.client import Gateway, eventually - -_QUERY: Final = "integration capped search" -_TEXT_BLOCK: Final = {"type": "text", "text": "searching once more"} -_NOT_INTERCEPTED: Final = "native tool reached the provider" -_FINAL_BLOCK: Final = {"type": "text", "text": "answered from the stored backend"} -_OWNED_RESULT_TEXT: Final = "Title: Owned result\nURL: https://owned.invalid/a\nSnippet: owned snippet" -_SEARCH_RESULT_BLOCK: Final = { - "type": "web_search_result", - "url": "https://owned.invalid/a", - "title": "Owned result", - "page_age": None, - "encrypted_content": "", - "snippet": "owned snippet", -} - - -def _search_tool_use(identity: str) -> dict[str, object]: - return {"type": "tool_use", "id": identity, "name": "litellm_web_search", "input": {"query": _QUERY}} - - -def _anthropic_reply(identity: str, content: list[dict[str, object]], stop_reason: str) -> Reply: - return Reply( - body=json.dumps( - { - "id": identity, - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": content, - "stop_reason": stop_reason, - "stop_sequence": None, - "usage": {"input_tokens": 10, "output_tokens": 4}, - } - ).encode() - ) - - -@pytest.mark.covers( - "other.provider_wire.anthropic.websearch_interception_capped_loop_ends_turn_without_internal_tool_use" -) -def test_capped_websearch_interception_loop_ends_turn_instead_of_exposing_internal_tool_use( - gateway: Gateway, tmp_path: Path -) -> None: - identity: Final = "websearch-wire-" + uuid.uuid4().hex - searched: Final = threading.Event() - - def respond(request: Request) -> Reply: - parts: Final = urlsplit(request.target) - if request.method == "GET" and parts.path == "/search": - assert parse_qs(parts.query)["q"] == [_QUERY], request.target - searched.set() - return Reply( - body=json.dumps( - { - "results": [ - {"title": "Owned result", "url": "https://owned.invalid/a", "content": "owned snippet"} - ] - } - ).encode() - ) - assert request.method == "POST" and parts.path == "/v1/messages", request.target - body: Final = json.loads(request.body) - if any(tool.get("type") == "web_search_20250305" for tool in body["tools"]): - return _anthropic_reply(identity, [{"type": "text", "text": _NOT_INTERCEPTED}], "end_turn") - assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] - return _anthropic_reply(identity, [_TEXT_BLOCK, _search_tool_use(identity)], "tool_use") - - def send(candidate: Gateway, model: str) -> httpx.Response: - return candidate.request( - "POST", - "/v1/messages", - { - "model": model, - "max_tokens": 64, - "messages": [{"role": "user", "content": identity + " attempt " + uuid.uuid4().hex}], - "tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 3}], - }, - ) - - def searched_through_proxy(response: httpx.Response) -> bool: - return searched.is_set() and _NOT_INTERCEPTED not in response.text - - with wire_server(respond) as wire: - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["search_tools"] = [ - { - "search_tool_name": "integration-searxng", - "litellm_params": {"search_provider": "searxng", "api_base": wire.url}, - } - ] - config["litellm_settings"].update( - { - "callbacks": ["websearch_interception"], - "websearch_interception_params": { - "enabled": True, - "enabled_providers": ["anthropic"], - "search_tool_name": "integration-searxng", - }, - } - ) - path: Final = tmp_path / "websearch.yaml" - path.write_text(yaml.safe_dump(config)) - with owned_proxy(gateway, tmp_path, {}, config=path) as candidate, candidate.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" - ) - response: Final = eventually(lambda: send(candidate, model), searched_through_proxy, seconds=40) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["stop_reason"] == "end_turn", response.text - content: Final = body["content"] - assert [block["type"] for block in content] == ["server_tool_use", "web_search_tool_result", "text"], ( - response.text - ) - assert content[0]["name"] == "web_search" and content[0]["input"] == {"query": _QUERY}, response.text - assert content[1]["tool_use_id"] == content[0]["id"], response.text - assert content[1]["content"] == [_SEARCH_RESULT_BLOCK], response.text - assert content[2] == _TEXT_BLOCK, response.text - targets: Final = tuple((request.method, urlsplit(request.target).path) for request in wire.drain()) - assert targets[-3:] == (("POST", "/v1/messages"), ("GET", "/search"), ("POST", "/v1/messages")), targets - - -@pytest.mark.covers("other.provider_wire.anthropic.websearch_interception_uses_database_search_tool_backend") -def test_database_created_search_tool_backend_receives_the_intercepted_query_over_a_same_named_config_tool( - gateway: Gateway, tmp_path: Path -) -> None: - identity: Final = "websearch-db-" + uuid.uuid4().hex - tool_name: Final = "integration-db-searxng-" + uuid.uuid4().hex - searched: Final = threading.Event() - - def respond(request: Request) -> Reply: - parts: Final = urlsplit(request.target) - if request.method == "GET" and parts.path == "/database/search": - assert parse_qs(parts.query)["q"] == [_QUERY], request.target - searched.set() - return Reply( - body=json.dumps( - { - "results": [ - {"title": "Owned result", "url": "https://owned.invalid/a", "content": "owned snippet"} - ] - } - ).encode() - ) - assert request.method == "POST" and parts.path == "/v1/messages", request.target - body: Final = json.loads(request.body) - if any(tool.get("type") == "web_search_20250305" for tool in body["tools"]): - return _anthropic_reply(identity, [{"type": "text", "text": _NOT_INTERCEPTED}], "end_turn") - assert [tool["name"] for tool in body["tools"]] == ["litellm_web_search"], body["tools"] - results: Final = [ - block - for message in body["messages"] - if isinstance(message["content"], list) - for block in message["content"] - if block["type"] == "tool_result" - ] - if not results: - return _anthropic_reply(identity, [_TEXT_BLOCK, _search_tool_use(identity)], "tool_use") - assert results == [{"type": "tool_result", "tool_use_id": identity, "content": _OWNED_RESULT_TEXT}], results - return _anthropic_reply(identity, [_FINAL_BLOCK], "end_turn") - - def send(candidate: Gateway, model: str) -> httpx.Response: - return candidate.request( - "POST", - "/v1/messages", - { - "model": model, - "max_tokens": 64, - "messages": [{"role": "user", "content": identity + " attempt " + uuid.uuid4().hex}], - "tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 3}], - }, - ) - - def searched_through_proxy(response: httpx.Response) -> bool: - return searched.is_set() and _NOT_INTERCEPTED not in response.text - - with wire_server(respond) as wire, gateway.scenario() as scenario: - created: Final = gateway.post( - "/search_tools", - { - "search_tool": { - "search_tool_name": tool_name, - "litellm_params": {"search_provider": "searxng", "api_base": wire.url + "/database"}, - } - }, - ) - scenario.cleanups.callback(gateway.request, "DELETE", f"/search_tools/{created['search_tool_id']}") - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["search_tools"] = [ - { - "search_tool_name": tool_name, - "litellm_params": {"search_provider": "searxng", "api_base": wire.url + "/config"}, - } - ] - config["litellm_settings"].update( - { - "callbacks": ["websearch_interception"], - "websearch_interception_params": { - "enabled": True, - "enabled_providers": ["anthropic"], - "search_tool_name": tool_name, - }, - } - ) - path: Final = tmp_path / "websearch-db.yaml" - path.write_text(yaml.safe_dump(config)) - environment: Final = {"ANTHROPIC_API_BASE": wire.url} - with owned_proxy(gateway, tmp_path, environment, config=path) as candidate, candidate.scenario() as models: - model: Final = models.model( - model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" - ) - response: Final = eventually(lambda: send(candidate, model), searched_through_proxy, seconds=40) - assert response.status_code == 200, response.text - body: Final = response.json() - assert body["stop_reason"] == "end_turn", response.text - assert body["content"][-1] == _FINAL_BLOCK, response.text - found: Final = [ - (result["url"], result["title"]) - for block in body["content"] - if block["type"] == "web_search_tool_result" - for result in block["content"] - ] - assert found == [("https://owned.invalid/a", "Owned result")], response.text - assert "litellm_web_search" not in response.text, response.text - targets: Final = tuple((request.method, urlsplit(request.target).path) for request in wire.drain()) - assert targets[-3:] == (("POST", "/v1/messages"), ("GET", "/database/search"), ("POST", "/v1/messages")), ( - targets - ) From 3cc45e02201417d0de9c788101e70295f8701373 Mon Sep 17 00:00:00 2001 From: kerry Date: Thu, 1 Oct 2026 00:44:34 +0000 Subject: [PATCH 18/19] test(anthropic): cover native reasoning translation, response and pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...opic_reasoning_request_translation_wire.py | 366 ++++++++++++++++++ .../test_anthropic_reasoning_response_wire.py | 137 +++++++ ..._anthropic_reasoning_token_pricing_wire.py | 71 ++++ 3 files changed, 574 insertions(+) create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py create mode 100644 tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py new file mode 100644 index 00000000000..f15cabda823 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py @@ -0,0 +1,366 @@ +import json +import uuid +from collections.abc import Mapping +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_HAIKU_4_5: Final = "claude-haiku-4-5" +_OPUS_4_5: Final = "claude-opus-4-5" +_OPUS_4_6: Final = "claude-opus-4-6" +_OPUS_4_7: Final = "claude-opus-4-7" +_FABLE_5_1: Final = "claude-fable-5-1" +_ADAPTIVE: Final = {"type": "adaptive", "display": "omitted"} +_ADAPTIVE_SUMMARIZED: Final = {"type": "adaptive", "display": "summarized"} + + +def _budget(tokens: int) -> dict[str, JsonValue]: + return {"type": "enabled", "budget_tokens": tokens} + + +def _client_body(**reasoning: JsonValue) -> dict[str, JsonValue]: + base: Final = { + key: value + for key, value in cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}").items() + if key != "thinking" + } + return {**base, "stream": False, **reasoning} + + +def _without(body: Mapping[str, JsonValue], *keys: str) -> dict[str, JsonValue]: + return {key: value for key, value in body.items() if key not in keys} + + +def _diff(expected: Mapping[str, JsonValue], body: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +def _forwarded_body( + gateway: Gateway, upstream_model: str, client_body: Mapping[str, JsonValue] +) -> dict[str, JsonValue]: + def respond(request: Request) -> Reply: + return Reply( + body=json.dumps( + { + "id": f"msg_{uuid.uuid4().hex}", + "type": "message", + "role": "assistant", + "model": upstream_model, + "content": [{"type": "text", "text": "PONG"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 4}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY + ) + response: Final = gateway.request("POST", "/v1/messages", {**client_body, "model": model}) + assert response.status_code == 200, response.text + received: Final = wire.drain() + assert len(received) == 1, received + return cc.JSON_OBJECT.validate_json(received[0].body) + + +def _assert_forwarded( + gateway: Gateway, + upstream_model: str, + client_body: dict[str, JsonValue], + expected_changes: Mapping[str, JsonValue], + removed: tuple[str, ...], +) -> None: + expected: Final = {**_without(client_body, *removed), **expected_changes, "model": upstream_model} + body: Final = _forwarded_body(gateway, upstream_model, client_body) + assert body == expected, _diff(expected, body) + + +@pytest.mark.parametrize( + ("upstream_model", "effort", "expected_changes", "removed"), + ( + pytest.param( + _OPUS_4_5, + "high", + {}, + ("thinking",), + id="opus-4.5-keeps-supported-effort-drops-adaptive", + ), + pytest.param( + _OPUS_4_5, + "xhigh", + {"thinking": _budget(8192)}, + ("output_config",), + id="opus-4.5-xhigh-falls-back-to-budget", + ), + pytest.param(_HAIKU_4_5, "low", {"thinking": _budget(1024)}, ("output_config",), id="haiku-4.5-low"), + pytest.param(_HAIKU_4_5, "medium", {"thinking": _budget(2048)}, ("output_config",), id="haiku-4.5-medium"), + pytest.param(_HAIKU_4_5, "high", {"thinking": _budget(4096)}, ("output_config",), id="haiku-4.5-high"), + pytest.param(_HAIKU_4_5, "xhigh", {"thinking": _budget(8192)}, ("output_config",), id="haiku-4.5-xhigh"), + pytest.param(_HAIKU_4_5, "max", {"thinking": _budget(16384)}, ("output_config",), id="haiku-4.5-max"), + pytest.param(_OPUS_4_6, "high", {}, (), id="opus-4.6-adaptive-unchanged"), + pytest.param(_OPUS_4_7, "xhigh", {}, (), id="opus-4.7-adaptive-unchanged"), + ), +) +def test_adaptive_thinking_and_effort_are_reshaped_only_for_models_without_adaptive_thinking( + gateway: Gateway, + upstream_model: str, + effort: str, + expected_changes: dict[str, JsonValue], + removed: tuple[str, ...], +) -> None: + client_body: Final = _client_body(thinking=dict(_ADAPTIVE), output_config={"effort": effort}) + _assert_forwarded(gateway, upstream_model, client_body, expected_changes, removed) + + +def test_adaptive_effort_fallback_budget_is_capped_below_max_tokens(gateway: Gateway) -> None: + client_body: Final = _client_body(thinking=dict(_ADAPTIVE), output_config={"effort": "max"}, max_tokens=4000) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {"thinking": _budget(3999)}, ("output_config",)) + + +def test_adaptive_effort_fallback_drops_thinking_when_max_tokens_cannot_fit_the_minimum_budget( + gateway: Gateway, +) -> None: + client_body: Final = _client_body(thinking=dict(_ADAPTIVE), output_config={"effort": "high"}, max_tokens=1024) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {}, ("thinking", "output_config")) + + +@pytest.mark.parametrize( + ("budget_tokens", "effort"), + ( + pytest.param(1024, "low", id="below-medium-threshold"), + pytest.param(2048, "medium", id="medium-threshold"), + pytest.param(4096, "high", id="high-threshold"), + pytest.param(8192, "xhigh", id="xhigh-threshold"), + ), +) +def test_legacy_thinking_budget_becomes_adaptive_effort_on_models_that_reject_budgets( + gateway: Gateway, budget_tokens: int, effort: str +) -> None: + client_body: Final = _client_body(thinking=_budget(budget_tokens)) + _assert_forwarded( + gateway, + _OPUS_4_7, + client_body, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": effort}}, + (), + ) + + +def test_legacy_thinking_translation_keeps_the_callers_effort(gateway: Gateway) -> None: + client_body: Final = _client_body(thinking=_budget(8192), output_config={"effort": "medium"}) + _assert_forwarded(gateway, _OPUS_4_7, client_body, {"thinking": {"type": "adaptive"}}, ()) + + +@pytest.mark.parametrize( + ("upstream_model", "removed"), + ( + pytest.param(_FABLE_5_1, ("thinking",), id="always-on-model-drops-disabled"), + pytest.param(_OPUS_4_7, (), id="other-model-keeps-disabled"), + ), +) +def test_disabled_thinking_is_dropped_only_for_always_on_thinking_models( + gateway: Gateway, upstream_model: str, removed: tuple[str, ...] +) -> None: + client_body: Final = _client_body(thinking={"type": "disabled"}) + _assert_forwarded(gateway, upstream_model, client_body, {}, removed) + + +@pytest.mark.parametrize( + ("reasoning_effort", "effort"), + ( + pytest.param("minimal", "low", id="minimal"), + pytest.param("low", "low", id="low"), + pytest.param("medium", "medium", id="medium"), + pytest.param("high", "high", id="high"), + pytest.param("xhigh", "xhigh", id="xhigh"), + pytest.param("max", "max", id="max"), + ), +) +def test_reasoning_effort_becomes_adaptive_thinking_and_effort_on_adaptive_models( + gateway: Gateway, reasoning_effort: str, effort: str +) -> None: + client_body: Final = _client_body(reasoning_effort=reasoning_effort) + _assert_forwarded( + gateway, + _OPUS_4_7, + client_body, + {"thinking": dict(_ADAPTIVE_SUMMARIZED), "output_config": {"effort": effort}}, + ("reasoning_effort",), + ) + + +@pytest.mark.parametrize( + ("reasoning_effort", "budget_tokens"), + ( + pytest.param("minimal", 1024, id="minimal"), + pytest.param("low", 1024, id="low"), + pytest.param("medium", 2048, id="medium"), + pytest.param("high", 4096, id="high"), + pytest.param("xhigh", 8192, id="xhigh"), + pytest.param("max", 16384, id="max"), + ), +) +def test_reasoning_effort_becomes_a_thinking_budget_on_models_without_adaptive_thinking( + gateway: Gateway, reasoning_effort: str, budget_tokens: int +) -> None: + client_body: Final = _client_body(reasoning_effort=reasoning_effort) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {"thinking": _budget(budget_tokens)}, ("reasoning_effort",)) + + +def test_reasoning_effort_none_clears_thinking_and_effort(gateway: Gateway) -> None: + client_body: Final = _client_body( + reasoning_effort="none", thinking=dict(_ADAPTIVE), output_config={"effort": "high"} + ) + _assert_forwarded(gateway, _OPUS_4_7, client_body, {}, ("reasoning_effort", "thinking", "output_config")) + + +def test_caller_thinking_wins_over_reasoning_effort(gateway: Gateway) -> None: + client_body: Final = _client_body(reasoning_effort="high", thinking=_budget(2000)) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {}, ("reasoning_effort",)) + + +def test_caller_effort_wins_over_reasoning_effort(gateway: Gateway) -> None: + client_body: Final = _client_body(reasoning_effort="high", output_config={"effort": "low"}) + _assert_forwarded(gateway, _OPUS_4_7, client_body, {"thinking": dict(_ADAPTIVE_SUMMARIZED)}, ("reasoning_effort",)) + + +def test_reasoning_effort_budget_is_capped_below_max_tokens(gateway: Gateway) -> None: + client_body: Final = _client_body(reasoning_effort="max", max_tokens=4000) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {"thinking": _budget(3999)}, ("reasoning_effort",)) + + +def test_reasoning_effort_is_dropped_when_max_tokens_cannot_fit_the_minimum_budget(gateway: Gateway) -> None: + client_body: Final = _client_body(reasoning_effort="high", max_tokens=1024) + _assert_forwarded(gateway, _HAIKU_4_5, client_body, {}, ("reasoning_effort",)) + + +@pytest.mark.parametrize( + ("upstream_model", "reasoning", "expected_changes", "removed"), + ( + pytest.param( + _HAIKU_4_5, + {"thinking": dict(_ADAPTIVE), "output_config": {"effort": "high"}}, + {"thinking": _budget(4096)}, + ("output_config", "temperature"), + id="haiku-4.5-effort-translated-to-budget", + ), + pytest.param( + _OPUS_4_5, + {"thinking": dict(_ADAPTIVE), "output_config": {"effort": "high"}}, + {}, + ("thinking", "temperature"), + id="opus-4.5-effort-kept", + ), + pytest.param(_HAIKU_4_5, {"thinking": _budget(2048)}, {}, ("temperature",), id="haiku-4.5-legacy-budget"), + ), +) +def test_non_default_temperature_is_dropped_when_a_non_adaptive_model_thinks( + gateway: Gateway, + upstream_model: str, + reasoning: dict[str, JsonValue], + expected_changes: dict[str, JsonValue], + removed: tuple[str, ...], +) -> None: + client_body: Final = _client_body(temperature=0, **reasoning) + _assert_forwarded(gateway, upstream_model, client_body, expected_changes, removed) + + +@pytest.mark.parametrize( + ("upstream_model", "temperature", "reasoning"), + ( + pytest.param(_HAIKU_4_5, 1, {"thinking": _budget(2048)}, id="temperature-1-with-thinking"), + pytest.param(_HAIKU_4_5, 0, {}, id="temperature-0-without-thinking"), + pytest.param( + _OPUS_4_6, + 0, + {"thinking": dict(_ADAPTIVE), "output_config": {"effort": "high"}}, + id="adaptive-model", + ), + ), +) +def test_temperature_is_kept_when_it_does_not_conflict_with_thinking( + gateway: Gateway, upstream_model: str, temperature: int, reasoning: dict[str, JsonValue] +) -> None: + client_body: Final = _client_body(temperature=temperature, **reasoning) + _assert_forwarded(gateway, upstream_model, client_body, {}, ()) + + +_SIGNED_THINKING: Final = {"type": "thinking", "thinking": "check the config first", "signature": "EqQBCkgIBRABGAIiQL"} +_TOOL_CALL: Final = {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}} +_TOOL_RESULT: Final = {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_01", "content": "ok"}]} + + +def _history_body(assistant_content: tuple[dict[str, JsonValue], ...]) -> dict[str, JsonValue]: + base: Final = _client_body(thinking=_budget(2048)) + first_turn: Final = base["messages"] + assert isinstance(first_turn, list) + return {**base, "messages": [*first_turn, {"role": "assistant", "content": list(assistant_content)}, _TOOL_RESULT]} + + +def _with_assistant_content( + body: Mapping[str, JsonValue], assistant_content: tuple[dict[str, JsonValue], ...] +) -> dict[str, JsonValue]: + messages: Final = body["messages"] + assert isinstance(messages, list) + return { + **body, + "messages": [*messages[:-2], {"role": "assistant", "content": list(assistant_content)}, messages[-1]], + } + + +def test_encrypted_reasoning_from_another_provider_is_stripped_and_anthropic_signed_thinking_is_kept( + gateway: Gateway, +) -> None: + client_body: Final = _history_body( + ( + {"type": "thinking", "thinking": "bridge reasoning", "signature": "litellm_encrypted_reasoning:gAAAAB"}, + {"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:gAAAAC"}, + _SIGNED_THINKING, + _TOOL_CALL, + ) + ) + expected: Final = {**_with_assistant_content(client_body, (_SIGNED_THINKING, _TOOL_CALL)), "model": _HAIKU_4_5} + body: Final = _forwarded_body(gateway, _HAIKU_4_5, client_body) + assert body == expected, _diff(expected, body) + + +def test_empty_thinking_block_is_stripped_and_redacted_thinking_is_kept(gateway: Gateway) -> None: + redacted: Final = {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"} + client_body: Final = _history_body( + ({"type": "thinking", "thinking": "", "signature": "EqQBCkgIBRABGAIiQM"}, redacted, _TOOL_CALL) + ) + expected: Final = {**_with_assistant_content(client_body, (redacted, _TOOL_CALL)), "model": _HAIKU_4_5} + body: Final = _forwarded_body(gateway, _HAIKU_4_5, client_body) + assert body == expected, _diff(expected, body) + + +@pytest.mark.parametrize( + ("upstream_model", "reasoning_effort"), + ( + pytest.param(_HAIKU_4_5, "turbo", id="unknown-value"), + pytest.param(_OPUS_4_6, "xhigh", id="level-the-model-lacks"), + ), +) +def test_unsupported_reasoning_effort_is_rejected_before_reaching_anthropic( + gateway: Gateway, upstream_model: str, reasoning_effort: str +) -> None: + with wire_server(lambda request: Reply()) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY + ) + response: Final = gateway.request( + "POST", "/v1/messages", {**_client_body(reasoning_effort=reasoning_effort), "model": model} + ) + assert response.status_code == 400, response.text + assert response.json()["error"]["type"] == "invalid_request_error", response.text + assert wire.drain() == () diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py new file mode 100644 index 00000000000..bd2659b6242 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py @@ -0,0 +1,137 @@ +import json +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_MODEL: Final = "claude-haiku-4-5" +_THINKING: Final = "the user wants a single word" +_SIGNATURE: Final = "EqQBCkgIBRABGAIiQLz" +_REDACTED: Final = "EmwKAhgBEgy3va3pzixlit" + + +def _reasoning_stream(identity: str) -> tuple[bytes, ...]: + return ( + cc.sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 1}, + }, + }, + ), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": "", "signature": ""}, + }, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": _THINKING}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": _SIGNATURE}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + cc.sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": 1, + "content_block": {"type": "redacted_thinking", "data": _REDACTED}, + }, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), + cc.sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 2, "content_block": {"type": "text", "text": ""}}, + ), + cc.sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 2, "delta": {"type": "text_delta", "text": "PONG"}}, + ), + cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), + cc.sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 30}, + }, + ), + cc.sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def _reasoning_message(identity: str) -> dict[str, JsonValue]: + return { + "id": identity, + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [ + {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, + {"type": "redacted_thinking", "data": _REDACTED}, + {"type": "text", "text": "PONG"}, + ], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 30}, + } + + +def _content_events(stream: str) -> tuple[tuple[str, dict[str, object]], ...]: + return tuple(event for event in cc.sse_events(stream) if event[0].startswith("content_block_")) + + +def _client_body(stream: bool) -> dict[str, JsonValue]: + return { + **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "stream": stream, + } + + +def test_streamed_thinking_signature_and_redacted_thinking_reach_the_client_unchanged(gateway: Gateway) -> None: + identity: Final = f"msg_{uuid.uuid4().hex}" + upstream_frames: Final = _reasoning_stream(identity) + + def respond(request: Request) -> Reply: + return Reply(chunks=upstream_frames, content_type="text/event-stream") + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=True), "model": model}) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert _content_events(response.text) == _content_events(b"".join(upstream_frames).decode()), response.text + + +def test_non_streamed_thinking_signature_and_redacted_thinking_reach_the_client_unchanged(gateway: Gateway) -> None: + identity: Final = f"msg_{uuid.uuid4().hex}" + upstream_message: Final = _reasoning_message(identity) + + def respond(request: Request) -> Reply: + return Reply(body=json.dumps(upstream_message).encode()) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=False), "model": model}) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert response.json()["content"] == upstream_message["content"], response.text diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py new file mode 100644 index 00000000000..ed12e0216ef --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py @@ -0,0 +1,71 @@ +import json +import uuid +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = "claude-haiku-4-5" +_INPUT_RATE: Final = 1e-6 +_OUTPUT_RATE: Final = 2e-6 +_REASONING_RATE: Final = 7e-6 + + +def test_reported_thinking_tokens_are_billed_at_the_reasoning_rate_and_the_rest_at_the_output_rate( + gateway: Gateway, +) -> None: + identity: Final = f"msg_{uuid.uuid4().hex}" + + def respond(request: Request) -> Reply: + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [ + {"type": "thinking", "thinking": "count the words", "signature": "EqQBCkgIBRABGAIiQLz"}, + {"type": "text", "text": "PONG"}, + ], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": { + "input_tokens": 100, + "output_tokens": 50, + "output_tokens_details": {"thinking_tokens": 30}, + }, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{_MODEL}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=_INPUT_RATE, + output_cost_per_token=_OUTPUT_RATE, + output_cost_per_reasoning_token=_REASONING_RATE, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + { + **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "stream": False, + "model": model, + }, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=70, + ) + assert float(rows[0]["spend"]) == pytest.approx(100 * _INPUT_RATE + 20 * _OUTPUT_RATE + 30 * _REASONING_RATE) From d85430cde75c478fd0381e974d837989f61fb7c6 Mon Sep 17 00:00:00 2001 From: kerry Date: Thu, 1 Oct 2026 00:53:29 +0000 Subject: [PATCH 19/19] test(anthropic): narrow PR to new reasoning tests, restore moved files and drop non-reasoning tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/README.md | 2 - tests/integration/_support/claude_code.py | 1 - ...est_anthropic_prompt_cache_pricing_wire.py | 85 --------- .../context/test_anthropic_compaction_wire.py | 108 ----------- ...ropic_bare_string_content_rejected_wire.py | 32 ---- ...est_anthropic_slow_upstream_cutoff_wire.py | 64 ------- .../test_anthropic_request_fidelity_wire.py | 71 -------- .../test_anthropic_document_input_wire.py | 133 -------------- .../test_anthropic_image_input_wire.py | 67 ------- .../test_anthropic_advisor_wire.py | 0 ...t_anthropic_legacy_thinking_budget_wire.py | 0 ..._anthropic_messages_live_lifecycle_wire.py | 0 .../test_anthropic_messages_timeout_wire.py | 0 ...anthropic_thinking_signature_retry_wire.py | 0 ..._tokens_wire.py => test_anthropic_wire.py} | 78 ++++++++ .../test_websearch_interception_wire.py | 0 .../tools/test_anthropic_tool_loop_wire.py | 165 ----------------- ...est_anthropic_web_search_citations_wire.py | 170 ------------------ .../test_anthropic_long_context_beta_wire.py | 53 ------ .../test_overloaded_deployment_fallback.py | 93 ---------- .../test_client_disconnect_still_bills.py | 85 --------- 21 files changed, 78 insertions(+), 1129 deletions(-) delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py rename tests/integration/messages_endpoint/providers/anthropic/{tools => }/test_anthropic_advisor_wire.py (100%) rename tests/integration/messages_endpoint/providers/anthropic/{reasoning => }/test_anthropic_legacy_thinking_budget_wire.py (100%) rename tests/integration/messages_endpoint/providers/anthropic/{streaming => }/test_anthropic_messages_live_lifecycle_wire.py (100%) rename tests/integration/messages_endpoint/providers/anthropic/{errors => }/test_anthropic_messages_timeout_wire.py (100%) rename tests/integration/messages_endpoint/providers/anthropic/{reasoning => }/test_anthropic_thinking_signature_retry_wire.py (100%) rename tests/integration/messages_endpoint/providers/anthropic/{caching/test_anthropic_tool_history_cache_tokens_wire.py => test_anthropic_wire.py} (63%) rename tests/integration/messages_endpoint/providers/anthropic/{tools => }/test_websearch_interception_wire.py (100%) delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py delete mode 100644 tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py delete mode 100644 tests/integration/routing/test_overloaded_deployment_fallback.py delete mode 100644 tests/integration/streaming/test_client_disconnect_still_bills.py diff --git a/tests/integration/README.md b/tests/integration/README.md index 1af3004b40e..c559e7545e0 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -30,8 +30,6 @@ Streaming checks send real HTTP transfer chunks, including one-byte partitions, The `messages_endpoint/` directory holds `/v1/messages` endpoint contracts: native-provider backends under `providers/` (`anthropic`, `bedrock`, `gemini`) and the translation bridges (`responses_bridge`, `chat_bridge`) at the top level. It runs in the providers shard; `run.py` selects test files recursively under each scheduled directory -A provider folder holds only what depends on that provider's wire format, and each subfolder is one feature that provider implements its own way: `headers/`, `streaming/`, `reasoning/`, `tools/`, `caching/`, `usage/` (reading the provider's token counts and pricing them), `multimodal/`, `context/` and `errors/`. A test goes in the folder of the feature it varies; one that fits no single folder tests two things and gets split. Behavior every provider shares, such as fallback or billing after a client disconnect, lives in the feature directory it exercises (`routing/`, `streaming/`, `spend/`). `_support/claude_code.py` holds a captured Claude Code request and stream builders that any directory can use as a realistic agent payload - The sdk shard exercises the SDK's own HTTP clients against local protocol peers with no gateway in the path, so a case here fails only when the client library or its wire behavior changes. The HTTP/2 case runs a hypercorn TLS peer offering h2 and http/1.1 over ALPN, drives the sync and async httpx handlers at it with `LITELLM_HTTP2` off and on, and asserts the version both the client and the peer observed on the wire. Put a test here only when it needs no proxy, database or Redis; a case that reaches the gateway belongs in one of the other shards The extensions shard uses the built-in generic callback and guardrail transports. It checks callback correlation and credential exclusion, guardrail rewriting and denial, retained OpenAI consumers and A2A wire versions. CircleCI runs it on parallel nodes, and each node starts its own database, Redis, upstream and proxy and runs its share of the group's files serially, split by recorded timings with `circleci tests split`. Tests keep the isolation of a serial run; they still must not assume a particular set of sibling files. `run.py --list` prints a group's files and `run.py ...` runs a subset of them diff --git a/tests/integration/_support/claude_code.py b/tests/integration/_support/claude_code.py index 909728ae5b0..044713e9656 100644 --- a/tests/integration/_support/claude_code.py +++ b/tests/integration/_support/claude_code.py @@ -9,7 +9,6 @@ from pydantic import JsonValue, TypeAdapter JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) ANTHROPIC_API_KEY: Final = "synthetic-anthropic-key" -SONNET: Final = "claude-sonnet-4-5" FABLE: Final = "claude-fable-5-1" OPUS: Final = "claude-opus-5-5" CLI_BETA: Final = ( diff --git a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py deleted file mode 100644 index 06318b52aee..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_prompt_cache_pricing_wire.py +++ /dev/null @@ -1,85 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_USAGE: Final = { - "input_tokens": 10, - "cache_read_input_tokens": 3000, - "cache_creation_input_tokens": 200, - "output_tokens": 5, -} - - -def test_cached_turn_charges_cache_read_and_creation_rates(gateway: Gateway) -> None: - identity: Final = f"msg_pc_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**turn2, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply(content_type="text/event-stream", chunks=cc.text_stream(identity, cc.FABLE, "PROBE", _USAGE)) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - output_cost_per_token=5e-6, - cache_read_input_token_cost=1e-7, - cache_creation_input_token_cost=1.25e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - usage: Final = events[0][1]["message"]["usage"] - assert usage["cache_read_input_tokens"] == 3000, usage - assert usage["cache_creation_input_tokens"] == 200, usage - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(10 * 1e-6 + 3000 * 1e-7 + 200 * 1.25e-6 + 5 * 5e-6), dict( - rows[0] - ) diff --git a/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py b/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py deleted file mode 100644 index c6ac00c20e8..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/context/test_anthropic_compaction_wire.py +++ /dev/null @@ -1,108 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_CONTEXT_MANAGEMENT: Final = { - "edits": [ - {"type": "clear_thinking_20251015", "keep": "all"}, - {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}, - ] -} -_COMPACTION_BLOCK: Final = {"type": "compaction", "content": ""} - - -def _compaction_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": dict(_COMPACTION_BLOCK)}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 5}, - "context_management": { - "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] - }, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_compaction_edit_and_applied_edit_block_round_trip_through_anthropic(gateway: Gateway) -> None: - identity: Final = f"msg_cm_{uuid.uuid4().hex}" - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "context_management": _CONTEXT_MANAGEMENT, - } - turn3: Final = cc.tool_loop_turn2( - request_body, - ( - dict(_COMPACTION_BLOCK), - {"type": "text", "text": "continuing after compaction"}, - ), - (), - ) - first_expected: Final = {**request_body, "model": cc.FABLE} - second_expected: Final = {**turn3, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == first_expected: - assert body["context_management"] == _CONTEXT_MANAGEMENT - return Reply(content_type="text/event-stream", chunks=_compaction_stream(identity)) - assert body == second_expected, { - key: (second_expected.get(key), body.get(key)) - for key in second_expected.keys() | body.keys() - if second_expected.get(key) != body.get(key) - } - assert body["context_management"] == _CONTEXT_MANAGEMENT - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream("msg_cm_next", cc.FABLE, "OK", {"input_tokens": 20, "output_tokens": 2}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**request_body, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - events: Final = cc.sse_events(response1.text) - assert events[1][1]["content_block"] == _COMPACTION_BLOCK, events[1] - deltas: Final = [data for event, data in events if event == "message_delta"] - assert len(deltas) == 1 and deltas[0].get("context_management") == { - "applied_edits": [{"type": "compact_20260112", "compacted_at": "2026-09-26T00:00:00Z"}] - }, events - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py deleted file mode 100644 index ffea1830e54..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_bare_string_content_rejected_wire.py +++ /dev/null @@ -1,32 +0,0 @@ -import json -import time -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway, eventually, object_value -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - - -@pytest.mark.covers("other.provider_wire.anthropic.bare_string_content_item_is_client_error") -@pytest.mark.parametrize( - "text", [pytest.param("what type of file is this?", id="type_word"), pytest.param("hello", id="plain")] -) -def test_anthropic_bare_string_content_item_is_rejected_as_client_error_before_the_wire( - gateway: Gateway, text: str -) -> None: - def respond(request: Request) -> Reply: - raise AssertionError(f"upstream must not be reached: {request.target}") - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" - ) - response: Final = gateway.request( - "POST", - "/v1/chat/completions", - {"model": model, "max_tokens": 16, "timeout": 5, "messages": [{"role": "system", "content": [text]}]}, - ) - assert response.status_code == 400, response.text - assert wire.drain() == () diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py b/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py deleted file mode 100644 index 7532afe718d..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_slow_upstream_cutoff_wire.py +++ /dev/null @@ -1,64 +0,0 @@ -import json -import time -import uuid -from typing import Final - -import pytest -from integration._support.client import Gateway, eventually, object_value -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - - -@pytest.mark.covers("other.provider_wire.anthropic.messages_request_timeout_reaches_transport") -def test_anthropic_messages_slow_upstream_is_cut_off_at_the_deployment_request_timeout(gateway: Gateway) -> None: - identity: Final = "anthropic-timeout-" + uuid.uuid4().hex - prompt: Final = f"slow answer {identity}" - - def respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages" - assert request.headers["x-api-key"] == "synthetic-anthropic-key" - body: Final = json.loads(request.body) - assert body["model"] == "claude-sonnet-4-5-20250929" - assert body["max_tokens"] == 16 - assert body["messages"] == [{"role": "user", "content": prompt}] - assert not { - "timeout", - "request_timeout", - "stream_chunk_size", - "litellm_params", - "litellm_metadata", - "rpm", - "tpm", - }.intersection(body) - time.sleep(1.5) - return Reply( - body=json.dumps( - { - "id": identity, - "type": "message", - "role": "assistant", - "model": "claude-sonnet-4-5-20250929", - "content": [{"type": "text", "text": "late"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 3, "output_tokens": 1}, - } - ).encode() - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model="anthropic/claude-sonnet-4-5-20250929", - api_base=wire.url, - api_key="synthetic-anthropic-key", - request_timeout=0.3, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, - headers={"anthropic-version": "2023-06-01"}, - ) - assert response.status_code == 408, response.text - assert "Timeout" in response.json()["error"]["message"], response.text - assert eventually(wire.drain, lambda requests: len(requests) == 1, seconds=5, return_last_on_timeout=True) diff --git a/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py b/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py deleted file mode 100644 index a5a57737fc1..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/headers/test_anthropic_request_fidelity_wire.py +++ /dev/null @@ -1,71 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_MODEL: Final = cc.SONNET - - -def test_streaming_request_reaches_anthropic_intact_and_streams_back(gateway: Gateway) -> None: - identity: Final = f"msg_cc_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - cli_beta: Final = frozenset(cc.CLI_BETA.split(",")) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - assert request.headers["x-api-key"] == cc.ANTHROPIC_API_KEY - assert request.headers["anthropic-version"] == "2023-06-01" - assert frozenset(request.headers.get("anthropic-beta", "").split(",")) == cli_beta, request.headers.get( - "anthropic-beta" - ) - assert "authorization" not in request.headers, dict(request.headers) - assert all(gateway.key not in value for value in request.headers.values()), dict(request.headers) - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": _MODEL} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, _MODEL, "PONG", {"input_tokens": 12, "output_tokens": 4}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key), - ) - assert response.status_code == 200, response.text - assert response.headers["content-type"].startswith("text/event-stream"), dict(response.headers) - events: Final = cc.sse_events(response.text) - assert [event for event, _ in events] == [ - "message_start", - "content_block_start", - "content_block_delta", - "content_block_stop", - "message_delta", - "message_stop", - ] - assert events[2][1]["delta"] == {"type": "text_delta", "text": "PONG"} - assert events[4][1]["delta"]["stop_reason"] == "end_turn" - assert events[4][1]["usage"]["output_tokens"] == 4 - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["prompt_tokens"] == 12 and rows[0]["completion_tokens"] == 4 diff --git a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py deleted file mode 100644 index 48b914a7374..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_document_input_wire.py +++ /dev/null @@ -1,133 +0,0 @@ -import base64 -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_PDF_BYTES: Final = ( - b"%PDF-1.1\n" - b"1 0 obj<>endobj\n" - b"2 0 obj<>endobj\n" - b"3 0 obj<>endobj\n" - b"trailer<>\n%%EOF" -) -_DOC_BLOCK: Final = { - "type": "document", - "source": {"type": "base64", "data": base64.b64encode(_PDF_BYTES).decode(), "media_type": "application/pdf"}, - "citations": {"enabled": True}, -} - - -def _cited_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "A page."}}, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 0, - "delta": { - "type": "citations_delta", - "citation": { - "type": "page_location", - "document_index": 0, - "document_title": "dot.pdf", - "start_page_number": 1, - "end_page_number": 1, - "cited_text": "Page", - }, - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 6}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_base64_pdf_document_with_citations_reaches_anthropic_identical(gateway: Gateway) -> None: - request_body: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "messages": [ - { - "role": "user", - "content": [ - dict(_DOC_BLOCK), - {"type": "text", "text": f"What is on page one? {uuid.uuid4().hex}"}, - ], - } - ], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - return Reply(content_type="text/event-stream", chunks=_cited_stream(f"msg_doc_{uuid.uuid4().hex}")) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - citations: Final = [ - data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" - ] - assert citations == [ - { - "type": "citations_delta", - "citation": { - "type": "page_location", - "document_index": 0, - "document_title": "dot.pdf", - "start_page_number": 1, - "end_page_number": 1, - "cited_text": "Page", - }, - } - ], citations - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py b/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py deleted file mode 100644 index 8a929ee2a06..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/multimodal/test_anthropic_image_input_wire.py +++ /dev/null @@ -1,67 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway -from integration._support.wire import Reply, Request, wire_server - -_PNG_B64: Final = "iVBORw0KGgoAAAANSUhEUgAAAAQAAAAECAIAAAAmkwkpAAAAEElEQVR4nGP4z8AARwzEcQCukw/x0F8jngAAAABJRU5ErkJggg==" -_IMAGE_BLOCK: Final = { - "type": "image", - "source": {"type": "base64", "data": _PNG_B64, "media_type": "image/png"}, -} - - -def test_tool_result_image_block_and_pasted_image_reach_anthropic_identical(gateway: Gateway) -> None: - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/dot.png and say what colour it is", - ) - with_image_result: Final = cc.tool_loop_turn2( - turn1, - ({"type": "tool_use", "id": "toolu_img", "name": "Read", "input": {"file_path": "/tmp/cc_probe/dot.png"}},), - (("toolu_img", [dict(_IMAGE_BLOCK)]),), - ) - pasted: Final = { - **cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000), - "messages": [ - { - "role": "user", - "content": [ - dict(_IMAGE_BLOCK), - {"type": "text", "text": f"What colour is this? {uuid.uuid4().hex}"}, - ], - } - ], - } - first_expected: Final = {**with_image_result, "model": cc.FABLE} - second_expected: Final = {**pasted, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body != first_expected: - assert body == second_expected, body - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream( - f"msg_img_{uuid.uuid4().hex}", cc.FABLE, "RED", {"input_tokens": 20, "output_tokens": 2} - ), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**with_image_result, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - response2: Final = gateway.request( - "POST", "/v1/messages", {**pasted, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_advisor_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_advisor_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_legacy_thinking_budget_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_legacy_thinking_budget_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/streaming/test_anthropic_messages_live_lifecycle_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_live_lifecycle_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/errors/test_anthropic_messages_timeout_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_messages_timeout_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_thinking_signature_retry_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_thinking_signature_retry_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py similarity index 63% rename from tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py index eff5fd565b5..7e9c5be227a 100644 --- a/tests/integration/messages_endpoint/providers/anthropic/caching/test_anthropic_tool_history_cache_tokens_wire.py +++ b/tests/integration/messages_endpoint/providers/anthropic/test_anthropic_wire.py @@ -126,3 +126,81 @@ def test_anthropic_tool_history_and_cache_tokens_keep_wire_and_accounting_contra parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) assert parsed["cost_breakdown"]["input_cost"] == pytest.approx(0.0245) assert parsed["cost_breakdown"]["output_cost"] == pytest.approx(0.008) + + +@pytest.mark.covers("other.provider_wire.anthropic.bare_string_content_item_is_client_error") +@pytest.mark.parametrize( + "text", [pytest.param("what type of file is this?", id="type_word"), pytest.param("hello", id="plain")] +) +def test_anthropic_bare_string_content_item_is_rejected_as_client_error_before_the_wire( + gateway: Gateway, text: str +) -> None: + def respond(request: Request) -> Reply: + raise AssertionError(f"upstream must not be reached: {request.target}") + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "max_tokens": 16, "timeout": 5, "messages": [{"role": "system", "content": [text]}]}, + ) + assert response.status_code == 400, response.text + assert wire.drain() == () + + +@pytest.mark.covers("other.provider_wire.anthropic.messages_request_timeout_reaches_transport") +def test_anthropic_messages_slow_upstream_is_cut_off_at_the_deployment_request_timeout(gateway: Gateway) -> None: + identity: Final = "anthropic-timeout-" + uuid.uuid4().hex + prompt: Final = f"slow answer {identity}" + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + assert request.headers["x-api-key"] == "synthetic-anthropic-key" + body: Final = json.loads(request.body) + assert body["model"] == "claude-sonnet-4-5-20250929" + assert body["max_tokens"] == 16 + assert body["messages"] == [{"role": "user", "content": prompt}] + assert not { + "timeout", + "request_timeout", + "stream_chunk_size", + "litellm_params", + "litellm_metadata", + "rpm", + "tpm", + }.intersection(body) + time.sleep(1.5) + return Reply( + body=json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "late"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 3, "output_tokens": 1}, + } + ).encode() + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=wire.url, + api_key="synthetic-anthropic-key", + request_timeout=0.3, + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {"model": model, "max_tokens": 16, "messages": [{"role": "user", "content": prompt}]}, + headers={"anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 408, response.text + assert "Timeout" in response.json()["error"]["message"], response.text + assert eventually(wire.drain, lambda requests: len(requests) == 1, seconds=5, return_last_on_timeout=True) diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py b/tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py similarity index 100% rename from tests/integration/messages_endpoint/providers/anthropic/tools/test_websearch_interception_wire.py rename to tests/integration/messages_endpoint/providers/anthropic/test_websearch_interception_wire.py diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py deleted file mode 100644 index f3fb9786e02..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_tool_loop_wire.py +++ /dev/null @@ -1,165 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server -from pydantic import JsonValue - -_THINKING: Final = "need to read the file" -_SIGNATURE: Final = "sig_probe_1" -_USAGE: Final = {"input_tokens": 20, "output_tokens": 10} - - -def _diff(expected: dict[str, JsonValue], body: dict[str, JsonValue]) -> dict[str, JsonValue]: - return { - key: {"expected": expected.get(key), "upstream": body.get(key)} - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - - -def test_tool_loop_round_trips_thinking_tool_use_and_tool_result(gateway: Gateway) -> None: - identity1: Final = f"msg_tl1_{uuid.uuid4().hex}" - identity2: Final = f"msg_tl2_{uuid.uuid4().hex}" - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", - ) - calls: Final = (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - ), - (("toolu_read_1", "1\tPROBE\n2\t"),), - ) - first_expected: Final = {**turn1, "model": cc.FABLE} - second_expected: Final = {**turn2, "model": cc.FABLE} - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - if body == first_expected: - return Reply( - content_type="text/event-stream", - chunks=cc.tool_use_stream(identity1, cc.FABLE, _THINKING, _SIGNATURE, calls, _USAGE), - ) - assert body == second_expected, _diff(second_expected, body) - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity2, cc.FABLE, "PROBE", {"input_tokens": 30, "output_tokens": 3}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) - response1: Final = gateway.request( - "POST", "/v1/messages", {**turn1, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response1.status_code == 200, response1.text - events: Final = cc.sse_events(response1.text) - assert [ - ( - event, - data.get("delta", {}).get( - "type", data.get("content_block", {}).get("type", data.get("delta", {}).get("stop_reason")) - ), - ) - for event, data in events - ] == [ - ("message_start", None), - ("content_block_start", "thinking"), - ("content_block_delta", "thinking_delta"), - ("content_block_delta", "signature_delta"), - ("content_block_stop", None), - ("content_block_start", "tool_use"), - ("content_block_delta", "input_json_delta"), - ("content_block_delta", "input_json_delta"), - ("content_block_stop", None), - ("message_delta", "tool_use"), - ("message_stop", None), - ] - assert events[5][1]["content_block"]["id"] == "toolu_read_1" - assert events[5][1]["content_block"]["name"] == "Read" - partial: Final = events[6][1]["delta"]["partial_json"] + events[7][1]["delta"]["partial_json"] - assert partial == '{"file_path": "/tmp/cc_probe/hello.txt"}' - response2: Final = gateway.request( - "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers - ) - assert response2.status_code == 200, response2.text - events2: Final = cc.sse_events(response2.text) - assert events2[2][1]["delta"] == {"type": "text_delta", "text": "PROBE"} - assert events2[4][1]["delta"]["stop_reason"] == "end_turn" - bodies: Final = tuple(cc.JSON_OBJECT.validate_json(request.body) for request in wire.drain()) - assert bodies == (first_expected, second_expected), bodies - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity2,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert rows[0]["prompt_tokens"] == 30 - - -def test_parallel_tool_results_reach_anthropic_in_client_order(gateway: Gateway) -> None: - turn1: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Read /tmp/cc_probe/hello.txt and /tmp/cc_probe/world.txt and reply with both words", - ) - turn2: Final = cc.tool_loop_turn2( - turn1, - ( - {"type": "thinking", "thinking": _THINKING, "signature": _SIGNATURE}, - { - "type": "tool_use", - "id": "toolu_read_1", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/hello.txt"}, - }, - { - "type": "tool_use", - "id": "toolu_read_2", - "name": "Read", - "input": {"file_path": "/tmp/cc_probe/world.txt"}, - }, - ), - (("toolu_read_2", "1\tPROBE2\n2\t"), ("toolu_read_1", "1\tPROBE\n2\t")), - ) - - def respond(request: Request) -> Reply: - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**turn2, "model": cc.FABLE} - assert body == expected, _diff(expected, body) - results: Final = [block for block in body["messages"][3]["content"] if block["type"] == "tool_result"] - assert [block["tool_use_id"] for block in results] == ["toolu_read_2", "toolu_read_1"] - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(f"msg_mt_{uuid.uuid4().hex}", cc.FABLE, "PROBE PROBE2", _USAGE), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**turn2, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 diff --git a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py b/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py deleted file mode 100644 index 66852e6e21b..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/tools/test_anthropic_web_search_citations_wire.py +++ /dev/null @@ -1,170 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -WEB_SEARCH_TOOL: Final = {"type": "web_search_20250305", "name": "web_search", "max_uses": 8} - - -def _web_search_stream(identity: str) -> tuple[bytes, ...]: - return ( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.FABLE, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 20, "output_tokens": 1, "server_tool_use": {"web_search_requests": 1}}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": 0, - "content_block": {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {}}, - }, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 0, - "delta": {"type": "input_json_delta", "partial_json": '{"query": "current LiteLLM version"}'}, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "content_block_start", - { - "type": "content_block_start", - "index": 1, - "content_block": { - "type": "web_search_tool_result", - "tool_use_id": "srvtoolu_1", - "content": [ - { - "type": "web_search_result", - "title": "litellm releases", - "url": "https://example.com/litellm", - "page_age": None, - "encrypted_content": "enc_ws_1", - } - ], - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 1}), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 2, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 2, "delta": {"type": "text_delta", "text": "1.104.0"}}, - ), - cc.sse_frame( - "content_block_delta", - { - "type": "content_block_delta", - "index": 2, - "delta": { - "type": "citations_delta", - "citation": { - "type": "web_search_result_location", - "url": "https://example.com/litellm", - "title": "litellm releases", - "cited_text": "version 1.104.0", - "encrypted_index": "eidx_1", - }, - }, - }, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 2}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 15, "server_tool_use": {"web_search_requests": 1}}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ) - - -def test_web_search_tool_passthrough_and_cited_response(gateway: Gateway) -> None: - identity: Final = f"msg_ws_{uuid.uuid4().hex}" - base: Final = cc.frontier_request( - f"cache-bust-{uuid.uuid4().hex}", - "high", - 64000, - prompt_text="Use web search to find the current LiteLLM version and answer in one word", - ) - request_body: Final = { - **base, - "tools": [*base["tools"], WEB_SEARCH_TOOL], - } - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - body: Final = cc.JSON_OBJECT.validate_json(request.body) - expected: Final = {**request_body, "model": cc.FABLE} - assert body == expected, { - key: (expected.get(key), body.get(key)) - for key in expected.keys() | body.keys() - if expected.get(key) != body.get(key) - } - assert body["tools"][-1] == WEB_SEARCH_TOOL - assert len({tool["name"] for tool in body["tools"]}) == len(body["tools"]), body["tools"] - return Reply(content_type="text/event-stream", chunks=_web_search_stream(identity)) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - output_cost_per_token=5e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), - ) - assert response.status_code == 200, response.text - events: Final = cc.sse_events(response.text) - started: Final = [ - (data["index"], data["content_block"]["type"]) for event, data in events if event == "content_block_start" - ] - assert started == [(0, "server_tool_use"), (1, "web_search_tool_result"), (2, "text")], started - citations: Final = [ - data["delta"] for event, data in events if data.get("delta", {}).get("type") == "citations_delta" - ] - assert len(citations) == 1 and citations[0]["citation"]["url"] == "https://example.com/litellm", citations - start_usage: Final = events[0][1]["message"]["usage"] - assert start_usage["server_tool_use"]["web_search_requests"] == 1, start_usage - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - token_cost: Final = 20 * 1e-6 + 15 * 5e-6 - assert float(rows[0]["spend"]) >= token_cost, dict(rows[0]) diff --git a/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py b/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py deleted file mode 100644 index 7fe405ca439..00000000000 --- a/tests/integration/messages_endpoint/providers/anthropic/usage/test_anthropic_long_context_beta_wire.py +++ /dev/null @@ -1,53 +0,0 @@ -import uuid -from typing import Final - -import pytest -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - -_BETA_1M: Final = f"{cc.FRONTIER_CLI_BETA.replace(',effort-2025-11-24', ',context-1m-2025-08-07,effort-2025-11-24')}" - - -def test_1m_context_beta_forwarded_and_tiered_prompt_priced_above_200k(gateway: Gateway) -> None: - identity: Final = f"msg_1m_{uuid.uuid4().hex}" - request_body: Final = cc.frontier_request(f"cache-bust-{uuid.uuid4().hex}", "high", 64000) - - def respond(request: Request) -> Reply: - assert request.method == "POST" - assert request.target == "/v1/messages", request.target - upstream_beta: Final = request.headers.get("anthropic-beta", "") - assert upstream_beta.split(",").count("context-1m-2025-08-07") == 1, upstream_beta - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, cc.FABLE, "PONG", {"input_tokens": 250000, "output_tokens": 100}), - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model( - model=f"anthropic/{cc.FABLE}", - api_base=wire.url, - api_key=cc.ANTHROPIC_API_KEY, - input_cost_per_token=1e-6, - input_cost_per_token_above_200k_tokens=2e-6, - output_cost_per_token=5e-6, - ) - response: Final = gateway.request( - "POST", - "/v1/messages", - {**request_body, "model": model}, - params={"beta": "true"}, - headers=cc.cli_headers(gateway.key, _BETA_1M), - ) - assert response.status_code == 200, response.text - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT spend, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - ) - assert float(rows[0]["spend"]) == pytest.approx(250000 * 2e-6 + 100 * 5e-6), dict(rows[0]) diff --git a/tests/integration/routing/test_overloaded_deployment_fallback.py b/tests/integration/routing/test_overloaded_deployment_fallback.py deleted file mode 100644 index 7f5e669472d..00000000000 --- a/tests/integration/routing/test_overloaded_deployment_fallback.py +++ /dev/null @@ -1,93 +0,0 @@ -import json -import uuid -from pathlib import Path -from typing import Final - -import yaml -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.process import owned_proxy -from integration._support.wire import Reply, Request, wire_server - - -def _error_529() -> Reply: - return Reply( - status=529, - body=json.dumps({"type": "error", "error": {"type": "overloaded_error", "message": "Overloaded"}}).encode(), - ) - - -def test_overloaded_primary_falls_back_to_second_deployment(gateway: Gateway, tmp_path: Path) -> None: - identity: Final = f"msg_fb_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - - def respond_fallback(request: Request) -> Reply: - return Reply( - content_type="text/event-stream", - chunks=cc.text_stream(identity, cc.SONNET, "PONG", {"input_tokens": 12, "output_tokens": 4}), - ) - - with ( - wire_server(lambda request: _error_529()) as primary, - wire_server(respond_fallback) as fallback, - ): - config: Final = { - **yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()), - "model_list": [ - { - "model_name": "cc-primary", - "litellm_params": { - "model": f"anthropic/{cc.SONNET}", - "api_key": cc.ANTHROPIC_API_KEY, - "api_base": primary.url, - "model_info": {"id": "primary-cc"}, - }, - }, - { - "model_name": "cc-fallback-group", - "litellm_params": { - "model": f"anthropic/{cc.SONNET}", - "api_key": cc.ANTHROPIC_API_KEY, - "api_base": fallback.url, - "model_info": {"id": "fallback-cc"}, - }, - }, - ], - "router_settings": { - "num_retries": 0, - "disable_cooldowns": True, - "fallbacks": [{"cc-primary": ["cc-fallback-group"]}], - }, - } - path: Final = tmp_path / "fallbacks.yaml" - path.write_text(yaml.safe_dump(config)) - with owned_proxy( - gateway, tmp_path, {"REDIS_HOST": "127.0.0.1", "REDIS_PORT": "6379"}, config=path - ) as candidate: - with candidate.client.stream( - "POST", - "/v1/messages", - params={"beta": "true"}, - json={**request_body, "model": "cc-primary"}, - headers={**cc.cli_headers(candidate.key), "authorization": f"Bearer {candidate.key}"}, - ) as response: - assert response.status_code == 200, response.status_code - body: Final = "".join(response.iter_text()) - assert "message_stop" in body, body - assert "PONG" in body, body - deployments: Final = candidate.get("/model/info")["data"] - fallback_id: Final = next( - entry["model_info"]["id"] - for entry in deployments - if entry["litellm_params"]["api_base"] == fallback.url - ) - assert response.headers.get("x-litellm-model-id") == fallback_id, dict(response.headers) - assert len(primary.drain()) == 1 - assert len(fallback.drain()) == 1 - rows: Final = eventually( - lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), - lambda values: len(values) == 1, - seconds=70, - ) - assert len(rows) == 1 diff --git a/tests/integration/streaming/test_client_disconnect_still_bills.py b/tests/integration/streaming/test_client_disconnect_still_bills.py deleted file mode 100644 index 19a52f274a1..00000000000 --- a/tests/integration/streaming/test_client_disconnect_still_bills.py +++ /dev/null @@ -1,85 +0,0 @@ -import uuid -from typing import Final - -from integration._support import claude_code as cc -from integration._support.client import Gateway, eventually -from integration._support.database import read_rows -from integration._support.wire import Reply, Request, wire_server - - -def test_client_disconnect_mid_stream_still_bills_the_message(gateway: Gateway) -> None: - identity: Final = f"msg_dc_{uuid.uuid4().hex}" - request_body: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") - - def respond(request: Request) -> Reply: - return Reply( - content_type="text/event-stream", - chunks=( - cc.sse_frame( - "message_start", - { - "type": "message_start", - "message": { - "id": identity, - "type": "message", - "role": "assistant", - "model": cc.SONNET, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 1}, - }, - }, - ), - cc.sse_frame( - "content_block_start", - {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, - ), - cc.sse_frame( - "content_block_delta", - {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "PONG"}}, - ), - cc.sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), - cc.sse_frame( - "message_delta", - { - "type": "message_delta", - "delta": {"stop_reason": "end_turn", "stop_sequence": None}, - "usage": {"output_tokens": 4}, - }, - ), - cc.sse_frame("message_stop", {"type": "message_stop"}), - ), - pause_between_chunks=3.0, - ) - - with wire_server(respond) as wire, gateway.scenario() as scenario: - model: Final = scenario.model(model=f"anthropic/{cc.SONNET}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) - with gateway.client.stream( - "POST", - "/v1/messages", - params={"beta": "true"}, - json={**request_body, "model": model}, - headers={ - **cc.cli_headers(gateway.key), - "authorization": f"Bearer {gateway.key}", - }, - ) as response: - assert response.status_code == 200, response.status_code - first: Final = next(response.iter_text()) - assert "message_start" in first, first - assert len(wire.drain()) == 1 - rows: Final = eventually( - lambda: read_rows( - 'SELECT prompt_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), - ), - lambda values: len(values) == 1, - seconds=70, - return_last_on_timeout=True, - ) - assert rows and rows[0]["prompt_tokens"] == 12, rows - assert wire.disconnected.empty(), ( - "closing the client stream must not abort the upstream call before it finishes; " - f"wire recorded a disconnect on {wire.disconnected.get_nowait()}" - )