diff --git a/tests/integration/_support/claude_code.py b/tests/integration/_support/claude_code.py
new file mode 100644
index 00000000000..7bb05ce941e
--- /dev/null
+++ b/tests/integration/_support/claude_code.py
@@ -0,0 +1,1009 @@
+"""Shared Claude Code-shaped request builders and upstream stream fixtures for integration contracts."""
+
+import json
+from collections.abc import Mapping
+from dataclasses import dataclass
+from functools import reduce
+from itertools import chain
+from typing import Final
+
+from integration._support.wire import Request
+from pydantic import JsonValue, TypeAdapter
+
+JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
+ANTHROPIC_API_KEY: Final = "synthetic-anthropic-key"
+FABLE: Final = "claude-fable-5-1"
+OPUS: Final = "claude-opus-5-5"
+CLI_BETA: Final = (
+ "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13,"
+ "context-management-2025-06-27,prompt-caching-scope-2026-01-05"
+)
+FRONTIER_CLI_BETA: Final = (
+ f"{CLI_BETA},mid-conversation-system-2026-04-07,per-turn-control-2026-07-01,"
+ "mid-conversation-tool-changes-2026-07-01,effort-2025-11-24"
+)
+CACHE: Final = {"type": "ephemeral"}
+CONTEXT_MANAGEMENT: Final = {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]}
+THINKING_BUDGET: Final = {"budget_tokens": 31999, "type": "enabled", "display": "omitted"}
+THINKING_ADAPTIVE: Final = {"type": "adaptive", "display": "omitted"}
+CLAUDE_CODE_REASONING_BETAS: Final = (
+ "effort-2025-11-24",
+ "interleaved-thinking-2025-05-14",
+ "thinking-token-count-2026-05-13",
+)
+REASONING_FIELDS: Final = ("thinking", "output_config", "reasoning_effort", "temperature")
+_OUTPUT_USAGE_KEYS: Final = frozenset({"output_tokens", "output_tokens_details"})
+METADATA_USER_ID: Final = json.dumps(
+ {
+ "device_id": "0" * 64,
+ "account_uuid": "",
+ "session_id": "00000000-0000-4000-8000-000000000000",
+ }
+)
+
+
+def schema(properties: JsonValue, required: tuple[str, ...]) -> dict[str, JsonValue]:
+ return {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": properties,
+ "required": list(required),
+ "additionalProperties": False,
+ }
+
+
+def field(description: str, **extra: JsonValue) -> dict[str, JsonValue]:
+ return {"description": description, **extra}
+
+
+def tools() -> tuple[dict[str, JsonValue], ...]:
+ _MAX: Final = 9007199254740991
+ return (
+ {
+ "name": "Agent",
+ "description": "Launch a new agent to handle complex, multi-step tasks.",
+ "input_schema": schema(
+ {
+ "description": field("A short (3-5 word) description of the task", type="string"),
+ "prompt": field("The task for the agent to perform", type="string"),
+ "subagent_type": field("The type of specialized agent to use for this task", type="string"),
+ "model": field(
+ "Optional model override for this agent.",
+ type="string",
+ enum=["sonnet", "opus", "haiku", "fable"],
+ ),
+ "run_in_background": field(
+ "Agents run in the background by default; you will be notified when one completes.",
+ type="boolean",
+ ),
+ "isolation": field("Isolation mode.", type="string", enum=["worktree", "remote"]),
+ },
+ ("description", "prompt"),
+ ),
+ },
+ {
+ "name": "Bash",
+ "description": "Executes a given bash command and returns its output.",
+ "input_schema": schema(
+ {
+ "command": field("The command to execute", type="string"),
+ "timeout": field("Optional timeout in milliseconds (max 600000)", type="number"),
+ "description": field(
+ "Clear, concise description of what this command does in active voice.",
+ type="string",
+ ),
+ "run_in_background": field("Set to true to run this command in the background.", type="boolean"),
+ "dangerouslyDisableSandbox": field(
+ "Set this to true to dangerously override sandbox mode and run commands without sandboxing.",
+ type="boolean",
+ ),
+ },
+ ("command",),
+ ),
+ },
+ {
+ "name": "CronCreate",
+ "description": "Schedule a prompt to be enqueued at a future time.",
+ "input_schema": schema(
+ {
+ "cron": field(
+ 'Standard 5-field cron expression in local time: "M H DoM Mon DoW" (e.g.',
+ type="string",
+ ),
+ "prompt": field("The prompt to enqueue at each fire time.", type="string"),
+ "recurring": field(
+ "true (default) = fire on every cron match until deleted or auto-expired after 7 days.",
+ type="boolean",
+ ),
+ "durable": field(
+ "true = persist to .claude/scheduled_tasks.json and survive restarts.",
+ type="boolean",
+ ),
+ },
+ ("cron", "prompt"),
+ ),
+ },
+ {
+ "name": "CronDelete",
+ "description": "Cancel a cron job previously scheduled with CronCreate.",
+ "input_schema": schema(
+ {
+ "id": field("Job ID returned by CronCreate.", type="string"),
+ },
+ ("id",),
+ ),
+ },
+ {
+ "name": "CronList",
+ "description": "List all cron jobs scheduled via CronCreate, both durable (.claude/scheduled_tasks.json) and session-only.",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {},
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "Edit",
+ "description": "Performs exact string replacements in files.",
+ "input_schema": schema(
+ {
+ "file_path": field("The absolute path to the file to modify", type="string"),
+ "old_string": field("The text to replace", type="string"),
+ "new_string": field(
+ "The text to replace it with (must be different from old_string)", type="string"
+ ),
+ "replace_all": field(
+ "Replace all occurrences of old_string (default false)",
+ default=False,
+ type="boolean",
+ ),
+ },
+ ("file_path", "old_string", "new_string"),
+ ),
+ },
+ {
+ "name": "EnterWorktree",
+ "description": "Use this tool ONLY when explicitly instructed to work in a worktree — either by the user directly, or by project instruc",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {
+ "name": field("Optional name for a new worktree.", type="string"),
+ "path": field(
+ "Path to an existing worktree to switch into instead of creating a new one.",
+ type="string",
+ ),
+ },
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "ExitWorktree",
+ "description": "Exit a worktree session created by EnterWorktree and return the session to the original working directory.",
+ "input_schema": schema(
+ {
+ "action": field(
+ '"keep" leaves the worktree and branch on disk; "remove" deletes both.',
+ type="string",
+ enum=["keep", "remove"],
+ ),
+ "discard_changes": field(
+ 'Required true when action is "remove" and the worktree has uncommitted files or unmerged commits.',
+ type="boolean",
+ ),
+ },
+ ("action",),
+ ),
+ },
+ {
+ "name": "ListAgents",
+ "description": "Lists agents you can SendMessage to — in-process subagents you spawned, the teammates on your team, other local Claude s",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {
+ "channel": field("Not available in this build; leave unset.", type="string", maxLength=256),
+ "q": field("Not available in this build; leave unset.", type="string", maxLength=256),
+ },
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "NotebookEdit",
+ "description": "Replaces, inserts, or deletes a single cell in a Jupyter notebook (.ipynb file).",
+ "input_schema": schema(
+ {
+ "notebook_path": field(
+ "The absolute path to the Jupyter notebook file to edit (must be absolute, not relative)",
+ type="string",
+ ),
+ "cell_id": field("The ID of the cell to edit.", type="string"),
+ "new_source": field("The new source for the cell", type="string"),
+ "cell_type": field(
+ "The type of the cell (code or markdown).",
+ type="string",
+ enum=["code", "markdown"],
+ ),
+ "edit_mode": field(
+ "The type of edit to make (replace, insert, delete).",
+ type="string",
+ enum=["replace", "insert", "delete"],
+ ),
+ },
+ ("notebook_path", "new_source"),
+ ),
+ },
+ {
+ "name": "Read",
+ "description": "Reads a file from the local filesystem.",
+ "input_schema": schema(
+ {
+ "file_path": field("The absolute path to the file to read", type="string"),
+ "offset": field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX),
+ "limit": field(
+ "The number of lines to read.",
+ type="integer",
+ exclusiveMinimum=0,
+ maximum=_MAX,
+ ),
+ "pages": field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"),
+ },
+ ("file_path",),
+ ),
+ },
+ {
+ "name": "ReportFindings",
+ "description": "Report code-review findings as a typed list so the host UI can render them.",
+ "input_schema": schema(
+ {
+ "level": field(
+ "Effort level the review ran at",
+ type="string",
+ enum=["low", "medium", "high", "xhigh", "max"],
+ ),
+ "findings": field(
+ "Verified findings, most-severe first; empty if none survived",
+ maxItems=32,
+ type="array",
+ items={
+ "type": "object",
+ "properties": {
+ "file": field("Repo-relative path of the file the finding is in", type="string"),
+ "line": field(
+ "1-indexed line the finding anchors to",
+ type="integer",
+ minimum=-_MAX,
+ maximum=_MAX,
+ ),
+ "summary": field("One-sentence statement of the defect", type="string"),
+ "short_summary": field(
+ "Compressed label for compact UI (≤60 chars): the claim alone, no rationale or consequence clause",
+ type="string",
+ maxLength=60,
+ ),
+ "failure_scenario": field("Concrete inputs/state → wrong output/crash", type="string"),
+ "category": field(
+ "Short kebab-case slug of the finding type, e.g.",
+ type="string",
+ maxLength=40,
+ ),
+ "verdict": field(
+ "Set when a verify pass ran; absent on inline-only reviews",
+ type="string",
+ enum=["CONFIRMED", "PLAUSIBLE"],
+ ),
+ "outcome": field(
+ "Set ONLY when re-reporting after applying fixes: what happened to this finding",
+ type="string",
+ enum=["fixed", "skipped", "no_change_needed"],
+ ),
+ },
+ "required": ["file", "summary", "failure_scenario"],
+ "additionalProperties": False,
+ },
+ ),
+ },
+ ("findings",),
+ ),
+ },
+ {
+ "name": "ScheduleWakeup",
+ "description": "Schedule when to resume work in /loop dynamic mode — the user invoked /loop without an interval, asking you to self-pace",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {
+ "delaySeconds": field("Seconds from now to wake up.", type="number"),
+ "reason": field("One short sentence explaining the chosen delay.", type="string"),
+ "prompt": field("The /loop input to fire on wake-up.", type="string"),
+ "stop": field(
+ "Set to true to end the dynamic loop immediately instead of scheduling another wakeup.",
+ type="boolean",
+ ),
+ "noop": field(
+ "true = nothing changed (you checked and there is nothing to report).",
+ type="boolean",
+ ),
+ },
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "SendMessage",
+ "description": "# SendMessage\n\nSend a message to another agent.",
+ "input_schema": schema(
+ {
+ "to": field(
+ 'Recipient: a name from ListAgents (append its " [ref]" only when a listing or an error shows one), a teammate name, "mai',
+ type="string",
+ allOf=[{"pattern": "^[^\\n\\r]*$"}, {"pattern": "^[\\s\\S]{0,300}$"}],
+ ),
+ "summary": field(
+ "A 5-10 word label for your own transcript row (not transmitted — the recipient previews the first line of `message`).",
+ type="string",
+ maxLength=200,
+ ),
+ "message": field("Plain text message content.", default="", type="string"),
+ "notify_when_idle": field(
+ "Ask a session ON THIS MACHINE to send you ONE notice when it next goes idle (finishes its turn with nothing queued) or e",
+ type="boolean",
+ ),
+ },
+ ("to", "message"),
+ ),
+ },
+ {
+ "name": "Skill",
+ "description": "Invoke a skill.",
+ "input_schema": schema(
+ {
+ "skill": field("The name of a skill from the available-skills list.", type="string"),
+ "args": field("Optional arguments for the skill", type="string"),
+ },
+ ("skill",),
+ ),
+ },
+ {
+ "name": "TaskCreate",
+ "description": "Use this tool to create a structured task list for your current coding session.",
+ "input_schema": schema(
+ {
+ "subject": field("A brief title for the task", type="string"),
+ "description": field("What needs to be done", type="string"),
+ "activeForm": field(
+ 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")',
+ type="string",
+ ),
+ "metadata": field(
+ "Arbitrary metadata to attach to the task",
+ type="object",
+ propertyNames={"type": "string"},
+ additionalProperties={},
+ ),
+ },
+ ("subject", "description"),
+ ),
+ },
+ {
+ "name": "TaskGet",
+ "description": "Use this tool to retrieve a task by its ID from the task list.",
+ "input_schema": schema(
+ {
+ "taskId": field("The ID of the task to retrieve", type="string"),
+ },
+ ("taskId",),
+ ),
+ },
+ {
+ "name": "TaskList",
+ "description": "Use this tool to list all tasks in the task list.",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {},
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "TaskStop",
+ "description": "- Stops a running background task by its ID\n- Takes a task_id parameter identifying the task to stop\n- To stop an agent-",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {
+ "task_id": field("The ID of the background task to stop.", type="string"),
+ "shell_id": field("Deprecated: use task_id instead", type="string"),
+ },
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "TaskUpdate",
+ "description": "Use this tool to update a task in the task list.",
+ "input_schema": schema(
+ {
+ "taskId": field("The ID of the task to update", type="string"),
+ "subject": field("New subject for the task", type="string"),
+ "description": field("New description for the task", type="string"),
+ "activeForm": field(
+ 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")',
+ type="string",
+ ),
+ "status": field(
+ "New status for the task",
+ anyOf=[
+ {"type": "string", "enum": ["pending", "in_progress", "completed"]},
+ {"type": "string", "const": "deleted"},
+ ],
+ ),
+ "addBlocks": field("Task IDs that this task blocks", type="array", items={"type": "string"}),
+ "addBlockedBy": field("Task IDs that block this task", type="array", items={"type": "string"}),
+ "owner": field("New owner for the task", type="string"),
+ "metadata": field(
+ "Metadata keys to merge into the task.",
+ type="object",
+ propertyNames={"type": "string"},
+ additionalProperties={},
+ ),
+ },
+ ("taskId",),
+ ),
+ },
+ {
+ "name": "WebFetch",
+ "description": "IMPORTANT: WebFetch WILL FAIL for authenticated or private URLs.",
+ "input_schema": schema(
+ {
+ "url": field("The URL to fetch content from", type="string", format="uri"),
+ "prompt": field("The prompt to run on the fetched content", type="string"),
+ },
+ ("url", "prompt"),
+ ),
+ },
+ {
+ "name": "WebSearch",
+ "description": "- Allows Claude to search the web and use the results to inform responses\n- Provides up-to-date information for current ",
+ "input_schema": schema(
+ {
+ "query": field("The search query to use", type="string", minLength=2),
+ "allowed_domains": field(
+ "Only include search results from these domains", type="array", items={"type": "string"}
+ ),
+ "blocked_domains": field(
+ "Never include search results from these domains", type="array", items={"type": "string"}
+ ),
+ },
+ ("query",),
+ ),
+ },
+ {
+ "name": "Workflow",
+ "description": "Execute a workflow script that orchestrates multiple subagents deterministically.",
+ "input_schema": {
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
+ "type": "object",
+ "properties": {
+ "script": field("Self-contained workflow script.", type="string", maxLength=524288),
+ "name": field(
+ "Name of a predefined workflow (built-in or from .claude/workflows/).", type="string"
+ ),
+ "description": field(
+ "Ignored — set the workflow description in the script's `meta` block.", type="string"
+ ),
+ "title": field("Ignored — set the workflow title in the script's `meta` block.", type="string"),
+ "args": field("Optional input value exposed to the script as the global `args`, verbatim."),
+ "scriptPath": field("Path to a workflow script file on disk.", type="string"),
+ "resumeFromRunId": field(
+ "Run ID of a prior Workflow invocation to resume from.",
+ type="string",
+ pattern="^wf_[a-z0-9-]{6,}$",
+ ),
+ },
+ "additionalProperties": False,
+ },
+ },
+ {
+ "name": "Write",
+ "description": "Writes a file to the local filesystem.",
+ "input_schema": schema(
+ {
+ "file_path": field(
+ "The absolute path to the file to write (must be absolute, not relative)", type="string"
+ ),
+ "content": field("The content to write to the file", type="string"),
+ },
+ ("file_path", "content"),
+ ),
+ },
+ )
+
+
+def system_blocks() -> tuple[dict[str, JsonValue], ...]:
+ return (
+ {"type": "text", "text": "x-anthropic-billing-header: cc_version=2.1.283.00; cc_entrypoint=sdk-cli;"},
+ {"type": "text", "text": "Synthetic agent identity system prompt.", "cache_control": CACHE},
+ {"type": "text", "text": "Synthetic interactive agent instructions.", "cache_control": CACHE},
+ )
+
+
+def claude_code_request(cache_bust: str) -> dict[str, JsonValue]:
+ reminders: Final = (
+ f"\n{cache_bust}\n",
+ "\nSynthetic model identity reminder.\n",
+ "\nSynthetic agent types reminder.\n",
+ "\nSynthetic skills reminder.\n",
+ "\n15000000 tokens left\n",
+ "\nSynthetic date reminder.\n",
+ "\nSynthetic attribution reminder.\n",
+ )
+ return {
+ "model": "",
+ "system": list(system_blocks()),
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ *[{"type": "text", "text": reminder} for reminder in reminders],
+ {"type": "text", "text": "Reply with exactly the word PONG", "cache_control": CACHE},
+ ],
+ }
+ ],
+ "tools": list(tools()),
+ "metadata": {"user_id": METADATA_USER_ID},
+ "max_tokens": 32000,
+ "thinking": dict(THINKING_BUDGET),
+ "context_management": dict(CONTEXT_MANAGEMENT),
+ "stream": True,
+ }
+
+
+def frontier_request(
+ cache_bust: str,
+ effort: str,
+ max_tokens: int,
+ prompt_text: str = "Reply with exactly the word PONG",
+ stream: bool = True,
+) -> dict[str, JsonValue]:
+ return {
+ "model": "",
+ "system": list(system_blocks()),
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {"type": "text", "text": f"\n{cache_bust}\n"},
+ {"type": "text", "text": prompt_text},
+ ],
+ },
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "# Environment\nSynthetic environment block.",
+ "cache_control": CACHE,
+ }
+ ],
+ },
+ ],
+ "tools": list(tools()),
+ "metadata": {"user_id": METADATA_USER_ID},
+ "max_tokens": max_tokens,
+ "thinking": dict(THINKING_ADAPTIVE),
+ "context_management": dict(CONTEXT_MANAGEMENT),
+ "output_config": {"effort": effort},
+ "stream": stream,
+ }
+
+
+def tool_loop_turn2(
+ base: dict[str, JsonValue],
+ assistant_content: tuple[dict[str, JsonValue], ...],
+ tool_results: tuple[tuple[str, JsonValue], ...],
+) -> dict[str, JsonValue]:
+ return {
+ **base,
+ "messages": [
+ *base["messages"],
+ {"role": "assistant", "content": list(assistant_content)},
+ {
+ "role": "user",
+ "content": [
+ {"tool_use_id": tool_use_id, "type": "tool_result", "content": content}
+ for tool_use_id, content in tool_results
+ ],
+ },
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "14999970 tokens left",
+ "cache_control": CACHE,
+ },
+ {
+ "type": "text",
+ "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.",
+ },
+ ],
+ },
+ ],
+ }
+
+
+def cli_headers(key: str, beta: str = CLI_BETA) -> dict[str, str]:
+ return {
+ "accept": "application/json",
+ "content-type": "application/json",
+ "user-agent": "claude-cli/2.1.283 (external, sdk-cli)",
+ "x-claude-code-session-id": "00000000-0000-4000-8000-000000000000",
+ "x-stainless-arch": "x64",
+ "x-stainless-lang": "js",
+ "x-stainless-os": "Linux",
+ "x-stainless-package-version": "0.112.1",
+ "x-stainless-retry-count": "0",
+ "x-stainless-runtime": "node",
+ "x-stainless-runtime-version": "v26.3.0",
+ "x-stainless-timeout": "600",
+ "anthropic-beta": beta,
+ "anthropic-dangerous-direct-browser-access": "true",
+ "anthropic-version": "2023-06-01",
+ "x-app": "cli",
+ "x-api-key": key,
+ }
+
+
+def sse_frame(event: str, data: JsonValue) -> bytes:
+ return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode()
+
+
+def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]:
+ frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip())
+ return tuple(
+ (
+ event,
+ json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))),
+ )
+ for frame in frames
+ if (event := next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: ")))
+ != "ping"
+ )
+
+
+def reasoning_betas(anthropic_beta: str) -> tuple[str, ...]:
+ return tuple(sorted(beta for beta in anthropic_beta.split(",") if beta in CLAUDE_CODE_REASONING_BETAS))
+
+
+def body_diff(expected: Mapping[str, JsonValue], body: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ return {
+ key: {"expected": expected.get(key), "upstream": body.get(key)}
+ for key in expected.keys() | body.keys()
+ if expected.get(key) != body.get(key)
+ }
+
+
+@dataclass(frozen=True, slots=True)
+class Forwarded:
+ model: JsonValue
+ reasoning: dict[str, JsonValue]
+ assistant_history: tuple[JsonValue, ...]
+ other_changes: dict[str, JsonValue]
+ reasoning_betas: tuple[str, ...]
+
+
+def _is_assistant_turn(message: JsonValue) -> bool:
+ return isinstance(message, dict) and message.get("role") == "assistant"
+
+
+def _assistant_history(body: Mapping[str, JsonValue]) -> tuple[JsonValue, ...]:
+ messages: Final = body.get("messages")
+ if not isinstance(messages, list):
+ return ()
+ return tuple(
+ message["content"] for message in messages if isinstance(message, dict) and _is_assistant_turn(message)
+ )
+
+
+def _without_assistant_turns(messages: JsonValue) -> JsonValue:
+ if not isinstance(messages, list):
+ return messages
+ return [message for message in messages if not _is_assistant_turn(message)]
+
+
+def _unrelated_fields(body: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ excluded: Final = frozenset({*REASONING_FIELDS, "model"})
+ return {
+ key: _without_assistant_turns(value) if key == "messages" else value
+ for key, value in body.items()
+ if key not in excluded
+ }
+
+
+def forwarded(sent: Mapping[str, JsonValue], request: Request) -> Forwarded:
+ body: Final = JSON_OBJECT.validate_json(request.body)
+ return Forwarded(
+ model=body.get("model"),
+ reasoning={field: body[field] for field in REASONING_FIELDS if field in body},
+ assistant_history=_assistant_history(body),
+ other_changes=body_diff(_unrelated_fields(sent), _unrelated_fields(body)),
+ reasoning_betas=reasoning_betas(request.headers.get("anthropic-beta", "")),
+ )
+
+
+def _appended(block: Mapping[str, JsonValue], key: str, text: str) -> dict[str, JsonValue]:
+ return {**block, key: f"{block.get(key) or ''}{text}"}
+
+
+def _with_delta(block: dict[str, JsonValue], delta: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ match delta:
+ case {"type": "thinking_delta", "thinking": str(text)}:
+ return _appended(block, "thinking", text)
+ case {"type": "signature_delta", "signature": str(text)}:
+ return _appended(block, "signature", text)
+ case {"type": "text_delta", "text": str(text)}:
+ return _appended(block, "text", text)
+ case {"type": "input_json_delta", "partial_json": str(text)}:
+ return _appended(block, "partial_json", text)
+ case _:
+ return block
+
+
+def _with_event(
+ blocks: tuple[dict[str, JsonValue], ...], event: tuple[str, dict[str, JsonValue]]
+) -> tuple[dict[str, JsonValue], ...]:
+ match event:
+ case ("content_block_start", {"content_block": dict() as block}):
+ return (*blocks, JSON_OBJECT.validate_python(block))
+ case ("content_block_delta", {"index": int(index), "delta": dict() as delta}):
+ return (
+ *blocks[:index],
+ _with_delta(blocks[index], JSON_OBJECT.validate_python(delta)),
+ *blocks[index + 1 :],
+ )
+ case _:
+ return blocks
+
+
+def _finished(block: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ partial_json: Final = block.get("partial_json")
+ if not isinstance(partial_json, str):
+ return dict(block)
+ return {**{key: value for key, value in block.items() if key != "partial_json"}, "input": json.loads(partial_json)}
+
+
+def _client_events(stream: str) -> tuple[tuple[str, dict[str, JsonValue]], ...]:
+ return tuple((event, JSON_OBJECT.validate_python(data)) for event, data in sse_events(stream))
+
+
+def _stopped_indices(events: tuple[tuple[str, dict[str, JsonValue]], ...]) -> frozenset[JsonValue]:
+ return frozenset(data.get("index") for event, data in events if event == "content_block_stop")
+
+
+def streamed_content(stream: str) -> list[dict[str, JsonValue]]:
+ events: Final = _client_events(stream)
+ stopped: Final = _stopped_indices(events)
+ blocks: Final = reduce(_with_event, events, ())
+ return [_finished(block) for index, block in enumerate(blocks) if index in stopped]
+
+
+def streamed_usage(stream: str) -> JsonValue:
+ return next(data.get("usage") for event, data in reversed(_client_events(stream)) if event == "message_delta")
+
+
+def _start_usage(usage: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ return {key: value for key, value in usage.items() if key not in _OUTPUT_USAGE_KEYS}
+
+
+def _delta_usage(usage: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ return {key: value for key, value in usage.items() if key in _OUTPUT_USAGE_KEYS}
+
+
+def _block_start(index: int, block: Mapping[str, JsonValue]) -> bytes:
+ return sse_frame("content_block_start", {"type": "content_block_start", "index": index, "content_block": block})
+
+
+def _block_delta(index: int, delta: Mapping[str, JsonValue]) -> bytes:
+ return sse_frame("content_block_delta", {"type": "content_block_delta", "index": index, "delta": delta})
+
+
+def _block_stop(index: int) -> bytes:
+ return sse_frame("content_block_stop", {"type": "content_block_stop", "index": index})
+
+
+def _block_frames(index: int, block: Mapping[str, JsonValue]) -> tuple[bytes, ...]:
+ match block.get("type"):
+ case "thinking":
+ return (
+ _block_start(index, {"type": "thinking", "thinking": "", "signature": ""}),
+ _block_delta(index, {"type": "thinking_delta", "thinking": block["thinking"]}),
+ _block_delta(index, {"type": "signature_delta", "signature": block["signature"]}),
+ _block_stop(index),
+ )
+ case "text":
+ return (
+ _block_start(index, {"type": "text", "text": ""}),
+ _block_delta(index, {"type": "text_delta", "text": block["text"]}),
+ _block_stop(index),
+ )
+ case _:
+ return (_block_start(index, block), _block_stop(index))
+
+
+def message_reply(
+ identity: str, model: str, content: tuple[dict[str, JsonValue], ...], usage: dict[str, JsonValue]
+) -> bytes:
+ return json.dumps(
+ {
+ "id": identity,
+ "type": "message",
+ "role": "assistant",
+ "model": model,
+ "content": list(content),
+ "stop_reason": "end_turn",
+ "stop_sequence": None,
+ "usage": usage,
+ }
+ ).encode()
+
+
+def message_stream(
+ identity: str, model: str, content: tuple[dict[str, JsonValue], ...], usage: dict[str, JsonValue]
+) -> tuple[bytes, ...]:
+ start: Final = sse_frame(
+ "message_start",
+ {
+ "type": "message_start",
+ "message": {
+ "id": identity,
+ "type": "message",
+ "role": "assistant",
+ "model": model,
+ "content": [],
+ "stop_reason": None,
+ "stop_sequence": None,
+ "usage": _start_usage(usage),
+ },
+ },
+ )
+ delta: Final = sse_frame(
+ "message_delta",
+ {
+ "type": "message_delta",
+ "delta": {"stop_reason": "end_turn", "stop_sequence": None},
+ "usage": _delta_usage(usage),
+ },
+ )
+ blocks: Final = chain.from_iterable(_block_frames(index, block) for index, block in enumerate(content))
+ return (start, *blocks, delta, sse_frame("message_stop", {"type": "message_stop"}))
+
+
+def text_stream(identity: str, model: str, text: str, usage: dict[str, int]) -> tuple[bytes, ...]:
+ return (
+ sse_frame(
+ "message_start",
+ {
+ "type": "message_start",
+ "message": {
+ "id": identity,
+ "type": "message",
+ "role": "assistant",
+ "model": model,
+ "content": [],
+ "stop_reason": None,
+ "stop_sequence": None,
+ "usage": _start_usage(usage),
+ },
+ },
+ ),
+ sse_frame(
+ "content_block_start",
+ {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}},
+ ),
+ sse_frame(
+ "content_block_delta",
+ {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}},
+ ),
+ sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}),
+ sse_frame(
+ "message_delta",
+ {
+ "type": "message_delta",
+ "delta": {"stop_reason": "end_turn", "stop_sequence": None},
+ "usage": {"output_tokens": usage["output_tokens"]},
+ },
+ ),
+ sse_frame("message_stop", {"type": "message_stop"}),
+ )
+
+
+def _tool_use_frames(index: int, tool_id: str, name: str, tool_input: JsonValue) -> tuple[bytes, ...]:
+ arguments: Final = json.dumps(tool_input)
+ return (
+ sse_frame(
+ "content_block_start",
+ {
+ "type": "content_block_start",
+ "index": index,
+ "content_block": {"type": "tool_use", "id": tool_id, "name": name, "input": {}},
+ },
+ ),
+ sse_frame(
+ "content_block_delta",
+ {
+ "type": "content_block_delta",
+ "index": index,
+ "delta": {"type": "input_json_delta", "partial_json": arguments[: len(arguments) // 2]},
+ },
+ ),
+ sse_frame(
+ "content_block_delta",
+ {
+ "type": "content_block_delta",
+ "index": index,
+ "delta": {"type": "input_json_delta", "partial_json": arguments[len(arguments) // 2 :]},
+ },
+ ),
+ sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}),
+ )
+
+
+def tool_use_stream(
+ identity: str,
+ model: str,
+ thinking: str,
+ signature: str,
+ tool_calls: tuple[tuple[str, str, JsonValue], ...],
+ usage: dict[str, int],
+) -> tuple[bytes, ...]:
+ head: Final = (
+ sse_frame(
+ "message_start",
+ {
+ "type": "message_start",
+ "message": {
+ "id": identity,
+ "type": "message",
+ "role": "assistant",
+ "model": model,
+ "content": [],
+ "stop_reason": None,
+ "stop_sequence": None,
+ "usage": _start_usage(usage),
+ },
+ },
+ ),
+ sse_frame(
+ "content_block_start",
+ {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}},
+ ),
+ sse_frame(
+ "content_block_delta",
+ {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": thinking}},
+ ),
+ sse_frame(
+ "content_block_delta",
+ {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": signature}},
+ ),
+ sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}),
+ )
+ tail: Final = (
+ sse_frame(
+ "message_delta",
+ {
+ "type": "message_delta",
+ "delta": {"stop_reason": "tool_use", "stop_sequence": None},
+ "usage": {"output_tokens": usage["output_tokens"]},
+ },
+ ),
+ sse_frame("message_stop", {"type": "message_stop"}),
+ )
+ frames: Final = (
+ *head,
+ *chain.from_iterable(
+ _tool_use_frames(index, tool_id, name, tool_input)
+ for index, (tool_id, name, tool_input) in enumerate(tool_calls, start=1)
+ ),
+ *tail,
+ )
+ return frames
diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py
new file mode 100644
index 00000000000..ac2247ff09a
--- /dev/null
+++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py
@@ -0,0 +1,81 @@
+import uuid
+from typing import Final
+
+from integration._support import claude_code as cc
+from integration._support.client import Gateway
+from integration._support.wire import Reply, Request, wire_server
+
+_READ_A: Final = {"type": "tool_use", "id": "toolu_a", "name": "Read", "input": {"file_path": "/tmp/cc_probe/a.txt"}}
+_READ_B: Final = {"type": "tool_use", "id": "toolu_b", "name": "Read", "input": {"file_path": "/tmp/cc_probe/b.txt"}}
+
+
+def test_interleaved_thinking_history_reaches_anthropic_and_interleaved_blocks_stream_back(gateway: Gateway) -> None:
+ turn1: Final = cc.frontier_request(
+ f"cache-bust-{uuid.uuid4().hex}",
+ "high",
+ 64000,
+ prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words",
+ )
+ turn2: Final = cc.tool_loop_turn2(
+ turn1, ({"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A), (("toolu_a", "ALPHA"),)
+ )
+ turn3: Final = cc.tool_loop_turn2(
+ turn2,
+ ({"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B),
+ (("toolu_b", "BRAVO"),),
+ )
+
+ def respond(request: Request) -> Reply:
+ if b"toolu_b" not in request.body:
+ return Reply(
+ content_type="text/event-stream",
+ chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}),
+ )
+ return Reply(
+ content_type="text/event-stream",
+ chunks=cc.message_stream(
+ f"msg_il_{uuid.uuid4().hex}",
+ cc.FABLE,
+ (
+ {"type": "thinking", "thinking": "got B", "signature": "sig3"},
+ {"type": "text", "text": "got B"},
+ {
+ "type": "tool_use",
+ "id": "toolu_c",
+ "name": "Read",
+ "input": {"file_path": "/tmp/cc_probe/c.txt"},
+ },
+ ),
+ {"input_tokens": 20, "output_tokens": 12},
+ ),
+ )
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
+ headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA)
+ response2: Final = gateway.request(
+ "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers
+ )
+ assert response2.status_code == 200, response2.text
+ response3: Final = gateway.request(
+ "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers
+ )
+ assert response3.status_code == 200, response3.text
+ received: Final = wire.drain()
+ assert len(received) == 2, received
+ second: Final = cc.forwarded(turn2, received[0])
+ third: Final = cc.forwarded(turn3, received[1])
+ assert second.assistant_history == ([{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A],)
+ assert third.assistant_history == (
+ [{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A],
+ [{"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B],
+ ), third.assistant_history
+ adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}
+ assert (second.reasoning, third.reasoning) == (adaptive_high, adaptive_high)
+ assert (second.other_changes, third.other_changes) == ({}, {})
+ assert (second.reasoning_betas, third.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2
+ assert cc.streamed_content(response3.text) == [
+ {"type": "thinking", "thinking": "got B", "signature": "sig3"},
+ {"type": "text", "text": "got B"},
+ {"type": "tool_use", "id": "toolu_c", "name": "Read", "input": {"file_path": "/tmp/cc_probe/c.txt"}},
+ ], response3.text
diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py
new file mode 100644
index 00000000000..003fe870f16
--- /dev/null
+++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py
@@ -0,0 +1,81 @@
+import uuid
+from typing import Final
+
+from integration._support import claude_code as cc
+from integration._support.client import Gateway
+from integration._support.wire import Reply, Request, wire_server
+
+
+def test_mid_loop_model_switch_replays_thinking_history_and_reasoning_unchanged(gateway: Gateway) -> None:
+ turn1: Final = cc.frontier_request(
+ f"cache-bust-{uuid.uuid4().hex}",
+ "high",
+ 64000,
+ prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word",
+ )
+ turn2: Final = cc.tool_loop_turn2(
+ turn1,
+ (
+ {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"},
+ {
+ "type": "tool_use",
+ "id": "toolu_read_1",
+ "name": "Read",
+ "input": {"file_path": "/tmp/cc_probe/hello.txt"},
+ },
+ ),
+ (("toolu_read_1", "1\tPROBE\n2\t"),),
+ )
+
+ def respond(request: Request) -> Reply:
+ if b"tool_result" not in request.body:
+ return Reply(
+ content_type="text/event-stream",
+ chunks=cc.tool_use_stream(
+ f"msg_{uuid.uuid4().hex}",
+ cc.FABLE,
+ "need to read the file",
+ "sig_anthropic_1",
+ (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),),
+ {"input_tokens": 20, "output_tokens": 10},
+ ),
+ )
+ return Reply(
+ content_type="text/event-stream",
+ chunks=cc.text_stream(
+ f"msg_{uuid.uuid4().hex}", cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3}
+ ),
+ )
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
+ opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
+ headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA)
+ response1: Final = gateway.request(
+ "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers
+ )
+ assert response1.status_code == 200, response1.text
+ response2: Final = gateway.request(
+ "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers
+ )
+ assert response2.status_code == 200, response2.text
+ received: Final = wire.drain()
+ assert len(received) == 2, received
+ to_fable: Final = cc.forwarded(turn1, received[0])
+ to_opus: Final = cc.forwarded(turn2, received[1])
+ assert (to_fable.model, to_opus.model) == (cc.FABLE, cc.OPUS), received
+ assert to_opus.assistant_history == (
+ [
+ {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"},
+ {
+ "type": "tool_use",
+ "id": "toolu_read_1",
+ "name": "Read",
+ "input": {"file_path": "/tmp/cc_probe/hello.txt"},
+ },
+ ],
+ ), to_opus.assistant_history
+ adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}
+ assert (to_fable.reasoning, to_opus.reasoning) == (adaptive_high, adaptive_high)
+ assert (to_fable.other_changes, to_opus.other_changes) == ({}, {})
+ assert (to_fable.reasoning_betas, to_opus.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2
diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py
new file mode 100644
index 00000000000..4fc99e2dd67
--- /dev/null
+++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py
@@ -0,0 +1,482 @@
+import uuid
+from collections.abc import Mapping
+from typing import Final
+
+import pytest
+from integration._support import claude_code as cc
+from integration._support.client import Gateway
+from integration._support.wire import Reply, Request, wire_server
+from pydantic import JsonValue
+
+
+def _claude_code_turn(sent: Mapping[str, JsonValue]) -> dict[str, JsonValue]:
+ default_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")
+ without_reasoning: Final = {key: value for key, value in default_turn.items() if key != "thinking"}
+ return {**without_reasoning, "stream": False, **sent}
+
+
+def _forward(gateway: Gateway, upstream_model: str, sent: Mapping[str, JsonValue]) -> cc.Forwarded:
+ client_body: Final = _claude_code_turn(sent)
+
+ def respond(request: Request) -> Reply:
+ return Reply(
+ body=cc.message_reply(
+ f"msg_{uuid.uuid4().hex}",
+ upstream_model,
+ ({"type": "text", "text": "PONG"},),
+ {"input_tokens": 12, "output_tokens": 4},
+ )
+ )
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(
+ model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY
+ )
+ response: Final = gateway.request(
+ "POST",
+ "/v1/messages",
+ {**client_body, "model": model},
+ headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA),
+ )
+ assert response.status_code == 200, response.text
+ received: Final = wire.drain()
+ assert len(received) == 1, received
+ return cc.forwarded(client_body, received[0])
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "sent", "received"),
+ (
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "low"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 1024}},
+ id="haiku-4.5-low",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "medium"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ id="haiku-4.5-medium",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 4096}},
+ id="haiku-4.5-high",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 8192}},
+ id="haiku-4.5-xhigh",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "max"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 16384}},
+ id="haiku-4.5-max",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "max"},
+ "max_tokens": 4000,
+ },
+ {"thinking": {"type": "enabled", "budget_tokens": 3999}},
+ id="haiku-4.5-budget-capped-below-max-tokens",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ "max_tokens": 1024,
+ },
+ {},
+ id="haiku-4.5-max-tokens-below-minimum-budget",
+ ),
+ pytest.param(
+ "claude-opus-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ {"output_config": {"effort": "high"}},
+ id="opus-4.5-keeps-effort-drops-adaptive",
+ ),
+ pytest.param(
+ "claude-opus-4-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ {"thinking": {"type": "enabled", "budget_tokens": 8192}},
+ id="opus-4.5-xhigh-falls-back-to-budget",
+ ),
+ pytest.param(
+ "claude-opus-4-6",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ id="opus-4.6-unchanged",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ id="opus-4.7-unchanged",
+ ),
+ pytest.param(
+ "claude-fable-5-1",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}},
+ id="fable-5.1-unchanged",
+ ),
+ pytest.param(
+ "claude-opus-5-5",
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}},
+ id="opus-5.5-unchanged",
+ ),
+ ),
+)
+def test_adaptive_thinking_and_effort_are_rewritten_only_for_models_without_adaptive_thinking(
+ gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, upstream_model, sent)
+ assert forwarded.reasoning == received, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "sent", "received"),
+ (
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "enabled", "budget_tokens": 1024}},
+ {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
+ id="opus-4.7-1024-is-low",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
+ id="opus-4.7-2048-is-medium",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "enabled", "budget_tokens": 4096}},
+ {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
+ id="opus-4.7-4096-is-high",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "enabled", "budget_tokens": 8192}},
+ {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
+ id="opus-4.7-8192-is-xhigh",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "enabled", "budget_tokens": 8192}, "output_config": {"effort": "medium"}},
+ {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
+ id="opus-4.7-keeps-the-callers-effort",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ id="haiku-4.5-unchanged",
+ ),
+ ),
+)
+def test_legacy_thinking_budget_becomes_adaptive_effort_only_on_models_that_reject_budgets(
+ gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, upstream_model, sent)
+ assert forwarded.reasoning == received, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "sent", "received"),
+ (
+ pytest.param(
+ "claude-fable-5-1",
+ {"thinking": {"type": "disabled"}},
+ {},
+ id="fable-5.1-always-thinks-so-disabled-is-dropped",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"thinking": {"type": "disabled"}},
+ {"thinking": {"type": "disabled"}},
+ id="opus-4.7-keeps-disabled",
+ ),
+ ),
+)
+def test_disabled_thinking_is_dropped_only_for_always_on_thinking_models(
+ gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, upstream_model, sent)
+ assert forwarded.reasoning == received, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "sent", "received"),
+ (
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "minimal"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
+ id="opus-4.7-minimal",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "low"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
+ id="opus-4.7-low",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "medium"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "medium"}},
+ id="opus-4.7-medium",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "high"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}},
+ id="opus-4.7-high",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "xhigh"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "xhigh"}},
+ id="opus-4.7-xhigh",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "max"},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "max"}},
+ id="opus-4.7-max",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "minimal"},
+ {"thinking": {"type": "enabled", "budget_tokens": 1024}},
+ id="haiku-4.5-minimal",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "low"},
+ {"thinking": {"type": "enabled", "budget_tokens": 1024}},
+ id="haiku-4.5-low",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "medium"},
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ id="haiku-4.5-medium",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "high"},
+ {"thinking": {"type": "enabled", "budget_tokens": 4096}},
+ id="haiku-4.5-high",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "xhigh"},
+ {"thinking": {"type": "enabled", "budget_tokens": 8192}},
+ id="haiku-4.5-xhigh",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "max"},
+ {"thinking": {"type": "enabled", "budget_tokens": 16384}},
+ id="haiku-4.5-max",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "max", "max_tokens": 4000},
+ {"thinking": {"type": "enabled", "budget_tokens": 3999}},
+ id="haiku-4.5-budget-capped-below-max-tokens",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "high", "max_tokens": 1024},
+ {},
+ id="haiku-4.5-max-tokens-below-minimum-budget",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {
+ "reasoning_effort": "none",
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ },
+ {},
+ id="none-clears-thinking-and-effort",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"reasoning_effort": "high", "thinking": {"type": "enabled", "budget_tokens": 2000}},
+ {"thinking": {"type": "enabled", "budget_tokens": 2000}},
+ id="callers-thinking-wins",
+ ),
+ pytest.param(
+ "claude-opus-4-7",
+ {"reasoning_effort": "high", "output_config": {"effort": "low"}},
+ {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}},
+ id="callers-effort-wins",
+ ),
+ ),
+)
+def test_reasoning_effort_becomes_the_thinking_shape_each_model_accepts(
+ gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, upstream_model, sent)
+ assert forwarded.reasoning == received, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "sent", "received"),
+ (
+ pytest.param(
+ "claude-haiku-4-5",
+ {
+ "temperature": 0,
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ },
+ {"thinking": {"type": "enabled", "budget_tokens": 4096}},
+ id="haiku-4.5-drops-temperature-0-with-effort",
+ ),
+ pytest.param(
+ "claude-opus-4-5",
+ {
+ "temperature": 0,
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ },
+ {"output_config": {"effort": "high"}},
+ id="opus-4.5-drops-temperature-0-with-effort",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"temperature": 0, "thinking": {"type": "enabled", "budget_tokens": 2048}},
+ {"thinking": {"type": "enabled", "budget_tokens": 2048}},
+ id="haiku-4.5-drops-temperature-0-with-budget",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}},
+ {"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}},
+ id="haiku-4.5-keeps-temperature-1",
+ ),
+ pytest.param(
+ "claude-haiku-4-5",
+ {"temperature": 0},
+ {"temperature": 0},
+ id="haiku-4.5-keeps-temperature-without-thinking",
+ ),
+ pytest.param(
+ "claude-opus-4-6",
+ {
+ "temperature": 0,
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ },
+ {
+ "temperature": 0,
+ "thinking": {"type": "adaptive", "display": "omitted"},
+ "output_config": {"effort": "high"},
+ },
+ id="opus-4.6-adaptive-keeps-temperature",
+ ),
+ ),
+)
+def test_temperature_is_dropped_only_when_a_non_adaptive_model_thinks(
+ gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, upstream_model, sent)
+ assert forwarded.reasoning == received, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+def _tool_loop(assistant_content: list[JsonValue]) -> dict[str, JsonValue]:
+ first_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")["messages"]
+ assert isinstance(first_turn, list)
+ tool_result: Final = {"type": "tool_result", "tool_use_id": "toolu_01", "content": "ok"}
+ return {
+ "thinking": {"type": "enabled", "budget_tokens": 2048},
+ "messages": [
+ *first_turn,
+ {"role": "assistant", "content": assistant_content},
+ {"role": "user", "content": [tool_result]},
+ ],
+ }
+
+
+@pytest.mark.parametrize(
+ ("sent_history", "received_history"),
+ (
+ pytest.param(
+ [
+ {"type": "thinking", "thinking": "bridge reasoning", "signature": "litellm_encrypted_reasoning:gAAAAB"},
+ {"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:gAAAAC"},
+ {"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"},
+ {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
+ ],
+ [
+ {"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"},
+ {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
+ ],
+ id="encrypted-reasoning-from-another-provider-stripped-anthropic-signed-kept",
+ ),
+ pytest.param(
+ [
+ {"type": "thinking", "thinking": "", "signature": "EqQBCkgIBRABGAIiQM"},
+ {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"},
+ {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
+ ],
+ [
+ {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"},
+ {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}},
+ ],
+ id="empty-thinking-stripped-redacted-thinking-kept",
+ ),
+ ),
+)
+def test_thinking_history_keeps_only_blocks_anthropic_can_verify(
+ gateway: Gateway, sent_history: list[JsonValue], received_history: list[JsonValue]
+) -> None:
+ forwarded: Final = _forward(gateway, "claude-haiku-4-5", _tool_loop(sent_history))
+ assert forwarded.assistant_history == (received_history,), forwarded.assistant_history
+ assert forwarded.reasoning == {"thinking": {"type": "enabled", "budget_tokens": 2048}}, forwarded
+ assert forwarded.other_changes == {}, forwarded.other_changes
+ assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas
+
+
+@pytest.mark.parametrize(
+ ("upstream_model", "reasoning_effort"),
+ (
+ pytest.param("claude-haiku-4-5", "turbo", id="unknown-value"),
+ pytest.param("claude-opus-4-6", "xhigh", id="level-the-model-lacks"),
+ ),
+)
+def test_unsupported_reasoning_effort_is_rejected_before_reaching_anthropic(
+ gateway: Gateway, upstream_model: str, reasoning_effort: str
+) -> None:
+ with wire_server(lambda request: Reply()) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(
+ model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY
+ )
+ response: Final = gateway.request(
+ "POST", "/v1/messages", {**_claude_code_turn({"reasoning_effort": reasoning_effort}), "model": model}
+ )
+ assert response.status_code == 400, response.text
+ assert response.json()["error"]["type"] == "invalid_request_error", response.text
+ assert wire.drain() == ()
diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py
new file mode 100644
index 00000000000..1ada28b3355
--- /dev/null
+++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py
@@ -0,0 +1,63 @@
+import uuid
+from typing import Final
+
+from integration._support import claude_code as cc
+from integration._support.client import Gateway
+from integration._support.wire import Reply, Request, wire_server
+from pydantic import JsonValue
+
+_MODEL: Final = "claude-haiku-4-5"
+_ANTHROPIC_CONTENT: Final = (
+ {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
+ {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
+ {"type": "text", "text": "PONG"},
+)
+_ANTHROPIC_USAGE: Final = {"input_tokens": 12, "output_tokens": 30, "output_tokens_details": {"thinking_tokens": 20}}
+
+
+def _client_body(stream: bool) -> dict[str, JsonValue]:
+ return {
+ **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"),
+ "thinking": {"type": "enabled", "budget_tokens": 2048},
+ "stream": stream,
+ }
+
+
+def test_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None:
+ def respond(request: Request) -> Reply:
+ return Reply(
+ chunks=cc.message_stream(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE),
+ content_type="text/event-stream",
+ )
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
+ response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=True), "model": model})
+ assert response.status_code == 200, response.text
+ assert len(wire.drain()) == 1
+ assert cc.streamed_content(response.text) == [
+ {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
+ {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
+ {"type": "text", "text": "PONG"},
+ ], response.text
+ assert cc.streamed_usage(response.text) == {
+ "output_tokens": 30,
+ "output_tokens_details": {"thinking_tokens": 20},
+ }, response.text
+
+
+def test_non_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None:
+ def respond(request: Request) -> Reply:
+ return Reply(body=cc.message_reply(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE))
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY)
+ response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=False), "model": model})
+ assert response.status_code == 200, response.text
+ assert len(wire.drain()) == 1
+ assert response.json()["content"] == [
+ {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"},
+ {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"},
+ {"type": "text", "text": "PONG"},
+ ], response.text
+ assert response.json()["usage"]["output_tokens_details"] == {"thinking_tokens": 20}, response.text
diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py
new file mode 100644
index 00000000000..cf3c709028c
--- /dev/null
+++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py
@@ -0,0 +1,64 @@
+import uuid
+from typing import Final
+
+import pytest
+from integration._support import claude_code as cc
+from integration._support.client import Gateway, eventually
+from integration._support.database import read_rows
+from integration._support.wire import Reply, Request, wire_server
+
+_MODEL: Final = "claude-haiku-4-5"
+_INPUT_RATE: Final = 1e-6
+_OUTPUT_RATE: Final = 2e-6
+_REASONING_RATE: Final = 7e-6
+_CONTENT: Final = (
+ {"type": "thinking", "thinking": "count the words", "signature": "EqQBCkgIBRABGAIiQLz"},
+ {"type": "text", "text": "PONG"},
+)
+_USAGE: Final = {"input_tokens": 100, "output_tokens": 50, "output_tokens_details": {"thinking_tokens": 30}}
+
+
+def _reply(identity: str, stream: bool) -> Reply:
+ if stream:
+ return Reply(chunks=cc.message_stream(identity, _MODEL, _CONTENT, _USAGE), content_type="text/event-stream")
+ return Reply(body=cc.message_reply(identity, _MODEL, _CONTENT, _USAGE))
+
+
+@pytest.mark.parametrize("stream", (pytest.param(False, id="non-streamed"), pytest.param(True, id="streamed")))
+def test_reported_thinking_tokens_are_billed_at_the_reasoning_rate_and_the_rest_at_the_output_rate(
+ gateway: Gateway, stream: bool
+) -> None:
+ identity: Final = f"msg_{uuid.uuid4().hex}"
+
+ def respond(request: Request) -> Reply:
+ return _reply(identity, stream)
+
+ with wire_server(respond) as wire, gateway.scenario() as scenario:
+ model: Final = scenario.model(
+ model=f"anthropic/{_MODEL}",
+ api_base=wire.url,
+ api_key=cc.ANTHROPIC_API_KEY,
+ input_cost_per_token=_INPUT_RATE,
+ output_cost_per_token=_OUTPUT_RATE,
+ output_cost_per_reasoning_token=_REASONING_RATE,
+ )
+ body: Final = {
+ **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"),
+ "thinking": {"type": "enabled", "budget_tokens": 2048},
+ "stream": stream,
+ "model": model,
+ }
+ response: Final = gateway.request("POST", "/v1/messages", body)
+ assert response.status_code == 200, response.text
+ assert len(wire.drain()) == 1
+ rows: Final = eventually(
+ lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)),
+ lambda values: len(values) == 1,
+ seconds=70,
+ )
+ input_tokens, output_tokens, thinking_tokens = 100, 50, 30
+ assert float(rows[0]["spend"]) == pytest.approx(
+ input_tokens * _INPUT_RATE
+ + (output_tokens - thinking_tokens) * _OUTPUT_RATE
+ + thinking_tokens * _REASONING_RATE
+ ), rows