diff --git a/tests/integration/_support/claude_code.py b/tests/integration/_support/claude_code.py new file mode 100644 index 00000000000..7bb05ce941e --- /dev/null +++ b/tests/integration/_support/claude_code.py @@ -0,0 +1,1009 @@ +"""Shared Claude Code-shaped request builders and upstream stream fixtures for integration contracts.""" + +import json +from collections.abc import Mapping +from dataclasses import dataclass +from functools import reduce +from itertools import chain +from typing import Final + +from integration._support.wire import Request +from pydantic import JsonValue, TypeAdapter + +JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +ANTHROPIC_API_KEY: Final = "synthetic-anthropic-key" +FABLE: Final = "claude-fable-5-1" +OPUS: Final = "claude-opus-5-5" +CLI_BETA: Final = ( + "claude-code-20250219,interleaved-thinking-2025-05-14,thinking-token-count-2026-05-13," + "context-management-2025-06-27,prompt-caching-scope-2026-01-05" +) +FRONTIER_CLI_BETA: Final = ( + f"{CLI_BETA},mid-conversation-system-2026-04-07,per-turn-control-2026-07-01," + "mid-conversation-tool-changes-2026-07-01,effort-2025-11-24" +) +CACHE: Final = {"type": "ephemeral"} +CONTEXT_MANAGEMENT: Final = {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]} +THINKING_BUDGET: Final = {"budget_tokens": 31999, "type": "enabled", "display": "omitted"} +THINKING_ADAPTIVE: Final = {"type": "adaptive", "display": "omitted"} +CLAUDE_CODE_REASONING_BETAS: Final = ( + "effort-2025-11-24", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", +) +REASONING_FIELDS: Final = ("thinking", "output_config", "reasoning_effort", "temperature") +_OUTPUT_USAGE_KEYS: Final = frozenset({"output_tokens", "output_tokens_details"}) +METADATA_USER_ID: Final = json.dumps( + { + "device_id": "0" * 64, + "account_uuid": "", + "session_id": "00000000-0000-4000-8000-000000000000", + } +) + + +def schema(properties: JsonValue, required: tuple[str, ...]) -> dict[str, JsonValue]: + return { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": properties, + "required": list(required), + "additionalProperties": False, + } + + +def field(description: str, **extra: JsonValue) -> dict[str, JsonValue]: + return {"description": description, **extra} + + +def tools() -> tuple[dict[str, JsonValue], ...]: + _MAX: Final = 9007199254740991 + return ( + { + "name": "Agent", + "description": "Launch a new agent to handle complex, multi-step tasks.", + "input_schema": schema( + { + "description": field("A short (3-5 word) description of the task", type="string"), + "prompt": field("The task for the agent to perform", type="string"), + "subagent_type": field("The type of specialized agent to use for this task", type="string"), + "model": field( + "Optional model override for this agent.", + type="string", + enum=["sonnet", "opus", "haiku", "fable"], + ), + "run_in_background": field( + "Agents run in the background by default; you will be notified when one completes.", + type="boolean", + ), + "isolation": field("Isolation mode.", type="string", enum=["worktree", "remote"]), + }, + ("description", "prompt"), + ), + }, + { + "name": "Bash", + "description": "Executes a given bash command and returns its output.", + "input_schema": schema( + { + "command": field("The command to execute", type="string"), + "timeout": field("Optional timeout in milliseconds (max 600000)", type="number"), + "description": field( + "Clear, concise description of what this command does in active voice.", + type="string", + ), + "run_in_background": field("Set to true to run this command in the background.", type="boolean"), + "dangerouslyDisableSandbox": field( + "Set this to true to dangerously override sandbox mode and run commands without sandboxing.", + type="boolean", + ), + }, + ("command",), + ), + }, + { + "name": "CronCreate", + "description": "Schedule a prompt to be enqueued at a future time.", + "input_schema": schema( + { + "cron": field( + 'Standard 5-field cron expression in local time: "M H DoM Mon DoW" (e.g.', + type="string", + ), + "prompt": field("The prompt to enqueue at each fire time.", type="string"), + "recurring": field( + "true (default) = fire on every cron match until deleted or auto-expired after 7 days.", + type="boolean", + ), + "durable": field( + "true = persist to .claude/scheduled_tasks.json and survive restarts.", + type="boolean", + ), + }, + ("cron", "prompt"), + ), + }, + { + "name": "CronDelete", + "description": "Cancel a cron job previously scheduled with CronCreate.", + "input_schema": schema( + { + "id": field("Job ID returned by CronCreate.", type="string"), + }, + ("id",), + ), + }, + { + "name": "CronList", + "description": "List all cron jobs scheduled via CronCreate, both durable (.claude/scheduled_tasks.json) and session-only.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + }, + { + "name": "Edit", + "description": "Performs exact string replacements in files.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to modify", type="string"), + "old_string": field("The text to replace", type="string"), + "new_string": field( + "The text to replace it with (must be different from old_string)", type="string" + ), + "replace_all": field( + "Replace all occurrences of old_string (default false)", + default=False, + type="boolean", + ), + }, + ("file_path", "old_string", "new_string"), + ), + }, + { + "name": "EnterWorktree", + "description": "Use this tool ONLY when explicitly instructed to work in a worktree — either by the user directly, or by project instruc", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "name": field("Optional name for a new worktree.", type="string"), + "path": field( + "Path to an existing worktree to switch into instead of creating a new one.", + type="string", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "ExitWorktree", + "description": "Exit a worktree session created by EnterWorktree and return the session to the original working directory.", + "input_schema": schema( + { + "action": field( + '"keep" leaves the worktree and branch on disk; "remove" deletes both.', + type="string", + enum=["keep", "remove"], + ), + "discard_changes": field( + 'Required true when action is "remove" and the worktree has uncommitted files or unmerged commits.', + type="boolean", + ), + }, + ("action",), + ), + }, + { + "name": "ListAgents", + "description": "Lists agents you can SendMessage to — in-process subagents you spawned, the teammates on your team, other local Claude s", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "channel": field("Not available in this build; leave unset.", type="string", maxLength=256), + "q": field("Not available in this build; leave unset.", type="string", maxLength=256), + }, + "additionalProperties": False, + }, + }, + { + "name": "NotebookEdit", + "description": "Replaces, inserts, or deletes a single cell in a Jupyter notebook (.ipynb file).", + "input_schema": schema( + { + "notebook_path": field( + "The absolute path to the Jupyter notebook file to edit (must be absolute, not relative)", + type="string", + ), + "cell_id": field("The ID of the cell to edit.", type="string"), + "new_source": field("The new source for the cell", type="string"), + "cell_type": field( + "The type of the cell (code or markdown).", + type="string", + enum=["code", "markdown"], + ), + "edit_mode": field( + "The type of edit to make (replace, insert, delete).", + type="string", + enum=["replace", "insert", "delete"], + ), + }, + ("notebook_path", "new_source"), + ), + }, + { + "name": "Read", + "description": "Reads a file from the local filesystem.", + "input_schema": schema( + { + "file_path": field("The absolute path to the file to read", type="string"), + "offset": field("The line number to start reading from.", type="integer", minimum=0, maximum=_MAX), + "limit": field( + "The number of lines to read.", + type="integer", + exclusiveMinimum=0, + maximum=_MAX, + ), + "pages": field('Page range for PDF files (e.g., "1-5", "3", "10-20").', type="string"), + }, + ("file_path",), + ), + }, + { + "name": "ReportFindings", + "description": "Report code-review findings as a typed list so the host UI can render them.", + "input_schema": schema( + { + "level": field( + "Effort level the review ran at", + type="string", + enum=["low", "medium", "high", "xhigh", "max"], + ), + "findings": field( + "Verified findings, most-severe first; empty if none survived", + maxItems=32, + type="array", + items={ + "type": "object", + "properties": { + "file": field("Repo-relative path of the file the finding is in", type="string"), + "line": field( + "1-indexed line the finding anchors to", + type="integer", + minimum=-_MAX, + maximum=_MAX, + ), + "summary": field("One-sentence statement of the defect", type="string"), + "short_summary": field( + "Compressed label for compact UI (≤60 chars): the claim alone, no rationale or consequence clause", + type="string", + maxLength=60, + ), + "failure_scenario": field("Concrete inputs/state → wrong output/crash", type="string"), + "category": field( + "Short kebab-case slug of the finding type, e.g.", + type="string", + maxLength=40, + ), + "verdict": field( + "Set when a verify pass ran; absent on inline-only reviews", + type="string", + enum=["CONFIRMED", "PLAUSIBLE"], + ), + "outcome": field( + "Set ONLY when re-reporting after applying fixes: what happened to this finding", + type="string", + enum=["fixed", "skipped", "no_change_needed"], + ), + }, + "required": ["file", "summary", "failure_scenario"], + "additionalProperties": False, + }, + ), + }, + ("findings",), + ), + }, + { + "name": "ScheduleWakeup", + "description": "Schedule when to resume work in /loop dynamic mode — the user invoked /loop without an interval, asking you to self-pace", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "delaySeconds": field("Seconds from now to wake up.", type="number"), + "reason": field("One short sentence explaining the chosen delay.", type="string"), + "prompt": field("The /loop input to fire on wake-up.", type="string"), + "stop": field( + "Set to true to end the dynamic loop immediately instead of scheduling another wakeup.", + type="boolean", + ), + "noop": field( + "true = nothing changed (you checked and there is nothing to report).", + type="boolean", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "SendMessage", + "description": "# SendMessage\n\nSend a message to another agent.", + "input_schema": schema( + { + "to": field( + 'Recipient: a name from ListAgents (append its " [ref]" only when a listing or an error shows one), a teammate name, "mai', + type="string", + allOf=[{"pattern": "^[^\\n\\r]*$"}, {"pattern": "^[\\s\\S]{0,300}$"}], + ), + "summary": field( + "A 5-10 word label for your own transcript row (not transmitted — the recipient previews the first line of `message`).", + type="string", + maxLength=200, + ), + "message": field("Plain text message content.", default="", type="string"), + "notify_when_idle": field( + "Ask a session ON THIS MACHINE to send you ONE notice when it next goes idle (finishes its turn with nothing queued) or e", + type="boolean", + ), + }, + ("to", "message"), + ), + }, + { + "name": "Skill", + "description": "Invoke a skill.", + "input_schema": schema( + { + "skill": field("The name of a skill from the available-skills list.", type="string"), + "args": field("Optional arguments for the skill", type="string"), + }, + ("skill",), + ), + }, + { + "name": "TaskCreate", + "description": "Use this tool to create a structured task list for your current coding session.", + "input_schema": schema( + { + "subject": field("A brief title for the task", type="string"), + "description": field("What needs to be done", type="string"), + "activeForm": field( + 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")', + type="string", + ), + "metadata": field( + "Arbitrary metadata to attach to the task", + type="object", + propertyNames={"type": "string"}, + additionalProperties={}, + ), + }, + ("subject", "description"), + ), + }, + { + "name": "TaskGet", + "description": "Use this tool to retrieve a task by its ID from the task list.", + "input_schema": schema( + { + "taskId": field("The ID of the task to retrieve", type="string"), + }, + ("taskId",), + ), + }, + { + "name": "TaskList", + "description": "Use this tool to list all tasks in the task list.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + }, + { + "name": "TaskStop", + "description": "- Stops a running background task by its ID\n- Takes a task_id parameter identifying the task to stop\n- To stop an agent-", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "task_id": field("The ID of the background task to stop.", type="string"), + "shell_id": field("Deprecated: use task_id instead", type="string"), + }, + "additionalProperties": False, + }, + }, + { + "name": "TaskUpdate", + "description": "Use this tool to update a task in the task list.", + "input_schema": schema( + { + "taskId": field("The ID of the task to update", type="string"), + "subject": field("New subject for the task", type="string"), + "description": field("New description for the task", type="string"), + "activeForm": field( + 'Present continuous form shown in spinner when in_progress (e.g., "Running tests")', + type="string", + ), + "status": field( + "New status for the task", + anyOf=[ + {"type": "string", "enum": ["pending", "in_progress", "completed"]}, + {"type": "string", "const": "deleted"}, + ], + ), + "addBlocks": field("Task IDs that this task blocks", type="array", items={"type": "string"}), + "addBlockedBy": field("Task IDs that block this task", type="array", items={"type": "string"}), + "owner": field("New owner for the task", type="string"), + "metadata": field( + "Metadata keys to merge into the task.", + type="object", + propertyNames={"type": "string"}, + additionalProperties={}, + ), + }, + ("taskId",), + ), + }, + { + "name": "WebFetch", + "description": "IMPORTANT: WebFetch WILL FAIL for authenticated or private URLs.", + "input_schema": schema( + { + "url": field("The URL to fetch content from", type="string", format="uri"), + "prompt": field("The prompt to run on the fetched content", type="string"), + }, + ("url", "prompt"), + ), + }, + { + "name": "WebSearch", + "description": "- Allows Claude to search the web and use the results to inform responses\n- Provides up-to-date information for current ", + "input_schema": schema( + { + "query": field("The search query to use", type="string", minLength=2), + "allowed_domains": field( + "Only include search results from these domains", type="array", items={"type": "string"} + ), + "blocked_domains": field( + "Never include search results from these domains", type="array", items={"type": "string"} + ), + }, + ("query",), + ), + }, + { + "name": "Workflow", + "description": "Execute a workflow script that orchestrates multiple subagents deterministically.", + "input_schema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "script": field("Self-contained workflow script.", type="string", maxLength=524288), + "name": field( + "Name of a predefined workflow (built-in or from .claude/workflows/).", type="string" + ), + "description": field( + "Ignored — set the workflow description in the script's `meta` block.", type="string" + ), + "title": field("Ignored — set the workflow title in the script's `meta` block.", type="string"), + "args": field("Optional input value exposed to the script as the global `args`, verbatim."), + "scriptPath": field("Path to a workflow script file on disk.", type="string"), + "resumeFromRunId": field( + "Run ID of a prior Workflow invocation to resume from.", + type="string", + pattern="^wf_[a-z0-9-]{6,}$", + ), + }, + "additionalProperties": False, + }, + }, + { + "name": "Write", + "description": "Writes a file to the local filesystem.", + "input_schema": schema( + { + "file_path": field( + "The absolute path to the file to write (must be absolute, not relative)", type="string" + ), + "content": field("The content to write to the file", type="string"), + }, + ("file_path", "content"), + ), + }, + ) + + +def system_blocks() -> tuple[dict[str, JsonValue], ...]: + return ( + {"type": "text", "text": "x-anthropic-billing-header: cc_version=2.1.283.00; cc_entrypoint=sdk-cli;"}, + {"type": "text", "text": "Synthetic agent identity system prompt.", "cache_control": CACHE}, + {"type": "text", "text": "Synthetic interactive agent instructions.", "cache_control": CACHE}, + ) + + +def claude_code_request(cache_bust: str) -> dict[str, JsonValue]: + reminders: Final = ( + f"\n{cache_bust}\n", + "\nSynthetic model identity reminder.\n", + "\nSynthetic agent types reminder.\n", + "\nSynthetic skills reminder.\n", + "\n15000000 tokens left\n", + "\nSynthetic date reminder.\n", + "\nSynthetic attribution reminder.\n", + ) + return { + "model": "", + "system": list(system_blocks()), + "messages": [ + { + "role": "user", + "content": [ + *[{"type": "text", "text": reminder} for reminder in reminders], + {"type": "text", "text": "Reply with exactly the word PONG", "cache_control": CACHE}, + ], + } + ], + "tools": list(tools()), + "metadata": {"user_id": METADATA_USER_ID}, + "max_tokens": 32000, + "thinking": dict(THINKING_BUDGET), + "context_management": dict(CONTEXT_MANAGEMENT), + "stream": True, + } + + +def frontier_request( + cache_bust: str, + effort: str, + max_tokens: int, + prompt_text: str = "Reply with exactly the word PONG", + stream: bool = True, +) -> dict[str, JsonValue]: + return { + "model": "", + "system": list(system_blocks()), + "messages": [ + { + "role": "user", + "content": [ + {"type": "text", "text": f"\n{cache_bust}\n"}, + {"type": "text", "text": prompt_text}, + ], + }, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "# Environment\nSynthetic environment block.", + "cache_control": CACHE, + } + ], + }, + ], + "tools": list(tools()), + "metadata": {"user_id": METADATA_USER_ID}, + "max_tokens": max_tokens, + "thinking": dict(THINKING_ADAPTIVE), + "context_management": dict(CONTEXT_MANAGEMENT), + "output_config": {"effort": effort}, + "stream": stream, + } + + +def tool_loop_turn2( + base: dict[str, JsonValue], + assistant_content: tuple[dict[str, JsonValue], ...], + tool_results: tuple[tuple[str, JsonValue], ...], +) -> dict[str, JsonValue]: + return { + **base, + "messages": [ + *base["messages"], + {"role": "assistant", "content": list(assistant_content)}, + { + "role": "user", + "content": [ + {"tool_use_id": tool_use_id, "type": "tool_result", "content": content} + for tool_use_id, content in tool_results + ], + }, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "14999970 tokens left", + "cache_control": CACHE, + }, + { + "type": "text", + "text": "First privately list what you need next; then request every item that doesn't depend on another's result in this one response.", + }, + ], + }, + ], + } + + +def cli_headers(key: str, beta: str = CLI_BETA) -> dict[str, str]: + return { + "accept": "application/json", + "content-type": "application/json", + "user-agent": "claude-cli/2.1.283 (external, sdk-cli)", + "x-claude-code-session-id": "00000000-0000-4000-8000-000000000000", + "x-stainless-arch": "x64", + "x-stainless-lang": "js", + "x-stainless-os": "Linux", + "x-stainless-package-version": "0.112.1", + "x-stainless-retry-count": "0", + "x-stainless-runtime": "node", + "x-stainless-runtime-version": "v26.3.0", + "x-stainless-timeout": "600", + "anthropic-beta": beta, + "anthropic-dangerous-direct-browser-access": "true", + "anthropic-version": "2023-06-01", + "x-app": "cli", + "x-api-key": key, + } + + +def sse_frame(event: str, data: JsonValue) -> bytes: + return f"event: {event}\ndata: {json.dumps(data)}\n\n".encode() + + +def sse_events(text: str) -> tuple[tuple[str, dict[str, object]], ...]: + frames: Final = tuple(frame for frame in text.split("\n\n") if frame.strip()) + return tuple( + ( + event, + json.loads(next(line.removeprefix("data: ") for line in frame.splitlines() if line.startswith("data: "))), + ) + for frame in frames + if (event := next(line.removeprefix("event: ") for line in frame.splitlines() if line.startswith("event: "))) + != "ping" + ) + + +def reasoning_betas(anthropic_beta: str) -> tuple[str, ...]: + return tuple(sorted(beta for beta in anthropic_beta.split(",") if beta in CLAUDE_CODE_REASONING_BETAS)) + + +def body_diff(expected: Mapping[str, JsonValue], body: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return { + key: {"expected": expected.get(key), "upstream": body.get(key)} + for key in expected.keys() | body.keys() + if expected.get(key) != body.get(key) + } + + +@dataclass(frozen=True, slots=True) +class Forwarded: + model: JsonValue + reasoning: dict[str, JsonValue] + assistant_history: tuple[JsonValue, ...] + other_changes: dict[str, JsonValue] + reasoning_betas: tuple[str, ...] + + +def _is_assistant_turn(message: JsonValue) -> bool: + return isinstance(message, dict) and message.get("role") == "assistant" + + +def _assistant_history(body: Mapping[str, JsonValue]) -> tuple[JsonValue, ...]: + messages: Final = body.get("messages") + if not isinstance(messages, list): + return () + return tuple( + message["content"] for message in messages if isinstance(message, dict) and _is_assistant_turn(message) + ) + + +def _without_assistant_turns(messages: JsonValue) -> JsonValue: + if not isinstance(messages, list): + return messages + return [message for message in messages if not _is_assistant_turn(message)] + + +def _unrelated_fields(body: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + excluded: Final = frozenset({*REASONING_FIELDS, "model"}) + return { + key: _without_assistant_turns(value) if key == "messages" else value + for key, value in body.items() + if key not in excluded + } + + +def forwarded(sent: Mapping[str, JsonValue], request: Request) -> Forwarded: + body: Final = JSON_OBJECT.validate_json(request.body) + return Forwarded( + model=body.get("model"), + reasoning={field: body[field] for field in REASONING_FIELDS if field in body}, + assistant_history=_assistant_history(body), + other_changes=body_diff(_unrelated_fields(sent), _unrelated_fields(body)), + reasoning_betas=reasoning_betas(request.headers.get("anthropic-beta", "")), + ) + + +def _appended(block: Mapping[str, JsonValue], key: str, text: str) -> dict[str, JsonValue]: + return {**block, key: f"{block.get(key) or ''}{text}"} + + +def _with_delta(block: dict[str, JsonValue], delta: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + match delta: + case {"type": "thinking_delta", "thinking": str(text)}: + return _appended(block, "thinking", text) + case {"type": "signature_delta", "signature": str(text)}: + return _appended(block, "signature", text) + case {"type": "text_delta", "text": str(text)}: + return _appended(block, "text", text) + case {"type": "input_json_delta", "partial_json": str(text)}: + return _appended(block, "partial_json", text) + case _: + return block + + +def _with_event( + blocks: tuple[dict[str, JsonValue], ...], event: tuple[str, dict[str, JsonValue]] +) -> tuple[dict[str, JsonValue], ...]: + match event: + case ("content_block_start", {"content_block": dict() as block}): + return (*blocks, JSON_OBJECT.validate_python(block)) + case ("content_block_delta", {"index": int(index), "delta": dict() as delta}): + return ( + *blocks[:index], + _with_delta(blocks[index], JSON_OBJECT.validate_python(delta)), + *blocks[index + 1 :], + ) + case _: + return blocks + + +def _finished(block: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + partial_json: Final = block.get("partial_json") + if not isinstance(partial_json, str): + return dict(block) + return {**{key: value for key, value in block.items() if key != "partial_json"}, "input": json.loads(partial_json)} + + +def _client_events(stream: str) -> tuple[tuple[str, dict[str, JsonValue]], ...]: + return tuple((event, JSON_OBJECT.validate_python(data)) for event, data in sse_events(stream)) + + +def _stopped_indices(events: tuple[tuple[str, dict[str, JsonValue]], ...]) -> frozenset[JsonValue]: + return frozenset(data.get("index") for event, data in events if event == "content_block_stop") + + +def streamed_content(stream: str) -> list[dict[str, JsonValue]]: + events: Final = _client_events(stream) + stopped: Final = _stopped_indices(events) + blocks: Final = reduce(_with_event, events, ()) + return [_finished(block) for index, block in enumerate(blocks) if index in stopped] + + +def streamed_usage(stream: str) -> JsonValue: + return next(data.get("usage") for event, data in reversed(_client_events(stream)) if event == "message_delta") + + +def _start_usage(usage: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return {key: value for key, value in usage.items() if key not in _OUTPUT_USAGE_KEYS} + + +def _delta_usage(usage: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return {key: value for key, value in usage.items() if key in _OUTPUT_USAGE_KEYS} + + +def _block_start(index: int, block: Mapping[str, JsonValue]) -> bytes: + return sse_frame("content_block_start", {"type": "content_block_start", "index": index, "content_block": block}) + + +def _block_delta(index: int, delta: Mapping[str, JsonValue]) -> bytes: + return sse_frame("content_block_delta", {"type": "content_block_delta", "index": index, "delta": delta}) + + +def _block_stop(index: int) -> bytes: + return sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}) + + +def _block_frames(index: int, block: Mapping[str, JsonValue]) -> tuple[bytes, ...]: + match block.get("type"): + case "thinking": + return ( + _block_start(index, {"type": "thinking", "thinking": "", "signature": ""}), + _block_delta(index, {"type": "thinking_delta", "thinking": block["thinking"]}), + _block_delta(index, {"type": "signature_delta", "signature": block["signature"]}), + _block_stop(index), + ) + case "text": + return ( + _block_start(index, {"type": "text", "text": ""}), + _block_delta(index, {"type": "text_delta", "text": block["text"]}), + _block_stop(index), + ) + case _: + return (_block_start(index, block), _block_stop(index)) + + +def message_reply( + identity: str, model: str, content: tuple[dict[str, JsonValue], ...], usage: dict[str, JsonValue] +) -> bytes: + return json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": list(content), + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": usage, + } + ).encode() + + +def message_stream( + identity: str, model: str, content: tuple[dict[str, JsonValue], ...], usage: dict[str, JsonValue] +) -> tuple[bytes, ...]: + start: Final = sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": _start_usage(usage), + }, + }, + ) + delta: Final = sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": _delta_usage(usage), + }, + ) + blocks: Final = chain.from_iterable(_block_frames(index, block) for index, block in enumerate(content)) + return (start, *blocks, delta, sse_frame("message_stop", {"type": "message_stop"})) + + +def text_stream(identity: str, model: str, text: str, usage: dict[str, int]) -> tuple[bytes, ...]: + return ( + sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": _start_usage(usage), + }, + }, + ), + sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}}, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + sse_frame("message_stop", {"type": "message_stop"}), + ) + + +def _tool_use_frames(index: int, tool_id: str, name: str, tool_input: JsonValue) -> tuple[bytes, ...]: + arguments: Final = json.dumps(tool_input) + return ( + sse_frame( + "content_block_start", + { + "type": "content_block_start", + "index": index, + "content_block": {"type": "tool_use", "id": tool_id, "name": name, "input": {}}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[: len(arguments) // 2]}, + }, + ), + sse_frame( + "content_block_delta", + { + "type": "content_block_delta", + "index": index, + "delta": {"type": "input_json_delta", "partial_json": arguments[len(arguments) // 2 :]}, + }, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": index}), + ) + + +def tool_use_stream( + identity: str, + model: str, + thinking: str, + signature: str, + tool_calls: tuple[tuple[str, str, JsonValue], ...], + usage: dict[str, int], +) -> tuple[bytes, ...]: + head: Final = ( + sse_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": model, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": _start_usage(usage), + }, + }, + ), + sse_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": ""}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": thinking}}, + ), + sse_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": signature}}, + ), + sse_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + ) + tail: Final = ( + sse_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "tool_use", "stop_sequence": None}, + "usage": {"output_tokens": usage["output_tokens"]}, + }, + ), + sse_frame("message_stop", {"type": "message_stop"}), + ) + frames: Final = ( + *head, + *chain.from_iterable( + _tool_use_frames(index, tool_id, name, tool_input) + for index, (tool_id, name, tool_input) in enumerate(tool_calls, start=1) + ), + *tail, + ) + return frames diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py new file mode 100644 index 00000000000..ac2247ff09a --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_interleaved_thinking_history_wire.py @@ -0,0 +1,81 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + +_READ_A: Final = {"type": "tool_use", "id": "toolu_a", "name": "Read", "input": {"file_path": "/tmp/cc_probe/a.txt"}} +_READ_B: Final = {"type": "tool_use", "id": "toolu_b", "name": "Read", "input": {"file_path": "/tmp/cc_probe/b.txt"}} + + +def test_interleaved_thinking_history_reaches_anthropic_and_interleaved_blocks_stream_back(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/a.txt then /tmp/cc_probe/b.txt one at a time and reply with both words", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, ({"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A), (("toolu_a", "ALPHA"),) + ) + turn3: Final = cc.tool_loop_turn2( + turn2, + ({"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B), + (("toolu_b", "BRAVO"),), + ) + + def respond(request: Request) -> Reply: + if b"toolu_b" not in request.body: + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream("msg_il_turn2", cc.FABLE, "got A", {"input_tokens": 20, "output_tokens": 4}), + ) + return Reply( + content_type="text/event-stream", + chunks=cc.message_stream( + f"msg_il_{uuid.uuid4().hex}", + cc.FABLE, + ( + {"type": "thinking", "thinking": "got B", "signature": "sig3"}, + {"type": "text", "text": "got B"}, + { + "type": "tool_use", + "id": "toolu_c", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/c.txt"}, + }, + ), + {"input_tokens": 20, "output_tokens": 12}, + ), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + response3: Final = gateway.request( + "POST", "/v1/messages", {**turn3, "model": model}, params={"beta": "true"}, headers=headers + ) + assert response3.status_code == 200, response3.text + received: Final = wire.drain() + assert len(received) == 2, received + second: Final = cc.forwarded(turn2, received[0]) + third: Final = cc.forwarded(turn3, received[1]) + assert second.assistant_history == ([{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A],) + assert third.assistant_history == ( + [{"type": "thinking", "thinking": "plan", "signature": "sig1"}, _READ_A], + [{"type": "thinking", "thinking": "got A", "signature": "sig2"}, {"type": "text", "text": "got A"}, _READ_B], + ), third.assistant_history + adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}} + assert (second.reasoning, third.reasoning) == (adaptive_high, adaptive_high) + assert (second.other_changes, third.other_changes) == ({}, {}) + assert (second.reasoning_betas, third.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2 + assert cc.streamed_content(response3.text) == [ + {"type": "thinking", "thinking": "got B", "signature": "sig3"}, + {"type": "text", "text": "got B"}, + {"type": "tool_use", "id": "toolu_c", "name": "Read", "input": {"file_path": "/tmp/cc_probe/c.txt"}}, + ], response3.text diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py new file mode 100644 index 00000000000..003fe870f16 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_model_switch_history_wire.py @@ -0,0 +1,81 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server + + +def test_mid_loop_model_switch_replays_thinking_history_and_reasoning_unchanged(gateway: Gateway) -> None: + turn1: Final = cc.frontier_request( + f"cache-bust-{uuid.uuid4().hex}", + "high", + 64000, + prompt_text="Read /tmp/cc_probe/hello.txt and reply with its single word", + ) + turn2: Final = cc.tool_loop_turn2( + turn1, + ( + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ), + (("toolu_read_1", "1\tPROBE\n2\t"),), + ) + + def respond(request: Request) -> Reply: + if b"tool_result" not in request.body: + return Reply( + content_type="text/event-stream", + chunks=cc.tool_use_stream( + f"msg_{uuid.uuid4().hex}", + cc.FABLE, + "need to read the file", + "sig_anthropic_1", + (("toolu_read_1", "Read", {"file_path": "/tmp/cc_probe/hello.txt"}),), + {"input_tokens": 20, "output_tokens": 10}, + ), + ) + return Reply( + content_type="text/event-stream", + chunks=cc.text_stream( + f"msg_{uuid.uuid4().hex}", cc.OPUS, "PROBE", {"input_tokens": 30, "output_tokens": 3} + ), + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + fable: Final = scenario.model(model=f"anthropic/{cc.FABLE}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + opus: Final = scenario.model(model=f"anthropic/{cc.OPUS}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + headers: Final = cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA) + response1: Final = gateway.request( + "POST", "/v1/messages", {**turn1, "model": fable}, params={"beta": "true"}, headers=headers + ) + assert response1.status_code == 200, response1.text + response2: Final = gateway.request( + "POST", "/v1/messages", {**turn2, "model": opus}, params={"beta": "true"}, headers=headers + ) + assert response2.status_code == 200, response2.text + received: Final = wire.drain() + assert len(received) == 2, received + to_fable: Final = cc.forwarded(turn1, received[0]) + to_opus: Final = cc.forwarded(turn2, received[1]) + assert (to_fable.model, to_opus.model) == (cc.FABLE, cc.OPUS), received + assert to_opus.assistant_history == ( + [ + {"type": "thinking", "thinking": "need to read the file", "signature": "sig_anthropic_1"}, + { + "type": "tool_use", + "id": "toolu_read_1", + "name": "Read", + "input": {"file_path": "/tmp/cc_probe/hello.txt"}, + }, + ], + ), to_opus.assistant_history + adaptive_high: Final = {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}} + assert (to_fable.reasoning, to_opus.reasoning) == (adaptive_high, adaptive_high) + assert (to_fable.other_changes, to_opus.other_changes) == ({}, {}) + assert (to_fable.reasoning_betas, to_opus.reasoning_betas) == (cc.CLAUDE_CODE_REASONING_BETAS,) * 2 diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py new file mode 100644 index 00000000000..4fc99e2dd67 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_request_translation_wire.py @@ -0,0 +1,482 @@ +import uuid +from collections.abc import Mapping +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + + +def _claude_code_turn(sent: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + default_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}") + without_reasoning: Final = {key: value for key, value in default_turn.items() if key != "thinking"} + return {**without_reasoning, "stream": False, **sent} + + +def _forward(gateway: Gateway, upstream_model: str, sent: Mapping[str, JsonValue]) -> cc.Forwarded: + client_body: Final = _claude_code_turn(sent) + + def respond(request: Request) -> Reply: + return Reply( + body=cc.message_reply( + f"msg_{uuid.uuid4().hex}", + upstream_model, + ({"type": "text", "text": "PONG"},), + {"input_tokens": 12, "output_tokens": 4}, + ) + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY + ) + response: Final = gateway.request( + "POST", + "/v1/messages", + {**client_body, "model": model}, + headers=cc.cli_headers(gateway.key, cc.FRONTIER_CLI_BETA), + ) + assert response.status_code == 200, response.text + received: Final = wire.drain() + assert len(received) == 1, received + return cc.forwarded(client_body, received[0]) + + +@pytest.mark.parametrize( + ("upstream_model", "sent", "received"), + ( + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "low"}}, + {"thinking": {"type": "enabled", "budget_tokens": 1024}}, + id="haiku-4.5-low", + ), + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "medium"}}, + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="haiku-4.5-medium", + ), + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + {"thinking": {"type": "enabled", "budget_tokens": 4096}}, + id="haiku-4.5-high", + ), + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + {"thinking": {"type": "enabled", "budget_tokens": 8192}}, + id="haiku-4.5-xhigh", + ), + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "max"}}, + {"thinking": {"type": "enabled", "budget_tokens": 16384}}, + id="haiku-4.5-max", + ), + pytest.param( + "claude-haiku-4-5", + { + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "max"}, + "max_tokens": 4000, + }, + {"thinking": {"type": "enabled", "budget_tokens": 3999}}, + id="haiku-4.5-budget-capped-below-max-tokens", + ), + pytest.param( + "claude-haiku-4-5", + { + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + "max_tokens": 1024, + }, + {}, + id="haiku-4.5-max-tokens-below-minimum-budget", + ), + pytest.param( + "claude-opus-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + {"output_config": {"effort": "high"}}, + id="opus-4.5-keeps-effort-drops-adaptive", + ), + pytest.param( + "claude-opus-4-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + {"thinking": {"type": "enabled", "budget_tokens": 8192}}, + id="opus-4.5-xhigh-falls-back-to-budget", + ), + pytest.param( + "claude-opus-4-6", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + id="opus-4.6-unchanged", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + id="opus-4.7-unchanged", + ), + pytest.param( + "claude-fable-5-1", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "high"}}, + id="fable-5.1-unchanged", + ), + pytest.param( + "claude-opus-5-5", + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + {"thinking": {"type": "adaptive", "display": "omitted"}, "output_config": {"effort": "xhigh"}}, + id="opus-5.5-unchanged", + ), + ), +) +def test_adaptive_thinking_and_effort_are_rewritten_only_for_models_without_adaptive_thinking( + gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue] +) -> None: + forwarded: Final = _forward(gateway, upstream_model, sent) + assert forwarded.reasoning == received, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +@pytest.mark.parametrize( + ("upstream_model", "sent", "received"), + ( + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "enabled", "budget_tokens": 1024}}, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}}, + id="opus-4.7-1024-is-low", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}}, + id="opus-4.7-2048-is-medium", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "enabled", "budget_tokens": 4096}}, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}}, + id="opus-4.7-4096-is-high", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "enabled", "budget_tokens": 8192}}, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}}, + id="opus-4.7-8192-is-xhigh", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "enabled", "budget_tokens": 8192}, "output_config": {"effort": "medium"}}, + {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}}, + id="opus-4.7-keeps-the-callers-effort", + ), + pytest.param( + "claude-haiku-4-5", + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="haiku-4.5-unchanged", + ), + ), +) +def test_legacy_thinking_budget_becomes_adaptive_effort_only_on_models_that_reject_budgets( + gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue] +) -> None: + forwarded: Final = _forward(gateway, upstream_model, sent) + assert forwarded.reasoning == received, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +@pytest.mark.parametrize( + ("upstream_model", "sent", "received"), + ( + pytest.param( + "claude-fable-5-1", + {"thinking": {"type": "disabled"}}, + {}, + id="fable-5.1-always-thinks-so-disabled-is-dropped", + ), + pytest.param( + "claude-opus-4-7", + {"thinking": {"type": "disabled"}}, + {"thinking": {"type": "disabled"}}, + id="opus-4.7-keeps-disabled", + ), + ), +) +def test_disabled_thinking_is_dropped_only_for_always_on_thinking_models( + gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue] +) -> None: + forwarded: Final = _forward(gateway, upstream_model, sent) + assert forwarded.reasoning == received, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +@pytest.mark.parametrize( + ("upstream_model", "sent", "received"), + ( + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "minimal"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}}, + id="opus-4.7-minimal", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "low"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}}, + id="opus-4.7-low", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "medium"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "medium"}}, + id="opus-4.7-medium", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "high"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}}, + id="opus-4.7-high", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "xhigh"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "xhigh"}}, + id="opus-4.7-xhigh", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "max"}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "max"}}, + id="opus-4.7-max", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "minimal"}, + {"thinking": {"type": "enabled", "budget_tokens": 1024}}, + id="haiku-4.5-minimal", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "low"}, + {"thinking": {"type": "enabled", "budget_tokens": 1024}}, + id="haiku-4.5-low", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "medium"}, + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="haiku-4.5-medium", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "high"}, + {"thinking": {"type": "enabled", "budget_tokens": 4096}}, + id="haiku-4.5-high", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "xhigh"}, + {"thinking": {"type": "enabled", "budget_tokens": 8192}}, + id="haiku-4.5-xhigh", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "max"}, + {"thinking": {"type": "enabled", "budget_tokens": 16384}}, + id="haiku-4.5-max", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "max", "max_tokens": 4000}, + {"thinking": {"type": "enabled", "budget_tokens": 3999}}, + id="haiku-4.5-budget-capped-below-max-tokens", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "high", "max_tokens": 1024}, + {}, + id="haiku-4.5-max-tokens-below-minimum-budget", + ), + pytest.param( + "claude-opus-4-7", + { + "reasoning_effort": "none", + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + }, + {}, + id="none-clears-thinking-and-effort", + ), + pytest.param( + "claude-haiku-4-5", + {"reasoning_effort": "high", "thinking": {"type": "enabled", "budget_tokens": 2000}}, + {"thinking": {"type": "enabled", "budget_tokens": 2000}}, + id="callers-thinking-wins", + ), + pytest.param( + "claude-opus-4-7", + {"reasoning_effort": "high", "output_config": {"effort": "low"}}, + {"thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "low"}}, + id="callers-effort-wins", + ), + ), +) +def test_reasoning_effort_becomes_the_thinking_shape_each_model_accepts( + gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue] +) -> None: + forwarded: Final = _forward(gateway, upstream_model, sent) + assert forwarded.reasoning == received, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +@pytest.mark.parametrize( + ("upstream_model", "sent", "received"), + ( + pytest.param( + "claude-haiku-4-5", + { + "temperature": 0, + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + }, + {"thinking": {"type": "enabled", "budget_tokens": 4096}}, + id="haiku-4.5-drops-temperature-0-with-effort", + ), + pytest.param( + "claude-opus-4-5", + { + "temperature": 0, + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + }, + {"output_config": {"effort": "high"}}, + id="opus-4.5-drops-temperature-0-with-effort", + ), + pytest.param( + "claude-haiku-4-5", + {"temperature": 0, "thinking": {"type": "enabled", "budget_tokens": 2048}}, + {"thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="haiku-4.5-drops-temperature-0-with-budget", + ), + pytest.param( + "claude-haiku-4-5", + {"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}}, + {"temperature": 1, "thinking": {"type": "enabled", "budget_tokens": 2048}}, + id="haiku-4.5-keeps-temperature-1", + ), + pytest.param( + "claude-haiku-4-5", + {"temperature": 0}, + {"temperature": 0}, + id="haiku-4.5-keeps-temperature-without-thinking", + ), + pytest.param( + "claude-opus-4-6", + { + "temperature": 0, + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + }, + { + "temperature": 0, + "thinking": {"type": "adaptive", "display": "omitted"}, + "output_config": {"effort": "high"}, + }, + id="opus-4.6-adaptive-keeps-temperature", + ), + ), +) +def test_temperature_is_dropped_only_when_a_non_adaptive_model_thinks( + gateway: Gateway, upstream_model: str, sent: dict[str, JsonValue], received: dict[str, JsonValue] +) -> None: + forwarded: Final = _forward(gateway, upstream_model, sent) + assert forwarded.reasoning == received, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +def _tool_loop(assistant_content: list[JsonValue]) -> dict[str, JsonValue]: + first_turn: Final = cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}")["messages"] + assert isinstance(first_turn, list) + tool_result: Final = {"type": "tool_result", "tool_use_id": "toolu_01", "content": "ok"} + return { + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "messages": [ + *first_turn, + {"role": "assistant", "content": assistant_content}, + {"role": "user", "content": [tool_result]}, + ], + } + + +@pytest.mark.parametrize( + ("sent_history", "received_history"), + ( + pytest.param( + [ + {"type": "thinking", "thinking": "bridge reasoning", "signature": "litellm_encrypted_reasoning:gAAAAB"}, + {"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:gAAAAC"}, + {"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"}, + {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}}, + ], + [ + {"type": "thinking", "thinking": "check the config", "signature": "EqQBCkgIBRABGAIiQL"}, + {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}}, + ], + id="encrypted-reasoning-from-another-provider-stripped-anthropic-signed-kept", + ), + pytest.param( + [ + {"type": "thinking", "thinking": "", "signature": "EqQBCkgIBRABGAIiQM"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"}, + {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}}, + ], + [ + {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzix"}, + {"type": "tool_use", "id": "toolu_01", "name": "Read", "input": {"file_path": "/repo/config.yaml"}}, + ], + id="empty-thinking-stripped-redacted-thinking-kept", + ), + ), +) +def test_thinking_history_keeps_only_blocks_anthropic_can_verify( + gateway: Gateway, sent_history: list[JsonValue], received_history: list[JsonValue] +) -> None: + forwarded: Final = _forward(gateway, "claude-haiku-4-5", _tool_loop(sent_history)) + assert forwarded.assistant_history == (received_history,), forwarded.assistant_history + assert forwarded.reasoning == {"thinking": {"type": "enabled", "budget_tokens": 2048}}, forwarded + assert forwarded.other_changes == {}, forwarded.other_changes + assert forwarded.reasoning_betas == cc.CLAUDE_CODE_REASONING_BETAS, forwarded.reasoning_betas + + +@pytest.mark.parametrize( + ("upstream_model", "reasoning_effort"), + ( + pytest.param("claude-haiku-4-5", "turbo", id="unknown-value"), + pytest.param("claude-opus-4-6", "xhigh", id="level-the-model-lacks"), + ), +) +def test_unsupported_reasoning_effort_is_rejected_before_reaching_anthropic( + gateway: Gateway, upstream_model: str, reasoning_effort: str +) -> None: + with wire_server(lambda request: Reply()) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{upstream_model}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY + ) + response: Final = gateway.request( + "POST", "/v1/messages", {**_claude_code_turn({"reasoning_effort": reasoning_effort}), "model": model} + ) + assert response.status_code == 400, response.text + assert response.json()["error"]["type"] == "invalid_request_error", response.text + assert wire.drain() == () diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py new file mode 100644 index 00000000000..1ada28b3355 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_response_wire.py @@ -0,0 +1,63 @@ +import uuid +from typing import Final + +from integration._support import claude_code as cc +from integration._support.client import Gateway +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_MODEL: Final = "claude-haiku-4-5" +_ANTHROPIC_CONTENT: Final = ( + {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"}, + {"type": "text", "text": "PONG"}, +) +_ANTHROPIC_USAGE: Final = {"input_tokens": 12, "output_tokens": 30, "output_tokens_details": {"thinking_tokens": 20}} + + +def _client_body(stream: bool) -> dict[str, JsonValue]: + return { + **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "stream": stream, + } + + +def test_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None: + def respond(request: Request) -> Reply: + return Reply( + chunks=cc.message_stream(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE), + content_type="text/event-stream", + ) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=True), "model": model}) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert cc.streamed_content(response.text) == [ + {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"}, + {"type": "text", "text": "PONG"}, + ], response.text + assert cc.streamed_usage(response.text) == { + "output_tokens": 30, + "output_tokens_details": {"thinking_tokens": 20}, + }, response.text + + +def test_non_streamed_thinking_blocks_and_thinking_token_count_reach_the_client_unchanged(gateway: Gateway) -> None: + def respond(request: Request) -> Reply: + return Reply(body=cc.message_reply(f"msg_{uuid.uuid4().hex}", _MODEL, _ANTHROPIC_CONTENT, _ANTHROPIC_USAGE)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=cc.ANTHROPIC_API_KEY) + response: Final = gateway.request("POST", "/v1/messages", {**_client_body(stream=False), "model": model}) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert response.json()["content"] == [ + {"type": "thinking", "thinking": "the user wants a single word", "signature": "EqQBCkgIBRABGAIiQLz"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy3va3pzixlit"}, + {"type": "text", "text": "PONG"}, + ], response.text + assert response.json()["usage"]["output_tokens_details"] == {"thinking_tokens": 20}, response.text diff --git a/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py new file mode 100644 index 00000000000..cf3c709028c --- /dev/null +++ b/tests/integration/messages_endpoint/providers/anthropic/reasoning/test_anthropic_reasoning_token_pricing_wire.py @@ -0,0 +1,64 @@ +import uuid +from typing import Final + +import pytest +from integration._support import claude_code as cc +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.wire import Reply, Request, wire_server + +_MODEL: Final = "claude-haiku-4-5" +_INPUT_RATE: Final = 1e-6 +_OUTPUT_RATE: Final = 2e-6 +_REASONING_RATE: Final = 7e-6 +_CONTENT: Final = ( + {"type": "thinking", "thinking": "count the words", "signature": "EqQBCkgIBRABGAIiQLz"}, + {"type": "text", "text": "PONG"}, +) +_USAGE: Final = {"input_tokens": 100, "output_tokens": 50, "output_tokens_details": {"thinking_tokens": 30}} + + +def _reply(identity: str, stream: bool) -> Reply: + if stream: + return Reply(chunks=cc.message_stream(identity, _MODEL, _CONTENT, _USAGE), content_type="text/event-stream") + return Reply(body=cc.message_reply(identity, _MODEL, _CONTENT, _USAGE)) + + +@pytest.mark.parametrize("stream", (pytest.param(False, id="non-streamed"), pytest.param(True, id="streamed"))) +def test_reported_thinking_tokens_are_billed_at_the_reasoning_rate_and_the_rest_at_the_output_rate( + gateway: Gateway, stream: bool +) -> None: + identity: Final = f"msg_{uuid.uuid4().hex}" + + def respond(request: Request) -> Reply: + return _reply(identity, stream) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{_MODEL}", + api_base=wire.url, + api_key=cc.ANTHROPIC_API_KEY, + input_cost_per_token=_INPUT_RATE, + output_cost_per_token=_OUTPUT_RATE, + output_cost_per_reasoning_token=_REASONING_RATE, + ) + body: Final = { + **cc.claude_code_request(f"cache-bust-{uuid.uuid4().hex}"), + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "stream": stream, + "model": model, + } + response: Final = gateway.request("POST", "/v1/messages", body) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + rows: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=70, + ) + input_tokens, output_tokens, thinking_tokens = 100, 50, 30 + assert float(rows[0]["spend"]) == pytest.approx( + input_tokens * _INPUT_RATE + + (output_tokens - thinking_tokens) * _OUTPUT_RATE + + thinking_tokens * _REASONING_RATE + ), rows