"""thinking x Vertex AI. Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Anthropic's models on Google Cloud Vertex AI, enable extended thinking via `--effort high`, and assert that the upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by `tests/e2e/claude_code/conftest.py`: tests/e2e/claude_code/thinking/test_vertex_ai.py ^^^^^^^^ ^^^^^^^^^ feature_id provider """ from __future__ import annotations from typing import Any, Mapping, Sequence import pytest from claude_code._env import require_proxy from claude_code.cli_driver import ( ClaudeCLIError, failure_diagnostic, run_claude_models_parallel, ) VERTEX_AI_MODELS = [ "claude-haiku-4-5-vertex", "claude-sonnet-4-5-vertex", "claude-opus-4-7-vertex", ] THINKING_ARGS = ["--effort", "max"] THINKING_PROMPT = ( "I have a 3-gallon jug and a 5-gallon jug. How can I measure " "exactly 4 gallons of water? Think through the steps carefully." ) def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: for event in events: if event.get("type") != "assistant": continue message = event.get("message") or {} content = message.get("content") if not isinstance(content, list): continue for block in content: if isinstance(block, dict) and block.get("type") == "thinking": return True return False @pytest.mark.covers("llm.messages.vertex.thinking.nonstream.works") def test_thinking_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url, api_key = require_proxy(compat_result) outcomes = run_claude_models_parallel( models=VERTEX_AI_MODELS, prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, extra_args=THINKING_ARGS, ) failures = [] for model in VERTEX_AI_MODELS: outcome = outcomes[model] if isinstance(outcome, ClaudeCLIError): error = f"[{model}] {outcome}" compat_result.add({"status": "fail", "error": error}) failures.append(error) continue if outcome.exit_code != 0: error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" compat_result.add({"status": "fail", "error": error}) failures.append(error) continue if not _has_thinking_block(outcome.events): error = ( f"[{model}] no `thinking` content block observed in stream-json events" ) compat_result.add({"status": "fail", "error": error}) failures.append(error) continue compat_result.add({"status": "pass"}) if failures: pytest.fail("; ".join(failures), pytrace=False)