mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
Anthropic and Microsoft announced Claude Haiku 4.5, Sonnet 4.5/4.6, and
Opus 4.1/4.6/4.7 in Microsoft Foundry on 2025-11-18, so the matrix's
Azure column should exercise a real route through the LiteLLM proxy
rather than report not_applicable.
Foundry serves Claude on an Anthropic-shape /anthropic/v1/messages
endpoint (not the Azure OpenAI chat-completions route), and LiteLLM
already supports it via the azure_ai/claude-* provider prefix
(litellm/llms/azure_ai/anthropic/{handler,transformation,messages_transformation}.py).
- test_config.yaml: add 3 azure aliases pointing at azure_ai/claude-*
with AZURE_FOUNDRY_API_BASE / AZURE_FOUNDRY_API_KEY env
- 6x test_azure.py: replace not_applicable stubs with real run_claude
drivers, mirroring the existing test_vertex_ai.py shape exactly
- sample_compatibility-matrix.json: Azure cells flip to pass
- _builder_unit_tests: pin the new invariant (run_claude is used,
not_applicable is gone) and feed pass results across all 5 providers
in the 6x5 golden test
116 lines
3.8 KiB
Python
116 lines
3.8 KiB
Python
"""extended_thinking x Azure (Microsoft Foundry).
|
|
|
|
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
|
Claude requests to Anthropic's models hosted in Microsoft Foundry on
|
|
Azure, enable extended thinking via `MAX_THINKING_TOKENS`, and assert
|
|
that the upstream returned a `thinking` content block.
|
|
|
|
Foundry's Claude deployments advertise `supports_reasoning: true` in
|
|
LiteLLM's pricing metadata; the `thinking={"type": "enabled", ...}`
|
|
parameter passes through `azure_ai/claude-*` to Foundry's
|
|
`/anthropic/v1/messages` endpoint unchanged. Note that
|
|
`claude-opus-4-7-preview` documents thinking as not supported on
|
|
Foundry; if that lands, this row may flip to a partial pass and we'll
|
|
re-evaluate.
|
|
|
|
The (feature, provider) for this cell is inferred from the file path by
|
|
`tests/claude_code/conftest.py`:
|
|
|
|
tests/claude_code/extended_thinking/test_azure.py
|
|
^^^^^^^^^^^^^^^^^ ^^^^^
|
|
feature_id provider
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from typing import Any, Mapping, Sequence
|
|
|
|
import pytest
|
|
|
|
from tests.claude_code.cli_driver import ClaudeCLIError, run_claude
|
|
|
|
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
|
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
|
|
|
AZURE_MODELS = [
|
|
"claude-haiku-4-5-azure",
|
|
"claude-sonnet-4-6-azure",
|
|
"claude-opus-4-7-azure",
|
|
]
|
|
|
|
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
|
THINKING_PROMPT = (
|
|
"Think step by step: if I have three apples and eat two, how many remain? "
|
|
"Answer with the single digit only."
|
|
)
|
|
|
|
|
|
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|
for event in events:
|
|
if event.get("type") != "assistant":
|
|
continue
|
|
message = event.get("message") or {}
|
|
content = message.get("content")
|
|
if not isinstance(content, list):
|
|
continue
|
|
for block in content:
|
|
if isinstance(block, dict) and block.get("type") == "thinking":
|
|
return True
|
|
return False
|
|
|
|
|
|
@pytest.mark.parametrize("model", AZURE_MODELS)
|
|
def test_extended_thinking_azure(compat_result, model):
|
|
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
|
enabled and assert a `thinking` content block was emitted."""
|
|
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
|
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
|
if not base_url or not api_key:
|
|
compat_result.set(
|
|
{
|
|
"status": "fail",
|
|
"error": (
|
|
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
|
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
|
),
|
|
}
|
|
)
|
|
pytest.fail(
|
|
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
|
)
|
|
|
|
try:
|
|
result = run_claude(
|
|
prompt=THINKING_PROMPT,
|
|
model=model,
|
|
base_url=base_url,
|
|
api_key=api_key,
|
|
extra_env=THINKING_ENV,
|
|
)
|
|
except ClaudeCLIError as exc:
|
|
compat_result.set({"status": "fail", "error": f"[{model}] {exc}"})
|
|
pytest.fail(str(exc), pytrace=False)
|
|
return
|
|
|
|
if result.exit_code != 0:
|
|
compat_result.set(
|
|
{
|
|
"status": "fail",
|
|
"error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}",
|
|
}
|
|
)
|
|
pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False)
|
|
return
|
|
|
|
if not _has_thinking_block(result.events):
|
|
compat_result.set(
|
|
{
|
|
"status": "fail",
|
|
"error": f"[{model}] no `thinking` content block observed in stream-json events",
|
|
}
|
|
)
|
|
pytest.fail(f"no thinking block for {model}", pytrace=False)
|
|
return
|
|
|
|
compat_result.set({"status": "pass"})
|