litellm/tests/claude_code/tool_use_streaming/test_azure.py
mateo-berri d05e45893e compat-matrix: parallel-fanout refactor + 5 new feature dirs + rate limiter
- Switch all per-cell tests from @pytest.mark.parametrize("model", ...)
  (3 sequential invocations) to a single test that fans out to all 3
  Claude tiers via run_claude_models_parallel. Per-cell wall time is now
  bounded by the slowest model rather than the sum.

- Add 5 new v0 feature dirs (5 providers each, 25 new test files):
    web_search, pdf_input, prompt_caching_1h,
    tool_use_streaming, thinking_with_tool_use
  Manifest expanded to match.

- Add cross-process token-bucket rate limiter (rate_limiter.py + tests)
  so xdist workers stay under per-provider req/s limits during full-grid
  runs. New env knobs: LITELLM_COMPAT_RATE_{ANTHROPIC,AZURE,VERTEX_AI,
  BEDROCK_CONVERSE,BEDROCK_INVOKE}.

- conftest.py: write per-worker shards under <artifact>.shards/, merge
  in the controller; preserve the "don't write empty artifact" guard so
  unit-test runs don't clobber a real compat-results.json.

- Vertex test_config.yaml: route project/location through env so the
  cron VM can target a different GCP project than the upstream default.

- Add run_compat.sh wrapper for binary-searching ideal req/s per
  provider against compat-rate-limit-summary.json output.
2026-05-06 23:31:19 +00:00

122 lines
3.8 KiB
Python

"""tool_use_streaming x Microsoft Foundry (Azure).
Drive the real `claude` CLI in headless `--output-format stream-json`
mode against a running LiteLLM proxy that routes Claude requests to
Microsoft Foundry's Anthropic deployments on Azure, ask Claude to
invoke a built-in tool (`Bash`), and assert that the upstream (a)
emitted a `tool_use` content block and (b) actually streamed events
incrementally.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/tool_use_streaming/test_azure.py
^^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from tests.claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
TOOL_USE_PROMPT = (
"Use the Bash tool to run the command `echo pong` and report what it printed."
)
TOOL_USE_ARGS = ["--allowed-tools", "Bash"]
MIN_STREAM_EVENTS = 4
def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_use":
return True
return False
def test_tool_use_streaming_azure(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=TOOL_USE_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=TOOL_USE_ARGS,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_tool_use_event(outcome.events):
error = (
f"[{model}] no tool_use content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if len(outcome.events) < MIN_STREAM_EVENTS:
error = (
f"[{model}] only {len(outcome.events)} stream-json events observed "
f"(< {MIN_STREAM_EVENTS}); proxy likely buffered the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)