mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-22 00:31:44 +00:00
Matrix grows from 11 to 15 feature rows. All new tests collected + 180 unit tests still pass; smoke runs hit real LiteLLM bug surfaces on bedrock_invoke, bedrock_converse, and vertex_ai (cells correctly red in PR #142). Rename ------ `extended_thinking` -> `thinking` (directory, manifest id+name, 5 test fn names, 5 docstrings, builder unit-test fixtures, sample JSON, run_compat.sh). Existing test logic already covers both manual (`thinking.type=enabled`, Haiku 4.5) and adaptive (`thinking.type=adaptive`, Opus 4.7) shapes because Claude Code picks the shape per model from `--effort max`; the name change just stops the column from looking like a Claude 3.7 reference. New rows -------- - structured_outputs (5 files, CLI `--json-schema`). Claude Code synthesizes a single `StructuredOutput` tool from the schema and surfaces the tool_use input as `structured_output` on the trailing `result` event. Test ships its own `_validate_against_schema` so we don't take a jsonschema dep just for matrix surface. - count_tokens (5 files, HTTP probe). POSTs the proxy's `/v1/messages/count_tokens` directly and asserts the response is `{input_tokens: positive int}`. No CLI hook exists for this endpoint; the test goes through the new http_probe helper instead. - tool_search (5 files, HTTP probe). Sends `tools: [{type: tool_search_tool_regex_20251119, name: tool_search_tool_regex}]` and asserts the proxy doesn't 400. MCP fan-out via `--mcp-config` would also exercise the tool-search beta header path, but it's flaky w.r.t. Claude Code's internal tool-deferral threshold; the HTTP probe hits the actual bug surface (per-provider beta-header translation `advanced-tool-use-2025-11-20` vs `tool-search-tool-2025-10-19`). - long_context_1m (5 files, CLI `--betas context-1m-2025-08-07 --max-budget-usd 6`). A ~210k-token padded prompt over stdin exercises the 1M-context beta. Sonnet 4.6 + Opus 4.7 only -- Haiku 4.5's window is 200k, so it's excluded from MODELS (not marked not_applicable) to keep the per-cell aggregator semantics intact. Prompt uses a document-style preamble + 8 cycling pangrams rather than repeating identical chunks; without that, Opus 4.7 trips the safety filter mid-response with a Usage Policy refusal. `--max-budget-usd 6` is a runaway-loop guard, ~2x worst-case Opus per-cell spend. New helper ---------- `tests/claude_code/http_probe.py`: shared `ProbeResult` dataclass plus per-endpoint `probe_*` + `assert_*_shape` pairs for the HTTP-probe rows. Uses httpx with `anthropic-version: 2023-06-01` and a 30s timeout.
141 lines
2.7 KiB
JSON
141 lines
2.7 KiB
JSON
{
|
|
"schema_version": "1",
|
|
"generated_at": "2026-04-25T00:00:00Z",
|
|
"litellm_version": "v1.83.0-stable",
|
|
"claude_code_version": "2.1.120",
|
|
"providers": [
|
|
"anthropic",
|
|
"bedrock_invoke",
|
|
"bedrock_converse",
|
|
"vertex_ai",
|
|
"azure"
|
|
],
|
|
"features": [
|
|
{
|
|
"id": "basic_messaging_non_streaming",
|
|
"name": "Basic messaging (non-streaming)",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"id": "basic_messaging_streaming",
|
|
"name": "Basic messaging (streaming)",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"id": "tool_use",
|
|
"name": "Tool use",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"id": "prompt_caching_5m",
|
|
"name": "Prompt caching (5m TTL)",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"id": "vision",
|
|
"name": "Vision",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"id": "thinking",
|
|
"name": "Thinking",
|
|
"providers": {
|
|
"anthropic": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_invoke": {
|
|
"status": "pass"
|
|
},
|
|
"bedrock_converse": {
|
|
"status": "pass"
|
|
},
|
|
"vertex_ai": {
|
|
"status": "pass"
|
|
},
|
|
"azure": {
|
|
"status": "pass"
|
|
}
|
|
}
|
|
}
|
|
]
|
|
}
|