mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
LiteLLM sends the Bedrock Mantle GPT rows through Bedrock's Responses endpoint, which refuses minimal on gpt-5.4 and gpt-5.5 like every other Bedrock GPT row. The earlier commit measured the raw chat endpoint, which accepts it, and dropped the flag by mistake. The ladder test now matches what the proxy path returns
268 lines
11 KiB
Python
268 lines
11 KiB
Python
from __future__ import annotations
|
|
|
|
import importlib.util
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
from types import MappingProxyType
|
|
from typing import Final
|
|
|
|
import jsonschema
|
|
import pytest
|
|
|
|
from litellm.llms.openai.chat.gpt_5_transformation import is_gpt_reasoning_series_name
|
|
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
|
|
|
|
REPO_ROOT = Path(__file__).parents[2]
|
|
GENERATOR_PATH = REPO_ROOT / "ci_cd" / "generate_model_prices_schema.py"
|
|
PRICES_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
|
BACKUP_PRICES_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
|
SCHEMA_PATH = REPO_ROOT / "model_prices_and_context_window.schema.json"
|
|
|
|
|
|
def build_validator(schema: dict) -> jsonschema.Draft202012Validator:
|
|
return jsonschema.Draft202012Validator(schema, format_checker=jsonschema.Draft202012Validator.FORMAT_CHECKER)
|
|
|
|
|
|
def load_generator():
|
|
spec = importlib.util.spec_from_file_location("generate_model_prices_schema", GENERATOR_PATH)
|
|
assert spec is not None and spec.loader is not None
|
|
module = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(module)
|
|
return module
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def committed_schema() -> dict:
|
|
return json.loads(SCHEMA_PATH.read_text())
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def prices() -> dict:
|
|
return json.loads(PRICES_PATH.read_text())
|
|
|
|
|
|
def test_committed_schema_matches_generator_output(prices: dict, committed_schema: dict):
|
|
generator = load_generator()
|
|
regenerated = json.loads(generator.render(generator.build_schema(prices)))
|
|
assert regenerated == committed_schema, (
|
|
"model_prices_and_context_window.schema.json is out of sync; "
|
|
"run `python ci_cd/generate_model_prices_schema.py` and commit the result"
|
|
)
|
|
|
|
|
|
def test_prices_file_validates_against_committed_schema(prices: dict, committed_schema: dict):
|
|
validator = build_validator(committed_schema)
|
|
errors = [
|
|
f"{'.'.join(str(part) for part in error.absolute_path)}: {error.message}"
|
|
for error in validator.iter_errors(prices)
|
|
]
|
|
assert errors == []
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"entry",
|
|
[
|
|
{"litellm_provider": "openai", "mode": "chat", "input_cost_per_token": "0.01"},
|
|
{"litellm_provider": "openai", "mode": "chat", "input_cost_per_token": -1},
|
|
{"litellm_provider": "openai", "mode": "not_a_real_mode"},
|
|
{"mode": "chat"},
|
|
{"litellm_provider": "openai", "deprecation_date": "June 2026"},
|
|
{"litellm_provider": "openai", "deprecation_date": "2026-99-99"},
|
|
{"litellm_provider": "openai", "deprecation_date": "2026-13-01"},
|
|
{"litellm_provider": "openai", "deprecation_date": "2026-01-32"},
|
|
{"litellm_provider": "openai", "deprecation_date": "2026-01-00"},
|
|
{"litellm_provider": "openai", "deprecation_date": "2026-02-31"},
|
|
{"litellm_provider": "openai", "supported_modalities": ["smell"]},
|
|
{"litellm_provider": "openai", "supports_vision": "yes"},
|
|
{"litellm_provider": "openai", "max_tokens": 8191.5},
|
|
{"litellm_provider": "openai", "tiered_pricing": [{"unknown_tier_field": 1}]},
|
|
],
|
|
ids=[
|
|
"cost_as_string",
|
|
"negative_cost",
|
|
"unknown_mode",
|
|
"missing_provider",
|
|
"non_iso_deprecation_date",
|
|
"impossible_month_and_day",
|
|
"month_out_of_range",
|
|
"day_out_of_range",
|
|
"day_zero",
|
|
"calendar_impossible_day",
|
|
"unknown_modality",
|
|
"boolean_flag_as_string",
|
|
"fractional_max_tokens",
|
|
"unknown_tiered_pricing_field",
|
|
],
|
|
)
|
|
def test_schema_rejects_malformed_entries(committed_schema: dict, entry: dict):
|
|
validator = build_validator(committed_schema)
|
|
assert not validator.is_valid({"some-model": entry})
|
|
|
|
|
|
def test_schema_accepts_minimal_and_unknown_optional_fields(committed_schema: dict):
|
|
validator = build_validator(committed_schema)
|
|
assert validator.is_valid({"some-model": {"litellm_provider": "openai"}})
|
|
assert validator.is_valid({"some-model": {"litellm_provider": "openai", "brand_new_field": {"nested": True}}})
|
|
|
|
|
|
def test_schema_accepts_cache_creation_cost_inside_a_pricing_tier(committed_schema: dict):
|
|
validator = build_validator(committed_schema)
|
|
entry = {
|
|
"litellm_provider": "dashscope",
|
|
"mode": "chat",
|
|
"tiered_pricing": [
|
|
{
|
|
"range": [0, 256000],
|
|
"input_cost_per_token": 3.25e-07,
|
|
"output_cost_per_token": 1.95e-06,
|
|
"cache_creation_input_token_cost": 4.063e-07,
|
|
"cache_read_input_token_cost": 3.25e-08,
|
|
}
|
|
],
|
|
}
|
|
assert validator.is_valid({"some-model": entry})
|
|
|
|
|
|
def find_duplicate_keys(path: Path) -> list[str]:
|
|
duplicates: list[str] = []
|
|
|
|
def record_duplicates(pairs):
|
|
seen: set[str] = set()
|
|
for key, _ in pairs:
|
|
if key in seen:
|
|
duplicates.append(key)
|
|
seen.add(key)
|
|
return dict(pairs)
|
|
|
|
json.loads(path.read_text(), object_pairs_hook=record_duplicates)
|
|
return duplicates
|
|
|
|
|
|
@pytest.mark.parametrize("path", (PRICES_PATH, BACKUP_PRICES_PATH), ids=("main", "backup"))
|
|
def test_price_map_has_no_duplicate_keys(path: Path):
|
|
assert find_duplicate_keys(path) == [], (
|
|
f"{path.name} defines the same key twice; JSON parsers keep only the last "
|
|
"occurrence, so the earlier entry's fields are silently dropped. This is what "
|
|
"a clean text merge of two branches that both added a model looks like: "
|
|
"deduplicate the keys into one entry"
|
|
)
|
|
|
|
|
|
DATED_VARIANT = re.compile(r"^(.*?)-(\d{4}-\d{2}-\d{2})$")
|
|
SERVICE_TIER_SUFFIXES = ("_flex", "_priority")
|
|
|
|
|
|
def tier_anchor(tier_key: str) -> str:
|
|
matched = next(suffix for suffix in SERVICE_TIER_SUFFIXES if tier_key.endswith(suffix))
|
|
return tier_key[: -len(matched)]
|
|
|
|
|
|
def test_dated_variants_carry_base_alias_service_tier_pricing(prices: dict):
|
|
drifted = [
|
|
f"{name}: missing {tier_key}={base[tier_key]} (base alias {match.group(1)})"
|
|
for name, entry in prices.items()
|
|
if isinstance(entry, dict)
|
|
for match in [DATED_VARIANT.match(name)]
|
|
if match is not None
|
|
for base in [prices.get(match.group(1))]
|
|
if isinstance(base, dict)
|
|
for tier_key in base
|
|
if tier_key.endswith(SERVICE_TIER_SUFFIXES)
|
|
and tier_anchor(tier_key) in base
|
|
and entry.get(tier_anchor(tier_key)) == base[tier_anchor(tier_key)]
|
|
and entry.get(tier_key) != base[tier_key]
|
|
]
|
|
assert drifted == [], (
|
|
"dated model variants are missing flex/priority pricing their base alias has; "
|
|
"sync the tier keys so service-tier requests against pinned snapshots are not "
|
|
"billed at standard rates:\n" + "\n".join(drifted)
|
|
)
|
|
|
|
|
|
OPENAI_REASONING_FAMILY_MARKERS = ("codex", "deep-research", "chat-latest")
|
|
|
|
|
|
def is_openai_o_series(name: str) -> bool:
|
|
return len(name) > 1 and name[0] == "o" and name[1].isdigit()
|
|
|
|
|
|
def is_openai_reasoning_family(name: str) -> bool:
|
|
base = name.split("/")[-1].removeprefix("ft:")
|
|
if "search-api" in base:
|
|
return False
|
|
return (
|
|
is_openai_o_series(base)
|
|
or is_gpt_reasoning_series_name(base)
|
|
or any(marker in base for marker in OPENAI_REASONING_FAMILY_MARKERS)
|
|
)
|
|
|
|
|
|
def test_openai_reasoning_family_entries_carry_supports_reasoning(prices: dict):
|
|
unflagged = [
|
|
name
|
|
for name, entry in prices.items()
|
|
if isinstance(entry, dict)
|
|
and entry.get("litellm_provider") == "openai"
|
|
and is_openai_reasoning_family(name)
|
|
and entry.get("supports_reasoning") is not True
|
|
]
|
|
assert unflagged == [], (
|
|
"OpenAI o-series, gpt-5+, codex, deep-research, and chat-latest models are reasoning "
|
|
"models, and the Responses API drops the `reasoning` param for any mapped OpenAI model "
|
|
"whose entry lacks supports_reasoning; flag these entries:\n" + "\n".join(unflagged)
|
|
)
|
|
|
|
|
|
def test_chat_latest_declares_the_one_effort_openai_accepts(prices: dict):
|
|
"""OpenAI rejects every reasoning.effort on chat-latest except medium, and a reasoning entry
|
|
with no declared levels resolves to None, which lets /model_group/info and the dashboard offer
|
|
levels the upstream will 400 on."""
|
|
assert resolve_supported_reasoning_efforts(prices["chat-latest"], deployment_is_mapped=True) == ("medium",)
|
|
|
|
|
|
BEDROCK_OPENAI_GPT_MARKERS: Final = ("openai.gpt-5.4", "openai.gpt-5.5", "openai.gpt-5.6", "openai.gpt-6-astra")
|
|
BEDROCK_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse", "bedrock_mantle"))
|
|
BEDROCK_ROW_PREFIXES: Final = ("bedrock_mantle/", "us.", "global.")
|
|
GPT_5_4_BEDROCK_LADDER: Final = ("none", "low", "medium", "high", "xhigh")
|
|
GPT_5_6_BEDROCK_LADDER: Final = ("none", "low", "medium", "high", "xhigh", "max")
|
|
GPT_6_ASTRA_BEDROCK_LADDER: Final = ("low", "medium", "high", "xhigh", "max")
|
|
BEDROCK_OPENAI_GPT_LADDERS: Final = MappingProxyType(
|
|
{
|
|
"bedrock_mantle/openai.gpt-5.4": GPT_5_4_BEDROCK_LADDER,
|
|
"bedrock_mantle/openai.gpt-5.5": GPT_5_4_BEDROCK_LADDER,
|
|
**{
|
|
f"{prefix}openai.gpt-5.6-{variant}": GPT_5_6_BEDROCK_LADDER
|
|
for prefix in BEDROCK_ROW_PREFIXES
|
|
for variant in ("luna", "sol", "terra")
|
|
},
|
|
**{f"{prefix}openai.gpt-6-astra": GPT_6_ASTRA_BEDROCK_LADDER for prefix in BEDROCK_ROW_PREFIXES},
|
|
}
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("name", "ladder"), tuple(BEDROCK_OPENAI_GPT_LADDERS.items()), ids=tuple(BEDROCK_OPENAI_GPT_LADDERS)
|
|
)
|
|
def test_bedrock_openai_gpt_rows_advertise_the_ladder_bedrock_accepts(prices: dict, name: str, ladder: tuple[str, ...]):
|
|
"""Each ladder is the set of levels Bedrock answered 200 to for that row through the proxy on
|
|
2026-09-11 (PR #40740): the Mantle rows go out over its Responses endpoint and the Converse rows
|
|
over inference profiles. Bedrock differs from the direct OpenAI rows in two places, gpt-5.6 and
|
|
gpt-6-astra take max there, and gpt-6-astra refuses none; minimal is refused on every row.
|
|
xhigh and max are opt-in for the resolver, so a row missing either flag silently drops that
|
|
level from every group it belongs to."""
|
|
assert resolve_supported_reasoning_efforts(prices[name], deployment_is_mapped=True) == ladder
|
|
|
|
|
|
def test_every_bedrock_openai_gpt_row_advertises_xhigh(prices: dict):
|
|
"""The GovCloud and gpt-5.6-cyber rows cannot be called from our account, so they carry the
|
|
family's xhigh flag rather than a measured ladder."""
|
|
missing: Final = [
|
|
name
|
|
for name, entry in prices.items()
|
|
if isinstance(entry, dict)
|
|
and entry.get("litellm_provider") in BEDROCK_PROVIDERS
|
|
and any(marker in name for marker in BEDROCK_OPENAI_GPT_MARKERS)
|
|
and "xhigh" not in (resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) or ())
|
|
]
|
|
assert missing == []
|