From 77885779caebecf54e29dc77ef13a321dc7a5f45 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 10 Jul 2026 20:41:35 -0700 Subject: [PATCH 1/5] refactor(fallback-generalizations): split rules into routing and provider-neutral capability kinds --- basedpyright-code-budget.json | 10 +- .../fallback_generalizations.py | 209 ++++---- .../get_llm_provider_logic.py | 8 +- ...odel_prices_and_context_window_backup.json | 49 +- litellm/utils.py | 25 +- model_prices_and_context_window.json | 49 +- tests/local_testing/test_get_llm_provider.py | 10 +- .../test_fallback_generalizations.py | 472 ++++++++++-------- .../test_get_model_cost_map.py | 49 +- type-discipline-budget.json | 4 +- 10 files changed, 474 insertions(+), 411 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index cb3427bed4d..8c5bf165c7e 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5900 }, "reportMissingTypeArgument": { - "limit": 15918 + "limit": 15913 }, "reportMissingTypeStubs": { "limit": 41 @@ -105,13 +105,13 @@ "limit": 113 }, "reportUnknownMemberType": { - "limit": 40541 + "limit": 40539 }, "reportUnknownParameterType": { - "limit": 20418 + "limit": 20413 }, "reportUnknownVariableType": { - "limit": 32151 + "limit": 32144 }, "reportUnnecessaryCast": { "limit": 177 @@ -123,7 +123,7 @@ "limit": 7 }, "reportUnnecessaryIsInstance": { - "limit": 1212 + "limit": 1211 }, "reportUntypedBaseClass": { "limit": 165 diff --git a/litellm/litellm_core_utils/fallback_generalizations.py b/litellm/litellm_core_utils/fallback_generalizations.py index 8b5a797cbd7..c9b6d83e9f1 100644 --- a/litellm/litellm_core_utils/fallback_generalizations.py +++ b/litellm/litellm_core_utils/fallback_generalizations.py @@ -3,151 +3,166 @@ Declarative fallback generalizations for unknown / newly-released models. The ``fallback_generalizations`` block in ``model_prices_and_context_window.json`` holds an ordered list of rules. Each rule pairs a single case-insensitive regex -with the metadata to apply when a model name has no exact entry in the cost map. -The metadata is a partial cost-map entry: ``litellm_provider`` drives provider -routing, and the remaining fields (``mode``, ``supports_*``, context window, -pricing, ...) drive ``get_model_info`` / ``supports_*``. +with a ``model_info`` dict, and the structure of ``model_info`` decides which of +two kinds the rule is. -Precedence: rules are evaluated in file order and the first match wins. Callers -with extra constraints (model-info resolution checks the provider) use -``match_all_fallback_generalizations`` to skip inapplicable earlier rules instead -of discarding the model name. Rules are consulted only after exact and -case-insensitive lookups miss, so an exact entry always takes precedence over a -rule. +A ROUTING rule carries exactly one ``model_info`` key, ``litellm_provider``. It is +consumed only by ``get_llm_provider`` bare-id inference: the first routing rule +whose regex matches decides the provider. Routing rules never contribute to model +info. + +A CAPABILITY rule carries any ``model_info`` keys except ``litellm_provider`` +(``mode``, ``supports_*``, context window, pricing, ...). It is consumed by +``get_model_info`` fallback resolution: the ``model_info`` of ALL capability rules +whose regex matches is unioned in file order, with later rules overriding earlier +ones on key conflicts, and the caller backfills ``litellm_provider`` with the +provider it requested. If no capability rule matches, model-info resolution misses +as if no rules existed. + +A rule that mixes ``litellm_provider`` with other ``model_info`` keys is invalid: +``set_fallback_generalizations`` logs a warning and skips it at install time (a +warning rather than a crash, because released proxies fetch this JSON remotely). +Rules are only consulted after exact and case-insensitive lookups miss, so an +exact cost-map entry always takes precedence over any rule. Patterns are matched case-insensitively with ``re.search`` and are not implicitly -anchored: a rule must include ``^`` and ``$`` (as the shipped rules do) to bind to -the whole model name, otherwise it matches as a substring. Keeping anchoring in the -regex makes the rule the single, self-contained source of truth for what it matches. - -A rule may set ``extends`` to the ``name`` of another rule to inherit that rule's -``model_info``; the rule's own ``model_info`` overrides the inherited keys, so a -narrow rule (for example a version-gated capability flag) carries only its delta -instead of duplicating the parent's pricing block. Inheritance is resolved once, -at install time, against each rule's raw (unresolved) ``model_info``; it is a -single level (a parent that itself extends is not chained). +anchored: a rule must include ``^`` and ``$`` to bind to the whole model name, +otherwise it matches as a substring. Keeping anchoring in the regex makes the rule +the single, self-contained source of truth for what it matches. Any other keys on a rule (for example a free-text ``description`` documenting what the regex matches) are ignored by the engine and exist only for the reader. -The compiled-regex list is built once and cached. ``match_fallback_generalization`` -is O(number of rules); callers must only invoke it on a cache miss. +Rules are compiled and classified once, at install time. The match functions are +O(number of rules); callers must only invoke them on a cache miss. """ import re -from typing import Optional +from dataclasses import dataclass +from typing import Optional, Union from litellm._logging import verbose_logger NAME_FIELD = "name" PATTERN_FIELD = "pattern" MODEL_INFO_FIELD = "model_info" -EXTENDS_FIELD = "extends" +PROVIDER_KEY = "litellm_provider" -def _resolve_extends(rules: list) -> list: - """Expand ``extends`` inheritance so each rule's ``model_info`` is self-contained. +@dataclass(frozen=True, slots=True) +class _RoutingRule: + pattern: re.Pattern + provider: str - A rule with ``extends: `` is rewritten with ``model_info`` set to the parent's - ``model_info`` overlaid by its own. Resolution is single-level and uses each rule's - raw ``model_info`` as the parent source. Non-dict rules and dangling parents are - passed through unchanged. - """ - base_by_name = { - rule[NAME_FIELD]: rule[MODEL_INFO_FIELD] - for rule in rules - if isinstance(rule, dict) - and isinstance(rule.get(NAME_FIELD), str) - and isinstance(rule.get(MODEL_INFO_FIELD), dict) - } - def resolved(rule: dict) -> dict: - parent_name = rule.get(EXTENDS_FIELD) - own_info = rule.get(MODEL_INFO_FIELD) - parent_info = base_by_name.get(parent_name) if isinstance(parent_name, str) else None - if parent_info is None or not isinstance(own_info, dict): - return rule - return {**rule, MODEL_INFO_FIELD: {**parent_info, **own_info}} +@dataclass(frozen=True, slots=True) +class _CapabilityRule: + pattern: re.Pattern + model_info: dict - return [resolved(rule) if isinstance(rule, dict) else rule for rule in rules] + +_CompiledRule = Union[_RoutingRule, _CapabilityRule] + + +def _compile_rule(rule: object) -> Optional[_CompiledRule]: + if not isinstance(rule, dict): + return None + pattern = rule.get(PATTERN_FIELD) + model_info = rule.get(MODEL_INFO_FIELD) + if not isinstance(pattern, str) or not isinstance(model_info, dict): + verbose_logger.warning( + "LiteLLM: skipping malformed fallback generalization rule %s (needs string '%s' and dict '%s').", + rule.get(NAME_FIELD, pattern), + PATTERN_FIELD, + MODEL_INFO_FIELD, + ) + return None + try: + compiled = re.compile(pattern, re.IGNORECASE) + except re.error as e: + verbose_logger.warning( + "LiteLLM: skipping fallback generalization rule with invalid regex %r: %s", + pattern, + e, + ) + return None + if PROVIDER_KEY not in model_info: + return _CapabilityRule(pattern=compiled, model_info=model_info) + provider = model_info[PROVIDER_KEY] + if len(model_info) == 1 and isinstance(provider, str): + return _RoutingRule(pattern=compiled, provider=provider) + verbose_logger.warning( + "LiteLLM: skipping invalid fallback generalization rule %s: a routing rule's '%s' must contain " + "'%s' as its only key (a string), and a capability rule must not contain '%s' at all.", + rule.get(NAME_FIELD, pattern), + MODEL_INFO_FIELD, + PROVIDER_KEY, + PROVIDER_KEY, + ) + return None class _FallbackGeneralizations: - """Holds the active rule list and its lazily-compiled regex cache.""" + """Holds the raw rule list and its install-time-compiled routing and capability rules.""" def __init__(self) -> None: - self.rules: list[dict] = [] - self._compiled: Optional[list[tuple[re.Pattern, dict]]] = None + self.rules: list = [] + self.routing_rules: tuple = () + self.capability_rules: tuple = () - def set_rules(self, rules: Optional[list[dict]]) -> None: - self.rules = rules if isinstance(rules, list) else [] - self._compiled = None + def set_rules(self, rules: Optional[list]) -> None: + installed = rules if isinstance(rules, list) else [] + compiled = tuple(c for c in (_compile_rule(rule) for rule in installed) if c is not None) + self.rules = installed + self.routing_rules = tuple(rule for rule in compiled if isinstance(rule, _RoutingRule)) + self.capability_rules = tuple(rule for rule in compiled if isinstance(rule, _CapabilityRule)) - def _compile(self) -> list[tuple[re.Pattern, dict]]: - compiled: list[tuple[re.Pattern, dict]] = [] - for rule in self.rules: - if not isinstance(rule, dict): - continue - pattern = rule.get(PATTERN_FIELD) - model_info = rule.get(MODEL_INFO_FIELD) - if not isinstance(pattern, str) or not isinstance(model_info, dict): - verbose_logger.warning( - "LiteLLM: skipping malformed fallback generalization rule %s (needs string '%s' and dict '%s').", - rule.get("name", pattern), - PATTERN_FIELD, - MODEL_INFO_FIELD, - ) - continue - try: - compiled.append((re.compile(pattern, re.IGNORECASE), model_info)) - except re.error as e: - verbose_logger.warning( - "LiteLLM: skipping fallback generalization rule with invalid regex %r: %s", - pattern, - e, - ) - return compiled - - def matches(self, model: str) -> list[dict]: + def match_routing(self, model: str) -> Optional[str]: if not model: - return [] - if self._compiled is None: - self._compiled = self._compile() - return [dict(model_info) for pattern, model_info in self._compiled if pattern.search(model) is not None] + return None + return next( + (rule.provider for rule in self.routing_rules if rule.pattern.search(model) is not None), + None, + ) - def match(self, model: str) -> Optional[dict]: - return next(iter(self.matches(model)), None) + def match_capabilities(self, model: str) -> Optional[dict]: + if not model: + return None + matched = tuple(rule.model_info for rule in self.capability_rules if rule.pattern.search(model) is not None) + if not matched: + return None + return {key: value for model_info in matched for key, value in model_info.items()} _registry = _FallbackGeneralizations() -def set_fallback_generalizations(rules: Optional[list[dict]]) -> None: - """Install the active rule list and invalidate the compiled-regex cache. +def set_fallback_generalizations(rules: Optional[list]) -> None: + """Install the active rule list, compiling and classifying each rule. - ``extends`` inheritance is resolved here, once, before the rules are stored. + Malformed, invalid-regex, and mixed-kind rules are warned about and skipped here. Called once when the model cost map is loaded (and again on any reload). """ - _registry.set_rules(_resolve_extends(rules) if isinstance(rules, list) else rules) + _registry.set_rules(rules) -def get_fallback_generalization_rules() -> list[dict]: +def get_fallback_generalization_rules() -> list: """Return the raw rule list (read-only view for callers/tests).""" return _registry.rules -def match_fallback_generalization(model: str) -> Optional[dict]: - """Return the ``model_info`` of the first rule whose regex matches ``model``. +def match_routing_generalization(model: str) -> Optional[str]: + """Return the provider of the first routing rule whose regex matches ``model``. O(number of rules). Only call this once exact lookups have missed. """ - return _registry.match(model) + return _registry.match_routing(model) -def match_all_fallback_generalizations(model: str) -> list[dict]: - """Return the ``model_info`` of every rule whose regex matches ``model``, in rule order. +def match_capability_generalizations(model: str) -> Optional[dict]: + """Return the union of the ``model_info`` of every capability rule matching ``model``. - Lets a caller with extra constraints (e.g. a provider match) skip an - inapplicable earlier rule instead of discarding the whole candidate. + Later rules override earlier ones on key conflicts. Returns ``None`` when no + capability rule matches. O(number of rules); only call once exact lookups have missed. """ - return _registry.matches(model) + return _registry.match_capabilities(model) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index fa8ffe1ba1d..487a7b7e25f 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -4,7 +4,7 @@ from urllib.parse import urlparse import litellm from litellm.constants import REPLICATE_MODEL_NAME_WITH_ID_LENGTH from litellm.litellm_core_utils.fallback_generalizations import ( - match_fallback_generalization, + match_routing_generalization, ) from litellm.llms.openai_like.json_loader import JSONProviderRegistry from litellm.secret_managers.main import get_secret, get_secret_str @@ -474,12 +474,10 @@ def get_llm_provider( custom_llm_provider = "sap" # Last resort for an otherwise-unknown model: a declarative - # fallback-generalization rule (e.g. routes future claude-* to anthropic). + # fallback-generalization routing rule (e.g. routes future claude-* to anthropic). # Exact provider matches above always win; this only runs on a miss. if not custom_llm_provider: - generalization = match_fallback_generalization(model) - if generalization is not None: - custom_llm_provider = generalization.get("litellm_provider") or None + custom_llm_provider = match_routing_generalization(model) if not custom_llm_provider: if litellm.suppress_debug_info is False: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6ecadaf1fe6..c483b3288e2 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -45050,31 +45050,26 @@ "fallback_generalizations": { "rules": [ { - "name": "bedrock-anthropic-claude-mid-conversation-system", - "pattern": "anthropic\\.claude-[a-z]+-(?:4[-._](?:[89]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", - "description": "Bedrock Invoke ids for Claude 4.8 or higher: anthropic.claude- with minor 4.8 through 4.99, any 5.x or later major-minor, or a bare 5+ major, which also admits new families such as fable. These models accept mid-conversation role system messages in place (verified live on Opus 4.8, Sonnet 5 and Fable 5), so unmapped future Bedrock Claudes keep the cache-preserving in-place handling instead of the hoist-all default. Listed first so bare-id provider inference, which takes the first pattern hit, resolves these Bedrock ids to bedrock; model-info resolution skips provider-mismatched rules either way.", - "extends": "anthropic-claude", + "name": "bedrock-claude-ids", + "pattern": "anthropic\\.claude-", + "description": "Any Bedrock-syntax Claude id: the dotted anthropic.claude- segment appears in bare (anthropic.claude-...), region-prefixed (us./eu./au./jp./apac.) and global.-prefixed ids, for every version. Routes these to bedrock before the bare-id Anthropic rule is consulted.", "model_info": { - "litellm_provider": "bedrock", - "supports_adaptive_thinking": true, - "supports_mid_conversation_system": true + "litellm_provider": "bedrock" } }, { - "name": "anthropic-claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new families with no code change.", - "extends": "anthropic-claude", - "model_info": { - "supports_adaptive_thinking": true - } - }, - { - "name": "anthropic-claude", + "name": "anthropic-claude-ids", "pattern": "^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$", - "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Version-neutral fallback that gives an unmapped Claude provider routing and baseline capabilities; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", + "description": "A bare Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", + "model_info": { + "litellm_provider": "anthropic" + } + }, + { + "name": "claude-family-baseline", + "pattern": "claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?", + "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...). Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", "model_info": { - "litellm_provider": "anthropic", "mode": "chat", "max_input_tokens": 200000, "max_output_tokens": 64000, @@ -45090,6 +45085,22 @@ "supports_pdf_input": true, "supports_system_messages": true } + }, + { + "name": "claude-adaptive-thinking", + "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", + "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions with no code change.", + "model_info": { + "supports_adaptive_thinking": true + } + }, + { + "name": "claude-mid-conversation-system", + "pattern": "claude-[a-z]+-(?:4[-._](?:[89]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", + "description": "Claude at version 4.8 or higher, in any id shape that contains claude--: minors 4.8 through 4.99, any later major-minor, and bare 5+ majors so a new family like claude-fable-5 matches. Anthropic introduced mid-conversation system messages with Opus 4.8 and every newer Claude keeps them; 4.7 and below reject the system role inside messages.", + "model_info": { + "supports_mid_conversation_system": true + } } ] } diff --git a/litellm/utils.py b/litellm/utils.py index 731be992af0..b25841e5389 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -61,7 +61,7 @@ from litellm._lazy_imports import ( ) from litellm._uuid import uuid from litellm.litellm_core_utils.fallback_generalizations import ( - match_all_fallback_generalizations, + match_capability_generalizations, ) from litellm.constants import ( DEFAULT_CHAT_COMPLETION_PARAM_VALUES, @@ -5043,25 +5043,28 @@ def _get_model_info_from_generalization( potential_model_names: PotentialModelNamesAndCustomLLMProvider, custom_llm_provider: Optional[str], ) -> Optional[tuple[str, dict]]: - """Resolve an unmapped model via a declarative fallback-generalization rule. + """Resolve an unmapped model via the declarative capability generalization rules. Tries the same name candidates as the exact lookups, in the same order, and - returns ``(matched_name, model_info)`` for the first matching rule that also - satisfies the provider constraint; a rule scoped to another provider is - skipped in favor of later rules rather than discarding the candidate. - O(number of rules); only call after the exact lookups have missed. + returns ``(matched_name, model_info)`` for the first candidate matched by at + least one capability rule, with ``litellm_provider`` backfilled from the + provider the caller requested. O(number of rules); only call after the exact + lookups have missed. """ - candidates = [ + candidates = ( potential_model_names["combined_model_name"], model, potential_model_names["split_model"], potential_model_names["combined_stripped_model_name"], potential_model_names["stripped_model_name"], - ] + ) for candidate in candidates: - for generalized_info in match_all_fallback_generalizations(candidate): - if _check_provider_match(model_info=generalized_info, custom_llm_provider=custom_llm_provider): - return candidate, generalized_info + generalized_info = match_capability_generalizations(candidate) + if generalized_info is None: + continue + if custom_llm_provider is None: + return candidate, generalized_info + return candidate, {**generalized_info, "litellm_provider": custom_llm_provider} return None diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 31a6146cbd6..ee0b1193990 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -45283,31 +45283,26 @@ "fallback_generalizations": { "rules": [ { - "name": "bedrock-anthropic-claude-mid-conversation-system", - "pattern": "anthropic\\.claude-[a-z]+-(?:4[-._](?:[89]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", - "description": "Bedrock Invoke ids for Claude 4.8 or higher: anthropic.claude- with minor 4.8 through 4.99, any 5.x or later major-minor, or a bare 5+ major, which also admits new families such as fable. These models accept mid-conversation role system messages in place (verified live on Opus 4.8, Sonnet 5 and Fable 5), so unmapped future Bedrock Claudes keep the cache-preserving in-place handling instead of the hoist-all default. Listed first so bare-id provider inference, which takes the first pattern hit, resolves these Bedrock ids to bedrock; model-info resolution skips provider-mismatched rules either way.", - "extends": "anthropic-claude", + "name": "bedrock-claude-ids", + "pattern": "anthropic\\.claude-", + "description": "Any Bedrock-syntax Claude id: the dotted anthropic.claude- segment appears in bare (anthropic.claude-...), region-prefixed (us./eu./au./jp./apac.) and global.-prefixed ids, for every version. Routes these to bedrock before the bare-id Anthropic rule is consulted.", "model_info": { - "litellm_provider": "bedrock", - "supports_adaptive_thinking": true, - "supports_mid_conversation_system": true + "litellm_provider": "bedrock" } }, { - "name": "anthropic-claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new families with no code change.", - "extends": "anthropic-claude", - "model_info": { - "supports_adaptive_thinking": true - } - }, - { - "name": "anthropic-claude", + "name": "anthropic-claude-ids", "pattern": "^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$", - "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Version-neutral fallback that gives an unmapped Claude provider routing and baseline capabilities; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", + "description": "A bare Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", + "model_info": { + "litellm_provider": "anthropic" + } + }, + { + "name": "claude-family-baseline", + "pattern": "claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?", + "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...). Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", "model_info": { - "litellm_provider": "anthropic", "mode": "chat", "max_input_tokens": 200000, "max_output_tokens": 64000, @@ -45323,6 +45318,22 @@ "supports_pdf_input": true, "supports_system_messages": true } + }, + { + "name": "claude-adaptive-thinking", + "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", + "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions with no code change.", + "model_info": { + "supports_adaptive_thinking": true + } + }, + { + "name": "claude-mid-conversation-system", + "pattern": "claude-[a-z]+-(?:4[-._](?:[89]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", + "description": "Claude at version 4.8 or higher, in any id shape that contains claude--: minors 4.8 through 4.99, any later major-minor, and bare 5+ majors so a new family like claude-fable-5 matches. Anthropic introduced mid-conversation system messages with Opus 4.8 and every newer Claude keeps them; 4.7 and below reject the system role inside messages.", + "model_info": { + "supports_mid_conversation_system": true + } } ] } diff --git a/tests/local_testing/test_get_llm_provider.py b/tests/local_testing/test_get_llm_provider.py index 9ecb639fc3a..4c3e13da17a 100644 --- a/tests/local_testing/test_get_llm_provider.py +++ b/tests/local_testing/test_get_llm_provider.py @@ -509,10 +509,10 @@ def shipped_generalizations(): class TestClaudeModelPatternMatching: """ - The ``anthropic-claude`` fallback generalization rule routes future Claude - models to the Anthropic provider without requiring a + The ``anthropic-claude-ids`` fallback generalization routing rule routes future + Claude models to the Anthropic provider without requiring a model_prices_and_context_window.json entry. These tests exercise the rule - end-to-end through ``get_llm_provider`` and ``match_fallback_generalization``. + end-to-end through ``get_llm_provider`` and ``match_routing_generalization``. """ @pytest.mark.parametrize( @@ -556,10 +556,10 @@ class TestClaudeModelPatternMatching: self, model, shipped_generalizations ): from litellm.litellm_core_utils.fallback_generalizations import ( - match_fallback_generalization, + match_routing_generalization, ) - assert match_fallback_generalization(model) is None + assert match_routing_generalization(model) is None def test_routing_comes_from_the_rule_not_python(self, shipped_generalizations): """With the rule cleared, an unknown claude must no longer route to diff --git a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py index cf74bed9c15..b911a46443c 100644 --- a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py +++ b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py @@ -1,11 +1,13 @@ """ Tests for the declarative fallback-generalizations mechanism. -Covers both the pure module (litellm.litellm_core_utils.fallback_generalizations) -and its end-to-end wiring into provider routing (get_llm_provider) and model-info -resolution (get_model_info / supports_*). +Covers the pure module (litellm.litellm_core_utils.fallback_generalizations): the +routing/capability rule split, install-time validation, capability unioning; and +its end-to-end wiring into provider routing (get_llm_provider) and model-info +resolution (get_model_info) including the shipped rules in the bundled cost map. """ +import logging import os import sys @@ -14,10 +16,11 @@ import pytest sys.path.insert(0, os.path.abspath("../../..")) import litellm +from litellm._logging import verbose_logger from litellm.litellm_core_utils.fallback_generalizations import ( get_fallback_generalization_rules, - match_all_fallback_generalizations, - match_fallback_generalization, + match_capability_generalizations, + match_routing_generalization, set_fallback_generalizations, ) @@ -32,92 +35,143 @@ def restore_generalizations(): set_fallback_generalizations(previous) +class _RecordingHandler(logging.Handler): + def __init__(self): + super().__init__(level=logging.WARNING) + self.messages = [] + + def emit(self, record): + self.messages.append(record.getMessage()) + + +@pytest.fixture +def warning_messages(): + handler = _RecordingHandler() + previous_level = verbose_logger.level + verbose_logger.setLevel(logging.WARNING) + verbose_logger.addHandler(handler) + try: + yield handler.messages + finally: + verbose_logger.removeHandler(handler) + verbose_logger.setLevel(previous_level) + + # --------------------------------------------------------------------------- # -# Pure module behaviour +# Engine: routing rules # --------------------------------------------------------------------------- # -def test_match_returns_model_info_of_first_matching_rule(restore_generalizations): +def test_routing_inference_first_match_wins(restore_generalizations): + restore_generalizations( + [ + {"name": "first", "pattern": r"^acme-", "model_info": {"litellm_provider": "openai"}}, + {"name": "second", "pattern": r"^acme-pro-", "model_info": {"litellm_provider": "anthropic"}}, + ] + ) + assert match_routing_generalization("acme-pro-1") == "openai" + assert match_routing_generalization("gpt-4o") is None + assert match_routing_generalization("") is None + + +def test_capability_rules_do_not_route(restore_generalizations): + restore_generalizations([{"name": "caps", "pattern": r"^acme-", "model_info": {"supports_vision": True}}]) + assert match_routing_generalization("acme-pro-1") is None + + +def test_routing_match_is_case_insensitive(restore_generalizations): + restore_generalizations( + [{"name": "r", "pattern": r"^claude-opus", "model_info": {"litellm_provider": "anthropic"}}] + ) + assert match_routing_generalization("CLAUDE-OPUS-9-9") == "anthropic" + + +# --------------------------------------------------------------------------- # +# Engine: capability rules +# --------------------------------------------------------------------------- # + + +def test_capability_union_is_last_wins_in_file_order(restore_generalizations): restore_generalizations( [ { - "name": "first", + "name": "broad", "pattern": r"^acme-", - "model_info": {"litellm_provider": "openai", "tag": "first"}, + "model_info": {"mode": "chat", "supports_vision": True, "max_input_tokens": 1000}, }, { - "name": "second", + "name": "narrow", "pattern": r"^acme-pro-", - "model_info": {"litellm_provider": "anthropic", "tag": "second"}, + "model_info": {"supports_vision": False, "supports_reasoning": True}, }, ] ) - # Both rules match "acme-pro-1"; first-in-list wins (documented precedence). - matched = match_fallback_generalization("acme-pro-1") - assert matched is not None - assert matched["tag"] == "first" + assert match_capability_generalizations("acme-pro-1") == { + "mode": "chat", + "supports_vision": False, + "max_input_tokens": 1000, + "supports_reasoning": True, + } + assert match_capability_generalizations("acme-basic-1") == { + "mode": "chat", + "supports_vision": True, + "max_input_tokens": 1000, + } -def test_match_all_returns_every_matching_rule_in_order(restore_generalizations): +def test_routing_rules_are_excluded_from_capability_results(restore_generalizations): restore_generalizations( [ - { - "name": "first", - "pattern": r"^acme-", - "model_info": {"litellm_provider": "openai", "tag": "first"}, - }, - { - "name": "second", - "pattern": r"^acme-pro-", - "model_info": {"litellm_provider": "anthropic", "tag": "second"}, - }, + {"name": "route", "pattern": r"^acme-", "model_info": {"litellm_provider": "openai"}}, + {"name": "caps", "pattern": r"^acme-pro-", "model_info": {"supports_vision": True}}, ] ) - assert [m["tag"] for m in match_all_fallback_generalizations("acme-pro-1")] == ["first", "second"] - assert match_all_fallback_generalizations("gpt-4o") == [] + assert match_capability_generalizations("acme-pro-1") == {"supports_vision": True} + assert match_capability_generalizations("acme-basic-1") is None -def test_provider_scoped_rule_is_skipped_for_other_providers(restore_generalizations): - """Model-info resolution must fall through a provider-mismatched earlier rule to a - later applicable one, instead of discarding the model name at the first pattern hit.""" - restore_generalizations( - [ - { - "name": "bedrock-scoped", - "pattern": r"^acme-", - "model_info": {"litellm_provider": "bedrock", "supports_vision": False}, - }, - { - "name": "openai-scoped", - "pattern": r"^acme-", - "model_info": {"litellm_provider": "openai", "mode": "chat", "supports_vision": True}, - }, - ] - ) - litellm.get_model_info.cache_clear() - info = litellm.get_model_info("acme-fast-1", custom_llm_provider="openai") - assert info["litellm_provider"] == "openai" - assert info["supports_vision"] is True - - -def test_match_is_case_insensitive(restore_generalizations): - restore_generalizations( - [{"name": "r", "pattern": r"^claude-opus", "model_info": {"ok": True}}] - ) - assert match_fallback_generalization("CLAUDE-OPUS-9-9") == {"ok": True} - - -def test_no_match_returns_none(restore_generalizations): - restore_generalizations( - [{"name": "r", "pattern": r"^claude-", "model_info": {"ok": True}}] - ) - assert match_fallback_generalization("gpt-4o") is None - assert match_fallback_generalization("") is None - - -def test_empty_rules_match_nothing(restore_generalizations): +def test_no_capability_match_returns_none(restore_generalizations): + restore_generalizations([{"name": "r", "pattern": r"^claude-", "model_info": {"ok": True}}]) + assert match_capability_generalizations("gpt-4o") is None + assert match_capability_generalizations("") is None restore_generalizations([]) - assert match_fallback_generalization("claude-opus-9-9") is None + assert match_capability_generalizations("claude-opus-9-9") is None + + +def test_reinstalling_rules_replaces_compiled_rules(restore_generalizations): + restore_generalizations([{"name": "r", "pattern": r"^aaa", "model_info": {"v": 1}}]) + assert match_capability_generalizations("aaa-1") == {"v": 1} + set_fallback_generalizations([{"name": "r", "pattern": r"^bbb", "model_info": {"v": 2}}]) + assert match_capability_generalizations("aaa-1") is None + assert match_capability_generalizations("bbb-1") == {"v": 2} + + +# --------------------------------------------------------------------------- # +# Engine: install-time validation +# --------------------------------------------------------------------------- # + + +def test_mixed_kind_rule_warns_and_is_skipped(restore_generalizations, warning_messages): + restore_generalizations( + [ + { + "name": "mixed-kind", + "pattern": r"^acme-", + "model_info": {"litellm_provider": "anthropic", "supports_vision": True}, + }, + {"name": "good-caps", "pattern": r"^acme-pro-", "model_info": {"supports_reasoning": True}}, + ] + ) + assert any("mixed-kind" in message and "litellm_provider" in message for message in warning_messages) + assert match_routing_generalization("acme-pro-1") is None + assert match_capability_generalizations("acme-pro-1") == {"supports_reasoning": True} + + +def test_non_string_provider_rule_warns_and_is_skipped(restore_generalizations, warning_messages): + restore_generalizations([{"name": "bad-provider", "pattern": r"^acme-", "model_info": {"litellm_provider": 42}}]) + assert any("bad-provider" in message for message in warning_messages) + assert match_routing_generalization("acme-1") is None + assert match_capability_generalizations("acme-1") is None def test_malformed_rules_are_skipped_not_fatal(restore_generalizations): @@ -131,69 +185,7 @@ def test_malformed_rules_are_skipped_not_fatal(restore_generalizations): {"name": "good", "pattern": r"^claude-", "model_info": {"good": True}}, ] ) - # Non-dict entries and dicts with bad fields are all skipped; the one - # valid rule still matches. - assert match_fallback_generalization("claude-opus-9-9") == {"good": True} - - -def test_setting_rules_invalidates_compiled_cache(restore_generalizations): - restore_generalizations([{"name": "r", "pattern": r"^aaa", "model_info": {"v": 1}}]) - assert match_fallback_generalization("aaa-1") == {"v": 1} - # Re-install different rules; the compiled cache must be rebuilt. - set_fallback_generalizations( - [{"name": "r", "pattern": r"^bbb", "model_info": {"v": 2}}] - ) - assert match_fallback_generalization("aaa-1") is None - assert match_fallback_generalization("bbb-1") == {"v": 2} - - -def test_extends_inherits_parent_and_own_overrides(restore_generalizations): - """A rule's ``extends`` pulls in the parent's model_info; its own keys win on conflict, - so a narrow rule carries only its delta instead of duplicating the parent.""" - restore_generalizations( - [ - { - "name": "base", - "pattern": r"^base-only$", - "model_info": { - "litellm_provider": "anthropic", - "input_cost_per_token": 5e-06, - "supports_vision": True, - }, - }, - { - "name": "child", - "pattern": r"^kid-", - "extends": "base", - "model_info": { - "supports_adaptive_thinking": True, - "supports_vision": False, - }, - }, - ] - ) - matched = match_fallback_generalization("kid-1") - assert matched == { - "litellm_provider": "anthropic", - "input_cost_per_token": 5e-06, - "supports_vision": False, - "supports_adaptive_thinking": True, - } - - -def test_extends_with_unknown_parent_keeps_own_model_info(restore_generalizations): - """A dangling ``extends`` is non-fatal: the rule resolves to its own model_info.""" - restore_generalizations( - [ - { - "name": "orphan", - "pattern": r"^orphan-", - "extends": "does-not-exist", - "model_info": {"litellm_provider": "openai"}, - } - ] - ) - assert match_fallback_generalization("orphan-1") == {"litellm_provider": "openai"} + assert match_capability_generalizations("claude-opus-9-9") == {"good": True} # --------------------------------------------------------------------------- # @@ -201,91 +193,61 @@ def test_extends_with_unknown_parent_keeps_own_model_info(restore_generalization # --------------------------------------------------------------------------- # -@pytest.fixture -def myco_rule(restore_generalizations): - """A self-contained rule carrying provider, pricing, context and capabilities.""" +def test_unknown_model_routes_via_routing_rule(restore_generalizations): + restore_generalizations([{"name": "myco", "pattern": r"^myco-", "model_info": {"litellm_provider": "openai"}}]) + _, provider, _, _ = litellm.get_llm_provider(model="myco-fast-1") + assert provider == "openai" + + +def test_capability_info_backfills_requested_provider(restore_generalizations): restore_generalizations( [ { - "name": "myco", - "pattern": r"^myco-[a-z]+-\d+$", + "name": "beeco-caps", + "pattern": r"^beeco-[a-z]+-\d+$", "model_info": { - "litellm_provider": "openai", "mode": "chat", - "input_cost_per_token": 1e-06, - "output_cost_per_token": 2e-06, "max_input_tokens": 12345, - "max_output_tokens": 678, "supports_vision": True, "supports_function_calling": True, }, } ] ) - return "myco-fast-1" - - -def test_unknown_model_routes_via_rule(myco_rule): - _, provider, _, _ = litellm.get_llm_provider(model=myco_rule) - assert provider == "openai" - - -def test_unknown_model_gets_pricing_context_and_capabilities(myco_rule): - info = litellm.get_model_info(myco_rule) - assert info["litellm_provider"] == "openai" - assert info["input_cost_per_token"] == 1e-06 - assert info["output_cost_per_token"] == 2e-06 + litellm.get_model_info.cache_clear() + info = litellm.get_model_info("beeco-fast-1", custom_llm_provider="groq") + assert info["litellm_provider"] == "groq" assert info["max_input_tokens"] == 12345 assert info["supports_vision"] is True + other = litellm.get_model_info("beeco-fast-1", custom_llm_provider="openai") + assert other["litellm_provider"] == "openai" -def test_supports_helper_reads_through_generalization(myco_rule): - assert litellm.supports_vision(myco_rule) is True - assert litellm.supports_function_calling(myco_rule) is True +def test_routing_only_match_does_not_resolve_model_info(restore_generalizations): + restore_generalizations([{"name": "route", "pattern": r"^ceeco-", "model_info": {"litellm_provider": "openai"}}]) + litellm.get_model_info.cache_clear() + with pytest.raises(Exception): + litellm.get_model_info("ceeco-fast-1", custom_llm_provider="openai") def test_exact_entry_takes_precedence_over_rule(restore_generalizations): - """An exact cost-map entry must win over a rule that also matches it.""" restore_generalizations( - [ - { - "name": "shadow-gpt4o", - "pattern": r"^gpt-4o$", - "model_info": { - "litellm_provider": "anthropic", - "input_cost_per_token": 999.0, - }, - } - ] + [{"name": "shadow-gpt4o", "pattern": r"^gpt-4o$", "model_info": {"input_cost_per_token": 999.0}}] ) + litellm.get_model_info.cache_clear() info = litellm.get_model_info("gpt-4o") - # Resolved from the real exact entry, not the shadowing rule. assert info["litellm_provider"] == "openai" assert info["input_cost_per_token"] != 999.0 -def test_unknown_model_without_matching_rule_still_unmapped(restore_generalizations): - restore_generalizations( - [ - { - "name": "claude", - "pattern": r"^claude-", - "model_info": {"litellm_provider": "anthropic"}, - } - ] - ) - with pytest.raises(Exception): - litellm.get_model_info("totally-unknown-model-xyz") - - # --------------------------------------------------------------------------- # -# Shipped anthropic-claude rule +# Shipped rules (bundled cost map) # --------------------------------------------------------------------------- # @pytest.fixture def shipped_cost_map(monkeypatch): - """Activate the bundled cost map so the shipped anthropic-claude rule is installed.""" + """Activate the bundled cost map so the shipped rules are installed.""" original_cost = litellm.model_cost previous_rules = list(get_fallback_generalization_rules()) monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") @@ -299,40 +261,115 @@ def shipped_cost_map(monkeypatch): set_fallback_generalizations(previous_rules) -def test_shipped_rule_marks_unmapped_high_version_claude_adaptive_without_pricing( - shipped_cost_map, -): - """An unmapped Claude >= 4.6 resolves via the version-gated adaptive-thinking rule, which - inherits routing and capabilities from the base rule and adds ``supports_adaptive_thinking``. - The rule carries no pricing, so cost stays unpriced (zero, not a fabricated number) rather - than reporting a confidently-wrong price.""" - model = "claude-opus-9-9" +def test_shipped_bare_claude_id_routes_to_anthropic(shipped_cost_map): + _, provider, _, _ = litellm.get_llm_provider(model="claude-haiku-4-6") + assert provider == "anthropic" + + +def test_shipped_bedrock_syntax_claude_id_routes_to_bedrock(shipped_cost_map): + """Regression: a bedrock-syntax id must infer bedrock even when its version also + matches an unanchored Anthropic capability pattern. The old first-match-wins engine + routed global.anthropic.claude-haiku-4-6 to anthropic via the adaptive rule.""" + for model in [ + "global.anthropic.claude-haiku-4-6", + "us.anthropic.claude-haiku-4-6", + "anthropic.claude-haiku-4-6", + "eu.anthropic.claude-opus-5-0", + ]: + assert model not in litellm.model_cost + _, provider, _, _ = litellm.get_llm_provider(model=model) + assert provider == "bedrock", model + + +def test_shipped_rules_resolve_unmapped_bedrock_claude_with_bedrock_provider(shipped_cost_map): + model = "us.anthropic.claude-haiku-4-6" assert model not in litellm.model_cost - info = litellm.get_model_info(model) - assert info["litellm_provider"] == "anthropic" + info = litellm.get_model_info(model, custom_llm_provider="bedrock") + assert info["litellm_provider"] == "bedrock" assert info["supports_adaptive_thinking"] is True assert info["supports_function_calling"] is True + assert info["max_input_tokens"] == 200000 + assert info.get("supports_mid_conversation_system") is None assert not info.get("input_cost_per_token") assert not info.get("output_cost_per_token") -def test_shipped_rule_resolves_unmapped_low_version_claude_without_adaptive(shipped_cost_map): - """An unmapped Claude < 4.6 falls through to the version-neutral anthropic-claude rule: it - gets provider routing and baseline capabilities but no ``supports_adaptive_thinking`` flag, - so a sub-4.6 alias such as ``claude-opus-4-0`` resolves yet is never marked adaptive.""" - model = "claude-opus-4-0" +def test_shipped_rules_stack_adaptive_and_mid_conversation_flags(shipped_cost_map): + model = "claude-opus-4-9" assert model not in litellm.model_cost - info = litellm.get_model_info(model) + info = litellm.get_model_info(model, custom_llm_provider="anthropic") assert info["litellm_provider"] == "anthropic" + assert info["supports_adaptive_thinking"] is True + assert info["supports_mid_conversation_system"] is True + assert info["supports_function_calling"] is True + + +@pytest.mark.parametrize( + "model,provider", + [ + ("claude-opus-4-9@20260101", "vertex_ai"), + ("databricks-claude-opus-5-1", "databricks"), + ], +) +def test_shipped_rules_are_provider_neutral_for_unmapped_ids(shipped_cost_map, model, provider): + assert model not in litellm.model_cost + info = litellm.get_model_info(model, custom_llm_provider=provider) + assert info["litellm_provider"] == provider + assert info["supports_adaptive_thinking"] is True + assert info["supports_mid_conversation_system"] is True + assert info["supports_function_calling"] is True + + +@pytest.mark.parametrize( + "model,provider,adaptive,mid_conversation", + [ + ("us.anthropic.claude-opus-4-5", "bedrock", None, None), + ("claude-haiku-4-6", "anthropic", True, None), + ("claude-haiku-4-7", "anthropic", True, None), + ("claude-haiku-4-8", "anthropic", True, True), + ("claude-haiku-4-9", "anthropic", True, True), + ("claude-haiku-4-10", "anthropic", True, True), + ("claude-haiku-5-0", "anthropic", True, True), + ("claude-sonnet-5-1", "anthropic", True, True), + ], +) +def test_shipped_version_boundaries(shipped_cost_map, model, provider, adaptive, mid_conversation): + assert model not in litellm.model_cost + info = litellm.get_model_info(model, custom_llm_provider=provider) + assert info["litellm_provider"] == provider assert info["supports_function_calling"] is True - assert info.get("supports_adaptive_thinking") is None assert not info.get("input_cost_per_token") + assert info.get("supports_adaptive_thinking") is adaptive, model + assert info.get("supports_mid_conversation_system") is mid_conversation, model + + +def test_shipped_mid_conversation_rule_covers_new_families_like_fable(shipped_cost_map): + """The 4.8+ mid-conversation gate accepts bare 5+ majors, so a new family shaped + like claude-fable-5 is covered. The adaptive rule keeps its opus/sonnet/haiku + family list, so a fable-shaped id gets no adaptive flag from the rules.""" + model = "claude-fable-5-1" + assert model not in litellm.model_cost + info = litellm.get_model_info(model, custom_llm_provider="anthropic") + assert info["supports_mid_conversation_system"] is True + assert info.get("supports_adaptive_thinking") is None + assert info["supports_function_calling"] is True + + +def test_shipped_exact_entry_beats_rules(shipped_cost_map): + model = "us.anthropic.claude-sonnet-4-6" + assert model in litellm.model_cost + info = litellm.get_model_info(model, custom_llm_provider="bedrock") + assert info["litellm_provider"] == "bedrock_converse" + assert info["input_cost_per_token"] == 3.3e-06 + assert info["max_input_tokens"] == 1000000 + assert info["supports_adaptive_thinking"] is True + assert info.get("supports_mid_conversation_system") is None def test_shipped_adaptive_rule_gates_on_version_not_pricing(shipped_cost_map): - """The version-gated ``anthropic-claude-adaptive-thinking`` rule marks an unmapped - Claude adaptive only from >= 4.6, including provider-prefixed ids the anchored pricing - rule cannot match, while leaving < 4.6 (and the dated Opus 4.0 form) non-adaptive.""" + """The version-gated adaptive-thinking capability rule marks an unmapped Claude + adaptive only from >= 4.6, including provider-prefixed ids the anchored routing + rule cannot match, while leaving the dated Opus 4.0 form non-adaptive.""" from litellm.llms.anthropic.common_utils import AnthropicModelInfo adaptive = "us.anthropic.claude-opus-4-9" @@ -343,12 +380,10 @@ def test_shipped_adaptive_rule_gates_on_version_not_pricing(shipped_cost_map): assert AnthropicModelInfo._is_adaptive_thinking_model(non_adaptive) is False -def test_shipped_bedrock_rule_resolves_unmapped_future_claude_for_bedrock(shipped_cost_map): - """An unmapped Bedrock Claude >= 4.8 resolves via the bedrock-scoped - ``bedrock-anthropic-claude-mid-conversation-system`` rule even when the lookup - carries ``custom_llm_provider="bedrock"``, which the provider check uses to drop - the anthropic-scoped rules. It inherits base capabilities, gains both - version-gated flags, and stays unpriced.""" +def test_shipped_rules_resolve_unmapped_future_bedrock_claude_with_both_flags(shipped_cost_map): + """An unmapped Bedrock Claude >= 4.8 resolves for custom_llm_provider="bedrock" with + baseline capabilities, both version-gated flags, the bedrock provider backfilled, and + no fabricated pricing.""" model = "us.anthropic.claude-opus-4-9" assert model not in litellm.model_cost info = litellm.get_model_info(model, custom_llm_provider="bedrock") @@ -359,11 +394,11 @@ def test_shipped_bedrock_rule_resolves_unmapped_future_claude_for_bedrock(shippe assert not info.get("input_cost_per_token") -def test_shipped_bedrock_mid_conversation_rule_gates_on_version_and_naming(shipped_cost_map): - """The bedrock rule only claims Bedrock-style ids at 4.8+, bare 5+ majors and - new families included; pre-4.8 Bedrock ids and native ids never gain the flag, - and the rule outranks the anthropic-scoped ones for Bedrock ids because it is - listed first.""" +def test_shipped_mid_conversation_gate_on_bedrock_ids(shipped_cost_map): + """Bedrock-syntax ids gain ``supports_mid_conversation_system`` only from 4.8 upward, + bare 5+ majors and new families included; 4.7-and-below Bedrock ids never gain it. + The flag comes from the provider-neutral capability rule rather than a bedrock-scoped + one, so the same gate covers native and vertex-shaped ids too.""" for flagged in ( "us.anthropic.claude-opus-4-8", "jp.anthropic.claude-opus-4-8", @@ -371,17 +406,14 @@ def test_shipped_bedrock_mid_conversation_rule_gates_on_version_and_naming(shipp "us.anthropic.claude-fable-5", "anthropic.claude-sonnet-5-20260101-v1:0", ): - matched = match_fallback_generalization(flagged) + matched = match_capability_generalizations(flagged) assert matched is not None, flagged - assert matched["litellm_provider"] == "bedrock", flagged assert matched["supports_mid_conversation_system"] is True, flagged for unflagged in ( "us.anthropic.claude-opus-4-7", "us.anthropic.claude-sonnet-4-6", "us.anthropic.claude-haiku-4-5-20251001-v1:0", "anthropic.claude-3-5-sonnet-20240620-v1:0", - "claude-opus-4-9", - "claude-sonnet-5", ): - matched = match_fallback_generalization(unflagged) + matched = match_capability_generalizations(unflagged) assert matched is None or not matched.get("supports_mid_conversation_system"), unflagged diff --git a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py index 924aaa7775d..bdd71f28b1a 100644 --- a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py +++ b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py @@ -13,7 +13,8 @@ sys.path.insert(0, os.path.abspath("../../..")) from litellm.litellm_core_utils.fallback_generalizations import ( get_fallback_generalization_rules, - match_fallback_generalization, + match_capability_generalizations, + match_routing_generalization, set_fallback_generalizations, ) from litellm.litellm_core_utils.get_model_cost_map import ( @@ -102,9 +103,7 @@ def test_finalize_pops_key_and_installs_rules(): # The reserved key is removed from the returned model map ... assert FALLBACK_GENERALIZATIONS_KEY not in finalized # ... and its rules are installed into the generalizations module. - assert match_fallback_generalization("widget-9") == { - "litellm_provider": "openai" - } + assert match_routing_generalization("widget-9") == "openai" finally: set_fallback_generalizations(previous) @@ -116,27 +115,25 @@ def test_finalize_with_no_block_clears_rules(): [{"name": "stale", "pattern": r"^x", "model_info": {"a": 1}}] ) _finalize_model_cost_map(_make_models(2)) - assert match_fallback_generalization("x-1") is None + assert match_capability_generalizations("x-1") is None finally: set_fallback_generalizations(previous) -def test_shipped_backup_carries_the_anthropic_claude_rule(): - """The bundled backup must ship the anthropic-claude rule so a fresh install - (or an offline fallback) routes unknown Claude models without code changes.""" +def test_shipped_backup_carries_the_claude_routing_rules(): + """The bundled backup must ship the Claude routing rules so a fresh install + (or an offline fallback) routes unknown Claude models without code changes. + Bedrock-syntax ids must hit the bedrock rule before the bare-id Anthropic rule.""" backup = GetModelCostMap.load_local_model_cost_map() rules = backup.get(FALLBACK_GENERALIZATIONS_KEY, {}).get("rules", []) - names = {r.get("name") for r in rules} - assert "anthropic-claude" in names - - rule = next(r for r in rules if r.get("name") == "anthropic-claude") - assert rule["model_info"]["litellm_provider"] == "anthropic" + names = [r.get("name") for r in rules] + assert names.index("bedrock-claude-ids") < names.index("anthropic-claude-ids") previous = list(get_fallback_generalization_rules()) try: set_fallback_generalizations(rules) - matched = match_fallback_generalization("claude-opus-4-9") - assert matched is not None and matched["litellm_provider"] == "anthropic" + assert match_routing_generalization("claude-opus-4-9") == "anthropic" + assert match_routing_generalization("global.anthropic.claude-opus-4-9") == "bedrock" finally: set_fallback_generalizations(previous) @@ -147,23 +144,19 @@ def test_shipped_backup_marks_claude_4_6_plus_adaptive_not_4_0(): route) and on the version-gated anthropic-claude-adaptive-thinking rule for unmapped future Claudes, while leaving the dated Claude 4.0 names ("...-4-20250514") unflagged so a date can never be mistaken for a 4.6+ minor - version. The version-neutral anthropic-claude pricing rule must not flag it, so - an unmapped sub-4.6 name is priced but stays non-adaptive. The adaptive rule must - inherit pricing from the pricing rule via ``extends`` and carry only its delta, so - the Opus-tier price block is never duplicated across rules.""" + version. The version-neutral claude-family-baseline capability rule must not flag + it, so an unmapped sub-4.6 name resolves but stays non-adaptive. The adaptive rule + carries only its delta; capability unioning stacks it onto the baseline, so the + baseline block is never duplicated across rules and no rule needs ``extends``.""" backup = GetModelCostMap.load_local_model_cost_map() rules = backup[FALLBACK_GENERALIZATIONS_KEY]["rules"] - pricing_rule = next(r for r in rules if r.get("name") == "anthropic-claude") - adaptive_rule = next( - r for r in rules if r.get("name") == "anthropic-claude-adaptive-thinking" - ) - assert "supports_adaptive_thinking" not in pricing_rule["model_info"] - assert adaptive_rule["model_info"]["supports_adaptive_thinking"] is True - - assert "extends" not in pricing_rule - assert adaptive_rule.get("extends") == "anthropic-claude" + baseline_rule = next(r for r in rules if r.get("name") == "claude-family-baseline") + adaptive_rule = next(r for r in rules if r.get("name") == "claude-adaptive-thinking") + assert "supports_adaptive_thinking" not in baseline_rule["model_info"] + assert "litellm_provider" not in baseline_rule["model_info"] assert adaptive_rule["model_info"] == {"supports_adaptive_thinking": True} + assert all("extends" not in r for r in rules) for adaptive in [ "anthropic.claude-opus-4-8", diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 2c44ab5a049..545f5ac0e94 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -1,9 +1,9 @@ { "LIT001": { - "limit": 23452 + "limit": 23439 }, "LIT002": { - "limit": 27522 + "limit": 27517 }, "LIT003": { "limit": 292 From 1ccc3382d9c6c0233845949c196239cc192a5014 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 10 Jul 2026 21:00:33 -0700 Subject: [PATCH 2/5] feat(fallback-generalizations): widen adaptive-thinking gate to any claude family at major 5+ --- ...odel_prices_and_context_window_backup.json | 4 +-- model_prices_and_context_window.json | 4 +-- .../test_fallback_generalizations.py | 29 ++++++++++++++++--- .../anthropic/test_anthropic_common_utils.py | 23 ++++++++------- 4 files changed, 42 insertions(+), 18 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c483b3288e2..868ba5c43e3 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -45088,8 +45088,8 @@ }, { "name": "claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions with no code change.", + "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))|claude-[a-z]+-(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?", + "description": "Claude opus/sonnet/haiku at version 4.6 or higher (4.6 through 4.99, then any later major-minor), plus any claude-- id at major version 5 or higher, bare major or major-minor, so a new family shaped like claude-fable-5 is covered by the same gate as the core families. 4.x coverage stays limited to opus/sonnet/haiku. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", "model_info": { "supports_adaptive_thinking": true } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ee0b1193990..37ce3d270c8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -45321,8 +45321,8 @@ }, { "name": "claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions with no code change.", + "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))|claude-[a-z]+-(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?", + "description": "Claude opus/sonnet/haiku at version 4.6 or higher (4.6 through 4.99, then any later major-minor), plus any claude-- id at major version 5 or higher, bare major or major-minor, so a new family shaped like claude-fable-5 is covered by the same gate as the core families. 4.x coverage stays limited to opus/sonnet/haiku. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", "model_info": { "supports_adaptive_thinking": true } diff --git a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py index b911a46443c..e4407b547cc 100644 --- a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py +++ b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py @@ -343,14 +343,35 @@ def test_shipped_version_boundaries(shipped_cost_map, model, provider, adaptive, assert info.get("supports_mid_conversation_system") is mid_conversation, model -def test_shipped_mid_conversation_rule_covers_new_families_like_fable(shipped_cost_map): - """The 4.8+ mid-conversation gate accepts bare 5+ majors, so a new family shaped - like claude-fable-5 is covered. The adaptive rule keeps its opus/sonnet/haiku - family list, so a fable-shaped id gets no adaptive flag from the rules.""" +def test_shipped_rules_cover_new_families_like_fable_at_5_plus(shipped_cost_map): + """Both version gates accept any claude-- id at major 5 or higher, bare + major or major-minor, so a new family shaped like claude-fable-5 gets adaptive + thinking and mid-conversation system support without a cost-map entry.""" model = "claude-fable-5-1" assert model not in litellm.model_cost info = litellm.get_model_info(model, custom_llm_provider="anthropic") assert info["supports_mid_conversation_system"] is True + assert info["supports_adaptive_thinking"] is True + assert info["supports_function_calling"] is True + + +def test_shipped_rules_flag_bare_5_plus_majors_of_any_family(shipped_cost_map): + """A bare 5+ major with no minor gets both flags at the rule level; the mapped + claude-fable-5 entry itself still resolves from the cost map, so this pins the + pattern via the capability union rather than get_model_info.""" + matched = match_capability_generalizations("claude-fable-5") + assert matched is not None + assert matched["supports_adaptive_thinking"] is True + assert matched["supports_mid_conversation_system"] is True + + +def test_shipped_adaptive_rule_keeps_4x_coverage_to_core_families(shipped_cost_map): + """The family-agnostic adaptive branch starts at major 5; a non-core family at + 4.x gets the 4.8+ mid-conversation flag and baseline caps but never adaptive.""" + model = "claude-newfam-4-9" + assert model not in litellm.model_cost + info = litellm.get_model_info(model, custom_llm_provider="anthropic") + assert info["supports_mid_conversation_system"] is True assert info.get("supports_adaptive_thinking") is None assert info["supports_function_calling"] is True diff --git a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py index f7a234c8ae8..316f0d6ddf7 100644 --- a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py +++ b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py @@ -1649,17 +1649,18 @@ class TestClaudeOpus48AdaptiveThinking: @pytest.mark.parametrize( "model", [ - "us.anthropic.claude-fable-5-preview", - "claude-fable-5-preview", + "us.anthropic.claude-fable-preview", + "claude-fable-preview", ], ) def test_unmapped_aliases_without_parseable_version_stay_non_adaptive( self, local_model_cost_map, model ): """An alias absent from the map, not matched by any ``fallback_generalizations`` - rule, and without a parseable opus/sonnet/haiku >= 4.6 family version stays - non-adaptive. ``fable`` is outside the version-rule family set, so neither the - cost map nor the declarative rule marks it adaptive.""" + rule, and without any parseable family version stays non-adaptive. ``fable`` + without a major version matches neither the core-family 4.6+ gate nor the + family-agnostic 5+ gate, so neither the cost map nor the declarative rule marks + it adaptive.""" import litellm from litellm.llms.anthropic.common_utils import AnthropicModelInfo @@ -1676,16 +1677,18 @@ class TestClaudeOpus48AdaptiveThinking: "claude-opus-5-0", "claude-opus-4-10", "claude-opus-4-8-some-future-suffix", + "claude-fable-5-preview", + "us.anthropic.claude-fable-5-preview", ], ) def test_adaptive_thinking_version_fallback_for_unmapped_high_versions( self, local_model_cost_map, model ): - """Provider-prefixed or suffixed Claude names that resolve to no mapped entry and - are not matched by the anchored ``anthropic-claude`` pricing rule still resolve to - adaptive when their opus/sonnet/haiku family version is >= 4.6. The version gate is - the declarative ``anthropic-claude-adaptive-thinking`` rule, so 5.x, 6.x and any - later family are covered with no code change.""" + """Provider-prefixed or suffixed Claude names that resolve to no mapped entry + still resolve to adaptive when their opus/sonnet/haiku family version is >= 4.6 + or any family's major version is >= 5. The version gate is the declarative + ``claude-adaptive-thinking`` rule, so 5.x, 6.x and any later family are covered + with no code change.""" import litellm from litellm.llms.anthropic.common_utils import AnthropicModelInfo From 1329036ca1cc8e3ec07274d02ff80af2cffa07b2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 11 Jul 2026 00:00:41 -0700 Subject: [PATCH 3/5] fix(fallback-generalizations): tolerate legacy remote rule schema and keep register_model cache-pricing inheritance --- basedpyright-code-budget.json | 8 +- .../fallback_generalizations.py | 86 ++++++++++++++----- litellm/utils.py | 27 +++++- .../test_fallback_generalizations.py | 76 ++++++++++++++-- type-discipline-budget.json | 4 +- 5 files changed, 164 insertions(+), 37 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 8c5bf165c7e..c29adbd3044 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5900 }, "reportMissingTypeArgument": { - "limit": 15913 + "limit": 15908 }, "reportMissingTypeStubs": { "limit": 41 @@ -108,10 +108,10 @@ "limit": 40539 }, "reportUnknownParameterType": { - "limit": 20413 + "limit": 20408 }, "reportUnknownVariableType": { - "limit": 32144 + "limit": 32143 }, "reportUnnecessaryCast": { "limit": 177 @@ -123,7 +123,7 @@ "limit": 7 }, "reportUnnecessaryIsInstance": { - "limit": 1211 + "limit": 1210 }, "reportUntypedBaseClass": { "limit": 165 diff --git a/litellm/litellm_core_utils/fallback_generalizations.py b/litellm/litellm_core_utils/fallback_generalizations.py index c9b6d83e9f1..410bb9623fe 100644 --- a/litellm/litellm_core_utils/fallback_generalizations.py +++ b/litellm/litellm_core_utils/fallback_generalizations.py @@ -19,9 +19,18 @@ ones on key conflicts, and the caller backfills ``litellm_provider`` with the provider it requested. If no capability rule matches, model-info resolution misses as if no rules existed. -A rule that mixes ``litellm_provider`` with other ``model_info`` keys is invalid: -``set_fallback_generalizations`` logs a warning and skips it at install time (a -warning rather than a crash, because released proxies fetch this JSON remotely). +LEGACY-SCHEMA SHIM (temporary, until the new-schema JSON reaches main): released +proxies fetch this JSON remotely from main, whose block still ships the old schema +where a rule mixes ``litellm_provider`` with capability keys and may inherit a +parent's ``model_info`` via ``extends``. Such a legacy rule is tolerated rather +than skipped: ``extends`` is resolved once at install time (single level, against +raw parents), and the resolved rule acts as BOTH kinds, a routing rule (its +``litellm_provider`` participates in first-hit inference) and a capability rule +(its full ``model_info``, provider included, participates in the union). New-schema +rules never mix the two and never use ``extends``. A rule whose +``litellm_provider`` is not a string is invalid and is warned about and skipped +(a warning rather than a crash, for the same remote-fetch reason). + Rules are only consulted after exact and case-insensitive lookups miss, so an exact cost-map entry always takes precedence over any rule. @@ -47,6 +56,36 @@ NAME_FIELD = "name" PATTERN_FIELD = "pattern" MODEL_INFO_FIELD = "model_info" PROVIDER_KEY = "litellm_provider" +LEGACY_EXTENDS_FIELD = "extends" + + +def _resolve_legacy_extends(rules: list) -> list: + """Expand legacy ``extends`` inheritance so each rule's ``model_info`` is self-contained. + + Compatibility shim for the old remote schema: single level, resolved against each + parent's raw ``model_info``, with the child's own keys winning on conflict. Non-dict + rules and dangling parents pass through unchanged; new-schema rules carry no + ``extends`` and are untouched. + """ + base_by_name = { + rule[NAME_FIELD]: rule[MODEL_INFO_FIELD] + for rule in rules + if isinstance(rule, dict) + and isinstance(rule.get(NAME_FIELD), str) + and isinstance(rule.get(MODEL_INFO_FIELD), dict) + } + + def resolved(rule: object) -> object: + if not isinstance(rule, dict): + return rule + parent_name = rule.get(LEGACY_EXTENDS_FIELD) + own_info = rule.get(MODEL_INFO_FIELD) + parent_info = base_by_name.get(parent_name) if isinstance(parent_name, str) else None + if parent_info is None or not isinstance(own_info, dict): + return rule + return {**rule, MODEL_INFO_FIELD: {**parent_info, **own_info}} + + return [resolved(rule) for rule in rules] @dataclass(frozen=True, slots=True) @@ -64,9 +103,9 @@ class _CapabilityRule: _CompiledRule = Union[_RoutingRule, _CapabilityRule] -def _compile_rule(rule: object) -> Optional[_CompiledRule]: +def _compile_rule(rule: object) -> tuple[_CompiledRule, ...]: if not isinstance(rule, dict): - return None + return () pattern = rule.get(PATTERN_FIELD) model_info = rule.get(MODEL_INFO_FIELD) if not isinstance(pattern, str) or not isinstance(model_info, dict): @@ -76,7 +115,7 @@ def _compile_rule(rule: object) -> Optional[_CompiledRule]: PATTERN_FIELD, MODEL_INFO_FIELD, ) - return None + return () try: compiled = re.compile(pattern, re.IGNORECASE) except re.error as e: @@ -85,21 +124,24 @@ def _compile_rule(rule: object) -> Optional[_CompiledRule]: pattern, e, ) - return None + return () if PROVIDER_KEY not in model_info: - return _CapabilityRule(pattern=compiled, model_info=model_info) + return (_CapabilityRule(pattern=compiled, model_info=model_info),) provider = model_info[PROVIDER_KEY] - if len(model_info) == 1 and isinstance(provider, str): - return _RoutingRule(pattern=compiled, provider=provider) - verbose_logger.warning( - "LiteLLM: skipping invalid fallback generalization rule %s: a routing rule's '%s' must contain " - "'%s' as its only key (a string), and a capability rule must not contain '%s' at all.", - rule.get(NAME_FIELD, pattern), - MODEL_INFO_FIELD, - PROVIDER_KEY, - PROVIDER_KEY, + if not isinstance(provider, str): + verbose_logger.warning( + "LiteLLM: skipping invalid fallback generalization rule %s: '%s' in '%s' must be a string.", + rule.get(NAME_FIELD, pattern), + PROVIDER_KEY, + MODEL_INFO_FIELD, + ) + return () + if len(model_info) == 1: + return (_RoutingRule(pattern=compiled, provider=provider),) + return ( + _RoutingRule(pattern=compiled, provider=provider), + _CapabilityRule(pattern=compiled, model_info=model_info), ) - return None class _FallbackGeneralizations: @@ -112,7 +154,7 @@ class _FallbackGeneralizations: def set_rules(self, rules: Optional[list]) -> None: installed = rules if isinstance(rules, list) else [] - compiled = tuple(c for c in (_compile_rule(rule) for rule in installed) if c is not None) + compiled = tuple(kind for rule in _resolve_legacy_extends(installed) for kind in _compile_rule(rule)) self.rules = installed self.routing_rules = tuple(rule for rule in compiled if isinstance(rule, _RoutingRule)) self.capability_rules = tuple(rule for rule in compiled if isinstance(rule, _CapabilityRule)) @@ -140,8 +182,10 @@ _registry = _FallbackGeneralizations() def set_fallback_generalizations(rules: Optional[list]) -> None: """Install the active rule list, compiling and classifying each rule. - Malformed, invalid-regex, and mixed-kind rules are warned about and skipped here. - Called once when the model cost map is loaded (and again on any reload). + Legacy ``extends`` inheritance is resolved here, once, before classification; + a legacy rule mixing ``litellm_provider`` with capability keys installs as both + kinds. Malformed and invalid-regex rules are warned about and skipped. Called + once when the model cost map is loaded (and again on any reload). """ _registry.set_rules(rules) diff --git a/litellm/utils.py b/litellm/utils.py index b25841e5389..907834dc5a4 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2651,6 +2651,26 @@ def _resolve_builtin_model_cost_entry(key: str, provider: str) -> Optional[Dict[ return None +def _get_builtin_model_info_for_registration(model: str) -> Optional[ModelInfo]: + """Resolve ``model`` to its built-in cost-map entry for registration merging. + + Returns ``None`` when the lookup raises or when it resolved via a + fallback-generalization capability rule, detected as the resolved key missing + ``litellm.model_cost`` while matching a capability rule. A rule-derived entry + carries no pricing, so treating it as a hit would skip the built-in + cache-pricing inheritance for prefix-mangled keys. + """ + try: + info = get_model_info(model=model) + except Exception: + return None + if info["key"] in litellm.model_cost: + return info + if match_capability_generalizations(info["key"]) is None: + return info + return None + + def register_model(model_cost: Union[str, dict]): """ Register new / Override existing models (and their pricing) to specific providers. @@ -2691,10 +2711,11 @@ def register_model(model_cost: Union[str, dict]): existing_model = litellm.model_cost.get(key, {}) model_cost_key = key else: - try: - existing_model = cast(dict, get_model_info(model=key)) + builtin_model_info = _get_builtin_model_info_for_registration(model=_key_str) + if builtin_model_info is not None: + existing_model = cast(dict, builtin_model_info) model_cost_key = existing_model["key"] - except Exception: + else: existing_model = {} model_cost_key = key builtin_entry = _resolve_builtin_model_cost_entry(key=_key_str, provider=provider) diff --git a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py index e4407b547cc..652f99ee402 100644 --- a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py +++ b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py @@ -147,24 +147,86 @@ def test_reinstalling_rules_replaces_compiled_rules(restore_generalizations): # --------------------------------------------------------------------------- # -# Engine: install-time validation +# Engine: install-time validation and legacy-schema shim # --------------------------------------------------------------------------- # -def test_mixed_kind_rule_warns_and_is_skipped(restore_generalizations, warning_messages): +def test_legacy_mixed_rule_acts_as_both_kinds(restore_generalizations): + """A legacy rule mixing ``litellm_provider`` with capability keys routes AND + contributes its full model_info (provider included) to the capability union.""" restore_generalizations( [ { - "name": "mixed-kind", + "name": "legacy-mixed", "pattern": r"^acme-", "model_info": {"litellm_provider": "anthropic", "supports_vision": True}, }, - {"name": "good-caps", "pattern": r"^acme-pro-", "model_info": {"supports_reasoning": True}}, + {"name": "new-caps", "pattern": r"^acme-pro-", "model_info": {"supports_reasoning": True}}, ] ) - assert any("mixed-kind" in message and "litellm_provider" in message for message in warning_messages) - assert match_routing_generalization("acme-pro-1") is None - assert match_capability_generalizations("acme-pro-1") == {"supports_reasoning": True} + assert match_routing_generalization("acme-pro-1") == "anthropic" + assert match_capability_generalizations("acme-pro-1") == { + "litellm_provider": "anthropic", + "supports_vision": True, + "supports_reasoning": True, + } + + +LEGACY_MAIN_RULES = [ + { + "name": "anthropic-claude-adaptive-thinking", + "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))", + "description": "Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new families with no code change.", + "extends": "anthropic-claude", + "model_info": {"supports_adaptive_thinking": True}, + }, + { + "name": "anthropic-claude", + "pattern": "^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$", + "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Version-neutral fallback that gives an unmapped Claude provider routing and baseline capabilities; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", + "model_info": { + "litellm_provider": "anthropic", + "mode": "chat", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "supports_function_calling": True, + "supports_parallel_function_calling": True, + "supports_vision": True, + "supports_tool_choice": True, + "supports_assistant_prefill": True, + "supports_prompt_caching": True, + "supports_response_schema": True, + "supports_reasoning": True, + "supports_pdf_input": True, + "supports_system_messages": True, + }, + }, +] + + +def test_legacy_main_schema_keeps_unmapped_claude_working(restore_generalizations): + """Pins the remote-map transition window: a released proxy running this engine + against main's old-schema block (mixed provider+capability rule plus ``extends``, + copied verbatim above) must keep unmapped-Claude inference and info resolution + working until the new-schema JSON reaches main.""" + restore_generalizations([dict(rule) for rule in LEGACY_MAIN_RULES]) + litellm.get_model_info.cache_clear() + + _, provider, _, _ = litellm.get_llm_provider(model="claude-opus-9-9") + assert provider == "anthropic" + + info = litellm.get_model_info("claude-opus-9-9") + assert info["litellm_provider"] == "anthropic" + assert info["supports_adaptive_thinking"] is True + assert info["supports_function_calling"] is True + assert info["max_input_tokens"] == 200000 + assert not info.get("input_cost_per_token") + + low = litellm.get_model_info("claude-opus-4-0") + assert low["litellm_provider"] == "anthropic" + assert low["supports_function_calling"] is True + assert low.get("supports_adaptive_thinking") is None def test_non_string_provider_rule_warns_and_is_skipped(restore_generalizations, warning_messages): diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 545f5ac0e94..6bc301baa0e 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -1,9 +1,9 @@ { "LIT001": { - "limit": 23439 + "limit": 23424 }, "LIT002": { - "limit": 27517 + "limit": 27514 }, "LIT003": { "limit": 292 From 81e639398ad4801170c0e08886548bb0b4257558 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 11 Jul 2026 00:49:05 -0700 Subject: [PATCH 4/5] fix(fallback-generalizations): let exact cost-map entries beat capability rules across lookup-candidate ladders --- basedpyright-code-budget.json | 8 +++--- litellm/utils.py | 11 ++++++-- .../test_fallback_generalizations.py | 25 +++++++++++++++++++ .../test_anthropic_claude3_transformation.py | 14 +++++------ type-discipline-budget.json | 4 +-- 5 files changed, 47 insertions(+), 15 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index c29adbd3044..edfb3536ad3 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5900 }, "reportMissingTypeArgument": { - "limit": 15908 + "limit": 15903 }, "reportMissingTypeStubs": { "limit": 41 @@ -108,10 +108,10 @@ "limit": 40539 }, "reportUnknownParameterType": { - "limit": 20408 + "limit": 20403 }, "reportUnknownVariableType": { - "limit": 32143 + "limit": 32141 }, "reportUnnecessaryCast": { "limit": 177 @@ -123,7 +123,7 @@ "limit": 7 }, "reportUnnecessaryIsInstance": { - "limit": 1210 + "limit": 1209 }, "reportUntypedBaseClass": { "limit": 165 diff --git a/litellm/utils.py b/litellm/utils.py index 907834dc5a4..18b89ee0d13 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5069,8 +5069,13 @@ def _get_model_info_from_generalization( Tries the same name candidates as the exact lookups, in the same order, and returns ``(matched_name, model_info)`` for the first candidate matched by at least one capability rule, with ``litellm_provider`` backfilled from the - provider the caller requested. O(number of rules); only call after the exact - lookups have missed. + provider the caller requested. Rules lose to exact entries: if ANY candidate is + an exact ``litellm.model_cost`` key (necessarily provider-mismatched, or the + exact lookups would have returned it), the model is known rather than unmapped, + and resolving it from rules would hand an unpriced rule-derived entry to + callers whose fallback ladder (e.g. the cost calculator's model-name variants) + still had a priced exact name to try. O(number of rules); only call after the + exact lookups have missed. """ candidates = ( potential_model_names["combined_model_name"], @@ -5079,6 +5084,8 @@ def _get_model_info_from_generalization( potential_model_names["combined_stripped_model_name"], potential_model_names["stripped_model_name"], ) + if any(_get_model_cost_key(candidate) is not None for candidate in candidates): + return None for candidate in candidates: generalized_info = match_capability_generalizations(candidate) if generalized_info is None: diff --git a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py index 652f99ee402..69020fceba1 100644 --- a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py +++ b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py @@ -449,6 +449,31 @@ def test_shipped_exact_entry_beats_rules(shipped_cost_map): assert info.get("supports_mid_conversation_system") is None +def test_shipped_rules_lose_to_exact_entries_across_cost_ladder_variants(shipped_cost_map): + """A route-mangled variant of an exactly-mapped model must never resolve from + rules. The cost calculator tries model-name variants in order; a rule-derived + unpriced entry served for an early variant (here bedrock/claude-haiku-4-5-20251001, + whose bare form is exactly mapped under anthropic) would zero out the bill even + though the exact priced bedrock entry is one variant later. An exactly-mapped id + under a mismatched provider raises instead of resolving from rules.""" + from litellm import completion_cost + from litellm.types.utils import ModelResponse, Usage + + assert "claude-haiku-4-5-20251001" in litellm.model_cost + with pytest.raises(Exception): + litellm.get_model_info("claude-haiku-4-5-20251001", custom_llm_provider="bedrock") + + entry = litellm.model_cost["us.anthropic.claude-haiku-4-5-20251001-v1:0"] + response = ModelResponse(model="claude-haiku-4-5-20251001", usage=Usage(prompt_tokens=100, completion_tokens=50)) + cost = completion_cost( + completion_response=response, + model="bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0", + custom_llm_provider="bedrock", + ) + assert cost == 100 * entry["input_cost_per_token"] + 50 * entry["output_cost_per_token"] + assert cost > 0 + + def test_shipped_adaptive_rule_gates_on_version_not_pricing(shipped_cost_map): """The version-gated adaptive-thinking capability rule marks an unmapped Claude adaptive only from >= 4.6, including provider-prefixed ids the anchored routing diff --git a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py index 7d104ff1f2b..52f136a1fd9 100644 --- a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py @@ -2130,9 +2130,9 @@ def test_bedrock_invoke_transform_hoists_all_system_for_unmapped_model(local_mod def test_bedrock_invoke_transform_keeps_system_in_place_for_unmapped_future_claude(local_model_cost_map): """An unmapped Bedrock Claude at 4.8 or higher resolves through the - ``bedrock-anthropic-claude-mid-conversation-system`` fallback rule, so a - future model that has not landed in the cost map yet keeps the - cache-preserving in-place behavior instead of falling back to hoist-all.""" + ``claude-mid-conversation-system`` capability rule, so a future model that + has not landed in the cost map yet keeps the cache-preserving in-place + behavior instead of falling back to hoist-all.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -2159,9 +2159,9 @@ def test_bedrock_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_ """Exact cost-map hits resolve before fallback-generalization rules, so a mapped Bedrock Claude 4.8+ entry without ``supports_mid_conversation_system`` silently loses the cache-preserving in-place handling that the - ``bedrock-anthropic-claude-mid-conversation-system`` rule grants unmapped - ids. Every mapped entry the rule's own pattern matches must carry the flag - explicitly.""" + ``claude-mid-conversation-system`` capability rule grants unmapped ids. + Every mapped bedrock entry the rule's own pattern matches must carry the + flag explicitly.""" import re import litellm @@ -2171,7 +2171,7 @@ def test_bedrock_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_ cost_map = json.load(f) rules = cost_map["fallback_generalizations"]["rules"] pattern = re.compile( - next(r["pattern"] for r in rules if r["name"] == "bedrock-anthropic-claude-mid-conversation-system"), + next(r["pattern"] for r in rules if r["name"] == "claude-mid-conversation-system"), re.IGNORECASE, ) missing = [ diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 6bc301baa0e..87b4c96e323 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -1,9 +1,9 @@ { "LIT001": { - "limit": 23424 + "limit": 23409 }, "LIT002": { - "limit": 27514 + "limit": 27511 }, "LIT003": { "limit": 292 From 6fa088224bc2022c7541ee44cf02c0bd6dd2942e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 11 Jul 2026 10:42:47 -0700 Subject: [PATCH 5/5] fix(fallback-generalizations): cover bare Claude majors in baseline and routing, require claude- prefix in adaptive gate --- ...odel_prices_and_context_window_backup.json | 12 ++--- model_prices_and_context_window.json | 12 ++--- .../test_fallback_generalizations.py | 50 ++++++++++++++++--- .../anthropic/test_anthropic_common_utils.py | 4 +- 4 files changed, 57 insertions(+), 21 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 868ba5c43e3..b4ae842c4e6 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -45059,16 +45059,16 @@ }, { "name": "anthropic-claude-ids", - "pattern": "^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$", - "description": "A bare Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", + "pattern": "^claude-[a-z]+-\\d+(?:[-.]\\d+)?(?:-\\d{8})?$", + "description": "A bare Claude family-major id with an optional minor and an optional 8-digit date suffix, anchored to the whole name, so claude-newfamily-5 routes like claude-newfamily-5-1 does. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", "model_info": { "litellm_provider": "anthropic" } }, { "name": "claude-family-baseline", - "pattern": "claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?", - "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...). Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", + "pattern": "claude-[a-z]+-\\d+(?:[-.]\\d+)?(?:-\\d{8})?", + "description": "Any Claude family-major id with an optional minor and an optional 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...), so bare majors like claude-newfamily-5 get the same baseline as claude-newfamily-5-1. Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", "model_info": { "mode": "chat", "max_input_tokens": 200000, @@ -45088,8 +45088,8 @@ }, { "name": "claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))|claude-[a-z]+-(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher (4.6 through 4.99, then any later major-minor), plus any claude-- id at major version 5 or higher, bare major or major-minor, so a new family shaped like claude-fable-5 is covered by the same gate as the core families. 4.x coverage stays limited to opus/sonnet/haiku. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", + "pattern": "claude-[a-z]+-(?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", + "description": "Claude at version 4.6 or higher, in any id shape that contains claude--: minors 4.6 through 4.99, any later major-minor, and bare 5+ majors so a new family shaped like claude-fable-5 matches. Requiring the claude- prefix keeps non-Claude names such as team-sonnet-5-1 out. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", "model_info": { "supports_adaptive_thinking": true } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 37ce3d270c8..94d9f6496bd 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -45292,16 +45292,16 @@ }, { "name": "anthropic-claude-ids", - "pattern": "^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$", - "description": "A bare Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", + "pattern": "^claude-[a-z]+-\\d+(?:[-.]\\d+)?(?:-\\d{8})?$", + "description": "A bare Claude family-major id with an optional minor and an optional 8-digit date suffix, anchored to the whole name, so claude-newfamily-5 routes like claude-newfamily-5-1 does. Routes an unmapped Claude id that carries no provider namespace to the Anthropic API.", "model_info": { "litellm_provider": "anthropic" } }, { "name": "claude-family-baseline", - "pattern": "claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?", - "description": "Any Claude family-major-minor id, optionally with an 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...). Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", + "pattern": "claude-[a-z]+-\\d+(?:[-.]\\d+)?(?:-\\d{8})?", + "description": "Any Claude family-major id with an optional minor and an optional 8-digit date suffix, under any provider namespace (bare, bedrock-dotted, vertex, databricks, ...), so bare majors like claude-newfamily-5 get the same baseline as claude-newfamily-5-1. Carries the model-family facts every Claude shares; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.", "model_info": { "mode": "chat", "max_input_tokens": 200000, @@ -45321,8 +45321,8 @@ }, { "name": "claude-adaptive-thinking", - "pattern": "(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))|claude-[a-z]+-(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?", - "description": "Claude opus/sonnet/haiku at version 4.6 or higher (4.6 through 4.99, then any later major-minor), plus any claude-- id at major version 5 or higher, bare major or major-minor, so a new family shaped like claude-fable-5 is covered by the same gate as the core families. 4.x coverage stays limited to opus/sonnet/haiku. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", + "pattern": "claude-[a-z]+-(?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d)(?!\\d)(?:[-._]\\d{1,2}(?!\\d))?)", + "description": "Claude at version 4.6 or higher, in any id shape that contains claude--: minors 4.6 through 4.99, any later major-minor, and bare 5+ majors so a new family shaped like claude-fable-5 matches. Requiring the claude- prefix keeps non-Claude names such as team-sonnet-5-1 out. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new versions and new families with no code change.", "model_info": { "supports_adaptive_thinking": true } diff --git a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py index 69020fceba1..0ea09463c16 100644 --- a/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py +++ b/tests/test_litellm/litellm_core_utils/test_fallback_generalizations.py @@ -427,15 +427,51 @@ def test_shipped_rules_flag_bare_5_plus_majors_of_any_family(shipped_cost_map): assert matched["supports_mid_conversation_system"] is True -def test_shipped_adaptive_rule_keeps_4x_coverage_to_core_families(shipped_cost_map): - """The family-agnostic adaptive branch starts at major 5; a non-core family at - 4.x gets the 4.8+ mid-conversation flag and baseline caps but never adaptive.""" - model = "claude-newfam-4-9" +def test_shipped_version_gates_are_family_agnostic_at_4x(shipped_cost_map): + """Both version gates apply to any claude-- id, 4.x included: a non-core + family at 4.9 gets adaptive and mid-conversation, while the same family at 4.5 + gets baseline only. Only opus/sonnet/haiku ever shipped 4.x ids, so the + family-agnostic 4.6+ gate changes nothing for real models.""" + high = litellm.get_model_info("claude-newfam-4-9", custom_llm_provider="anthropic") + assert high["supports_adaptive_thinking"] is True + assert high["supports_mid_conversation_system"] is True + assert high["supports_function_calling"] is True + + low = litellm.get_model_info("claude-newfam-4-5", custom_llm_provider="anthropic") + assert low.get("supports_adaptive_thinking") is None + assert low.get("supports_mid_conversation_system") is None + assert low["supports_function_calling"] is True + + +def test_shipped_rules_give_bare_majors_the_full_baseline_union(shipped_cost_map): + """A bare-major unmapped id (no minor) resolves the same baseline union as its + major-minor sibling: the baseline pattern's minor is optional, so claude-newt-5 + is not left with version flags but no mode, token limits, or capability facts.""" + model = "anthropic/claude-newt-5" assert model not in litellm.model_cost - info = litellm.get_model_info(model, custom_llm_provider="anthropic") - assert info["supports_mid_conversation_system"] is True - assert info.get("supports_adaptive_thinking") is None + info = litellm.get_model_info(model) + assert info["litellm_provider"] == "anthropic" + assert info["mode"] == "chat" + assert info["max_tokens"] == 64000 assert info["supports_function_calling"] is True + assert info["supports_adaptive_thinking"] is True + assert info["supports_mid_conversation_system"] is True + + +def test_shipped_routing_rule_covers_bare_majors(shipped_cost_map): + _, provider, _, _ = litellm.get_llm_provider(model="claude-newt-5") + assert provider == "anthropic" + + +def test_shipped_adaptive_rule_requires_claude_prefix(shipped_cost_map): + """A non-Claude name embedding a core-family 4.6+/5.x version substring must not + resolve from the rules; serving it a zero-priced rule entry would silently + swallow cost tracking for arbitrary custom deployment names.""" + model = "openai/team-sonnet-5-1-alias" + assert model not in litellm.model_cost + assert match_capability_generalizations("team-sonnet-5-1-alias") is None + with pytest.raises(Exception): + litellm.get_model_info(model) def test_shipped_exact_entry_beats_rules(shipped_cost_map): diff --git a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py index 316f0d6ddf7..4bd270bb7fb 100644 --- a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py +++ b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py @@ -1685,8 +1685,8 @@ class TestClaudeOpus48AdaptiveThinking: self, local_model_cost_map, model ): """Provider-prefixed or suffixed Claude names that resolve to no mapped entry - still resolve to adaptive when their opus/sonnet/haiku family version is >= 4.6 - or any family's major version is >= 5. The version gate is the declarative + still resolve to adaptive when the id carries claude-- at version 4.6 + or higher, bare 5+ majors included. The version gate is the declarative ``claude-adaptive-thinking`` rule, so 5.x, 6.x and any later family are covered with no code change.""" import litellm