fix(cost): mirror batch long-context keys on custom pricing params

Register the two *_above_272k_tokens_batches keys on CustomPricingLiteLLMParams so a per-deployment override stays out of the shared backend key, add them to the inline model-info schema and alias-count tests, and build LiteLLM_Params and GenericLiteLLMParams through model_validate at the two dict-splat call sites so basedpyright's reportArgumentType budget ratchets down instead of blocking the new fields.
This commit is contained in:
mateo-berri 2026-09-04 21:40:16 -07:00
parent b7f7d5c42f
commit 2c182f4b93
7 changed files with 17 additions and 9 deletions

View file

@ -3,7 +3,7 @@
"limit": 14074
},
"reportArgumentType": {
"limit": 2206
"limit": 1921
},
"reportAssignmentType": {
"limit": 319

View file

@ -62,7 +62,7 @@ class AzurePassthroughConfig(BasePassthroughConfig):
) -> dict:
return BaseAzureLLM._base_validate_azure_environment(
headers=headers,
litellm_params=GenericLiteLLMParams(**{**litellm_params, "api_key": api_key}),
litellm_params=GenericLiteLLMParams.model_validate({**litellm_params, "api_key": api_key}),
)
@staticmethod

View file

@ -8643,12 +8643,8 @@ class Router:
if ptu_error is not None and is_ptu_cost_attribution_enabled():
raise ValueError(ptu_error)
zeroed_pricing: Final = zeroed_ptu_pricing(_model_info, _litellm_params) if config_sourced else None
litellm_params: Final[LiteLLM_Params] = LiteLLM_Params(
**(
_litellm_params
if zeroed_pricing is None
else MappingProxyType({**_litellm_params, **zeroed_pricing})
)
litellm_params: Final[LiteLLM_Params] = LiteLLM_Params.model_validate(
_litellm_params if zeroed_pricing is None else MappingProxyType({**_litellm_params, **zeroed_pricing})
)
warn_on_provider_credential_mismatch(model_name=_model_name, litellm_params=_litellm_params)
deployment = Deployment(

View file

@ -3481,6 +3481,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
input_cost_per_token_above_200k_tokens_priority: float | None = None
input_cost_per_token_above_272k_tokens_priority: float | None = None
input_cost_per_token_above_272k_tokens_flex: float | None = None
input_cost_per_token_above_272k_tokens_batches: float | None = None
input_cost_per_query: float | None = None
input_cost_per_image: float | None = None
input_cost_per_image_above_128k_tokens: float | None = None
@ -3501,6 +3502,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
output_cost_per_token_above_200k_tokens_priority: float | None = None
output_cost_per_token_above_272k_tokens_priority: float | None = None
output_cost_per_token_above_272k_tokens_flex: float | None = None
output_cost_per_token_above_272k_tokens_batches: float | None = None
output_cost_per_character_above_128k_tokens: float | None = None
output_cost_per_image: float | None = None
output_cost_per_image_token: float | None = None

View file

@ -1776,7 +1776,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
sol = litellm.model_cost["gpt-5.6-sol"]
cost_fields = sorted(field for field in sol if "cost" in field)
assert len(cost_fields) == 27
assert len(cost_fields) == 29
for field in cost_fields:
assert alias.get(field) == sol.get(field), field

View file

@ -959,12 +959,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"input_cost_per_token_priority": {"type": "number"},
"input_cost_per_token_above_200k_tokens_priority": {"type": "number"},
"input_cost_per_token_above_272k_tokens_priority": {"type": "number"},
"input_cost_per_token_above_272k_tokens_batches": {"type": "number"},
"input_cost_per_token_above_272k_tokens_flex": {"type": "number"},
"input_cost_per_audio_token_priority": {"type": "number"},
"output_cost_per_token_flex": {"type": "number"},
"output_cost_per_token_priority": {"type": "number"},
"output_cost_per_token_above_200k_tokens_priority": {"type": "number"},
"output_cost_per_token_above_272k_tokens_priority": {"type": "number"},
"output_cost_per_token_above_272k_tokens_batches": {"type": "number"},
"output_cost_per_token_above_272k_tokens_flex": {"type": "number"},
"regional_endpoint_uplift_multiplier": {"type": "number"},
"regional_processing_uplift_multiplier_eu": {"type": "number"},

View file

@ -29355,6 +29355,8 @@ export interface components {
input_cost_per_token_above_200k_tokens_priority?: number | null;
/** Input Cost Per Token Above 272K Tokens */
input_cost_per_token_above_272k_tokens?: number | null;
/** Input Cost Per Token Above 272K Tokens Batches */
input_cost_per_token_above_272k_tokens_batches?: number | null;
/** Input Cost Per Token Above 272K Tokens Flex */
input_cost_per_token_above_272k_tokens_flex?: number | null;
/** Input Cost Per Token Above 272K Tokens Priority */
@ -29460,6 +29462,8 @@ export interface components {
output_cost_per_token_above_200k_tokens_priority?: number | null;
/** Output Cost Per Token Above 272K Tokens */
output_cost_per_token_above_272k_tokens?: number | null;
/** Output Cost Per Token Above 272K Tokens Batches */
output_cost_per_token_above_272k_tokens_batches?: number | null;
/** Output Cost Per Token Above 272K Tokens Flex */
output_cost_per_token_above_272k_tokens_flex?: number | null;
/** Output Cost Per Token Above 272K Tokens Priority */
@ -39455,6 +39459,8 @@ export interface components {
input_cost_per_token_above_200k_tokens_priority?: number | null;
/** Input Cost Per Token Above 272K Tokens */
input_cost_per_token_above_272k_tokens?: number | null;
/** Input Cost Per Token Above 272K Tokens Batches */
input_cost_per_token_above_272k_tokens_batches?: number | null;
/** Input Cost Per Token Above 272K Tokens Flex */
input_cost_per_token_above_272k_tokens_flex?: number | null;
/** Input Cost Per Token Above 272K Tokens Priority */
@ -39560,6 +39566,8 @@ export interface components {
output_cost_per_token_above_200k_tokens_priority?: number | null;
/** Output Cost Per Token Above 272K Tokens */
output_cost_per_token_above_272k_tokens?: number | null;
/** Output Cost Per Token Above 272K Tokens Batches */
output_cost_per_token_above_272k_tokens_batches?: number | null;
/** Output Cost Per Token Above 272K Tokens Flex */
output_cost_per_token_above_272k_tokens_flex?: number | null;
/** Output Cost Per Token Above 272K Tokens Priority */