mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): mirror batch long-context keys on custom pricing params
Register the two *_above_272k_tokens_batches keys on CustomPricingLiteLLMParams so a per-deployment override stays out of the shared backend key, add them to the inline model-info schema and alias-count tests, and build LiteLLM_Params and GenericLiteLLMParams through model_validate at the two dict-splat call sites so basedpyright's reportArgumentType budget ratchets down instead of blocking the new fields.
This commit is contained in:
parent
b7f7d5c42f
commit
2c182f4b93
7 changed files with 17 additions and 9 deletions
|
|
@ -3,7 +3,7 @@
|
|||
"limit": 14074
|
||||
},
|
||||
"reportArgumentType": {
|
||||
"limit": 2206
|
||||
"limit": 1921
|
||||
},
|
||||
"reportAssignmentType": {
|
||||
"limit": 319
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ class AzurePassthroughConfig(BasePassthroughConfig):
|
|||
) -> dict:
|
||||
return BaseAzureLLM._base_validate_azure_environment(
|
||||
headers=headers,
|
||||
litellm_params=GenericLiteLLMParams(**{**litellm_params, "api_key": api_key}),
|
||||
litellm_params=GenericLiteLLMParams.model_validate({**litellm_params, "api_key": api_key}),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -8643,12 +8643,8 @@ class Router:
|
|||
if ptu_error is not None and is_ptu_cost_attribution_enabled():
|
||||
raise ValueError(ptu_error)
|
||||
zeroed_pricing: Final = zeroed_ptu_pricing(_model_info, _litellm_params) if config_sourced else None
|
||||
litellm_params: Final[LiteLLM_Params] = LiteLLM_Params(
|
||||
**(
|
||||
_litellm_params
|
||||
if zeroed_pricing is None
|
||||
else MappingProxyType({**_litellm_params, **zeroed_pricing})
|
||||
)
|
||||
litellm_params: Final[LiteLLM_Params] = LiteLLM_Params.model_validate(
|
||||
_litellm_params if zeroed_pricing is None else MappingProxyType({**_litellm_params, **zeroed_pricing})
|
||||
)
|
||||
warn_on_provider_credential_mismatch(model_name=_model_name, litellm_params=_litellm_params)
|
||||
deployment = Deployment(
|
||||
|
|
|
|||
|
|
@ -3481,6 +3481,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
input_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
input_cost_per_query: float | None = None
|
||||
input_cost_per_image: float | None = None
|
||||
input_cost_per_image_above_128k_tokens: float | None = None
|
||||
|
|
@ -3501,6 +3502,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
output_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
output_cost_per_character_above_128k_tokens: float | None = None
|
||||
output_cost_per_image: float | None = None
|
||||
output_cost_per_image_token: float | None = None
|
||||
|
|
|
|||
|
|
@ -1776,7 +1776,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
|
|||
sol = litellm.model_cost["gpt-5.6-sol"]
|
||||
|
||||
cost_fields = sorted(field for field in sol if "cost" in field)
|
||||
assert len(cost_fields) == 27
|
||||
assert len(cost_fields) == 29
|
||||
|
||||
for field in cost_fields:
|
||||
assert alias.get(field) == sol.get(field), field
|
||||
|
|
|
|||
|
|
@ -959,12 +959,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_token_priority": {"type": "number"},
|
||||
"input_cost_per_token_above_200k_tokens_priority": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_priority": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_batches": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_flex": {"type": "number"},
|
||||
"input_cost_per_audio_token_priority": {"type": "number"},
|
||||
"output_cost_per_token_flex": {"type": "number"},
|
||||
"output_cost_per_token_priority": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens_priority": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_priority": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_batches": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_flex": {"type": "number"},
|
||||
"regional_endpoint_uplift_multiplier": {"type": "number"},
|
||||
"regional_processing_uplift_multiplier_eu": {"type": "number"},
|
||||
|
|
|
|||
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -29355,6 +29355,8 @@ export interface components {
|
|||
input_cost_per_token_above_200k_tokens_priority?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens */
|
||||
input_cost_per_token_above_272k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Batches */
|
||||
input_cost_per_token_above_272k_tokens_batches?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Flex */
|
||||
input_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Priority */
|
||||
|
|
@ -29460,6 +29462,8 @@ export interface components {
|
|||
output_cost_per_token_above_200k_tokens_priority?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens */
|
||||
output_cost_per_token_above_272k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Batches */
|
||||
output_cost_per_token_above_272k_tokens_batches?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Flex */
|
||||
output_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Priority */
|
||||
|
|
@ -39455,6 +39459,8 @@ export interface components {
|
|||
input_cost_per_token_above_200k_tokens_priority?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens */
|
||||
input_cost_per_token_above_272k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Batches */
|
||||
input_cost_per_token_above_272k_tokens_batches?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Flex */
|
||||
input_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Priority */
|
||||
|
|
@ -39560,6 +39566,8 @@ export interface components {
|
|||
output_cost_per_token_above_200k_tokens_priority?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens */
|
||||
output_cost_per_token_above_272k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Batches */
|
||||
output_cost_per_token_above_272k_tokens_batches?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Flex */
|
||||
output_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Priority */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue