litellm/model_prices_and_context_window.schema.json
mateo-berri be594f5984 feat(guardrails): count bedrock guardrail cost against spend and budgets
Price ApplyGuardrail usage units recorded by PR #37225 with a new
bedrock/guardrails entry in the model cost map (regional override via
bedrock/{region}/guardrails), add the per-request guardrail_cost to the
standard logging payload's response_cost and CostBreakdown, surface it in
the x-litellm-response-cost header, and bill blocked requests through the
failure hook so key and team budgets see what AWS bills
2026-08-18 14:16:07 -07:00

778 lines
26 KiB
JSON

{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "LiteLLM model_prices_and_context_window.json",
"description": "Schema for LiteLLM's model price and context window registry (https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). Every top-level key except 'sample_spec' and 'fallback_generalizations' is a model id, optionally prefixed with its provider (e.g. 'azure/gpt-5.4'), mapping to a model entry. All costs are USD per unit. New optional fields are added regularly, so consumers should ignore unknown fields rather than reject them.",
"type": "object",
"properties": {
"sample_spec": {
"type": "object",
"description": "Documentation placeholder illustrating the entry shape; not a real model and not schema-conformant (several values are prose)."
},
"fallback_generalizations": {
"type": "object",
"description": "Regex rules that generalize unknown model ids to known families; not a model entry.",
"properties": {
"rules": {
"type": "array",
"items": {
"type": "object",
"properties": {
"name": {
"type": "string"
},
"pattern": {
"type": "string"
},
"description": {
"type": "string"
}
},
"required": [
"name",
"pattern"
],
"additionalProperties": true
}
}
},
"additionalProperties": false
}
},
"additionalProperties": {
"$ref": "#/$defs/modelEntry"
},
"$defs": {
"modelEntry": {
"type": "object",
"description": "Pricing, limits, and capability flags for one model. Fields other than litellm_provider are optional; boolean capability flags are simply omitted when unknown or false.",
"required": [
"litellm_provider"
],
"properties": {
"annotation_cost_per_page": {
"type": "number",
"minimum": 0
},
"audio_transcription_config": {
"type": "string"
},
"bedrock_converse_supports_strict_tools": {
"type": "boolean"
},
"bedrock_output_config_effort_ceiling": {
"type": "string",
"description": "Highest reasoning effort the Bedrock output_config accepts for this model.",
"enum": [
"low",
"medium",
"high",
"max",
"xhigh"
]
},
"cache_creation_input_audio_token_cost": {
"type": "number",
"minimum": 0
},
"cache_creation_input_token_cost": {
"type": "number",
"minimum": 0,
"description": "USD per token written to the provider's prompt cache."
},
"cache_creation_input_token_cost_above_1hr": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_200k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_272k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"cache_creation_input_token_cost_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"cache_creation_input_token_cost_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"cache_read_input_audio_token_cost": {
"type": "number",
"minimum": 0
},
"cache_read_input_token_cost": {
"type": "number",
"minimum": 0,
"description": "USD per prompt token served from the provider's prompt cache."
},
"cache_read_input_token_cost_above_200k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_above_200k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"cache_read_input_token_cost_above_272k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"cache_read_input_token_cost_above_272k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"cache_read_input_token_cost_above_512k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"cache_read_input_token_cost_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"citation_cost_per_token": {
"type": "number",
"minimum": 0
},
"code_interpreter_cost_per_session": {
"type": "number",
"minimum": 0
},
"comment": {
"type": "string"
},
"deprecation_date": {
"type": "string",
"description": "Date the provider deprecates the model, YYYY-MM-DD.",
"format": "date",
"pattern": "^\\d{4}-(0[1-9]|1[0-2])-(0[1-9]|[12]\\d|3[01])$"
},
"gemini_audio_only_live": {
"type": "boolean"
},
"gemini_native_audio": {
"type": "boolean"
},
"guardrail_cost_per_unit": {
"type": "object",
"description": "USD cost per billable guardrail unit, keyed by the provider's usage counter name (e.g. Bedrock's contentPolicyUnits).",
"additionalProperties": {
"type": "number",
"minimum": 0
}
},
"input_cost_per_audio_per_second": {
"type": "number",
"minimum": 0
},
"input_cost_per_audio_per_second_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_audio_token": {
"type": "number",
"minimum": 0
},
"input_cost_per_audio_token_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"input_cost_per_character": {
"type": "number",
"minimum": 0
},
"input_cost_per_character_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_image": {
"type": "number",
"minimum": 0
},
"input_cost_per_image_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_image_token": {
"type": "number",
"minimum": 0
},
"input_cost_per_pixel": {
"type": "number",
"minimum": 0
},
"input_cost_per_query": {
"type": "number",
"minimum": 0
},
"input_cost_per_request": {
"type": "number",
"minimum": 0
},
"input_cost_per_second": {
"type": "number",
"minimum": 0
},
"input_cost_per_token": {
"type": "number",
"minimum": 0,
"description": "USD per prompt token."
},
"input_cost_per_token_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_token_above_200k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_token_above_200k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"input_cost_per_token_above_256k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_token_above_272k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_token_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"input_cost_per_token_above_272k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"input_cost_per_token_above_512k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_token_batches": {
"type": "number",
"minimum": 0,
"description": "USD per prompt token via the provider's batch API."
},
"input_cost_per_token_cache_hit": {
"type": "number",
"minimum": 0
},
"input_cost_per_token_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"input_cost_per_token_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"input_cost_per_video_per_second": {
"type": "number",
"minimum": 0
},
"input_cost_per_video_per_second_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_video_per_second_above_15s_interval": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_cost_per_video_per_second_above_8s_interval": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"input_dbu_cost_per_token": {
"type": "number",
"minimum": 0
},
"litellm_provider": {
"type": "string",
"description": "LiteLLM provider slug; one of https://docs.litellm.ai/docs/providers."
},
"max_input_tokens": {
"type": "integer",
"minimum": 0,
"description": "Maximum prompt/context tokens the model accepts."
},
"max_output_tokens": {
"type": "integer",
"minimum": 0,
"description": "Maximum tokens the model can generate in one response."
},
"max_tokens": {
"type": "integer",
"minimum": 0,
"description": "Legacy field: max output tokens if the provider specifies it, else max input tokens."
},
"metadata": {
"type": "object",
"description": "Free-form notes about the entry (e.g. pricing derivation)."
},
"mode": {
"type": "string",
"description": "Primary API surface / task type of the model.",
"enum": [
"audio_speech",
"audio_transcription",
"chat",
"completion",
"embedding",
"guardrail",
"image_edit",
"image_generation",
"moderation",
"ocr",
"realtime",
"rerank",
"responses",
"search",
"vector_store",
"video_generation"
]
},
"ocr_cost_per_credit": {
"type": "number",
"minimum": 0
},
"ocr_cost_per_page": {
"type": "number",
"minimum": 0
},
"output_cost_per_audio_token": {
"type": "number",
"minimum": 0
},
"output_cost_per_character": {
"type": "number",
"minimum": 0
},
"output_cost_per_character_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_image": {
"type": "number",
"minimum": 0
},
"output_cost_per_image_token": {
"type": "number",
"minimum": 0
},
"output_cost_per_pixel": {
"type": "number",
"minimum": 0
},
"output_cost_per_reasoning_token": {
"type": "number",
"minimum": 0,
"description": "USD per reasoning/thinking token, when billed separately."
},
"output_cost_per_second": {
"type": "number",
"minimum": 0
},
"output_cost_per_second_1080p": {
"type": "number",
"minimum": 0
},
"output_cost_per_token": {
"type": "number",
"minimum": 0,
"description": "USD per generated token."
},
"output_cost_per_token_above_128k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_token_above_200k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_token_above_200k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"output_cost_per_token_above_256k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_token_above_272k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_token_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"output_cost_per_token_above_272k_tokens_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"output_cost_per_token_above_512k_tokens": {
"type": "number",
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"output_cost_per_token_batches": {
"type": "number",
"minimum": 0,
"description": "USD per generated token via the provider's batch API."
},
"output_cost_per_token_flex": {
"type": "number",
"minimum": 0,
"description": "Flex service-tier rate for the same-named base field."
},
"output_cost_per_token_priority": {
"type": "number",
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"output_cost_per_video_per_second": {
"type": "number",
"minimum": 0
},
"output_cost_per_video_token": {
"type": "number",
"minimum": 0
},
"output_dbu_cost_per_token": {
"type": "number",
"minimum": 0
},
"output_vector_size": {
"type": "integer",
"minimum": 0,
"description": "Embedding dimension for embedding models."
},
"prompt_cache_min_tokens": {
"type": "integer",
"minimum": 0,
"description": "Smallest prefix the provider will actually cache; absent means the provider default applies."
},
"provider_specific_entry": {
"type": "object",
"description": "Provider-internal routing hints (e.g. bedrock_invocation_schema)."
},
"regional_processing_uplift_multiplier_eu": {
"type": "number",
"minimum": 1,
"description": "Multiplier applied to all token costs for EU data residency (e.g. 1.10 = +10%)."
},
"regional_processing_uplift_multiplier_us": {
"type": "number",
"minimum": 1,
"description": "Multiplier applied to all token costs for US data residency (e.g. 1.10 = +10%)."
},
"rpm": {
"type": "integer",
"minimum": 0,
"description": "Provider default requests-per-minute limit."
},
"search_context_cost_per_query": {
"type": "object",
"description": "USD cost per web search query, keyed by search context size.",
"properties": {
"search_context_size_low": {
"type": "number",
"minimum": 0
},
"search_context_size_medium": {
"type": "number",
"minimum": 0
},
"search_context_size_high": {
"type": "number",
"minimum": 0
}
},
"additionalProperties": false
},
"source": {
"type": "string",
"description": "URL of the provider pricing/model page this entry was taken from."
},
"supported_endpoints": {
"type": "array",
"description": "OpenAI-style API routes this model can be called through, e.g. /v1/chat/completions.",
"items": {
"type": "string"
}
},
"supported_modalities": {
"type": "array",
"description": "Input modalities the model accepts.",
"items": {
"type": "string",
"enum": [
"text",
"image",
"audio",
"video"
]
}
},
"supported_output_modalities": {
"type": "array",
"description": "Output modalities the model can produce.",
"items": {
"type": "string",
"enum": [
"text",
"image",
"audio",
"video",
"code"
]
}
},
"supported_regions": {
"type": "array",
"description": "Cloud regions the model is available in ('global' or region ids).",
"items": {
"type": "string"
}
},
"supports_adaptive_thinking": {
"type": "boolean"
},
"supports_assistant_prefill": {
"type": "boolean"
},
"supports_audio_input": {
"type": "boolean"
},
"supports_audio_output": {
"type": "boolean"
},
"supports_computer_use": {
"type": "boolean"
},
"supports_embedding_image_input": {
"type": "boolean"
},
"supports_function_calling": {
"type": "boolean"
},
"supports_image_input": {
"type": "boolean"
},
"supports_image_size": {
"type": "boolean"
},
"supports_low_reasoning_effort": {
"type": "boolean"
},
"supports_max_reasoning_effort": {
"type": "boolean"
},
"supports_mid_conversation_system": {
"type": "boolean"
},
"supports_minimal_reasoning_effort": {
"type": "boolean"
},
"supports_multimodal": {
"type": "boolean"
},
"supports_native_streaming": {
"type": "boolean"
},
"supports_native_structured_output": {
"type": "boolean"
},
"supports_none_reasoning_effort": {
"type": "boolean"
},
"supports_nova_canvas_image_edit": {
"type": "boolean"
},
"supports_output_config": {
"type": "boolean"
},
"supports_parallel_function_calling": {
"type": "boolean"
},
"supports_parallel_tool_use_config": {
"type": "boolean"
},
"supports_pdf_input": {
"type": "boolean"
},
"supports_prompt_caching": {
"type": "boolean"
},
"supports_reasoning": {
"type": "boolean"
},
"supports_response_schema": {
"type": "boolean"
},
"supports_sampling_params": {
"type": "boolean"
},
"supports_speed": {
"type": "boolean"
},
"supports_system_messages": {
"type": "boolean"
},
"supports_tool_choice": {
"type": "boolean"
},
"supports_tool_search": {
"type": "boolean"
},
"supports_url_context": {
"type": "boolean"
},
"supports_video_input": {
"type": "boolean"
},
"supports_vision": {
"type": "boolean"
},
"supports_web_search": {
"type": "boolean"
},
"supports_xhigh_reasoning_effort": {
"type": "boolean"
},
"tiered_pricing": {
"type": "array",
"description": "Context-length or result-count tiered rates; each tier's costs apply within its range.",
"items": {
"type": "object",
"properties": {
"range": {
"type": "array",
"description": "[min, max] prompt-token span this tier applies to.",
"items": {
"type": "number",
"minimum": 0
},
"minItems": 2,
"maxItems": 2
},
"max_results_range": {
"type": "array",
"description": "[min, max] result-count span this tier applies to (search models).",
"items": {
"type": "number",
"minimum": 0
},
"minItems": 2,
"maxItems": 2
},
"input_cost_per_token": {
"type": "number",
"minimum": 0
},
"output_cost_per_token": {
"type": "number",
"minimum": 0
},
"output_cost_per_reasoning_token": {
"type": "number",
"minimum": 0
},
"cache_read_input_token_cost": {
"type": "number",
"minimum": 0
},
"cache_creation_input_token_cost": {
"type": "number",
"minimum": 0
},
"input_cost_per_query": {
"type": "number",
"minimum": 0
}
},
"additionalProperties": false
}
},
"tpm": {
"type": "integer",
"minimum": 0,
"description": "Provider default tokens-per-minute limit."
},
"use_openai_responses_path": {
"type": "boolean"
},
"uses_embed_content": {
"type": "boolean"
},
"web_search_billing_unit": {
"type": "string",
"description": "Whether web search is billed per query or per prompt.",
"enum": [
"per_query",
"per_prompt"
]
}
},
"additionalProperties": true
}
}
}