mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
feat(model_prices): add databricks kimi k3 and glm 5.2 endpoints
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
52b7bea6f3
commit
04593cfe48
3 changed files with 142 additions and 1 deletions
|
|
@ -15167,6 +15167,34 @@
|
|||
"output_dbu_cost_per_token": 7.143e-06,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-glm-5-2": {
|
||||
"cache_creation_input_token_cost": 1.4e-06,
|
||||
"cache_read_input_token_cost": 2.5998e-07,
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"input_dbu_cost_per_token": 2e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"metadata": {
|
||||
"notes": "Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.39999e-06,
|
||||
"output_dbu_cost_per_token": 6.2857e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving",
|
||||
"supported_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"databricks/databricks-gpt-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
|
|
@ -15434,6 +15462,35 @@
|
|||
"output_vector_size": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-kimi-k3": {
|
||||
"cache_creation_input_token_cost": 2.99999e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.99999e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"metadata": {
|
||||
"notes": "Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.500002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-llama-2-70b-chat": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
|
|
|
|||
|
|
@ -15167,6 +15167,34 @@
|
|||
"output_dbu_cost_per_token": 7.143e-06,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-glm-5-2": {
|
||||
"cache_creation_input_token_cost": 1.4e-06,
|
||||
"cache_read_input_token_cost": 2.5998e-07,
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"input_dbu_cost_per_token": 2e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"metadata": {
|
||||
"notes": "Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.39999e-06,
|
||||
"output_dbu_cost_per_token": 6.2857e-05,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving",
|
||||
"supported_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"databricks/databricks-gpt-5": {
|
||||
"cache_creation_input_token_cost": 1.24999e-06,
|
||||
"cache_read_input_token_cost": 1.2502e-07,
|
||||
|
|
@ -15434,6 +15462,35 @@
|
|||
"output_vector_size": 1024,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving"
|
||||
},
|
||||
"databricks/databricks-kimi-k3": {
|
||||
"cache_creation_input_token_cost": 2.99999e-06,
|
||||
"cache_read_input_token_cost": 3.0002e-07,
|
||||
"input_cost_per_token": 2.99999e-06,
|
||||
"input_dbu_cost_per_token": 4.2857e-05,
|
||||
"litellm_provider": "databricks",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"metadata": {
|
||||
"notes": "Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."
|
||||
},
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.500002e-05,
|
||||
"output_dbu_cost_per_token": 0.000214286,
|
||||
"source": "https://www.databricks.com/product/pricing/foundation-model-serving",
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"databricks/databricks-llama-2-70b-chat": {
|
||||
"cache_creation_input_token_cost": 5.0001e-07,
|
||||
"cache_read_input_token_cost": 5.0001e-07,
|
||||
|
|
|
|||
|
|
@ -61,7 +61,13 @@ PUBLISHED_DBU_PER_MILLION: Final = {
|
|||
"databricks/databricks-gemini-3-1-flash-lite": ("4.464", "26.786", "4.464", "0.446"),
|
||||
"databricks/databricks-gemini-2-5-pro": ("22.321", "178.571", "22.321", "2.232"),
|
||||
"databricks/databricks-gemini-2-5-flash": ("5.357", "44.643", "5.357", "0.536"),
|
||||
"databricks/databricks-kimi-k3": ("42.857", "214.286", "42.857", "4.286"),
|
||||
"databricks/databricks-glm-5-2": ("20.000", "62.857", "20.000", "3.714"),
|
||||
}
|
||||
MILLION_TOKEN_CONTEXT_MODELS: Final = (
|
||||
"databricks/databricks-kimi-k3",
|
||||
"databricks/databricks-glm-5-2",
|
||||
)
|
||||
PROMOTIONAL_DISCOUNT: Final = 0.80
|
||||
PROMOTION_EXPIRES: Final = "2027-01-31"
|
||||
ENTRIES_STORING_PROMOTIONAL_RATE: Final = (
|
||||
|
|
@ -211,7 +217,28 @@ def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_ra
|
|||
assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS)
|
||||
@pytest.mark.parametrize("model", MILLION_TOKEN_CONTEXT_MODELS)
|
||||
def test_million_token_context_models_price_and_size_at_published_values(
|
||||
local_model_cost_map: None,
|
||||
model: str,
|
||||
) -> None:
|
||||
info: Final = _model_info(model)
|
||||
input_dbu, output_dbu, _, cache_read_dbu = PUBLISHED_DBU_PER_MILLION[model]
|
||||
|
||||
assert info["input_cost_per_token"] == _dollars_per_token(input_dbu)
|
||||
assert info["output_cost_per_token"] == _dollars_per_token(output_dbu)
|
||||
assert info["cache_read_input_token_cost"] == _dollars_per_token(cache_read_dbu)
|
||||
assert info["max_input_tokens"] == 1000000
|
||||
assert info["mode"] == "chat"
|
||||
assert info["supports_prompt_caching"] is True
|
||||
|
||||
|
||||
def test_kimi_k3_accepts_images_while_glm_5_2_is_text_only(local_model_cost_map: None) -> None:
|
||||
assert _model_info("databricks/databricks-kimi-k3")["supports_vision"] is True
|
||||
assert _model_info("databricks/databricks-glm-5-2")["supports_vision"] is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS + MILLION_TOKEN_CONTEXT_MODELS)
|
||||
def test_backup_price_map_matches_main(model: str) -> None:
|
||||
main_cost: Final = json.loads(MAIN_PRICES.read_text())
|
||||
backup_cost: Final = json.loads(BACKUP_PRICES.read_text())
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue