From 1c289e5ecd1a022e71fface263992708d1b536e8 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:31:27 -0700 Subject: [PATCH] fix(prices): add baseten/zai-org/GLM-5.3-Fast pricing (#42764) * fix(prices): add baseten/zai-org/GLM-5.3-Fast pricing with cost tracking e2e Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): assert message instead of comment on breakdown row Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(baseten): drop the live e2e cost tracking test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: kerry --- ...odel_prices_and_context_window_backup.json | 24 +++++++++++++++++++ model_prices_and_context_window.json | 24 +++++++++++++++++++ tests/test_litellm/test_cost_calculator.py | 18 ++++++++++++++ 3 files changed, 66 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index da8013ae709..4a058eafd1f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -64063,6 +64063,30 @@ "supports_tool_choice": true, "supports_vision": true }, + "baseten/zai-org/GLM-5.3-Fast": { + "cache_read_input_token_cost": 2.1e-07, + "input_cost_per_token": 2.1e-06, + "litellm_provider": "baseten", + "max_input_tokens": 1048576, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 6.6e-06, + "source": "https://www.baseten.co/library/glm-53-fast/", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "openrouter/minimax/minimax-m3": { "input_cost_per_token": 3e-07, "output_cost_per_token": 1.2e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index da8013ae709..4a058eafd1f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -64063,6 +64063,30 @@ "supports_tool_choice": true, "supports_vision": true }, + "baseten/zai-org/GLM-5.3-Fast": { + "cache_read_input_token_cost": 2.1e-07, + "input_cost_per_token": 2.1e-06, + "litellm_provider": "baseten", + "max_input_tokens": 1048576, + "max_output_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 6.6e-06, + "source": "https://www.baseten.co/library/glm-53-fast/", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "openrouter/minimax/minimax-m3": { "input_cost_per_token": 3e-07, "output_cost_per_token": 1.2e-06, diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 10a904d6141..f7d6cfaf079 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -4635,6 +4635,24 @@ def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_car assert info["supports_pdf_input"] is False +def test_baseten_glm_5_3_fast_is_priced_from_registry(_local_model_cost_map: None) -> None: + model: Final = "baseten/zai-org/GLM-5.3-Fast" + prompt_tokens: Final = 1000 + completion_tokens: Final = 500 + + prompt_usd, completion_usd = litellm.cost_per_token( + model=model, + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + ) + + entry: Final = litellm.model_cost[model] + assert prompt_usd == pytest.approx(prompt_tokens * entry["input_cost_per_token"]) + assert completion_usd == pytest.approx(completion_tokens * entry["output_cost_per_token"]) + assert prompt_usd > 0 + assert completion_usd > 0 + + def test_completion_cost_charges_explicit_per_token_rates_over_registered_ones( monkeypatch: pytest.MonkeyPatch, ) -> None: