From c452f7b713f9cefa0a9983b8499855235d8514ab Mon Sep 17 00:00:00 2001 From: AlexBGoode Date: Thu, 7 May 2026 18:38:39 +0000 Subject: [PATCH] feat(catalog): add zai/glm-5.1, zai/glm-4.7-flash, openrouter/z-ai/glm-5.1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three new pricing entries for Z.AI's GLM family, all sourced from official upstream documentation: * `zai/glm-5.1` (Z.AI direct, provider=zai) - input $1.40/1M, cached input $0.26/1M, output $4.40/1M - context 200K in / 128K out (matching sibling zai/glm-5 / zai/glm-4.7) - source: https://docs.z.ai/guides/overview/pricing * `zai/glm-4.7-flash` (Z.AI direct, provider=zai) - all four price columns published as Free on the Z.AI pricing page (input / cached input / cache storage / output) - context 200K in / 128K out - supports_prompt_caching: true and supports_reasoning: true to match sibling zai/glm-4.7 conventions; the Z.AI pricing page lists a cached- input column for this model (free), so the model is caching-eligible - source: https://docs.z.ai/guides/overview/pricing * `openrouter/z-ai/glm-5.1` (OpenRouter route, provider=openrouter) - input $1.05/1M, cached input $0.525/1M, output $3.50/1M - max_input_tokens 202752, max_output_tokens 65535 (matches OpenRouter's `top_provider.max_completion_tokens` for this model; intentionally lower than Z.AI direct's 128K since this is the OR endpoint's actual ceiling) - source: https://openrouter.ai/z-ai/glm-5.1 All three entries follow the conventions of surrounding entries (zai/glm-5 and zai/glm-4.7 for the zai/* namespace, openrouter/z-ai/glm-5 for the openrouter/z-ai/* namespace). Why: direct Z.AI users and openrouter users on these models currently get response_cost = 0.0 in spend logs because the cost calculator falls back to the static catalog and finds nothing for these newer models. With these entries, cost calc picks them up by router_model_id, so spend tracking and budget enforcement work without per-deployment input_cost_per_token / output_cost_per_token overrides on every install. Verified empirically against: - https://docs.z.ai/guides/overview/pricing (2026-05-07) for the two zai/* entries. - https://openrouter.ai/api/v1/models (2026-05-07) for the openrouter/z-ai/glm-5.1 entry. The 65535 max_output_tokens value comes directly from `top_provider.max_completion_tokens` in the JSON response. Test plan: - `python3 -c "import json; json.load(open('model_prices_and_context_window.json'))"` — JSON parses. - All three keys / values inspected against neighbour conventions; no duplicate fields, no removed fields. - No code-path changes, so no test additions are needed in tests/litellm/. --- model_prices_and_context_window.json | 46 ++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 92e87c00ef6..9f233b6aa03 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27403,6 +27403,22 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "openrouter/z-ai/glm-5.1": { + "input_cost_per_token": 1.05e-06, + "output_cost_per_token": 3.5e-06, + "cache_read_input_token_cost": 5.25e-07, + "cache_creation_input_token_cost": 0.0, + "litellm_provider": "openrouter", + "max_input_tokens": 202752, + "max_output_tokens": 65535, + "max_tokens": 65535, + "mode": "chat", + "source": "https://openrouter.ai/z-ai/glm-5.1", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, "openrouter/minimax/minimax-m2.1": { "input_cost_per_token": 2.7e-07, "output_cost_per_token": 1.2e-06, @@ -35130,6 +35146,21 @@ "supports_tool_choice": true, "source": "https://docs.z.ai/guides/overview/pricing" }, + "zai/glm-5.1": { + "cache_creation_input_token_cost": 0, + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "litellm_provider": "zai", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "source": "https://docs.z.ai/guides/overview/pricing" + }, "zai/glm-5-code": { "cache_creation_input_token_cost": 0, "cache_read_input_token_cost": 3e-07, @@ -35160,6 +35191,21 @@ "supports_tool_choice": true, "source": "https://docs.z.ai/guides/overview/pricing" }, + "zai/glm-4.7-flash": { + "cache_creation_input_token_cost": 0, + "cache_read_input_token_cost": 0, + "input_cost_per_token": 0, + "output_cost_per_token": 0, + "litellm_provider": "zai", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "source": "https://docs.z.ai/guides/overview/pricing" + }, "zai/glm-4.6": { "cache_creation_input_token_cost": 0, "cache_read_input_token_cost": 1.1e-07,