From af179be681d997d5d91251021f7a385534783e23 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 15:26:44 -0700 Subject: [PATCH] fix(together_ai): stop writing context_length as max_output_tokens in the serverless sync The Together catalog exposes only context_length, so the sync was recording every chat model's context window as its output ceiling. New entries now carry max_input_tokens and the legacy max_tokens from the catalog and get an output ceiling only from a reviewed capability rule. GLM-5.2 and GLM-5.3-Flash rules carry the documented 128K ceiling, and the 26 other inflated together_ai chat entries drop max_output_tokens in both registry copies. --- ...odel_prices_and_context_window_backup.json | 34 +++---------------- model_prices_and_context_window.json | 34 +++---------------- scripts/sync_together_ai_models.py | 24 ++++++++----- .../test_sync_together_ai_models.py | 31 +++++++++++++++-- .../test_together_ai_model_metadata.py | 16 ++++++++- 5 files changed, 68 insertions(+), 71 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ec175025b42..4aa5f44ef63 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -38583,7 +38583,6 @@ "input_cost_per_token": 1.04e-06, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.04e-06, @@ -38705,7 +38704,6 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 6e-07, @@ -38752,7 +38750,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38770,7 +38767,6 @@ "input_cost_per_token": 4.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38788,7 +38784,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 256000, - "max_output_tokens": 256000, "max_tokens": 256000, "metadata": { "successor": "together_ai/moonshotai/Kimi-K3" @@ -38868,7 +38863,6 @@ "input_cost_per_token": 3e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -38885,7 +38879,6 @@ "input_cost_per_token": 0.0, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0.0, @@ -38895,7 +38888,6 @@ "input_cost_per_token": 1.7e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 2.5e-07, @@ -38911,7 +38903,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 3e-06, @@ -38923,7 +38914,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 7.5e-06, @@ -38934,7 +38924,6 @@ "input_cost_per_token": 3.2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 1.28e-06, @@ -38945,7 +38934,6 @@ "input_cost_per_token": 2e-06, "litellm_provider": "together_ai", "max_input_tokens": 1010000, - "max_output_tokens": 1010000, "max_tokens": 1010000, "mode": "chat", "output_cost_per_token": 6e-06, @@ -38956,7 +38944,6 @@ "input_cost_per_token": 1e-07, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1e-07, @@ -38967,7 +38954,6 @@ "input_cost_per_token": 1.4e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2.8e-07, @@ -38984,7 +38970,6 @@ "input_cost_per_token": 1.74e-06, "litellm_provider": "together_ai", "max_input_tokens": 512000, - "max_output_tokens": 512000, "max_tokens": 512000, "mode": "chat", "output_cost_per_token": 3.48e-06, @@ -39001,7 +38986,6 @@ "input_cost_per_token": 1.32e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 3.96e-06, @@ -39017,7 +39001,6 @@ "input_cost_per_token": 6e-08, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.2e-07, @@ -39027,7 +39010,6 @@ "input_cost_per_token": 3.9e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 9.7e-07, @@ -39053,7 +39035,6 @@ "input_cost_per_token": 2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2e-07, @@ -39064,7 +39045,6 @@ "input_cost_per_token": 3.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -39077,7 +39057,6 @@ "input_cost_per_token": 9.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 4e-06, @@ -39094,7 +39073,6 @@ "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -39118,7 +39096,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 512288, - "max_output_tokens": 512288, "max_tokens": 512288, "mode": "chat", "output_cost_per_token": 3.6e-06, @@ -39135,7 +39112,6 @@ "input_cost_per_token": 2.8e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 8.6e-07, @@ -39146,7 +39122,6 @@ "input_cost_per_token": 1e-06, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 4.05e-06, @@ -39162,7 +39137,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -39174,8 +39148,8 @@ "input_cost_per_token": 1.4e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 4.4e-06, "source": "https://docs.together.ai/docs/serverless-models", @@ -39191,8 +39165,8 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 5e-07, "source": "https://docs.together.ai/docs/serverless-models", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ec175025b42..4aa5f44ef63 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -38583,7 +38583,6 @@ "input_cost_per_token": 1.04e-06, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.04e-06, @@ -38705,7 +38704,6 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 6e-07, @@ -38752,7 +38750,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38770,7 +38767,6 @@ "input_cost_per_token": 4.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38788,7 +38784,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 256000, - "max_output_tokens": 256000, "max_tokens": 256000, "metadata": { "successor": "together_ai/moonshotai/Kimi-K3" @@ -38868,7 +38863,6 @@ "input_cost_per_token": 3e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -38885,7 +38879,6 @@ "input_cost_per_token": 0.0, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0.0, @@ -38895,7 +38888,6 @@ "input_cost_per_token": 1.7e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 2.5e-07, @@ -38911,7 +38903,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 3e-06, @@ -38923,7 +38914,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 7.5e-06, @@ -38934,7 +38924,6 @@ "input_cost_per_token": 3.2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 1.28e-06, @@ -38945,7 +38934,6 @@ "input_cost_per_token": 2e-06, "litellm_provider": "together_ai", "max_input_tokens": 1010000, - "max_output_tokens": 1010000, "max_tokens": 1010000, "mode": "chat", "output_cost_per_token": 6e-06, @@ -38956,7 +38944,6 @@ "input_cost_per_token": 1e-07, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1e-07, @@ -38967,7 +38954,6 @@ "input_cost_per_token": 1.4e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2.8e-07, @@ -38984,7 +38970,6 @@ "input_cost_per_token": 1.74e-06, "litellm_provider": "together_ai", "max_input_tokens": 512000, - "max_output_tokens": 512000, "max_tokens": 512000, "mode": "chat", "output_cost_per_token": 3.48e-06, @@ -39001,7 +38986,6 @@ "input_cost_per_token": 1.32e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 3.96e-06, @@ -39017,7 +39001,6 @@ "input_cost_per_token": 6e-08, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.2e-07, @@ -39027,7 +39010,6 @@ "input_cost_per_token": 3.9e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 9.7e-07, @@ -39053,7 +39035,6 @@ "input_cost_per_token": 2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2e-07, @@ -39064,7 +39045,6 @@ "input_cost_per_token": 3.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -39077,7 +39057,6 @@ "input_cost_per_token": 9.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 4e-06, @@ -39094,7 +39073,6 @@ "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -39118,7 +39096,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 512288, - "max_output_tokens": 512288, "max_tokens": 512288, "mode": "chat", "output_cost_per_token": 3.6e-06, @@ -39135,7 +39112,6 @@ "input_cost_per_token": 2.8e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 8.6e-07, @@ -39146,7 +39122,6 @@ "input_cost_per_token": 1e-06, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 4.05e-06, @@ -39162,7 +39137,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -39174,8 +39148,8 @@ "input_cost_per_token": 1.4e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 4.4e-06, "source": "https://docs.together.ai/docs/serverless-models", @@ -39191,8 +39165,8 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 5e-07, "source": "https://docs.together.ai/docs/serverless-models", diff --git a/scripts/sync_together_ai_models.py b/scripts/sync_together_ai_models.py index 97b308fb660..12b128890f1 100644 --- a/scripts/sync_together_ai_models.py +++ b/scripts/sync_together_ai_models.py @@ -187,9 +187,22 @@ CAPABILITY_RULES: Final = ( _rule("thinkingmachines/Inkling-Small", "reviewed for the LIT-5968 backfill; no tool or vision support documented"), _rule( "zai-org/GLM-5.2", - "reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2", + "reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2;" + " 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.2", **_TOOLS, supports_reasoning=True, + max_output_tokens=128000, + max_tokens=128000, + ), + _rule( + "zai-org/GLM-5.3-Flash", + "reviewed for LIT-6489 against https://www.together.ai/models/glm-5-3-flash;" + " 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.3", + **_TOOLS, + supports_reasoning=True, + supports_vision=True, + max_output_tokens=128000, + max_tokens=128000, ), ) @@ -288,15 +301,10 @@ def _api_fields(model: CatalogModel) -> RegistryEntry: def _new_entry(model: CatalogModel, mode: str) -> RegistryEntry: rule: Final = RULES_BY_ID.get(model.id) - length_fields: Final = ( - {} - if model.context_length is None - else {"max_input_tokens": model.context_length, "max_tokens": model.context_length} - | ({"max_output_tokens": model.context_length} if mode == "chat" else {}) - ) + legacy_ceiling: Final = {} if model.context_length is None else {"max_tokens": model.context_length} merged: Final = { **_api_fields(model), - **length_fields, + **legacy_ceiling, "litellm_provider": PROVIDER, "mode": mode, "source": SOURCE_URL, diff --git a/tests/test_litellm/test_sync_together_ai_models.py b/tests/test_litellm/test_sync_together_ai_models.py index 7c1287e94b8..b8a85bcfbdc 100644 --- a/tests/test_litellm/test_sync_together_ai_models.py +++ b/tests/test_litellm/test_sync_together_ai_models.py @@ -105,7 +105,6 @@ def test_added_chat_model_matches_reviewed_registry_shape() -> None: "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -138,7 +137,35 @@ def test_moderation_type_maps_to_chat_mode() -> None: outcome = sync.compute_sync({}, RECORDED_CATALOG, RECORDED_DOC) guard = outcome.cost_map["together_ai/meta-llama/Llama-Guard-4-12B"] assert guard["mode"] == "chat" - assert guard["max_output_tokens"] == 1048576 + assert "max_output_tokens" not in guard + + +def test_output_ceiling_comes_from_the_rule_never_from_context_length() -> None: + glm = next(model for model in RECORDED_CATALOG if model.id == "zai-org/GLM-5.2") + fresh = sync.compute_sync({}, [_chat_model("acme/unreviewed", ctx=1048576), glm], _doc({"x": "2026-01-01"})) + unreviewed = fresh.cost_map["together_ai/acme/unreviewed"] + assert "max_output_tokens" not in unreviewed + assert (unreviewed["max_input_tokens"], unreviewed["max_tokens"]) == (1048576, 1048576) + reviewed = fresh.cost_map["together_ai/zai-org/GLM-5.2"] + assert (reviewed["max_input_tokens"], reviewed["max_output_tokens"], reviewed["max_tokens"]) == ( + 1048575, + 128000, + 128000, + ) + inflated = { + "together_ai/zai-org/GLM-5.2": { + "input_cost_per_token": 1.4e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 1048575, + "max_output_tokens": 1048575, + "max_tokens": 1048575, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + } + } + corrected = sync.compute_sync(inflated, [glm], _doc({"x": "2026-01-01"})) + assert corrected.cost_map["together_ai/zai-org/GLM-5.2"]["max_output_tokens"] == 128000 + assert any("max_output_tokens: 1048575 -> 128000" in line for line in corrected.updated) def test_docs_removed_but_live_model_stays_live_with_warning() -> None: diff --git a/tests/test_litellm/test_together_ai_model_metadata.py b/tests/test_litellm/test_together_ai_model_metadata.py index 45f0370386b..eda8190ad7a 100644 --- a/tests/test_litellm/test_together_ai_model_metadata.py +++ b/tests/test_litellm/test_together_ai_model_metadata.py @@ -107,6 +107,8 @@ def test_together_glm_52_pricing(cost_map: CostMap): info = cost_map["together_ai/zai-org/GLM-5.2"] assert info["input_cost_per_token"] == 1.4e-06 assert info["output_cost_per_token"] == 4.4e-06 + assert info["max_input_tokens"] == 1048575 + assert info["max_output_tokens"] == 128000 assert info["supports_function_calling"] is True assert info["supports_reasoning"] is True @@ -117,7 +119,7 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap): assert info["output_cost_per_token"] == 5e-07 assert info["cache_read_input_token_cost"] == 3e-08 assert info["max_input_tokens"] == 1048575 - assert info["max_output_tokens"] == 1048575 + assert info["max_output_tokens"] == 128000 assert info["supports_function_calling"] is True assert info["supports_parallel_function_calling"] is True assert info["supports_prompt_caching"] is True @@ -127,6 +129,18 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap): assert info["supports_reasoning"] is True +def test_together_chat_entries_never_carry_context_length_as_output_ceiling(cost_map: CostMap): + inflated = sorted( + model + for model, info in cost_map.items() + if info.get("litellm_provider") == "together_ai" + and info.get("mode") == "chat" + and "max_output_tokens" in info + and info["max_output_tokens"] == info.get("max_input_tokens") + ) + assert inflated == [] + + def test_together_multilingual_e5_embedding_entry(cost_map: CostMap): info = cost_map["together_ai/intfloat/multilingual-e5-large-instruct"] assert info["mode"] == "embedding"