diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 05fa5df6855..05c1cfd3179 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -38678,7 +38678,6 @@ "input_cost_per_token": 1.04e-06, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.04e-06, @@ -38800,7 +38799,6 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 6e-07, @@ -38847,7 +38845,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38865,7 +38862,6 @@ "input_cost_per_token": 4.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38883,7 +38879,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 256000, - "max_output_tokens": 256000, "max_tokens": 256000, "metadata": { "successor": "together_ai/moonshotai/Kimi-K3" @@ -38963,7 +38958,6 @@ "input_cost_per_token": 3e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -38980,7 +38974,6 @@ "input_cost_per_token": 0.0, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0.0, @@ -38990,7 +38983,6 @@ "input_cost_per_token": 1.7e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 2.5e-07, @@ -39006,7 +38998,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 3e-06, @@ -39018,7 +39009,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 7.5e-06, @@ -39029,7 +39019,6 @@ "input_cost_per_token": 3.2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 1.28e-06, @@ -39040,7 +39029,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1010000, - "max_output_tokens": 1010000, "max_tokens": 1010000, "mode": "chat", "output_cost_per_token": 6.25e-06, @@ -39051,7 +39039,6 @@ "input_cost_per_token": 1e-07, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1e-07, @@ -39062,7 +39049,6 @@ "input_cost_per_token": 1.4e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2.8e-07, @@ -39079,7 +39065,6 @@ "input_cost_per_token": 1.74e-06, "litellm_provider": "together_ai", "max_input_tokens": 512000, - "max_output_tokens": 512000, "max_tokens": 512000, "mode": "chat", "output_cost_per_token": 3.48e-06, @@ -39096,7 +39081,6 @@ "input_cost_per_token": 1.32e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 3.96e-06, @@ -39112,7 +39096,6 @@ "input_cost_per_token": 6e-08, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.2e-07, @@ -39122,7 +39105,6 @@ "input_cost_per_token": 3.9e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 9.7e-07, @@ -39148,7 +39130,6 @@ "input_cost_per_token": 2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2e-07, @@ -39159,7 +39140,6 @@ "input_cost_per_token": 3.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -39172,7 +39152,6 @@ "input_cost_per_token": 9.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 4e-06, @@ -39189,7 +39168,6 @@ "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -39213,7 +39191,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 512288, - "max_output_tokens": 512288, "max_tokens": 512288, "mode": "chat", "output_cost_per_token": 3.6e-06, @@ -39230,7 +39207,6 @@ "input_cost_per_token": 2.8e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 8.6e-07, @@ -39241,7 +39217,6 @@ "input_cost_per_token": 1e-06, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 4.05e-06, @@ -39257,7 +39232,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -39269,8 +39243,8 @@ "input_cost_per_token": 1.4e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 4.4e-06, "source": "https://docs.together.ai/docs/serverless-models", @@ -39303,8 +39277,8 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 5e-07, "source": "https://docs.together.ai/docs/serverless-models", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 05fa5df6855..05c1cfd3179 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -38678,7 +38678,6 @@ "input_cost_per_token": 1.04e-06, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.04e-06, @@ -38800,7 +38799,6 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 6e-07, @@ -38847,7 +38845,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38865,7 +38862,6 @@ "input_cost_per_token": 4.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 200000, - "max_output_tokens": 200000, "max_tokens": 200000, "metadata": { "successor": "together_ai/zai-org/GLM-5.2" @@ -38883,7 +38879,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 256000, - "max_output_tokens": 256000, "max_tokens": 256000, "metadata": { "successor": "together_ai/moonshotai/Kimi-K3" @@ -38963,7 +38958,6 @@ "input_cost_per_token": 3e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -38980,7 +38974,6 @@ "input_cost_per_token": 0.0, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 0.0, @@ -38990,7 +38983,6 @@ "input_cost_per_token": 1.7e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 2.5e-07, @@ -39006,7 +38998,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 3e-06, @@ -39018,7 +39009,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 7.5e-06, @@ -39029,7 +39019,6 @@ "input_cost_per_token": 3.2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1000000, - "max_output_tokens": 1000000, "max_tokens": 1000000, "mode": "chat", "output_cost_per_token": 1.28e-06, @@ -39040,7 +39029,6 @@ "input_cost_per_token": 2.5e-06, "litellm_provider": "together_ai", "max_input_tokens": 1010000, - "max_output_tokens": 1010000, "max_tokens": 1010000, "mode": "chat", "output_cost_per_token": 6.25e-06, @@ -39051,7 +39039,6 @@ "input_cost_per_token": 1e-07, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1e-07, @@ -39062,7 +39049,6 @@ "input_cost_per_token": 1.4e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2.8e-07, @@ -39079,7 +39065,6 @@ "input_cost_per_token": 1.74e-06, "litellm_provider": "together_ai", "max_input_tokens": 512000, - "max_output_tokens": 512000, "max_tokens": 512000, "mode": "chat", "output_cost_per_token": 3.48e-06, @@ -39096,7 +39081,6 @@ "input_cost_per_token": 1.32e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 3.96e-06, @@ -39112,7 +39096,6 @@ "input_cost_per_token": 6e-08, "litellm_provider": "together_ai", "max_input_tokens": 32768, - "max_output_tokens": 32768, "max_tokens": 32768, "mode": "chat", "output_cost_per_token": 1.2e-07, @@ -39122,7 +39105,6 @@ "input_cost_per_token": 3.9e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 9.7e-07, @@ -39148,7 +39130,6 @@ "input_cost_per_token": 2e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 2e-07, @@ -39159,7 +39140,6 @@ "input_cost_per_token": 3.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 131072, - "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 1.5e-06, @@ -39172,7 +39152,6 @@ "input_cost_per_token": 9.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 4e-06, @@ -39189,7 +39168,6 @@ "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -39213,7 +39191,6 @@ "input_cost_per_token": 6e-07, "litellm_provider": "together_ai", "max_input_tokens": 512288, - "max_output_tokens": 512288, "max_tokens": 512288, "mode": "chat", "output_cost_per_token": 3.6e-06, @@ -39230,7 +39207,6 @@ "input_cost_per_token": 2.8e-07, "litellm_provider": "together_ai", "max_input_tokens": 262144, - "max_output_tokens": 262144, "max_tokens": 262144, "mode": "chat", "output_cost_per_token": 8.6e-07, @@ -39241,7 +39217,6 @@ "input_cost_per_token": 1e-06, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 4.05e-06, @@ -39257,7 +39232,6 @@ "input_cost_per_token": 5e-07, "litellm_provider": "together_ai", "max_input_tokens": 524288, - "max_output_tokens": 524288, "max_tokens": 524288, "mode": "chat", "output_cost_per_token": 1.2e-06, @@ -39269,8 +39243,8 @@ "input_cost_per_token": 1.4e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 4.4e-06, "source": "https://docs.together.ai/docs/serverless-models", @@ -39303,8 +39277,8 @@ "input_cost_per_token": 1.5e-07, "litellm_provider": "together_ai", "max_input_tokens": 1048575, - "max_output_tokens": 1048575, - "max_tokens": 1048575, + "max_output_tokens": 128000, + "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 5e-07, "source": "https://docs.together.ai/docs/serverless-models", diff --git a/scripts/sync_together_ai_models.py b/scripts/sync_together_ai_models.py index 97b308fb660..12b128890f1 100644 --- a/scripts/sync_together_ai_models.py +++ b/scripts/sync_together_ai_models.py @@ -187,9 +187,22 @@ CAPABILITY_RULES: Final = ( _rule("thinkingmachines/Inkling-Small", "reviewed for the LIT-5968 backfill; no tool or vision support documented"), _rule( "zai-org/GLM-5.2", - "reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2", + "reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2;" + " 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.2", **_TOOLS, supports_reasoning=True, + max_output_tokens=128000, + max_tokens=128000, + ), + _rule( + "zai-org/GLM-5.3-Flash", + "reviewed for LIT-6489 against https://www.together.ai/models/glm-5-3-flash;" + " 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.3", + **_TOOLS, + supports_reasoning=True, + supports_vision=True, + max_output_tokens=128000, + max_tokens=128000, ), ) @@ -288,15 +301,10 @@ def _api_fields(model: CatalogModel) -> RegistryEntry: def _new_entry(model: CatalogModel, mode: str) -> RegistryEntry: rule: Final = RULES_BY_ID.get(model.id) - length_fields: Final = ( - {} - if model.context_length is None - else {"max_input_tokens": model.context_length, "max_tokens": model.context_length} - | ({"max_output_tokens": model.context_length} if mode == "chat" else {}) - ) + legacy_ceiling: Final = {} if model.context_length is None else {"max_tokens": model.context_length} merged: Final = { **_api_fields(model), - **length_fields, + **legacy_ceiling, "litellm_provider": PROVIDER, "mode": mode, "source": SOURCE_URL, diff --git a/tests/test_litellm/test_sync_together_ai_models.py b/tests/test_litellm/test_sync_together_ai_models.py index 7c1287e94b8..b8a85bcfbdc 100644 --- a/tests/test_litellm/test_sync_together_ai_models.py +++ b/tests/test_litellm/test_sync_together_ai_models.py @@ -105,7 +105,6 @@ def test_added_chat_model_matches_reviewed_registry_shape() -> None: "input_cost_per_token": 3e-06, "litellm_provider": "together_ai", "max_input_tokens": 1048576, - "max_output_tokens": 1048576, "max_tokens": 1048576, "mode": "chat", "output_cost_per_token": 1.5e-05, @@ -138,7 +137,35 @@ def test_moderation_type_maps_to_chat_mode() -> None: outcome = sync.compute_sync({}, RECORDED_CATALOG, RECORDED_DOC) guard = outcome.cost_map["together_ai/meta-llama/Llama-Guard-4-12B"] assert guard["mode"] == "chat" - assert guard["max_output_tokens"] == 1048576 + assert "max_output_tokens" not in guard + + +def test_output_ceiling_comes_from_the_rule_never_from_context_length() -> None: + glm = next(model for model in RECORDED_CATALOG if model.id == "zai-org/GLM-5.2") + fresh = sync.compute_sync({}, [_chat_model("acme/unreviewed", ctx=1048576), glm], _doc({"x": "2026-01-01"})) + unreviewed = fresh.cost_map["together_ai/acme/unreviewed"] + assert "max_output_tokens" not in unreviewed + assert (unreviewed["max_input_tokens"], unreviewed["max_tokens"]) == (1048576, 1048576) + reviewed = fresh.cost_map["together_ai/zai-org/GLM-5.2"] + assert (reviewed["max_input_tokens"], reviewed["max_output_tokens"], reviewed["max_tokens"]) == ( + 1048575, + 128000, + 128000, + ) + inflated = { + "together_ai/zai-org/GLM-5.2": { + "input_cost_per_token": 1.4e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 1048575, + "max_output_tokens": 1048575, + "max_tokens": 1048575, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + } + } + corrected = sync.compute_sync(inflated, [glm], _doc({"x": "2026-01-01"})) + assert corrected.cost_map["together_ai/zai-org/GLM-5.2"]["max_output_tokens"] == 128000 + assert any("max_output_tokens: 1048575 -> 128000" in line for line in corrected.updated) def test_docs_removed_but_live_model_stays_live_with_warning() -> None: diff --git a/tests/test_litellm/test_together_ai_model_metadata.py b/tests/test_litellm/test_together_ai_model_metadata.py index 76f3d7f3f9f..c9e2863d240 100644 --- a/tests/test_litellm/test_together_ai_model_metadata.py +++ b/tests/test_litellm/test_together_ai_model_metadata.py @@ -108,6 +108,8 @@ def test_together_glm_52_pricing(cost_map: CostMap): info = cost_map["together_ai/zai-org/GLM-5.2"] assert info["input_cost_per_token"] == 1.4e-06 assert info["output_cost_per_token"] == 4.4e-06 + assert info["max_input_tokens"] == 1048575 + assert info["max_output_tokens"] == 128000 assert info["supports_function_calling"] is True assert info["supports_reasoning"] is True @@ -118,7 +120,7 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap): assert info["output_cost_per_token"] == 5e-07 assert info["cache_read_input_token_cost"] == 3e-08 assert info["max_input_tokens"] == 1048575 - assert info["max_output_tokens"] == 1048575 + assert info["max_output_tokens"] == 128000 assert info["supports_function_calling"] is True assert info["supports_parallel_function_calling"] is True assert info["supports_prompt_caching"] is True @@ -128,6 +130,18 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap): assert info["supports_reasoning"] is True +def test_together_chat_entries_never_carry_context_length_as_output_ceiling(cost_map: CostMap): + inflated = sorted( + model + for model, info in cost_map.items() + if info.get("litellm_provider") == "together_ai" + and info.get("mode") == "chat" + and "max_output_tokens" in info + and info["max_output_tokens"] == info.get("max_input_tokens") + ) + assert inflated == [] + + def test_together_multilingual_e5_embedding_entry(cost_map: CostMap): info = cost_map["together_ai/intfloat/multilingual-e5-large-instruct"] assert info["mode"] == "embedding"