Merge pull request #38820 from BerriAI/litellm_fix_together_sync_output_ceiling

fix(together_ai): stop writing context_length as max_output_tokens in the serverless sync
This commit is contained in:
Mateo Wang 2026-08-29 16:44:56 -07:00 committed by GitHub
commit 42d8360f29
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 68 additions and 71 deletions

View file

@ -38678,7 +38678,6 @@
"input_cost_per_token": 1.04e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.04e-06,
@ -38800,7 +38799,6 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 6e-07,
@ -38847,7 +38845,6 @@
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_output_tokens": 200000,
"max_tokens": 200000,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
@ -38865,7 +38862,6 @@
"input_cost_per_token": 4.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_output_tokens": 200000,
"max_tokens": 200000,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
@ -38883,7 +38879,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 256000,
"max_output_tokens": 256000,
"max_tokens": 256000,
"metadata": {
"successor": "together_ai/moonshotai/Kimi-K3"
@ -38963,7 +38958,6 @@
"input_cost_per_token": 3e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
@ -38980,7 +38974,6 @@
"input_cost_per_token": 0.0,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 0.0,
@ -38990,7 +38983,6 @@
"input_cost_per_token": 1.7e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 2.5e-07,
@ -39006,7 +38998,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 3e-06,
@ -39018,7 +39009,6 @@
"input_cost_per_token": 2.5e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 7.5e-06,
@ -39029,7 +39019,6 @@
"input_cost_per_token": 3.2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 1.28e-06,
@ -39040,7 +39029,6 @@
"input_cost_per_token": 2.5e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1010000,
"max_output_tokens": 1010000,
"max_tokens": 1010000,
"mode": "chat",
"output_cost_per_token": 6.25e-06,
@ -39051,7 +39039,6 @@
"input_cost_per_token": 1e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1e-07,
@ -39062,7 +39049,6 @@
"input_cost_per_token": 1.4e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
@ -39079,7 +39065,6 @@
"input_cost_per_token": 1.74e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 512000,
"max_output_tokens": 512000,
"max_tokens": 512000,
"mode": "chat",
"output_cost_per_token": 3.48e-06,
@ -39096,7 +39081,6 @@
"input_cost_per_token": 1.32e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
@ -39112,7 +39096,6 @@
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.2e-07,
@ -39122,7 +39105,6 @@
"input_cost_per_token": 3.9e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 9.7e-07,
@ -39148,7 +39130,6 @@
"input_cost_per_token": 2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 2e-07,
@ -39159,7 +39140,6 @@
"input_cost_per_token": 3.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.5e-06,
@ -39172,7 +39152,6 @@
"input_cost_per_token": 9.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 4e-06,
@ -39189,7 +39168,6 @@
"input_cost_per_token": 3e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
@ -39213,7 +39191,6 @@
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 512288,
"max_output_tokens": 512288,
"max_tokens": 512288,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
@ -39230,7 +39207,6 @@
"input_cost_per_token": 2.8e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 8.6e-07,
@ -39241,7 +39217,6 @@
"input_cost_per_token": 1e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 4.05e-06,
@ -39257,7 +39232,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
@ -39269,8 +39243,8 @@
"input_cost_per_token": 1.4e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048575,
"max_output_tokens": 1048575,
"max_tokens": 1048575,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"source": "https://docs.together.ai/docs/serverless-models",
@ -39303,8 +39277,8 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048575,
"max_output_tokens": 1048575,
"max_tokens": 1048575,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 5e-07,
"source": "https://docs.together.ai/docs/serverless-models",

View file

@ -38678,7 +38678,6 @@
"input_cost_per_token": 1.04e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.04e-06,
@ -38800,7 +38799,6 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 6e-07,
@ -38847,7 +38845,6 @@
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_output_tokens": 200000,
"max_tokens": 200000,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
@ -38865,7 +38862,6 @@
"input_cost_per_token": 4.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_output_tokens": 200000,
"max_tokens": 200000,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
@ -38883,7 +38879,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 256000,
"max_output_tokens": 256000,
"max_tokens": 256000,
"metadata": {
"successor": "together_ai/moonshotai/Kimi-K3"
@ -38963,7 +38958,6 @@
"input_cost_per_token": 3e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
@ -38980,7 +38974,6 @@
"input_cost_per_token": 0.0,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 0.0,
@ -38990,7 +38983,6 @@
"input_cost_per_token": 1.7e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 2.5e-07,
@ -39006,7 +38998,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 3e-06,
@ -39018,7 +39009,6 @@
"input_cost_per_token": 2.5e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 7.5e-06,
@ -39029,7 +39019,6 @@
"input_cost_per_token": 3.2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1000000,
"max_output_tokens": 1000000,
"max_tokens": 1000000,
"mode": "chat",
"output_cost_per_token": 1.28e-06,
@ -39040,7 +39029,6 @@
"input_cost_per_token": 2.5e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1010000,
"max_output_tokens": 1010000,
"max_tokens": 1010000,
"mode": "chat",
"output_cost_per_token": 6.25e-06,
@ -39051,7 +39039,6 @@
"input_cost_per_token": 1e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1e-07,
@ -39062,7 +39049,6 @@
"input_cost_per_token": 1.4e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
@ -39079,7 +39065,6 @@
"input_cost_per_token": 1.74e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 512000,
"max_output_tokens": 512000,
"max_tokens": 512000,
"mode": "chat",
"output_cost_per_token": 3.48e-06,
@ -39096,7 +39081,6 @@
"input_cost_per_token": 1.32e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
@ -39112,7 +39096,6 @@
"input_cost_per_token": 6e-08,
"litellm_provider": "together_ai",
"max_input_tokens": 32768,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.2e-07,
@ -39122,7 +39105,6 @@
"input_cost_per_token": 3.9e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 9.7e-07,
@ -39148,7 +39130,6 @@
"input_cost_per_token": 2e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 2e-07,
@ -39159,7 +39140,6 @@
"input_cost_per_token": 3.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 131072,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 1.5e-06,
@ -39172,7 +39152,6 @@
"input_cost_per_token": 9.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 4e-06,
@ -39189,7 +39168,6 @@
"input_cost_per_token": 3e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
@ -39213,7 +39191,6 @@
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 512288,
"max_output_tokens": 512288,
"max_tokens": 512288,
"mode": "chat",
"output_cost_per_token": 3.6e-06,
@ -39230,7 +39207,6 @@
"input_cost_per_token": 2.8e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 262144,
"max_output_tokens": 262144,
"max_tokens": 262144,
"mode": "chat",
"output_cost_per_token": 8.6e-07,
@ -39241,7 +39217,6 @@
"input_cost_per_token": 1e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 4.05e-06,
@ -39257,7 +39232,6 @@
"input_cost_per_token": 5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 524288,
"max_output_tokens": 524288,
"max_tokens": 524288,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
@ -39269,8 +39243,8 @@
"input_cost_per_token": 1.4e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048575,
"max_output_tokens": 1048575,
"max_tokens": 1048575,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"source": "https://docs.together.ai/docs/serverless-models",
@ -39303,8 +39277,8 @@
"input_cost_per_token": 1.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 1048575,
"max_output_tokens": 1048575,
"max_tokens": 1048575,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 5e-07,
"source": "https://docs.together.ai/docs/serverless-models",

View file

@ -187,9 +187,22 @@ CAPABILITY_RULES: Final = (
_rule("thinkingmachines/Inkling-Small", "reviewed for the LIT-5968 backfill; no tool or vision support documented"),
_rule(
"zai-org/GLM-5.2",
"reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2",
"reviewed for the LIT-5968 backfill against https://www.together.ai/models/glm-5-2;"
" 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.2",
**_TOOLS,
supports_reasoning=True,
max_output_tokens=128000,
max_tokens=128000,
),
_rule(
"zai-org/GLM-5.3-Flash",
"reviewed for LIT-6489 against https://www.together.ai/models/glm-5-3-flash;"
" 128K output ceiling per https://docs.z.ai/guides/llm/glm-5.3",
**_TOOLS,
supports_reasoning=True,
supports_vision=True,
max_output_tokens=128000,
max_tokens=128000,
),
)
@ -288,15 +301,10 @@ def _api_fields(model: CatalogModel) -> RegistryEntry:
def _new_entry(model: CatalogModel, mode: str) -> RegistryEntry:
rule: Final = RULES_BY_ID.get(model.id)
length_fields: Final = (
{}
if model.context_length is None
else {"max_input_tokens": model.context_length, "max_tokens": model.context_length}
| ({"max_output_tokens": model.context_length} if mode == "chat" else {})
)
legacy_ceiling: Final = {} if model.context_length is None else {"max_tokens": model.context_length}
merged: Final = {
**_api_fields(model),
**length_fields,
**legacy_ceiling,
"litellm_provider": PROVIDER,
"mode": mode,
"source": SOURCE_URL,

View file

@ -105,7 +105,6 @@ def test_added_chat_model_matches_reviewed_registry_shape() -> None:
"input_cost_per_token": 3e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
@ -138,7 +137,35 @@ def test_moderation_type_maps_to_chat_mode() -> None:
outcome = sync.compute_sync({}, RECORDED_CATALOG, RECORDED_DOC)
guard = outcome.cost_map["together_ai/meta-llama/Llama-Guard-4-12B"]
assert guard["mode"] == "chat"
assert guard["max_output_tokens"] == 1048576
assert "max_output_tokens" not in guard
def test_output_ceiling_comes_from_the_rule_never_from_context_length() -> None:
glm = next(model for model in RECORDED_CATALOG if model.id == "zai-org/GLM-5.2")
fresh = sync.compute_sync({}, [_chat_model("acme/unreviewed", ctx=1048576), glm], _doc({"x": "2026-01-01"}))
unreviewed = fresh.cost_map["together_ai/acme/unreviewed"]
assert "max_output_tokens" not in unreviewed
assert (unreviewed["max_input_tokens"], unreviewed["max_tokens"]) == (1048576, 1048576)
reviewed = fresh.cost_map["together_ai/zai-org/GLM-5.2"]
assert (reviewed["max_input_tokens"], reviewed["max_output_tokens"], reviewed["max_tokens"]) == (
1048575,
128000,
128000,
)
inflated = {
"together_ai/zai-org/GLM-5.2": {
"input_cost_per_token": 1.4e-06,
"litellm_provider": "together_ai",
"max_input_tokens": 1048575,
"max_output_tokens": 1048575,
"max_tokens": 1048575,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
}
}
corrected = sync.compute_sync(inflated, [glm], _doc({"x": "2026-01-01"}))
assert corrected.cost_map["together_ai/zai-org/GLM-5.2"]["max_output_tokens"] == 128000
assert any("max_output_tokens: 1048575 -> 128000" in line for line in corrected.updated)
def test_docs_removed_but_live_model_stays_live_with_warning() -> None:

View file

@ -108,6 +108,8 @@ def test_together_glm_52_pricing(cost_map: CostMap):
info = cost_map["together_ai/zai-org/GLM-5.2"]
assert info["input_cost_per_token"] == 1.4e-06
assert info["output_cost_per_token"] == 4.4e-06
assert info["max_input_tokens"] == 1048575
assert info["max_output_tokens"] == 128000
assert info["supports_function_calling"] is True
assert info["supports_reasoning"] is True
@ -118,7 +120,7 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap):
assert info["output_cost_per_token"] == 5e-07
assert info["cache_read_input_token_cost"] == 3e-08
assert info["max_input_tokens"] == 1048575
assert info["max_output_tokens"] == 1048575
assert info["max_output_tokens"] == 128000
assert info["supports_function_calling"] is True
assert info["supports_parallel_function_calling"] is True
assert info["supports_prompt_caching"] is True
@ -128,6 +130,18 @@ def test_together_glm_53_flash_pricing_and_capabilities(cost_map: CostMap):
assert info["supports_reasoning"] is True
def test_together_chat_entries_never_carry_context_length_as_output_ceiling(cost_map: CostMap):
inflated = sorted(
model
for model, info in cost_map.items()
if info.get("litellm_provider") == "together_ai"
and info.get("mode") == "chat"
and "max_output_tokens" in info
and info["max_output_tokens"] == info.get("max_input_tokens")
)
assert inflated == []
def test_together_multilingual_e5_embedding_entry(cost_map: CostMap):
info = cost_map["together_ai/intfloat/multilingual-e5-large-instruct"]
assert info["mode"] == "embedding"