From 3ff274207ef9c317e179f19d356a6904c8fa41a3 Mon Sep 17 00:00:00 2001 From: Fede Kamelhar Date: Thu, 11 Jun 2026 06:25:51 -0400 Subject: [PATCH] fix(oci): inject a default maxTokens so omitted max_tokens doesn't truncate responses (#30018) * fix(oci): inject default maxTokens so omitted max_tokens doesn't truncate OCI GenAI applies a tiny server-side maxTokens default (~20 tokens) when the request omits it, so any call that doesn't send max_tokens comes back cut off mid-string with finishReason "length". MLflow judges never send max_tokens, so their JSON responses arrived as unterminated strings and json.loads failed in MLflow's gateway adapter. When no maxTokens/maxCompletionTokens target is set, inject DEFAULT_OCI_CHAT_MAX_TOKENS (env-overridable, defaults 4096), mirroring the Anthropic config's default-max-tokens behaviour. An explicit max_tokens still wins, and reasoning models still route to maxCompletionTokens. Used a fixed default rather than the catalog max_output_tokens because the catalog value is unreliable for some models (grok-4 reports max_output_tokens equal to its context window, not a real output cap, which would risk 400s). Adds TestOCIDefaultMaxTokens covering Cohere and generic injection, the explicit-override case, and the reasoning maxCompletionTokens branch. * test(oci): e2e regression that omitted max_tokens isn't truncated Real-proxy integration test asserting a chat completion that omits max_tokens completes with finish_reason "stop" instead of being cut off at OCI's ~20-token server default. Fails before the maxTokens-default injection (finish_reason "length", ~19 tokens), passes after. * test(oci): update cohere default-params test for injected maxTokens test_cohere_default_parameters asserted no maxTokens was injected, encoding the old behaviour where OCI's ~20-token server default truncated responses. Now that transform_request injects DEFAULT_OCI_CHAT_MAX_TOKENS, assert maxTokens equals that default while the other params (topK/topP/frequencyPenalty) stay pass-through with no hardcoded default. * fix(oci): make DEFAULT_OCI_CHAT_MAX_TOKENS a plain constant Drop the os.getenv override. The env knob was not requested and introducing a new env var forced a cross-repo dependency on litellm-docs (test_env_keys.py validates every referenced env var against the docs table there). A plain 4096 constant keeps the PR self-contained; callers who want a different limit pass max_tokens explicitly per request. * fix(oci): route all OpenAI commercial models to maxCompletionTokens OCI serves OpenAI models (gpt-4.1, gpt-5.1 through 5.5, o-series) that the litellm catalog doesn't track, so the supports_reasoning lookup returned False for them and the provider sent maxTokens, which the reasoning families reject with HTTP 400. With the injected default maxTokens this broke every request to those models, not just ones with an explicit max_tokens. Route the whole openai.* vendor prefix to maxCompletionTokens since OpenAI accepts max_completion_tokens on every chat model; the openai.gpt-oss-* open weights are served by OCI's own stack and keep maxTokens. Verified live against gpt-5.2, gpt-5, gpt-4o, gpt-4.1, gpt-oss-120b, llama-3.3, command-a and grok-3-mini * test(oci): hoist transformation imports and drop unused ones Makes the generic-chat test file ruff-clean: the per-test local imports of OCIChatConfig/OCIVendors shadowed the module-level import (F811) and left it unused (F401), and json plus three OCI type imports were never referenced --- litellm/constants.py | 1 + litellm/llms/oci/chat/transformation.py | 23 +++++++-- .../integration/test_oci_proxy_integration.py | 37 ++++++++++++++ .../oci/chat/test_oci_chat_transformation.py | 40 +++++++++++++++ .../oci/chat/test_oci_cohere_tool_calls.py | 7 +-- .../llms/oci/chat/test_oci_generic_chat.py | 50 ++++++++++++++----- 6 files changed, 137 insertions(+), 21 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 714cfe9b114..ea6d16d583d 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -418,6 +418,7 @@ REPLICATE_POLLING_DELAY_SECONDS = float( DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS = int( os.getenv("DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS", 4096) ) +DEFAULT_OCI_CHAT_MAX_TOKENS = 4096 TOGETHER_AI_4_B = int(os.getenv("TOGETHER_AI_4_B", 4)) TOGETHER_AI_8_B = int(os.getenv("TOGETHER_AI_8_B", 8)) TOGETHER_AI_21_B = int(os.getenv("TOGETHER_AI_21_B", 21)) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index f050f9eea36..627d0ae8072 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -25,6 +25,7 @@ from typing import ( import httpx import litellm +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.litellm_core_utils.logging_utils import track_llm_api_timing from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.llms.custom_httpx.http_handler import ( @@ -87,15 +88,20 @@ STREAMING_TIMEOUT = 60 * 5 def _model_uses_max_completion_tokens(model: str) -> bool: """Return True for OCI-hosted models that require ``maxCompletionTokens``. - Reasoning models on OCI (e.g. the OpenAI GPT-5 family) reject ``maxTokens`` - with HTTP 400 and require ``maxCompletionTokens`` per OpenAI's reasoning-API - convention. Driven by ``supports_reasoning`` in - ``model_prices_and_context_window.json`` so new model families are picked - up via a catalog update rather than a code change. + OpenAI commercial models proxied through OCI (``openai.*``) reject + ``maxTokens`` with HTTP 400 on the reasoning families (gpt-5.x, o-series) + and accept ``maxCompletionTokens`` everywhere, so route the whole vendor + prefix to it rather than chasing each new release in + ``model_prices_and_context_window.json``. The ``openai.gpt-oss-*`` open + weights are served by OCI's own stack and keep ``maxTokens``. Any other + vendor falls back to the catalog's ``supports_reasoning`` flag. """ if not model: return False name = model[4:] if model.lower().startswith("oci/") else model + lowered = name.lower() + if lowered.startswith("openai."): + return not lowered.startswith("openai.gpt-oss") return supports_reasoning(model=name, custom_llm_provider="oci") @@ -451,6 +457,13 @@ class OCIChatConfig(BaseConfig): elif oci_alias in optional_params: selected_params[target] = optional_params[oci_alias] # type: ignore[index] + # OCI's server-side default token cap is tiny (~20 tokens), so an + # omitted max_tokens silently truncates the response mid-string. Most + # callers never send a limit (MLflow judges among them), so inject a + # sane default when one is absent, mirroring litellm's Anthropic config. + if max_tokens_key not in selected_params: + selected_params[max_tokens_key] = DEFAULT_OCI_CHAT_MAX_TOKENS + # OCI expects uppercase reasoning levels (LOW/MEDIUM/HIGH/NONE); OpenAI # clients send lowercase. OpenAI's "disable" maps to OCI's "NONE". if "reasoningEffort" in selected_params: diff --git a/tests/integration/test_oci_proxy_integration.py b/tests/integration/test_oci_proxy_integration.py index 8bfcdd90486..67e40ca814a 100644 --- a/tests/integration/test_oci_proxy_integration.py +++ b/tests/integration/test_oci_proxy_integration.py @@ -272,3 +272,40 @@ def test_model_list_advertises_oci_models(proxy_url: str) -> None: advertised = {row["id"] for row in r.json()["data"]} for expected in CHAT_MODELS + ["oci-embed"]: assert expected in advertised, f"{expected} missing from /v1/models: {advertised}" + + +def test_omitted_max_tokens_not_truncated(proxy_url: str) -> None: + """A request that omits max_tokens completes instead of being cut off. + + Regression for OCI's tiny server-side maxTokens default (~20 tokens): without + an injected default, a request that doesn't set max_tokens came back with + finish_reason "length" after ~19 tokens, so structured outputs (e.g. MLflow + judge JSON) arrived as unterminated strings. The OCI provider now injects a + sane default when the caller omits one. + """ + r = httpx.post( + f"{proxy_url}/v1/chat/completions", + headers=_auth_headers(), + json={ + "model": "oci-cohere-command", + "messages": [ + { + "role": "user", + "content": "In four or five complete sentences, explain why the sky appears blue.", + } + ], + }, + timeout=REQUEST_TIMEOUT_S, + ) + assert r.status_code == 200, f"omitted max_tokens -> {r.status_code}: {r.text}" + body = r.json() + choice = body["choices"][0] + assert ( + choice["finish_reason"] != "length" + ), f"response truncated by token cap: {choice}" + assert choice["finish_reason"] == "stop" + content = choice["message"].get("content") or "" + assert content.strip(), f"empty content: {choice}" + # The ~20-token server default truncated well before this; a complete + # four-to-five sentence answer comfortably exceeds it. + assert body["usage"]["completion_tokens"] > 50, body["usage"] diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index e0911e1ef31..94c8b23def3 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -11,6 +11,7 @@ import litellm sys.path.insert(0, os.path.abspath("../../../../..")) from litellm import ModelResponse +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.llms.oci.chat.transformation import ( OCIChatConfig, OCIRequestWrapper, @@ -104,6 +105,7 @@ class TestOCIChatConfig: "chatRequest": { "apiFormat": "GENERIC", "isStream": False, + "maxTokens": DEFAULT_OCI_CHAT_MAX_TOKENS, "messages": [ { "role": "USER", @@ -956,6 +958,44 @@ class TestOCICohereParamMapping: assert result.get("temperature") == 0.5 +class TestOCIDefaultMaxTokens: + """Regression for OCI's tiny server-side token cap (~20 tokens), which + silently truncated responses mid-string whenever the caller omitted + max_tokens (MLflow judges never send it, so their JSON came back cut off). + transform_request injects DEFAULT_OCI_CHAT_MAX_TOKENS when no limit is + supplied, and leaves an explicit limit untouched.""" + + def _chat_request(self, model: str, optional_params: dict) -> dict: + config = OCIChatConfig() + body = config.transform_request( + model=model, + messages=[{"role": "user", "content": "hi"}], + optional_params={**BASE_OCI_PARAMS, **optional_params}, + litellm_params={}, + headers={}, + ) + return body["chatRequest"] + + @pytest.mark.parametrize( + "model", ["cohere.command-latest", "meta.llama-3.3-70b-instruct"] + ) + def test_default_injected_when_max_tokens_omitted(self, model): + chat_request = self._chat_request(model, {}) + assert chat_request["maxTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS + + @pytest.mark.parametrize( + "model", ["cohere.command-latest", "meta.llama-3.3-70b-instruct"] + ) + def test_explicit_max_tokens_not_overridden(self, model): + chat_request = self._chat_request(model, {"max_tokens": 256}) + assert chat_request["maxTokens"] == 256 + + def test_reasoning_model_defaults_max_completion_tokens(self): + chat_request = self._chat_request("openai.gpt-5", {}) + assert chat_request["maxCompletionTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS + assert "maxTokens" not in chat_request + + class TestOCIReasoningEffort: """ Reasoning-effort handling for GENERIC reasoning models: diff --git a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py index cc914a22eeb..bee1e033502 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py @@ -5,6 +5,7 @@ import json from unittest.mock import patch, MagicMock from litellm import ModelResponse +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.llms.oci.chat.cohere import ( adapt_messages_to_cohere_standard, adapt_tool_definitions_to_cohere_standard, @@ -462,7 +463,8 @@ class TestOCICohereToolCalls: assert "tool_choice" not in supported_params def test_cohere_default_parameters(self): - """Test that Cohere requests do not inject hardcoded defaults — caller supplies all params.""" + """maxTokens is defaulted (OCI's server default truncates at ~20 tokens); + every other param is still pass-through with no hardcoded default.""" config = OCIChatConfig() messages = [{"role": "user", "content": "Hello"}] optional_params = {"oci_compartment_id": TEST_COMPARTMENT_ID} @@ -477,8 +479,7 @@ class TestOCICohereToolCalls: chat_request = transformed_request["chatRequest"] - # No hardcoded defaults injected — only pass through what the user supplies - assert "maxTokens" not in chat_request + assert chat_request["maxTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS assert "topK" not in chat_request assert "topP" not in chat_request assert "frequencyPenalty" not in chat_request diff --git a/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py b/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py index 7583e3bc183..a4a5f111513 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py @@ -2,7 +2,6 @@ Unit tests for litellm/llms/oci/chat/generic.py — error paths and stream handling. """ -import json import pytest from unittest.mock import MagicMock @@ -16,7 +15,12 @@ from litellm.llms.oci.chat.generic import ( handle_generic_response, handle_generic_stream_chunk, ) -from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIStreamWrapper +from litellm.llms.oci.chat.transformation import ( + OCIChatConfig, + OCIStreamWrapper, + OCIVendors, + _model_uses_max_completion_tokens, +) from litellm.llms.oci.common_utils import OCIError # --------------------------------------------------------------------------- @@ -271,7 +275,6 @@ class TestHandleGenericStreamChunk: assert result.choices[0].index == 0 def test_image_content_in_stream_raises(self): - from litellm.types.llms.oci import OCIImageContentPart, OCIImageUrl, OCIMessage chunk = { "apiFormat": "GENERIC", @@ -368,10 +371,6 @@ def _register_oci_gpt5_in_catalog(): class TestGpt5MaxCompletionTokens: def test_helper_detects_gpt5_family(self, _register_oci_gpt5_in_catalog): - from litellm.llms.oci.chat.transformation import ( - _model_uses_max_completion_tokens, - ) - assert _model_uses_max_completion_tokens("openai.gpt-5") is True assert _model_uses_max_completion_tokens("openai.gpt-5-mini") is True assert _model_uses_max_completion_tokens("openai.gpt-5-nano") is True @@ -382,11 +381,40 @@ class TestGpt5MaxCompletionTokens: assert _model_uses_max_completion_tokens("cohere.command-latest") is False assert _model_uses_max_completion_tokens("") is False + def test_helper_covers_openai_models_absent_from_catalog(self): + """OCI keeps adding OpenAI models (gpt-4.1, gpt-5.1..5.5, o-series) + faster than the litellm catalog tracks them. The vendor-prefix rule + must route them to maxCompletionTokens even with no catalog entry, + since OpenAI accepts max_completion_tokens on every chat model while + the reasoning families hard-reject max_tokens.""" + import litellm + + for name in ( + "openai.gpt-5.2", + "openai.gpt-4.1", + "openai.o3", + "oci/openai.gpt-5.1-codex", + ): + assert f"oci/{name.removeprefix('oci/')}" not in litellm.model_cost + assert _model_uses_max_completion_tokens(name) is True + + assert _model_uses_max_completion_tokens("openai.gpt-oss-20b") is False + + def test_default_injection_uses_max_completion_tokens_for_uncataloged_gpt(self): + """Regression: with the injected default maxTokens, a GPT model absent + from the catalog got "maxTokens" on every request and OCI returned 400 + ("Use 'max_completion_tokens' instead") even when the caller never set + max_tokens.""" + from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS + + cfg = OCIChatConfig() + out = cfg._get_optional_params(OCIVendors.GENERIC, {}, model="openai.gpt-5.2") + assert out.get("maxCompletionTokens") == DEFAULT_OCI_CHAT_MAX_TOKENS + assert "maxTokens" not in out + def test_gpt5_routes_max_tokens_to_max_completion_tokens( self, _register_oci_gpt5_in_catalog ): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() # Both shapes optional_params can take after upstream map_openai_params: # 1. openai-side key still present @@ -404,8 +432,6 @@ class TestGpt5MaxCompletionTokens: assert "maxTokens" not in out_b def test_non_gpt5_keeps_max_tokens(self): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() out = cfg._get_optional_params( OCIVendors.GENERIC, @@ -416,8 +442,6 @@ class TestGpt5MaxCompletionTokens: assert "maxCompletionTokens" not in out def test_cohere_reasoning_model_keeps_max_tokens(self): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() out = cfg._get_optional_params( OCIVendors.COHERE,