diff --git a/litellm/constants.py b/litellm/constants.py index 714cfe9b114..ea6d16d583d 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -418,6 +418,7 @@ REPLICATE_POLLING_DELAY_SECONDS = float( DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS = int( os.getenv("DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS", 4096) ) +DEFAULT_OCI_CHAT_MAX_TOKENS = 4096 TOGETHER_AI_4_B = int(os.getenv("TOGETHER_AI_4_B", 4)) TOGETHER_AI_8_B = int(os.getenv("TOGETHER_AI_8_B", 8)) TOGETHER_AI_21_B = int(os.getenv("TOGETHER_AI_21_B", 21)) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index f050f9eea36..627d0ae8072 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -25,6 +25,7 @@ from typing import ( import httpx import litellm +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.litellm_core_utils.logging_utils import track_llm_api_timing from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.llms.custom_httpx.http_handler import ( @@ -87,15 +88,20 @@ STREAMING_TIMEOUT = 60 * 5 def _model_uses_max_completion_tokens(model: str) -> bool: """Return True for OCI-hosted models that require ``maxCompletionTokens``. - Reasoning models on OCI (e.g. the OpenAI GPT-5 family) reject ``maxTokens`` - with HTTP 400 and require ``maxCompletionTokens`` per OpenAI's reasoning-API - convention. Driven by ``supports_reasoning`` in - ``model_prices_and_context_window.json`` so new model families are picked - up via a catalog update rather than a code change. + OpenAI commercial models proxied through OCI (``openai.*``) reject + ``maxTokens`` with HTTP 400 on the reasoning families (gpt-5.x, o-series) + and accept ``maxCompletionTokens`` everywhere, so route the whole vendor + prefix to it rather than chasing each new release in + ``model_prices_and_context_window.json``. The ``openai.gpt-oss-*`` open + weights are served by OCI's own stack and keep ``maxTokens``. Any other + vendor falls back to the catalog's ``supports_reasoning`` flag. """ if not model: return False name = model[4:] if model.lower().startswith("oci/") else model + lowered = name.lower() + if lowered.startswith("openai."): + return not lowered.startswith("openai.gpt-oss") return supports_reasoning(model=name, custom_llm_provider="oci") @@ -451,6 +457,13 @@ class OCIChatConfig(BaseConfig): elif oci_alias in optional_params: selected_params[target] = optional_params[oci_alias] # type: ignore[index] + # OCI's server-side default token cap is tiny (~20 tokens), so an + # omitted max_tokens silently truncates the response mid-string. Most + # callers never send a limit (MLflow judges among them), so inject a + # sane default when one is absent, mirroring litellm's Anthropic config. + if max_tokens_key not in selected_params: + selected_params[max_tokens_key] = DEFAULT_OCI_CHAT_MAX_TOKENS + # OCI expects uppercase reasoning levels (LOW/MEDIUM/HIGH/NONE); OpenAI # clients send lowercase. OpenAI's "disable" maps to OCI's "NONE". if "reasoningEffort" in selected_params: diff --git a/tests/integration/test_oci_proxy_integration.py b/tests/integration/test_oci_proxy_integration.py index 8bfcdd90486..67e40ca814a 100644 --- a/tests/integration/test_oci_proxy_integration.py +++ b/tests/integration/test_oci_proxy_integration.py @@ -272,3 +272,40 @@ def test_model_list_advertises_oci_models(proxy_url: str) -> None: advertised = {row["id"] for row in r.json()["data"]} for expected in CHAT_MODELS + ["oci-embed"]: assert expected in advertised, f"{expected} missing from /v1/models: {advertised}" + + +def test_omitted_max_tokens_not_truncated(proxy_url: str) -> None: + """A request that omits max_tokens completes instead of being cut off. + + Regression for OCI's tiny server-side maxTokens default (~20 tokens): without + an injected default, a request that doesn't set max_tokens came back with + finish_reason "length" after ~19 tokens, so structured outputs (e.g. MLflow + judge JSON) arrived as unterminated strings. The OCI provider now injects a + sane default when the caller omits one. + """ + r = httpx.post( + f"{proxy_url}/v1/chat/completions", + headers=_auth_headers(), + json={ + "model": "oci-cohere-command", + "messages": [ + { + "role": "user", + "content": "In four or five complete sentences, explain why the sky appears blue.", + } + ], + }, + timeout=REQUEST_TIMEOUT_S, + ) + assert r.status_code == 200, f"omitted max_tokens -> {r.status_code}: {r.text}" + body = r.json() + choice = body["choices"][0] + assert ( + choice["finish_reason"] != "length" + ), f"response truncated by token cap: {choice}" + assert choice["finish_reason"] == "stop" + content = choice["message"].get("content") or "" + assert content.strip(), f"empty content: {choice}" + # The ~20-token server default truncated well before this; a complete + # four-to-five sentence answer comfortably exceeds it. + assert body["usage"]["completion_tokens"] > 50, body["usage"] diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index e0911e1ef31..94c8b23def3 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -11,6 +11,7 @@ import litellm sys.path.insert(0, os.path.abspath("../../../../..")) from litellm import ModelResponse +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.llms.oci.chat.transformation import ( OCIChatConfig, OCIRequestWrapper, @@ -104,6 +105,7 @@ class TestOCIChatConfig: "chatRequest": { "apiFormat": "GENERIC", "isStream": False, + "maxTokens": DEFAULT_OCI_CHAT_MAX_TOKENS, "messages": [ { "role": "USER", @@ -956,6 +958,44 @@ class TestOCICohereParamMapping: assert result.get("temperature") == 0.5 +class TestOCIDefaultMaxTokens: + """Regression for OCI's tiny server-side token cap (~20 tokens), which + silently truncated responses mid-string whenever the caller omitted + max_tokens (MLflow judges never send it, so their JSON came back cut off). + transform_request injects DEFAULT_OCI_CHAT_MAX_TOKENS when no limit is + supplied, and leaves an explicit limit untouched.""" + + def _chat_request(self, model: str, optional_params: dict) -> dict: + config = OCIChatConfig() + body = config.transform_request( + model=model, + messages=[{"role": "user", "content": "hi"}], + optional_params={**BASE_OCI_PARAMS, **optional_params}, + litellm_params={}, + headers={}, + ) + return body["chatRequest"] + + @pytest.mark.parametrize( + "model", ["cohere.command-latest", "meta.llama-3.3-70b-instruct"] + ) + def test_default_injected_when_max_tokens_omitted(self, model): + chat_request = self._chat_request(model, {}) + assert chat_request["maxTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS + + @pytest.mark.parametrize( + "model", ["cohere.command-latest", "meta.llama-3.3-70b-instruct"] + ) + def test_explicit_max_tokens_not_overridden(self, model): + chat_request = self._chat_request(model, {"max_tokens": 256}) + assert chat_request["maxTokens"] == 256 + + def test_reasoning_model_defaults_max_completion_tokens(self): + chat_request = self._chat_request("openai.gpt-5", {}) + assert chat_request["maxCompletionTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS + assert "maxTokens" not in chat_request + + class TestOCIReasoningEffort: """ Reasoning-effort handling for GENERIC reasoning models: diff --git a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py index cc914a22eeb..bee1e033502 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py @@ -5,6 +5,7 @@ import json from unittest.mock import patch, MagicMock from litellm import ModelResponse +from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS from litellm.llms.oci.chat.cohere import ( adapt_messages_to_cohere_standard, adapt_tool_definitions_to_cohere_standard, @@ -462,7 +463,8 @@ class TestOCICohereToolCalls: assert "tool_choice" not in supported_params def test_cohere_default_parameters(self): - """Test that Cohere requests do not inject hardcoded defaults — caller supplies all params.""" + """maxTokens is defaulted (OCI's server default truncates at ~20 tokens); + every other param is still pass-through with no hardcoded default.""" config = OCIChatConfig() messages = [{"role": "user", "content": "Hello"}] optional_params = {"oci_compartment_id": TEST_COMPARTMENT_ID} @@ -477,8 +479,7 @@ class TestOCICohereToolCalls: chat_request = transformed_request["chatRequest"] - # No hardcoded defaults injected — only pass through what the user supplies - assert "maxTokens" not in chat_request + assert chat_request["maxTokens"] == DEFAULT_OCI_CHAT_MAX_TOKENS assert "topK" not in chat_request assert "topP" not in chat_request assert "frequencyPenalty" not in chat_request diff --git a/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py b/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py index 7583e3bc183..a4a5f111513 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_generic_chat.py @@ -2,7 +2,6 @@ Unit tests for litellm/llms/oci/chat/generic.py — error paths and stream handling. """ -import json import pytest from unittest.mock import MagicMock @@ -16,7 +15,12 @@ from litellm.llms.oci.chat.generic import ( handle_generic_response, handle_generic_stream_chunk, ) -from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIStreamWrapper +from litellm.llms.oci.chat.transformation import ( + OCIChatConfig, + OCIStreamWrapper, + OCIVendors, + _model_uses_max_completion_tokens, +) from litellm.llms.oci.common_utils import OCIError # --------------------------------------------------------------------------- @@ -271,7 +275,6 @@ class TestHandleGenericStreamChunk: assert result.choices[0].index == 0 def test_image_content_in_stream_raises(self): - from litellm.types.llms.oci import OCIImageContentPart, OCIImageUrl, OCIMessage chunk = { "apiFormat": "GENERIC", @@ -368,10 +371,6 @@ def _register_oci_gpt5_in_catalog(): class TestGpt5MaxCompletionTokens: def test_helper_detects_gpt5_family(self, _register_oci_gpt5_in_catalog): - from litellm.llms.oci.chat.transformation import ( - _model_uses_max_completion_tokens, - ) - assert _model_uses_max_completion_tokens("openai.gpt-5") is True assert _model_uses_max_completion_tokens("openai.gpt-5-mini") is True assert _model_uses_max_completion_tokens("openai.gpt-5-nano") is True @@ -382,11 +381,40 @@ class TestGpt5MaxCompletionTokens: assert _model_uses_max_completion_tokens("cohere.command-latest") is False assert _model_uses_max_completion_tokens("") is False + def test_helper_covers_openai_models_absent_from_catalog(self): + """OCI keeps adding OpenAI models (gpt-4.1, gpt-5.1..5.5, o-series) + faster than the litellm catalog tracks them. The vendor-prefix rule + must route them to maxCompletionTokens even with no catalog entry, + since OpenAI accepts max_completion_tokens on every chat model while + the reasoning families hard-reject max_tokens.""" + import litellm + + for name in ( + "openai.gpt-5.2", + "openai.gpt-4.1", + "openai.o3", + "oci/openai.gpt-5.1-codex", + ): + assert f"oci/{name.removeprefix('oci/')}" not in litellm.model_cost + assert _model_uses_max_completion_tokens(name) is True + + assert _model_uses_max_completion_tokens("openai.gpt-oss-20b") is False + + def test_default_injection_uses_max_completion_tokens_for_uncataloged_gpt(self): + """Regression: with the injected default maxTokens, a GPT model absent + from the catalog got "maxTokens" on every request and OCI returned 400 + ("Use 'max_completion_tokens' instead") even when the caller never set + max_tokens.""" + from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS + + cfg = OCIChatConfig() + out = cfg._get_optional_params(OCIVendors.GENERIC, {}, model="openai.gpt-5.2") + assert out.get("maxCompletionTokens") == DEFAULT_OCI_CHAT_MAX_TOKENS + assert "maxTokens" not in out + def test_gpt5_routes_max_tokens_to_max_completion_tokens( self, _register_oci_gpt5_in_catalog ): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() # Both shapes optional_params can take after upstream map_openai_params: # 1. openai-side key still present @@ -404,8 +432,6 @@ class TestGpt5MaxCompletionTokens: assert "maxTokens" not in out_b def test_non_gpt5_keeps_max_tokens(self): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() out = cfg._get_optional_params( OCIVendors.GENERIC, @@ -416,8 +442,6 @@ class TestGpt5MaxCompletionTokens: assert "maxCompletionTokens" not in out def test_cohere_reasoning_model_keeps_max_tokens(self): - from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors - cfg = OCIChatConfig() out = cfg._get_optional_params( OCIVendors.COHERE,