fix(oci): route all OpenAI commercial models to maxCompletionTokens

OCI serves OpenAI models (gpt-4.1, gpt-5.1 through 5.5, o-series) that
the litellm catalog doesn't track, so the supports_reasoning lookup
returned False for them and the provider sent maxTokens, which the
reasoning families reject with HTTP 400. With the injected default
maxTokens this broke every request to those models, not just ones with
an explicit max_tokens. Route the whole openai.* vendor prefix to
maxCompletionTokens since OpenAI accepts max_completion_tokens on every
chat model; the openai.gpt-oss-* open weights are served by OCI's own
stack and keep maxTokens. Verified live against gpt-5.2, gpt-5, gpt-4o,
gpt-4.1, gpt-oss-120b, llama-3.3, command-a and grok-3-mini
This commit is contained in:
Federico Kamelhar 2026-06-10 11:47:41 -04:00
parent b57a6dd986
commit f4a030243d
2 changed files with 45 additions and 5 deletions

View file

@ -88,15 +88,20 @@ STREAMING_TIMEOUT = 60 * 5
def _model_uses_max_completion_tokens(model: str) -> bool:
"""Return True for OCI-hosted models that require ``maxCompletionTokens``.
Reasoning models on OCI (e.g. the OpenAI GPT-5 family) reject ``maxTokens``
with HTTP 400 and require ``maxCompletionTokens`` per OpenAI's reasoning-API
convention. Driven by ``supports_reasoning`` in
``model_prices_and_context_window.json`` so new model families are picked
up via a catalog update rather than a code change.
OpenAI commercial models proxied through OCI (``openai.*``) reject
``maxTokens`` with HTTP 400 on the reasoning families (gpt-5.x, o-series)
and accept ``maxCompletionTokens`` everywhere, so route the whole vendor
prefix to it rather than chasing each new release in
``model_prices_and_context_window.json``. The ``openai.gpt-oss-*`` open
weights are served by OCI's own stack and keep ``maxTokens``. Any other
vendor falls back to the catalog's ``supports_reasoning`` flag.
"""
if not model:
return False
name = model[4:] if model.lower().startswith("oci/") else model
lowered = name.lower()
if lowered.startswith("openai."):
return not lowered.startswith("openai.gpt-oss")
return supports_reasoning(model=name, custom_llm_provider="oci")

View file

@ -382,6 +382,41 @@ class TestGpt5MaxCompletionTokens:
assert _model_uses_max_completion_tokens("cohere.command-latest") is False
assert _model_uses_max_completion_tokens("") is False
def test_helper_covers_openai_models_absent_from_catalog(self):
"""OCI keeps adding OpenAI models (gpt-4.1, gpt-5.1..5.5, o-series)
faster than the litellm catalog tracks them. The vendor-prefix rule
must route them to maxCompletionTokens even with no catalog entry,
since OpenAI accepts max_completion_tokens on every chat model while
the reasoning families hard-reject max_tokens."""
import litellm
from litellm.llms.oci.chat.transformation import (
_model_uses_max_completion_tokens,
)
for name in (
"openai.gpt-5.2",
"openai.gpt-4.1",
"openai.o3",
"oci/openai.gpt-5.1-codex",
):
assert f"oci/{name.removeprefix('oci/')}" not in litellm.model_cost
assert _model_uses_max_completion_tokens(name) is True
assert _model_uses_max_completion_tokens("openai.gpt-oss-20b") is False
def test_default_injection_uses_max_completion_tokens_for_uncataloged_gpt(self):
"""Regression: with the injected default maxTokens, a GPT model absent
from the catalog got "maxTokens" on every request and OCI returned 400
("Use 'max_completion_tokens' instead") even when the caller never set
max_tokens."""
from litellm.constants import DEFAULT_OCI_CHAT_MAX_TOKENS
from litellm.llms.oci.chat.transformation import OCIChatConfig, OCIVendors
cfg = OCIChatConfig()
out = cfg._get_optional_params(OCIVendors.GENERIC, {}, model="openai.gpt-5.2")
assert out.get("maxCompletionTokens") == DEFAULT_OCI_CHAT_MAX_TOKENS
assert "maxTokens" not in out
def test_gpt5_routes_max_tokens_to_max_completion_tokens(
self, _register_oci_gpt5_in_catalog
):