diff --git a/litellm/llms/azure/responses/transformation.py b/litellm/llms/azure/responses/transformation.py index 2a82b42df7b..7e18af86f7d 100644 --- a/litellm/llms/azure/responses/transformation.py +++ b/litellm/llms/azure/responses/transformation.py @@ -38,6 +38,15 @@ class AzureOpenAIResponsesAPIConfig(OpenAIResponsesAPIConfig): def _effort_resolves_to_none(model: str, effort: str | None) -> bool: return AzureOpenAIGPT5Config.effort_resolves_to_none(model, effort) + @staticmethod + def _is_unsupported_reasoning_effort(model: str, effort: str | None) -> bool: + """Read the effort capability from the Azure map entry, like the sibling checks above. + + The inherited OpenAI lookup would resolve a bare Azure deployment name (e.g. ``gpt-5.4``) + against the OpenAI entry, which can explicitly disable an effort Azure supports. + """ + return AzureOpenAIGPT5Config.is_reasoning_effort_unsupported(model, effort) + def get_supported_openai_params(self, model: str) -> list: """ Azure Responses API does not support context_management (compaction). diff --git a/litellm/llms/openai/chat/gpt_5_transformation.py b/litellm/llms/openai/chat/gpt_5_transformation.py index bf6b52225f2..b795d77efb8 100644 --- a/litellm/llms/openai/chat/gpt_5_transformation.py +++ b/litellm/llms/openai/chat/gpt_5_transformation.py @@ -163,6 +163,15 @@ class OpenAIGPT5Config(OpenAIGPTConfig): key=f"supports_{level}_reasoning_effort", ) + @classmethod + def is_reasoning_effort_unsupported(cls, model: str, level: str | None) -> bool: + """Return whether GPT-5 capability metadata rejects this effort level.""" + if level == "xhigh": + return not cls._supports_reasoning_effort_level(model, level) + if level in ("minimal", "low"): + return cls._is_reasoning_effort_level_explicitly_disabled(model, level) + return False + @classmethod def effort_resolves_to_none(cls, model: str, effective_effort: str | None) -> bool: """Whether this request's reasoning effort ends up as "none", which is the single diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 6c1d8698652..0d523507a2a 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -150,6 +150,13 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): return OpenAIGPT5Config.effort_resolves_to_none(model, effort) + @staticmethod + def _is_unsupported_reasoning_effort(model: str, effort: str | None) -> bool: + """Apply the GPT-5 reasoning-effort capability flags used by chat completions.""" + from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config + + return OpenAIGPT5Config.is_reasoning_effort_unsupported(model, effort) + @staticmethod def _supports_reasoning_param(model: str) -> bool: from litellm.utils import _get_model_info_helper @@ -214,7 +221,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): GPT-5 models have restrictions on temperature and top_p (only temperature=1 is accepted, and top_p is rejected, unless reasoning.effort resolves to - 'none' on models that support it). + 'none' on models that support it). Reasoning effort values the model map + disables are rejected or dropped first, mirroring chat completions. Apply the same validation used by the chat completions path. """ params: Final = dict(response_api_optional_params) @@ -243,8 +251,27 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): if self._is_gpt_5_model(model=lookup_name): reasoning: Final = params.get("reasoning") or {} effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None + unsupported_effort: Final = self._is_unsupported_reasoning_effort(lookup_name, effort) + should_drop_effort: Final = unsupported_effort and (drop_params or litellm.drop_params) + + if unsupported_effort: + if should_drop_effort: + if isinstance(reasoning, dict): + updated_reasoning: Final = reasoning.copy() + updated_reasoning.pop("effort", None) + if updated_reasoning: + params["reasoning"] = updated_reasoning + else: + params.pop("reasoning", None) + else: + raise litellm.UnsupportedParamsError( + message=f"reasoning.effort={effort} is not supported for this model.", + status_code=400, + ) + + effective_effort: Final = None if should_drop_effort else effort supports_none: Final = self._supports_reasoning_effort_none(model=lookup_name) - effort_is_none: Final = supports_none and self._effort_resolves_to_none(lookup_name, effort) + effort_is_none: Final = supports_none and self._effort_resolves_to_none(lookup_name, effective_effort) temperature: Final = params.get("temperature") if temperature is not None and temperature != 1: diff --git a/litellm/proxy/auth/auth_object_prefetch.py b/litellm/proxy/auth/auth_object_prefetch.py index 52e26e885c9..503223e7d87 100644 --- a/litellm/proxy/auth/auth_object_prefetch.py +++ b/litellm/proxy/auth/auth_object_prefetch.py @@ -14,7 +14,6 @@ from pydantic import BaseModel, TypeAdapter, ValidationError from litellm._logging import verbose_proxy_logger from litellm.caching.redis_cache import RedisCache -from litellm.constants import DEFAULT_IN_MEMORY_TTL from litellm.models.organization import LiteLLM_OrganizationTable from litellm.models.team import LiteLLM_TeamTableCachedObj from litellm.models.team_membership import LiteLLM_TeamMembership @@ -190,14 +189,17 @@ def _iter_entries(refs: AuthObjectRefs, management_ttl: float) -> Iterator[_Cach None, ) if refs.organization_id is not None: + # Organization entries use the management TTL like every other prefetched object: a + # 5s fuse expires before slow requests reach the getters, sending them back to + # the DB the prefetch was meant to spare. yield _CacheEntry( - f"org_id:{refs.organization_id}", "organization_row", LiteLLM_OrganizationTable, DEFAULT_IN_MEMORY_TTL + f"org_id:{refs.organization_id}", "organization_row", LiteLLM_OrganizationTable, management_ttl ) yield _CacheEntry( f"org_id:{refs.organization_id}:with_budget", "organization_row", LiteLLM_OrganizationTable, - DEFAULT_IN_MEMORY_TTL, + management_ttl, ) if refs.project_id is not None: yield _CacheEntry(f"project_id:{refs.project_id}", "project_row", LiteLLM_ProjectTableCachedObj, management_ttl) diff --git a/tests/test_litellm/proxy/auth/test_auth_object_prefetch.py b/tests/test_litellm/proxy/auth/test_auth_object_prefetch.py index 0fd0dda3017..58761e270a0 100644 --- a/tests/test_litellm/proxy/auth/test_auth_object_prefetch.py +++ b/tests/test_litellm/proxy/auth/test_auth_object_prefetch.py @@ -181,8 +181,8 @@ async def test_cold_regime_is_one_mget_one_query_and_the_getters_never_touch_io_ assert sets == sorted( [ f"SET {TEAM_ID}_{USER_ID} ttl=5", - f"SET org_id:{ORG_ID} ttl=5", - f"SET org_id:{ORG_ID}:with_budget ttl=5", + f"SET org_id:{ORG_ID} ttl=60", + f"SET org_id:{ORG_ID}:with_budget ttl=60", f"SET {USER_ID} ttl=60", f"SET team_id:{TEAM_ID} ttl=60", f"SET team_membership:{USER_ID}:{TEAM_ID} ttl=None", diff --git a/tests/unit/llms/openai/responses/test_reasoning_effort_capability.py b/tests/unit/llms/openai/responses/test_reasoning_effort_capability.py new file mode 100644 index 00000000000..585a0ea3b5f --- /dev/null +++ b/tests/unit/llms/openai/responses/test_reasoning_effort_capability.py @@ -0,0 +1,86 @@ +import re + +import pytest + +import litellm +from litellm.llms.azure.responses.transformation import AzureOpenAIResponsesAPIConfig +from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig + + +@pytest.mark.parametrize("effort", ["minimal", "low"]) +def test_rejects_explicitly_unsupported_lower_reasoning_effort(effort: str) -> None: + config = OpenAIResponsesAPIConfig() + + with pytest.raises(litellm.UnsupportedParamsError, match=f"reasoning.effort={effort}"): + config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": effort}}, + model="gpt-5.5-pro", + drop_params=False, + ) + + +def test_keeps_supported_reasoning_effort() -> None: + config = OpenAIResponsesAPIConfig() + + result = config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": "medium"}}, + model="gpt-5.5-pro", + drop_params=False, + ) + + assert result["reasoning"] == {"effort": "medium"} + + +def test_drop_params_removes_only_unsupported_effort() -> None: + config = OpenAIResponsesAPIConfig() + + result = config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal", "summary": "detailed"}}, + model="gpt-5.5-pro", + drop_params=True, + ) + + assert result["reasoning"] == {"summary": "detailed"} + + +def test_drop_params_revalidates_temperature_after_effort_drop() -> None: + """Dropping an unsupported effort revalidates temperature against the emptied effort. + + `gpt-5.1` really declares `xhigh` unsupported while supporting (and defaulting to) + `none`, so no capability stubbing is needed: the dropped effort resolves to `none` + and a valid non-default temperature survives. + """ + config = OpenAIResponsesAPIConfig() + + result = config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": "xhigh"}, "temperature": 0.5}, + model="gpt-5.1", + drop_params=True, + ) + + assert "reasoning" not in result + assert result["temperature"] == 0.5 + + +def test_azure_config_uses_azure_capability_map_for_effort() -> None: + """A bare Azure deployment name must resolve against the Azure map entry. + + OpenAI's `gpt-5.4` entry explicitly disables `minimal` while Azure's enables it; + the OpenAI lookup would wrongly reject an effort Azure supports. + """ + openai_config = OpenAIResponsesAPIConfig() + with pytest.raises(litellm.UnsupportedParamsError, match=re.escape("reasoning.effort=minimal")): + openai_config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal"}}, + model="gpt-5.4", + drop_params=False, + ) + + azure_config = AzureOpenAIResponsesAPIConfig() + result = azure_config.map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal"}}, + model="gpt-5.4", + drop_params=False, + ) + + assert result["reasoning"] == {"effort": "minimal"}