From b9e46cbdb75675decb93e5ddd340f979145d7388 Mon Sep 17 00:00:00 2001 From: Darien Kindlund Date: Fri, 24 Apr 2026 11:48:39 -0400 Subject: [PATCH 1/7] fix(adapters,vertex): pass output_config through to backends that accept it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves the silent strip of Anthropic Structured Outputs across the Vertex AI Claude transformation paths and the Anthropic-adapter re-merge. Consolidates and supersedes four stalled community PRs addressing overlapping aspects of the same root bug: - #23475 (Vertex AI Claude blanket-strip removal) - #23396 (Vertex AI Claude conditional passthrough) - #23706 (Anthropic adapter exclude output_config from non-Anthropic backends) - #22727 (Anthropic adapter strip output_config for non-Anthropic backends) Closes / addresses: #23380 (Vertex AI Claude output_config drop), related: #26423, #25079, #24549, #25971, #25957, #26163, #24856. What was broken --------------- * Vertex AI Claude paths called ``data.pop("output_config")`` and ``data.pop("output_format")`` unconditionally even when Vertex accepted those fields. Callers asking for Structured Outputs got a 200 with prose and never knew the schema constraints had been silently dropped (often masked for months by permissive fallback parsers). * The ``/v1/messages`` -> ``/chat/completions`` adapter (``LiteLLMMessagesToCompletionTransformationHandler``) re-merged the raw Anthropic-shaped ``output_config`` into ``completion_kwargs`` AFTER the translator already mapped its meaningful parts to ``response_format`` / ``reasoning_effort``. Non-Anthropic backends (Azure OpenAI, Fireworks, Bedrock Nova, etc.) then 400'd with "Extra inputs are not permitted". Approach -------- Vertex AI Claude (chat-completion + experimental_pass_through paths): Replace the unconditional pop with a sanitizer ``_sanitize_vertex_anthropic_output_params`` that strips only the Vertex-unsupported keys (today: ``effort``) from ``output_config`` while forwarding ``format`` and the legacy top-level ``output_format``. Defensive: non-dict ``output_config`` values are dropped to avoid sending malformed payloads downstream. Greptile P1 from PR #23396 addressed: when ``output_config`` carries both ``format`` and ``effort``, the prior conditional pass-through forwarded ``effort`` and reproduced the 400. The new helper filters per-key. Anthropic ``/v1/messages`` adapter: Add ``output_config`` to a named module-level constant ``ANTHROPIC_ONLY_REQUEST_KEYS`` and wire it into ``excluded_keys`` so the post-translation re-merge skips re-adding the raw key. This fixes the 400 on non-Anthropic backends and avoids the conflicting duplicate (``response_format`` + raw ``output_config``) on Anthropic-family backends. Greptile P2 from PR #23706 addressed: the constant gives reviewers one grep target instead of an inline literal that silently grows. Greptile P2 from PR #22727 addressed: ``extra_kwargs or {}`` is replaced with explicit ``is None`` checks so empty-dict callers no longer skip the fallback path. Tests ----- * tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/ test_vertex_ai_partner_models_anthropic_transformation.py: - 5 new/updated cases plus a direct unit test for ``_sanitize_vertex_anthropic_output_params``. - Updated ``test_vertex_ai_claude_sonnet_4_5_structured_output_fix`` so its mock-injected ``output_format`` is asserted to FLOW THROUGH (the original test asserted the now-buggy strip behavior). * tests/test_litellm/llms/anthropic/experimental_pass_through/ adapters/test_handler_output_config_passthrough.py (new): - Constant export sanity, output_config strip with ``effort`` only, output_config strip with ``format`` only, regression guard that unrelated extras still flow, explicit-empty-dict path, and the ``extra_kwargs=None`` no-crash path. Test-quality fixes incorporated from Greptile review on the superseded PRs: * No ``inspect.getsource`` source-text assertions (PR #24114 / #23475). * ``sys.path`` insertion is anchored to ``__file__`` (PR #23706). * Assertion messages are positional, not tuple (PR #24114-class bug). * No ``or {}`` masking explicit empty dicts in helper signatures (PR #22727). Verified locally: 26/26 pass with this commit. The new tests fail (or fail to import) on ``main`` without it. Out of scope ------------ * The ``max_tokens`` capping logic from PR #22727 — independent concern, deserves its own PR with a focused test plan. * Architectural rework of the ``excluded_keys`` mechanism (Greptile P2 on PR #23706 noted point-fix growth). The named constant gives maintainers a clear place to extend; a registry-based approach would be a follow-up. Co-Authored-By: netbrah Co-Authored-By: s-zx Co-Authored-By: invoicepulse Co-Authored-By: cfdude Co-Authored-By: Claude Opus 4.7 (1M context) --- .../adapters/handler.py | 36 ++- .../transformation.py | 13 +- .../anthropic/transformation.py | 50 +++- .../test_handler_output_config_passthrough.py | 167 ++++++++++++ ...partner_models_anthropic_transformation.py | 243 +++++++++++++----- 5 files changed, 431 insertions(+), 78 deletions(-) create mode 100644 tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index d16f5afb45c..10455825e41 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -27,6 +27,16 @@ from litellm.utils import get_model_info if TYPE_CHECKING: pass + +# Anthropic-only fields that the translator above already maps into the +# OpenAI-format completion_kwargs (output_config → reasoning_effort / +# response_format, etc.). They must be filtered out of the raw +# extra_kwargs re-merge below or non-Anthropic backends reject the call +# with 400 "Extra inputs are not permitted". Add new entries here when +# extending AnthropicMessagesRequestOptionalParams with another Anthropic- +# specific key. +ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset({"output_config"}) + ######################################################## # init adapter ANTHROPIC_ADAPTER = AnthropicAdapter() @@ -202,8 +212,12 @@ class LiteLLMMessagesToCompletionTransformationHandler: request_data["output_format"] = output_format # Extract output_config from extra_kwargs so the translator can use it - # (e.g. output_config.effort for adaptive thinking → reasoning_effort) - extra_kwargs = extra_kwargs or {} + # (e.g. output_config.effort for adaptive thinking → reasoning_effort, + # output_config.format → response_format for structured outputs). + # Use explicit None check rather than `or {}` so an explicit empty dict + # caller-passed argument is preserved (matters for tests that drive + # the fallback inference path). + extra_kwargs = extra_kwargs if extra_kwargs is not None else {} if "output_config" in extra_kwargs: request_data["output_config"] = extra_kwargs["output_config"] @@ -225,8 +239,22 @@ class LiteLLMMessagesToCompletionTransformationHandler: "include_usage": True, } - excluded_keys = {"anthropic_messages"} - extra_kwargs = extra_kwargs or {} + # Keys that must NOT be forwarded as raw extras into the OpenAI-format + # ``completion_kwargs`` after translation. The translator above has + # already consumed the meaningful parts of these inputs (e.g. + # ``output_config.format`` → ``response_format``, ``output_config.effort`` + # → ``reasoning_effort`` for non-Claude targets). Re-adding the raw + # Anthropic-shaped key here causes 400 "Extra inputs are not permitted" + # on non-Anthropic backends (Azure OpenAI, Fireworks, Bedrock Nova, + # etc.) and is silently lossy on Anthropic-family targets, which would + # see the translated key ``response_format`` AND a duplicate, conflicting + # ``output_config``. + # + # Maintainability: when adding a new Anthropic-only request param to + # ``AnthropicMessagesRequestOptionalParams``, also extend + # ``ANTHROPIC_ONLY_REQUEST_KEYS`` here so it doesn't silently leak. + excluded_keys = ANTHROPIC_ONLY_REQUEST_KEYS | {"anthropic_messages"} + extra_kwargs = extra_kwargs if extra_kwargs is not None else {} for key, value in extra_kwargs.items(): if ( key == "litellm_logging_obj" diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 5c3bbf61ee2..9080ac02330 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -13,6 +13,7 @@ from litellm.types.llms.vertex_ai import VertexPartnerProvider from litellm.types.router import GenericLiteLLMParams from ....vertex_llm_base import VertexBase +from ..transformation import _sanitize_vertex_anthropic_output_params class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, VertexBase): @@ -158,12 +159,10 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert "model", None ) # do not pass model in request body to vertex ai - anthropic_messages_request.pop( - "output_format", None - ) # do not pass output_format in request body to vertex ai - vertex ai does not support output_format as yet - - anthropic_messages_request.pop( - "output_config", None - ) # do not pass output_config in request body to vertex ai - vertex ai does not support output_config + # Vertex AI Claude accepts ``output_config.format`` (structured outputs) + # and ``output_format``, but rejects ``output_config.effort`` with 400 + # "Extra inputs are not permitted". Sanitize in place so the supported + # bits flow through. + _sanitize_vertex_anthropic_output_params(anthropic_messages_request) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 504914c4796..a9dea6646ff 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -11,6 +11,45 @@ from litellm.types.utils import ModelResponse from ....anthropic.chat.transformation import AnthropicConfig +# Keys inside ``output_config`` that Vertex AI Claude does not accept. +# Today only ``effort`` triggers "Extra inputs are not permitted"; add new +# entries here as Vertex parity drifts. Keep this list narrow — anything +# Vertex DOES accept (e.g. ``format`` for structured outputs) must be +# preserved so callers can rely on Anthropic-native features. +_VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS = frozenset({"effort"}) + + +def _sanitize_vertex_anthropic_output_params(data: dict) -> None: + """ + Strip Vertex-unsupported keys from ``output_config`` / ``output_format`` + in-place; forward whatever remains. + + Behavior: + * ``output_config`` containing only unsupported keys (e.g. ``effort`` + alone) is removed entirely so the request body has no empty dict. + * ``output_config`` containing a mix of supported + unsupported keys has + the unsupported subset filtered out and the rest forwarded. + * ``output_config`` that is supported in full passes through unchanged. + * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). + * Non-dict values for ``output_config`` are dropped to avoid sending + malformed payloads downstream. + """ + output_config = data.get("output_config") + if output_config is None: + return + if not isinstance(output_config, dict): + data.pop("output_config", None) + return + sanitized = { + k: v + for k, v in output_config.items() + if k not in _VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS + } + if sanitized: + data["output_config"] = sanitized + else: + data.pop("output_config", None) + class VertexAIError(Exception): def __init__(self, status_code, message): @@ -105,11 +144,12 @@ class VertexAIAnthropicConfig(AnthropicConfig): data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter - # VertexAI doesn't support output_format parameter, remove it if present - data.pop("output_format", None) - - # VertexAI doesn't support output_config parameter, remove it if present - data.pop("output_config", None) + # Vertex AI Claude accepts ``output_config.format`` (structured outputs / + # JSON Schema) but NOT ``output_config.effort`` — sending ``effort`` to + # Vertex returns 400 "Extra inputs are not permitted". Sanitize in place: + # forward the structured-output bits, drop the unsupported keys. + # Same treatment for the legacy top-level ``output_format`` field. + _sanitize_vertex_anthropic_output_params(data) tools = optional_params.get("tools") tool_search_used = self.is_tool_search_used(tools) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py new file mode 100644 index 00000000000..50e5fd884e9 --- /dev/null +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py @@ -0,0 +1,167 @@ +""" +Regression tests for output_config passthrough through the Anthropic +``/v1/messages`` → ``/chat/completions`` adapter. + +Background — what was broken: +* When a client sent ``output_config`` to ``/v1/messages`` and the request + was routed to a non-Anthropic backend (Azure OpenAI, Fireworks, Bedrock + Nova, etc.), the adapter forwarded the raw Anthropic-shaped ``output_config`` + field as-is into the OpenAI-format ``completion_kwargs``. The non-Anthropic + backend then rejected the request with 400 "Extra inputs are not permitted". +* The translator above the re-merge already extracts the meaningful parts of + ``output_config`` (``format`` → ``response_format``, ``effort`` → + ``reasoning_effort`` for non-Claude targets), so re-adding the raw key was + always either redundant (Anthropic-family) or harmful (non-Anthropic). + +Tests cover (consolidating PRs #23706 and #22727): +1. ``output_config`` is excluded from the post-translation re-merge. +2. ``ANTHROPIC_ONLY_REQUEST_KEYS`` constant is exported and contains + ``output_config`` so future maintainers know where to extend it. +3. The translator-extracted fields (``response_format`` / ``reasoning_effort``) + are still present after the strip — the strip removes only the raw + Anthropic-shaped duplicate. +4. Helper-level coverage for empty ``extra_kwargs`` (PR #22727 Greptile P2 — + the original ``or {}`` pattern silently substituted a default and prevented + the fallback inference path from being exercised). +""" + +import os +import sys +from unittest.mock import MagicMock, patch + +import pytest + +# Anchor sys.path to this file's location — not the working-directory-relative +# pattern Greptile flagged on PR #23706. Resolves correctly regardless of +# where pytest is invoked from. +sys.path.insert( + 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")) +) + +from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( + ANTHROPIC_ONLY_REQUEST_KEYS, + LiteLLMMessagesToCompletionTransformationHandler, +) + +MESSAGES = [{"role": "user", "content": "hello"}] + + +def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): + """ + Drive ``_prepare_completion_kwargs`` with the minimum scaffolding needed. + + Uses an explicit-None check on ``extra_kwargs`` so callers can test the + falsy-empty-dict path. The fallback ``or {}`` pattern PR #22727 used here + masked the no-extra-kwargs case from ever exercising the test's intent. + """ + return LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( + max_tokens=overrides.get("max_tokens", 1024), + messages=overrides.get("messages", MESSAGES), + model=model, + metadata=None, + stop_sequences=None, + stream=False, + system=None, + temperature=None, + thinking=None, + tool_choice=None, + tools=None, + top_k=None, + top_p=None, + output_format=None, + extra_kwargs=extra_kwargs, + ) + + +class TestAnthropicOnlyRequestKeysExport: + """The exclusion list must be a public, named constant for maintainability — + Greptile P2 on PR #23706: ``excluded_keys`` was silently growing as a + point-fix pattern. A named module-level constant gives reviewers a single + grep target when extending Anthropic-only fields.""" + + def test_constant_exposed(self): + assert isinstance(ANTHROPIC_ONLY_REQUEST_KEYS, frozenset) + + def test_contains_output_config(self): + assert "output_config" in ANTHROPIC_ONLY_REQUEST_KEYS + + +class TestOutputConfigStrippedFromCompletionKwargs: + """``output_config`` must not survive the post-translation re-merge into + ``completion_kwargs`` regardless of the target provider — the translator + has already consumed its meaningful parts.""" + + def test_output_config_with_effort_is_stripped(self): + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": {"effort": "high"}, + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + + # Returns (completion_kwargs, original_messages, ...) — first element + # is the dict we care about. + completion_kwargs = result[0] if isinstance(result, tuple) else result + assert "output_config" not in completion_kwargs, ( + "Raw output_config must not be forwarded — non-Anthropic backends " + "reject it with 400 'Extra inputs are not permitted'" + ) + + def test_output_config_with_format_is_stripped_format_already_translated(self): + """Even when ``output_config`` carries useful structured-output info, + the raw key must be excluded — the translator above has already mapped + ``output_config.format`` to ``response_format`` (the OpenAI-shaped key + the downstream backend understands).""" + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": { + "format": {"type": "json_schema", "schema": {"type": "object"}} + }, + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "output_config" not in completion_kwargs + + def test_other_extra_kwargs_still_passed_through(self): + """Regression guard: the strip must be narrow. Unrelated fields like + ``api_key`` / ``timeout`` continue to flow through.""" + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": {"effort": "high"}, + "timeout": 30, + "user": "end-user-123", + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "output_config" not in completion_kwargs + assert completion_kwargs.get("timeout") == 30 + assert completion_kwargs.get("user") == "end-user-123" + + +class TestEmptyExtraKwargsPath: + """Greptile P2 on PR #22727: ``extra_kwargs or {default}`` substitutes a + default for an explicitly-passed empty dict, hiding the no-extra-kwargs + path. The new explicit-None pattern lets ``extra_kwargs={}`` reach the + code under test as written.""" + + def test_explicit_empty_dict_does_not_substitute_default(self): + # Explicit empty dict must be honored — not silently replaced with a + # default that adds back a custom_llm_provider this test wants absent. + result = _call_prepare(extra_kwargs={}) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + # No output_config because nothing supplied it. + assert "output_config" not in completion_kwargs + + def test_none_extra_kwargs_handled_safely(self): + """The signature documents ``extra_kwargs: Optional[Dict] = None``; + passing None must not crash with KeyError or AttributeError.""" + result = _call_prepare(extra_kwargs=None) + # Just exercising the path; assert no exception and we get back a + # dict-like result. + completion_kwargs = result[0] if isinstance(result, tuple) else result + assert isinstance(completion_kwargs, dict) diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 79fc66a74b8..376c48d9e95 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -203,12 +203,15 @@ def test_vertex_ai_anthropic_structured_output_header_not_added(): def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): """ Test fix for issue #18625: Claude Sonnet 4.5 on VertexAI should use tool-based - structured outputs instead of output_format parameter. + structured outputs when ``response_format`` is supplied via the OpenAI-compat + interface (``map_openai_params``). This test verifies that: - 1. Claude Sonnet 4.5 uses tool-based structured outputs on VertexAI - 2. output_format parameter is removed from the final request - 3. The fix prevents "Extra inputs are not permitted" error + 1. Claude Sonnet 4.5 uses tool-based structured outputs when ``response_format`` + is given to the OpenAI-compat path (the path that triggered #18625). + 2. ``output_format`` is forwarded to Vertex AI when present — Vertex now + accepts the field; the prior blanket-strip behavior was the silent drop + of Anthropic Structured Outputs that this PR fixes. """ config = VertexAIAnthropicConfig() @@ -294,11 +297,15 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): headers={}, ) - # Verify that output_format was removed (fixes the "Extra inputs are not permitted" error) + # output_format is now forwarded to Vertex (Vertex parity has shifted — + # it accepts the field and uses it to enforce the JSON schema). The + # prior behavior silently stripped it, hiding Structured Outputs from + # callers who explicitly requested them. + assert "output_format" in final_data + assert final_data["output_format"]["type"] == "json_schema" assert ( - "output_format" not in final_data - ), "output_format should be removed for VertexAI" - assert "model" not in final_data, "model should be removed for VertexAI" + "model" not in final_data + ), "model is still stripped (Vertex routes by URL)" assert "tools" in final_data, "tools should still be present" assert "tool_choice" in final_data, "tool_choice should still be present" @@ -491,28 +498,22 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea ), "Header should be removed if no supported values remain" -def test_vertex_ai_anthropic_output_config_dropped(): +def test_vertex_ai_anthropic_output_config_effort_only_dropped(): """ - Test that output_config parameter is dropped from Vertex AI Anthropic requests. - - Vertex AI does not support the output_config parameter (used for effort settings - in Anthropic API). This test ensures it's properly removed to prevent - "Extra inputs are not permitted" errors. + ``output_config`` containing only ``effort`` (an Anthropic-only key Vertex + rejects with "Extra inputs are not permitted") is dropped entirely so the + request body has no empty dict. """ config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "What is 2+2?"}] - headers = {} + headers: dict = {} - # Simulate optional_params with output_config that would be passed in optional_params = { "max_tokens": 1024, - "output_config": { - "effort": "high" # This is Anthropic-specific and not supported by Vertex AI - }, + "output_config": {"effort": "high"}, } - # Call transform_request which should drop output_config result = config.transform_request( model="claude-3-5-sonnet-20241022", messages=messages, @@ -521,54 +522,144 @@ def test_vertex_ai_anthropic_output_config_dropped(): headers=headers, ) - # Verify output_config was removed assert ( "output_config" not in result - ), "output_config should be dropped from Vertex AI Anthropic requests" - - # Verify other parameters are preserved - assert result["max_tokens"] == 1024, "max_tokens should be preserved" - assert "messages" in result, "messages should be present" + ), "output_config containing only effort must be dropped" + assert result["max_tokens"] == 1024 + assert "messages" in result -def test_vertex_ai_anthropic_output_format_and_output_config_both_dropped(): +def test_vertex_ai_anthropic_output_config_format_passes_through(): """ - Test that both output_format and output_config are dropped from Vertex AI requests. - - This ensures that even if both parameters somehow make it to the transform_request, - they are properly cleaned up before sending to Vertex AI. + ``output_config`` containing structured-output ``format`` is FORWARDED to + Vertex AI Claude — Vertex now accepts it and uses it for JSON Schema + enforcement. Previously the entire field was being silently stripped, so + Anthropic Structured Outputs never engaged on Vertex even when callers + requested it. """ config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "Return a person object."}] + output_config = { + "format": { + "type": "json_schema", + "schema": { + "type": "object", + "additionalProperties": False, + "properties": { + "name": {"type": "string"}, + "age": {"type": "integer"}, + }, + }, + } + } + optional_params = {"max_tokens": 1024, "output_config": output_config} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"] == output_config + + +def test_vertex_ai_anthropic_output_config_format_plus_effort_strips_only_effort(): + """ + Greptile P1 on PR #23396: when ``output_config`` contains BOTH ``format`` + and ``effort``, the prior conditional-passthrough logic forwarded the + full dict including the unsupported ``effort`` key, reproducing the + 400 error the fix was meant to resolve. Only ``effort`` (and any future + Vertex-unsupported keys) should be filtered; ``format`` must survive. + """ + config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "Return a person object."}] + + output_config = { + "format": { + "type": "json_schema", + "schema": { + "type": "object", + "additionalProperties": False, + "properties": {"name": {"type": "string"}}, + }, + }, + "effort": "high", + } + optional_params = {"max_tokens": 1024, "output_config": output_config} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "output_config" in result + assert ( + "effort" not in result["output_config"] + ), "effort must be stripped — Vertex returns 400 on unknown keys" + assert result["output_config"]["format"] == output_config["format"] + + +def test_vertex_ai_anthropic_output_config_non_dict_dropped(): + """Defensive: if ``output_config`` is somehow not a dict, drop it rather + than forwarding malformed data downstream.""" + config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "hi"}] + optional_params = {"max_tokens": 64, "output_config": "not-a-dict"} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "output_config" not in result + + +def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_dropped(): + """ + When the request carries both ``output_format`` (top-level structured + outputs) AND an ``output_config`` whose only useful key for Vertex is + ``effort``: ``output_format`` must be forwarded (Vertex accepts it), + while ``output_config`` is dropped because Vertex returns 400 on + ``effort``. This replaces the old "drop both" behavior, which was the + silent strip the bug report flagged. + """ + config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "Extract structured data"}] - headers = {} + + output_format = { + "type": "json_schema", + "json_schema": { + "name": "data", + "schema": { + "type": "object", + "properties": {"result": {"type": "string"}}, + }, + }, + } optional_params = { "max_tokens": 2048, - "output_format": { - "type": "json_schema", - "json_schema": { - "name": "data", - "schema": { - "type": "object", - "properties": {"result": {"type": "string"}}, - }, - }, - }, + "output_format": output_format, "output_config": {"effort": "high"}, } - # Simulate parent class creating test_data with both parameters - # (as if the parent transform_request added them) test_data = { "model": "claude-3-5-sonnet-20241022", "messages": messages, "max_tokens": 2048, - "output_format": optional_params["output_format"], - "output_config": optional_params["output_config"], + "output_format": output_format, + "output_config": {"effort": "high"}, } - # Mock the parent transform_request to return data with both parameters original_transform = config.__class__.__bases__[0].transform_request def mock_transform_request( @@ -584,22 +675,50 @@ def test_vertex_ai_anthropic_output_format_and_output_config_both_dropped(): messages=messages, optional_params=optional_params, litellm_params={}, - headers=headers, + headers={}, ) - # Verify both were removed - assert ( - "output_format" not in result - ), "output_format should be dropped from Vertex AI requests" - assert ( - "output_config" not in result - ), "output_config should be dropped from Vertex AI requests" - - # Verify essential params are preserved - assert result["max_tokens"] == 2048, "max_tokens should be preserved" - assert "messages" in result, "messages should be present" - assert "model" not in result, "model should also be dropped for Vertex AI" - + # output_format flows through unchanged — Vertex AI Claude accepts it. + assert result["output_format"] == output_format + # output_config containing only ``effort`` is dropped to avoid the + # 400 "Extra inputs are not permitted" the silent strip used to mask. + assert "output_config" not in result + assert result["max_tokens"] == 2048 + assert "model" not in result, "model is still stripped (Vertex routes by URL)" finally: - # Restore original method config.__class__.__bases__[0].transform_request = original_transform + + +def test_sanitize_vertex_anthropic_output_params_unit(): + """Direct unit coverage for the helper itself (used by both Vertex + Anthropic transformation paths). Mirrors the integration assertions + above without going through the full ``transform_request`` stack.""" + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( + _sanitize_vertex_anthropic_output_params, + ) + + # No-op when output_config absent. + data: dict = {"max_tokens": 8} + _sanitize_vertex_anthropic_output_params(data) + assert data == {"max_tokens": 8} + + # Effort-only → dropped entirely. + data = {"output_config": {"effort": "high"}} + _sanitize_vertex_anthropic_output_params(data) + assert "output_config" not in data + + # Format-only → preserved unchanged. + fmt = {"format": {"type": "json_schema", "schema": {"type": "object"}}} + data = {"output_config": dict(fmt)} + _sanitize_vertex_anthropic_output_params(data) + assert data["output_config"] == fmt + + # Mixed → effort filtered, format kept. + data = {"output_config": {"format": fmt["format"], "effort": "high"}} + _sanitize_vertex_anthropic_output_params(data) + assert data["output_config"] == fmt + + # Non-dict → dropped defensively. + data = {"output_config": "garbage"} + _sanitize_vertex_anthropic_output_params(data) + assert "output_config" not in data From 79517bc6282c43d0d844b6d3f7b2597e3c2735bd Mon Sep 17 00:00:00 2001 From: Darien Kindlund Date: Fri, 24 Apr 2026 12:03:55 -0400 Subject: [PATCH 2/7] fix: address Greptile review feedback on PR #26439 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three concerns raised by bot reviewers, all addressed: 1. CodeQL cyclic-import warning ``experimental_pass_through/transformation.py`` imported from the parent ``..transformation`` module, which CodeQL flagged as a potential cycle. Extracted the helper into a new leaf module ``vertex_ai_partner_models/anthropic/output_params_utils.py`` that has no heavy imports of its own. Both transformation files now import from it cleanly. Renamed the helper from the underscore- prefixed ``_sanitize_vertex_anthropic_output_params`` to the public ``sanitize_vertex_anthropic_output_params`` since it is now shared across modules. 2. Greptile P2: redundant ``None`` guard on ``extra_kwargs`` ``handler.py`` had two ``extra_kwargs = extra_kwargs if ... else {}`` coercions; the second was a no-op because line 220 already coerced. Removed the second one and added a NOTE comment so future readers understand ``extra_kwargs`` is guaranteed non-None at the point of use. 3. Greptile P2: misleading "already translated" docstring The docstring claimed the translator above mapped ``output_config.format`` to ``response_format``, but Greptile correctly traced the code and found that only the legacy top-level ``output_format`` was being translated — ``output_config.format`` was being silently dropped on the adapter path. Two-part fix: a. Code: extended ``_translate_output_format_to_openai`` to accept both shapes (top-level ``output_format`` AND ``output_config.format`` sub-key). Top-level still takes precedence when both are supplied. This means callers using the newer Anthropic Structured Outputs API now have their schema properly forwarded to non-Anthropic backends as ``response_format``. b. Tests: rewrote the misleading docstring to describe what actually happens, plus added two new tests: * ``test_output_format_top_level_still_translates`` — regression guard for the legacy path * ``test_output_format_takes_precedence_over_output_config_format`` — documents the precedence rule explicitly Tests: 28/28 pass (was 26/26 before; +2 for the new translation behavior + precedence). All run in ~0.5s, no real network calls. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../adapters/handler.py | 3 +- .../adapters/transformation.py | 23 +++-- .../transformation.py | 4 +- .../anthropic/output_params_utils.py | 50 +++++++++++ .../anthropic/transformation.py | 42 +-------- .../test_handler_output_config_passthrough.py | 85 ++++++++++++++++--- ...partner_models_anthropic_transformation.py | 14 +-- 7 files changed, 156 insertions(+), 65 deletions(-) create mode 100644 litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 10455825e41..ac55aac8062 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -254,7 +254,8 @@ class LiteLLMMessagesToCompletionTransformationHandler: # ``AnthropicMessagesRequestOptionalParams``, also extend # ``ANTHROPIC_ONLY_REQUEST_KEYS`` here so it doesn't silently leak. excluded_keys = ANTHROPIC_ONLY_REQUEST_KEYS | {"anthropic_messages"} - extra_kwargs = extra_kwargs if extra_kwargs is not None else {} + # NOTE: extra_kwargs was already coerced from None to {} at the top of + # this method (line ~220). It is guaranteed to be a dict here. for key, value in extra_kwargs.items(): if ( key == "litellm_logging_obj" diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 20fa4f125de..f7bc67ccd3a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -664,7 +664,7 @@ class LiteLLMAnthropicMessagesAdapter: @staticmethod def translate_anthropic_thinking_to_reasoning_effort( - thinking: Dict[str, Any] + thinking: Dict[str, Any], ) -> Optional[str]: """ Translate Anthropic's thinking parameter to OpenAI's reasoning_effort. @@ -1081,10 +1081,23 @@ class LiteLLMAnthropicMessagesAdapter: anthropic_message_request: AnthropicMessagesRequest, new_kwargs: ChatCompletionRequest, ) -> None: - """Translate output_format to response_format when applicable.""" - if "output_format" not in anthropic_message_request: - return - output_format = anthropic_message_request["output_format"] + """Translate Anthropic structured-output config to OpenAI ``response_format``. + + Accepts either the legacy top-level ``output_format`` field OR the + newer ``output_config.format`` (sub-key on ``output_config``) so that + both shapes flow through to non-Anthropic backends as + ``response_format``. Without the ``output_config.format`` branch, + callers using the new Anthropic Structured Outputs API would have + their schema silently dropped on the adapter path — only the legacy + top-level ``output_format`` was being mapped. + + ``output_format`` takes precedence when both are provided. + """ + output_format: Any = anthropic_message_request.get("output_format") + if not output_format: + output_config = anthropic_message_request.get("output_config") + if isinstance(output_config, dict): + output_format = output_config.get("format") if not output_format: return response_format = self.translate_anthropic_output_format_to_openai( diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 9080ac02330..d450f7a4635 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -13,7 +13,7 @@ from litellm.types.llms.vertex_ai import VertexPartnerProvider from litellm.types.router import GenericLiteLLMParams from ....vertex_llm_base import VertexBase -from ..transformation import _sanitize_vertex_anthropic_output_params +from ..output_params_utils import sanitize_vertex_anthropic_output_params class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, VertexBase): @@ -163,6 +163,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert # and ``output_format``, but rejects ``output_config.effort`` with 400 # "Extra inputs are not permitted". Sanitize in place so the supported # bits flow through. - _sanitize_vertex_anthropic_output_params(anthropic_messages_request) + sanitize_vertex_anthropic_output_params(anthropic_messages_request) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py new file mode 100644 index 00000000000..982d8edbf20 --- /dev/null +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py @@ -0,0 +1,50 @@ +""" +Shared sanitization for ``output_config`` / ``output_format`` on Vertex AI +Claude. Lives in its own module so both the chat-completion transformation +(``transformation.py``) and the Messages pass-through transformation +(``experimental_pass_through/transformation.py``) can import it without +forming a cycle through the parent module's heavier imports. + +CodeQL flagged the ``..transformation`` import path as a potential cyclic +import; extracting the helper into a leaf module resolves the warning and +keeps the parent module's import surface narrow. +""" + +# Keys inside ``output_config`` that Vertex AI Claude does not accept. +# Today only ``effort`` triggers "Extra inputs are not permitted"; add new +# entries here as Vertex parity drifts. Keep this list narrow — anything +# Vertex DOES accept (e.g. ``format`` for structured outputs) must be +# preserved so callers can rely on Anthropic-native features. +VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset({"effort"}) + + +def sanitize_vertex_anthropic_output_params(data: dict) -> None: + """ + Strip Vertex-unsupported keys from ``output_config`` / + ``output_format`` in-place; forward whatever remains. + + Behavior: + * ``output_config`` containing only unsupported keys (e.g. ``effort`` + alone) is removed entirely so the request body has no empty dict. + * ``output_config`` containing a mix of supported + unsupported keys + has the unsupported subset filtered out and the rest forwarded. + * ``output_config`` that is supported in full passes through unchanged. + * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). + * Non-dict values for ``output_config`` are dropped to avoid sending + malformed payloads downstream. + """ + output_config = data.get("output_config") + if output_config is None: + return + if not isinstance(output_config, dict): + data.pop("output_config", None) + return + sanitized = { + k: v + for k, v in output_config.items() + if k not in VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS + } + if sanitized: + data["output_config"] = sanitized + else: + data.pop("output_config", None) diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index a9dea6646ff..914c7e92e5e 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -10,45 +10,7 @@ from litellm.types.llms.openai import AllMessageValues from litellm.types.utils import ModelResponse from ....anthropic.chat.transformation import AnthropicConfig - -# Keys inside ``output_config`` that Vertex AI Claude does not accept. -# Today only ``effort`` triggers "Extra inputs are not permitted"; add new -# entries here as Vertex parity drifts. Keep this list narrow — anything -# Vertex DOES accept (e.g. ``format`` for structured outputs) must be -# preserved so callers can rely on Anthropic-native features. -_VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS = frozenset({"effort"}) - - -def _sanitize_vertex_anthropic_output_params(data: dict) -> None: - """ - Strip Vertex-unsupported keys from ``output_config`` / ``output_format`` - in-place; forward whatever remains. - - Behavior: - * ``output_config`` containing only unsupported keys (e.g. ``effort`` - alone) is removed entirely so the request body has no empty dict. - * ``output_config`` containing a mix of supported + unsupported keys has - the unsupported subset filtered out and the rest forwarded. - * ``output_config`` that is supported in full passes through unchanged. - * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). - * Non-dict values for ``output_config`` are dropped to avoid sending - malformed payloads downstream. - """ - output_config = data.get("output_config") - if output_config is None: - return - if not isinstance(output_config, dict): - data.pop("output_config", None) - return - sanitized = { - k: v - for k, v in output_config.items() - if k not in _VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS - } - if sanitized: - data["output_config"] = sanitized - else: - data.pop("output_config", None) +from .output_params_utils import sanitize_vertex_anthropic_output_params class VertexAIError(Exception): @@ -149,7 +111,7 @@ class VertexAIAnthropicConfig(AnthropicConfig): # Vertex returns 400 "Extra inputs are not permitted". Sanitize in place: # forward the structured-output bits, drop the unsupported keys. # Same treatment for the legacy top-level ``output_format`` field. - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) tools = optional_params.get("tools") tool_search_used = self.is_tool_search_used(tools) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py index 50e5fd884e9..615dc5cfebc 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py @@ -46,10 +46,13 @@ from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( MESSAGES = [{"role": "user", "content": "hello"}] -def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): +def _call_prepare(extra_kwargs, model="gpt-4o", output_format=None, **overrides): """ Drive ``_prepare_completion_kwargs`` with the minimum scaffolding needed. + ``output_format`` is a top-level parameter on the function, so callers + pass it explicitly here rather than tucking it into ``extra_kwargs``. + Uses an explicit-None check on ``extra_kwargs`` so callers can test the falsy-empty-dict path. The fallback ``or {}`` pattern PR #22727 used here masked the no-extra-kwargs case from ever exercising the test's intent. @@ -68,7 +71,7 @@ def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): tools=None, top_k=None, top_p=None, - output_format=None, + output_format=output_format, extra_kwargs=extra_kwargs, ) @@ -107,22 +110,84 @@ class TestOutputConfigStrippedFromCompletionKwargs: "reject it with 400 'Extra inputs are not permitted'" ) - def test_output_config_with_format_is_stripped_format_already_translated(self): - """Even when ``output_config`` carries useful structured-output info, - the raw key must be excluded — the translator above has already mapped - ``output_config.format`` to ``response_format`` (the OpenAI-shaped key - the downstream backend understands).""" + def test_output_config_format_translated_to_response_format(self): + """When ``output_config`` carries structured-output ``format``, the + translator now maps it to OpenAI's ``response_format`` so non-Anthropic + backends see the schema in their native shape. The raw + ``output_config`` key is still stripped from ``completion_kwargs`` — + only the translated ``response_format`` survives. + + Before this PR, only the legacy top-level ``output_format`` was + translated; ``output_config.format`` was silently dropped on the + adapter path even when the schema was correctly supplied (issue + flagged by Greptile review of the initial fix). + """ + schema = { + "type": "object", + "additionalProperties": False, + "properties": {"name": {"type": "string"}}, + } extra_kwargs = { "custom_llm_provider": "azure", - "output_config": { - "format": {"type": "json_schema", "schema": {"type": "object"}} - }, + "output_config": {"format": {"type": "json_schema", "schema": schema}}, } result = _call_prepare(extra_kwargs=extra_kwargs) completion_kwargs = result[0] if isinstance(result, tuple) else result + # Raw Anthropic-shaped key is gone (would 400 on non-Anthropic backends). assert "output_config" not in completion_kwargs + # Translated OpenAI-shaped key is present so the schema actually + # reaches the downstream backend. + assert "response_format" in completion_kwargs, ( + "output_config.format must be translated to response_format — " + "without this, structured-output schemas are silently dropped on " + "the adapter path" + ) + + def test_output_format_top_level_still_translates(self): + """Regression guard: the legacy top-level ``output_format`` field must + continue to translate to ``response_format``. The new + ``output_config.format`` path must not break this existing behavior.""" + schema = {"type": "object", "properties": {"name": {"type": "string"}}} + result = _call_prepare( + extra_kwargs={"custom_llm_provider": "azure"}, + output_format={"type": "json_schema", "schema": schema}, + ) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "response_format" in completion_kwargs + + def test_output_format_takes_precedence_over_output_config_format(self): + """When both top-level ``output_format`` and ``output_config.format`` + are present, the legacy top-level ``output_format`` wins. Documents + which one the translator picks rather than leaving it implementation- + defined.""" + winning_schema = { + "type": "object", + "properties": {"top_level": {"type": "string"}}, + } + losing_schema = { + "type": "object", + "properties": {"nested": {"type": "string"}}, + } + result = _call_prepare( + extra_kwargs={ + "custom_llm_provider": "azure", + "output_config": { + "format": {"type": "json_schema", "schema": losing_schema} + }, + }, + output_format={"type": "json_schema", "schema": winning_schema}, + ) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "response_format" in completion_kwargs + # Verify the winning_schema (top-level output_format) was used, + # not the losing one nested under output_config. + rendered = str(completion_kwargs["response_format"]) + assert "top_level" in rendered + assert "nested" not in rendered def test_other_extra_kwargs_still_passed_through(self): """Regression guard: the strip must be narrow. Unrelated fields like diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 376c48d9e95..d0be476d72e 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -693,32 +693,32 @@ def test_sanitize_vertex_anthropic_output_params_unit(): """Direct unit coverage for the helper itself (used by both Vertex Anthropic transformation paths). Mirrors the integration assertions above without going through the full ``transform_request`` stack.""" - from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( - _sanitize_vertex_anthropic_output_params, + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.output_params_utils import ( + sanitize_vertex_anthropic_output_params, ) # No-op when output_config absent. data: dict = {"max_tokens": 8} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data == {"max_tokens": 8} # Effort-only → dropped entirely. data = {"output_config": {"effort": "high"}} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert "output_config" not in data # Format-only → preserved unchanged. fmt = {"format": {"type": "json_schema", "schema": {"type": "object"}}} data = {"output_config": dict(fmt)} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data["output_config"] == fmt # Mixed → effort filtered, format kept. data = {"output_config": {"format": fmt["format"], "effort": "high"}} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data["output_config"] == fmt # Non-dict → dropped defensively. data = {"output_config": "garbage"} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert "output_config" not in data From e2d0fd9eacdeadf46c0e18057e5f51ea1f28eb49 Mon Sep 17 00:00:00 2001 From: RoomWithOutRoof <166608075+Jah-yee@users.noreply.github.com> Date: Sun, 26 Apr 2026 01:45:01 +0800 Subject: [PATCH 3/7] fix: remove duplicate MAX_SIZE + add Cloudflare response_text support (#26385) - Remove duplicate MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB definition (kept the one with default 1024, removed the one with default 512) - Add fallback from 'response' to 'response_text' key in Cloudflare Workers AI transformation for newer Nemotron models Co-authored-by: yuneng-jiang Co-authored-by: Jah-yee <110645028+Jah-yee@users.noreply.github.com> --- litellm/constants.py | 3 --- litellm/llms/cloudflare/chat/transformation.py | 6 +++--- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 012599ab6ab..385e3723bee 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -409,9 +409,6 @@ CACHED_STREAMING_CHUNK_DELAY = float(os.getenv("CACHED_STREAMING_CHUNK_DELAY", 0 AUDIO_SPEECH_CHUNK_SIZE = int( os.getenv("AUDIO_SPEECH_CHUNK_SIZE", 8192) ) # chunk_size for audio speech streaming. Balance between latency and memory usage -MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( - os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 512) -) DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2000)) #### Networking settings #### # Sentinel used when `REQUEST_TIMEOUT` is unset: `litellm.request_timeout` keeps this diff --git a/litellm/llms/cloudflare/chat/transformation.py b/litellm/llms/cloudflare/chat/transformation.py index 9e59782bf73..d0c2e86f708 100644 --- a/litellm/llms/cloudflare/chat/transformation.py +++ b/litellm/llms/cloudflare/chat/transformation.py @@ -147,9 +147,9 @@ class CloudflareChatConfig(BaseConfig): ) -> ModelResponse: completion_response = raw_response.json() - model_response.choices[0].message.content = completion_response["result"][ # type: ignore - "response" - ] + # Support both "response" and "response_text" keys (newer models like Nemotron use "response_text") + result = completion_response["result"] + model_response.choices[0].message.content = result.get("response") if result.get("response") is not None else result.get("response_text", "") # type: ignore prompt_tokens = litellm.utils.get_token_count(messages=messages, model=model) completion_tokens = len( From 334aedf2d49cbc1992b153610950a12b6603a97d Mon Sep 17 00:00:00 2001 From: Blossom Date: Sun, 26 Apr 2026 01:52:17 +0800 Subject: [PATCH 4/7] fix(ui): add missing 'zai' (Z.AI / Zhipu AI) provider to Add-Model dropdown (#25482) (#26419) The Z.AI (Zhipu AI) provider was missing from the Add-Model dropdown in the admin UI, even though the rest of the stack already supports it: - /public/providers returns 'zai' in the provider list - provider_endpoints_support.json includes a full 'zai' entry with endpoints and a docs URL (https://docs.litellm.ai/docs/providers/zai) - Backend routing works for zai/* models (e.g. zai/glm-4.5, zai/glm-5) - There are many zai/* entries in model_prices_and_context_window.json The dropdown is driven by the hard-coded Providers enum and provider_map in provider_info_helpers.tsx, which did not include 'zai', so users could not select Z.AI when adding a model through the UI. This PR: - Adds Providers.ZAI ('Z.AI (Zhipu AI)') to the enum. - Maps it to 'zai' in provider_map so the UI round-trips the existing backend provider key. - Wires a reasonable placeholder 'zai/glm-4.5' in getPlaceholder, since glm-4.5 is an established zai/* model in the pricing catalog. - Adds two regression tests in provider_info_helpers.test.tsx: 1. getProviderLogoAndName('zai') resolves to Providers.ZAI. 2. getPlaceholder(Providers.ZAI) returns 'zai/glm-4.5'. No logo asset is added in this PR; getProviderLogoAndName already gracefully returns an empty logo string for providers missing from providerLogoMap, matching the existing pattern for several other providers. A follow-up can add a dedicated logo. Fixes #25482 Co-authored-by: yuneng-jiang --- .../src/components/provider_info_helpers.test.tsx | 14 ++++++++++++++ .../src/components/provider_info_helpers.tsx | 4 ++++ 2 files changed, 18 insertions(+) diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index a8021f94d84..fa014de4e62 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -68,6 +68,16 @@ describe("provider_info_helpers", () => { expect(result.logo).toBe(providerLogoMap[Providers.OpenAI]); }); + it("should resolve the zai (Z.AI) provider value to the Z.AI display name", () => { + // Regression test for https://github.com/BerriAI/litellm/issues/25482 — + // the backend already returns `zai` from /public/providers and the docs + // have a dedicated page, but the UI dropdown was missing an entry, so + // `getProviderLogoAndName("zai")` previously returned the raw value as + // the display name (no mapping). + const result = getProviderLogoAndName("zai"); + expect(result.displayName).toBe(Providers.ZAI); + }); + it("should return provider value as display name when no mapping exists", () => { const unknownProvider = "unknown_provider"; const result = getProviderLogoAndName(unknownProvider); @@ -156,6 +166,10 @@ describe("provider_info_helpers", () => { expect(getPlaceholder(Providers.WATSONX)).toBe("watsonx/ibm/granite-3-3-8b-instruct"); }); + it("should return zai/glm-4.5 placeholder for Z.AI provider", () => { + expect(getPlaceholder(Providers.ZAI)).toBe("zai/glm-4.5"); + }); + it("should return default gpt-3.5-turbo placeholder for unknown provider", () => { expect(getPlaceholder("UnknownProvider" as any)).toBe("gpt-3.5-turbo"); }); diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index e833d0eb4fb..62c0633d117 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -103,6 +103,7 @@ export enum Providers { WATSONX_TEXT = "Watsonx Text", xAI = "xAI", XINFERENCE = "Xinference", + ZAI = "Z.AI (Zhipu AI)", } export const provider_map: Record = { @@ -210,6 +211,7 @@ export const provider_map: Record = { WATSONX_TEXT: "watsonx_text", xAI: "xai", XINFERENCE: "xinference", + ZAI: "zai", }; const asset_logos_folder = "../ui/assets/logos/"; @@ -366,6 +368,8 @@ export const getPlaceholder = (selectedProvider: string): string => { return "watsonx/ibm/granite-3-3-8b-instruct"; } else if (selectedProvider === Providers.Cursor) { return "cursor/claude-4-sonnet"; + } else if (selectedProvider === Providers.ZAI) { + return "zai/glm-4.5"; } else { return "gpt-3.5-turbo"; } From f63a6f1b263130e8529b699acb19b7481489518e Mon Sep 17 00:00:00 2001 From: Yufeng He <40085740+he-yufeng@users.noreply.github.com> Date: Sun, 26 Apr 2026 03:13:13 +0800 Subject: [PATCH 5/7] fix(proxy): set verbose_logger level when LITELLM_LOG=INFO (#26401) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes #26396. The `LITELLM_LOG=INFO` branch in proxy_server only set `verbose_router_logger` and `verbose_proxy_logger`. The third logger `verbose_logger` (used by e.g. `token_based_routing.py`) inherited the Python root default (WARNING) and its INFO-level messages were silently filtered — inconsistent with the neighbouring DEBUG branch which configures all three and with the `debug=True` / `detailed_debug` paths above. Include `verbose_logger` in the INFO branch as well so all three loggers behave the same. Co-authored-by: yuneng-jiang Co-authored-by: Yufeng He <40085740+universeplayer@users.noreply.github.com> --- litellm/proxy/proxy_server.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 4ca6895c2ce..4afac7173cd 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5845,10 +5845,15 @@ async def initialize( # noqa: PLR0915 if litellm_log_setting.upper() == "INFO": import logging - from litellm._logging import verbose_proxy_logger, verbose_router_logger + from litellm._logging import ( + verbose_logger, + verbose_proxy_logger, + verbose_router_logger, + ) # this must ALWAYS remain logging.INFO, DO NOT MODIFY THIS + verbose_logger.setLevel(level=logging.INFO) # set package log to info verbose_router_logger.setLevel( level=logging.INFO ) # set router logs to info From 98a9005c765cf6ceee0eec498e3517166c0e0b7e Mon Sep 17 00:00:00 2001 From: Alvin Tang Date: Sun, 26 Apr 2026 05:11:51 +0800 Subject: [PATCH 6/7] fix(arize): _set_usage_outputs handles raw OpenAI Pydantic CompletionUsage (#26506) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * [Feat] Day-0 support for GPT-5.5 and GPT-5.5 Pro (#26449) * feat(openai): day-0 support for GPT-5.5 and GPT-5.5 Pro Add pricing + capability entries for the new GPT-5.5 family launched by OpenAI on 2026-04-24: - gpt-5.5 / gpt-5.5-2026-04-23 (chat): $5/$30/$0.50 per 1M input/output/cached input - gpt-5.5-pro / gpt-5.5-pro-2026-04-23 (responses-only): $60/$360/$6 per 1M input/output/cached input Other fees (long-context >272k, flex, batches, priority, cache discounts) follow the same ratios as GPT-5.4, with context window retained at 1.05M input / 128K output. No transformation / classifier code changes are required: OpenAIGPT5Config.is_model_gpt_5_4_plus_model() already matches 5.5+ via numeric version parsing, and model registration is driven from the JSON. The existing responses-API bridge for tools + reasoning_effort (litellm/main.py:970) already covers gpt-5.5-pro. Tests: - GPT5_MODELS regression list now covers gpt-5.5-pro and dated variants - New test_generic_cost_per_token_gpt55_pro cost-calc test - Updated test_generic_cost_per_token_gpt55 for long-context fields * fix(openai): mirror reasoning_effort flags onto gpt-5.5 dated variants gpt-5.5-2026-04-23 and gpt-5.5-pro-2026-04-23 were missing the supports_none_reasoning_effort, supports_xhigh_reasoning_effort, and supports_minimal_reasoning_effort flags that their non-dated counterparts define. Reasoning-effort routing in OpenAIGPT5Config is fully capability-driven from these JSON flags — since an absent flag is treated as False for opt-in levels (xhigh), users pinning to a dated snapshot would silently lose xhigh support and diverge from the base alias on logprobs + flexible temperature handling. Copy the flags onto both dated variants so every dated snapshot inherits the base model's reasoning-effort capability profile. Adds a parametrized regression test that asserts supports_{none,minimal,xhigh}_reasoning_effort parity between each dated variant and its non-dated counterpart, preventing future drift when new snapshots are added. * [Feat] Add azure/gpt-5.5 + azure/gpt-5.5-pro entries (+ dated variants) (#26361) * feat(azure): add azure/gpt-5.5 + azure/gpt-5.5-pro entries (+ dated variants) Azure variants of OpenAI's GPT-5.5 family. Microsoft has not yet shipped GPT-5.5 on Azure OpenAI (latest GA on the Foundry models page is GPT-5.4 as of 2026-04-24), but adding the entries day-0 mirrors the established precedent for azure/gpt-5.4* (which were in the cost map before the Azure rollout) so cost tracking and capability flags work the moment customers deploy. Schema follows the existing azure/gpt-5.4* shape: - Same base/long-context pricing as openai/gpt-5.5*: $5/$30 chat, $60/$360 pro per 1M, with priority tier 2x base - Azure variants drop the flex/batches keys (Azure has no flex tier) but keep priority pricing, matching gpt-5.4* precedent - mode=chat for the thinking model, mode=responses for pro reasoning_effort capability flags mirror the OpenAI variants exactly since Azure proxies the same API contract: minimal rejection on both chat and pro, low/none rejection on pro. Once #26456 (which sets supports_low_reasoning_effort + minimal=false on openai/gpt-5.5*) lands, OpenAI and Azure flag profiles align. Tests pin entry presence + pricing for all four Azure variants and verify the live-API-derived reasoning_effort flags. * test: register supports_low_reasoning_effort in cost-map JSON schema azure/gpt-5.5-pro and azure/gpt-5.5-pro-2026-04-23 added in this branch carry supports_low_reasoning_effort=false. The strict 'additionalProperties: false' schema in test_aaamodel_prices_and_context_window_json_is_valid rejected the new key. Register it alongside the other supports_*_reasoning_effort entries. Note: the runtime side of this flag (code that reads it) lands in #26456. Until that PR merges the flag is inert for both Azure and OpenAI pro entries, but having the schema accept it lets cost-map tests pass on either merge order. * fix(arize/langfuse_otel): handle Pydantic usage objects without `.get` `_set_usage_outputs` called `usage.get(...)` and `usage.get('output_tokens_details', {}).get('reasoning_tokens')`. These crash with `AttributeError: 'CompletionUsage' object has no attribute 'get'` when `usage` (or the nested token-details object) is a raw OpenAI Pydantic model rather than a dict / litellm `Usage` wrapper. Reproduces on the langfuse_otel + arize Responses API logging paths. Fixes #13672. Changes: - Add `_safe_get(obj, key, default)` that prefers dict-style `.get` when available and otherwise falls back to `getattr`. Works uniformly for dicts, litellm's `Usage`, and plain Pydantic models like `openai.types.completion_usage.CompletionUsage` / `CompletionTokensDetails` / `OutputTokensDetails`. - Use `_safe_get` for total / completion / prompt / output tokens. - Look for reasoning tokens in `completion_tokens_details` (Chat Completions API) before falling back to `output_tokens_details` (Responses API). Previously reasoning tokens from the Chat Completions API were silently dropped. Tests: - `test_set_usage_outputs_pydantic_completion_usage` — covers the chat completions path with raw `CompletionUsage` + `CompletionTokensDetails`. - `test_set_usage_outputs_pydantic_response_api_usage` — covers the Responses API path with a Pydantic usage object lacking `.get`. Both tests fail on main before this commit and pass after. --------- Co-authored-by: yuneng-jiang Co-authored-by: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Co-authored-by: alvinttang Co-authored-by: Krrish Dholakia --- litellm/integrations/arize/_utils.py | 42 ++++- ...odel_prices_and_context_window_backup.json | 163 ++++++++++++++++++ model_prices_and_context_window.json | 163 ++++++++++++++++++ .../integrations/arize/test_arize_utils.py | 85 +++++++++ tests/test_litellm/test_utils.py | 1 + 5 files changed, 450 insertions(+), 4 deletions(-) diff --git a/litellm/integrations/arize/_utils.py b/litellm/integrations/arize/_utils.py index 8dfaa8b1425..a1bf65141c9 100644 --- a/litellm/integrations/arize/_utils.py +++ b/litellm/integrations/arize/_utils.py @@ -220,23 +220,57 @@ def _set_structured_outputs(span: "Span", response_obj, msg_attrs, span_attrs): safe_set_attribute(span, f"{prefix}.{msg_attrs.MESSAGE_ROLE}", message_role) +def _safe_get(obj, key, default=None): + """Read ``key`` from a dict-like or Pydantic-model-like object. + + The arize/langfuse_otel logger receives ``usage`` objects from many sources: + plain dicts, litellm ``Usage`` (which exposes ``.get``), and raw OpenAI + Pydantic models (e.g. ``openai.types.completion_usage.CompletionUsage`` and + nested ``CompletionTokensDetails`` / ``OutputTokensDetails``) which do NOT + expose ``.get``. Calling ``.get`` on the latter raised ``AttributeError`` — + see https://github.com/BerriAI/litellm/issues/13672. + """ + if obj is None: + return default + getter = getattr(obj, "get", None) + if callable(getter): + try: + return getter(key, default) + except TypeError: + # Some objects expose `.get` with a different signature + pass + return getattr(obj, key, default) + + def _set_usage_outputs(span: "Span", response_obj, span_attrs): usage = response_obj and response_obj.get("usage") if not usage: return safe_set_attribute( - span, span_attrs.LLM_TOKEN_COUNT_TOTAL, usage.get("total_tokens") + span, span_attrs.LLM_TOKEN_COUNT_TOTAL, _safe_get(usage, "total_tokens") + ) + completion_tokens = _safe_get(usage, "completion_tokens") or _safe_get( + usage, "output_tokens" ) - completion_tokens = usage.get("completion_tokens") or usage.get("output_tokens") if completion_tokens: safe_set_attribute( span, span_attrs.LLM_TOKEN_COUNT_COMPLETION, completion_tokens ) - prompt_tokens = usage.get("prompt_tokens") or usage.get("input_tokens") + prompt_tokens = _safe_get(usage, "prompt_tokens") or _safe_get( + usage, "input_tokens" + ) if prompt_tokens: safe_set_attribute(span, span_attrs.LLM_TOKEN_COUNT_PROMPT, prompt_tokens) - reasoning_tokens = usage.get("output_tokens_details", {}).get("reasoning_tokens") + + # Reasoning tokens live in `completion_tokens_details` for Chat Completions + # API (Usage) and in `output_tokens_details` for Responses API + # (ResponseAPIUsage). Both nested objects may be plain Pydantic models + # without `.get`. + token_details = _safe_get(usage, "completion_tokens_details") or _safe_get( + usage, "output_tokens_details" + ) + reasoning_tokens = _safe_get(token_details, "reasoning_tokens") if reasoning_tokens: safe_set_attribute( span, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index f6de40717d1..5cccd5f00af 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -4645,6 +4645,169 @@ "supports_vision": true, "supports_web_search": true }, + "azure/gpt-5.5": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": true, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false + }, + "azure/gpt-5.5-2026-04-23": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true + }, + "azure/gpt-5.5-pro": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": false, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false + }, + "azure/gpt-5.5-pro-2026-04-23": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "azure/gpt-5.4-mini": { "cache_read_input_token_cost": 7.5e-08, "input_cost_per_token": 7.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 2d13c5cd00f..12a0d8fe0a7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -4659,6 +4659,169 @@ "supports_vision": true, "supports_web_search": true }, + "azure/gpt-5.5": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": true, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false + }, + "azure/gpt-5.5-2026-04-23": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true + }, + "azure/gpt-5.5-pro": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": false, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false + }, + "azure/gpt-5.5-pro-2026-04-23": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "azure/gpt-5.4-mini": { "cache_read_input_token_cost": 7.5e-08, "input_cost_per_token": 7.5e-07, diff --git a/tests/test_litellm/integrations/arize/test_arize_utils.py b/tests/test_litellm/integrations/arize/test_arize_utils.py index a87a4167899..86c5448d468 100644 --- a/tests/test_litellm/integrations/arize/test_arize_utils.py +++ b/tests/test_litellm/integrations/arize/test_arize_utils.py @@ -273,6 +273,91 @@ def test_arize_set_attributes_responses_api(): ) +def test_set_usage_outputs_pydantic_completion_usage(): + """ + Regression test for https://github.com/BerriAI/litellm/issues/13672 + + `_set_usage_outputs` previously called `usage.get(...)` which crashes when + `usage` is a plain Pydantic model (e.g. openai.types.completion_usage.CompletionUsage) + that does not implement dict-style `.get()`. Same crash for nested + `output_tokens_details` / `completion_tokens_details`. + + The function must: + 1. Read total/prompt/completion tokens from a Pydantic usage without `.get`. + 2. Read reasoning_tokens from `completion_tokens_details` (chat completions API) + OR `output_tokens_details` (responses API), even when those nested objects + are Pydantic models without `.get`. + 3. Not raise AttributeError; not call span.record_exception. + """ + from unittest.mock import MagicMock + + from openai.types.completion_usage import ( + CompletionTokensDetails, + CompletionUsage, + ) + + from litellm.integrations.arize._utils import _set_usage_outputs + + span = MagicMock() + + # Plain OpenAI Pydantic model — has no `.get()` + usage = CompletionUsage( + completion_tokens=60, + prompt_tokens=40, + total_tokens=100, + completion_tokens_details=CompletionTokensDetails(reasoning_tokens=25), + ) + assert not hasattr(usage, "get"), "precondition: CompletionUsage must lack .get" + + response_obj = {"usage": usage} + + # Must not raise + _set_usage_outputs(span, response_obj, SpanAttributes) + + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 100) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60) + # reasoning_tokens for chat completions live in completion_tokens_details + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25 + ) + + +def test_set_usage_outputs_pydantic_response_api_usage(): + """ + Same crash also affects Responses API with `output_tokens_details` as a + Pydantic model that lacks `.get()`. Verifies the responses-API path. + """ + from unittest.mock import MagicMock + + from litellm.integrations.arize._utils import _set_usage_outputs + from litellm.types.llms.openai import OutputTokensDetails + + # Build an object that mimics openai ResponsesAPI usage but lacks `.get` + # (uses a plain class — not BaseLiteLLMOpenAIResponseObject) + class PlainResponsesUsage: + def __init__(self): + self.total_tokens = 370 + self.input_tokens = 120 + self.output_tokens = 250 + self.output_tokens_details = OutputTokensDetails(reasoning_tokens=180) + + usage = PlainResponsesUsage() + assert not hasattr(usage, "get") + + span = MagicMock() + response_obj = {"usage": usage} + + _set_usage_outputs(span, response_obj, SpanAttributes) + + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250) + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180 + ) + + class TestArizeLogger(CustomLogger): """ Custom logger implementation to capture standard_callback_dynamic_params. diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index dc344a433bc..93c61e003d9 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -769,6 +769,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "uses_embed_content": {"type": "boolean"}, "supports_reasoning": {"type": "boolean"}, "supports_minimal_reasoning_effort": {"type": "boolean"}, + "supports_low_reasoning_effort": {"type": "boolean"}, "supports_none_reasoning_effort": {"type": "boolean"}, "supports_xhigh_reasoning_effort": {"type": "boolean"}, "supports_max_reasoning_effort": {"type": "boolean"}, From b07908133b828d62dc74cc6c0834c56a4b2768f1 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 2 May 2026 05:59:15 +0000 Subject: [PATCH 7/7] fix(cloudflare): support response_text in streaming chunk parser Newer Cloudflare Workers AI models (e.g. Nemotron) emit 'response_text' instead of 'response' on streamed chunks. The non-streaming path was already updated to fall back to 'response_text' (#26385), but the streaming chunk parser still only read 'response', which caused streaming requests against those models to silently produce empty content. Mirror the non-streaming fallback in CloudflareChatResponseIterator.chunk_parser and add a streaming test for the response_text shape. Co-authored-by: Mateo Wang --- .../llms/cloudflare/chat/transformation.py | 4 +- tests/llm_translation/test_cloudflare.py | 81 +++++++++++++++++++ 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/litellm/llms/cloudflare/chat/transformation.py b/litellm/llms/cloudflare/chat/transformation.py index d0c2e86f708..aef41079a0a 100644 --- a/litellm/llms/cloudflare/chat/transformation.py +++ b/litellm/llms/cloudflare/chat/transformation.py @@ -199,8 +199,10 @@ class CloudflareChatResponseIterator(BaseModelResponseIterator): index = int(chunk.get("index", 0)) - if "response" in chunk: + if "response" in chunk and chunk["response"] is not None: text = chunk["response"] + elif "response_text" in chunk and chunk["response_text"] is not None: + text = chunk["response_text"] returned_chunk = GenericStreamingChunk( text=text, diff --git a/tests/llm_translation/test_cloudflare.py b/tests/llm_translation/test_cloudflare.py index 0c799b4f399..5a6a0008398 100644 --- a/tests/llm_translation/test_cloudflare.py +++ b/tests/llm_translation/test_cloudflare.py @@ -43,6 +43,14 @@ def _streaming_chunks() -> list[str]: ] +def _streaming_chunks_response_text() -> list[str]: + return [ + json.dumps({"response_text": "I am"}), + json.dumps({"response_text": " a language"}), + json.dumps({"response_text": " model."}), + ] + + @pytest.mark.parametrize("sync_mode", [True, False]) def test_completion_cloudflare(sync_mode): messages = [{"role": "user", "content": "what llm are you"}] @@ -145,3 +153,76 @@ def test_completion_cloudflare_stream(sync_mode): if c.choices[0].delta.content ) assert "language" in content.lower() + + +@pytest.mark.parametrize("sync_mode", [True, False]) +def test_completion_cloudflare_stream_response_text(sync_mode): + """Newer Cloudflare Workers AI models (e.g. Nemotron) emit `response_text` + instead of `response` in streamed chunks. The iterator must surface that + text so streaming output is not silently empty. + """ + messages = [{"role": "user", "content": "what llm are you"}] + raw_chunks = _streaming_chunks_response_text() + + if sync_mode: + + def _iter_lines(): + for chunk in raw_chunks: + yield f"data: {chunk}" + yield "data: [DONE]" + + mock_resp = MagicMock() + mock_resp.iter_lines.return_value = _iter_lines() + mock_resp.status_code = 200 + mock_resp.headers = {"content-type": "text/event-stream"} + + with patch.object(HTTPHandler, "post", return_value=mock_resp) as mock_post: + response = completion( + model="cloudflare/@cf/nvidia/nemotron-mini-4b-instruct", + messages=messages, + max_tokens=15, + stream=True, + api_base=FAKE_API_BASE, + api_key=FAKE_API_KEY, + ) + chunks_received = list(response) + mock_post.assert_called_once() + else: + + async def _aiter_lines(): + for chunk in raw_chunks: + yield f"data: {chunk}" + yield "data: [DONE]" + + mock_resp = MagicMock() + mock_resp.aiter_lines.return_value = _aiter_lines() + mock_resp.status_code = 200 + mock_resp.headers = {"content-type": "text/event-stream"} + + async def _run(): + with patch.object( + AsyncHTTPHandler, "post", new_callable=AsyncMock, return_value=mock_resp + ) as mock_post: + resp = await acompletion( + model="cloudflare/@cf/nvidia/nemotron-mini-4b-instruct", + messages=messages, + max_tokens=15, + stream=True, + api_base=FAKE_API_BASE, + api_key=FAKE_API_KEY, + ) + received = [] + async for chunk in resp: + received.append(chunk) + mock_post.assert_called_once() + return received + + chunks_received = asyncio.run(_run()) + + assert len(chunks_received) > 0 + content = "".join( + c.choices[0].delta.content + for c in chunks_received + if c.choices[0].delta.content + ) + assert "language" in content.lower()