From 30bee8adc9dbec8e548bca17c2be03bf8151df10 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Mon, 14 Sep 2026 09:32:08 +0200 Subject: [PATCH 01/24] feat(hosted_vllm): opt in to forwarding assistant reasoning content --- .../litellm_core_utils/get_litellm_params.py | 4 +- .../hosted_vllm/chat/reasoning_content.md | 39 ++++++ .../llms/hosted_vllm/chat/transformation.py | 31 ++++- litellm/types/router.py | 1 + litellm/types/utils.py | 1 + .../test_router_forward_reasoning_content.py | 119 ++++++++++++++++++ .../test_hosted_vllm_chat_transformation.py | 84 +++++++++++++ .../test_litellm_completion_responses.py | 86 +++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 12 +- 9 files changed, 371 insertions(+), 6 deletions(-) create mode 100644 litellm/llms/hosted_vllm/chat/reasoning_content.md create mode 100644 tests/router_unit_tests/test_router_forward_reasoning_content.py diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py index edd2e88f95c..4eead556aa2 100644 --- a/litellm/litellm_core_utils/get_litellm_params.py +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -26,7 +26,7 @@ AWS_CREDENTIAL_KWARGS_KEYS: Final = frozenset( # Keys `completion()` forwards from its own kwargs into `get_litellm_params`, # which are otherwise invisible to it because that call site passes explicit # named arguments rather than `**kwargs`. -FORWARDED_KWARGS_KEYS: Final = AWS_CREDENTIAL_KWARGS_KEYS +FORWARDED_KWARGS_KEYS: Final = AWS_CREDENTIAL_KWARGS_KEYS | frozenset({"forward_reasoning_content"}) # Pre-define optional kwargs keys as frozenset for O(1) lookups # These are extracted from kwargs only if present, avoiding unnecessary .get() calls @@ -59,7 +59,7 @@ OPTIONAL_KWARGS_KEYS: Final = ( "use_xai_oauth", } ) - | AWS_CREDENTIAL_KWARGS_KEYS + | FORWARDED_KWARGS_KEYS ) # Backward-compatible alias for existing imports/tests. diff --git a/litellm/llms/hosted_vllm/chat/reasoning_content.md b/litellm/llms/hosted_vllm/chat/reasoning_content.md new file mode 100644 index 00000000000..b35093346d6 --- /dev/null +++ b/litellm/llms/hosted_vllm/chat/reasoning_content.md @@ -0,0 +1,39 @@ +# Forwarding assistant reasoning to hosted vLLM + +Some vLLM backends and chat templates accept previous assistant reasoning alongside content and tool calls. Set `forward_reasoning_content: true` on an individual model entry to forward the `reasoning_content` field supplied by the client + +```yaml +model_list: + - model_name: reasoning-history + litellm_params: + model: hosted_vllm/your-served-model + api_base: http://localhost:8000/v1 + forward_reasoning_content: true + - model_name: default-history + litellm_params: + model: hosted_vllm/your-served-model + api_base: http://localhost:8000/v1 + forward_reasoning_content: false +``` + +The default is false. Omitting the option or setting it to false retains the existing removal of assistant `reasoning_content`. The option is local to each request, applies only to `hosted_vllm`, and is not sent to the backend. It can also be passed to `litellm.completion` or `litellm.acompletion` + +When enabled, LiteLLM forwards the supplied field without trimming it or inserting it into visible `content`. Content and tool calls keep their existing transformations. It does not reconstruct missing reasoning or convert Anthropic `thinking_blocks`, signatures, or redacted thinking. Their existing handling is unchanged + +For Responses requests routed through Chat Completions, also set `use_chat_completions_api: true`. The option preserves the `reasoning_content` produced by that bridge from supported Responses reasoning input. It does not change native vLLM Responses requests or make opaque encrypted reasoning portable + +## Backend compatibility + +Enable this only after checking both your vLLM request parser and model chat template. A backend returning reasoning in its responses does not necessarily accept reasoning in previous assistant messages + +The vLLM v0.12.0 and v0.13.0 chat parsers accept `reasoning` and `reasoning_content` and expose both names to the template. Newer vLLM versions may normalize the deprecated input name `reasoning_content` to `reasoning` before rendering. LiteLLM forwards its existing `reasoning_content` field, without adding a duplicate `reasoning` field or selecting behavior by model name. Older releases, vendor forks and custom templates need separate verification + +Sources: [vLLM v0.12.0 chat parser](https://github.com/vllm-project/vllm/blob/v0.12.0/vllm/entrypoints/chat_utils.py#L1532), [vLLM v0.13.0 chat parser](https://github.com/vllm-project/vllm/blob/v0.13.0/vllm/entrypoints/chat_utils.py) + +## Template policy and performance + +`forward_reasoning_content` controls transport. A model-specific setting such as `chat_template_kwargs.preserve_thinking` controls which received history its template uses. In the Qwen3.8-Flash-Next template, `preserve_thinking: false` removes reasoning from earlier user turns but still retains reasoning within the current tool sequence. It does not mean that every assistant reasoning field should be deleted + +See [Qwen3.8-Flash-Next preserved thinking](https://huggingface.co/Qwen/Qwen3.8-Flash-Next#disable-preserved-thinking) + +Forwarding history can change the rendered prompt and token prefix. Boundary whitespace normalization may have no effect when the template trims that field. Validate rendered tokens with the actual tokenizer and template before drawing cache conclusions. This option provides no measured latency, cache-hit or reasoning-quality guarantee diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 29dc485732f..c7516adb8de 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -4,6 +4,7 @@ Translate from OpenAI's `/v1/chat/completions` to VLLM's `/v1/chat/completions` import json from collections.abc import Coroutine +from copy import deepcopy from typing import Any, Final, Literal, cast, overload from litellm.litellm_core_utils.prompt_templates.common_utils import ( @@ -145,6 +146,33 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): return ChatCompletionVideoObject(type="video_url", video_url=ChatCompletionVideoUrlObject(url=file_data)) raise ValueError("file_id or file_data is required") + def transform_request( + self, + model: str, + messages: list[AllMessageValues], # mutable-ok: provider request contract + optional_params: dict, # mutable-ok: provider request contract + litellm_params: dict, # mutable-ok: provider request contract + headers: dict, # mutable-ok: provider request contract + ) -> dict: # mutable-ok: provider request contract + request_messages: Final = deepcopy(messages) + if litellm_params.get("forward_reasoning_content") is not True: + for message in request_messages: + if message["role"] == "assistant": + message.pop("reasoning_content", None) + return super().transform_request(model, request_messages, optional_params, litellm_params, headers) + + async def async_transform_request( + self, + model: str, + messages: list[AllMessageValues], # mutable-ok: provider request contract + optional_params: dict, # mutable-ok: provider request contract + litellm_params: dict, # mutable-ok: provider request contract + headers: dict, # mutable-ok: provider request contract + ) -> dict: # mutable-ok: provider request contract + return await super().async_transform_request( + model, deepcopy(messages), optional_params, litellm_params, headers + ) + @overload def _transform_messages( self, messages: list[AllMessageValues], model: str, is_async: Literal[True] @@ -164,13 +192,12 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): """ Support translating: - video files from file_id or file_data to video_url - - thinking_blocks and reasoning_content on assistant messages are removed, + - thinking_blocks on assistant messages are removed, and content lists are converted to strings for vLLM compatibility """ for message in messages: if message["role"] == "assistant": message.pop("thinking_blocks", None) - message.pop("reasoning_content", None) existing_content = message.get("content") if isinstance(existing_content, list): text_parts = [] diff --git a/litellm/types/router.py b/litellm/types/router.py index c7363502017..040a2d2e026 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -357,6 +357,7 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): ) model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False + forward_reasoning_content: bool | None = False model_info: dict | None = None mock_response: str | ModelResponse | Exception | Any | None = None diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 00c55b35182..0ce75cd9893 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3826,6 +3826,7 @@ all_litellm_params = ( "budget_duration", "use_in_pass_through", "merge_reasoning_content_in_choices", + "forward_reasoning_content", "litellm_credential_name", "allowed_openai_params", "litellm_session_id", diff --git a/tests/router_unit_tests/test_router_forward_reasoning_content.py b/tests/router_unit_tests/test_router_forward_reasoning_content.py new file mode 100644 index 00000000000..548b5cca84c --- /dev/null +++ b/tests/router_unit_tests/test_router_forward_reasoning_content.py @@ -0,0 +1,119 @@ +import json +from copy import deepcopy +from typing import Final + +import httpx +import pytest +import respx + +import litellm +from litellm import Router + +URL: Final = "https://reasoning-test.invalid/v1/chat/completions" +MODEL: Final = "hosted_vllm/reasoning-test" +REASONING: Final = "Inspect both tool results before answering." + + +def _messages(): + return [ + {"role": "user", "content": "Compare both records"}, + { + "role": "assistant", + "content": None, + "reasoning_content": REASONING, + "tool_calls": [ + {"id": f"call_{index}", "type": "function", "function": {"name": "lookup", "arguments": "{}"}} + for index in (1, 2) + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": "first record"}, + {"role": "tool", "tool_call_id": "call_2", "content": "second record"}, + ] + + +def _route(mock: respx.MockRouter): + return mock.post(URL).respond( + 200, + json={ + "id": "chatcmpl-reasoning-test", + "object": "chat.completion", + "created": 1, + "model": "reasoning-test", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "Compared"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12}, + }, + ) + + +def _assert_wire(request: httpx.Request, enabled: bool): + body: Final = json.loads(request.content) + assert body["model"] == "reasoning-test" + assert "forward_reasoning_content" not in body + assert "forward_reasoning_content" not in request.content.decode() + messages: Final = body["messages"] + assert [message["role"] for message in messages] == ["user", "assistant", "tool", "tool"] + assert [tool["id"] for tool in messages[1]["tool_calls"]] == ["call_1", "call_2"] + assert [message["tool_call_id"] for message in messages[2:]] == ["call_1", "call_2"] + assert [message["content"] for message in messages[2:]] == ["first record", "second record"] + assert messages[1].get("reasoning_content") == (REASONING if enabled else None) + assert request.content.decode().count(REASONING) == int(enabled) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("async_mode", [False, True], ids=["completion", "acompletion"]) +@pytest.mark.parametrize("forward", [None, False, True], ids=["absent", "false", "true"]) +async def test_direct_completion_reasoning_flag_reaches_final_wire( + async_mode: bool, forward: bool | None, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + messages: Final = _messages() + original: Final = deepcopy(messages) + kwargs: Final = { + "model": MODEL, + "api_base": URL.removesuffix("/chat/completions"), + "api_key": "test-key", + "messages": messages, + **({} if forward is None else {"forward_reasoning_content": forward}), + } + with respx.mock(assert_all_called=True) as mock: + route: Final = _route(mock) + response: Final = await litellm.acompletion(**kwargs) if async_mode else litellm.completion(**kwargs) + assert response.choices[0].message.content == "Compared" + assert route.call_count == 1 + _assert_wire(route.calls[0].request, forward is True) + assert messages == original + + +@pytest.mark.asyncio +@pytest.mark.parametrize("async_mode", [False, True], ids=["completion", "acompletion"]) +async def test_router_aliases_isolate_reasoning_flag_on_same_backend(async_mode: bool, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + model_list: Final = [ + { + "model_name": alias, + "litellm_params": { + "model": MODEL, + "api_base": URL.removesuffix("/chat/completions"), + "api_key": "test-key", + **({} if forward is None else {"forward_reasoning_content": forward}), + }, + } + for alias, forward in (("default", None), ("disabled", False), ("enabled", True)) + ] + original_models: Final = deepcopy(model_list) + router: Final = Router(model_list=model_list, num_retries=0) + messages: Final = _messages() + original_messages: Final = deepcopy(messages) + with respx.mock(assert_all_called=True) as mock: + route: Final = _route(mock) + for alias in ("enabled", "default", "disabled", "enabled"): + response = ( + await router.acompletion(model=alias, messages=messages) + if async_mode + else router.completion(model=alias, messages=messages) + ) + assert response.choices[0].message.content == "Compared" + _assert_wire(route.calls[-1].request, alias == "enabled") + assert messages == original_messages + assert route.call_count == 4 + assert model_list == original_models diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index 82b05601a85..d7cc1a26729 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -1,6 +1,8 @@ import json +from copy import deepcopy from unittest.mock import MagicMock, patch +import pytest from litellm.constants import ( DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, @@ -9,6 +11,88 @@ from litellm.constants import ( from litellm.llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig +@pytest.mark.parametrize("params", [{}, {"forward_reasoning_content": False}, {"forward_reasoning_content": True}]) +@pytest.mark.parametrize("is_async", [False, True]) +@pytest.mark.asyncio +async def test_forward_reasoning_content_preserves_only_explicit_history(params, is_async): + config = HostedVLLMChatConfig() + messages = [ + {"role": "user", "content": "Check the counter"}, + { + "role": "assistant", + "content": None, + "reasoning_content": " synthetic history\n", + "thinking_blocks": [{"type": "thinking", "thinking": "Do not convert this", "signature": "sig"}], + "tool_calls": [{"id": "call_1", "type": "function", "function": {"name": "read", "arguments": "{}"}}], + }, + {"role": "tool", "tool_call_id": "call_1", "content": "7"}, + { + "role": "assistant", + "content": None, + "thinking_blocks": [{"type": "thinking", "thinking": "Never synthesize history", "signature": "sig"}], + "tool_calls": [{"id": "call_2", "type": "function", "function": {"name": "verify", "arguments": "{}"}}], + }, + {"role": "tool", "tool_call_id": "call_2", "content": "verified"}, + ] + original = deepcopy(messages) + arguments = dict( + model="qwen3.8-flash-next", messages=messages, optional_params={}, litellm_params=params, headers={} + ) + result = await config.async_transform_request(**arguments) if is_async else config.transform_request(**arguments) + expected = deepcopy(original) + expected[1].pop("thinking_blocks") + expected[3].pop("thinking_blocks") + if params.get("forward_reasoning_content") is not True: + expected[1].pop("reasoning_content") + assert result["messages"] == expected + assert messages == original + assert "forward_reasoning_content" not in result + + +@pytest.mark.asyncio +async def test_forward_reasoning_content_reused_config_and_caller_are_isolated(): + config = HostedVLLMChatConfig() + messages = [{"role": "assistant", "content": "answer", "reasoning_content": "synthetic history"}] + original = deepcopy(messages) + for enabled in (False, True, False, True): + for transform in (config.transform_request, config.async_transform_request): + result = transform( + model="qwen3.8-flash-next", + messages=messages, + optional_params={}, + litellm_params={"forward_reasoning_content": enabled}, + headers={}, + ) + if transform == config.async_transform_request: + result = await result + assert result["messages"] == (original if enabled else [{"role": "assistant", "content": "answer"}]) + assert messages == original + + +@pytest.mark.asyncio +async def test_forward_reasoning_content_keeps_async_content_conversion(): + class AsyncContentConfig(HostedVLLMChatConfig): + async def _async_transform_content_item(self, content_item): + return {"type": "image_url", "image_url": {"url": "data:image/png;base64,c3ludGhldGlj"}} + + config = AsyncContentConfig() + messages = [ + {"role": "user", "content": [{"type": "image_url", "image_url": {"url": "https://example.invalid/image.png"}}]}, + {"role": "assistant", "content": "answer", "reasoning_content": "synthetic history"}, + ] + original = deepcopy(messages) + result = await config.async_transform_request( + model="qwen3.8-flash-next", + messages=messages, + optional_params={}, + litellm_params={"forward_reasoning_content": True}, + headers={}, + ) + assert result["messages"][0]["content"][0]["image_url"]["url"] == "data:image/png;base64,c3ludGhldGlj" + assert result["messages"][1]["reasoning_content"] == "synthetic history" + assert messages == original + + def test_hosted_vllm_chat_transformation_file_url(): config = HostedVLLMChatConfig() video_url = "https://example.com/video.mp4" diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 3c78bbf79d7..8e522f73bd6 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2,7 +2,9 @@ import json from copy import deepcopy from typing import Final, Literal +import httpx import pytest +import respx from openai.types.responses.response_function_web_search import ( ActionFind, ActionOpenPage, @@ -30,6 +32,90 @@ from litellm.types.utils import ( ) +@pytest.mark.asyncio +@pytest.mark.parametrize("async_mode", [False, True], ids=["responses", "aresponses"]) +@pytest.mark.parametrize("forward", [None, False, True], ids=["absent", "false", "true"]) +@pytest.mark.parametrize("sequential", [False, True], ids=["parallel-tools", "sequential-tools"]) +async def test_hosted_vllm_responses_reasoning_and_parallel_tools_final_wire( + async_mode: bool, forward: bool | None, sequential: bool, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + reasoning: Final = "Inspect both tool results before answering." + next_reasoning: Final = "Use the first result to inspect the second." + input_items: Final = [ + {"role": "user", "content": "Compare both records"}, + { + "type": "reasoning", + "id": "rs_previous", + "summary": [], + "content": [{"type": "reasoning_text", "text": reasoning}], + }, + {"type": "function_call", "call_id": "call_1", "name": "lookup", "arguments": "{}"}, + *( + [ + {"type": "function_call_output", "call_id": "call_1", "output": "first record"}, + { + "type": "reasoning", + "id": "rs_next", + "summary": [], + "content": [{"type": "reasoning_text", "text": next_reasoning}], + }, + ] + if sequential + else [] + ), + {"type": "function_call", "call_id": "call_2", "name": "lookup", "arguments": "{}"}, + *([] if sequential else [{"type": "function_call_output", "call_id": "call_1", "output": "first record"}]), + {"type": "function_call_output", "call_id": "call_2", "output": "second record"}, + ] + original: Final = deepcopy(input_items) + kwargs: Final = { + "model": "hosted_vllm/reasoning-test", + "input": input_items, + "api_base": "https://responses-reasoning-test.invalid/v1", + "api_key": "test-key", + "use_chat_completions_api": True, + **({} if forward is None else {"forward_reasoning_content": forward}), + } + with respx.mock(assert_all_called=True) as mock: + route: Final = mock.post("https://responses-reasoning-test.invalid/v1/chat/completions").mock( + return_value=httpx.Response( + 200, + json={ + "id": "chatcmpl-bridge-reasoning", + "object": "chat.completion", + "created": 1, + "model": "reasoning-test", + "choices": [ + {"index": 0, "message": {"role": "assistant", "content": "Compared"}, "finish_reason": "stop"} + ], + "usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12}, + }, + ) + ) + response: Final = await litellm.aresponses(**kwargs) if async_mode else litellm.responses(**kwargs) + assert response.output[0].content[0].text == "Compared" + assert route.call_count == 1 + payload: Final = json.loads(route.calls[0].request.content) + assert payload["model"] == "reasoning-test" + assert "forward_reasoning_content" not in route.calls[0].request.content.decode() + assert "use_chat_completions_api" not in payload + messages: Final = payload["messages"] + assert [message["role"] for message in messages] == ( + ["user", "assistant", "tool", "assistant", "tool"] if sequential else ["user", "assistant", "tool", "tool"] + ) + assert [tool["id"] for message in messages for tool in message.get("tool_calls", [])] == ["call_1", "call_2"] + results: Final = [message for message in messages if message["role"] == "tool"] + assert [message["tool_call_id"] for message in results] == ["call_1", "call_2"] + assert [message["content"] for message in results] == ["first record", "second record"] + assert messages[1].get("reasoning_content") == (reasoning if forward is True else None) + assert route.calls[0].request.content.decode().count(reasoning) == int(forward is True) + if sequential: + assert messages[3].get("reasoning_content") == (next_reasoning if forward is True else None) + assert route.calls[0].request.content.decode().count(next_reasoning) == int(forward is True) + assert input_items == original + + class TestLiteLLMCompletionResponsesConfig: def test_transform_input_file_item_to_file_item_with_file_id(self): """Test transformation of input_file item with file_id to Chat Completion file format""" diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 7eadaa6c991..e118181ba18 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -16781,7 +16781,6 @@ export interface paths { * - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking. * - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" } * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. - * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) @@ -16887,7 +16886,6 @@ export interface paths { * - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking. * - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" } * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. - * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) @@ -29713,6 +29711,11 @@ export interface components { default_api_key_tpm_limit?: number | null; /** Drop Params */ drop_params?: boolean | string | null; + /** + * Forward Reasoning Content + * @default false + */ + forward_reasoning_content: boolean | null; /** Gcs Bucket Name */ gcs_bucket_name?: string | null; /** Google Maps Grounding Cost Per Query */ @@ -39927,6 +39930,11 @@ export interface components { default_api_key_tpm_limit?: number | null; /** Drop Params */ drop_params?: boolean | string | null; + /** + * Forward Reasoning Content + * @default false + */ + forward_reasoning_content: boolean | null; /** Gcs Bucket Name */ gcs_bucket_name?: string | null; /** Google Maps Grounding Cost Per Query */ From 8a82ea72ac6e393207aac1e8995a169fa4cdc68e Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Mon, 14 Sep 2026 11:09:57 +0200 Subject: [PATCH 02/24] fix(hosted_vllm): isolate forwarded reasoning in response caches --- litellm/caching/caching.py | 7 ++ .../hosted_vllm/chat/reasoning_content.md | 39 ------ .../test_router_forward_reasoning_content.py | 111 ++++++++++++++++++ tests/test_litellm/caching/test_caching.py | 46 ++++++++ 4 files changed, 164 insertions(+), 39 deletions(-) delete mode 100644 litellm/llms/hosted_vllm/chat/reasoning_content.md diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index d6dd2a073af..27f13abb140 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -366,6 +366,13 @@ class Cache: param_value = kwargs[param] cache_key += f"{param}: {param_value}" + nested_litellm_params: Final = kwargs.get("litellm_params") or {} + forward_reasoning_content: Final = kwargs.get( + "forward_reasoning_content", nested_litellm_params.get("forward_reasoning_content") + ) + if forward_reasoning_content is True: + cache_key += "forward_reasoning_content: True" + if is_semantic_cache: cache_key += self._get_semantic_cache_tenant_scope(kwargs) diff --git a/litellm/llms/hosted_vllm/chat/reasoning_content.md b/litellm/llms/hosted_vllm/chat/reasoning_content.md deleted file mode 100644 index b35093346d6..00000000000 --- a/litellm/llms/hosted_vllm/chat/reasoning_content.md +++ /dev/null @@ -1,39 +0,0 @@ -# Forwarding assistant reasoning to hosted vLLM - -Some vLLM backends and chat templates accept previous assistant reasoning alongside content and tool calls. Set `forward_reasoning_content: true` on an individual model entry to forward the `reasoning_content` field supplied by the client - -```yaml -model_list: - - model_name: reasoning-history - litellm_params: - model: hosted_vllm/your-served-model - api_base: http://localhost:8000/v1 - forward_reasoning_content: true - - model_name: default-history - litellm_params: - model: hosted_vllm/your-served-model - api_base: http://localhost:8000/v1 - forward_reasoning_content: false -``` - -The default is false. Omitting the option or setting it to false retains the existing removal of assistant `reasoning_content`. The option is local to each request, applies only to `hosted_vllm`, and is not sent to the backend. It can also be passed to `litellm.completion` or `litellm.acompletion` - -When enabled, LiteLLM forwards the supplied field without trimming it or inserting it into visible `content`. Content and tool calls keep their existing transformations. It does not reconstruct missing reasoning or convert Anthropic `thinking_blocks`, signatures, or redacted thinking. Their existing handling is unchanged - -For Responses requests routed through Chat Completions, also set `use_chat_completions_api: true`. The option preserves the `reasoning_content` produced by that bridge from supported Responses reasoning input. It does not change native vLLM Responses requests or make opaque encrypted reasoning portable - -## Backend compatibility - -Enable this only after checking both your vLLM request parser and model chat template. A backend returning reasoning in its responses does not necessarily accept reasoning in previous assistant messages - -The vLLM v0.12.0 and v0.13.0 chat parsers accept `reasoning` and `reasoning_content` and expose both names to the template. Newer vLLM versions may normalize the deprecated input name `reasoning_content` to `reasoning` before rendering. LiteLLM forwards its existing `reasoning_content` field, without adding a duplicate `reasoning` field or selecting behavior by model name. Older releases, vendor forks and custom templates need separate verification - -Sources: [vLLM v0.12.0 chat parser](https://github.com/vllm-project/vllm/blob/v0.12.0/vllm/entrypoints/chat_utils.py#L1532), [vLLM v0.13.0 chat parser](https://github.com/vllm-project/vllm/blob/v0.13.0/vllm/entrypoints/chat_utils.py) - -## Template policy and performance - -`forward_reasoning_content` controls transport. A model-specific setting such as `chat_template_kwargs.preserve_thinking` controls which received history its template uses. In the Qwen3.8-Flash-Next template, `preserve_thinking: false` removes reasoning from earlier user turns but still retains reasoning within the current tool sequence. It does not mean that every assistant reasoning field should be deleted - -See [Qwen3.8-Flash-Next preserved thinking](https://huggingface.co/Qwen/Qwen3.8-Flash-Next#disable-preserved-thinking) - -Forwarding history can change the rendered prompt and token prefix. Boundary whitespace normalization may have no effect when the template trims that field. Validate rendered tokens with the actual tokenizer and template before drawing cache conclusions. This option provides no measured latency, cache-hit or reasoning-quality guarantee diff --git a/tests/router_unit_tests/test_router_forward_reasoning_content.py b/tests/router_unit_tests/test_router_forward_reasoning_content.py index 548b5cca84c..e0d8cc3b4bc 100644 --- a/tests/router_unit_tests/test_router_forward_reasoning_content.py +++ b/tests/router_unit_tests/test_router_forward_reasoning_content.py @@ -1,3 +1,4 @@ +import asyncio import json from copy import deepcopy from typing import Final @@ -8,6 +9,8 @@ import respx import litellm from litellm import Router +from litellm.caching.caching import Cache +from litellm.caching.caching_handler import _PENDING_CACHE_WRITES URL: Final = "https://reasoning-test.invalid/v1/chat/completions" MODEL: Final = "hosted_vllm/reasoning-test" @@ -117,3 +120,111 @@ async def test_router_aliases_isolate_reasoning_flag_on_same_backend(async_mode: assert messages == original_messages assert route.call_count == 4 assert model_list == original_models + + +@pytest.mark.asyncio +@pytest.mark.parametrize("async_mode", [False, True], ids=["sync", "async"]) +@pytest.mark.parametrize("surface", ["sdk", "router", "responses"]) +async def test_local_cache_separates_forwarded_reasoning_history( + async_mode: bool, surface: str, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "cache", Cache(type="local", namespace="reasoning-cache-test")) + messages: Final = _messages() + original_messages: Final = deepcopy(messages) + input_items: Final = [ + {"role": "user", "content": "Compare both records"}, + { + "type": "reasoning", + "id": "rs_previous", + "summary": [], + "content": [{"type": "reasoning_text", "text": REASONING}], + }, + {"type": "function_call", "call_id": "call_1", "name": "lookup", "arguments": "{}"}, + {"type": "function_call", "call_id": "call_2", "name": "lookup", "arguments": "{}"}, + {"type": "function_call_output", "call_id": "call_1", "output": "first record"}, + {"type": "function_call_output", "call_id": "call_2", "output": "second record"}, + ] + original_input: Final = deepcopy(input_items) + router: Final = Router( + model_list=[ + { + "model_name": alias, + "litellm_params": { + "model": MODEL, + "api_base": URL.removesuffix("/chat/completions"), + "api_key": "test-key", + **({} if enabled is None else {"forward_reasoning_content": enabled}), + }, + } + for alias, enabled in (("default", None), ("disabled", False), ("enabled", True)) + ], + cache_responses=True, + caching_groups=[("default", "disabled", "enabled")], + num_retries=0, + ) + + def backend(request: httpx.Request) -> httpx.Response: + body: Final = json.loads(request.content) + assert "forward_reasoning_content" not in body + enabled: Final = body["messages"][1].get("reasoning_content") == REASONING + return httpx.Response( + 200, + json={ + "id": "provider-cache-reasoning", + "object": "chat.completion", + "created": 1, + "model": "reasoning-test", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "forwarded" if enabled else "omitted"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12}, + }, + ) + + with respx.mock(assert_all_called=True) as mock: + route: Final = mock.post(URL).mock(side_effect=backend) + for alias, forward, expected_calls in ( + ("default", None, 1), + ("disabled", False, 1), + ("enabled", True, 2), + ("disabled", False, 2), + ("enabled", True, 2), + ): + kwargs: Final = { + "model": MODEL, + "api_base": URL.removesuffix("/chat/completions"), + "api_key": "test-key", + "caching": True, + **({} if forward is None else {"forward_reasoning_content": forward}), + } + if surface == "router": + response = ( + await router.acompletion(model=alias, messages=messages) + if async_mode + else router.completion(model=alias, messages=messages) + ) + elif surface == "responses": + response = ( + await litellm.aresponses(input=input_items, use_chat_completions_api=True, **kwargs) + if async_mode + else litellm.responses(input=input_items, use_chat_completions_api=True, **kwargs) + ) + else: + response = ( + await litellm.acompletion(messages=messages, **kwargs) + if async_mode + else litellm.completion(messages=messages, **kwargs) + ) + await asyncio.gather(*tuple(_PENDING_CACHE_WRITES)) + content: Final = ( + response.output[0].content[0].text if surface == "responses" else response.choices[0].message.content + ) + assert content == ("forwarded" if forward is True else "omitted") + assert route.call_count == expected_calls + assert messages == original_messages + assert input_items == original_input diff --git a/tests/test_litellm/caching/test_caching.py b/tests/test_litellm/caching/test_caching.py index 4d0ec0fb677..12e77118b6d 100644 --- a/tests/test_litellm/caching/test_caching.py +++ b/tests/test_litellm/caching/test_caching.py @@ -90,6 +90,52 @@ def _semantic_cache(**cache_kwargs): ) +@pytest.mark.parametrize("semantic", [False, True]) +def test_reasoning_forwarding_cache_scope_preserves_groups_namespace_and_tenant(semantic): + cache = _semantic_cache(namespace="reasoning-test") if semantic else Cache(type="local", namespace="reasoning-test") + + def key(alias="first", tenant="tenant-a", namespace="reasoning-test", nested=False, forward=None, prompt="hi"): + return cache.get_cache_key( + model="hosted_vllm/reasoning-test", + messages=[{"role": "user", "content": prompt}], + metadata={"model_group": alias, "caching_groups": [("first", "second")], "user_api_key": tenant}, + cache={"namespace": namespace}, + **( + {"litellm_params": {"forward_reasoning_content": forward}} + if nested + else {"forward_reasoning_content": forward} + ), + ) + + default = key() + enabled = key(forward=True) + assert default == key(forward=False) == key(nested=True, forward=False) + assert enabled != default + assert enabled == key(nested=True, forward=True) == key(alias="second", forward=True) + assert enabled.startswith("reasoning-test:") + assert enabled != key(namespace="other-namespace", forward=True) + if semantic: + assert enabled != key(tenant="tenant-b", forward=True) + assert enabled == key(prompt="hello", forward=True) + else: + assert enabled != key(prompt="hello", forward=True) + + +def test_reasoning_forwarding_cache_key_preserves_legacy_disabled_key_and_top_level_precedence(): + cache = Cache(type="local", namespace="reasoning-cache-test") + request = {"model": "hosted_vllm/reasoning-test", "messages": [{"role": "user", "content": "hi"}]} + legacy = "reasoning-cache-test:fca1120c8360f4b9ca0cd9b52f981f290a6eec25a8c6256033a81edcc713618c" + assert cache.get_cache_key(**request) == legacy + assert cache.get_cache_key(**request, forward_reasoning_content=False) == legacy + assert ( + cache.get_cache_key( + **request, forward_reasoning_content=False, litellm_params={"forward_reasoning_content": True} + ) + == legacy + ) + assert cache.get_cache_key(**request, litellm_params={"forward_reasoning_content": True}) != legacy + + @pytest.mark.parametrize( "cache_type", [LiteLLMCacheType.REDIS_SEMANTIC, LiteLLMCacheType.VALKEY_SEMANTIC], From 9e47c58e1d656cddaefd51e53fc3940d14ccfced Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 08:02:35 +0200 Subject: [PATCH 03/24] feat: configure historical assistant reasoning field --- litellm/caching/caching.py | 5 + .../litellm_core_utils/get_litellm_params.py | 4 +- .../reasoning_content_utils.py | 37 +++++ .../llms/hosted_vllm/chat/transformation.py | 7 +- .../llms/openai/chat/gpt_transformation.py | 17 ++- litellm/proxy/_lazy_openapi_snapshot.json | 2 +- litellm/types/router.py | 1 + litellm/types/utils.py | 1 + .../test_router_forward_reasoning_content.py | 54 +++++-- tests/test_litellm/caching/test_caching.py | 39 +++++ .../test_hosted_vllm_chat_transformation.py | 142 ++++++++++++++++++ .../test_litellm_completion_responses.py | 38 ++++- ui/litellm-dashboard/src/lib/http/schema.d.ts | 12 ++ 13 files changed, 338 insertions(+), 21 deletions(-) create mode 100644 litellm/litellm_core_utils/reasoning_content_utils.py diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index 27f13abb140..3b681b29859 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -372,6 +372,11 @@ class Cache: ) if forward_reasoning_content is True: cache_key += "forward_reasoning_content: True" + reasoning_content_field: Final = kwargs.get( + "reasoning_content_field", nested_litellm_params.get("reasoning_content_field") + ) + if reasoning_content_field == "reasoning": + cache_key += "reasoning_content_field: reasoning" if is_semantic_cache: cache_key += self._get_semantic_cache_tenant_scope(kwargs) diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py index 4eead556aa2..fca670103f7 100644 --- a/litellm/litellm_core_utils/get_litellm_params.py +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -26,7 +26,9 @@ AWS_CREDENTIAL_KWARGS_KEYS: Final = frozenset( # Keys `completion()` forwards from its own kwargs into `get_litellm_params`, # which are otherwise invisible to it because that call site passes explicit # named arguments rather than `**kwargs`. -FORWARDED_KWARGS_KEYS: Final = AWS_CREDENTIAL_KWARGS_KEYS | frozenset({"forward_reasoning_content"}) +FORWARDED_KWARGS_KEYS: Final = AWS_CREDENTIAL_KWARGS_KEYS | frozenset( + {"forward_reasoning_content", "reasoning_content_field"} +) # Pre-define optional kwargs keys as frozenset for O(1) lookups # These are extracted from kwargs only if present, avoiding unnecessary .get() calls diff --git a/litellm/litellm_core_utils/reasoning_content_utils.py b/litellm/litellm_core_utils/reasoning_content_utils.py new file mode 100644 index 00000000000..ace40b18bc8 --- /dev/null +++ b/litellm/litellm_core_utils/reasoning_content_utils.py @@ -0,0 +1,37 @@ +from collections.abc import Mapping, Sequence +from copy import deepcopy +from types import MappingProxyType +from typing import ( + Final, + cast, # noqa: TID251 # Preserves arbitrary provider fields without lossy TypedDict validation. +) + +from litellm.types.llms.openai import AllMessageValues + + +def normalize_reasoning_content( + messages: Sequence[AllMessageValues], *, forward: bool = True +) -> list[AllMessageValues]: # mutable-ok: provider request contract + def normalize_message(message: AllMessageValues) -> AllMessageValues: + if message["role"] != "assistant": + return message + history: Final[Mapping[str, object]] = message + reasoning: Final = ( + history.get("reasoning") if history.get("reasoning") is not None else history.get("reasoning_content") + ) + normalized: Final[Mapping[str, object]] = MappingProxyType( + { + **MappingProxyType( + {key: value for key, value in history.items() if key not in ("reasoning", "reasoning_content")} + ), + **( + MappingProxyType({"reasoning": reasoning}) + if forward and reasoning is not None + else MappingProxyType({}) + ), + } + ) + result: Final = dict(normalized) # mutable-ok: provider request contract + return cast(AllMessageValues, result) # cast-ok: only optional reasoning keys change + + return [normalize_message(message) for message in deepcopy(messages)] # mutable-ok: provider request contract diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index c7516adb8de..fe28e11a2ee 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -11,6 +11,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, ) from litellm.litellm_core_utils.prompt_templates.factory import _parse_mime_type +from litellm.litellm_core_utils.reasoning_content_utils import normalize_reasoning_content from litellm.litellm_core_utils.reasoning_effort_utils import ( reasoning_effort_from_thinking_budget, ) @@ -154,7 +155,11 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): litellm_params: dict, # mutable-ok: provider request contract headers: dict, # mutable-ok: provider request contract ) -> dict: # mutable-ok: provider request contract - request_messages: Final = deepcopy(messages) + request_messages: Final = ( + normalize_reasoning_content(messages, forward=litellm_params.get("forward_reasoning_content") is True) + if litellm_params.get("reasoning_content_field") == "reasoning" + else deepcopy(messages) + ) if litellm_params.get("forward_reasoning_content") is not True: for message in request_messages: if message["role"] == "assistant": diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index 9b410cf073e..b3824ef77a7 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -30,6 +30,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_convert_url_to_base64, convert_url_to_base64, ) +from litellm.litellm_core_utils.reasoning_content_utils import normalize_reasoning_content from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException @@ -477,7 +478,13 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): Returns: dict: The transformed request. Sent as the body of the API call. """ - messages = self._transform_messages(messages=messages, model=model) + request_messages: Final = ( + normalize_reasoning_content(messages) + if litellm_params.get("custom_llm_provider") == "openai" + and litellm_params.get("reasoning_content_field") == "reasoning" + else messages + ) + messages = self._transform_messages(messages=request_messages, model=model) if not self._should_preserve_cache_control_for_endpoint( litellm_params.get("custom_llm_provider"), litellm_params.get("api_base") ): @@ -506,7 +513,13 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): litellm_params: dict, headers: dict, ) -> dict: - transformed_messages = await self._transform_messages(messages=messages, model=model, is_async=True) + request_messages: Final = ( + normalize_reasoning_content(messages) + if litellm_params.get("custom_llm_provider") == "openai" + and litellm_params.get("reasoning_content_field") == "reasoning" + else messages + ) + transformed_messages = await self._transform_messages(messages=request_messages, model=model, is_async=True) if not self._should_preserve_cache_control_for_endpoint( litellm_params.get("custom_llm_provider"), litellm_params.get("api_base") ): diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 7a110eff080..4c4f2164a59 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -18968,7 +18968,7 @@ } } }, - "description": "\n Unified rate-limit error.\n\n Every rate-limit condition surfaced by litellm \u2014 whether it originated from\n an upstream LLM provider, a vendor batch endpoint, or one of litellm's own\n proxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\n max-iterations, etc.) \u2014 is raised as an instance of this class.\n\n The :attr:`category` attribute lets callers distinguish the source. See\n :class:`RateLimitErrorCategory` for the available values.\n " + "description": "\nUnified rate-limit error.\n\nEvery rate-limit condition surfaced by litellm \u2014 whether it originated from\nan upstream LLM provider, a vendor batch endpoint, or one of litellm's own\nproxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\nmax-iterations, etc.) \u2014 is raised as an instance of this class.\n\nThe :attr:`category` attribute lets callers distinguish the source. See\n:class:`RateLimitErrorCategory` for the available values.\n" }, "500": { "content": { diff --git a/litellm/types/router.py b/litellm/types/router.py index 040a2d2e026..ffbdbc96d10 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -358,6 +358,7 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False forward_reasoning_content: bool | None = False + reasoning_content_field: Literal["reasoning_content", "reasoning"] = "reasoning_content" model_info: dict | None = None mock_response: str | ModelResponse | Exception | Any | None = None diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 0ce75cd9893..8ab4d55437e 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3827,6 +3827,7 @@ all_litellm_params = ( "use_in_pass_through", "merge_reasoning_content_in_choices", "forward_reasoning_content", + "reasoning_content_field", "litellm_credential_name", "allowed_openai_params", "litellm_session_id", diff --git a/tests/router_unit_tests/test_router_forward_reasoning_content.py b/tests/router_unit_tests/test_router_forward_reasoning_content.py index e0d8cc3b4bc..24923154978 100644 --- a/tests/router_unit_tests/test_router_forward_reasoning_content.py +++ b/tests/router_unit_tests/test_router_forward_reasoning_content.py @@ -124,10 +124,14 @@ async def test_router_aliases_isolate_reasoning_flag_on_same_backend(async_mode: @pytest.mark.asyncio @pytest.mark.parametrize("async_mode", [False, True], ids=["sync", "async"]) -@pytest.mark.parametrize("surface", ["sdk", "router", "responses"]) +@pytest.mark.parametrize("surface", ["sdk", "router", "responses", "router-responses"]) +@pytest.mark.parametrize("provider", ["hosted_vllm", "openai"]) +@pytest.mark.parametrize("normalize", [False, True]) async def test_local_cache_separates_forwarded_reasoning_history( - async_mode: bool, surface: str, monkeypatch: pytest.MonkeyPatch + async_mode: bool, surface: str, provider: str, normalize: bool, monkeypatch: pytest.MonkeyPatch ): + if provider == "openai" and not normalize: + pytest.skip("OpenAI does not apply hosted_vllm forwarding policy") monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) monkeypatch.setattr(litellm, "cache", Cache(type="local", namespace="reasoning-cache-test")) messages: Final = _messages() @@ -151,10 +155,22 @@ async def test_local_cache_separates_forwarded_reasoning_history( { "model_name": alias, "litellm_params": { - "model": MODEL, + "model": f"{provider}/reasoning-test", "api_base": URL.removesuffix("/chat/completions"), "api_key": "test-key", - **({} if enabled is None else {"forward_reasoning_content": enabled}), + "use_chat_completions_api": True, + **( + { + "forward_reasoning_content": True, + **( + {} + if enabled is None + else {"reasoning_content_field": "reasoning" if enabled else "reasoning_content"} + ), + } + if normalize + else ({} if enabled is None else {"forward_reasoning_content": enabled}) + ), }, } for alias, enabled in (("default", None), ("disabled", False), ("enabled", True)) @@ -167,7 +183,8 @@ async def test_local_cache_separates_forwarded_reasoning_history( def backend(request: httpx.Request) -> httpx.Response: body: Final = json.loads(request.content) assert "forward_reasoning_content" not in body - enabled: Final = body["messages"][1].get("reasoning_content") == REASONING + assert "reasoning_content_field" not in body + enabled: Final = body["messages"][1].get("reasoning" if normalize else "reasoning_content") == REASONING return httpx.Response( 200, json={ @@ -196,13 +213,30 @@ async def test_local_cache_separates_forwarded_reasoning_history( ("enabled", True, 2), ): kwargs: Final = { - "model": MODEL, + "model": f"{provider}/reasoning-test", "api_base": URL.removesuffix("/chat/completions"), "api_key": "test-key", "caching": True, - **({} if forward is None else {"forward_reasoning_content": forward}), + **( + { + "forward_reasoning_content": True, + **( + {} + if forward is None + else {"reasoning_content_field": "reasoning" if forward else "reasoning_content"} + ), + } + if normalize + else ({} if forward is None else {"forward_reasoning_content": forward}) + ), } - if surface == "router": + if surface == "router-responses": + response = ( + await router.aresponses(model=alias, input=input_items) + if async_mode + else router.responses(model=alias, input=input_items) + ) + elif surface == "router": response = ( await router.acompletion(model=alias, messages=messages) if async_mode @@ -222,7 +256,9 @@ async def test_local_cache_separates_forwarded_reasoning_history( ) await asyncio.gather(*tuple(_PENDING_CACHE_WRITES)) content: Final = ( - response.output[0].content[0].text if surface == "responses" else response.choices[0].message.content + response.output[0].content[0].text + if surface in ("responses", "router-responses") + else response.choices[0].message.content ) assert content == ("forwarded" if forward is True else "omitted") assert route.call_count == expected_calls diff --git a/tests/test_litellm/caching/test_caching.py b/tests/test_litellm/caching/test_caching.py index 12e77118b6d..250ff93ee48 100644 --- a/tests/test_litellm/caching/test_caching.py +++ b/tests/test_litellm/caching/test_caching.py @@ -298,3 +298,42 @@ def test_exact_cache_key_includes_anthropic_messages_params(anthropic_param): assert baseline != cache.get_cache_key( model="claude-sonnet-4-5", messages=messages, **anthropic_param ) + + +@pytest.mark.parametrize("semantic", [False, True]) +@pytest.mark.parametrize("provider", ["hosted_vllm", "openai"]) +def test_reasoning_field_cache_identity(semantic: bool, provider: str): + cache = _semantic_cache(namespace="history-field") if semantic else Cache(type="local", namespace="history-field") + request = { + "model": f"{provider}/reasoning-test", + "messages": [{"role": "user", "content": "hi"}], + "metadata": {"model_group": "first", "caching_groups": [("first", "second")], "user_api_key": "tenant-a"}, + } + legacy = cache.get_cache_key(**request) + assert cache.get_cache_key(**request, reasoning_content_field="reasoning_content") == legacy + assert cache.get_cache_key(**request, litellm_params={"reasoning_content_field": "reasoning_content"}) == legacy + normalized = cache.get_cache_key(**request, reasoning_content_field="reasoning") + assert normalized != legacy + assert normalized == cache.get_cache_key(**request, litellm_params={"reasoning_content_field": "reasoning"}) + assert normalized != cache.get_cache_key( + **request, reasoning_content_field="reasoning", forward_reasoning_content=True + ) + assert ( + cache.get_cache_key( + **request, + reasoning_content_field="reasoning_content", + litellm_params={"reasoning_content_field": "reasoning"}, + ) + == legacy + ) + assert normalized == cache.get_cache_key( + **{**request, "metadata": {**request["metadata"], "model_group": "second"}}, reasoning_content_field="reasoning" + ) + assert normalized != cache.get_cache_key( + **request, reasoning_content_field="reasoning", cache={"namespace": "other"} + ) + if semantic: + assert normalized != cache.get_cache_key( + **{**request, "metadata": {**request["metadata"], "user_api_key": "tenant-b"}}, + reasoning_content_field="reasoning", + ) diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index d7cc1a26729..0b70e763f34 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -1,5 +1,11 @@ import json from copy import deepcopy +from typing import Final + +import httpx +import respx + +import litellm from unittest.mock import MagicMock, patch import pytest @@ -517,3 +523,139 @@ def test_hosted_vllm_custom_tools_use_top_level_input_schema(): assert tools[0]["function"]["name"] == "search" assert tools[0]["function"]["description"] == "Search docs" assert tools[0]["function"]["parameters"] == input_schema + + +@pytest.mark.asyncio +@pytest.mark.parametrize("provider", ["hosted_vllm", "openai"]) +@pytest.mark.parametrize("is_async", [False, True]) +@pytest.mark.parametrize("via_router", [False, True]) +@pytest.mark.parametrize("forward", [None, False, True]) +@pytest.mark.parametrize( + "history, expected", + [ + ({"reasoning_content": "source"}, "source"), + ({"reasoning": "target"}, "target"), + ({"reasoning_content": "same", "reasoning": "same"}, "same"), + ({"reasoning_content": "source", "reasoning": "target"}, "target"), + ({"reasoning_content": "source", "reasoning": None}, "source"), + ({"reasoning_content": "source", "reasoning": ""}, ""), + ({"reasoning_content": "", "reasoning": None}, ""), + ({"reasoning_content": None, "reasoning": None}, None), + ({}, None), + ], +) +async def test_reasoning_field_sdk_router_final_wire( + provider: str, + is_async: bool, + via_router: bool, + forward: bool | None, + history: dict[str, str | None], + expected: str | None, + monkeypatch: pytest.MonkeyPatch, +): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + messages: Final = [ + {"role": "user", "content": "Check both records"}, + *[ + message + for index in (1, 2) + for message in ( + { + "role": "assistant", + "content": f"Checking {index}", + **history, + "tool_calls": [ + {"id": f"call_{index}", "type": "function", "function": {"name": "lookup", "arguments": "{}"}} + ], + }, + {"role": "tool", "tool_call_id": f"call_{index}", "content": f"record {index}"}, + ) + ], + ] + original: Final = deepcopy(messages) + base_params: Final = { + "model": f"{provider}/reasoning-test", + "api_base": "https://history-field.invalid/v1", + "api_key": "test-key", + **({} if forward is None else {"forward_reasoning_content": forward}), + } + router: Final = litellm.Router( + model_list=[ + {"model_name": alias, "litellm_params": {**base_params, **params}} + for alias, params in ( + ("legacy", {}), + ("normalized", {"reasoning_content_field": "reasoning"}), + ("explicit-default", {"reasoning_content_field": "reasoning_content"}), + ) + ], + num_retries=0, + ) + with respx.mock(assert_all_called=True) as mock: + route: Final = mock.post("https://history-field.invalid/v1/chat/completions").respond( + 200, + json={ + "id": "chatcmpl-history", + "object": "chat.completion", + "created": 1, + "model": "reasoning-test", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "Done"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 1, "total_tokens": 11}, + }, + ) + for alias, field in (("normalized", "reasoning"), ("legacy", None), ("explicit-default", "reasoning_content")): + kwargs: Final = ( + {"model": alias, "messages": messages} + if via_router + else { + **base_params, + "messages": messages, + **({} if field is None else {"reasoning_content_field": field}), + } + ) + client: Final = router if via_router else litellm + response: Final = await client.acompletion(**kwargs) if is_async else client.completion(**kwargs) + assert response.choices[0].message.content == "Done" + payload: Final = json.loads(route.calls[-1].request.content) + forwarded: Final = provider == "openai" or forward is True + expected_history: Final = ( + ({"reasoning": expected} if expected is not None and forwarded else {}) + if field == "reasoning" + else { + key: value + for key, value in history.items() + if value is not None and (forwarded or key != "reasoning_content") + } + ) + assert payload["messages"] == [ + ( + {**{key: value for key, value in message.items() if key not in history}, **expected_history} + if message["role"] == "assistant" + else message + ) + for message in original + ] + assert "reasoning_content_field" not in payload + assert "forward_reasoning_content" not in payload + assert messages == original + assert route.call_count == 3 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("is_async", [False, True]) +@pytest.mark.parametrize("provider", ["deepinfra", "together_ai", None]) +async def test_reasoning_field_does_not_apply_to_inherited_provider(provider: str | None, is_async: bool): + from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig + + config: Final = OpenAIGPTConfig() + messages: Final = [{"role": "assistant", "content": "Done", "reasoning_content": "source", "reasoning": "target"}] + original: Final = deepcopy(messages) + kwargs: Final = { + "model": "reasoning-test", + "messages": messages, + "optional_params": {}, + "headers": {}, + "litellm_params": {"custom_llm_provider": provider, "reasoning_content_field": "reasoning"}, + } + result: Final = await config.async_transform_request(**kwargs) if is_async else config.transform_request(**kwargs) + assert result["messages"] == original + assert messages == original diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 8e522f73bd6..35b43d1b35c 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -36,8 +36,17 @@ from litellm.types.utils import ( @pytest.mark.parametrize("async_mode", [False, True], ids=["responses", "aresponses"]) @pytest.mark.parametrize("forward", [None, False, True], ids=["absent", "false", "true"]) @pytest.mark.parametrize("sequential", [False, True], ids=["parallel-tools", "sequential-tools"]) +@pytest.mark.parametrize("provider", ["hosted_vllm", "openai"]) +@pytest.mark.parametrize("field", [None, "reasoning"]) +@pytest.mark.parametrize("via_router", [False, True]) async def test_hosted_vllm_responses_reasoning_and_parallel_tools_final_wire( - async_mode: bool, forward: bool | None, sequential: bool, monkeypatch: pytest.MonkeyPatch + async_mode: bool, + forward: bool | None, + sequential: bool, + provider: str, + field: str | None, + via_router: bool, + monkeypatch: pytest.MonkeyPatch, ): monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) reasoning: Final = "Inspect both tool results before answering." @@ -70,13 +79,27 @@ async def test_hosted_vllm_responses_reasoning_and_parallel_tools_final_wire( ] original: Final = deepcopy(input_items) kwargs: Final = { - "model": "hosted_vllm/reasoning-test", + "model": f"{provider}/reasoning-test", "input": input_items, "api_base": "https://responses-reasoning-test.invalid/v1", "api_key": "test-key", "use_chat_completions_api": True, **({} if forward is None else {"forward_reasoning_content": forward}), + **({} if field is None else {"reasoning_content_field": field}), } + router: Final = litellm.Router( + model_list=[ + { + "model_name": "history-alias", + "litellm_params": {key: value for key, value in kwargs.items() if key != "input"}, + } + ], + num_retries=0, + ) + client: Final = router if via_router else litellm + request: Final = {"model": "history-alias", "input": input_items} if via_router else kwargs + forwarded: Final = provider == "openai" or forward is True + output_field: Final = field or "reasoning_content" with respx.mock(assert_all_called=True) as mock: route: Final = mock.post("https://responses-reasoning-test.invalid/v1/chat/completions").mock( return_value=httpx.Response( @@ -93,13 +116,14 @@ async def test_hosted_vllm_responses_reasoning_and_parallel_tools_final_wire( }, ) ) - response: Final = await litellm.aresponses(**kwargs) if async_mode else litellm.responses(**kwargs) + response: Final = await client.aresponses(**request) if async_mode else client.responses(**request) assert response.output[0].content[0].text == "Compared" assert route.call_count == 1 payload: Final = json.loads(route.calls[0].request.content) assert payload["model"] == "reasoning-test" assert "forward_reasoning_content" not in route.calls[0].request.content.decode() assert "use_chat_completions_api" not in payload + assert "reasoning_content_field" not in payload messages: Final = payload["messages"] assert [message["role"] for message in messages] == ( ["user", "assistant", "tool", "assistant", "tool"] if sequential else ["user", "assistant", "tool", "tool"] @@ -108,11 +132,11 @@ async def test_hosted_vllm_responses_reasoning_and_parallel_tools_final_wire( results: Final = [message for message in messages if message["role"] == "tool"] assert [message["tool_call_id"] for message in results] == ["call_1", "call_2"] assert [message["content"] for message in results] == ["first record", "second record"] - assert messages[1].get("reasoning_content") == (reasoning if forward is True else None) - assert route.calls[0].request.content.decode().count(reasoning) == int(forward is True) + assert messages[1].get(output_field) == (reasoning if forwarded else None) + assert route.calls[0].request.content.decode().count(reasoning) == int(forwarded) if sequential: - assert messages[3].get("reasoning_content") == (next_reasoning if forward is True else None) - assert route.calls[0].request.content.decode().count(next_reasoning) == int(forward is True) + assert messages[3].get(output_field) == (next_reasoning if forwarded else None) + assert route.calls[0].request.content.decode().count(next_reasoning) == int(forwarded) assert input_items == original diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index e118181ba18..adc6468820c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29885,6 +29885,12 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; + /** + * Reasoning Content Field + * @default reasoning_content + * @enum {string} + */ + reasoning_content_field: "reasoning_content" | "reasoning"; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ @@ -40104,6 +40110,12 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; + /** + * Reasoning Content Field + * @default reasoning_content + * @enum {string} + */ + reasoning_content_field: "reasoning_content" | "reasoning"; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ From 101537c231cb038fbc8df658047270834f882726 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 09:28:08 +0200 Subject: [PATCH 04/24] fix: preserve Python 3.12 OpenAPI snapshot formatting --- litellm/proxy/_lazy_openapi_snapshot.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 4c4f2164a59..7a110eff080 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -18968,7 +18968,7 @@ } } }, - "description": "\nUnified rate-limit error.\n\nEvery rate-limit condition surfaced by litellm \u2014 whether it originated from\nan upstream LLM provider, a vendor batch endpoint, or one of litellm's own\nproxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\nmax-iterations, etc.) \u2014 is raised as an instance of this class.\n\nThe :attr:`category` attribute lets callers distinguish the source. See\n:class:`RateLimitErrorCategory` for the available values.\n" + "description": "\n Unified rate-limit error.\n\n Every rate-limit condition surfaced by litellm \u2014 whether it originated from\n an upstream LLM provider, a vendor batch endpoint, or one of litellm's own\n proxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\n max-iterations, etc.) \u2014 is raised as an instance of this class.\n\n The :attr:`category` attribute lets callers distinguish the source. See\n :class:`RateLimitErrorCategory` for the available values.\n " }, "500": { "content": { From c84c73b3df5ad87ccd747f57e7a958676a1db155 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 09:30:47 +0200 Subject: [PATCH 05/24] fix: preserve reasoning field on partial model updates --- litellm/types/router.py | 2 +- .../test_model_management_endpoints.py | 32 +++++++++++++++++-- ui/litellm-dashboard/src/lib/http/schema.d.ts | 16 +++------- 3 files changed, 35 insertions(+), 15 deletions(-) diff --git a/litellm/types/router.py b/litellm/types/router.py index ffbdbc96d10..f3703f82711 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -358,7 +358,7 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False forward_reasoning_content: bool | None = False - reasoning_content_field: Literal["reasoning_content", "reasoning"] = "reasoning_content" + reasoning_content_field: Literal["reasoning_content", "reasoning"] | None = None model_info: dict | None = None mock_response: str | ModelResponse | Exception | Any | None = None diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index c3ad66397ea..86bbd0d3b4b 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -1168,7 +1168,8 @@ class TestUpdateModel: """ @pytest.mark.asyncio - async def test_update_model_clears_cache_after_db_write(self): + @pytest.mark.parametrize("reasoning_field", [None, "reasoning_content", "reasoning"]) + async def test_update_model_clears_cache_after_db_write(self, reasoning_field): """ Regression test for the stale-router bug: POST /model/update must refresh the in-memory router after persisting to LiteLLM_ProxyModelTable, otherwise @@ -1190,6 +1191,7 @@ class TestUpdateModel: existing_row.litellm_params = { "model": "openai/gpt-4o-mini", "api_key": "sk-existing", + "reasoning_content_field": "reasoning", } existing_row.model_dump.return_value = { "model_name": "gpt-4o-mini", @@ -1237,7 +1239,7 @@ class TestUpdateModel: ): await update_model( model_params=updateDeployment( - litellm_params=updateLiteLLMParams(guardrails=["g1"]), + litellm_params=updateLiteLLMParams(guardrails=["g1"], **({} if reasoning_field is None else {"reasoning_content_field": reasoning_field})), model_info=ModelInfo(id=model_id), ), user_api_key_dict=admin_user, @@ -1245,6 +1247,8 @@ class TestUpdateModel: mock_prisma.db.litellm_proxymodeltable.update.assert_awaited_once() mock_clear_cache.assert_awaited_once_with() + stored = json.loads(mock_prisma.db.litellm_proxymodeltable.update.call_args.kwargs["data"]["litellm_params"]) + assert stored["reasoning_content_field"] == (reasoning_field or "reasoning") class TestUpdatePublicModelGroups: @@ -6198,3 +6202,27 @@ class TestAccessGroupModelSync: assert "array_replace" in update_call.args[0] assert update_call.args[1:] == ("gpt-5.6", "gpt-5.6-eu") invalidate.assert_awaited_once_with(("ag-1",)) + + +@pytest.mark.parametrize("field", [None, "reasoning_content", "reasoning"]) +def test_model_patch_preserves_reasoning_field_unless_explicit(field, monkeypatch): + from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper + from litellm.proxy.management_endpoints.model_management_endpoints import update_db_model + + monkeypatch.setenv("LITELLM_SALT_KEY", "sk-test-reasoning-field") + deployment = Deployment( + model_name="reasoning-test", + litellm_params=LiteLLM_Params(model="openai/reasoning-test", reasoning_content_field="reasoning"), + model_info=ModelInfo(id="reasoning-row"), + ) + params = updateLiteLLMParams(tpm=123, **({} if field is None else {"reasoning_content_field": field})) + if field is None: + assert "reasoning_content_field" not in params.model_dump(exclude_none=True) + result = update_db_model(db_model=deployment, updated_patch=updateDeployment(litellm_params=params)) + stored = json.loads(result["litellm_params"]) + assert stored["tpm"] == 123 + assert ( + stored["reasoning_content_field"] if field is None + else decrypt_value_helper(value=stored["reasoning_content_field"], key="reasoning_content_field") + ) == (field or "reasoning") + assert deployment.litellm_params.reasoning_content_field == "reasoning" diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index adc6468820c..f0df54706b8 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29885,12 +29885,8 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; - /** - * Reasoning Content Field - * @default reasoning_content - * @enum {string} - */ - reasoning_content_field: "reasoning_content" | "reasoning"; + /** Reasoning Content Field */ + reasoning_content_field?: ("reasoning_content" | "reasoning") | null; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ @@ -40110,12 +40106,8 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; - /** - * Reasoning Content Field - * @default reasoning_content - * @enum {string} - */ - reasoning_content_field: "reasoning_content" | "reasoning"; + /** Reasoning Content Field */ + reasoning_content_field?: ("reasoning_content" | "reasoning") | null; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ From a1db7d6b0364150869bc330a6ce52a206231b975 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 09:44:48 +0200 Subject: [PATCH 06/24] fix: preserve reasoning forwarding on partial updates --- litellm/types/router.py | 2 +- .../test_model_management_endpoints.py | 29 +++++++++++++++---- ui/litellm-dashboard/src/lib/http/schema.d.ts | 14 +++------ 3 files changed, 28 insertions(+), 17 deletions(-) diff --git a/litellm/types/router.py b/litellm/types/router.py index f3703f82711..c8a53f3467b 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -357,7 +357,7 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): ) model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False - forward_reasoning_content: bool | None = False + forward_reasoning_content: bool | None = None reasoning_content_field: Literal["reasoning_content", "reasoning"] | None = None model_info: dict | None = None mock_response: str | ModelResponse | Exception | Any | None = None diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 86bbd0d3b4b..ba7bfdcdc21 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -1169,7 +1169,8 @@ class TestUpdateModel: @pytest.mark.asyncio @pytest.mark.parametrize("reasoning_field", [None, "reasoning_content", "reasoning"]) - async def test_update_model_clears_cache_after_db_write(self, reasoning_field): + @pytest.mark.parametrize("forward", [None, False, True]) + async def test_update_model_clears_cache_after_db_write(self, reasoning_field, forward): """ Regression test for the stale-router bug: POST /model/update must refresh the in-memory router after persisting to LiteLLM_ProxyModelTable, otherwise @@ -1192,6 +1193,7 @@ class TestUpdateModel: "model": "openai/gpt-4o-mini", "api_key": "sk-existing", "reasoning_content_field": "reasoning", + "forward_reasoning_content": True, } existing_row.model_dump.return_value = { "model_name": "gpt-4o-mini", @@ -1239,7 +1241,11 @@ class TestUpdateModel: ): await update_model( model_params=updateDeployment( - litellm_params=updateLiteLLMParams(guardrails=["g1"], **({} if reasoning_field is None else {"reasoning_content_field": reasoning_field})), + litellm_params=updateLiteLLMParams( + **({"guardrails": ["g1"]} if reasoning_field is None and forward is None else {}), + **({} if reasoning_field is None else {"reasoning_content_field": reasoning_field}), + **({} if forward is None else {"forward_reasoning_content": forward}), + ), model_info=ModelInfo(id=model_id), ), user_api_key_dict=admin_user, @@ -1249,6 +1255,7 @@ class TestUpdateModel: mock_clear_cache.assert_awaited_once_with() stored = json.loads(mock_prisma.db.litellm_proxymodeltable.update.call_args.kwargs["data"]["litellm_params"]) assert stored["reasoning_content_field"] == (reasoning_field or "reasoning") + assert stored["forward_reasoning_content"] is (True if forward is None else forward) class TestUpdatePublicModelGroups: @@ -6205,24 +6212,34 @@ class TestAccessGroupModelSync: @pytest.mark.parametrize("field", [None, "reasoning_content", "reasoning"]) -def test_model_patch_preserves_reasoning_field_unless_explicit(field, monkeypatch): +@pytest.mark.parametrize("forward", [None, False, True]) +def test_model_patch_preserves_reasoning_field_unless_explicit(field, forward, monkeypatch): from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper from litellm.proxy.management_endpoints.model_management_endpoints import update_db_model monkeypatch.setenv("LITELLM_SALT_KEY", "sk-test-reasoning-field") deployment = Deployment( model_name="reasoning-test", - litellm_params=LiteLLM_Params(model="openai/reasoning-test", reasoning_content_field="reasoning"), + litellm_params=LiteLLM_Params(model="openai/reasoning-test", reasoning_content_field="reasoning", forward_reasoning_content=True), model_info=ModelInfo(id="reasoning-row"), ) - params = updateLiteLLMParams(tpm=123, **({} if field is None else {"reasoning_content_field": field})) + params = updateLiteLLMParams( + **({"tpm": 123} if field is None and forward is None else {}), + **({} if field is None else {"reasoning_content_field": field}), + **({} if forward is None else {"forward_reasoning_content": forward}), + ) + if forward is None: + assert "forward_reasoning_content" not in params.model_dump(exclude_none=True) if field is None: assert "reasoning_content_field" not in params.model_dump(exclude_none=True) result = update_db_model(db_model=deployment, updated_patch=updateDeployment(litellm_params=params)) stored = json.loads(result["litellm_params"]) - assert stored["tpm"] == 123 + if field is None and forward is None: + assert stored["tpm"] == 123 + assert stored["forward_reasoning_content"] is (True if forward is None else forward) assert ( stored["reasoning_content_field"] if field is None else decrypt_value_helper(value=stored["reasoning_content_field"], key="reasoning_content_field") ) == (field or "reasoning") assert deployment.litellm_params.reasoning_content_field == "reasoning" + assert deployment.litellm_params.forward_reasoning_content is True diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index f0df54706b8..a90fce927ca 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29711,11 +29711,8 @@ export interface components { default_api_key_tpm_limit?: number | null; /** Drop Params */ drop_params?: boolean | string | null; - /** - * Forward Reasoning Content - * @default false - */ - forward_reasoning_content: boolean | null; + /** Forward Reasoning Content */ + forward_reasoning_content?: boolean | null; /** Gcs Bucket Name */ gcs_bucket_name?: string | null; /** Google Maps Grounding Cost Per Query */ @@ -39932,11 +39929,8 @@ export interface components { default_api_key_tpm_limit?: number | null; /** Drop Params */ drop_params?: boolean | string | null; - /** - * Forward Reasoning Content - * @default false - */ - forward_reasoning_content: boolean | null; + /** Forward Reasoning Content */ + forward_reasoning_content?: boolean | null; /** Gcs Bucket Name */ gcs_bucket_name?: string | null; /** Google Maps Grounding Cost Per Query */ From 624f2f97b645d73227c851a1c04251d13bd6f8e0 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 10:04:36 +0200 Subject: [PATCH 07/24] fix: support encrypted reasoning field storage --- .../reasoning_content_utils.py | 11 ++-- .../llms/hosted_vllm/chat/transformation.py | 12 ++--- litellm/types/router.py | 5 +- .../test_hosted_vllm_chat_transformation.py | 7 ++- .../test_model_management_endpoints.py | 51 ++++++++++++++++--- ui/litellm-dashboard/src/lib/http/schema.d.ts | 14 +++-- 6 files changed, 73 insertions(+), 27 deletions(-) diff --git a/litellm/litellm_core_utils/reasoning_content_utils.py b/litellm/litellm_core_utils/reasoning_content_utils.py index ace40b18bc8..b307a85804e 100644 --- a/litellm/litellm_core_utils/reasoning_content_utils.py +++ b/litellm/litellm_core_utils/reasoning_content_utils.py @@ -10,23 +10,24 @@ from litellm.types.llms.openai import AllMessageValues def normalize_reasoning_content( - messages: Sequence[AllMessageValues], *, forward: bool = True + messages: Sequence[AllMessageValues], *, forward: bool = True, normalize: bool = True ) -> list[AllMessageValues]: # mutable-ok: provider request contract def normalize_message(message: AllMessageValues) -> AllMessageValues: if message["role"] != "assistant": return message + if not normalize and forward: + return message history: Final[Mapping[str, object]] = message + removed_fields: Final = ("reasoning", "reasoning_content") if normalize else ("reasoning_content",) reasoning: Final = ( history.get("reasoning") if history.get("reasoning") is not None else history.get("reasoning_content") ) normalized: Final[Mapping[str, object]] = MappingProxyType( { - **MappingProxyType( - {key: value for key, value in history.items() if key not in ("reasoning", "reasoning_content")} - ), + **MappingProxyType({key: value for key, value in history.items() if key not in removed_fields}), **( MappingProxyType({"reasoning": reasoning}) - if forward and reasoning is not None + if normalize and forward and reasoning is not None else MappingProxyType({}) ), } diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index fe28e11a2ee..be3c4206f5c 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -155,15 +155,11 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): litellm_params: dict, # mutable-ok: provider request contract headers: dict, # mutable-ok: provider request contract ) -> dict: # mutable-ok: provider request contract - request_messages: Final = ( - normalize_reasoning_content(messages, forward=litellm_params.get("forward_reasoning_content") is True) - if litellm_params.get("reasoning_content_field") == "reasoning" - else deepcopy(messages) + request_messages: Final = normalize_reasoning_content( + messages, + forward=litellm_params.get("forward_reasoning_content") is True, + normalize=litellm_params.get("reasoning_content_field") == "reasoning", ) - if litellm_params.get("forward_reasoning_content") is not True: - for message in request_messages: - if message["role"] == "assistant": - message.pop("reasoning_content", None) return super().transform_request(model, request_messages, optional_params, litellm_params, headers) async def async_transform_request( diff --git a/litellm/types/router.py b/litellm/types/router.py index c8a53f3467b..a7695622051 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -358,7 +358,10 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False forward_reasoning_content: bool | None = None - reasoning_content_field: Literal["reasoning_content", "reasoning"] | None = None + reasoning_content_field: str | None = Field( + default=None, + description="Historical assistant reasoning field: reasoning_content (default) or reasoning.", + ) model_info: dict | None = None mock_response: str | ModelResponse | Exception | Any | None = None diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index 0b70e763f34..e17e490b47f 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -586,6 +586,7 @@ async def test_reasoning_field_sdk_router_final_wire( ("legacy", {}), ("normalized", {"reasoning_content_field": "reasoning"}), ("explicit-default", {"reasoning_content_field": "reasoning_content"}), + ("unknown", {"reasoning_content_field": "unknown"}), ) ], num_retries=0, @@ -602,7 +603,9 @@ async def test_reasoning_field_sdk_router_final_wire( "usage": {"prompt_tokens": 10, "completion_tokens": 1, "total_tokens": 11}, }, ) - for alias, field in (("normalized", "reasoning"), ("legacy", None), ("explicit-default", "reasoning_content")): + for alias, field in ( + ("normalized", "reasoning"), ("legacy", None), ("explicit-default", "reasoning_content"), ("unknown", "unknown") + ): kwargs: Final = ( {"model": alias, "messages": messages} if via_router @@ -637,7 +640,7 @@ async def test_reasoning_field_sdk_router_final_wire( assert "reasoning_content_field" not in payload assert "forward_reasoning_content" not in payload assert messages == original - assert route.call_count == 3 + assert route.call_count == 4 @pytest.mark.asyncio diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index ba7bfdcdc21..fc1490a03de 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -1170,7 +1170,7 @@ class TestUpdateModel: @pytest.mark.asyncio @pytest.mark.parametrize("reasoning_field", [None, "reasoning_content", "reasoning"]) @pytest.mark.parametrize("forward", [None, False, True]) - async def test_update_model_clears_cache_after_db_write(self, reasoning_field, forward): + async def test_update_model_clears_cache_after_db_write(self, reasoning_field, forward, monkeypatch): """ Regression test for the stale-router bug: POST /model/update must refresh the in-memory router after persisting to LiteLLM_ProxyModelTable, otherwise @@ -1186,13 +1186,16 @@ class TestUpdateModel: updateLiteLLMParams, ) + from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper + + monkeypatch.setenv("LITELLM_SALT_KEY", "sk-test-reasoning-field") model_id = "db-model-under-test" existing_row = MagicMock() existing_row.litellm_params = { "model": "openai/gpt-4o-mini", "api_key": "sk-existing", - "reasoning_content_field": "reasoning", + "reasoning_content_field": encrypt_value_helper("reasoning"), "forward_reasoning_content": True, } existing_row.model_dump.return_value = { @@ -1228,10 +1231,6 @@ class TestUpdateModel: "litellm.proxy.management_endpoints.model_management_endpoints.ModelManagementAuthChecks.can_user_make_model_call", new=AsyncMock(return_value=None), ), - patch( - "litellm.proxy.management_endpoints.model_management_endpoints.encrypt_value_helper", - side_effect=lambda value: value, - ), patch( "litellm.proxy.management_endpoints.model_management_endpoints.clear_cache", new=AsyncMock( @@ -1254,7 +1253,9 @@ class TestUpdateModel: mock_prisma.db.litellm_proxymodeltable.update.assert_awaited_once() mock_clear_cache.assert_awaited_once_with() stored = json.loads(mock_prisma.db.litellm_proxymodeltable.update.call_args.kwargs["data"]["litellm_params"]) - assert stored["reasoning_content_field"] == (reasoning_field or "reasoning") + assert decrypt_value_helper(stored["reasoning_content_field"], key="reasoning_content_field") == (reasoning_field or "reasoning") + if reasoning_field is None: + assert stored["reasoning_content_field"] == existing_row.litellm_params["reasoning_content_field"] assert stored["forward_reasoning_content"] is (True if forward is None else forward) @@ -6243,3 +6244,39 @@ def test_model_patch_preserves_reasoning_field_unless_explicit(field, forward, m ) == (field or "reasoning") assert deployment.litellm_params.reasoning_content_field == "reasoning" assert deployment.litellm_params.forward_reasoning_content is True + + +@pytest.mark.asyncio +@pytest.mark.parametrize("field", ["reasoning_content", "reasoning"]) +async def test_reasoning_field_encrypted_db_round_trip(field, monkeypatch): + from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper + from litellm.proxy.management_endpoints.model_management_endpoints import get_db_model, update_db_model + + monkeypatch.setenv("LITELLM_SALT_KEY", "sk-test-reasoning-field") + initial = Deployment( + model_name="reasoning-test", + litellm_params=LiteLLM_Params(model="openai/reasoning-test", forward_reasoning_content=True), + model_info=ModelInfo(id="reasoning-row"), + ) + first = update_db_model( + db_model=initial, + updated_patch=updateDeployment(litellm_params=updateLiteLLMParams(reasoning_content_field=field)), + ) + encrypted = json.loads(first["litellm_params"]) + assert encrypted["reasoning_content_field"] != field + raw = {"model_name": first["model_name"], "litellm_params": encrypted, "model_info": {"id": "reasoning-row"}} + deployment = Deployment(**raw) + assert deployment.litellm_params.reasoning_content_field == encrypted["reasoning_content_field"] + row = MagicMock() + row.model_dump.return_value = raw + prisma = MagicMock() + prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=row) + loaded = await get_db_model("reasoning-row", prisma) + assert loaded.litellm_params.reasoning_content_field == encrypted["reasoning_content_field"] + second = update_db_model( + db_model=loaded, updated_patch=updateDeployment(litellm_params=updateLiteLLMParams(tpm=123)) + ) + stored = json.loads(second["litellm_params"]) + assert stored["reasoning_content_field"] == encrypted["reasoning_content_field"] + assert stored["forward_reasoning_content"] is True + assert decrypt_value_helper(stored["reasoning_content_field"], key="reasoning_content_field") == field diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index a90fce927ca..ce427314159 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29882,8 +29882,11 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; - /** Reasoning Content Field */ - reasoning_content_field?: ("reasoning_content" | "reasoning") | null; + /** + * Reasoning Content Field + * @description Historical assistant reasoning field: reasoning_content (default) or reasoning. + */ + reasoning_content_field?: string | null; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ @@ -40100,8 +40103,11 @@ export interface components { } | null; /** Quality Router Default Model */ quality_router_default_model?: string | null; - /** Reasoning Content Field */ - reasoning_content_field?: ("reasoning_content" | "reasoning") | null; + /** + * Reasoning Content Field + * @description Historical assistant reasoning field: reasoning_content (default) or reasoning. + */ + reasoning_content_field?: string | null; /** Region Name */ region_name?: string | null; /** Regional Endpoint Uplift Multiplier */ From 115f4fd89bc7555874ef81e625ebb5be3aecafc7 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 15 Sep 2026 10:37:07 +0200 Subject: [PATCH 08/24] fix: validate reasoning field before chat dispatch --- .../reasoning_content_utils.py | 13 ++++++ .../llms/hosted_vllm/chat/transformation.py | 25 ++++++----- .../llms/openai/chat/gpt_transformation.py | 13 ++++-- .../test_hosted_vllm_chat_transformation.py | 43 ++++++++++++++++--- 4 files changed, 76 insertions(+), 18 deletions(-) diff --git a/litellm/litellm_core_utils/reasoning_content_utils.py b/litellm/litellm_core_utils/reasoning_content_utils.py index b307a85804e..63ad70f5b27 100644 --- a/litellm/litellm_core_utils/reasoning_content_utils.py +++ b/litellm/litellm_core_utils/reasoning_content_utils.py @@ -6,9 +6,22 @@ from typing import ( cast, # noqa: TID251 # Preserves arbitrary provider fields without lossy TypedDict validation. ) +from litellm.exceptions import BadRequestError from litellm.types.llms.openai import AllMessageValues +def should_normalize_reasoning_content(field: object, *, model: str, provider: str) -> bool: + if field is None or field == "reasoning_content": + return False + if field == "reasoning": + return True + raise BadRequestError( + message="reasoning_content_field must be reasoning_content or reasoning", + model=model, + llm_provider=provider, + ) + + def normalize_reasoning_content( messages: Sequence[AllMessageValues], *, forward: bool = True, normalize: bool = True ) -> list[AllMessageValues]: # mutable-ok: provider request contract diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index be3c4206f5c..3f084ca6517 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -11,7 +11,10 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, ) from litellm.litellm_core_utils.prompt_templates.factory import _parse_mime_type -from litellm.litellm_core_utils.reasoning_content_utils import normalize_reasoning_content +from litellm.litellm_core_utils.reasoning_content_utils import ( + normalize_reasoning_content, + should_normalize_reasoning_content, +) from litellm.litellm_core_utils.reasoning_effort_utils import ( reasoning_effort_from_thinking_budget, ) @@ -151,14 +154,16 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): self, model: str, messages: list[AllMessageValues], # mutable-ok: provider request contract - optional_params: dict, # mutable-ok: provider request contract - litellm_params: dict, # mutable-ok: provider request contract - headers: dict, # mutable-ok: provider request contract - ) -> dict: # mutable-ok: provider request contract + optional_params: dict[str, object], # mutable-ok: provider request contract + litellm_params: dict[str, object], # mutable-ok: provider request contract + headers: dict[str, str], # mutable-ok: provider request contract + ) -> dict[str, object]: # mutable-ok: provider request contract request_messages: Final = normalize_reasoning_content( messages, forward=litellm_params.get("forward_reasoning_content") is True, - normalize=litellm_params.get("reasoning_content_field") == "reasoning", + normalize=should_normalize_reasoning_content( + litellm_params.get("reasoning_content_field"), model=model, provider="hosted_vllm" + ), ) return super().transform_request(model, request_messages, optional_params, litellm_params, headers) @@ -166,10 +171,10 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): self, model: str, messages: list[AllMessageValues], # mutable-ok: provider request contract - optional_params: dict, # mutable-ok: provider request contract - litellm_params: dict, # mutable-ok: provider request contract - headers: dict, # mutable-ok: provider request contract - ) -> dict: # mutable-ok: provider request contract + optional_params: dict[str, object], # mutable-ok: provider request contract + litellm_params: dict[str, object], # mutable-ok: provider request contract + headers: dict[str, str], # mutable-ok: provider request contract + ) -> dict[str, object]: # mutable-ok: provider request contract return await super().async_transform_request( model, deepcopy(messages), optional_params, litellm_params, headers ) diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index b3824ef77a7..fe5586daf7b 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -30,7 +30,10 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_convert_url_to_base64, convert_url_to_base64, ) -from litellm.litellm_core_utils.reasoning_content_utils import normalize_reasoning_content +from litellm.litellm_core_utils.reasoning_content_utils import ( + normalize_reasoning_content, + should_normalize_reasoning_content, +) from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException @@ -481,7 +484,9 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): request_messages: Final = ( normalize_reasoning_content(messages) if litellm_params.get("custom_llm_provider") == "openai" - and litellm_params.get("reasoning_content_field") == "reasoning" + and should_normalize_reasoning_content( + litellm_params.get("reasoning_content_field"), model=model, provider="openai" + ) else messages ) messages = self._transform_messages(messages=request_messages, model=model) @@ -516,7 +521,9 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): request_messages: Final = ( normalize_reasoning_content(messages) if litellm_params.get("custom_llm_provider") == "openai" - and litellm_params.get("reasoning_content_field") == "reasoning" + and should_normalize_reasoning_content( + litellm_params.get("reasoning_content_field"), model=model, provider="openai" + ) else messages ) transformed_messages = await self._transform_messages(messages=request_messages, model=model, is_async=True) diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index e17e490b47f..c133f23487c 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -586,7 +586,6 @@ async def test_reasoning_field_sdk_router_final_wire( ("legacy", {}), ("normalized", {"reasoning_content_field": "reasoning"}), ("explicit-default", {"reasoning_content_field": "reasoning_content"}), - ("unknown", {"reasoning_content_field": "unknown"}), ) ], num_retries=0, @@ -604,7 +603,7 @@ async def test_reasoning_field_sdk_router_final_wire( }, ) for alias, field in ( - ("normalized", "reasoning"), ("legacy", None), ("explicit-default", "reasoning_content"), ("unknown", "unknown") + ("normalized", "reasoning"), ("legacy", None), ("explicit-default", "reasoning_content") ): kwargs: Final = ( {"model": alias, "messages": messages} @@ -640,13 +639,14 @@ async def test_reasoning_field_sdk_router_final_wire( assert "reasoning_content_field" not in payload assert "forward_reasoning_content" not in payload assert messages == original - assert route.call_count == 4 + assert route.call_count == 3 @pytest.mark.asyncio @pytest.mark.parametrize("is_async", [False, True]) @pytest.mark.parametrize("provider", ["deepinfra", "together_ai", None]) -async def test_reasoning_field_does_not_apply_to_inherited_provider(provider: str | None, is_async: bool): +@pytest.mark.parametrize("field", ["reasoning", "invalid-selector"]) +async def test_reasoning_field_does_not_apply_to_inherited_provider(provider: str | None, is_async: bool, field: str): from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig config: Final = OpenAIGPTConfig() @@ -657,8 +657,41 @@ async def test_reasoning_field_does_not_apply_to_inherited_provider(provider: st "messages": messages, "optional_params": {}, "headers": {}, - "litellm_params": {"custom_llm_provider": provider, "reasoning_content_field": "reasoning"}, + "litellm_params": {"custom_llm_provider": provider, "reasoning_content_field": field}, } result: Final = await config.async_transform_request(**kwargs) if is_async else config.transform_request(**kwargs) assert result["messages"] == original assert messages == original + + +@pytest.mark.asyncio +@pytest.mark.parametrize("provider", ["hosted_vllm", "openai"]) +@pytest.mark.parametrize("is_async", [False, True]) +@pytest.mark.parametrize("via_router", [False, True]) +@pytest.mark.parametrize("bridge", [False, True]) +@pytest.mark.parametrize("field", ["reasonig", ""]) +@pytest.mark.parametrize("forward", [False, True]) +async def test_invalid_reasoning_field_fails_before_http( + provider, is_async, via_router, bridge, field, forward, monkeypatch +): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + messages = [{"role": "user", "content": "Hello"}] + original = deepcopy(messages) + params = { + "model": f"{provider}/reasoning-test", "api_key": "test-key", + "api_base": "https://invalid-reasoning-field.invalid/v1", + "reasoning_content_field": field, "forward_reasoning_content": forward, + **({"use_chat_completions_api": True} if bridge else {}), + } + router = litellm.Router(model_list=[{"model_name": "invalid-field", "litellm_params": params}], num_retries=0) + client = router if via_router else litellm + kwargs = {**({"model": "invalid-field"} if via_router else params), "input" if bridge else "messages": messages} + method = (client.aresponses if is_async else client.responses) if bridge else (client.acompletion if is_async else client.completion) + with respx.mock(assert_all_called=False) as mock: + with pytest.raises(litellm.BadRequestError) as error: + await method(**kwargs) if is_async else method(**kwargs) + assert error.value.status_code == 400 + assert "reasoning_content_field must be reasoning_content or reasoning" in str(error.value) + assert "reasonig" not in str(error.value) + assert len(mock.calls) == 0 + assert messages == original From a590abfc66e7ffb9ee1ad3c1ca9c5a6c518dafc2 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 16 Sep 2026 08:14:16 +0200 Subject: [PATCH 09/24] test: annotate reasoning history test parameters --- .../chat/test_hosted_vllm_chat_transformation.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index c133f23487c..fc20ca90faa 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -20,7 +20,9 @@ from litellm.llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig @pytest.mark.parametrize("params", [{}, {"forward_reasoning_content": False}, {"forward_reasoning_content": True}]) @pytest.mark.parametrize("is_async", [False, True]) @pytest.mark.asyncio -async def test_forward_reasoning_content_preserves_only_explicit_history(params, is_async): +async def test_forward_reasoning_content_preserves_only_explicit_history( + params: dict[str, bool], is_async: bool +) -> None: config = HostedVLLMChatConfig() messages = [ {"role": "user", "content": "Check the counter"}, @@ -672,8 +674,14 @@ async def test_reasoning_field_does_not_apply_to_inherited_provider(provider: st @pytest.mark.parametrize("field", ["reasonig", ""]) @pytest.mark.parametrize("forward", [False, True]) async def test_invalid_reasoning_field_fails_before_http( - provider, is_async, via_router, bridge, field, forward, monkeypatch -): + provider: str, + is_async: bool, + via_router: bool, + bridge: bool, + field: str, + forward: bool, + monkeypatch: pytest.MonkeyPatch, +) -> None: monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) messages = [{"role": "user", "content": "Hello"}] original = deepcopy(messages) From 7a6c2ac1fc9594f8ca8750377c7667c7a5952401 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Fri, 25 Sep 2026 22:24:57 +0200 Subject: [PATCH 10/24] test(types): register reasoning transport options in the owned inventory Upstream now enumerates every owned litellm_params name in tests/unit/types/test_litellm_params.py, so forward_reasoning_content and reasoning_content_field have to appear there next to the reasoning option they travel with. Without them the merged branch fails the exact-inventory test and its concatenation counterpart. --- tests/unit/types/test_litellm_params.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/unit/types/test_litellm_params.py b/tests/unit/types/test_litellm_params.py index e421321aaaa..3bddac93500 100644 --- a/tests/unit/types/test_litellm_params.py +++ b/tests/unit/types/test_litellm_params.py @@ -181,6 +181,8 @@ OPTION_NAMES: Final = ( "assistant_continue_message", "disable_add_transform_inline_image_block", "merge_reasoning_content_in_choices", + "forward_reasoning_content", + "reasoning_content_field", "enable_json_schema_validation", "complete_response", "stream_chunk_size", From 6716494e1fff59f024d03b7429df3ee817cc4c1d Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Fri, 25 Sep 2026 22:51:00 +0200 Subject: [PATCH 11/24] test(ci): drop the repeated UNIT_FLAG key in the shard paths test Upstream #43186 moved this file with "UNIT_FLAG" set twice in the same env literal, which fails F601 in the test-tree ruff config and turns the lint job red for every pull request that touches tests. Removing the duplicate keeps the effective environment identical. --- tests/unit/test_unit_shard_missing_paths.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/unit/test_unit_shard_missing_paths.py b/tests/unit/test_unit_shard_missing_paths.py index b528a75d0df..0360a227142 100644 --- a/tests/unit/test_unit_shard_missing_paths.py +++ b/tests/unit/test_unit_shard_missing_paths.py @@ -40,7 +40,6 @@ def _run_shard(tmp_path: Path, test_path: str, workers: str) -> subprocess.Compl "TEST_PATH": test_path, "UNIT_FLAG": "", "WORKERS": workers, - "UNIT_FLAG": "", }, capture_output=True, text=True, From 1e7d04a06536d9a713b1de0a00a555b1bf120ee3 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 30 Sep 2026 14:08:51 +0200 Subject: [PATCH 12/24] refactor(router): declare the reasoning transport options on the param models that use them Declaring `forward_reasoning_content` and `reasoning_content_field` on `GenericLiteLLMParams` reached every caller that builds those params from an unannotated `**kwargs`, because basedpyright reports one unknown-argument diagnostic per declared parameter at every such spread site: 113 sites, so the two fields added 226 diagnostics and pushed `reportUnknownArgumentType` past its ceiling without any of those files changing. `LiteLLM_Params` (deployment and creation) and `updateLiteLLMParams` (model update) are the models that carry the options, and neither is built by spreading unknown kwargs, so declaring them there keeps the API, the generated dashboard types and the partial-update behaviour identical while the shared signature stays as it was. --- litellm/types/router.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/litellm/types/router.py b/litellm/types/router.py index fc19efa4b58..7fa49352a9c 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -425,11 +425,6 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): ) model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: bool | None = False - forward_reasoning_content: bool | None = None - reasoning_content_field: str | None = Field( - default=None, - description="Historical assistant reasoning field: reasoning_content (default) or reasoning.", - ) model_info: dict | None = None mock_response: str | ModelResponse | Exception | object | None = None @@ -531,6 +526,11 @@ class LiteLLM_Params(GenericLiteLLMParams): model: str model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) + forward_reasoning_content: bool | None = None + reasoning_content_field: str | None = Field( + default=None, + description="Historical assistant reasoning field: reasoning_content (default) or reasoning.", + ) def __contains__(self, key) -> bool: # Define custom behavior for the 'in' operator @@ -553,6 +553,11 @@ class updateLiteLLMParams(GenericLiteLLMParams): # This class is used to update the LiteLLM_Params # only differece is model is optional model: str | None = None + forward_reasoning_content: bool | None = None + reasoning_content_field: str | None = Field( + default=None, + description="Historical assistant reasoning field: reasoning_content (default) or reasoning.", + ) class updateDeployment(BaseModel): From 8897879028ca32d517f4a0050300808d891fd356 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 30 Sep 2026 16:01:09 +0200 Subject: [PATCH 13/24] test(catalog): validate ultrafast pricing fields in the strict schema --- tests/unit/test_utils.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index afc449d22c4..1bb8d6a4a74 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -761,12 +761,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_computer_use": {"type": "boolean"}, "cache_creation_input_audio_token_cost": {"type": "number"}, "cache_creation_input_token_cost": {"type": "number"}, + "cache_creation_input_token_cost_ultrafast": {"type": "number"}, "cache_creation_input_token_cost_above_1hr": {"type": "number"}, "cache_creation_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_256k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens": {"type": "number"}, + "cache_creation_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_flex": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_priority": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens_batches": {"type": "number"}, @@ -775,12 +777,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, + "cache_read_input_token_cost_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens_batches": {"type": "number"}, "cache_read_input_token_cost_above_256k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_272k_tokens": {"type": "number"}, + "cache_read_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_272k_tokens_flex": {"type": "number"}, "cache_read_input_token_cost_above_512k_tokens": {"type": "number"}, "cache_read_input_token_cost_batches": {"type": "number"}, @@ -805,6 +809,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_token_above_200k_tokens_batches": {"type": "number"}, "input_cost_per_token_above_256k_tokens": {"type": "number"}, "input_cost_per_token_above_272k_tokens": {"type": "number"}, + "input_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"}, "input_cost_per_token_above_512k_tokens": {"type": "number"}, "cache_read_input_token_cost_flex": {"type": "number"}, "cache_read_input_token_cost_priority": {"type": "number"}, @@ -835,6 +840,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cost_per_second": {"type": "number"}, "input_cost_per_second": {"type": "number"}, "input_cost_per_token": {"type": "number"}, + "input_cost_per_token_ultrafast": {"type": "number"}, "input_cost_per_token_above_128k_tokens": {"type": "number"}, "input_cost_per_audio_token_batches": {"type": "number"}, "input_cost_per_image_token_batches": {"type": "number"}, @@ -904,12 +910,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_second_1080p": {"type": "number"}, "output_cost_per_second_4k": {"type": "number"}, "output_cost_per_token": {"type": "number"}, + "output_cost_per_token_ultrafast": {"type": "number"}, "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens_batches": {"type": "number"}, "output_cost_per_token_above_256k_tokens": {"type": "number"}, "output_cost_per_token_above_272k_tokens": {"type": "number"}, + "output_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"}, "output_cost_per_token_above_512k_tokens": {"type": "number"}, "output_cost_per_token_batches": {"type": "number"}, "output_cost_per_reasoning_token": {"type": "number"}, From e5098f608089c564bb1d5b04cba294de898d77c8 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 30 Sep 2026 16:01:09 +0200 Subject: [PATCH 14/24] chore(reasoning): remove narrative comments flagged in review --- litellm/litellm_core_utils/get_litellm_params.py | 3 --- .../management_endpoints/test_model_management_endpoints.py | 2 -- 2 files changed, 5 deletions(-) diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py index 70516f1adf1..b36c4d6db6f 100644 --- a/litellm/litellm_core_utils/get_litellm_params.py +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -32,9 +32,6 @@ AWS_CREDENTIAL_KWARGS_KEYS: Final = frozenset( PROVIDER_AFFINITY_HEADER_KWARG_KEY: Final = "provider_affinity_header" -# Internal reasoning-history options that `completion()` forwards from its own -# kwargs into `get_litellm_params`, which are otherwise invisible to it because -# that call site passes explicit named arguments rather than `**kwargs`. REASONING_TRANSPORT_KWARGS_KEYS: Final = frozenset({"forward_reasoning_content", "reasoning_content_field"}) # Pre-define optional kwargs keys as frozenset for O(1) lookups diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 3167e1e6cba..e756a039608 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -1425,8 +1425,6 @@ class TestUpdateModel: mock_prisma.db.litellm_proxymodeltable.update.assert_awaited_once() mock_clear_cache.assert_awaited_once_with() stored = json.loads(mock_prisma.db.litellm_proxymodeltable.update.call_args.kwargs["data"]["litellm_params"]) - # The endpoint encrypts every incoming litellm_param, so the stored reasoning field is - # read back through the real decrypt path instead of a stubbed encryption helper. assert decrypt_value_helper(stored["reasoning_content_field"], key="reasoning_content_field") == (reasoning_field or "reasoning") if reasoning_field is None: assert stored["reasoning_content_field"] == existing_row.litellm_params["reasoning_content_field"] From 82ef0b9135c37afcb1266020b4fbad328d0b4f33 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 30 Sep 2026 16:03:12 +0200 Subject: [PATCH 15/24] test(interactions): follow operation schemas and declared path parameters --- .../interactions/test_openapi_compliance.py | 176 ++++++++++++++---- 1 file changed, 137 insertions(+), 39 deletions(-) diff --git a/tests/unit/interactions/test_openapi_compliance.py b/tests/unit/interactions/test_openapi_compliance.py index d3f1183cea6..8e460cc7c38 100644 --- a/tests/unit/interactions/test_openapi_compliance.py +++ b/tests/unit/interactions/test_openapi_compliance.py @@ -9,7 +9,8 @@ Run with: pytest tests/unit/interactions/test_openapi_compliance.py -v import json import os -from typing import Any, Dict +import re +from typing import Any, Dict, Final from unittest.mock import MagicMock, patch import httpx @@ -44,6 +45,56 @@ def _declared_type_value(variant_schema: Dict[str, Any]) -> Any: return type_property.get("const") or (enum_values[0] if len(enum_values) == 1 else None) +def _resolve_local_ref(spec_dict: dict[str, Any], schema: dict[str, Any]) -> dict[str, Any]: + """Resolve component references used by operations, schemas, and parameters.""" + if "$ref" not in schema: + return schema + reference: Final = schema["$ref"] + assert reference.startswith("#/components/"), f"Expected a local component reference: {reference}" + category, name = reference.removeprefix("#/components/").split("/") + return spec_dict["components"][category][name.replace("~1", "/").replace("~0", "~")] + + +def _interaction_operation( + spec_dict: dict[str, Any], method: str, *, individual: bool = False +) -> tuple[str, dict[str, Any]]: + """Match collection or item routes exactly, independent of placeholder names.""" + pattern: Final = r"(?:/[^/]+)*/interactions" + (r"/(\{[^/{}]+\})" if individual else "") + matches: Final = tuple( + (path, path_item, match) + for path, path_item in spec_dict["paths"].items() + if (match := re.fullmatch(pattern, path)) and method in path_item + ) + assert len(matches) == 1, f"Expected one {method.upper()} interactions endpoint, got {matches}" + path, path_item, match = matches[0] + operation: Final = path_item[method] + if individual: + parameter_name: Final = match.group(1)[1:-1] + parameters: Final = { + (parameter["name"], parameter["in"]): parameter + for raw_parameter in (*path_item.get("parameters", ()), *operation.get("parameters", ())) + for parameter in (_resolve_local_ref(spec_dict, raw_parameter),) + } + parameter: Final = parameters.get((parameter_name, "path")) + assert parameter is not None, f"{path} must declare its interaction ID path parameter" + assert parameter.get("required") is True, f"{path} must require its interaction ID" + parameter_schema: Final = _resolve_local_ref(spec_dict, parameter["schema"]) + assert parameter_schema.get("type") == "string", f"{path} must accept a string interaction ID" + return path, operation + + +def _model_request_schema(spec_dict: dict[str, Any]) -> dict[str, Any]: + """Find the model variant of the JSON body declared by the create operation.""" + _, operation = _interaction_operation(spec_dict, "post") + request_body: Final = _resolve_local_ref(spec_dict, operation["requestBody"]) + assert request_body.get("required") is True, "Creating an interaction must require a request body" + schema: Final = _resolve_local_ref(spec_dict, request_body["content"]["application/json"]["schema"]) + variants: Final = tuple(_resolve_local_ref(spec_dict, variant) for variant in schema.get("oneOf", (schema,))) + model_variants: Final = tuple(variant for variant in variants if "model" in variant.get("properties", {})) + assert len(model_variants) == 1, f"Expected one model request variant, got {model_variants}" + return model_variants[0] + + @pytest.fixture(scope="module") def spec_dict() -> Dict[str, Any]: """Load raw spec dict for manual validation.""" @@ -60,12 +111,15 @@ class TestRequestCompliance: """Tests that our request bodies match the OpenAPI spec.""" def test_create_model_interaction_request_schema(self, spec_dict): - """Verify CreateModelInteractionParams schema fields.""" - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] + """Verify the model request schema declared by POST /interactions.""" + schema = _model_request_schema(spec_dict) # Required fields per spec assert "model" in schema["required"] - assert "input" in schema["required"] + for field in ("model", "input"): + assert field in schema["properties"] + assert schema["properties"][field].get("readOnly") is not True + assert _resolve_local_ref(spec_dict, schema["properties"][field]).get("readOnly") is not True # Check our supported optional fields exist in spec our_optional_fields = [ @@ -88,13 +142,8 @@ class TestRequestCompliance: def test_input_types_match_spec(self, spec_dict): """Verify input field supports string, Content, Content[], Turn[].""" - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] - input_schema = schema["properties"]["input"] - - # The input property may be inline oneOf or a $ref to InteractionsInput - if "$ref" in input_schema: - ref_name = input_schema["$ref"].split("/")[-1] - input_schema = spec_dict["components"]["schemas"][ref_name] + schema = _model_request_schema(spec_dict) + input_schema = _resolve_local_ref(spec_dict, schema["properties"]["input"]) # Should be oneOf with multiple types assert "oneOf" in input_schema @@ -295,45 +344,94 @@ class TestEndpointCompliance: def test_create_endpoint_exists(self, spec_dict): """Verify POST /interactions endpoint exists.""" - paths = spec_dict["paths"] - - # Find the create interactions endpoint - create_path = None - for path, methods in paths.items(): - if "interactions" in path and "post" in methods: - create_path = path - break - - assert create_path is not None, "POST /interactions endpoint not found" + create_path, _ = _interaction_operation(spec_dict, "post") print(f"✓ Create endpoint: POST {create_path}") def test_get_endpoint_exists(self, spec_dict): """Verify GET /interactions/{id} endpoint exists.""" - paths = spec_dict["paths"] - - get_path = None - for path, methods in paths.items(): - if "{id}" in path and "interactions" in path and "get" in methods: - get_path = path - break - - assert get_path is not None, "GET /interactions/{id} endpoint not found" + get_path, _ = _interaction_operation(spec_dict, "get", individual=True) print(f"✓ Get endpoint: GET {get_path}") def test_delete_endpoint_exists(self, spec_dict): """Verify DELETE /interactions/{id} endpoint exists.""" - paths = spec_dict["paths"] - - delete_path = None - for path, methods in paths.items(): - if "{id}" in path and "interactions" in path and "delete" in methods: - delete_path = path - break - - assert delete_path is not None, "DELETE /interactions/{id} endpoint not found" + delete_path, _ = _interaction_operation(spec_dict, "delete", individual=True) print(f"✓ Delete endpoint: DELETE {delete_path}") +class TestOperationResolution: + """Keep structural resolution strict without depending on generated names.""" + + @pytest.mark.parametrize("as_union", [False, True]) + def test_model_schema_comes_from_create_operation(self, as_union): + model_schema: Final = {"properties": {"model": {"type": "string"}}, "required": ["model"]} + reference: Final = {"$ref": "#/components/schemas/RenamedModelRequest"} + body_schema: Final = ( + {"oneOf": [{"properties": {"agent": {"type": "string"}}}, reference]} if as_union else reference + ) + spec: Final = { + "paths": { + "/{version}/interactions": { + "post": { + "requestBody": {"required": True, "content": {"application/json": {"schema": body_schema}}} + } + } + }, + "components": { + "schemas": {"RenamedModelRequest": model_schema, "CreateModelInteractionParams": {"properties": {}}} + }, + } + assert _model_request_schema(spec) is model_schema + + @pytest.mark.parametrize("method,shared", [("get", False), ("delete", True)]) + def test_item_route_accepts_a_renamed_declared_identifier(self, method, shared): + parameter: Final = {"name": "renamedId", "in": "path", "required": True, "schema": {"type": "string"}} + parameters: Final = [{"$ref": "#/components/parameters/Identifier"}] + operation: Final = {"parameters": [] if shared else parameters} + path: Final = "/{version}/interactions/{renamedId}" + spec: Final = { + "paths": {path: {"parameters": parameters if shared else [], method: operation}}, + "components": {"parameters": {"Identifier": parameter}}, + } + assert _interaction_operation(spec, method, individual=True) == (path, operation) + + @pytest.mark.parametrize( + "path,parameter,error", + [ + ( + "/interactions/{id}/cancel", + {"required": True, "type": "string"}, + "Expected one GET interactions endpoint", + ), + ( + "/other_interactions/{id}", + {"required": True, "type": "string"}, + "Expected one GET interactions endpoint", + ), + ("/interactions/{id}", {"required": False, "type": "string"}, "must require its interaction ID"), + ("/interactions/{id}", {"required": True, "type": "integer"}, "must accept a string interaction ID"), + ], + ) + def test_item_route_rejects_incompatible_contracts(self, path, parameter, error): + spec: Final = { + "paths": { + path: { + "get": { + "parameters": [ + { + "name": "id", + "in": "path", + "required": parameter["required"], + "schema": {"type": parameter["type"]}, + } + ] + } + } + } + } + with pytest.raises(AssertionError, match=error): + _interaction_operation(spec, "get", individual=True) + + if __name__ == "__main__": # Quick manual test import httpx From 2a7abe8ca5dc45b950ac530987bee0d102f035f8 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 30 Sep 2026 23:56:31 +0200 Subject: [PATCH 16/24] test: remove duplicate catalog fields after upstream merge --- tests/unit/test_utils.py | 8 -------- 1 file changed, 8 deletions(-) diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index ee2f1faf52c..85437fd78a9 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -770,14 +770,12 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_above_272k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_flex": {"type": "number"}, - "cache_creation_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_priority": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens_batches": {"type": "number"}, "cache_creation_input_token_cost_above_272k_tokens_batches": {"type": "number"}, "cache_creation_input_token_cost_batches": {"type": "number"}, "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, - "cache_creation_input_token_cost_ultrafast": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, "cache_read_input_token_cost_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, @@ -788,7 +786,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_read_input_token_cost_above_272k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_272k_tokens_flex": {"type": "number"}, - "cache_read_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_512k_tokens": {"type": "number"}, "input_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"}, "cache_read_input_token_cost_batches": {"type": "number"}, @@ -813,12 +810,10 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_token_above_200k_tokens_batches": {"type": "number"}, "input_cost_per_token_above_256k_tokens": {"type": "number"}, "input_cost_per_token_above_272k_tokens": {"type": "number"}, - "input_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"}, "input_cost_per_token_above_512k_tokens": {"type": "number"}, "cache_read_input_token_cost_flex": {"type": "number"}, "cache_read_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost_balanced": {"type": "number"}, - "cache_read_input_token_cost_ultrafast": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens_priority": {"type": "number"}, "cache_read_input_token_cost_above_272k_tokens_priority": {"type": "number"}, "input_cost_per_token_flex": {"type": "number"}, @@ -848,7 +843,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cost_per_second": {"type": "number"}, "input_cost_per_second": {"type": "number"}, "input_cost_per_token": {"type": "number"}, - "input_cost_per_token_ultrafast": {"type": "number"}, "input_cost_per_token_above_128k_tokens": {"type": "number"}, "input_cost_per_audio_token_batches": {"type": "number"}, "input_cost_per_image_token_batches": {"type": "number"}, @@ -918,14 +912,12 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_second_1080p": {"type": "number"}, "output_cost_per_second_4k": {"type": "number"}, "output_cost_per_token": {"type": "number"}, - "output_cost_per_token_ultrafast": {"type": "number"}, "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens_batches": {"type": "number"}, "output_cost_per_token_above_256k_tokens": {"type": "number"}, "output_cost_per_token_above_272k_tokens": {"type": "number"}, - "output_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"}, "output_cost_per_token_above_512k_tokens": {"type": "number"}, "output_cost_per_token_batches": {"type": "number"}, "output_cost_per_reasoning_token": {"type": "number"}, From 5b13a21517ad8b211b3f82f4d34d72ac15976cfd Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 00:41:25 +0200 Subject: [PATCH 17/24] fix: synchronize trace routes and generated API types --- backend/routes/allowlist.py | 1 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 8 ++++++-- 2 files changed, 7 insertions(+), 2 deletions(-) diff --git a/backend/routes/allowlist.py b/backend/routes/allowlist.py index 232561dd154..5c280c22d80 100644 --- a/backend/routes/allowlist.py +++ b/backend/routes/allowlist.py @@ -60,6 +60,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = ( # Tools / agents (registry & policy admin) "/v1/tool/", "/v1/agents", + "/v1/traces", # Guardrails admin "/v2/guardrails/", # MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 17ca656a402..ea4193209aa 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -75423,7 +75423,9 @@ export interface operations { }; get_agent_trace_v1_traces__trace_id__get: { parameters: { - query?: never; + query?: { + trace_ref?: string; + }; header?: never; path: { trace_id: string; @@ -75454,7 +75456,9 @@ export interface operations { }; get_agent_trace_span_v1_traces__trace_id__spans__span_id__get: { parameters: { - query?: never; + query?: { + trace_ref?: string; + }; header?: never; path: { trace_id: string; From 7b4df9a706e755c091a880ad6182e5b44f07dae1 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 02:38:47 +0200 Subject: [PATCH 18/24] test: harden batches mock scope and MCP cancellation deadline handshake - batches _raw_batches_request now provides the ASGI scope real starlette requests carry, so the OTLP-trace branch of _read_request_body no longer collapses the body and the metadata 400 names the offending field again - test_outer_deadline_delivers_session_termination gives the initialize/ tools-call handshake its own 2s budget and keeps the original 0.2s fail_after as the real outer cancellation deadline; the in-flight task is cancelled and awaited in finally so session-termination DELETE is delivered before the caller resumes Validated: batches 155 passed; test_mcp_client 409 passed; 14 cancel/teardown variants green under 8-way CPU load. --- .../proxy/batches_endpoints/test_endpoints.py | 14 ++++++++++++++ .../test_mcp_client.py | 19 ++++++++++++++----- 2 files changed, 28 insertions(+), 5 deletions(-) diff --git a/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py b/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py index 2d597abf3b8..8e3d25552e0 100644 --- a/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py @@ -1038,6 +1038,20 @@ def _raw_batches_request(body: Dict[str, Any]) -> MagicMock: request.headers = {"Content-Type": "application/json"} request.client = MagicMock() request.client.host = "127.0.0.1" + request.scope = { + "type": "http", + "asgi": {"version": "3.0", "spec_version": "2.3"}, + "http_version": "1.1", + "method": "POST", + "scheme": "http", + "path": "/v1/batches", + "raw_path": b"/v1/batches", + "query_string": b"", + "root_path": "", + "headers": [(b"content-type", b"application/json"), (b"host", b"localhost")], + "client": ("127.0.0.1", 54321), + "server": ("localhost", 8000), + } request.body = AsyncMock(return_value=json.dumps(body).encode()) return request diff --git a/tests/unit/experimental_mcp_client/test_mcp_client.py b/tests/unit/experimental_mcp_client/test_mcp_client.py index 1a56227b008..3860f0f44bb 100644 --- a/tests/unit/experimental_mcp_client/test_mcp_client.py +++ b/tests/unit/experimental_mcp_client/test_mcp_client.py @@ -2644,12 +2644,21 @@ async def test_outer_deadline_delivers_session_termination(termination: str, gro client: Final = _MockTransportClient(respond, server_url="https://example.com/mcp", timeout=30) async def invoke(): - with anyio.fail_after(0.2): - pending: Final = client.call_tool(CallToolRequestParams(name="slow", arguments={}), raise_on_error=raise_on_error) - if grouped: - await asyncio.gather(pending) - else: + pending: Final = asyncio.ensure_future(client.call_tool(CallToolRequestParams(name="slow", arguments={}), raise_on_error=raise_on_error)) + try: + with anyio.fail_after(2.0): + await started.wait() + with anyio.fail_after(0.2): + if grouped: + await asyncio.gather(pending) + else: + await pending + finally: + pending.cancel() + try: await pending + except BaseException: + pass before: Final = anyio.current_time() with pytest.raises(TimeoutError): From b273b979160d9b0db852940445e5a7f59f719996 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 02:47:31 +0200 Subject: [PATCH 19/24] style: apply ruff format to MCP client test tree file --- .../test_mcp_client.py | 120 +++++++++++++----- 1 file changed, 86 insertions(+), 34 deletions(-) diff --git a/tests/unit/experimental_mcp_client/test_mcp_client.py b/tests/unit/experimental_mcp_client/test_mcp_client.py index 3860f0f44bb..d16de968c25 100644 --- a/tests/unit/experimental_mcp_client/test_mcp_client.py +++ b/tests/unit/experimental_mcp_client/test_mcp_client.py @@ -597,7 +597,9 @@ class TestExecuteSessionOperationSurfacesTransportError: raise _FakeExceptionGroup("transport", [_FakeExceptionGroup("reader", failures)]) self._make_session(session_class, initialize) - expected: Final = close_error if failure_phase == "early" else connect_error if failure_phase == "mixed" else cancelled + expected: Final = ( + close_error if failure_phase == "early" else connect_error if failure_phase == "mixed" else cancelled + ) with pytest.raises(type(expected)) as caught: await client._execute_session_operation(self._make_transport(close_transport), AsyncMock(), http_client) assert caught.value is expected @@ -617,7 +619,6 @@ class TestExecuteSessionOperationSurfacesTransportError: result = await client._execute_session_operation(transport_ctx, _op) assert result == "done" - @pytest.mark.asyncio @patch("litellm.experimental_mcp_client.client.ClientSession") async def test_session_entry_failure_still_closes_transport(self, session_class): @@ -1235,11 +1236,7 @@ def test_mcp_extra_matches_proxy_extra_and_supports_streamable_http(): sdk2_names: Final = frozenset(("mcp", "httpx2", "pydantic")) mcp_extra: Final = {Requirement(req).name: req for req in extras["mcp"]} - assert mcp_extra == { - name: req - for req in extras["proxy"] - if (name := Requirement(req).name) in sdk2_names - } + assert mcp_extra == {name: req for req in extras["proxy"] if (name := Requirement(req).name) in sdk2_names} specifier: Final = Requirement(mcp_extra["mcp"]).specifier assert not specifier.contains("1.28.1") @@ -1639,7 +1636,9 @@ async def test_http_response_handler_preserves_success_and_http_errors(status_co @pytest.mark.asyncio async def test_http_status_check_allows_auth_refresh_before_rejecting() -> None: - from litellm.proxy._experimental.mcp_server.outbound_credentials.client_credentials import ClientCredentialsBearerAuth + from litellm.proxy._experimental.mcp_server.outbound_credentials.client_credentials import ( + ClientCredentialsBearerAuth, + ) seen = [] @@ -1892,14 +1891,20 @@ async def test_sse_read_failure_is_preserved() -> None: @pytest.mark.parametrize("protocol_version", ["auto", "2025-06-18"]) @pytest.mark.parametrize("transport", [MCPTransport.sse, MCPTransport.stdio]) @pytest.mark.parametrize("mode", ["ok", "closed", "silent"]) -async def test_transport_completion_and_normal_messages(transport: MCPTransport, mode: str, protocol_version: str) -> None: +async def test_transport_completion_and_normal_messages( + transport: MCPTransport, mode: str, protocol_version: str +) -> None: from mcp import ClientSession from litellm.proxy._experimental.mcp_server.rest_endpoints import _connection_error_message logging_callback: Final = AsyncMock() read_timeout: Final = 0.2 if mode == "silent" else 30 client: Final = MCPClient( - server_url="https://example.com/sse", transport_type=transport, timeout=read_timeout, logging_callback=logging_callback, protocol_version=protocol_version + server_url="https://example.com/sse", + transport_type=transport, + timeout=read_timeout, + logging_callback=logging_callback, + protocol_version=protocol_version, ) async def operation(session: ClientSession) -> CallToolResult: @@ -2461,8 +2466,15 @@ def test_client_import_before_proxy_credentials_succeeds_in_fresh_process(): import subprocess result = subprocess.run( - [sys.executable, "-c", "import litellm.experimental_mcp_client.client; from litellm.proxy._experimental.mcp_server.mcp_server_manager import MCPServerManager; print(MCPServerManager.__name__)"], - capture_output=True, text=True, timeout=60, check=False, + [ + sys.executable, + "-c", + "import litellm.experimental_mcp_client.client; from litellm.proxy._experimental.mcp_server.mcp_server_manager import MCPServerManager; print(MCPServerManager.__name__)", + ], + capture_output=True, + text=True, + timeout=60, + check=False, ) assert result.returncode == 0, result.stderr assert result.stdout.strip() == "MCPServerManager" @@ -2495,8 +2507,10 @@ async def test_request_auth_preview_uses_the_same_effective_headers_as_egress() from litellm.proxy._experimental.mcp_server.outbound_credentials.httpx_auth import StaticHeaderAuth client: Final = MCPClient( - server_url="https://upstream.example/mcp", auth_type=MCPAuth.bearer_token, - resolved_auth=StaticHeaderAuth("Bearer resolved"), extra_headers={"X-Trace": "trace"}, + server_url="https://upstream.example/mcp", + auth_type=MCPAuth.bearer_token, + resolved_auth=StaticHeaderAuth("Bearer resolved"), + extra_headers={"X-Trace": "trace"}, ) request: Final = await client.prepare_request_auth() assert request.method == "POST" @@ -2520,18 +2534,29 @@ async def test_expired_session_preserves_sdk_error_and_next_operation_reinitiali return httpx2.Response(202) requests.append((payload["method"], request.headers.get("mcp-session-id"))) if payload["method"] == "initialize": - return httpx2.Response(200, headers={"mcp-session-id": f"session-{len(requests)}"}, json={ - "jsonrpc": "2.0", "id": payload["id"], "result": { - "protocolVersion": "2025-06-18", "capabilities": {}, - "serverInfo": {"name": "expiry-test", "version": "1"}, + return httpx2.Response( + 200, + headers={"mcp-session-id": f"session-{len(requests)}"}, + json={ + "jsonrpc": "2.0", + "id": payload["id"], + "result": { + "protocolVersion": "2025-06-18", + "capabilities": {}, + "serverInfo": {"name": "expiry-test", "version": "1"}, + }, }, - }) + ) if len(requests) == 2: if rpc_error: - return httpx2.Response(404, json={ - "jsonrpc": "2.0", "id": payload["id"], - "error": {"code": METHOD_NOT_FOUND, "message": "Tool catalog unavailable"}, - }) + return httpx2.Response( + 404, + json={ + "jsonrpc": "2.0", + "id": payload["id"], + "error": {"code": METHOD_NOT_FOUND, "message": "Tool catalog unavailable"}, + }, + ) return httpx2.Response(404) return httpx2.Response(200, json={"jsonrpc": "2.0", "id": payload["id"], "result": {"tools": []}}) @@ -2547,7 +2572,12 @@ async def test_expired_session_preserves_sdk_error_and_next_operation_reinitiali streamable_http_client(client.server_url, http_client=http_client), lambda session: session.list_tools() ) assert result.tools == [] - assert requests == [("initialize", None), ("tools/list", "session-1"), ("initialize", None), ("tools/list", "session-3")] + assert requests == [ + ("initialize", None), + ("tools/list", "session-1"), + ("initialize", None), + ("tools/list", "session-3"), + ] @pytest.mark.asyncio @@ -2604,7 +2634,9 @@ def test_public_mcp_import_preserves_incompatible_sdk_error() -> None: @pytest.mark.parametrize("grouped", (False, True)) @pytest.mark.parametrize("raise_on_error", (False, True)) @pytest.mark.parametrize("termination", ("ok", "failure", "hang")) -async def test_outer_deadline_delivers_session_termination(termination: str, grouped: bool, raise_on_error: bool) -> None: +async def test_outer_deadline_delivers_session_termination( + termination: str, grouped: bool, raise_on_error: bool +) -> None: deleted: Final = asyncio.Event() started: Final = asyncio.Event() @@ -2644,7 +2676,9 @@ async def test_outer_deadline_delivers_session_termination(termination: str, gro client: Final = _MockTransportClient(respond, server_url="https://example.com/mcp", timeout=30) async def invoke(): - pending: Final = asyncio.ensure_future(client.call_tool(CallToolRequestParams(name="slow", arguments={}), raise_on_error=raise_on_error)) + pending: Final = asyncio.ensure_future( + client.call_tool(CallToolRequestParams(name="slow", arguments={}), raise_on_error=raise_on_error) + ) try: with anyio.fail_after(2.0): await started.wait() @@ -2858,7 +2892,9 @@ async def test_cancellation_delivers_termination_over_tcp( listener: Final = await asyncio.start_server(handle_connection, "127.0.0.1", 0) port: Final = listener.sockets[0].getsockname()[1] client: Final = MCPClient( - server_url=f"http://127.0.0.1:{port}/mcp", protocol_version=protocol_version, timeout=2 if cancel_mode == "read_timeout" else 30 + server_url=f"http://127.0.0.1:{port}/mcp", + protocol_version=protocol_version, + timeout=2 if cancel_mode == "read_timeout" else 30, ) async def calls(): @@ -2942,16 +2978,32 @@ async def test_configured_upstream_revision_is_offered_and_checked(revision, acc assert payload.params["protocolVersion"] == offered assert ("sampling" in payload.params["capabilities"]) == callbacks assert ("elicitation" in payload.params["capabilities"]) == callbacks - return httpx2.Response(200, json={ - "jsonrpc": "2.0", "id": payload.id, - "result": {"protocolVersion": offered if accepted else "unsupported", - "capabilities": {"tools": {}}, "serverInfo": {"name": "upstream", "version": "1"}}, - }) + return httpx2.Response( + 200, + json={ + "jsonrpc": "2.0", + "id": payload.id, + "result": { + "protocolVersion": offered if accepted else "unsupported", + "capabilities": {"tools": {}}, + "serverInfo": {"name": "upstream", "version": "1"}, + }, + }, + ) assert accepted, "No operation may execute after failed version negotiation" - return httpx2.Response(200, json={"jsonrpc": "2.0", "id": payload.id, "result": {"tools": [{"name": "echo", "inputSchema": {"type": "object"}}]}}) + return httpx2.Response( + 200, + json={ + "jsonrpc": "2.0", + "id": payload.id, + "result": {"tools": [{"name": "echo", "inputSchema": {"type": "object"}}]}, + }, + ) client = _MockTransportClient( - respond, server_url="https://example.com/mcp", protocol_version=revision, + respond, + server_url="https://example.com/mcp", + protocol_version=revision, sampling_callback=AsyncMock() if callbacks else None, elicitation_callback=AsyncMock() if callbacks else None, ) From 0b630f4885dce3ea2ae05be9ba9641c52ee81c8b Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 03:20:34 +0200 Subject: [PATCH 20/24] fix: update vulnerable GitPython and Tornado dependencies --- pyproject.toml | 3 ++- uv.lock | 31 ++++++++++++++++--------------- 2 files changed, 18 insertions(+), 16 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index a81c75c2e0b..71b6ea2ee30 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -335,7 +335,8 @@ exclude = [ [tool.uv] constraint-dependencies = [ - "tornado>=6.5.8", + "tornado>=6.5.10", + "gitpython>=3.1.62", "aiohttp>=3.14.2,<4.0", "packaging>=24.0", "soupsieve>=2.8.4", diff --git a/uv.lock b/uv.lock index 2d31641a5e3..7d62a0ab8c2 100644 --- a/uv.lock +++ b/uv.lock @@ -21,11 +21,12 @@ members = [ ] constraints = [ { name = "aiohttp", specifier = ">=3.14.2,<4.0" }, + { name = "gitpython", specifier = ">=3.1.62" }, { name = "httplib2", specifier = ">=0.32.0" }, { name = "packaging", specifier = ">=24.0" }, { name = "setuptools", specifier = ">=83.0.0" }, { name = "soupsieve", specifier = ">=2.8.4" }, - { name = "tornado", specifier = ">=6.5.8" }, + { name = "tornado", specifier = ">=6.5.10" }, ] overrides = [ { name = "cryptography", specifier = ">=50.0.0,<51.0" }, @@ -2375,14 +2376,14 @@ wheels = [ [[package]] name = "gitpython" -version = "3.1.61" +version = "3.1.62" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "gitdb" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6f/61/3285044215fb596bf093e39ccb96ece0a1076a8ca57a61e069a6a33cdb1b/gitpython-3.1.61.tar.gz", hash = "sha256:f51c24d8c0f733a195447385f5774a5dfe8767f5acfd7994a33755644c6ecc95", size = 231680, upload-time = "2026-08-28T11:01:13.761Z" } +sdist = { url = "https://files.pythonhosted.org/packages/e0/db/3ca813cbacb23ab6fe46ff38a9b5ef8e73e970c8051f2ce903aacafe0446/gitpython-3.1.62.tar.gz", hash = "sha256:1791de66309bc0c7cfca40bf8d2e3de7ca091cbf94e6051be1ad0722c61062af", size = 231728, upload-time = "2026-09-07T02:57:21.155Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/6f/5e/49cc172da4d0578644ba37cec5cb365b1fefc603b26edea9bcac1c7f830a/gitpython-3.1.61-py3-none-any.whl", hash = "sha256:8ab28c9da863cdd9e7d7694ec46cf3e6c9a12d8a30a1acd3447aec11975d530c", size = 222118, upload-time = "2026-08-28T11:01:12.262Z" }, + { url = "https://files.pythonhosted.org/packages/d6/0b/29d7965215f8ef830a7ca1f42997fe13e5693d85e9edb18f938d063ef5f2/gitpython-3.1.62-py3-none-any.whl", hash = "sha256:7002251225e10e29d2e1f49e6532613fe5d5d9f0b6f1f02997a52b38fe56899e", size = 222753, upload-time = "2026-09-07T02:57:19.762Z" }, ] [[package]] @@ -9834,19 +9835,19 @@ wheels = [ [[package]] name = "tornado" -version = "6.5.8" +version = "6.5.10" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/10/d3/343e5bb989d6515b1646cf3d40135d73f3d5e45339bded401b56cdac24dd/tornado-6.5.8.tar.gz", hash = "sha256:9452e1b208a8bd771e2cb1f2ff564985b9b214bdebbe622793e1799e0a6bd23f", size = 520493, upload-time = "2026-08-07T02:12:42.971Z" } +sdist = { url = "https://files.pythonhosted.org/packages/06/61/53d562a57b28c08eda40b258c0f975e360541943ad7c7bef897a40caafda/tornado-6.5.10.tar.gz", hash = "sha256:a6b1ccd08c04b4a06fb5aeb381be99de5ad1e5375c1785e31d78c880feb57687", size = 537910, upload-time = "2026-09-15T13:47:48.73Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f2/d5/007086fd8df5489338e204f65adce33fd4f21a4999dbb2b9cff2f897b5f4/tornado-6.5.8-cp39-abi3-macosx_10_9_universal2.whl", hash = "sha256:cc6aa787d7cfab7c3d35189dc7a56fbd2399a569624c730c6b55b3d6531d0403", size = 449487, upload-time = "2026-08-07T02:12:28.682Z" }, - { url = "https://files.pythonhosted.org/packages/70/c8/5a24a99495903f594f6a199dd7beead1cbc0a13e2cb9102727bcaaf2a997/tornado-6.5.8-cp39-abi3-macosx_10_9_x86_64.whl", hash = "sha256:9715b5eb79735b2bcd454ce216a9275b7c0470e64ea1bf5742f78b2f72b26eeb", size = 447649, upload-time = "2026-08-07T02:12:30.306Z" }, - { url = "https://files.pythonhosted.org/packages/6e/de/f2e733f386b85962d1b1dc82cd63d169b5b4580062b35397eac9244a41fe/tornado-6.5.8-cp39-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:547d63f450d570c14fe0e8db2cfb14c9bbd1c2503b4a6612586267955aa47b58", size = 450707, upload-time = "2026-08-07T02:12:31.95Z" }, - { url = "https://files.pythonhosted.org/packages/0b/94/20efeee9a01c141e9ac47c397f81679dfda24b32768fc4fff24e76d36c2c/tornado-6.5.8-cp39-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7e2360a0ffbe145eca8af0b19cb7203d79b1a98dd4cccdd6b368f6f49c2e3808", size = 451677, upload-time = "2026-08-07T02:12:33.512Z" }, - { url = "https://files.pythonhosted.org/packages/42/ec/a96ccb8ccf0de2b7bc2c5fa1608a4803735018242e90c4882365a9fd418f/tornado-6.5.8-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:5d242290bdf7ab3151bc1065fdd75c0dcc21cbc7b49f22a4c56329c2d6566d22", size = 451510, upload-time = "2026-08-07T02:12:35.346Z" }, - { url = "https://files.pythonhosted.org/packages/29/b5/93185859245ad3f00e62175f29607346788b696369347f0146e0421286bb/tornado-6.5.8-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:7b94ff0e128fe0542f3bd331fb44d06260fc4ac16881545159f34ef08aad4195", size = 450917, upload-time = "2026-08-07T02:12:36.963Z" }, - { url = "https://files.pythonhosted.org/packages/97/cf/fe33cf062834487d34d1559746a4a12521033c22645b6d74d4bca702e018/tornado-6.5.8-cp39-abi3-win32.whl", hash = "sha256:67832909c4779c64942380cb5f044a5c6163d00831472d80e25e115de9917836", size = 451952, upload-time = "2026-08-07T02:12:38.512Z" }, - { url = "https://files.pythonhosted.org/packages/cb/e1/468ad54333e92ccb62627e62cb88e5fc14a2171daa67ed47b1b8542d5b86/tornado-6.5.8-cp39-abi3-win_amd64.whl", hash = "sha256:11881db6b7c168494be2c2d12e65931451bdf7ee718535418ae1d8855dd5a0ee", size = 452391, upload-time = "2026-08-07T02:12:39.971Z" }, - { url = "https://files.pythonhosted.org/packages/ad/3e/cd5e4f06e34cde33b8ef66cf36aa2b5ad46354cc1af7d2136bbe365fee1d/tornado-6.5.8-cp39-abi3-win_arm64.whl", hash = "sha256:68a7468c7e289f8514d7d664101753903217eff1bb6822c6b5994a0b5f5bcb26", size = 451411, upload-time = "2026-08-07T02:12:41.469Z" }, + { url = "https://files.pythonhosted.org/packages/cd/5b/ff5fc58fa2427c30dea74c90053f4fc5eda1e7f3833ed3ecc7147fe2b311/tornado-6.5.10-cp39-abi3-macosx_10_9_universal2.whl", hash = "sha256:9261783640e23258694a9ff0795df430a5a7b0a651d3dd53dd0969ad6be16da7", size = 465883, upload-time = "2026-09-15T13:47:35.463Z" }, + { url = "https://files.pythonhosted.org/packages/ad/f5/cd7be26c34a3315532f3aef5f092465da8f59c334dd439d3c14aaef16461/tornado-6.5.10-cp39-abi3-macosx_10_9_x86_64.whl", hash = "sha256:83e6cf438b106c6b3852d70960967bb1b70c87438050dca0981e4b9aa751a4c1", size = 464046, upload-time = "2026-09-15T13:47:37.178Z" }, + { url = "https://files.pythonhosted.org/packages/60/33/df6d7d04854a58619f8349a51e3edb138324130a7562b0bb21f115bb940f/tornado-6.5.10-cp39-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:bdf942448169e5336451d0494d7e3d81cfa726d5aa312affdc4682dd62a62f6d", size = 467096, upload-time = "2026-09-15T13:47:38.559Z" }, + { url = "https://files.pythonhosted.org/packages/29/17/cc35dff68272d685cffd8600ffafbd8067e7d05e7348d9f80caddffbbd5f/tornado-6.5.10-cp39-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:69acca6501eed74582b76dbbceee2a91613f54728e3e418346000d7103101676", size = 468067, upload-time = "2026-09-15T13:47:40.085Z" }, + { url = "https://files.pythonhosted.org/packages/c3/01/6e5349b4e1a53a4b4972a6716785e1fe7407f312063c3972690af8ff301b/tornado-6.5.10-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:66aaa3f57d30c6e6becee83ff28055d5930ac724214bde99393eefda83d5e015", size = 467901, upload-time = "2026-09-15T13:47:41.576Z" }, + { url = "https://files.pythonhosted.org/packages/28/5e/b4facf94370dba006819c8d304376f8b9fbec6b935b5e51bf45823a9790b/tornado-6.5.10-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:4bd192b959f9128fb99b8898148070ba4574c9589b78bce42d1851131fe85828", size = 467308, upload-time = "2026-09-15T13:47:43.145Z" }, + { url = "https://files.pythonhosted.org/packages/56/ae/047938e828cafc8eca4c908fafb6588fee944e3af39a0af9d7b602499ae5/tornado-6.5.10-cp39-abi3-win32.whl", hash = "sha256:302eb1e0e3e159314eb591920529fdea80acca92df5510a2cec5bbd4f099ec72", size = 468387, upload-time = "2026-09-15T13:47:44.556Z" }, + { url = "https://files.pythonhosted.org/packages/d8/d4/5901517f05affd752490f6a654ba31b7474664e8dd80bd045a00c220bd88/tornado-6.5.10-cp39-abi3-win_amd64.whl", hash = "sha256:37ae8f150cecfdbf747fc4e12f5e9a97ecd8cf1d4cdb3f119e2de84b11196918", size = 468828, upload-time = "2026-09-15T13:47:45.961Z" }, + { url = "https://files.pythonhosted.org/packages/f3/1a/fd497f3a7f7b74bb04f4b94536b5c9f80742b5d50501fd27977652ddec16/tornado-6.5.10-cp39-abi3-win_arm64.whl", hash = "sha256:ce045d3c298fddd30e89a2777f97039d1b641eb9518ac7b26a4721903539c694", size = 467847, upload-time = "2026-09-15T13:47:47.283Z" }, ] [[package]] From 9738840e3f450adc98ebe9c67caf276950a24432 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 04:27:49 +0200 Subject: [PATCH 21/24] fix(rust_bridge): name NativeTraceStorage.query stub parameter as the native binding stubtest reports the stub parameter "sql" inconsistent with the pyo3 runtime parameter "query" (python-bridge routes/traces.rs), failing the rust-wheel job. Python callers pass the argument positionally, so only the stub name changes; upstream main still carries the stale name. --- litellm/rust_bridge/_native.pyi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/rust_bridge/_native.pyi b/litellm/rust_bridge/_native.pyi index ff8bc198f27..206c0f78ed8 100644 --- a/litellm/rust_bridge/_native.pyi +++ b/litellm/rust_bridge/_native.pyi @@ -31,7 +31,7 @@ class NativeTraceStorage: def ensure_schema(self, trace_retention_days: int, spend_log_retention_days: int) -> Future[None]: ... def insert_rows(self, table: str, rows: Sequence[Mapping[str, JsonValue]]) -> Future[None]: ... def lens_query(self, name: str, parameters: Mapping[str, str | int | Sequence[str]]) -> Future[str]: ... - def query(self, sql: str, parameters: Mapping[str, str | int | Sequence[str]]) -> Future[str]: ... + def query(self, query: str, parameters: Mapping[str, str | int | Sequence[str]]) -> Future[str]: ... @final class NativeDiagnosticProcessor: From b95c3dc0669ce2ce591407fcf2a5a9091bc06b0b Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 04:27:49 +0200 Subject: [PATCH 22/24] Revert "fix: update vulnerable GitPython and Tornado dependencies" This reverts commit d8f1ca1d39fcbc32aee77c2476be62e004c8d89b. --- pyproject.toml | 3 +-- uv.lock | 31 +++++++++++++++---------------- 2 files changed, 16 insertions(+), 18 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 71b6ea2ee30..a81c75c2e0b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -335,8 +335,7 @@ exclude = [ [tool.uv] constraint-dependencies = [ - "tornado>=6.5.10", - "gitpython>=3.1.62", + "tornado>=6.5.8", "aiohttp>=3.14.2,<4.0", "packaging>=24.0", "soupsieve>=2.8.4", diff --git a/uv.lock b/uv.lock index 7d62a0ab8c2..2d31641a5e3 100644 --- a/uv.lock +++ b/uv.lock @@ -21,12 +21,11 @@ members = [ ] constraints = [ { name = "aiohttp", specifier = ">=3.14.2,<4.0" }, - { name = "gitpython", specifier = ">=3.1.62" }, { name = "httplib2", specifier = ">=0.32.0" }, { name = "packaging", specifier = ">=24.0" }, { name = "setuptools", specifier = ">=83.0.0" }, { name = "soupsieve", specifier = ">=2.8.4" }, - { name = "tornado", specifier = ">=6.5.10" }, + { name = "tornado", specifier = ">=6.5.8" }, ] overrides = [ { name = "cryptography", specifier = ">=50.0.0,<51.0" }, @@ -2376,14 +2375,14 @@ wheels = [ [[package]] name = "gitpython" -version = "3.1.62" +version = "3.1.61" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "gitdb" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/e0/db/3ca813cbacb23ab6fe46ff38a9b5ef8e73e970c8051f2ce903aacafe0446/gitpython-3.1.62.tar.gz", hash = "sha256:1791de66309bc0c7cfca40bf8d2e3de7ca091cbf94e6051be1ad0722c61062af", size = 231728, upload-time = "2026-09-07T02:57:21.155Z" } +sdist = { url = "https://files.pythonhosted.org/packages/6f/61/3285044215fb596bf093e39ccb96ece0a1076a8ca57a61e069a6a33cdb1b/gitpython-3.1.61.tar.gz", hash = "sha256:f51c24d8c0f733a195447385f5774a5dfe8767f5acfd7994a33755644c6ecc95", size = 231680, upload-time = "2026-08-28T11:01:13.761Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/d6/0b/29d7965215f8ef830a7ca1f42997fe13e5693d85e9edb18f938d063ef5f2/gitpython-3.1.62-py3-none-any.whl", hash = "sha256:7002251225e10e29d2e1f49e6532613fe5d5d9f0b6f1f02997a52b38fe56899e", size = 222753, upload-time = "2026-09-07T02:57:19.762Z" }, + { url = "https://files.pythonhosted.org/packages/6f/5e/49cc172da4d0578644ba37cec5cb365b1fefc603b26edea9bcac1c7f830a/gitpython-3.1.61-py3-none-any.whl", hash = "sha256:8ab28c9da863cdd9e7d7694ec46cf3e6c9a12d8a30a1acd3447aec11975d530c", size = 222118, upload-time = "2026-08-28T11:01:12.262Z" }, ] [[package]] @@ -9835,19 +9834,19 @@ wheels = [ [[package]] name = "tornado" -version = "6.5.10" +version = "6.5.8" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/06/61/53d562a57b28c08eda40b258c0f975e360541943ad7c7bef897a40caafda/tornado-6.5.10.tar.gz", hash = "sha256:a6b1ccd08c04b4a06fb5aeb381be99de5ad1e5375c1785e31d78c880feb57687", size = 537910, upload-time = "2026-09-15T13:47:48.73Z" } +sdist = { url = "https://files.pythonhosted.org/packages/10/d3/343e5bb989d6515b1646cf3d40135d73f3d5e45339bded401b56cdac24dd/tornado-6.5.8.tar.gz", hash = "sha256:9452e1b208a8bd771e2cb1f2ff564985b9b214bdebbe622793e1799e0a6bd23f", size = 520493, upload-time = "2026-08-07T02:12:42.971Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/cd/5b/ff5fc58fa2427c30dea74c90053f4fc5eda1e7f3833ed3ecc7147fe2b311/tornado-6.5.10-cp39-abi3-macosx_10_9_universal2.whl", hash = "sha256:9261783640e23258694a9ff0795df430a5a7b0a651d3dd53dd0969ad6be16da7", size = 465883, upload-time = "2026-09-15T13:47:35.463Z" }, - { url = "https://files.pythonhosted.org/packages/ad/f5/cd7be26c34a3315532f3aef5f092465da8f59c334dd439d3c14aaef16461/tornado-6.5.10-cp39-abi3-macosx_10_9_x86_64.whl", hash = "sha256:83e6cf438b106c6b3852d70960967bb1b70c87438050dca0981e4b9aa751a4c1", size = 464046, upload-time = "2026-09-15T13:47:37.178Z" }, - { url = "https://files.pythonhosted.org/packages/60/33/df6d7d04854a58619f8349a51e3edb138324130a7562b0bb21f115bb940f/tornado-6.5.10-cp39-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:bdf942448169e5336451d0494d7e3d81cfa726d5aa312affdc4682dd62a62f6d", size = 467096, upload-time = "2026-09-15T13:47:38.559Z" }, - { url = "https://files.pythonhosted.org/packages/29/17/cc35dff68272d685cffd8600ffafbd8067e7d05e7348d9f80caddffbbd5f/tornado-6.5.10-cp39-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:69acca6501eed74582b76dbbceee2a91613f54728e3e418346000d7103101676", size = 468067, upload-time = "2026-09-15T13:47:40.085Z" }, - { url = "https://files.pythonhosted.org/packages/c3/01/6e5349b4e1a53a4b4972a6716785e1fe7407f312063c3972690af8ff301b/tornado-6.5.10-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:66aaa3f57d30c6e6becee83ff28055d5930ac724214bde99393eefda83d5e015", size = 467901, upload-time = "2026-09-15T13:47:41.576Z" }, - { url = "https://files.pythonhosted.org/packages/28/5e/b4facf94370dba006819c8d304376f8b9fbec6b935b5e51bf45823a9790b/tornado-6.5.10-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:4bd192b959f9128fb99b8898148070ba4574c9589b78bce42d1851131fe85828", size = 467308, upload-time = "2026-09-15T13:47:43.145Z" }, - { url = "https://files.pythonhosted.org/packages/56/ae/047938e828cafc8eca4c908fafb6588fee944e3af39a0af9d7b602499ae5/tornado-6.5.10-cp39-abi3-win32.whl", hash = "sha256:302eb1e0e3e159314eb591920529fdea80acca92df5510a2cec5bbd4f099ec72", size = 468387, upload-time = "2026-09-15T13:47:44.556Z" }, - { url = "https://files.pythonhosted.org/packages/d8/d4/5901517f05affd752490f6a654ba31b7474664e8dd80bd045a00c220bd88/tornado-6.5.10-cp39-abi3-win_amd64.whl", hash = "sha256:37ae8f150cecfdbf747fc4e12f5e9a97ecd8cf1d4cdb3f119e2de84b11196918", size = 468828, upload-time = "2026-09-15T13:47:45.961Z" }, - { url = "https://files.pythonhosted.org/packages/f3/1a/fd497f3a7f7b74bb04f4b94536b5c9f80742b5d50501fd27977652ddec16/tornado-6.5.10-cp39-abi3-win_arm64.whl", hash = "sha256:ce045d3c298fddd30e89a2777f97039d1b641eb9518ac7b26a4721903539c694", size = 467847, upload-time = "2026-09-15T13:47:47.283Z" }, + { url = "https://files.pythonhosted.org/packages/f2/d5/007086fd8df5489338e204f65adce33fd4f21a4999dbb2b9cff2f897b5f4/tornado-6.5.8-cp39-abi3-macosx_10_9_universal2.whl", hash = "sha256:cc6aa787d7cfab7c3d35189dc7a56fbd2399a569624c730c6b55b3d6531d0403", size = 449487, upload-time = "2026-08-07T02:12:28.682Z" }, + { url = "https://files.pythonhosted.org/packages/70/c8/5a24a99495903f594f6a199dd7beead1cbc0a13e2cb9102727bcaaf2a997/tornado-6.5.8-cp39-abi3-macosx_10_9_x86_64.whl", hash = "sha256:9715b5eb79735b2bcd454ce216a9275b7c0470e64ea1bf5742f78b2f72b26eeb", size = 447649, upload-time = "2026-08-07T02:12:30.306Z" }, + { url = "https://files.pythonhosted.org/packages/6e/de/f2e733f386b85962d1b1dc82cd63d169b5b4580062b35397eac9244a41fe/tornado-6.5.8-cp39-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:547d63f450d570c14fe0e8db2cfb14c9bbd1c2503b4a6612586267955aa47b58", size = 450707, upload-time = "2026-08-07T02:12:31.95Z" }, + { url = "https://files.pythonhosted.org/packages/0b/94/20efeee9a01c141e9ac47c397f81679dfda24b32768fc4fff24e76d36c2c/tornado-6.5.8-cp39-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7e2360a0ffbe145eca8af0b19cb7203d79b1a98dd4cccdd6b368f6f49c2e3808", size = 451677, upload-time = "2026-08-07T02:12:33.512Z" }, + { url = "https://files.pythonhosted.org/packages/42/ec/a96ccb8ccf0de2b7bc2c5fa1608a4803735018242e90c4882365a9fd418f/tornado-6.5.8-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:5d242290bdf7ab3151bc1065fdd75c0dcc21cbc7b49f22a4c56329c2d6566d22", size = 451510, upload-time = "2026-08-07T02:12:35.346Z" }, + { url = "https://files.pythonhosted.org/packages/29/b5/93185859245ad3f00e62175f29607346788b696369347f0146e0421286bb/tornado-6.5.8-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:7b94ff0e128fe0542f3bd331fb44d06260fc4ac16881545159f34ef08aad4195", size = 450917, upload-time = "2026-08-07T02:12:36.963Z" }, + { url = "https://files.pythonhosted.org/packages/97/cf/fe33cf062834487d34d1559746a4a12521033c22645b6d74d4bca702e018/tornado-6.5.8-cp39-abi3-win32.whl", hash = "sha256:67832909c4779c64942380cb5f044a5c6163d00831472d80e25e115de9917836", size = 451952, upload-time = "2026-08-07T02:12:38.512Z" }, + { url = "https://files.pythonhosted.org/packages/cb/e1/468ad54333e92ccb62627e62cb88e5fc14a2171daa67ed47b1b8542d5b86/tornado-6.5.8-cp39-abi3-win_amd64.whl", hash = "sha256:11881db6b7c168494be2c2d12e65931451bdf7ee718535418ae1d8855dd5a0ee", size = 452391, upload-time = "2026-08-07T02:12:39.971Z" }, + { url = "https://files.pythonhosted.org/packages/ad/3e/cd5e4f06e34cde33b8ef66cf36aa2b5ad46354cc1af7d2136bbe365fee1d/tornado-6.5.8-cp39-abi3-win_arm64.whl", hash = "sha256:68a7468c7e289f8514d7d664101753903217eff1bb6822c6b5994a0b5f5bcb26", size = 451411, upload-time = "2026-08-07T02:12:41.469Z" }, ] [[package]] From 8a6eee4d49a274ecceed4f5b536fe4e1d9c02af9 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 05:01:17 +0200 Subject: [PATCH 23/24] style: ruff format the merged OpenAPI compliance test file Whitespace-only normalization (blank-line collapse and line wrapping) required by the ruff format gate after merging upstream 424bfd8758; no semantic or runtime change. Verified: 8 passed/13 skipped identical to pre-format, ruff-tests and py_compile clean. --- .../interactions/test_openapi_compliance.py | 35 +++++++------------ 1 file changed, 12 insertions(+), 23 deletions(-) diff --git a/tests/unit/interactions/test_openapi_compliance.py b/tests/unit/interactions/test_openapi_compliance.py index e97cd617d19..b373fd95b28 100644 --- a/tests/unit/interactions/test_openapi_compliance.py +++ b/tests/unit/interactions/test_openapi_compliance.py @@ -33,13 +33,10 @@ def _load_openapi_spec_dict() -> Dict[str, Any]: return response.json() except Exception as e: # pragma: no cover - defensive, env-dependent pytest.skip( - f"Skipping Google Interactions OpenAPI compliance tests - " - f"unable to load spec from {OPENAPI_SPEC_URL}: {e}" + f"Skipping Google Interactions OpenAPI compliance tests - unable to load spec from {OPENAPI_SPEC_URL}: {e}" ) - - def _declared_type_value(variant_schema: Dict[str, Any]) -> Any: """The single `type` value a union variant pins, whether spelled as a const or a 1-item enum.""" type_property = variant_schema.get("properties", {}).get("type", {}) @@ -175,22 +172,18 @@ class TestRequestCompliance: discriminator = content_schema.get("discriminator") if discriminator is not None: - assert ( - discriminator.get("propertyName") == "type" - ), f"Content is discriminated on {discriminator.get('propertyName')!r}, not 'type'" + assert discriminator.get("propertyName") == "type", ( + f"Content is discriminated on {discriminator.get('propertyName')!r}, not 'type'" + ) variant_names = [ - option["$ref"].split("/")[-1] - for option in content_schema.get("oneOf", []) - if "$ref" in option + option["$ref"].split("/")[-1] for option in content_schema.get("oneOf", []) if "$ref" in option ] assert variant_names, f"Content is not a union of named variants: {content_schema}" mapping = (discriminator or {}).get("mapping") or {} type_values = { - variant: mapping_value - for mapping_value, ref in mapping.items() - for variant in [ref.split("/")[-1]] + variant: mapping_value for mapping_value, ref in mapping.items() for variant in [ref.split("/")[-1]] } or { variant: _declared_type_value(spec_dict["components"]["schemas"].get(variant, {})) for variant in variant_names @@ -241,7 +234,9 @@ class TestRequestCompliance: for option in spec_dict["components"]["schemas"]["Step"]["oneOf"] if "$ref" in option } - assert {"UserInputStep", "ModelOutputStep"} <= step_variants, f"Step union is missing role steps: {step_variants}" + assert {"UserInputStep", "ModelOutputStep"} <= step_variants, ( + f"Step union is missing role steps: {step_variants}" + ) for step_name, type_value in [("UserInputStep", "user_input"), ("ModelOutputStep", "model_output")]: step_schema = spec_dict["components"]["schemas"][step_name] @@ -311,9 +306,7 @@ class TestResponseCompliance: expected_fields = ["total_input_tokens", "total_output_tokens", "total_tokens"] for field in expected_fields: - assert ( - field in usage_schema["properties"] - ), f"Usage field '{field}' not in spec" + assert field in usage_schema["properties"], f"Usage field '{field}' not in spec" print(f"✓ Usage field '{field}' exists") @@ -332,9 +325,7 @@ class TestToolsCompliance: """Verify FunctionDeclaration schema for function tools.""" if "FunctionDeclaration" in spec_dict["components"]["schemas"]: func_schema = spec_dict["components"]["schemas"]["FunctionDeclaration"] - assert "name" in func_schema.get( - "properties", {} - ) or "name" in func_schema.get("required", []) + assert "name" in func_schema.get("properties", {}) or "name" in func_schema.get("required", []) print("✓ FunctionDeclaration schema found") else: print("⚠ FunctionDeclaration schema not found (may be nested)") @@ -449,6 +440,4 @@ if __name__ == "__main__": if method in ["get", "post", "delete", "put", "patch"]: print(f" {method.upper()} {path}") - print( - f"\nSchemas: {list(spec.get('components', {}).get('schemas', {}).keys())[:10]}..." - ) + print(f"\nSchemas: {list(spec.get('components', {}).get('schemas', {}).keys())[:10]}...") From 9c32039bd7b2897502a731e58c20fe35e6aec9ec Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 06:25:37 +0200 Subject: [PATCH 24/24] test(traces): align native tests with named read queries --- tests/test_litellm_rust/test_traces.py | 39 +++++++++++++++++++++----- 1 file changed, 32 insertions(+), 7 deletions(-) diff --git a/tests/test_litellm_rust/test_traces.py b/tests/test_litellm_rust/test_traces.py index 447ca2ce4bb..5f628a96edf 100644 --- a/tests/test_litellm_rust/test_traces.py +++ b/tests/test_litellm_rust/test_traces.py @@ -1,6 +1,7 @@ import base64 import gzip import json +import time from typing import Final from urllib.parse import parse_qs, urlsplit @@ -14,16 +15,24 @@ pytestmark = pytest.mark.requires_rust_extension @pytest.mark.asyncio async def test_trace_reader_projects_connection_and_parameters(recording_server: RecordingServer) -> None: - recording_server.enqueue(ResponseSpec(body={"data": [{"trace_id": "trace-1"}]})) + recording_server.enqueue(ResponseSpec(body={"data": [{"span_id": "span-1"}]})) reader_url: Final = recording_server.base_url.replace("http://", "http://reader:p%40ss%2Fword%25@") storage: Final = NativeTraceStorage("trace_test", recording_server.base_url, reader_url + "?database=wrong") - rows: Final = json.loads(await storage.query("SELECT {trace_id:String} AS trace_id", {"trace_id": "trace-1"})) + rows: Final = json.loads( + await storage.query( + "trace_spans", {"trace_id": "trace-1", "team_ids": [], "api_key_hash": "", "trace_ref": ""} + ) + ) request: Final = recording_server.requests[0] - parameters: Final = parse_qs(urlsplit(request.path).query) - assert rows == [{"trace_id": "trace-1"}] - assert request.raw_body == b"SELECT {trace_id:String} AS trace_id" + parameters: Final = parse_qs(urlsplit(request.path).query, keep_blank_values=True) + assert rows == {"data": [{"span_id": "span-1"}]} + assert b"FROM otel_traces AS o" in request.raw_body + assert b"WHERE o.TraceId = {trace_id:String}" in request.raw_body assert parameters["database"] == ["trace_test"] assert parameters["param_trace_id"] == ["trace-1"] + assert parameters["param_team_ids"] == ["[]"] + assert parameters["param_api_key_hash"] == [""] + assert parameters["param_trace_ref"] == [""] assert parameters["readonly"] == ["1"] assert "user" not in parameters assert "password" not in parameters @@ -35,6 +44,16 @@ async def test_trace_reader_rejects_success_status_with_embedded_error(recording recording_server.enqueue(ResponseSpec(body={"data": [], "exception": "query failed"})) storage: Final = NativeTraceStorage("trace_test", recording_server.base_url, recording_server.base_url) with pytest.raises(RuntimeError, match="invalid or failed JSON"): + await storage.query( + "trace_spans", {"trace_id": "trace-1", "team_ids": [], "api_key_hash": "", "trace_ref": ""} + ) + + +@pytest.mark.asyncio +async def test_trace_reader_rejects_arbitrary_sql_before_sending(recording_server: RecordingServer) -> None: + recording_server.expected_requests = 0 + storage: Final = NativeTraceStorage("trace_test", recording_server.base_url, recording_server.base_url) + with pytest.raises(ValueError, match="unknown ClickHouse read query"): await storage.query("SELECT 1", {}) @@ -73,9 +92,15 @@ async def test_schema_setup_uses_writer_credentials_and_rejects_failed_statement async def test_insert_encodes_and_sends_rows(recording_server: RecordingServer) -> None: recording_server.enqueue(ResponseSpec(body="")) storage: Final = NativeTraceStorage("trace_test", recording_server.base_url) - await storage.insert_rows("otel_traces", [{"Timestamp": 1_234_567_890, "Input": "hello"}]) + before_insert_ms: Final = time.time_ns() // 1_000_000 + await storage.insert_rows("otel_traces", [{"Timestamp": 1_234_567_890, "Input": "hello", "EngineReceivedMs": 0}]) + after_insert_ms: Final = time.time_ns() // 1_000_000 request: Final = recording_server.requests[0] - assert json.loads(gzip.decompress(request.raw_body)) == { + row: Final = json.loads(gzip.decompress(request.raw_body)) + assert type(row["EngineReceivedMs"]) is int + assert before_insert_ms <= row["EngineReceivedMs"] <= after_insert_ms + assert row == { + "EngineReceivedMs": row["EngineReceivedMs"], "Input": "hello", "Timestamp": "1970-01-01T00:00:01.23456789Z", }