From 14c97ba8db1cb4172e3276ba191c962be8a1cc11 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 20 Jul 2026 13:48:24 -0700 Subject: [PATCH 01/22] fix(proxy): make /cursor/chat/completions work with Cursor agent mode - delegate messages-shaped bodies to the standard chat completions handler - strip chat-only stream_options before the Responses pipeline - fix cursor_data_generator signature (request kwarg) and duck-type the stream gate so router-wrapped streams convert instead of leaking raw Responses events - convert custom_tool_call items and events to chat tool_calls in the streaming and non-streaming paths; remap streamed tool_call indices to 0-based sequential; accumulate raw and pydantic tool calls into one choice - normalize generic pydantic output items through the raw-dict handler --- .../transformation.py | 181 ++++++++----- .../proxy/response_api_endpoints/endpoints.py | 49 ++-- litellm/types/llms/openai.py | 4 + ...responses_transformation_transformation.py | 253 ++++++++++++++++++ .../response_api_endpoints/test_endpoints.py | 147 ++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 10 +- 6 files changed, 555 insertions(+), 89 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 89a44fcdeef..d1df75bde36 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -100,6 +100,32 @@ def _build_reasoning_item( } +def _tool_call_dict_from_output_item(item: dict[str, Any]) -> dict[str, Any]: + """Convert a ``function_call`` or ``custom_tool_call`` output item dict to a chat + completions tool_call dict. Custom (grammar/freeform) tool calls carry their raw + string payload in ``input`` rather than ``arguments``; both map to + ``function.arguments`` so chat clients (e.g. Cursor agent mode) receive them like + any other tool call. The single conversion rule shared by the non-streaming + accumulator and the streaming ``output_item.added`` branch.""" + from litellm.responses.litellm_completion_transformation.transformation import ( + LiteLLMCompletionResponsesConfig, + ) + + is_custom = item.get("type") == "custom_tool_call" + arguments = (item.get("input") if is_custom else item.get("arguments")) or "" + name = item.get("name") or ("custom_tool" if is_custom else "") + tool_call_dict: dict[str, Any] = { + "id": LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item(item.get("id"), item.get("call_id")), + "function": {"name": name, "arguments": arguments}, + "type": "function", + } + provider_specific_fields = item.get("provider_specific_fields") + if isinstance(provider_specific_fields, dict) and provider_specific_fields: + tool_call_dict["provider_specific_fields"] = provider_specific_fields + tool_call_dict["function"]["provider_specific_fields"] = provider_specific_fields + return tool_call_dict + + def _reasoning_item_to_response_input( r_item: Union[ChatCompletionReasoningItem, Dict[str, Any]], ) -> Dict[str, Any]: @@ -176,36 +202,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): choice = Choices(message=msg, finish_reason="stop", index=index) return choice, index + 1 - # Handle function_call items (e.g., from GPT-5 Codex format) - if item_type == "function_call": - # Extract provider_specific_fields if present and pass through as-is - provider_specific_fields = item.get("provider_specific_fields") - if provider_specific_fields and not isinstance(provider_specific_fields, dict): - provider_specific_fields = ( - dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {} - ) - - tool_call_dict = { - "id": item.get("call_id") or item.get("id", ""), - "function": { - "name": item.get("name", ""), - "arguments": item.get("arguments", ""), - }, - "type": "function", - } - - # Pass through provider_specific_fields as-is if present - if provider_specific_fields: - tool_call_dict["provider_specific_fields"] = provider_specific_fields - # Also add to function's provider_specific_fields for consistency - tool_call_dict["function"]["provider_specific_fields"] = provider_specific_fields - - msg = Message( - content=None, - tool_calls=[tool_call_dict], - ) - choice = Choices(message=msg, finish_reason="tool_calls", index=index) - return choice, index + 1 + # function_call / custom_tool_call dicts are intercepted and accumulated by + # _convert_response_output_to_choices before this callback is reached # Unknown or unsupported type return None, index @@ -562,11 +560,21 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): accumulated_tool_calls.append(tool_call_dict) tool_call_index += 1 - elif isinstance(item, dict) and handle_raw_dict_callback is not None: - # Handle raw dict responses (e.g., from GPT-5 Codex) - choice, index = handle_raw_dict_callback(item=item, index=index) - if choice is not None: - choices.append(choice) + elif isinstance(item, (dict, BaseModel)): + # Raw dict items (e.g., from GPT-5 Codex) and pydantic items matching no + # openai SDK class above: typed ResponseCustomToolCall and litellm's own + # GenericResponseOutputItem from the completion bridge both land here + raw_item = item if isinstance(item, dict) else item.model_dump() + if raw_item.get("type") in ("function_call", "custom_tool_call"): + # Tool calls accumulate into the single trailing tool_calls choice + # like the typed branches above; a choice per call would hide every + # call after choices[0] from chat clients + accumulated_tool_calls.append(_tool_call_dict_from_output_item(raw_item)) + tool_call_index += 1 + elif handle_raw_dict_callback is not None: + choice, index = handle_raw_dict_callback(item=raw_item, index=index) + if choice is not None: + choices.append(choice) else: pass # don't fail request if item in list is not supported @@ -1078,6 +1086,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): def __init__(self, streaming_response, sync_stream: bool, json_mode: Optional[bool] = False): super().__init__(streaming_response, sync_stream, json_mode) self._chat_completion_id: str | None = None + self._tool_call_index_map: dict[int, int] = {} def _handle_string_chunk( self, str_line: Union[str, "BaseModel"] @@ -1096,15 +1105,35 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): return self.chunk_parser(json.loads(str_line)) + @staticmethod + def _sequential_tool_call_index( + tool_call_index_map: dict[int, int] | None, + output_index: int, + ) -> int: + """Chat-completions tool_call indices must be 0-based and sequential, but + Responses API ``output_index`` counts every output item (reasoning, + message, ...), so the first tool call of a reasoning model arrives at + output_index >= 1 and strict SSE accumulators (e.g. Cursor agent mode) + misplace it. When a per-stream map is provided, remap each distinct + output_index to the next sequential slot; without a map (stateless + callers), fall back to the raw output_index.""" + if tool_call_index_map is None: + return output_index + if output_index not in tool_call_index_map: + tool_call_index_map[output_index] = len(tool_call_index_map) # mutable-ok: per-stream accumulator state + return tool_call_index_map[output_index] + @staticmethod def translate_responses_chunk_to_openai_stream( parsed_chunk: Union[dict, BaseModel], + tool_call_index_map: dict[int, int] | None = None, ) -> "ModelResponseStream": """ Translate a Responses API streaming chunk to OpenAI chat completion streaming format. Args: parsed_chunk: Dict containing the Responses API event chunk + tool_call_index_map: Per-stream output_index -> sequential tool_call index map Returns: ModelResponseStream: OpenAI-formatted streaming chunk @@ -1165,7 +1194,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): function_chunk = ChatCompletionToolCallFunctionChunk( name=output_item.get("name", None), - arguments=parsed_chunk.get("arguments", ""), + arguments=output_item.get("arguments") or parsed_chunk.get("arguments") or "", ) if provider_specific_fields: @@ -1175,7 +1204,9 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): LiteLLMCompletionResponsesConfig, ) - tool_call_index = parsed_chunk.get("output_index", 0) + tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( + tool_call_index_map, parsed_chunk.get("output_index", 0) + ) tool_call_chunk = ChatCompletionToolCallChunk( id=LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item( output_item.get("id"), output_item.get("call_id") @@ -1198,10 +1229,41 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): ) ] ) - elif event_type == "response.function_call_arguments.delta": + if output_item.get("type") == "custom_tool_call": + tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( + tool_call_index_map, parsed_chunk.get("output_index", 0) + ) + converted = _tool_call_dict_from_output_item(output_item) + return ModelResponseStream( + choices=[ + StreamingChoices( + index=0, + delta=Delta( + tool_calls=[ + ChatCompletionToolCallChunk( + id=converted["id"], + index=tool_call_index, + type="function", + function=ChatCompletionToolCallFunctionChunk( + name=converted["function"]["name"], + arguments=converted["function"]["arguments"], + ), + ) + ] + ), + finish_reason=None, + ) + ] + ) + elif event_type in ( + ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA, + ResponsesAPIStreamEvents.CUSTOM_TOOL_CALL_INPUT_DELTA, + ): content_part: Optional[str] = parsed_chunk.get("delta", None) if content_part: - tool_call_index = parsed_chunk.get("output_index", 0) + tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( + tool_call_index_map, parsed_chunk.get("output_index", 0) + ) return ModelResponseStream( choices=[ StreamingChoices( @@ -1225,39 +1287,12 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): elif event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE: # New output item added output_item = parsed_chunk.get("item", {}) - if output_item.get("type") == "function_call": - # Extract provider_specific_fields if present - provider_specific_fields = output_item.get("provider_specific_fields") - if provider_specific_fields and not isinstance(provider_specific_fields, dict): - provider_specific_fields = ( - dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {} - ) - - function_chunk = ChatCompletionToolCallFunctionChunk( - name=output_item.get("name", None), - arguments="", # responses API sends everything again, we don't - ) - - # Add provider_specific_fields to function if present - if provider_specific_fields: - function_chunk["provider_specific_fields"] = provider_specific_fields - - tool_call_index = parsed_chunk.get("output_index", 0) - tool_call_chunk = ChatCompletionToolCallChunk( - id=output_item.get("call_id"), - index=tool_call_index, - type="function", - function=function_chunk, - ) - - # Add provider_specific_fields if present - if provider_specific_fields: - tool_call_chunk.provider_specific_fields = provider_specific_fields # type: ignore - + if output_item.get("type") in ("function_call", "custom_tool_call"): # Do NOT emit finish_reason here — response.completed handles the terminal # finish_reason. Emitting "tool_calls" here would prematurely terminate # the stream before subsequent tool calls arrive (same fix as #17246 for - # the message-type branch). + # the message-type branch). The item's fields were already streamed via + # output_item.added and the argument delta events. return ModelResponseStream( choices=[ StreamingChoices( @@ -1316,7 +1351,9 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): output_items = response_data.get("output", []) if response_data else [] has_function_calls = any( - item.get("type") == "function_call" for item in output_items if isinstance(item, dict) + item.get("type") in ("function_call", "custom_tool_call") + for item in output_items + if isinstance(item, dict) ) finish_reason = "tool_calls" if has_function_calls else "stop" @@ -1386,7 +1423,9 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): """ verbose_logger.debug(f"Chat provider: transform_streaming_response called with chunk: {chunk}") return self._with_stream_scoped_id( - OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk) + OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( + chunk, tool_call_index_map=self._tool_call_index_map + ) ) def _with_stream_scoped_id(self, chunk: "ModelResponseStream") -> "ModelResponseStream": diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 05c36406f36..dcd7ba18e7f 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -294,11 +294,15 @@ async def cursor_chat_completions( user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), ): """ - Cursor-specific endpoint that accepts Responses API input format but returns chat completions format. - - This endpoint handles requests from Cursor IDE which sends Responses API format (`input` field) - but expects chat completions format response (`choices`, `messages`, etc.). - + Cursor BYOK endpoint. Accepts both request shapes Cursor sends to its OpenAI-compatible + base URL and always answers in chat completions format. + + Cursor agent mode sends Responses API format bodies (`input`, flat tool defs, `reasoning`, + custom tools) to the chat/completions path while expecting chat completions responses; + those are routed through the Responses API pipeline and converted back. Genuine chat + completions bodies (`messages` present) are routed through the standard chat completions + pipeline untouched. + ```bash curl -X POST http://localhost:4000/cursor/chat/completions \ -H "Content-Type: application/json" \ @@ -317,6 +321,7 @@ async def cursor_chat_completions( from litellm.proxy.proxy_server import ( _read_request_body, async_data_generator, + chat_completion, general_settings, llm_router, proxy_config, @@ -328,20 +333,28 @@ async def cursor_chat_completions( user_temperature, version, ) - from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator from litellm.types.llms.openai import ResponsesAPIResponse from litellm.types.utils import ModelResponse data = await _read_request_body(request=request) - # Convert 'messages' to 'input' for Responses API compatibility - # Cursor sends 'messages' but Responses API expects 'input' - if "messages" in data and "input" not in data: - data["input"] = data.pop("messages") + if "messages" in data: + # Genuine chat completions body (Cursor sends these for models whose BYOK it + # already fixed); delegate so behavior matches /chat/completions exactly + return await chat_completion( + request=request, + fastapi_response=fastapi_response, + model=None, + user_api_key_dict=user_api_key_dict, + ) + + # OpenAI's Responses API rejects chat-completions-only stream_options + # (Cursor sends include_usage); usage arrives via response.completed anyway + data.pop("stream_options", None) processor = ProxyBaseLLMRequestProcessing(data=data) - def cursor_data_generator(response, user_api_key_dict, request_data): + def cursor_data_generator(response, user_api_key_dict, request_data, request=None): """ Custom generator that transforms Responses API streaming chunks to chat completion chunks. @@ -349,17 +362,21 @@ async def cursor_chat_completions( to chat completion format that Cursor IDE expects. Args: - response: The streaming response (BaseResponsesAPIStreamingIterator or other) + response: The streaming Responses API event iterator (router-wrapped or not) user_api_key_dict: User API key authentication dict request_data: Request data containing model, logging_obj, etc. + request: The originating FastAPI request, forwarded for disconnect handling Returns: Async generator that yields SSE-formatted chat completion chunks """ - # If response is a BaseResponsesAPIStreamingIterator, transform it first - if isinstance(response, BaseResponsesAPIStreamingIterator): + # Any async-iterable here is a Responses API event stream needing conversion. + # Class-identity checks miss router-wrapped streams (e.g. + # HiddenParamsAsyncIteratorWrapper around LiteLLMCompletionStreamingIterator), + # which previously leaked raw Responses events to the client. + if hasattr(response, "__anext__"): # Transform Responses API iterator to chat completion iterator - # Cast to AsyncIterator[str] since BaseResponsesAPIStreamingIterator implements __aiter__/__anext__ + # Cast to AsyncIterator[str] since the stream implements __aiter__/__anext__ completion_stream = responses_api_bridge.transformation_handler.get_model_response_iterator( streaming_response=cast(AsyncIterator[str], response), sync_stream=False, @@ -378,12 +395,14 @@ async def cursor_chat_completions( response=streamwrapper, user_api_key_dict=user_api_key_dict, request_data=request_data, + request=request, ) # Otherwise, use the default generator return async_data_generator( response=response, user_api_key_dict=user_api_key_dict, request_data=request_data, + request=request, ) try: diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 314bb653196..0d064006412 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -1405,6 +1405,10 @@ class ResponsesAPIStreamEvents(str, Enum): FUNCTION_CALL_ARGUMENTS_DELTA = "response.function_call_arguments.delta" FUNCTION_CALL_ARGUMENTS_DONE = "response.function_call_arguments.done" + # Custom tool call events (grammar/freeform tools, e.g. Cursor agent tools) + CUSTOM_TOOL_CALL_INPUT_DELTA = "response.custom_tool_call_input.delta" + CUSTOM_TOOL_CALL_INPUT_DONE = "response.custom_tool_call_input.done" + # File search events FILE_SEARCH_CALL_IN_PROGRESS = "response.file_search_call.in_progress" FILE_SEARCH_CALL_SEARCHING = "response.file_search_call.searching" diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index a111b932f2c..36c32d4b3a0 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -2962,3 +2962,256 @@ async def test_acompletion_bridge_normalizes_stream_options_on_the_wire( assert "stream_options" not in request_body else: assert request_body["stream_options"] == expected_wire_stream_options + + +def test_chunk_parser_custom_tool_call_stream_sequence(): + """Cursor agent mode drives grammar/freeform ``custom_tool_call`` items (e.g. its + ApplyPatch tool). The stream converter must surface them as chat-completions + tool_call deltas: the added event opens the call (id from ``call_id``, name, empty + arguments), each ``custom_tool_call_input.delta`` streams arguments, the done event + must NOT finish the stream, and ``response.completed`` must report + finish_reason="tool_calls". Before the fix every one of these events fell through + to an empty-content chunk and the completed event said "stop", so Cursor never saw + the tool call and agent mode stalled.""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) + + added = iterator.chunk_parser( + { + "type": "response.output_item.added", + "output_index": 1, + "item": { + "type": "custom_tool_call", + "id": "ctc_1", + "call_id": "call_patch1", + "name": "ApplyPatch", + "input": "", + }, + } + ) + tool_call = added.choices[0].delta.tool_calls[0] + assert tool_call.id == "call_patch1" + assert tool_call.type == "function" + assert tool_call.function.name == "ApplyPatch" + assert tool_call.function.arguments == "" + assert tool_call.index == 0 + assert added.choices[0].finish_reason is None + + delta = iterator.chunk_parser( + { + "type": "response.custom_tool_call_input.delta", + "output_index": 1, + "delta": "*** Begin Patch", + } + ) + delta_tool_call = delta.choices[0].delta.tool_calls[0] + assert delta_tool_call.function.arguments == "*** Begin Patch" + assert delta_tool_call.index == 0 + assert delta.choices[0].finish_reason is None + + done = iterator.chunk_parser( + { + "type": "response.output_item.done", + "output_index": 1, + "item": { + "type": "custom_tool_call", + "call_id": "call_patch1", + "name": "ApplyPatch", + "input": "*** Begin Patch", + }, + } + ) + assert done.choices[0].finish_reason is None + + completed = iterator.chunk_parser( + { + "type": "response.completed", + "response": { + "output": [ + {"type": "reasoning", "id": "rs_1"}, + {"type": "custom_tool_call", "call_id": "call_patch1"}, + ], + "usage": {"input_tokens": 7, "output_tokens": 3, "total_tokens": 10}, + }, + } + ) + assert completed.choices[0].finish_reason == "tool_calls" + assert completed.usage is not None + assert completed.usage.total_tokens == 10 + + +def test_chunk_parser_remaps_tool_call_indices_sequentially(): + """Responses API output_index counts every output item, so a reasoning model's + first tool call arrives at output_index >= 1. Chat-completions clients accumulate + streamed tool_calls by index and expect the first call at 0; Cursor agent mode + misplaces calls when indices start above 0 (the community BYOK bridge assigns its + own sequential indices for the same reason). The iterator must remap each distinct + output_index to the next sequential slot and route argument deltas to the mapped + slot.""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) + + first = iterator.chunk_parser( + { + "type": "response.output_item.added", + "output_index": 2, + "item": { + "type": "function_call", + "id": "fc_1", + "call_id": "call_read1", + "name": "read_file", + "arguments": "", + }, + } + ) + assert first.choices[0].delta.tool_calls[0].index == 0 + + first_args = iterator.chunk_parser( + { + "type": "response.function_call_arguments.delta", + "output_index": 2, + "delta": '{"path":', + } + ) + assert first_args.choices[0].delta.tool_calls[0].index == 0 + + second = iterator.chunk_parser( + { + "type": "response.output_item.added", + "output_index": 4, + "item": { + "type": "function_call", + "id": "fc_2", + "call_id": "call_grep1", + "name": "grep", + "arguments": "", + }, + } + ) + assert second.choices[0].delta.tool_calls[0].index == 1 + + second_args = iterator.chunk_parser( + { + "type": "response.function_call_arguments.delta", + "output_index": 4, + "delta": '{"pattern":', + } + ) + assert second_args.choices[0].delta.tool_calls[0].index == 1 + + +def test_convert_response_output_custom_tool_call_to_tool_calls_choice(): + """Non-streaming twin of the custom_tool_call fix: a typed ResponseCustomToolCall + output item must become a chat tool_call (arguments = the raw custom input string, + id = call_id) in a finish_reason="tool_calls" choice instead of being silently + dropped, which left Cursor agent mode with an empty assistant message.""" + from openai.types.responses import ResponseCustomToolCall + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + item = ResponseCustomToolCall( + type="custom_tool_call", + id="ctc_9", + call_id="call_custom9", + name="ApplyPatch", + input="*** Begin Patch\n*** End Patch", + ) + + choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices([item]) + + assert len(choices) == 1 + choice = choices[0] + assert choice.finish_reason == "tool_calls" + tool_call = choice.message.tool_calls[0] + assert tool_call.id == "call_custom9" + assert tool_call.function.name == "ApplyPatch" + assert tool_call.function.arguments == "*** Begin Patch\n*** End Patch" + + +def test_convert_response_output_accumulates_raw_tool_calls_into_one_choice(): + """Raw dict and generic-pydantic tool-call items must accumulate into the single + trailing tool_calls choice exactly like typed items. Emitting one choice per tool + call (the old raw-dict behavior) hid every call after choices[0] from chat + clients, which read only the first choice; a multi-tool agent turn through the + completion bridge lost all but one call.""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + items = [ + { + "type": "function_call", + "id": "fc_1", + "call_id": "call_read42", + "name": "read_file", + "arguments": '{"path": "a.py"}', + }, + { + "type": "custom_tool_call", + "id": "ctc_1", + "call_id": "call_patch42", + "name": "ApplyPatch", + "input": "*** Begin Patch", + }, + ] + + choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices( + items, + handle_raw_dict_callback=handler._handle_raw_dict_response_item, + ) + + assert len(choices) == 1 + choice = choices[0] + assert choice.finish_reason == "tool_calls" + tool_calls = choice.message.tool_calls + assert len(tool_calls) == 2 + assert tool_calls[0].id == "call_read42" + assert tool_calls[0].function.name == "read_file" + assert tool_calls[0].function.arguments == '{"path": "a.py"}' + assert tool_calls[1].id == "call_patch42" + assert tool_calls[1].function.name == "ApplyPatch" + assert tool_calls[1].function.arguments == "*** Begin Patch" + + +def test_convert_response_output_generic_pydantic_message_item(): + """litellm's completion bridge (used for non-Responses-native providers behind the + router) emits GenericResponseOutputItem pydantic models rather than openai SDK + classes. The converter must normalize unrecognized pydantic items through the + raw-dict handler instead of dropping them; dropping them made transform_response + raise 'Unknown items in responses API response' on an otherwise-successful + completion (hit live via /cursor/chat/completions multi-turn tool round trips).""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.responses.main import GenericResponseOutputItem, OutputText + + handler = LiteLLMResponsesTransformationHandler() + item = GenericResponseOutputItem( + type="message", + id="msg_generic1", + status="completed", + role="assistant", + content=[OutputText(type="output_text", text="42", annotations=[])], + ) + + choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices( + [item], + handle_raw_dict_callback=handler._handle_raw_dict_response_item, + ) + + assert len(choices) == 1 + assert choices[0].message.content == "42" + assert choices[0].finish_reason == "stop" diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 07d1a9d14f9..9af1ef4766d 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -8,6 +8,7 @@ from unittest.mock import AsyncMock, MagicMock, patch import pytest from fastapi.testclient import TestClient +import litellm from litellm.proxy.proxy_server import app @@ -711,3 +712,149 @@ class TestManagedResponsesSameProvider: call_kwargs: dict = {} handler._inject_credentials(call_kwargs, model="vertex_ai/gemini-2.0-flash") assert "custom_llm_provider" not in call_kwargs + + +def _auth_override(): + from litellm.proxy._types import UserAPIKeyAuth + + return UserAPIKeyAuth(api_key="sk-test-cursor", user_id="cursor-user") + + +def test_cursor_chat_completions_messages_body_uses_chat_pipeline(): + """A genuine chat-completions body (``messages`` present; what Cursor sends for + models whose BYOK it already fixed) must run through the standard chat pipeline + untouched: multi-turn tool history (assistant tool_calls + role="tool" results) + and nested chat-format tool defs are valid there, while blindly renaming + ``messages`` to ``input`` (the pre-fix behavior) produced items the Responses API + rejects. Asserts acompletion is called with the exact messages and aresponses is + never touched.""" + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + + import litellm.proxy.proxy_server as ps + + messages = [ + {"role": "user", "content": "read a file"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_hist1", + "type": "function", + "function": {"name": "read_file", "arguments": '{"path": "a.py"}'}, + } + ], + }, + {"role": "tool", "tool_call_id": "call_hist1", "content": "file contents"}, + {"role": "user", "content": "now summarize"}, + ] + + mock_router = MagicMock() + mock_router.acompletion = AsyncMock( + return_value=litellm.ModelResponse( + id="chatcmpl-cursor-1", + choices=[ + { + "index": 0, + "message": {"role": "assistant", "content": "summary"}, + "finish_reason": "stop", + } + ], + model="gpt-4o", + ) + ) + mock_router.aresponses = AsyncMock() + mock_router.get_available_deployment = MagicMock(return_value=None) + + app.dependency_overrides[user_api_key_auth] = _auth_override + try: + with patch.object(ps, "llm_router", mock_router): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-4o", + "messages": messages, + "tools": [ + { + "type": "function", + "function": {"name": "read_file", "parameters": {"type": "object"}}, + } + ], + }, + headers={"Authorization": "Bearer sk-test-cursor"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200, response.text + body = response.json() + assert body["choices"][0]["message"]["content"] == "summary" + assert "output" not in body + + mock_router.acompletion.assert_called_once() + called_kwargs = mock_router.acompletion.call_args.kwargs + assert called_kwargs["messages"] == messages + assert "input" not in called_kwargs + mock_router.aresponses.assert_not_called() + + +def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_stream_options(): + """A Responses-shaped body (``input``, no ``messages``; what Cursor agent mode + sends) must run through the Responses pipeline with chat-completions output, and + ``stream_options`` (chat-completions-only; Cursor sends include_usage) must be + stripped before the Responses call since OpenAI's Responses API rejects it.""" + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.types.llms.openai import ResponsesAPIResponse + + import litellm.proxy.proxy_server as ps + + mock_router = MagicMock() + mock_router.aresponses = AsyncMock( + return_value=ResponsesAPIResponse( + id="resp_cursor_agent1", + created_at=1234567890, + model="gpt-4o", + object="response", + output=[ + ResponseOutputMessage( + id="msg_agent1", + type="message", + role="assistant", + status="completed", + content=[ + ResponseOutputText(type="output_text", text="agent reply", annotations=[]) + ], + ) + ], + ) + ) + mock_router.acompletion = AsyncMock() + + app.dependency_overrides[user_api_key_auth] = _auth_override + try: + with patch.object(ps, "llm_router", mock_router): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-4o", + "input": [{"role": "user", "content": "hello"}], + "stream_options": {"include_usage": True}, + }, + headers={"Authorization": "Bearer sk-test-cursor"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200, response.text + body = response.json() + assert body["choices"][0]["message"]["content"] == "agent reply" + assert "output" not in body + + mock_router.aresponses.assert_called_once() + called_kwargs = mock_router.aresponses.call_args.kwargs + assert "stream_options" not in called_kwargs + mock_router.acompletion.assert_not_called() diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 109d638fb9c..4e0b9b88995 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -2621,10 +2621,14 @@ export interface paths { put?: never; /** * Cursor Chat Completions - * @description Cursor-specific endpoint that accepts Responses API input format but returns chat completions format. + * @description Cursor BYOK endpoint. Accepts both request shapes Cursor sends to its OpenAI-compatible + * base URL and always answers in chat completions format. * - * This endpoint handles requests from Cursor IDE which sends Responses API format (`input` field) - * but expects chat completions format response (`choices`, `messages`, etc.). + * Cursor agent mode sends Responses API format bodies (`input`, flat tool defs, `reasoning`, + * custom tools) to the chat/completions path while expecting chat completions responses; + * those are routed through the Responses API pipeline and converted back. Genuine chat + * completions bodies (`messages` present) are routed through the standard chat completions + * pipeline untouched. * * ```bash * curl -X POST http://localhost:4000/cursor/chat/completions -H "Content-Type: application/json" -H "Authorization: Bearer sk-1234" -d '{ From b080454d1f58efa233c22e3b363ee8b72205aabe Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 20 Jul 2026 14:00:08 -0700 Subject: [PATCH 02/22] refactor(responses): clarify output_item.added tool branches as if/elif chain --- .../litellm_responses_transformation/transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index d1df75bde36..eb4acb86674 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -1229,7 +1229,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): ) ] ) - if output_item.get("type") == "custom_tool_call": + elif output_item.get("type") == "custom_tool_call": tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( tool_call_index_map, parsed_chunk.get("output_index", 0) ) From af1b7f1347e56c2f00c2a16829d75eb7c52327ea Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 20 Jul 2026 14:18:09 -0700 Subject: [PATCH 03/22] fix(proxy): strip stream_options without mutating the cached request body --- .../proxy/response_api_endpoints/endpoints.py | 7 +++-- .../response_api_endpoints/test_endpoints.py | 27 +++++++++++++++++-- 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index dcd7ba18e7f..9601e2d4fde 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -349,8 +349,11 @@ async def cursor_chat_completions( ) # OpenAI's Responses API rejects chat-completions-only stream_options - # (Cursor sends include_usage); usage arrives via response.completed anyway - data.pop("stream_options", None) + # (Cursor sends include_usage); usage arrives via response.completed anyway. + # Rebuild rather than pop: _read_request_body can return the request-scope + # cached parsed-body dict itself, and removing keys from it corrupts the + # cache's key snapshot so later readers get an empty body + data = {key: value for key, value in data.items() if key != "stream_options"} processor = ProxyBaseLLMRequestProcessing(data=data) diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 9af1ef4766d..0cc79658b6a 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -803,14 +803,30 @@ def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_s """A Responses-shaped body (``input``, no ``messages``; what Cursor agent mode sends) must run through the Responses pipeline with chat-completions output, and ``stream_options`` (chat-completions-only; Cursor sends include_usage) must be - stripped before the Responses call since OpenAI's Responses API rejects it.""" + stripped before the Responses call since OpenAI's Responses API rejects it. + Stripping must not mutate the dict _read_request_body returned: that can be the + request-scope cached parsed body itself, and removing a key from it corrupts the + cache's key snapshot so any later _read_request_body caller (spend tracking, + logging hooks) silently gets an empty body; a follow-up read must still see the + full original body.""" + import asyncio + from openai.types.responses import ResponseOutputMessage, ResponseOutputText from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.proxy.common_utils.http_parsing_utils import ( + _read_request_body as real_read_request_body, + ) from litellm.types.llms.openai import ResponsesAPIResponse import litellm.proxy.proxy_server as ps + captured_requests = [] + + async def capturing_read_request_body(request): + captured_requests.append(request) + return await real_read_request_body(request=request) + mock_router = MagicMock() mock_router.aresponses = AsyncMock( return_value=ResponsesAPIResponse( @@ -835,7 +851,9 @@ def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_s app.dependency_overrides[user_api_key_auth] = _auth_override try: - with patch.object(ps, "llm_router", mock_router): + with patch.object(ps, "llm_router", mock_router), patch.object( + ps, "_read_request_body", side_effect=capturing_read_request_body + ): client = TestClient(app) response = client.post( "/cursor/chat/completions", @@ -858,3 +876,8 @@ def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_s called_kwargs = mock_router.aresponses.call_args.kwargs assert "stream_options" not in called_kwargs mock_router.acompletion.assert_not_called() + + assert captured_requests + followup_body = asyncio.run(real_read_request_body(request=captured_requests[0])) + assert followup_body.get("stream_options") == {"include_usage": True} + assert followup_body.get("input") == [{"role": "user", "content": "hello"}] From 19c875fa82f35f1bb6c484ed7a5b1d8195d77ba3 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 20 Jul 2026 14:45:59 -0700 Subject: [PATCH 04/22] refactor(responses): route function_call added-events through the shared tool-call converter --- .../transformation.py | 57 ++++--------------- 1 file changed, 11 insertions(+), 46 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index eb4acb86674..9fd193b4eeb 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -120,7 +120,11 @@ def _tool_call_dict_from_output_item(item: dict[str, Any]) -> dict[str, Any]: "type": "function", } provider_specific_fields = item.get("provider_specific_fields") - if isinstance(provider_specific_fields, dict) and provider_specific_fields: + if provider_specific_fields and not isinstance(provider_specific_fields, dict): + provider_specific_fields = ( + dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else None + ) + if provider_specific_fields: tool_call_dict["provider_specific_fields"] = provider_specific_fields tool_call_dict["function"]["provider_specific_fields"] = provider_specific_fields return tool_call_dict @@ -1184,39 +1188,26 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): elif event_type == "response.output_item.added": # New output item added output_item = parsed_chunk.get("item", {}) - if output_item.get("type") == "function_call": - # Extract provider_specific_fields if present - provider_specific_fields = output_item.get("provider_specific_fields") - if provider_specific_fields and not isinstance(provider_specific_fields, dict): - provider_specific_fields = ( - dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {} - ) + if output_item.get("type") in ("function_call", "custom_tool_call"): + converted = _tool_call_dict_from_output_item(output_item) + provider_specific_fields = converted.get("provider_specific_fields") function_chunk = ChatCompletionToolCallFunctionChunk( - name=output_item.get("name", None), - arguments=output_item.get("arguments") or parsed_chunk.get("arguments") or "", + name=converted["function"]["name"] or None, + arguments=converted["function"]["arguments"] or parsed_chunk.get("arguments") or "", ) - if provider_specific_fields: function_chunk["provider_specific_fields"] = provider_specific_fields - from litellm.responses.litellm_completion_transformation.transformation import ( - LiteLLMCompletionResponsesConfig, - ) - tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( tool_call_index_map, parsed_chunk.get("output_index", 0) ) tool_call_chunk = ChatCompletionToolCallChunk( - id=LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item( - output_item.get("id"), output_item.get("call_id") - ), + id=converted["id"], index=tool_call_index, type="function", function=function_chunk, ) - - # Add provider_specific_fields if present if provider_specific_fields: tool_call_chunk.provider_specific_fields = provider_specific_fields # type: ignore @@ -1229,32 +1220,6 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): ) ] ) - elif output_item.get("type") == "custom_tool_call": - tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index( - tool_call_index_map, parsed_chunk.get("output_index", 0) - ) - converted = _tool_call_dict_from_output_item(output_item) - return ModelResponseStream( - choices=[ - StreamingChoices( - index=0, - delta=Delta( - tool_calls=[ - ChatCompletionToolCallChunk( - id=converted["id"], - index=tool_call_index, - type="function", - function=ChatCompletionToolCallFunctionChunk( - name=converted["function"]["name"], - arguments=converted["function"]["arguments"], - ), - ) - ] - ), - finish_reason=None, - ) - ] - ) elif event_type in ( ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA, ResponsesAPIStreamEvents.CUSTOM_TOOL_CALL_INPUT_DELTA, From 96916f29a60b005f7e75173ae3aa5c0f6adbafff Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 20 Jul 2026 16:36:57 -0700 Subject: [PATCH 05/22] feat(proxy): serve the OpenAI model list at /cursor/models for BYOK base URLs --- litellm/proxy/_types.py | 2 + .../proxy/response_api_endpoints/endpoints.py | 29 ++++++ .../response_api_endpoints/test_endpoints.py | 26 ++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 92 +++++++++++++++++++ 4 files changed, 149 insertions(+) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 23fe7af9994..f0cefb2d55b 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -287,6 +287,8 @@ class LiteLLMRoutes(enum.Enum): "/chat/completions", "/v1/chat/completions", "/cursor/chat/completions", + "/cursor/models", + "/cursor/v1/models", # completions "/engines/{model}/completions", "/openai/deployments/{model}/completions", diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 9601e2d4fde..80abdab8583 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -22,6 +22,8 @@ from litellm.types.responses.main import DeleteResponseResult router = APIRouter() +_user_api_key_auth_dep = Depends(user_api_key_auth) + @router.post( "/v1/responses", @@ -283,6 +285,33 @@ async def responses_api( ) +@router.get( + "/cursor/models", + dependencies=[Depends(user_api_key_auth)], + tags=["responses"], +) +@router.get( + "/cursor/v1/models", + dependencies=[Depends(user_api_key_auth)], + tags=["responses"], +) +async def cursor_model_list( + user_api_key_dict: UserAPIKeyAuth = _user_api_key_auth_dep, +): + """ + OpenAI-compatible model listing for the Cursor BYOK base URL. + + Clients pointed at `/cursor` as an OpenAI-compatible base URL resolve and + verify models via `GET {base}/models` (the OpenAI SDK contract). Without this + route those requests fall through to the Cursor Cloud Agents passthrough, which + demands a Cursor API key and 401s, so key verification silently fails before any + chat request is ever sent. Delegates to the standard `/v1/models` handler. + """ + from litellm.proxy.proxy_server import model_list + + return await model_list(user_api_key_dict=user_api_key_dict) + + @router.post( "/cursor/chat/completions", dependencies=[Depends(user_api_key_auth)], diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 0cc79658b6a..c41de4a8e40 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -881,3 +881,29 @@ def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_s followup_body = asyncio.run(real_read_request_body(request=captured_requests[0])) assert followup_body.get("stream_options") == {"include_usage": True} assert followup_body.get("input") == [{"role": "user", "content": "hello"}] + + +def test_cursor_models_route_delegates_to_model_list(): + """Clients pointed at /cursor as an OpenAI-compatible base URL resolve and + verify keys via GET {base}/models (the OpenAI SDK contract). Without a dedicated + route those requests fall through to the Cursor Cloud Agents passthrough and 401 + for lack of a Cursor API key, so BYOK verification fails before any chat request + is sent. Both /cursor/models and /cursor/v1/models must serve the standard model + list instead.""" + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + + import litellm.proxy.proxy_server as ps + + model_payload = {"data": [{"id": "gpt-5.6", "object": "model"}], "object": "list"} + + app.dependency_overrides[user_api_key_auth] = _auth_override + try: + with patch.object(ps, "model_list", AsyncMock(return_value=model_payload)) as mock_model_list: + client = TestClient(app) + for path in ("/cursor/models", "/cursor/v1/models"): + response = client.get(path, headers={"Authorization": "Bearer sk-test-cursor"}) + assert response.status_code == 200, f"{path}: {response.text}" + assert response.json() == model_payload + assert mock_model_list.call_count == 2 + finally: + app.dependency_overrides.pop(user_api_key_auth, None) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4e0b9b88995..1bc8839976d 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -2645,6 +2645,58 @@ export interface paths { patch?: never; trace?: never; }; + "/cursor/models": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Cursor Model List + * @description OpenAI-compatible model listing for the Cursor BYOK base URL. + * + * Clients pointed at `/cursor` as an OpenAI-compatible base URL resolve and + * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this + * route those requests fall through to the Cursor Cloud Agents passthrough, which + * demands a Cursor API key and 401s, so key verification silently fails before any + * chat request is ever sent. Delegates to the standard `/v1/models` handler. + */ + get: operations["cursor_model_list_cursor_models_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/cursor/v1/models": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Cursor Model List + * @description OpenAI-compatible model listing for the Cursor BYOK base URL. + * + * Clients pointed at `/cursor` as an OpenAI-compatible base URL resolve and + * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this + * route those requests fall through to the Cursor Cloud Agents passthrough, which + * demands a Cursor API key and 401s, so key verification silently fails before any + * chat request is ever sent. Delegates to the standard `/v1/models` handler. + */ + get: operations["cursor_model_list_cursor_v1_models_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/cursor/{endpoint}": { parameters: { query?: never; @@ -38506,6 +38558,46 @@ export interface operations { }; }; }; + cursor_model_list_cursor_models_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + }; + }; + cursor_model_list_cursor_v1_models_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + }; + }; cursor_proxy_route_cursor__endpoint__get: { parameters: { query?: never; From b45c99f6c58a1a6f903f7b4ad64a6150f04becc0 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 11:49:42 -0700 Subject: [PATCH 06/22] fix(litellm): support OpenAI chat completions custom tool calls end to end Cursor Ask mode sends chat bodies whose tools array mixes nested function tools with flat Responses-style custom tools; the /cursor messages arm now nests those before delegating, published via the request parsed-body cache. Core chat parsing gains first-class custom tool call types mirroring the openai SDK union: a single dict dispatch feeds the provider-dict sinks, Delta dispatch stops both stream re-parse sites from silently swallowing custom deltas, the chunk builder accumulates custom input for spend logs, function-assuming consumers (json-mode gate, multi_tool_use repair, helicone, lunary) skip custom entries, and the chat-to-responses bridge flattens nested custom tools to the Responses flat shape --- .../transformation.py | 12 ++ litellm/integrations/helicone.py | 7 +- litellm/integrations/lunary.py | 2 +- .../convert_dict_to_response.py | 25 ++-- .../streaming_chunk_builder_utils.py | 41 +++++- .../llms/openai/chat/gpt_transformation.py | 8 +- .../proxy/response_api_endpoints/endpoints.py | 27 +++- litellm/types/utils.py | 98 ++++++++++++-- ...responses_transformation_transformation.py | 34 +++++ ...responses_transformation_transformation.py | 1 + .../test_convert_dict_to_response.py | 104 +++++++++++++++ .../test_streaming_chunk_builder_utils.py | 35 +++++ .../test_streaming_handler.py | 99 ++++++++++++++ .../response_api_endpoints/test_endpoints.py | 123 ++++++++++++++++++ tests/test_litellm/types/test_types_utils.py | 89 +++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 29 ++++- 16 files changed, 704 insertions(+), 30 deletions(-) create mode 100644 tests/test_litellm/litellm_core_utils/llm_response_utils/test_convert_dict_to_response.py diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 9fd193b4eeb..75cee42dd55 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -20,6 +20,7 @@ from typing import ( cast, ) +from openai.types.responses.custom_tool_param import CustomToolParam from openai.types.responses.tool_param import FunctionToolParam from pydantic import BaseModel @@ -894,6 +895,17 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): description=function_tool.get("description"), ) ) + elif tool.get("type") == "custom" and isinstance(tool.get("custom"), dict): + custom_payload = tool["custom"] + flat_custom: CustomToolParam = { + "type": "custom", + "name": custom_payload.get("name", ""), + } + if custom_payload.get("description") is not None: + flat_custom["description"] = custom_payload["description"] + if custom_payload.get("format") is not None: + flat_custom["format"] = custom_payload["format"] + responses_tools.append(flat_custom) else: responses_tools.append(tool) # type: ignore diff --git a/litellm/integrations/helicone.py b/litellm/integrations/helicone.py index 21e9479491e..4c7a606c16f 100644 --- a/litellm/integrations/helicone.py +++ b/litellm/integrations/helicone.py @@ -59,12 +59,15 @@ class HeliconeLogger: content = [] if "tool_calls" in message and message["tool_calls"]: for tool_call in message["tool_calls"]: + function = tool_call.get("function") + if not function: + continue content.append( { "type": "tool_use", "id": tool_call["id"], - "name": tool_call["function"]["name"], - "input": tool_call["function"]["arguments"], + "name": function["name"], + "input": function["arguments"], } ) elif "content" in message and message["content"]: diff --git a/litellm/integrations/lunary.py b/litellm/integrations/lunary.py index aaf5751cb79..448580f0b2d 100644 --- a/litellm/integrations/lunary.py +++ b/litellm/integrations/lunary.py @@ -31,7 +31,7 @@ def parse_tool_calls(tool_calls): return serialized - return [clean_tool_call(tool_call) for tool_call in tool_calls] + return [clean_tool_call(tool_call) for tool_call in tool_calls if getattr(tool_call, "function", None) is not None] def parse_messages(input): diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 47daf33824e..c5cfdea9ffe 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -19,6 +19,7 @@ from litellm.types.llms.openai import ( ) from litellm.types.utils import ( ChatCompletionDeltaToolCall, + ChatCompletionMessageCustomToolCall, ChatCompletionMessageToolCall, ChatCompletionRedactedThinkingBlock, Choices, @@ -43,6 +44,7 @@ from litellm.types.utils import ( TranscriptionUsageDurationObject, TranscriptionUsageTokensObject, Usage, + chat_completion_tool_call_from_dict, ) from .get_headers import get_response_headers @@ -369,7 +371,7 @@ from collections import defaultdict def _handle_invalid_parallel_tool_calls( - tool_calls: List[ChatCompletionMessageToolCall], + tool_calls: List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]], ): """ Handle hallucinated parallel tool call from openai - https://community.openai.com/t/model-tries-to-call-unknown-function-multi-tool-use-parallel/490653 @@ -382,6 +384,8 @@ def _handle_invalid_parallel_tool_calls( try: replacements: Dict[int, List[ChatCompletionMessageToolCall]] = defaultdict(list) for i, tool_call in enumerate(tool_calls): + if isinstance(tool_call, ChatCompletionMessageCustomToolCall): + continue current_function = tool_call.function.name function_args = json.loads(tool_call.function.arguments) if current_function == "multi_tool_use.parallel": @@ -527,19 +531,20 @@ class LiteLLMResponseObjectHandler: def _should_convert_tool_call_to_json_mode( - tool_calls: Optional[Union[List[ChatCompletionMessageToolCall], List[DatabricksTool]]] = None, + tool_calls: Optional[ + Union[ + List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]], + List[DatabricksTool], + ] + ] = None, convert_tool_call_to_json_mode: Optional[bool] = None, ) -> bool: """ Determine if tool calls should be converted to JSON mode """ - if ( - convert_tool_call_to_json_mode - and tool_calls is not None - and len(tool_calls) == 1 - and tool_calls[0]["function"]["name"] == RESPONSE_FORMAT_TOOL_NAME - ): - return True + if convert_tool_call_to_json_mode and tool_calls is not None and len(tool_calls) == 1: + function = tool_calls[0].get("function") + return function is not None and function["name"] == RESPONSE_FORMAT_TOOL_NAME return False @@ -647,7 +652,7 @@ def convert_to_model_response_object( if tool_calls is not None: _openai_tool_calls = [] for _tc in tool_calls: - _openai_tc = ChatCompletionMessageToolCall(**_tc) + _openai_tc = chat_completion_tool_call_from_dict(_tc) _openai_tool_calls.append(_openai_tc) fixed_tool_calls = _handle_invalid_parallel_tool_calls(_openai_tool_calls) diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index d52d9849310..09bd55096e8 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -9,6 +9,8 @@ from litellm.types.llms.openai import ( from litellm.types.utils import ( CacheCreationTokenDetails, ChatCompletionAudioResponse, + ChatCompletionCustomToolCallPayload, + ChatCompletionMessageCustomToolCall, ChatCompletionMessageToolCall, Choices, CompletionTokensDetails, @@ -202,8 +204,10 @@ class ChunkProcessor: response = self.update_model_response_with_hidden_params(model_response=response, chunk=chunk) return response - def get_combined_tool_content(self, tool_call_chunks: List[Dict[str, Any]]) -> List[ChatCompletionMessageToolCall]: - tool_calls_list: List[ChatCompletionMessageToolCall] = [] + def get_combined_tool_content( + self, tool_call_chunks: List[Dict[str, Any]] + ) -> List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]]: + tool_calls_list: List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] = [] tool_call_map: Dict[int, Dict[str, Any]] = {} # Map to store tool calls by index for chunk in tool_call_chunks: @@ -219,12 +223,15 @@ class ChunkProcessor: # Check if tool_call has function (either as attribute or dict key) has_function = False + has_custom = False if isinstance(tool_call, dict): has_function = "function" in tool_call and tool_call["function"] is not None + has_custom = "custom" in tool_call and tool_call["custom"] is not None else: has_function = hasattr(tool_call, "function") and tool_call.function is not None + has_custom = getattr(tool_call, "custom", None) is not None - if not has_function: + if not has_function and not has_custom: continue # Get index (handle both dict and object) @@ -239,6 +246,8 @@ class ChunkProcessor: "name": None, "type": None, "arguments": [], + "custom_name": None, + "custom_input": [], "provider_specific_fields": None, } @@ -261,6 +270,13 @@ class ChunkProcessor: tool_call_map[index]["name"] = function.name if hasattr(function, "arguments") and function.arguments: tool_call_map[index]["arguments"].append(function.arguments) + + custom = tool_call.get("custom") + if isinstance(custom, dict): + if custom.get("name"): + tool_call_map[index]["custom_name"] = custom["name"] + if custom.get("input"): + tool_call_map[index]["custom_input"].append(custom["input"]) else: # tool_call is an object if hasattr(tool_call, "id") and tool_call.id: @@ -273,6 +289,13 @@ class ChunkProcessor: if hasattr(tool_call.function, "arguments") and tool_call.function.arguments: tool_call_map[index]["arguments"].append(tool_call.function.arguments) + custom = getattr(tool_call, "custom", None) + if custom is not None: + if getattr(custom, "name", None): + tool_call_map[index]["custom_name"] = custom.name + if getattr(custom, "input", None): + tool_call_map[index]["custom_input"].append(custom.input) + # Preserve provider_specific_fields from streaming chunks provider_fields = None if isinstance(tool_call, dict): @@ -299,7 +322,17 @@ class ChunkProcessor: # Convert the map to a list of tool calls for index in sorted(tool_call_map.keys()): tool_call_data = tool_call_map[index] - if tool_call_data["id"] and tool_call_data["name"]: + if tool_call_data["type"] == "custom" and tool_call_data["id"] and tool_call_data["custom_name"]: + tool_calls_list.append( + ChatCompletionMessageCustomToolCall( + id=tool_call_data["id"], + custom=ChatCompletionCustomToolCallPayload( + name=tool_call_data["custom_name"], + input="".join(tool_call_data["custom_input"]), + ), + ) + ) + elif tool_call_data["id"] and tool_call_data["name"]: combined_arguments = "".join(tool_call_data["arguments"]) or "{}" # Build function - provider_specific_fields should be on tool_call level, not function level diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index f2498c0a7e2..129a9b51d0d 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -50,12 +50,14 @@ from litellm.types.llms.openai import ( OpenAIMessageContentListBlock, ) from litellm.types.utils import ( + ChatCompletionMessageCustomToolCall, ChatCompletionMessageToolCall, Choices, Function, Message, ModelResponse, ModelResponseStream, + chat_completion_tool_call_from_dict, ) from litellm.utils import convert_to_model_response_object @@ -531,12 +533,14 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): for choice in choices: ## HANDLE JSON MODE - anthropic returns single function call] tool_calls = choice["message"].get("tool_calls", None) - new_tool_calls: Optional[List[ChatCompletionMessageToolCall]] = None + new_tool_calls: Optional[ + List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] + ] = None message_content = choice["message"].get("content", None) if tool_calls is not None: _openai_tool_calls = [] for _tc in tool_calls: - _openai_tc = ChatCompletionMessageToolCall(**_tc) # type: ignore + _openai_tc = chat_completion_tool_call_from_dict(dict(_tc)) _openai_tool_calls.append(_openai_tc) fixed_tool_calls = _handle_invalid_parallel_tool_calls(_openai_tool_calls) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 80abdab8583..f9b2cc79f73 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -24,6 +24,23 @@ router = APIRouter() _user_api_key_auth_dep = Depends(user_api_key_auth) +_FLAT_CUSTOM_TOOL_KEYS = ("name", "description", "format") +_FLAT_FUNCTION_TOOL_KEYS = ("name", "description", "parameters", "strict") + + +def _nest_flat_chat_tool(tool: object) -> object: + if not isinstance(tool, dict) or "name" not in tool: + return tool + if tool.get("type") == "custom" and "custom" not in tool: + return {"type": "custom", "custom": {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool}} + if tool.get("type") == "function" and "function" not in tool: + return {"type": "function", "function": {k: tool[k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool}} + return tool + + +def _nest_flat_chat_tools(tools: list) -> list: + return [_nest_flat_chat_tool(tool) for tool in tools] + @router.post( "/v1/responses", @@ -330,7 +347,9 @@ async def cursor_chat_completions( custom tools) to the chat/completions path while expecting chat completions responses; those are routed through the Responses API pipeline and converted back. Genuine chat completions bodies (`messages` present) are routed through the standard chat completions - pipeline untouched. + pipeline, after nesting any flat Responses-style tool defs Cursor mixes into the chat + `tools` array (e.g. `{"type": "custom", "name": "ApplyPatch", ...}`) into the chat + completions shape OpenAI requires (`{"type": "custom", "custom": {...}}`). ```bash curl -X POST http://localhost:4000/cursor/chat/completions \ @@ -347,6 +366,7 @@ async def cursor_chat_completions( responses_api_bridge, ) from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper + from litellm.proxy.common_utils.http_parsing_utils import _safe_set_request_parsed_body from litellm.proxy.proxy_server import ( _read_request_body, async_data_generator, @@ -370,6 +390,11 @@ async def cursor_chat_completions( if "messages" in data: # Genuine chat completions body (Cursor sends these for models whose BYOK it # already fixed); delegate so behavior matches /chat/completions exactly + tools = data.get("tools") + if isinstance(tools, list): + nested_tools = _nest_flat_chat_tools(tools) + if nested_tools != tools: + _safe_set_request_parsed_body(request=request, parsed_body={**data, "tools": nested_tools}) return await chat_completion( request=request, fastapi_response=fastapi_response, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 18991f53e6f..a1e52f7584f 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1084,6 +1084,71 @@ class ChatCompletionDeltaToolCall(OpenAIObject): setattr(self, key, value) +class ChatCompletionCustomToolCallPayload(OpenAIObject): + name: str + input: str + + def __contains__(self, key): + return hasattr(self, key) + + def get(self, key, default=None): + return getattr(self, key, default) + + def __getitem__(self, key): + return getattr(self, key) + + +class ChatCompletionDeltaCustomToolCallPayload(OpenAIObject): + name: str | None = None + input: str | None = None + + def __contains__(self, key): + return hasattr(self, key) + + def get(self, key, default=None): + return getattr(self, key, default) + + def __getitem__(self, key): + return getattr(self, key) + + +class ChatCompletionMessageCustomToolCall(OpenAIObject): + id: str + type: Literal["custom"] = "custom" + custom: ChatCompletionCustomToolCallPayload + + def __contains__(self, key): + return hasattr(self, key) + + def get(self, key, default=None): + return getattr(self, key, default) + + def __getitem__(self, key): + return getattr(self, key) + + def __setitem__(self, key, value): + setattr(self, key, value) + + +class ChatCompletionDeltaCustomToolCall(OpenAIObject): + id: str | None = None + type: str | None = None + custom: ChatCompletionDeltaCustomToolCallPayload + index: int + + def __contains__(self, key): + return hasattr(self, key) + + def get(self, key, default=None): + return getattr(self, key, default) + + def __getitem__(self, key): + return getattr(self, key) + + def __setitem__(self, key, value): + setattr(self, key, value) + + class ChatCompletionMessageToolCall(OpenAIObject): def __init__( self, @@ -1125,6 +1190,16 @@ class ChatCompletionMessageToolCall(OpenAIObject): setattr(self, key, value) +def chat_completion_tool_call_from_dict( + tool_call: dict, +) -> "ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall": + if tool_call.get("type") == "custom": + return ChatCompletionMessageCustomToolCall( + **{k: v for k, v in tool_call.items() if not (k == "function" and v is None)} + ) + return ChatCompletionMessageToolCall(**tool_call) + + from openai.types.chat.chat_completion_audio import ChatCompletionAudio @@ -1177,7 +1252,7 @@ def add_provider_specific_fields(object: BaseModel, provider_specific_fields: Op class Message(SafeAttributeModel, OpenAIObject): content: Optional[str] role: Literal["assistant", "user", "system", "tool", "function"] - tool_calls: Optional[List[ChatCompletionMessageToolCall]] + tool_calls: Optional[List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]]] function_call: Optional[FunctionCall] audio: Optional[ChatCompletionAudioResponse] = None images: Optional[List[ImageURLListItem]] = None @@ -1208,7 +1283,7 @@ class Message(SafeAttributeModel, OpenAIObject): "function_call": (FunctionCall(**function_call) if function_call is not None else None), "tool_calls": ( [ - (ChatCompletionMessageToolCall(**tool_call) if isinstance(tool_call, dict) else tool_call) + (chat_completion_tool_call_from_dict(tool_call) if isinstance(tool_call, dict) else tool_call) for tool_call in tool_calls ] if tool_calls is not None and len(tool_calls) > 0 @@ -1301,7 +1376,7 @@ class Delta(SafeAttributeModel, OpenAIObject): content: Optional[str] role: Optional[str] function_call: Optional[FunctionCall] - tool_calls: Optional[List[ChatCompletionDeltaToolCall]] + tool_calls: Optional[List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]]] audio: Optional[ChatCompletionAudioResponse] images: Optional[List[ImageURLListItem]] annotations: Optional[List[ChatCompletionAnnotation]] @@ -1339,17 +1414,24 @@ class Delta(SafeAttributeModel, OpenAIObject): function_call = FunctionCall(**function_call) if tool_calls is not None and isinstance(tool_calls, list): - coerced_tool_calls: List[ChatCompletionDeltaToolCall] = [] + coerced_tool_calls: List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]] = [] current_index = 0 for tool_call in tool_calls: if isinstance(tool_call, dict): if tool_call.get("index", None) is None: tool_call["index"] = current_index current_index += 1 - if tool_call.get("type", None) is None: - tool_call["type"] = "function" - coerced_tool_calls.append(ChatCompletionDeltaToolCall(**tool_call)) - elif isinstance(tool_call, ChatCompletionDeltaToolCall): + if tool_call.get("type") == "custom" or "custom" in tool_call: + coerced_tool_calls.append( + ChatCompletionDeltaCustomToolCall( + **{k: v for k, v in tool_call.items() if not (k == "function" and v is None)} + ) + ) + else: + if tool_call.get("type", None) is None: + tool_call["type"] = "function" + coerced_tool_calls.append(ChatCompletionDeltaToolCall(**tool_call)) + elif isinstance(tool_call, (ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall)): coerced_tool_calls.append(tool_call) tool_calls = coerced_tool_calls diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 36c32d4b3a0..767d1631649 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -3215,3 +3215,37 @@ def test_convert_response_output_generic_pydantic_message_item(): assert len(choices) == 1 assert choices[0].message.content == "42" assert choices[0].finish_reason == "stop" + + +def test_convert_tools_to_responses_format_flattens_nested_custom_tool(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + tools = [ + { + "type": "custom", + "custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}, + }, + {"type": "function", "function": {"name": "f", "parameters": {"type": "object"}}}, + ] + converted = handler._convert_tools_to_responses_format(tools) + assert converted[0] == { + "type": "custom", + "name": "ApplyPatch", + "description": "V4A patch", + "format": {"type": "text"}, + } + assert converted[1]["type"] == "function" + assert converted[1]["name"] == "f" + + +def test_convert_tools_to_responses_format_flattens_custom_tool_without_optional_keys(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + converted = handler._convert_tools_to_responses_format([{"type": "custom", "custom": {"name": "Minimal"}}]) + assert converted[0] == {"type": "custom", "name": "Minimal"} diff --git a/tests/test_litellm/completion_extras/test_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/test_litellm_responses_transformation_transformation.py index 05bdc40112c..626e554d476 100644 --- a/tests/test_litellm/completion_extras/test_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/test_litellm_responses_transformation_transformation.py @@ -257,3 +257,4 @@ def test_translate_responses_chunk_passthrough_chat_completion_chunk(): assert result.choices[0].delta.content == "Hi! How can I help?" assert result.choices[0].finish_reason is None + diff --git a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_convert_dict_to_response.py b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_convert_dict_to_response.py new file mode 100644 index 00000000000..293e5de304f --- /dev/null +++ b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_convert_dict_to_response.py @@ -0,0 +1,104 @@ +import os +import sys + +sys.path.insert(0, os.path.abspath("../../../..")) + +from litellm.constants import RESPONSE_FORMAT_TOOL_NAME +from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( + _handle_invalid_parallel_tool_calls, + _should_convert_tool_call_to_json_mode, + convert_to_model_response_object, +) +from litellm.types.utils import ( + ChatCompletionMessageCustomToolCall, + ChatCompletionMessageToolCall, + Function, + ModelResponse, +) + +OPENAI_CUSTOM_TOOL_CALL_RESPONSE = { + "id": "chatcmpl-abc", + "created": 1784657740, + "model": "gpt-5.6", + "object": "chat.completion", + "choices": [ + { + "finish_reason": "tool_calls", + "index": 0, + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_njxQ", + "type": "custom", + "custom": { + "name": "ApplyPatch", + "input": "*** Begin Patch\n*** Update File: main.py\n@@\n+def hello():\n+ print(\"Hello\")\n*** End Patch\n", + }, + } + ], + "refusal": None, + "annotations": [], + }, + } + ], + "usage": {"completion_tokens": 10, "prompt_tokens": 5, "total_tokens": 15}, +} + + +def test_convert_openai_custom_tool_call_response(): + result = convert_to_model_response_object( + response_object=OPENAI_CUSTOM_TOOL_CALL_RESPONSE, + model_response_object=ModelResponse(), + response_type="completion", + ) + tool_calls = result.choices[0].message.tool_calls + assert len(tool_calls) == 1 + assert isinstance(tool_calls[0], ChatCompletionMessageCustomToolCall) + dumped = tool_calls[0].model_dump() + assert dumped == OPENAI_CUSTOM_TOOL_CALL_RESPONSE["choices"][0]["message"]["tool_calls"][0] + assert result.choices[0].finish_reason == "tool_calls" + + +def test_should_convert_tool_call_to_json_mode_ignores_custom_tool_call(): + custom_tool_call = ChatCompletionMessageCustomToolCall( + id="call_c", + custom={"name": "ApplyPatch", "input": "patch"}, + ) + assert ( + _should_convert_tool_call_to_json_mode( + tool_calls=[custom_tool_call], + convert_tool_call_to_json_mode=True, + ) + is False + ) + + +def test_should_convert_tool_call_to_json_mode_still_matches_response_format_tool(): + response_format_call = ChatCompletionMessageToolCall( + id="call_f", + type="function", + function=Function(name=RESPONSE_FORMAT_TOOL_NAME, arguments='{"answer": 4}'), + ) + assert ( + _should_convert_tool_call_to_json_mode( + tool_calls=[response_format_call], + convert_tool_call_to_json_mode=True, + ) + is True + ) + + +def test_handle_invalid_parallel_tool_calls_skips_custom_tool_calls(): + custom_tool_call = ChatCompletionMessageCustomToolCall( + id="call_c", + custom={"name": "ApplyPatch", "input": "patch"}, + ) + function_tool_call = ChatCompletionMessageToolCall( + id="call_f", + type="function", + function=Function(name="get_weather", arguments='{"city": "SF"}'), + ) + result = _handle_invalid_parallel_tool_calls([custom_tool_call, function_tool_call]) + assert result == [custom_tool_call, function_tool_call] diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py index be8c5a05601..197adf80f03 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py @@ -992,3 +992,38 @@ def test_cost_field_in_usage_chunks(): assert usage.cost == 0.00025 assert usage.prompt_tokens == 10 assert usage.completion_tokens == 5 + + +def test_get_combined_tool_content_custom_tool_call(): + from litellm.litellm_core_utils.streaming_chunk_builder_utils import ChunkProcessor + from litellm.types.utils import ChatCompletionMessageCustomToolCall + + processor = ChunkProcessor.__new__(ChunkProcessor) + tool_call_chunks = [ + { + "choices": [ + { + "delta": { + "tool_calls": [ + { + "index": 0, + "id": "call_TBs", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": ""}, + } + ] + } + } + ] + }, + {"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** Begin Patch\n"}}]}}]}, + {"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** End Patch\n"}}]}}]}, + ] + combined = processor.get_combined_tool_content(tool_call_chunks) + assert len(combined) == 1 + assert isinstance(combined[0], ChatCompletionMessageCustomToolCall) + assert combined[0].model_dump() == { + "id": "call_TBs", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch\n*** End Patch\n"}, + } diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py index 514714136fd..48bc3709517 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py @@ -3355,3 +3355,102 @@ async def test_transport_read_error_before_finish_reason_raises(logging_obj: Log if chunk.choices and chunk.choices[0].finish_reason ] assert fabricated_finish_reasons == [] + + +def test_openai_custom_tool_call_stream_deltas_survive_conversion(logging_obj: Logging): + """ + Regression test: OpenAI chat completions custom tool calls stream as + delta.tool_calls entries with a `custom` payload and NO `function` key. + Delta() used to raise on those dicts and chunk_creator's except branch + replaced the choice with an empty Delta, silently dropping the entire + tool call from the client stream. + """ + from openai.types.chat.chat_completion_chunk import ChatCompletionChunk + + from litellm.types.utils import ChatCompletionDeltaCustomToolCall + + raw_chunks = [ + { + "id": "chatcmpl-custom", + "object": "chat.completion.chunk", + "created": 1784657671, + "model": "gpt-5.6", + "choices": [ + { + "index": 0, + "delta": { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "index": 0, + "id": "call_TBs", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": ""}, + } + ], + }, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-custom", + "object": "chat.completion.chunk", + "created": 1784657671, + "model": "gpt-5.6", + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** Begin Patch\n"}}]}, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-custom", + "object": "chat.completion.chunk", + "created": 1784657671, + "model": "gpt-5.6", + "choices": [ + { + "index": 0, + "delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** End Patch\n"}}]}, + "finish_reason": None, + } + ], + }, + { + "id": "chatcmpl-custom", + "object": "chat.completion.chunk", + "created": 1784657671, + "model": "gpt-5.6", + "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}], + }, + ] + sdk_chunks = [ChatCompletionChunk.construct(**raw) for raw in raw_chunks] + first_dumped = sdk_chunks[0].choices[0].model_dump() + assert first_dumped["delta"]["tool_calls"][0]["custom"] == {"name": "ApplyPatch", "input": ""} + + wrapper = CustomStreamWrapper( + completion_stream=iter(sdk_chunks), + model="gpt-5.6", + custom_llm_provider="openai", + logging_obj=logging_obj, + ) + + emitted = list(wrapper) + tool_call_deltas = [ + chunk.choices[0].delta.tool_calls[0] + for chunk in emitted + if chunk.choices and chunk.choices[0].delta and chunk.choices[0].delta.tool_calls + ] + assert len(tool_call_deltas) == 3 + assert isinstance(tool_call_deltas[0], ChatCompletionDeltaCustomToolCall) + assert tool_call_deltas[0].id == "call_TBs" + assert tool_call_deltas[0].type == "custom" + assert tool_call_deltas[0].custom.name == "ApplyPatch" + combined_input = "".join(tc.custom.input or "" for tc in tool_call_deltas) + assert combined_input == "*** Begin Patch\n*** End Patch\n" + finish_reasons = [chunk.choices[0].finish_reason for chunk in emitted if chunk.choices] + assert "tool_calls" in finish_reasons diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index c41de4a8e40..4ef338b35e4 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -907,3 +907,126 @@ def test_cursor_models_route_delegates_to_model_list(): assert mock_model_list.call_count == 2 finally: app.dependency_overrides.pop(user_api_key_auth, None) + + +class TestNestFlatChatTools: + def test_flat_custom_tool_is_nested(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + result = _nest_flat_chat_tools( + [{"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}] + ) + assert result == [ + { + "type": "custom", + "custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}, + } + ] + + def test_flat_function_tool_is_nested(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + result = _nest_flat_chat_tools( + [{"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}}] + ) + assert result == [ + { + "type": "function", + "function": {"name": "read_file", "description": "d", "parameters": {"type": "object"}}, + } + ] + + def test_already_nested_and_unrecognized_tools_pass_through_unchanged(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + tools = [ + {"type": "custom", "custom": {"name": "already_nested"}}, + {"type": "function", "function": {"name": "f", "parameters": {}}}, + {"type": "web_search"}, + {"type": "custom"}, + {"name": "typeless"}, + {}, + "junk", + None, + 42, + ] + assert _nest_flat_chat_tools(tools) == tools + + +class TestCursorMessagesArmToolNormalization: + @pytest.mark.asyncio + async def test_flat_custom_tool_nested_before_chat_completion_delegation(self): + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.proxy._types import UserAPIKeyAuth + + seen = {} + + async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict): + from litellm.proxy.common_utils.http_parsing_utils import _read_request_body + + seen["body"] = await _read_request_body(request=request) + return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []} + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-5.6", + "messages": [{"role": "user", "content": "use ApplyPatch"}], + "tools": [ + { + "type": "function", + "function": {"name": "read_file", "parameters": {"type": "object"}}, + }, + {"type": "custom", "name": "ApplyPatch", "description": "V4A patch"}, + ], + "tool_choice": "required", + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert seen["body"]["tools"] == [ + {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}}, + {"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}, + ] + assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}] + + @pytest.mark.asyncio + async def test_messages_body_without_flat_tools_leaves_parsed_body_cache_untouched(self): + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.proxy._types import UserAPIKeyAuth + + seen = {} + + async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict): + from litellm.proxy.common_utils.http_parsing_utils import _read_request_body + + seen["body"] = await _read_request_body(request=request) + return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []} + + body = { + "model": "gpt-5.6", + "messages": [{"role": "user", "content": "hi"}], + "tools": [{"type": "function", "function": {"name": "f", "parameters": {}}}], + } + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json=body, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert seen["body"]["tools"] == body["tools"] + assert seen["body"]["messages"] == body["messages"] diff --git a/tests/test_litellm/types/test_types_utils.py b/tests/test_litellm/types/test_types_utils.py index 320c46aed3b..4d08239360f 100644 --- a/tests/test_litellm/types/test_types_utils.py +++ b/tests/test_litellm/types/test_types_utils.py @@ -603,3 +603,92 @@ def test_delattr_fast_path_missing_attribute_is_noop(): del racy.x del racy.x +def test_chat_completion_tool_call_from_dict_custom(): + from litellm.types.utils import ( + ChatCompletionMessageCustomToolCall, + ChatCompletionMessageToolCall, + chat_completion_tool_call_from_dict, + ) + + custom_tc = { + "id": "call_njxQ", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch\n*** End Patch\n"}, + } + parsed = chat_completion_tool_call_from_dict(custom_tc) + assert isinstance(parsed, ChatCompletionMessageCustomToolCall) + assert parsed.model_dump() == custom_tc + + func_tc = {"id": "call_1", "type": "function", "function": {"name": "f", "arguments": "{}"}} + parsed_func = chat_completion_tool_call_from_dict(func_tc) + assert isinstance(parsed_func, ChatCompletionMessageToolCall) + assert "custom" not in parsed_func.model_dump() + + +def test_chat_completion_tool_call_from_dict_custom_strips_null_function(): + from litellm.types.utils import chat_completion_tool_call_from_dict + + sdk_shaped = { + "id": "call_x", + "type": "custom", + "function": None, + "custom": {"name": "ApplyPatch", "input": ""}, + } + parsed = chat_completion_tool_call_from_dict(sdk_shaped) + assert "function" not in parsed.model_dump() + + +def test_message_with_mixed_function_and_custom_tool_calls(): + from litellm.types.utils import ( + ChatCompletionMessageCustomToolCall, + ChatCompletionMessageToolCall, + Message, + ) + + message = Message( + content=None, + role="assistant", + tool_calls=[ + {"id": "call_c", "type": "custom", "custom": {"name": "ApplyPatch", "input": "patch"}}, + {"id": "call_f", "type": "function", "function": {"name": "f", "arguments": "{}"}}, + ], + ) + assert isinstance(message.tool_calls[0], ChatCompletionMessageCustomToolCall) + assert isinstance(message.tool_calls[1], ChatCompletionMessageToolCall) + dumped = message.model_dump()["tool_calls"] + assert dumped[0] == {"id": "call_c", "type": "custom", "custom": {"name": "ApplyPatch", "input": "patch"}} + assert "custom" not in dumped[1] + + +def test_delta_custom_tool_call_first_and_continuation_chunks(): + from litellm.types.utils import ChatCompletionDeltaCustomToolCall, Delta + + first_chunk_tc = { + "index": 0, + "id": "call_TBs", + "function": None, + "type": "custom", + "custom": {"name": "ApplyPatch", "input": ""}, + } + continuation_tc = {"index": 0, "id": None, "function": None, "type": None, "custom": {"input": "***"}} + + first_delta = Delta(role="assistant", tool_calls=[first_chunk_tc]) + assert isinstance(first_delta.tool_calls[0], ChatCompletionDeltaCustomToolCall) + first_dump = first_delta.model_dump()["tool_calls"][0] + assert first_dump["type"] == "custom" + assert first_dump["custom"] == {"name": "ApplyPatch", "input": ""} + assert "function" not in first_dump + + continuation_delta = Delta(tool_calls=[continuation_tc]) + cont_dump = continuation_delta.model_dump()["tool_calls"][0] + assert cont_dump["type"] is None + assert cont_dump["custom"]["input"] == "***" + assert "function" not in cont_dump + + +def test_delta_function_tool_call_unchanged_by_custom_support(): + from litellm.types.utils import ChatCompletionDeltaToolCall, Delta + + delta = Delta(tool_calls=[{"index": 0, "id": "c2", "type": "function", "function": {"name": "g", "arguments": ""}}]) + assert isinstance(delta.tool_calls[0], ChatCompletionDeltaToolCall) + assert "custom" not in delta.model_dump()["tool_calls"][0] diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 1bc8839976d..bf3cc75c796 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -2628,7 +2628,9 @@ export interface paths { * custom tools) to the chat/completions path while expecting chat completions responses; * those are routed through the Responses API pipeline and converted back. Genuine chat * completions bodies (`messages` present) are routed through the standard chat completions - * pipeline untouched. + * pipeline, after nesting any flat Responses-style tool defs Cursor mixes into the chat + * `tools` array (e.g. `{"type": "custom", "name": "ApplyPatch", ...}`) into the chat + * completions shape OpenAI requires (`{"type": "custom", "custom": {...}}`). * * ```bash * curl -X POST http://localhost:4000/cursor/chat/completions -H "Content-Type: application/json" -H "Authorization: Bearer sk-1234" -d '{ @@ -22195,6 +22197,15 @@ export interface components { */ type: "ephemeral"; }; + /** ChatCompletionCustomToolCallPayload */ + ChatCompletionCustomToolCallPayload: { + /** Input */ + input: string; + /** Name */ + name: string; + } & { + [key: string]: unknown; + }; /** ChatCompletionDeveloperMessage */ ChatCompletionDeveloperMessage: { cache_control?: components["schemas"]["ChatCompletionCachedContent"]; @@ -22281,6 +22292,20 @@ export interface components { /** Url */ url: string; }; + /** ChatCompletionMessageCustomToolCall */ + ChatCompletionMessageCustomToolCall: { + custom: components["schemas"]["ChatCompletionCustomToolCallPayload"]; + /** Id */ + id: string; + /** + * Type + * @default custom + * @constant + */ + type: "custom"; + } & { + [key: string]: unknown; + }; /** ChatCompletionMessageToolCall */ ChatCompletionMessageToolCall: { [key: string]: unknown; @@ -27909,7 +27934,7 @@ export interface components { /** Thinking Blocks */ thinking_blocks?: (components["schemas"]["ChatCompletionThinkingBlock"] | components["schemas"]["ChatCompletionRedactedThinkingBlock"])[] | null; /** Tool Calls */ - tool_calls: components["schemas"]["ChatCompletionMessageToolCall"][] | null; + tool_calls: (components["schemas"]["ChatCompletionMessageToolCall"] | components["schemas"]["ChatCompletionMessageCustomToolCall"])[] | null; } & { [key: string]: unknown; }; From b79b01e38afbe740530efb294c98e116d2bf9f6c Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 13:59:23 -0700 Subject: [PATCH 07/22] fix(proxy): translate custom tool grammar formats and tool_choice across API surfaces Cursor's ApplyPatch is a grammar-constrained custom tool; the Responses surface carries the grammar flat while chat completions wraps the same fields in a grammar object, so the nested envelope from the previous commit still 400d at OpenAI (tools[N].custom.format.grammar). Adds a shared flat to nested format helper pair in prompt_templates/common_utils used by the cursor messages arm and the chat-to-responses bridge, nests flat Responses-style tool_choice objects on the cursor arm, flattens chat custom tool_choice on the chat-to-responses bridge, and maps custom tool_choice to function tool_choice on the responses-to-chat bridge to match that bridge's custom-to-function tool downgrade --- .../transformation.py | 29 +++--- .../prompt_templates/common_utils.py | 25 +++++ .../proxy/response_api_endpoints/endpoints.py | 28 +++++- .../transformation.py | 6 ++ ...responses_transformation_transformation.py | 48 ++++++++++ ...ore_utils_prompt_templates_common_utils.py | 45 +++++++++ .../response_api_endpoints/test_endpoints.py | 96 ++++++++++++++++++- .../test_litellm_completion_responses.py | 21 ++++ 8 files changed, 282 insertions(+), 16 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 75cee42dd55..e1842a62d56 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -155,17 +155,20 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): pass def _normalize_tool_choice_for_responses_api(self, tool_choice: Any) -> Any: - """Chat tool_choice uses function.name; Responses API expects top-level name.""" - if not isinstance(tool_choice, dict) or tool_choice.get("type") != "function": + """Chat tool_choice nests the name under function/custom; Responses API expects top-level name.""" + if not isinstance(tool_choice, dict): + return tool_choice + choice_type = tool_choice.get("type") + if choice_type not in ("function", "custom"): return tool_choice if isinstance(tool_choice.get("name"), str) and tool_choice.get("name"): - # Return only Responses shape so stray chat ``function`` key is not sent upstream. - return {"type": "function", "name": tool_choice["name"]} - fn = tool_choice.get("function") - if isinstance(fn, dict): - fn_name = fn.get("name") - if isinstance(fn_name, str) and fn_name: - return {"type": "function", "name": fn_name} + # Return only Responses shape so stray chat ``function``/``custom`` keys are not sent upstream. + return {"type": choice_type, "name": tool_choice["name"]} + nested = tool_choice.get(choice_type) + if isinstance(nested, dict): + nested_name = nested.get("name") + if isinstance(nested_name, str) and nested_name: + return {"type": choice_type, "name": nested_name} return tool_choice def _handle_raw_dict_response_item(self, item: Dict[str, Any], index: int) -> Tuple[Optional[Any], int]: @@ -896,6 +899,10 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) ) elif tool.get("type") == "custom" and isinstance(tool.get("custom"), dict): + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_responses_shape, + ) + custom_payload = tool["custom"] flat_custom: CustomToolParam = { "type": "custom", @@ -903,8 +910,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): } if custom_payload.get("description") is not None: flat_custom["description"] = custom_payload["description"] - if custom_payload.get("format") is not None: - flat_custom["format"] = custom_payload["format"] + if isinstance(custom_payload.get("format"), dict): + flat_custom["format"] = convert_custom_tool_format_to_responses_shape(custom_payload["format"]) responses_tools.append(flat_custom) else: responses_tools.append(tool) # type: ignore diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index c43089950ee..3a7a710c6a9 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -1252,6 +1252,31 @@ def is_function_call(optional_params: dict) -> bool: return False +def convert_custom_tool_format_to_chat_shape(format_obj: dict) -> dict: + """ + Responses API grammar formats are flat ({"type": "grammar", "definition", "syntax"}); + Chat Completions wraps the same fields in a "grammar" object. Text formats are + identical on both surfaces and pass through, as does anything unrecognized. + """ + if format_obj.get("type") == "grammar" and "grammar" not in format_obj: + return { + "type": "grammar", + "grammar": {k: format_obj[k] for k in ("definition", "syntax") if k in format_obj}, + } + return format_obj + + +def convert_custom_tool_format_to_responses_shape(format_obj: dict) -> dict: + """ + Inverse of convert_custom_tool_format_to_chat_shape: unwrap the Chat Completions + "grammar" object into the flat Responses API grammar shape. + """ + grammar = format_obj.get("grammar") + if format_obj.get("type") == "grammar" and isinstance(grammar, dict): + return {"type": "grammar", **{k: grammar[k] for k in ("definition", "syntax") if k in grammar}} + return format_obj + + def get_file_ids_from_messages(messages: List[AllMessageValues]) -> List[str]: """ Gets file ids from messages diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index f9b2cc79f73..7b64bcda7ce 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -29,10 +29,17 @@ _FLAT_FUNCTION_TOOL_KEYS = ("name", "description", "parameters", "strict") def _nest_flat_chat_tool(tool: object) -> object: + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_chat_shape, + ) + if not isinstance(tool, dict) or "name" not in tool: return tool if tool.get("type") == "custom" and "custom" not in tool: - return {"type": "custom", "custom": {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool}} + payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} + if isinstance(payload.get("format"), dict): + payload = {**payload, "format": convert_custom_tool_format_to_chat_shape(payload["format"])} + return {"type": "custom", "custom": payload} if tool.get("type") == "function" and "function" not in tool: return {"type": "function", "function": {k: tool[k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool}} return tool @@ -42,6 +49,16 @@ def _nest_flat_chat_tools(tools: list) -> list: return [_nest_flat_chat_tool(tool) for tool in tools] +def _nest_flat_chat_tool_choice(tool_choice: object) -> object: + if not isinstance(tool_choice, dict) or "name" not in tool_choice: + return tool_choice + if tool_choice.get("type") == "custom" and "custom" not in tool_choice: + return {"type": "custom", "custom": {"name": tool_choice["name"]}} + if tool_choice.get("type") == "function" and "function" not in tool_choice: + return {"type": "function", "function": {"name": tool_choice["name"]}} + return tool_choice + + @router.post( "/v1/responses", dependencies=[Depends(user_api_key_auth)], @@ -391,10 +408,17 @@ async def cursor_chat_completions( # Genuine chat completions body (Cursor sends these for models whose BYOK it # already fixed); delegate so behavior matches /chat/completions exactly tools = data.get("tools") + tool_choice = data.get("tool_choice") + normalized: dict = {} if isinstance(tools, list): nested_tools = _nest_flat_chat_tools(tools) if nested_tools != tools: - _safe_set_request_parsed_body(request=request, parsed_body={**data, "tools": nested_tools}) + normalized["tools"] = nested_tools + nested_tool_choice = _nest_flat_chat_tool_choice(tool_choice) + if nested_tool_choice != tool_choice: + normalized["tool_choice"] = nested_tool_choice + if normalized: + _safe_set_request_parsed_body(request=request, parsed_body={**data, **normalized}) return await chat_completion( request=request, fastapi_response=fastapi_response, diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 6b1ca3564e3..176274d236f 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -162,6 +162,12 @@ class LiteLLMCompletionResponsesConfig: if function_name: return {"type": "function", "function": {"name": function_name}} return "required" + elif tool_choice_type == "custom": + custom = tool_choice.get("custom") + custom_name = tool_choice.get("name") or (custom.get("name") if isinstance(custom, dict) else None) + if custom_name: + return {"type": "function", "function": {"name": custom_name}} + return "required" # Return as-is for unknown formats return tool_choice diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 767d1631649..64112ed43e8 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -2475,6 +2475,15 @@ def test_map_optional_params_tool_choice_chat_nested_to_responses_api(): {"type": "function", "name": "foo"}, ), ({"type": "required"}, {"type": "required"}), + ( + {"type": "custom", "custom": {"name": "ApplyPatch"}}, + {"type": "custom", "name": "ApplyPatch"}, + ), + ( + {"type": "custom", "name": "ApplyPatch"}, + {"type": "custom", "name": "ApplyPatch"}, + ), + ({"type": "custom"}, {"type": "custom"}), ], ) def test_normalize_tool_choice_for_responses_api(tool_choice, expected): @@ -3249,3 +3258,42 @@ def test_convert_tools_to_responses_format_flattens_custom_tool_without_optional handler = LiteLLMResponsesTransformationHandler() converted = handler._convert_tools_to_responses_format([{"type": "custom", "custom": {"name": "Minimal"}}]) assert converted[0] == {"type": "custom", "name": "Minimal"} + + +def test_convert_tools_to_responses_format_unwraps_nested_grammar_format(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + converted = handler._convert_tools_to_responses_format( + [ + { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "format": { + "type": "grammar", + "grammar": {"definition": "start: patch", "syntax": "lark"}, + }, + }, + } + ] + ) + assert converted[0] == { + "type": "custom", + "name": "ApplyPatch", + "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, + } + + +def test_convert_tools_to_responses_format_text_format_passes_through(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + converted = handler._convert_tools_to_responses_format( + [{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}] + ) + assert converted[0] == {"type": "custom", "name": "A", "format": {"type": "text"}} diff --git a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py index 1b1db634ed2..3728cc80323 100644 --- a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py +++ b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_common_utils.py @@ -721,3 +721,48 @@ class TestUnpackLegacyDefs: out = unpack_legacy_defs(schema) assert "components" not in out assert out["properties"]["r0"]["properties"]["p0"] == {"type": "string"} + + +class TestCustomToolFormatShapeConversion: + def test_flat_grammar_to_chat_shape(self): + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_chat_shape, + ) + + assert convert_custom_tool_format_to_chat_shape( + {"type": "grammar", "definition": "start: patch", "syntax": "lark"} + ) == {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}} + + def test_nested_grammar_to_responses_shape(self): + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_responses_shape, + ) + + assert convert_custom_tool_format_to_responses_shape( + {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "regex"}} + ) == {"type": "grammar", "definition": "start: patch", "syntax": "regex"} + + def test_both_directions_are_idempotent_and_pass_text_through(self): + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_chat_shape, + convert_custom_tool_format_to_responses_shape, + ) + + flat = {"type": "grammar", "definition": "d", "syntax": "lark"} + nested = {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}} + text = {"type": "text"} + assert convert_custom_tool_format_to_chat_shape(nested) == nested + assert convert_custom_tool_format_to_responses_shape(flat) == flat + assert convert_custom_tool_format_to_chat_shape(text) == text + assert convert_custom_tool_format_to_responses_shape(text) == text + assert convert_custom_tool_format_to_chat_shape(convert_custom_tool_format_to_responses_shape(nested)) == nested + + def test_unrecognized_formats_pass_through(self): + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_chat_shape, + convert_custom_tool_format_to_responses_shape, + ) + + for weird in ({}, {"type": "grammar"}, {"type": "future_format", "x": 1}): + assert convert_custom_tool_format_to_chat_shape(dict(weird)) in (weird, {"type": "grammar", "grammar": {}}) + assert convert_custom_tool_format_to_responses_shape(dict(weird)) == weird diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 4ef338b35e4..86fafa40811 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -981,9 +981,18 @@ class TestCursorMessagesArmToolNormalization: "type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}, }, - {"type": "custom", "name": "ApplyPatch", "description": "V4A patch"}, + { + "type": "custom", + "name": "ApplyPatch", + "description": "V4A patch", + "format": { + "type": "grammar", + "definition": "start: patch", + "syntax": "lark", + }, + }, ], - "tool_choice": "required", + "tool_choice": {"type": "custom", "name": "ApplyPatch"}, }, headers={"Authorization": "Bearer sk-1234"}, ) @@ -993,8 +1002,19 @@ class TestCursorMessagesArmToolNormalization: assert response.status_code == 200 assert seen["body"]["tools"] == [ {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}}, - {"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}, + { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "description": "V4A patch", + "format": { + "type": "grammar", + "grammar": {"definition": "start: patch", "syntax": "lark"}, + }, + }, + }, ] + assert seen["body"]["tool_choice"] == {"type": "custom", "custom": {"name": "ApplyPatch"}} assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}] @pytest.mark.asyncio @@ -1030,3 +1050,73 @@ class TestCursorMessagesArmToolNormalization: assert response.status_code == 200 assert seen["body"]["tools"] == body["tools"] assert seen["body"]["messages"] == body["messages"] + + +class TestNestFlatChatToolGrammarFormat: + def test_flat_grammar_format_is_wrapped_for_chat(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + result = _nest_flat_chat_tools( + [ + { + "type": "custom", + "name": "ApplyPatch", + "description": "V4A patch", + "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, + } + ] + ) + assert result == [ + { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "description": "V4A patch", + "format": { + "type": "grammar", + "grammar": {"definition": "start: patch", "syntax": "lark"}, + }, + }, + } + ] + + def test_flat_text_format_is_copied_unchanged(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + result = _nest_flat_chat_tools( + [{"type": "custom", "name": "A", "format": {"type": "text"}}] + ) + assert result == [{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}] + + +class TestNestFlatChatToolChoice: + def test_flat_custom_tool_choice_is_nested(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice + + assert _nest_flat_chat_tool_choice({"type": "custom", "name": "ApplyPatch"}) == { + "type": "custom", + "custom": {"name": "ApplyPatch"}, + } + + def test_flat_function_tool_choice_is_nested(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice + + assert _nest_flat_chat_tool_choice({"type": "function", "name": "f"}) == { + "type": "function", + "function": {"name": "f"}, + } + + def test_non_flat_tool_choice_values_pass_through(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice + + for unchanged in ( + "auto", + "required", + None, + {"type": "custom", "custom": {"name": "x"}}, + {"type": "function", "function": {"name": "f"}}, + {"type": "auto"}, + {"name": "typeless"}, + 42, + ): + assert _nest_flat_chat_tool_choice(unchanged) == unchanged diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index d8e3f495ced..3f3f51f0d3f 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -959,6 +959,27 @@ class TestToolChoiceTransformation: ) assert result == {"type": "function", "function": {"name": "get_weather"}} + def test_transform_tool_choice_custom_follows_function_downgrade(self): + """ + This bridge downgrades custom tools to function tools + (convert_custom_tool_to_function_tool), so a custom tool_choice must become a + function tool_choice naming the same tool or it references a tool type absent + from the converted request. + """ + flat = LiteLLMCompletionResponsesConfig._transform_tool_choice( + {"type": "custom", "name": "ApplyPatch"} + ) + assert flat == {"type": "function", "function": {"name": "ApplyPatch"}} + + nested = LiteLLMCompletionResponsesConfig._transform_tool_choice( + {"type": "custom", "custom": {"name": "ApplyPatch"}} + ) + assert nested == {"type": "function", "function": {"name": "ApplyPatch"}} + + def test_transform_tool_choice_custom_without_name_falls_back_to_required(self): + result = LiteLLMCompletionResponsesConfig._transform_tool_choice({"type": "custom"}) + assert result == "required" + def test_transform_tool_choice_function_without_name_falls_back_to_required(self): """A function-type dict with no name still falls back to required""" result = LiteLLMCompletionResponsesConfig._transform_tool_choice( From ebe48d67de3e10f42a46bf19b1700331b72c1e14 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 14:28:03 -0700 Subject: [PATCH 08/22] fix(proxy): normalize each tool shape level independently on the Cursor messages arm Live Cursor Ask-mode captures show the shape dialects mix PER LEVEL: the tool envelope arrives chat-nested while the grammar format inside it is still Responses-flat, so a normalizer that pattern-matches whole-tool templates misses every hybrid. The cursor arm now normalizes the envelope level and the format level independently and idempotently, making it total over the envelope x format matrix; a parametrized 8-cell test pins every combination. The reference BYOK bridge was checked and forwards chat bodies verbatim, so there is no prior art for these hybrids --- .../proxy/response_api_endpoints/endpoints.py | 27 +++++--- .../response_api_endpoints/test_endpoints.py | 69 ++++++++++++++----- ui/litellm-dashboard/src/lib/http/schema.d.ts | 10 ++- 3 files changed, 76 insertions(+), 30 deletions(-) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 7b64bcda7ce..a980f85d406 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -33,14 +33,21 @@ def _nest_flat_chat_tool(tool: object) -> object: convert_custom_tool_format_to_chat_shape, ) - if not isinstance(tool, dict) or "name" not in tool: + if not isinstance(tool, dict): return tool - if tool.get("type") == "custom" and "custom" not in tool: - payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} + if tool.get("type") == "custom": + if isinstance(tool.get("custom"), dict): + envelope = tool + payload = tool["custom"] + elif "name" in tool: + envelope = {"type": "custom"} + payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} + else: + return tool if isinstance(payload.get("format"), dict): payload = {**payload, "format": convert_custom_tool_format_to_chat_shape(payload["format"])} - return {"type": "custom", "custom": payload} - if tool.get("type") == "function" and "function" not in tool: + return {**envelope, "custom": payload} + if tool.get("type") == "function" and "function" not in tool and "name" in tool: return {"type": "function", "function": {k: tool[k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool}} return tool @@ -364,9 +371,13 @@ async def cursor_chat_completions( custom tools) to the chat/completions path while expecting chat completions responses; those are routed through the Responses API pipeline and converted back. Genuine chat completions bodies (`messages` present) are routed through the standard chat completions - pipeline, after nesting any flat Responses-style tool defs Cursor mixes into the chat - `tools` array (e.g. `{"type": "custom", "name": "ApplyPatch", ...}`) into the chat - completions shape OpenAI requires (`{"type": "custom", "custom": {...}}`). + pipeline, after normalizing each level of the `tools` array and `tool_choice` to the chat + completions shapes OpenAI requires. Cursor mixes Responses API shapes into chat bodies + per level, independently: a flat tool def (`{"type": "custom", "name": "ApplyPatch", ...}`) + gets nested under `custom`, and a flat grammar format + (`{"type": "grammar", "definition", "syntax"}`) gets wrapped as + `{"type": "grammar", "grammar": {...}}` wherever it appears, including inside tool defs + Cursor already sent pre-nested. ```bash curl -X POST http://localhost:4000/cursor/chat/completions \ diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 86fafa40811..b8699d6ef8c 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1052,26 +1052,56 @@ class TestCursorMessagesArmToolNormalization: assert seen["body"]["messages"] == body["messages"] -class TestNestFlatChatToolGrammarFormat: - def test_flat_grammar_format_is_wrapped_for_chat(self): +class TestNestFlatChatToolShapeMatrix: + """ + Cursor mixes Responses API shapes into chat bodies PER LEVEL, independently + (live-captured: a pre-nested custom envelope carrying a flat grammar format). + Every cell of envelope x format must land on the canonical chat shape. + """ + + FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"} + NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}} + TEXT = {"type": "text"} + + @pytest.mark.parametrize("envelope", ["flat", "nested"]) + @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) + def test_every_envelope_and_format_combination_lands_canonical(self, envelope, format_shape): from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools - result = _nest_flat_chat_tools( - [ - { - "type": "custom", - "name": "ApplyPatch", - "description": "V4A patch", - "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, - } - ] - ) - assert result == [ + format_value = { + "absent": None, + "text": self.TEXT, + "flat_grammar": self.FLAT_GRAMMAR, + "nested_grammar": self.NESTED_GRAMMAR, + }[format_shape] + payload = {"name": "ApplyPatch", "description": "V4A patch"} + if format_value is not None: + payload["format"] = format_value + tool = {"type": "custom", "custom": payload} if envelope == "nested" else {"type": "custom", **payload} + + canonical_payload = {"name": "ApplyPatch", "description": "V4A patch"} + if format_shape in ("flat_grammar", "nested_grammar"): + canonical_payload["format"] = self.NESTED_GRAMMAR + elif format_shape == "text": + canonical_payload["format"] = self.TEXT + + assert _nest_flat_chat_tools([tool]) == [{"type": "custom", "custom": canonical_payload}] + + def test_nested_envelope_with_flat_grammar_matches_live_cursor_capture(self): + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + + cursor_tool = { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, + }, + } + assert _nest_flat_chat_tools([cursor_tool]) == [ { "type": "custom", "custom": { "name": "ApplyPatch", - "description": "V4A patch", "format": { "type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}, @@ -1080,13 +1110,14 @@ class TestNestFlatChatToolGrammarFormat: } ] - def test_flat_text_format_is_copied_unchanged(self): + def test_canonical_nested_tool_is_returned_equal(self): from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools - result = _nest_flat_chat_tools( - [{"type": "custom", "name": "A", "format": {"type": "text"}}] - ) - assert result == [{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}] + canonical = { + "type": "custom", + "custom": {"name": "A", "format": {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}}}, + } + assert _nest_flat_chat_tools([canonical]) == [canonical] class TestNestFlatChatToolChoice: diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index bf3cc75c796..94f633c676e 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -2628,9 +2628,13 @@ export interface paths { * custom tools) to the chat/completions path while expecting chat completions responses; * those are routed through the Responses API pipeline and converted back. Genuine chat * completions bodies (`messages` present) are routed through the standard chat completions - * pipeline, after nesting any flat Responses-style tool defs Cursor mixes into the chat - * `tools` array (e.g. `{"type": "custom", "name": "ApplyPatch", ...}`) into the chat - * completions shape OpenAI requires (`{"type": "custom", "custom": {...}}`). + * pipeline, after normalizing each level of the `tools` array and `tool_choice` to the chat + * completions shapes OpenAI requires. Cursor mixes Responses API shapes into chat bodies + * per level, independently: a flat tool def (`{"type": "custom", "name": "ApplyPatch", ...}`) + * gets nested under `custom`, and a flat grammar format + * (`{"type": "grammar", "definition", "syntax"}`) gets wrapped as + * `{"type": "grammar", "grammar": {...}}` wherever it appears, including inside tool defs + * Cursor already sent pre-nested. * * ```bash * curl -X POST http://localhost:4000/cursor/chat/completions -H "Content-Type: application/json" -H "Authorization: Bearer sk-1234" -d '{ From 6d102ea5599a1ecf9c8fb823a88a18fc8131cd0c Mon Sep 17 00:00:00 2001 From: Tin Date: Tue, 21 Jul 2026 16:11:35 -0700 Subject: [PATCH 09/22] fix(litellm): bridge gpt-5.4+ chat requests with tools when reasoning defaults on OpenAI enables reasoning by default for gpt-5.4+ (unset reasoning_effort means medium server-side) and Chat Completions rejects function tools whenever reasoning is on, so a tools request without an explicit reasoning_effort 400d instead of auto-bridging to the Responses API; the bridge heuristic now treats unset effort as reasoning-active and honors the documented escape hatch by keeping explicit "none" on chat completions. The cursor input arm also gains the mirror of the messages-arm normalization: chat-nested tool envelopes, grammar formats, and object tool_choice flatten to the Responses dialect before dispatch --- litellm/main.py | 13 +- .../proxy/response_api_endpoints/endpoints.py | 49 +++++++ .../response_api_endpoints/test_endpoints.py | 136 ++++++++++++++++++ tests/test_litellm/test_main.py | 72 +++++++++- 4 files changed, 262 insertions(+), 8 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index acdec7385da..43d021ebe8b 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1022,7 +1022,12 @@ def responses_api_bridge_check( # ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects # those keys. # - # - gpt-5.4+: tools + reasoning_effort (original) or any reasoning-summary alias. + # - gpt-5.4+: function tools with reasoning active must be bridged. OpenAI enables + # reasoning by default for these models (unset reasoning_effort means medium + # server-side), and Chat Completions rejects tools whenever reasoning is on + # ("Function tools with reasoning_effort are not supported ... use /v1/responses + # or set reasoning_effort to 'none'"), so only an explicit ``"none"`` keeps the + # request chat-servable. # - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning # summary alias is present with ``reasoning_effort`` (tools alone stay on chat). if ( @@ -1030,8 +1035,10 @@ def responses_api_bridge_check( and model_info.get("mode") != "responses" and OpenAIGPT5Config.is_model_gpt_5_model(model) and not OpenAIGPT5Config.is_model_gpt_5_search_model(model) - and reasoning_effort is not None - and (reasoning_summary is not None or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and tools)) + and ( + (reasoning_effort is not None and reasoning_summary is not None) + or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and tools and reasoning_effort != "none") + ) ): model_info["mode"] = "responses" model = model.replace("responses/", "") diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index a980f85d406..8a678601284 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -66,6 +66,47 @@ def _nest_flat_chat_tool_choice(tool_choice: object) -> object: return tool_choice +def _flatten_chat_tool_for_responses(tool: object) -> object: + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + convert_custom_tool_format_to_responses_shape, + ) + + if not isinstance(tool, dict): + return tool + if tool.get("type") == "custom": + if isinstance(tool.get("custom"), dict): + payload = {k: tool["custom"][k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool["custom"]} + elif "name" in tool: + payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} + else: + return tool + if isinstance(payload.get("format"), dict): + payload = {**payload, "format": convert_custom_tool_format_to_responses_shape(payload["format"])} + return {"type": "custom", **payload} + if tool.get("type") == "function" and isinstance(tool.get("function"), dict): + return { + "type": "function", + **{k: tool["function"][k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool["function"]}, + } + return tool + + +def _flatten_chat_tools_for_responses(tools: list) -> list: + return [_flatten_chat_tool_for_responses(tool) for tool in tools] + + +def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object: + if not isinstance(tool_choice, dict): + return tool_choice + choice_type = tool_choice.get("type") + if choice_type not in ("custom", "function"): + return tool_choice + nested = tool_choice.get(choice_type) + if isinstance(nested, dict) and isinstance(nested.get("name"), str): + return {"type": choice_type, "name": nested["name"]} + return tool_choice + + @router.post( "/v1/responses", dependencies=[Depends(user_api_key_auth)], @@ -444,6 +485,14 @@ async def cursor_chat_completions( # cache's key snapshot so later readers get an empty body data = {key: value for key, value in data.items() if key != "stream_options"} + tools = data.get("tools") + if isinstance(tools, list): + data = {**data, "tools": _flatten_chat_tools_for_responses(tools)} + tool_choice = data.get("tool_choice") + flattened_tool_choice = _flatten_chat_tool_choice_for_responses(tool_choice) + if flattened_tool_choice != tool_choice: + data = {**data, "tool_choice": flattened_tool_choice} + processor = ProxyBaseLLMRequestProcessing(data=data) def cursor_data_generator(response, user_api_key_dict, request_data, request=None): diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index b8699d6ef8c..04f027ac4bb 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1151,3 +1151,139 @@ class TestNestFlatChatToolChoice: 42, ): assert _nest_flat_chat_tool_choice(unchanged) == unchanged + + +class TestFlattenChatToolsForResponsesInputArm: + """ + Mirror of TestNestFlatChatToolShapeMatrix for the input arm: chat-nested shapes in a + Responses-shaped body must flatten to the Responses dialect, per level, idempotently. + """ + + FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"} + NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}} + + @pytest.mark.parametrize("envelope", ["flat", "nested"]) + @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) + def test_every_envelope_and_format_combination_lands_flat(self, envelope, format_shape): + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + + format_value = { + "absent": None, + "text": {"type": "text"}, + "flat_grammar": self.FLAT_GRAMMAR, + "nested_grammar": self.NESTED_GRAMMAR, + }[format_shape] + payload = {"name": "ApplyPatch", "description": "V4A patch"} + if format_value is not None: + payload["format"] = format_value + tool = {"type": "custom", "custom": payload} if envelope == "nested" else {"type": "custom", **payload} + + canonical = {"type": "custom", "name": "ApplyPatch", "description": "V4A patch"} + if format_shape in ("flat_grammar", "nested_grammar"): + canonical["format"] = self.FLAT_GRAMMAR + elif format_shape == "text": + canonical["format"] = {"type": "text"} + + assert _flatten_chat_tools_for_responses([tool]) == [canonical] + + def test_nested_function_tool_is_flattened_and_flat_passes_through(self): + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + + nested = {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}} + flat = {"type": "function", "name": "read_file", "parameters": {"type": "object"}} + assert _flatten_chat_tools_for_responses([nested]) == [flat] + assert _flatten_chat_tools_for_responses([flat]) == [flat] + + def test_unrecognized_entries_pass_through(self): + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + + entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}] + assert _flatten_chat_tools_for_responses(entries) == entries + + +class TestFlattenChatToolChoiceForResponsesInputArm: + def test_nested_custom_and_function_tool_choice_flatten(self): + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses + + assert _flatten_chat_tool_choice_for_responses({"type": "custom", "custom": {"name": "ApplyPatch"}}) == { + "type": "custom", + "name": "ApplyPatch", + } + assert _flatten_chat_tool_choice_for_responses({"type": "function", "function": {"name": "f"}}) == { + "type": "function", + "name": "f", + } + + def test_flat_and_string_tool_choice_pass_through(self): + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses + + for unchanged in ("auto", "required", None, {"type": "custom", "name": "x"}, {"type": "auto"}, 42): + assert _flatten_chat_tool_choice_for_responses(unchanged) == unchanged + + +class TestCursorInputArmFlattening: + @pytest.mark.asyncio + async def test_nested_chat_shapes_in_input_body_reach_aresponses_flattened(self): + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + from litellm.types.llms.openai import ResponsesAPIResponse + + mock_response = ResponsesAPIResponse( + id="resp_flat123", + created_at=1234567890, + model="gpt-5.6", + object="response", + output=[ + ResponseOutputMessage( + id="msg_flat123", + type="message", + role="assistant", + status="completed", + content=[ResponseOutputText(type="output_text", text="ok", annotations=[])], + ) + ], + ) + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.llm_router") as mock_router: + mock_router.aresponses = AsyncMock(return_value=mock_response) + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-5.6", + "input": [{"role": "user", "content": "use ApplyPatch"}], + "tools": [ + { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "format": { + "type": "grammar", + "grammar": {"definition": "start: patch", "syntax": "lark"}, + }, + }, + }, + {"type": "function", "name": "read_file", "parameters": {"type": "object"}}, + ], + "tool_choice": {"type": "custom", "custom": {"name": "ApplyPatch"}}, + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + call_kwargs = mock_router.aresponses.call_args.kwargs + assert call_kwargs["tools"] == [ + { + "type": "custom", + "name": "ApplyPatch", + "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, + }, + {"type": "function", "name": "read_file", "parameters": {"type": "object"}}, + ] + assert call_kwargs["tool_choice"] == {"type": "custom", "name": "ApplyPatch"} diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 4611aafa3c1..b4f94177f7d 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -810,8 +810,12 @@ def test_responses_api_bridge_check_azure_gpt_5_4_tools_plus_reasoning_routes_to assert model_info.get("mode") == "responses" -def test_responses_api_bridge_check_azure_gpt_5_4_tools_without_reasoning_stays_chat(): - """Azure gpt-5.4 with tools only should not be force-routed to Responses API.""" +def test_responses_api_bridge_check_azure_gpt_5_4_tools_with_default_reasoning_routes_to_responses(): + """ + Azure gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables + reasoning by default for gpt-5.4+, and Chat Completions rejects function tools + whenever reasoning is on. + """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: @@ -824,11 +828,15 @@ def test_responses_api_bridge_check_azure_gpt_5_4_tools_without_reasoning_stays_ ) assert model == "gpt-5.4" - assert model_info.get("mode") != "responses" + assert model_info.get("mode") == "responses" -def test_responses_api_bridge_check_gpt_5_4_tools_without_reasoning_stays_chat(): - """gpt-5.4 with tools only should not be force-routed to Responses API.""" +def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_to_responses(): + """ + gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning + by default for gpt-5.4+, and Chat Completions rejects function tools whenever + reasoning is on ("use /v1/responses or set reasoning_effort to 'none'"). + """ from litellm.main import responses_api_bridge_check with patch("litellm.main._get_model_info_helper") as mock_get_model_info: @@ -841,6 +849,60 @@ def test_responses_api_bridge_check_gpt_5_4_tools_without_reasoning_stays_chat() ) assert model == "gpt-5.4" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_gpt_5_4_tools_with_reasoning_none_stays_chat(): + """ + Explicit reasoning_effort "none" is OpenAI's documented escape hatch that keeps + function tools servable on Chat Completions; the bridge must not fire. + """ + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.4", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort="none", + ) + + assert model == "gpt-5.4" + assert model_info.get("mode") != "responses" + + +def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_responses(): + """A reasoning summary is Responses-only regardless of effort value.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.4", + custom_llm_provider="openai", + reasoning_effort="none", + reasoning_summary="detailed", + ) + + assert model == "gpt-5.4" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): + """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.1", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + ) + + assert model == "gpt-5.1" assert model_info.get("mode") != "responses" From 7276f44b1db7ccab847549acd298230fa74ad243 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 18:10:16 -0700 Subject: [PATCH 10/22] fix(litellm): gate the gpt-5.4+ responses bridge on function tools specifically OpenAI's chat completions rejection applies to function tools only; custom (grammar) tools are served natively with reasoning on, live-proven by a 200 on a custom-only gpt-5.6 chat request. Gating on any truthy tools needlessly bridged custom-only requests, and the bridge maps custom tool calls back function-shaped, so the native chat custom tool_call surface added earlier in this PR was bypassed exactly where chat serves it natively. The gate now checks for a function-type tool in either the nested chat or flat Responses def shape; the same coarseness existed on the explicit-effort arm before this PR and is fixed by the shared leg --- litellm/main.py | 20 +++++++---- tests/test_litellm/test_main.py | 59 +++++++++++++++++++++++++++++++++ 2 files changed, 73 insertions(+), 6 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 43d021ebe8b..8a9e0e37ace 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1022,14 +1022,20 @@ def responses_api_bridge_check( # ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects # those keys. # - # - gpt-5.4+: function tools with reasoning active must be bridged. OpenAI enables + # - gpt-5.4+: FUNCTION tools with reasoning active must be bridged. OpenAI enables # reasoning by default for these models (unset reasoning_effort means medium - # server-side), and Chat Completions rejects tools whenever reasoning is on - # ("Function tools with reasoning_effort are not supported ... use /v1/responses - # or set reasoning_effort to 'none'"), so only an explicit ``"none"`` keeps the - # request chat-servable. + # server-side), and Chat Completions rejects function tools whenever reasoning is + # on ("Function tools with reasoning_effort are not supported ... use + # /v1/responses or set reasoning_effort to 'none'"), so only an explicit + # ``"none"`` keeps the request chat-servable. Custom (grammar) tools are served + # natively by Chat Completions with reasoning on, so custom-only requests stay on + # chat and keep their native custom tool_call response shape. # - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning # summary alias is present with ``reasoning_effort`` (tools alone stay on chat). + has_function_tool = any( + (tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function") + for tool in (tools or []) + ) if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" @@ -1037,7 +1043,9 @@ def responses_api_bridge_check( and not OpenAIGPT5Config.is_model_gpt_5_search_model(model) and ( (reasoning_effort is not None and reasoning_summary is not None) - or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and tools and reasoning_effort != "none") + or ( + OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_effort != "none" + ) ) ): model_info["mode"] = "responses" diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index b4f94177f7d..4a0ec04bed1 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -889,6 +889,65 @@ def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_ assert model_info.get("mode") == "responses" +def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat(): + """ + Chat Completions serves custom (grammar) tools natively with reasoning on; only + FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat + so responses keep the native custom tool_call shape instead of the bridge's + function-shaped mapping. + """ + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}], + reasoning_effort=None, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + +def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses(): + """One function tool in the mix is enough to make chat unservable with reasoning on.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[ + {"type": "custom", "custom": {"name": "ApplyPatch"}}, + {"type": "function", "function": {"name": "shell"}}, + ], + reasoning_effort=None, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses(): + """Responses-style flat function tool defs still count as function tools.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}], + reasoning_effort=None, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" from litellm.main import responses_api_bridge_check From bbba450301344ee8b4f4981a32dfe7fc63d10f9b Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 20:13:02 -0700 Subject: [PATCH 11/22] fix(litellm): honor dict-form reasoning_effort in the bridge escape hatch and serialize custom tool calls in helicone and lunary logs The bridge gate compared reasoning_effort against the string "none", so litellm's dict form ({"effort": "none"}) wrongly bridged; the gate now reads the effort value from either form and treats a summary inside the dict as Responses-only regardless of effort. Helicone and lunary previously skipped custom tool calls entirely; both now serialize them (helicone as a tool_use block from the custom payload, lunary with the custom name and input in its function fields, keeping type custom), with new mapped tests for both integrations --- litellm/integrations/helicone.py | 29 +++++++---- litellm/integrations/lunary.py | 16 +++++- litellm/main.py | 8 +-- .../integrations/test_helicone.py | 51 +++++++++++++++++++ .../test_litellm/integrations/test_lunary.py | 40 +++++++++++++++ tests/test_litellm/test_main.py | 50 ++++++++++++++++++ 6 files changed, 180 insertions(+), 14 deletions(-) create mode 100644 tests/test_litellm/integrations/test_helicone.py create mode 100644 tests/test_litellm/integrations/test_lunary.py diff --git a/litellm/integrations/helicone.py b/litellm/integrations/helicone.py index 4c7a606c16f..c9346f7e6cf 100644 --- a/litellm/integrations/helicone.py +++ b/litellm/integrations/helicone.py @@ -60,16 +60,25 @@ class HeliconeLogger: if "tool_calls" in message and message["tool_calls"]: for tool_call in message["tool_calls"]: function = tool_call.get("function") - if not function: - continue - content.append( - { - "type": "tool_use", - "id": tool_call["id"], - "name": function["name"], - "input": function["arguments"], - } - ) + custom = tool_call.get("custom") + if function: + content.append( + { + "type": "tool_use", + "id": tool_call["id"], + "name": function["name"], + "input": function["arguments"], + } + ) + elif custom: + content.append( + { + "type": "tool_use", + "id": tool_call["id"], + "name": custom["name"], + "input": custom["input"], + } + ) elif "content" in message and message["content"]: content = [{"type": "text", "text": message["content"]}] diff --git a/litellm/integrations/lunary.py b/litellm/integrations/lunary.py index 448580f0b2d..94cb5bab8fe 100644 --- a/litellm/integrations/lunary.py +++ b/litellm/integrations/lunary.py @@ -20,6 +20,16 @@ def parse_tool_calls(tool_calls): return None def clean_tool_call(tool_call): + custom = getattr(tool_call, "custom", None) + if custom is not None: + return { + "type": tool_call.type, + "id": tool_call.id, + "function": { + "name": custom.name, + "arguments": custom.input, + }, + } serialized = { "type": tool_call.type, "id": tool_call.id, @@ -31,7 +41,11 @@ def parse_tool_calls(tool_calls): return serialized - return [clean_tool_call(tool_call) for tool_call in tool_calls if getattr(tool_call, "function", None) is not None] + return [ + clean_tool_call(tool_call) + for tool_call in tool_calls + if getattr(tool_call, "function", None) is not None or getattr(tool_call, "custom", None) is not None + ] def parse_messages(input): diff --git a/litellm/main.py b/litellm/main.py index 8a9e0e37ace..b6c6b44a6f7 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1036,6 +1036,10 @@ def responses_api_bridge_check( (tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function") for tool in (tools or []) ) + if isinstance(reasoning_effort, dict): + reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None + else: + reasoning_active = reasoning_effort != "none" if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" @@ -1043,9 +1047,7 @@ def responses_api_bridge_check( and not OpenAIGPT5Config.is_model_gpt_5_search_model(model) and ( (reasoning_effort is not None and reasoning_summary is not None) - or ( - OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_effort != "none" - ) + or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_active) ) ): model_info["mode"] = "responses" diff --git a/tests/test_litellm/integrations/test_helicone.py b/tests/test_litellm/integrations/test_helicone.py new file mode 100644 index 00000000000..eeef6fadbe5 --- /dev/null +++ b/tests/test_litellm/integrations/test_helicone.py @@ -0,0 +1,51 @@ +import os +import sys +import types + +sys.path.insert(0, os.path.abspath("../../..")) + +from litellm.integrations.helicone import HeliconeLogger + + +def _claude_mapping(messages, response_obj): + logger = HeliconeLogger.__new__(HeliconeLogger) + return logger.claude_mapping(model="gpt-5.6", messages=messages, response_obj=response_obj) + + +def test_claude_mapping_serializes_custom_tool_calls(monkeypatch): + try: + import anthropic # noqa: F401 + except ImportError: + stub = types.ModuleType("anthropic") + stub.HUMAN_PROMPT = "\n\nHuman:" + stub.AI_PROMPT = "\n\nAssistant:" + monkeypatch.setitem(sys.modules, "anthropic", stub) + response_obj = { + "id": "chatcmpl-1", + "choices": [ + { + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_c", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}, + }, + { + "id": "call_f", + "type": "function", + "function": {"name": "read_file", "arguments": '{"path": "a.py"}'}, + }, + ], + }, + "finish_reason": "tool_calls", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 2}, + } + mapped = _claude_mapping([{"role": "user", "content": "hi"}], response_obj) + tool_use_blocks = [b for b in mapped["content"] if b["type"] == "tool_use"] + assert {"type": "tool_use", "id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"} in tool_use_blocks + assert {"type": "tool_use", "id": "call_f", "name": "read_file", "input": '{"path": "a.py"}'} in tool_use_blocks diff --git a/tests/test_litellm/integrations/test_lunary.py b/tests/test_litellm/integrations/test_lunary.py new file mode 100644 index 00000000000..0a1ec100594 --- /dev/null +++ b/tests/test_litellm/integrations/test_lunary.py @@ -0,0 +1,40 @@ +import os +import sys + +sys.path.insert(0, os.path.abspath("../../..")) + +from litellm.integrations.lunary import parse_tool_calls +from litellm.types.utils import ( + ChatCompletionMessageCustomToolCall, + ChatCompletionMessageToolCall, + Function, +) + + +def test_parse_tool_calls_serializes_custom_tool_calls(): + custom_call = ChatCompletionMessageCustomToolCall( + id="call_c", + custom={"name": "ApplyPatch", "input": "*** Begin Patch"}, + ) + function_call = ChatCompletionMessageToolCall( + id="call_f", + type="function", + function=Function(name="read_file", arguments='{"path": "a.py"}'), + ) + parsed = parse_tool_calls([custom_call, function_call]) + assert parsed == [ + { + "type": "custom", + "id": "call_c", + "function": {"name": "ApplyPatch", "arguments": "*** Begin Patch"}, + }, + { + "type": "function", + "id": "call_f", + "function": {"name": "read_file", "arguments": '{"path": "a.py"}'}, + }, + ] + + +def test_parse_tool_calls_none_passthrough(): + assert parse_tool_calls(None) is None diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 4a0ec04bed1..60558760f8e 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -948,6 +948,56 @@ def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_respons assert model_info.get("mode") == "responses" +def test_responses_api_bridge_check_dict_effort_none_stays_chat(): + """The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort={"effort": "none"}, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + +def test_responses_api_bridge_check_dict_effort_active_routes_to_responses(): + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort={"effort": "low"}, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_responses(): + """A summary inside the dict form is Responses-only even when effort is none.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort={"effort": "none", "summary": "concise"}, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" from litellm.main import responses_api_bridge_check From e9d16bc35cb7e1a754114a1977cdcab441c9f39b Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 20:51:21 -0700 Subject: [PATCH 12/22] fix(litellm): make the responses bridge and cursor routing total over the surfaces they now serve Three gaps from the bridge becoming a mainstream path for chat traffic. The chat to responses message converter only mapped function tool_calls, so history carrying the native custom tool calls this PR introduced raised "tool call not supported" on follow-up turns; custom entries now map to custom_tool_call items and their results to custom_tool_call_output. The stream translator returned an empty delta for output_item.done on tool items, which left the responses guardrail handler's tool extraction permanently empty (dead on staging too, where the built chunk was discarded); stateless callers now receive the complete tool call while per-stream callers keep the suppressed delta that prevents client-side duplication. Cursor routing keyed on the presence of a messages key, so a null or empty stub next to a real agent-mode input array picked the chat arm; routing now keys on messages content --- .../transformation.py | 56 ++++++++-- .../proxy/response_api_endpoints/endpoints.py | 13 ++- ...responses_transformation_transformation.py | 103 ++++++++++++++++++ .../response_api_endpoints/test_endpoints.py | 59 ++++++++++ 4 files changed, 222 insertions(+), 9 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index e1842a62d56..768bf6c3e66 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -221,6 +221,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) -> Tuple[List[Any], Optional[str]]: input_items: List[Any] = [] instructions: Optional[str] = None + custom_tool_call_ids: set = set() for msg in messages: role = msg.get("role") @@ -266,18 +267,28 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): else: # Fallback: convert unexpected types to input_text tool_output = [{"type": "input_text", "text": str(content)}] - input_items.append( - { - "type": "function_call_output", - "call_id": tool_call_id, - "output": tool_output, - } - ) + if tool_call_id in custom_tool_call_ids: + input_items.append( + { + "type": "custom_tool_call_output", + "call_id": tool_call_id, + "output": content if isinstance(content, str) else tool_output, + } + ) + else: + input_items.append( + { + "type": "function_call_output", + "call_id": tool_call_id, + "output": tool_output, + } + ) elif role == "assistant" and tool_calls and isinstance(tool_calls, list): for r_item in _get_reasoning_items(msg): input_items.append(_reasoning_item_to_response_input(r_item)) for tool_call in tool_calls: function = tool_call.get("function") + custom = tool_call.get("custom") if function: input_tool_call: Dict[str, Any] = { "type": "function_call", @@ -288,6 +299,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): if "arguments" in function: input_tool_call["arguments"] = function["arguments"] input_items.append(input_tool_call) + elif isinstance(custom, dict): + custom_tool_call_ids.add(tool_call["id"]) + input_items.append( + { + "type": "custom_tool_call", + "call_id": tool_call["id"], + "name": custom.get("name", ""), + "input": custom.get("input", ""), + } + ) else: raise ValueError(f"tool call not supported: {tool_call}") elif content is not None: @@ -1272,6 +1293,27 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): # New output item added output_item = parsed_chunk.get("item", {}) if output_item.get("type") in ("function_call", "custom_tool_call"): + if tool_call_index_map is None: + # Stateless callers (the responses guardrail handler extracting + # tool calls from a buffered output_item.done) get the complete + # tool call; per-stream callers already received it via + # output_item.added and the argument delta events + return ModelResponseStream( + choices=[ + StreamingChoices( + index=0, + delta=Delta( + tool_calls=[ + { + **_tool_call_dict_from_output_item(dict(output_item)), + "index": parsed_chunk.get("output_index", 0), + } + ] + ), + finish_reason=None, + ) + ] + ) # Do NOT emit finish_reason here — response.completed handles the terminal # finish_reason. Emitting "tool_calls" here would prematurely terminate # the stream before subsequent tool calls arrive (same fix as #17246 for diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 8a678601284..8ee742c0d69 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -95,6 +95,13 @@ def _flatten_chat_tools_for_responses(tools: list) -> list: return [_flatten_chat_tool_for_responses(tool) for tool in tools] +def _is_chat_completions_body(data: dict) -> bool: + messages = data.get("messages") + if isinstance(messages, list) and len(messages) > 0: + return True + return "messages" in data and "input" not in data + + def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object: if not isinstance(tool_choice, dict): return tool_choice @@ -456,9 +463,11 @@ async def cursor_chat_completions( data = await _read_request_body(request=request) - if "messages" in data: + if _is_chat_completions_body(data): # Genuine chat completions body (Cursor sends these for models whose BYOK it - # already fixed); delegate so behavior matches /chat/completions exactly + # already fixed); delegate so behavior matches /chat/completions exactly. + # Keyed on messages CONTENT, not key presence: Cursor can send a null or + # empty messages stub alongside a real agent-mode input array tools = data.get("tools") tool_choice = data.get("tool_choice") normalized: dict = {} diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 64112ed43e8..b8bd5c951ee 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -3297,3 +3297,106 @@ def test_convert_tools_to_responses_format_text_format_passes_through(): [{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}] ) assert converted[0] == {"type": "custom", "name": "A", "format": {"type": "text"}} + + +def test_convert_chat_completion_messages_maps_custom_tool_call_history(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + {"role": "user", "content": "use ApplyPatch"}, + { + "role": "assistant", + "tool_calls": [ + { + "id": "call_c", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}, + }, + { + "id": "call_f", + "type": "function", + "function": {"name": "shell", "arguments": '{"cmd": "ls"}'}, + }, + ], + }, + {"role": "tool", "tool_call_id": "call_c", "content": "patch applied"}, + {"role": "tool", "tool_call_id": "call_f", "content": "a.py"}, + ] + ) + assert { + "type": "custom_tool_call", + "call_id": "call_c", + "name": "ApplyPatch", + "input": "*** Begin Patch", + } in input_items + assert {"type": "custom_tool_call_output", "call_id": "call_c", "output": "patch applied"} in input_items + assert {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'} in input_items + assert { + "type": "function_call_output", + "call_id": "call_f", + "output": [{"type": "input_text", "text": "a.py"}], + } in input_items + + +def test_convert_chat_completion_messages_still_rejects_unknown_tool_call_shape(): + import pytest + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + with pytest.raises(ValueError, match="tool call not supported"): + handler.convert_chat_completion_messages_to_responses_api( + [{"role": "assistant", "tool_calls": [{"id": "call_x", "type": "mystery"}]}] + ) + + +def test_output_item_done_stateless_emits_complete_tool_call(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + for item, expected_name, expected_args in ( + ( + {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'}, + "shell", + '{"cmd": "ls"}', + ), + ( + {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"}, + "ApplyPatch", + "*** Begin Patch", + ), + ): + chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( + {"type": "response.output_item.done", "output_index": 2, "item": item} + ) + tool_calls = chunk.choices[0].delta.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0].id == item["call_id"] + assert tool_calls[0].function.name == expected_name + assert tool_calls[0].function.arguments == expected_args + assert tool_calls[0].index == 2 + assert chunk.choices[0].finish_reason is None + + +def test_output_item_done_with_stream_map_keeps_empty_delta(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( + { + "type": "response.output_item.done", + "output_index": 0, + "item": {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "x"}, + }, + tool_call_index_map={0: 0}, + ) + assert chunk.choices[0].delta.tool_calls is None + assert chunk.choices[0].finish_reason is None diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 04f027ac4bb..6a1e0d0a494 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1287,3 +1287,62 @@ class TestCursorInputArmFlattening: {"type": "function", "name": "read_file", "parameters": {"type": "object"}}, ] assert call_kwargs["tool_choice"] == {"type": "custom", "name": "ApplyPatch"} + + +class TestChatCompletionsBodyDetection: + def test_routing_matrix(self): + from litellm.proxy.response_api_endpoints.endpoints import _is_chat_completions_body + + assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}]}) is True + assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}], "input": []}) is True + assert _is_chat_completions_body({"messages": None, "input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({"messages": [], "input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({"messages": None}) is True + assert _is_chat_completions_body({"messages": []}) is True + assert _is_chat_completions_body({"input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({}) is False + + @pytest.mark.asyncio + async def test_null_messages_stub_with_input_reaches_responses_arm(self): + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.types.llms.openai import ResponsesAPIResponse + + mock_response = ResponsesAPIResponse( + id="resp_stub1", + created_at=1234567890, + model="gpt-5.6", + object="response", + output=[ + ResponseOutputMessage( + id="msg_stub1", + type="message", + role="assistant", + status="completed", + content=[ResponseOutputText(type="output_text", text="ok", annotations=[])], + ) + ], + ) + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.llm_router") as mock_router: + mock_router.aresponses = AsyncMock(return_value=mock_response) + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-5.6", + "messages": None, + "input": [{"role": "user", "content": "hello"}], + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert mock_router.aresponses.call_args is not None + assert mock_router.aresponses.call_args.kwargs["input"] == [{"role": "user", "content": "hello"}] From a5ba1caac5c312ff2189b1200d1eda54b5cac8e6 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 21:06:41 -0700 Subject: [PATCH 13/22] test(helicone): stub the anthropic module unconditionally An import probe proves nothing about the real SDK: it may be absent (it lives in the proxy-runtime extra) and the tests/test_litellm/llms/anthropic test package can shadow it once collection puts that path on sys.path, which made the test order-sensitive across collection sets --- tests/test_litellm/integrations/test_helicone.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/tests/test_litellm/integrations/test_helicone.py b/tests/test_litellm/integrations/test_helicone.py index eeef6fadbe5..da07fa1a9bf 100644 --- a/tests/test_litellm/integrations/test_helicone.py +++ b/tests/test_litellm/integrations/test_helicone.py @@ -13,13 +13,15 @@ def _claude_mapping(messages, response_obj): def test_claude_mapping_serializes_custom_tool_calls(monkeypatch): - try: - import anthropic # noqa: F401 - except ImportError: - stub = types.ModuleType("anthropic") - stub.HUMAN_PROMPT = "\n\nHuman:" - stub.AI_PROMPT = "\n\nAssistant:" - monkeypatch.setitem(sys.modules, "anthropic", stub) + """ + Stub the anthropic module unconditionally: the SDK may be absent (it lives in the + proxy-runtime extra), and the tests/test_litellm/llms/anthropic test package can + shadow it on sys.path, so an import probe proves nothing about the real SDK. + """ + stub = types.ModuleType("anthropic") + stub.HUMAN_PROMPT = "\n\nHuman:" + stub.AI_PROMPT = "\n\nAssistant:" + monkeypatch.setitem(sys.modules, "anthropic", stub) response_obj = { "id": "chatcmpl-1", "choices": [ From d516a72c0587fe813d7fdce39e5741fd74f2f660 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 21:47:51 -0700 Subject: [PATCH 14/22] fix(litellm): scope the unset-effort responses bridge to constraint-enforcing endpoints Chat-only OpenAI-compatible backends registered under the openai provider with custom api_base and gpt-5.4+ model names served tools-without-reasoning fine and have no /responses route, so the unset-effort arm added for real OpenAI would have silently rerouted previously working deployments. The arm now fires only when api_base is unset (default OpenAI endpoint) or the provider is azure; an explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base. Flagged lines also modernized to PEP 604 --- .../convert_dict_to_response.py | 9 +-- .../llms/openai/chat/gpt_transformation.py | 4 +- litellm/main.py | 17 +++++- tests/test_litellm/test_main.py | 58 +++++++++++++++++++ 4 files changed, 78 insertions(+), 10 deletions(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index c5cfdea9ffe..1b23db87264 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -531,12 +531,9 @@ class LiteLLMResponseObjectHandler: def _should_convert_tool_call_to_json_mode( - tool_calls: Optional[ - Union[ - List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]], - List[DatabricksTool], - ] - ] = None, + tool_calls: ( + list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | list[DatabricksTool] | None + ) = None, convert_tool_call_to_json_mode: Optional[bool] = None, ) -> bool: """ diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index 129a9b51d0d..e4492a8aba6 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -533,9 +533,7 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): for choice in choices: ## HANDLE JSON MODE - anthropic returns single function call] tool_calls = choice["message"].get("tool_calls", None) - new_tool_calls: Optional[ - List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] - ] = None + new_tool_calls: list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | None = None message_content = choice["message"].get("content", None) if tool_calls is not None: _openai_tool_calls = [] diff --git a/litellm/main.py b/litellm/main.py index b6c6b44a6f7..d008d976130 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -986,6 +986,7 @@ def responses_api_bridge_check( tools: Optional[List[Any]] = None, reasoning_effort: Optional[Any] = None, reasoning_summary: Optional[Any] = None, + api_base: str | None = None, ) -> Tuple[dict, str]: model_info: Dict[str, Any] = {} @@ -1030,6 +1031,12 @@ def responses_api_bridge_check( # ``"none"`` keeps the request chat-servable. Custom (grammar) tools are served # natively by Chat Completions with reasoning on, so custom-only requests stay on # chat and keep their native custom tool_call response shape. + # - The UNSET-effort arm only fires against endpoints known to enforce that + # constraint (the default OpenAI endpoint, or Azure OpenAI where api_base is + # always set): chat-only OpenAI-compatible backends registered under the openai + # provider with a custom api_base and gpt-5.4+ model names serve tools without + # reasoning fine and have no /responses route, so they keep pre-existing + # behavior (bridge only on an explicit reasoning_effort). # - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning # summary alias is present with ``reasoning_effort`` (tools alone stay on chat). has_function_tool = any( @@ -1040,6 +1047,7 @@ def responses_api_bridge_check( reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None else: reasoning_active = reasoning_effort != "none" + on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or api_base is None if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" @@ -1047,7 +1055,12 @@ def responses_api_bridge_check( and not OpenAIGPT5Config.is_model_gpt_5_search_model(model) and ( (reasoning_effort is not None and reasoning_summary is not None) - or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_active) + or ( + OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) + and has_function_tool + and reasoning_active + and (reasoning_effort is not None or on_constraint_enforcing_endpoint) + ) ) ): model_info["mode"] = "responses" @@ -5173,6 +5186,7 @@ def completion( # type: ignore model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, + api_base=api_base, ) if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model): @@ -5412,6 +5426,7 @@ def completion( # type: ignore tools=tools, reasoning_effort=reasoning_effort, reasoning_summary=_reasoning_summary_for_bridge, + api_base=api_base, ) # Use base_model (the true underlying model) for Azure model-type diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 60558760f8e..f72d2b5e23b 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -998,6 +998,64 @@ def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_resp assert model_info.get("mode") == "responses" +def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat(): + """ + Chat-only OpenAI-compatible backends registered under the openai provider with a + custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and + have no /responses route; the unset-effort arm must not reroute them. + """ + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base="http://vllm.internal:8000/v1", + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + +def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes(): + """Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort="high", + api_base="http://vllm.internal:8000/v1", + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes(): + """Azure OpenAI always sets api_base and does enforce the constraint; keep bridging.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.4", + custom_llm_provider="azure", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base="https://myresource.openai.azure.com", + ) + + assert model == "gpt-5.4" + assert model_info.get("mode") == "responses" + + def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" from litellm.main import responses_api_bridge_check From cc00650fecfd9b3bb1b44806a5cf8dc71e043dcf Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Wed, 22 Jul 2026 15:46:55 -0700 Subject: [PATCH 15/22] fix(litellm): treat a blank api_base as the default OpenAI endpoint in the bridge gate A blank api_base (empty or whitespace) resolves to the default OpenAI base downstream but is not None, so the constraint-enforcing-endpoint check misclassified it as a custom backend and skipped the unset-effort auto-bridge, leaving gpt-5.4+ function-tool requests to 400 at OpenAI. The check now treats None, empty, and whitespace api_base alike; a real custom base still opts out. Verified with get_llm_provider, which passes a blank api_base through while resolving the provider to openai --- litellm/main.py | 5 ++++- tests/test_litellm/test_main.py | 23 +++++++++++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/litellm/main.py b/litellm/main.py index d008d976130..4aa6bf9a19b 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1047,7 +1047,10 @@ def responses_api_bridge_check( reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None else: reasoning_active = reasoning_effort != "none" - on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or api_base is None + # A blank api_base (None, "", or whitespace) is not a custom endpoint: it resolves + # to the default OpenAI base downstream, which does enforce the reasoning+tools + # constraint. Azure always targets an OpenAI-constraint endpoint regardless. + on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or not (api_base and api_base.strip()) if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index f72d2b5e23b..057d11e1ecd 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -998,6 +998,29 @@ def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_resp assert model_info.get("mode") == "responses" +@pytest.mark.parametrize("blank_api_base", [None, "", " ", "\t"]) +def test_responses_api_bridge_check_blank_api_base_is_default_openai(blank_api_base): + """ + A blank api_base (None, empty, or whitespace) resolves to the default OpenAI + endpoint downstream, which enforces the reasoning+tools constraint, so gpt-5.4+ + function-tool requests with unset reasoning_effort must still auto-bridge. + """ + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base=blank_api_base, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat(): """ Chat-only OpenAI-compatible backends registered under the openai provider with a From 56cc475c801db630077737bee23556bf0fe55d53 Mon Sep 17 00:00:00 2001 From: tin Date: Thu, 23 Jul 2026 00:46:13 +0000 Subject: [PATCH 16/22] refactor(cursor): trim LOC in cursor byok tool normalization - share one _CustomToolCallAccess mixin across the 4 new custom-tool classes instead of hand-rolling dict access on each - inline the single-use _nest_flat_chat_tools / _flatten_chat_tools_for_responses list wrappers at their call sites - drop _nest_flat_chat_tool_choice: it rewrote object-form chat tool_choice into {type,custom:{name}}, a shape OpenAI rejects; real Cursor never sends tool_choice on the messages arm, so pass it through unchanged --- .../proxy/response_api_endpoints/endpoints.py | 26 +--- litellm/types/utils.py | 64 +++------- .../response_api_endpoints/test_endpoints.py | 115 ++++++------------ 3 files changed, 58 insertions(+), 147 deletions(-) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 8ee742c0d69..a3c1100eec3 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -52,20 +52,6 @@ def _nest_flat_chat_tool(tool: object) -> object: return tool -def _nest_flat_chat_tools(tools: list) -> list: - return [_nest_flat_chat_tool(tool) for tool in tools] - - -def _nest_flat_chat_tool_choice(tool_choice: object) -> object: - if not isinstance(tool_choice, dict) or "name" not in tool_choice: - return tool_choice - if tool_choice.get("type") == "custom" and "custom" not in tool_choice: - return {"type": "custom", "custom": {"name": tool_choice["name"]}} - if tool_choice.get("type") == "function" and "function" not in tool_choice: - return {"type": "function", "function": {"name": tool_choice["name"]}} - return tool_choice - - def _flatten_chat_tool_for_responses(tool: object) -> object: from litellm.litellm_core_utils.prompt_templates.common_utils import ( convert_custom_tool_format_to_responses_shape, @@ -91,10 +77,6 @@ def _flatten_chat_tool_for_responses(tool: object) -> object: return tool -def _flatten_chat_tools_for_responses(tools: list) -> list: - return [_flatten_chat_tool_for_responses(tool) for tool in tools] - - def _is_chat_completions_body(data: dict) -> bool: messages = data.get("messages") if isinstance(messages, list) and len(messages) > 0: @@ -469,15 +451,11 @@ async def cursor_chat_completions( # Keyed on messages CONTENT, not key presence: Cursor can send a null or # empty messages stub alongside a real agent-mode input array tools = data.get("tools") - tool_choice = data.get("tool_choice") normalized: dict = {} if isinstance(tools, list): - nested_tools = _nest_flat_chat_tools(tools) + nested_tools = [_nest_flat_chat_tool(tool) for tool in tools] if nested_tools != tools: normalized["tools"] = nested_tools - nested_tool_choice = _nest_flat_chat_tool_choice(tool_choice) - if nested_tool_choice != tool_choice: - normalized["tool_choice"] = nested_tool_choice if normalized: _safe_set_request_parsed_body(request=request, parsed_body={**data, **normalized}) return await chat_completion( @@ -496,7 +474,7 @@ async def cursor_chat_completions( tools = data.get("tools") if isinstance(tools, list): - data = {**data, "tools": _flatten_chat_tools_for_responses(tools)} + data = {**data, "tools": [_flatten_chat_tool_for_responses(tool) for tool in tools]} tool_choice = data.get("tool_choice") flattened_tool_choice = _flatten_chat_tool_choice_for_responses(tool_choice) if flattened_tool_choice != tool_choice: diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a1e52f7584f..7ff12b617e3 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1084,70 +1084,42 @@ class ChatCompletionDeltaToolCall(OpenAIObject): setattr(self, key, value) -class ChatCompletionCustomToolCallPayload(OpenAIObject): +class _CustomToolCallAccess(OpenAIObject): + def __contains__(self, key): + return hasattr(self, key) + + def get(self, key, default=None): + return getattr(self, key, default) + + def __getitem__(self, key): + return getattr(self, key) + + def __setitem__(self, key, value): + setattr(self, key, value) + + +class ChatCompletionCustomToolCallPayload(_CustomToolCallAccess): name: str input: str - def __contains__(self, key): - return hasattr(self, key) - def get(self, key, default=None): - return getattr(self, key, default) - - def __getitem__(self, key): - return getattr(self, key) - - -class ChatCompletionDeltaCustomToolCallPayload(OpenAIObject): +class ChatCompletionDeltaCustomToolCallPayload(_CustomToolCallAccess): name: str | None = None input: str | None = None - def __contains__(self, key): - return hasattr(self, key) - def get(self, key, default=None): - return getattr(self, key, default) - - def __getitem__(self, key): - return getattr(self, key) - - -class ChatCompletionMessageCustomToolCall(OpenAIObject): +class ChatCompletionMessageCustomToolCall(_CustomToolCallAccess): id: str type: Literal["custom"] = "custom" custom: ChatCompletionCustomToolCallPayload - def __contains__(self, key): - return hasattr(self, key) - def get(self, key, default=None): - return getattr(self, key, default) - - def __getitem__(self, key): - return getattr(self, key) - - def __setitem__(self, key, value): - setattr(self, key, value) - - -class ChatCompletionDeltaCustomToolCall(OpenAIObject): +class ChatCompletionDeltaCustomToolCall(_CustomToolCallAccess): id: str | None = None type: str | None = None custom: ChatCompletionDeltaCustomToolCallPayload index: int - def __contains__(self, key): - return hasattr(self, key) - - def get(self, key, default=None): - return getattr(self, key, default) - - def __getitem__(self, key): - return getattr(self, key) - - def __setitem__(self, key, value): - setattr(self, key, value) - class ChatCompletionMessageToolCall(OpenAIObject): def __init__( diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 6a1e0d0a494..88e7bbc84d4 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -911,33 +911,29 @@ def test_cursor_models_route_delegates_to_model_list(): class TestNestFlatChatTools: def test_flat_custom_tool_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool - result = _nest_flat_chat_tools( - [{"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}] + result = _nest_flat_chat_tool( + {"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}} ) - assert result == [ - { - "type": "custom", - "custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}, - } - ] + assert result == { + "type": "custom", + "custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}, + } def test_flat_function_tool_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool - result = _nest_flat_chat_tools( - [{"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}}] + result = _nest_flat_chat_tool( + {"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}} ) - assert result == [ - { - "type": "function", - "function": {"name": "read_file", "description": "d", "parameters": {"type": "object"}}, - } - ] + assert result == { + "type": "function", + "function": {"name": "read_file", "description": "d", "parameters": {"type": "object"}}, + } def test_already_nested_and_unrecognized_tools_pass_through_unchanged(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool tools = [ {"type": "custom", "custom": {"name": "already_nested"}}, @@ -950,7 +946,7 @@ class TestNestFlatChatTools: None, 42, ] - assert _nest_flat_chat_tools(tools) == tools + assert [_nest_flat_chat_tool(tool) for tool in tools] == tools class TestCursorMessagesArmToolNormalization: @@ -1014,7 +1010,7 @@ class TestCursorMessagesArmToolNormalization: }, }, ] - assert seen["body"]["tool_choice"] == {"type": "custom", "custom": {"name": "ApplyPatch"}} + assert seen["body"]["tool_choice"] == {"type": "custom", "name": "ApplyPatch"} assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}] @pytest.mark.asyncio @@ -1066,7 +1062,7 @@ class TestNestFlatChatToolShapeMatrix: @pytest.mark.parametrize("envelope", ["flat", "nested"]) @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) def test_every_envelope_and_format_combination_lands_canonical(self, envelope, format_shape): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool format_value = { "absent": None, @@ -1085,10 +1081,10 @@ class TestNestFlatChatToolShapeMatrix: elif format_shape == "text": canonical_payload["format"] = self.TEXT - assert _nest_flat_chat_tools([tool]) == [{"type": "custom", "custom": canonical_payload}] + assert _nest_flat_chat_tool(tool) == {"type": "custom", "custom": canonical_payload} def test_nested_envelope_with_flat_grammar_matches_live_cursor_capture(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool cursor_tool = { "type": "custom", @@ -1097,60 +1093,25 @@ class TestNestFlatChatToolShapeMatrix: "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, }, } - assert _nest_flat_chat_tools([cursor_tool]) == [ - { - "type": "custom", - "custom": { - "name": "ApplyPatch", - "format": { - "type": "grammar", - "grammar": {"definition": "start: patch", "syntax": "lark"}, - }, + assert _nest_flat_chat_tool(cursor_tool) == { + "type": "custom", + "custom": { + "name": "ApplyPatch", + "format": { + "type": "grammar", + "grammar": {"definition": "start: patch", "syntax": "lark"}, }, - } - ] + }, + } def test_canonical_nested_tool_is_returned_equal(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tools + from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool canonical = { "type": "custom", "custom": {"name": "A", "format": {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}}}, } - assert _nest_flat_chat_tools([canonical]) == [canonical] - - -class TestNestFlatChatToolChoice: - def test_flat_custom_tool_choice_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice - - assert _nest_flat_chat_tool_choice({"type": "custom", "name": "ApplyPatch"}) == { - "type": "custom", - "custom": {"name": "ApplyPatch"}, - } - - def test_flat_function_tool_choice_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice - - assert _nest_flat_chat_tool_choice({"type": "function", "name": "f"}) == { - "type": "function", - "function": {"name": "f"}, - } - - def test_non_flat_tool_choice_values_pass_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool_choice - - for unchanged in ( - "auto", - "required", - None, - {"type": "custom", "custom": {"name": "x"}}, - {"type": "function", "function": {"name": "f"}}, - {"type": "auto"}, - {"name": "typeless"}, - 42, - ): - assert _nest_flat_chat_tool_choice(unchanged) == unchanged + assert _nest_flat_chat_tool(canonical) == canonical class TestFlattenChatToolsForResponsesInputArm: @@ -1165,7 +1126,7 @@ class TestFlattenChatToolsForResponsesInputArm: @pytest.mark.parametrize("envelope", ["flat", "nested"]) @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) def test_every_envelope_and_format_combination_lands_flat(self, envelope, format_shape): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses format_value = { "absent": None, @@ -1184,21 +1145,21 @@ class TestFlattenChatToolsForResponsesInputArm: elif format_shape == "text": canonical["format"] = {"type": "text"} - assert _flatten_chat_tools_for_responses([tool]) == [canonical] + assert _flatten_chat_tool_for_responses(tool) == canonical def test_nested_function_tool_is_flattened_and_flat_passes_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses nested = {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}} flat = {"type": "function", "name": "read_file", "parameters": {"type": "object"}} - assert _flatten_chat_tools_for_responses([nested]) == [flat] - assert _flatten_chat_tools_for_responses([flat]) == [flat] + assert _flatten_chat_tool_for_responses(nested) == flat + assert _flatten_chat_tool_for_responses(flat) == flat def test_unrecognized_entries_pass_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tools_for_responses + from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}] - assert _flatten_chat_tools_for_responses(entries) == entries + assert [_flatten_chat_tool_for_responses(entry) for entry in entries] == entries class TestFlattenChatToolChoiceForResponsesInputArm: From c7c656e8a9132fd583a053ee9cf3d6da95535f4b Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Wed, 22 Jul 2026 23:51:19 -0700 Subject: [PATCH 17/22] fix(cursor): convert tools and tool_choice through one envelope rule Chat Completions nests a named tool_choice under its tool type while the Responses API keeps it flat; ChatCompletionNamedToolChoiceParam and ChatCompletionNamedToolChoiceCustomParam both mark the nested key required. The messages arm normalized tool definitions but forwarded tool_choice at whatever level Cursor sent it, so a flat {"type": "custom", "name": "ApplyPatch"} reached OpenAI unchanged and was rejected while the tool defs beside it nested correctly A tool definition and a named tool_choice carry the same envelope, so both now convert through a single _convert_tool_envelope, and _normalize_tool_dialect moves tools and tool_choice together on each arm. That covers all four cells of {tool def, tool_choice} x {to chat, to responses} and removes the shape where one field can be converted while the other is missed, replacing three helpers with two and cutting 24 lines Also restores the end-to-end assertion that a flat tool_choice reaches chat_completion nested, which had been flipped to pin the passthrough behavior --- .../proxy/response_api_endpoints/endpoints.py | 109 ++++------ .../response_api_endpoints/test_endpoints.py | 195 +++++++++--------- 2 files changed, 140 insertions(+), 164 deletions(-) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index a3c1100eec3..b67352f2057 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -24,57 +24,47 @@ router = APIRouter() _user_api_key_auth_dep = Depends(user_api_key_auth) -_FLAT_CUSTOM_TOOL_KEYS = ("name", "description", "format") -_FLAT_FUNCTION_TOOL_KEYS = ("name", "description", "parameters", "strict") +_TOOL_PAYLOAD_KEYS = { + "custom": ("name", "description", "format"), + "function": ("name", "description", "parameters", "strict"), +} -def _nest_flat_chat_tool(tool: object) -> object: +def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object: from litellm.litellm_core_utils.prompt_templates.common_utils import ( convert_custom_tool_format_to_chat_shape, - ) - - if not isinstance(tool, dict): - return tool - if tool.get("type") == "custom": - if isinstance(tool.get("custom"), dict): - envelope = tool - payload = tool["custom"] - elif "name" in tool: - envelope = {"type": "custom"} - payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} - else: - return tool - if isinstance(payload.get("format"), dict): - payload = {**payload, "format": convert_custom_tool_format_to_chat_shape(payload["format"])} - return {**envelope, "custom": payload} - if tool.get("type") == "function" and "function" not in tool and "name" in tool: - return {"type": "function", "function": {k: tool[k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool}} - return tool - - -def _flatten_chat_tool_for_responses(tool: object) -> object: - from litellm.litellm_core_utils.prompt_templates.common_utils import ( convert_custom_tool_format_to_responses_shape, ) - if not isinstance(tool, dict): - return tool - if tool.get("type") == "custom": - if isinstance(tool.get("custom"), dict): - payload = {k: tool["custom"][k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool["custom"]} - elif "name" in tool: - payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool} - else: - return tool - if isinstance(payload.get("format"), dict): - payload = {**payload, "format": convert_custom_tool_format_to_responses_shape(payload["format"])} - return {"type": "custom", **payload} - if tool.get("type") == "function" and isinstance(tool.get("function"), dict): - return { - "type": "function", - **{k: tool["function"][k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool["function"]}, - } - return tool + if not isinstance(obj, dict): + return obj + tool_type = obj.get("type") + payload_keys = _TOOL_PAYLOAD_KEYS.get(tool_type) + if payload_keys is None: + return obj + nested = obj.get(tool_type) + source = nested if isinstance(nested, dict) else obj + if source is obj and "name" not in obj: + return obj + payload = {key: source[key] for key in payload_keys if key in source} + if isinstance(payload.get("format"), dict): + convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape + payload = {**payload, "format": convert(payload["format"])} + return {"type": tool_type, tool_type: payload} if to_chat else {"type": tool_type, **payload} + + +def _normalize_tool_dialect(data: dict, *, to_chat: bool) -> dict: + converted: dict = {} + tools = data.get("tools") + if isinstance(tools, list): + normalized_tools = [_convert_tool_envelope(tool, to_chat=to_chat) for tool in tools] + if normalized_tools != tools: + converted["tools"] = normalized_tools + tool_choice = data.get("tool_choice") + normalized_choice = _convert_tool_envelope(tool_choice, to_chat=to_chat) + if normalized_choice != tool_choice: + converted["tool_choice"] = normalized_choice + return {**data, **converted} if converted else data def _is_chat_completions_body(data: dict) -> bool: @@ -84,18 +74,6 @@ def _is_chat_completions_body(data: dict) -> bool: return "messages" in data and "input" not in data -def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object: - if not isinstance(tool_choice, dict): - return tool_choice - choice_type = tool_choice.get("type") - if choice_type not in ("custom", "function"): - return tool_choice - nested = tool_choice.get(choice_type) - if isinstance(nested, dict) and isinstance(nested.get("name"), str): - return {"type": choice_type, "name": nested["name"]} - return tool_choice - - @router.post( "/v1/responses", dependencies=[Depends(user_api_key_auth)], @@ -450,14 +428,9 @@ async def cursor_chat_completions( # already fixed); delegate so behavior matches /chat/completions exactly. # Keyed on messages CONTENT, not key presence: Cursor can send a null or # empty messages stub alongside a real agent-mode input array - tools = data.get("tools") - normalized: dict = {} - if isinstance(tools, list): - nested_tools = [_nest_flat_chat_tool(tool) for tool in tools] - if nested_tools != tools: - normalized["tools"] = nested_tools - if normalized: - _safe_set_request_parsed_body(request=request, parsed_body={**data, **normalized}) + normalized = _normalize_tool_dialect(data, to_chat=True) + if normalized is not data: + _safe_set_request_parsed_body(request=request, parsed_body=normalized) return await chat_completion( request=request, fastapi_response=fastapi_response, @@ -472,13 +445,7 @@ async def cursor_chat_completions( # cache's key snapshot so later readers get an empty body data = {key: value for key, value in data.items() if key != "stream_options"} - tools = data.get("tools") - if isinstance(tools, list): - data = {**data, "tools": [_flatten_chat_tool_for_responses(tool) for tool in tools]} - tool_choice = data.get("tool_choice") - flattened_tool_choice = _flatten_chat_tool_choice_for_responses(tool_choice) - if flattened_tool_choice != tool_choice: - data = {**data, "tool_choice": flattened_tool_choice} + data = _normalize_tool_dialect(data, to_chat=False) processor = ProxyBaseLLMRequestProcessing(data=data) diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 88e7bbc84d4..fa5a16a9f3b 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -911,10 +911,11 @@ def test_cursor_models_route_delegates_to_model_list(): class TestNestFlatChatTools: def test_flat_custom_tool_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - result = _nest_flat_chat_tool( - {"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}} + result = _convert_tool_envelope( + {"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}, + to_chat=True, ) assert result == { "type": "custom", @@ -922,10 +923,11 @@ class TestNestFlatChatTools: } def test_flat_function_tool_is_nested(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - result = _nest_flat_chat_tool( - {"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}} + result = _convert_tool_envelope( + {"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}}, + to_chat=True, ) assert result == { "type": "function", @@ -933,7 +935,7 @@ class TestNestFlatChatTools: } def test_already_nested_and_unrecognized_tools_pass_through_unchanged(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope tools = [ {"type": "custom", "custom": {"name": "already_nested"}}, @@ -946,7 +948,7 @@ class TestNestFlatChatTools: None, 42, ] - assert [_nest_flat_chat_tool(tool) for tool in tools] == tools + assert [_convert_tool_envelope(tool, to_chat=True) for tool in tools] == tools class TestCursorMessagesArmToolNormalization: @@ -1010,7 +1012,7 @@ class TestCursorMessagesArmToolNormalization: }, }, ] - assert seen["body"]["tool_choice"] == {"type": "custom", "name": "ApplyPatch"} + assert seen["body"]["tool_choice"] == {"type": "custom", "custom": {"name": "ApplyPatch"}} assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}] @pytest.mark.asyncio @@ -1048,21 +1050,23 @@ class TestCursorMessagesArmToolNormalization: assert seen["body"]["messages"] == body["messages"] -class TestNestFlatChatToolShapeMatrix: +class TestToolEnvelopeConversionMatrix: """ Cursor mixes Responses API shapes into chat bodies PER LEVEL, independently (live-captured: a pre-nested custom envelope carrying a flat grammar format). - Every cell of envelope x format must land on the canonical chat shape. + Tool definitions and tool_choice share one envelope rule, so every cell of + direction x envelope x format must land on that direction's canonical shape. """ FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"} NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}} TEXT = {"type": "text"} + @pytest.mark.parametrize("to_chat", [True, False]) @pytest.mark.parametrize("envelope", ["flat", "nested"]) @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) - def test_every_envelope_and_format_combination_lands_canonical(self, envelope, format_shape): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + def test_every_direction_envelope_and_format_lands_canonical(self, to_chat, envelope, format_shape): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope format_value = { "absent": None, @@ -1077,109 +1081,114 @@ class TestNestFlatChatToolShapeMatrix: canonical_payload = {"name": "ApplyPatch", "description": "V4A patch"} if format_shape in ("flat_grammar", "nested_grammar"): - canonical_payload["format"] = self.NESTED_GRAMMAR + canonical_payload["format"] = self.NESTED_GRAMMAR if to_chat else self.FLAT_GRAMMAR elif format_shape == "text": canonical_payload["format"] = self.TEXT + expected = ( + {"type": "custom", "custom": canonical_payload} if to_chat else {"type": "custom", **canonical_payload} + ) - assert _nest_flat_chat_tool(tool) == {"type": "custom", "custom": canonical_payload} + assert _convert_tool_envelope(tool, to_chat=to_chat) == expected def test_nested_envelope_with_flat_grammar_matches_live_cursor_capture(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - cursor_tool = { + cursor_tool = {"type": "custom", "custom": {"name": "ApplyPatch", "format": self.FLAT_GRAMMAR}} + assert _convert_tool_envelope(cursor_tool, to_chat=True) == { "type": "custom", - "custom": { - "name": "ApplyPatch", - "format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"}, - }, - } - assert _nest_flat_chat_tool(cursor_tool) == { - "type": "custom", - "custom": { - "name": "ApplyPatch", - "format": { - "type": "grammar", - "grammar": {"definition": "start: patch", "syntax": "lark"}, - }, - }, + "custom": {"name": "ApplyPatch", "format": self.NESTED_GRAMMAR}, } - def test_canonical_nested_tool_is_returned_equal(self): - from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool + @pytest.mark.parametrize("to_chat", [True, False]) + def test_conversion_is_idempotent(self, to_chat): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - canonical = { - "type": "custom", - "custom": {"name": "A", "format": {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}}}, - } - assert _nest_flat_chat_tool(canonical) == canonical + once = _convert_tool_envelope({"type": "custom", "name": "A", "format": self.FLAT_GRAMMAR}, to_chat=to_chat) + assert _convert_tool_envelope(once, to_chat=to_chat) == once - -class TestFlattenChatToolsForResponsesInputArm: - """ - Mirror of TestNestFlatChatToolShapeMatrix for the input arm: chat-nested shapes in a - Responses-shaped body must flatten to the Responses dialect, per level, idempotently. - """ - - FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"} - NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}} - - @pytest.mark.parametrize("envelope", ["flat", "nested"]) - @pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"]) - def test_every_envelope_and_format_combination_lands_flat(self, envelope, format_shape): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses - - format_value = { - "absent": None, - "text": {"type": "text"}, - "flat_grammar": self.FLAT_GRAMMAR, - "nested_grammar": self.NESTED_GRAMMAR, - }[format_shape] - payload = {"name": "ApplyPatch", "description": "V4A patch"} - if format_value is not None: - payload["format"] = format_value - tool = {"type": "custom", "custom": payload} if envelope == "nested" else {"type": "custom", **payload} - - canonical = {"type": "custom", "name": "ApplyPatch", "description": "V4A patch"} - if format_shape in ("flat_grammar", "nested_grammar"): - canonical["format"] = self.FLAT_GRAMMAR - elif format_shape == "text": - canonical["format"] = {"type": "text"} - - assert _flatten_chat_tool_for_responses(tool) == canonical - - def test_nested_function_tool_is_flattened_and_flat_passes_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses + def test_nested_function_tool_flattens_and_flat_passes_through(self): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope nested = {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}} flat = {"type": "function", "name": "read_file", "parameters": {"type": "object"}} - assert _flatten_chat_tool_for_responses(nested) == flat - assert _flatten_chat_tool_for_responses(flat) == flat + assert _convert_tool_envelope(nested, to_chat=False) == flat + assert _convert_tool_envelope(flat, to_chat=False) == flat - def test_unrecognized_entries_pass_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses + @pytest.mark.parametrize("to_chat", [True, False]) + def test_unrecognized_entries_pass_through(self, to_chat): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}] - assert [_flatten_chat_tool_for_responses(entry) for entry in entries] == entries + entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}, 42, {"type": "auto"}] + assert [_convert_tool_envelope(entry, to_chat=to_chat) for entry in entries] == entries -class TestFlattenChatToolChoiceForResponsesInputArm: - def test_nested_custom_and_function_tool_choice_flatten(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses +class TestToolChoiceSharesTheToolEnvelopeRule: + """ + tool_choice carries the same {"type": T, T: {...}} chat envelope as a tool + definition, so it converts through the same function in both directions. + OpenAI requires the nested key on chat (SDK ChatCompletionNamedToolChoiceParam + and ChatCompletionNamedToolChoiceCustomParam both mark it Required). + """ - assert _flatten_chat_tool_choice_for_responses({"type": "custom", "custom": {"name": "ApplyPatch"}}) == { - "type": "custom", + @pytest.mark.parametrize("choice_type", ["custom", "function"]) + def test_flat_tool_choice_is_nested_for_chat(self, choice_type): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope + + assert _convert_tool_envelope({"type": choice_type, "name": "ApplyPatch"}, to_chat=True) == { + "type": choice_type, + choice_type: {"name": "ApplyPatch"}, + } + + @pytest.mark.parametrize("choice_type", ["custom", "function"]) + def test_nested_tool_choice_is_flattened_for_responses(self, choice_type): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope + + assert _convert_tool_envelope({"type": choice_type, choice_type: {"name": "ApplyPatch"}}, to_chat=False) == { + "type": choice_type, "name": "ApplyPatch", } - assert _flatten_chat_tool_choice_for_responses({"type": "function", "function": {"name": "f"}}) == { - "type": "function", - "name": "f", - } - def test_flat_and_string_tool_choice_pass_through(self): - from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses + @pytest.mark.parametrize("to_chat", [True, False]) + def test_sentinel_and_malformed_tool_choice_pass_through(self, to_chat): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope - for unchanged in ("auto", "required", None, {"type": "custom", "name": "x"}, {"type": "auto"}, 42): - assert _flatten_chat_tool_choice_for_responses(unchanged) == unchanged + for unchanged in ("auto", "required", "none", None, {"type": "auto"}, 42): + assert _convert_tool_envelope(unchanged, to_chat=to_chat) == unchanged + + +class TestNormalizeToolDialectCoversBothFields: + """ + The regression that motivated one normalizer: tools were converted while + tool_choice was left flat, so OpenAI rejected the request. Both fields move + together in a single call, on both arms. + """ + + @pytest.mark.parametrize("to_chat", [True, False]) + def test_tools_and_tool_choice_convert_together(self, to_chat): + from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect + + flat = {"type": "custom", "name": "ApplyPatch"} + nested = {"type": "custom", "custom": {"name": "ApplyPatch"}} + source = flat if to_chat else nested + expected = nested if to_chat else flat + + out = _normalize_tool_dialect({"messages": [], "tools": [source], "tool_choice": source}, to_chat=to_chat) + assert out["tools"] == [expected] + assert out["tool_choice"] == expected + + def test_body_needing_no_conversion_is_returned_by_identity(self): + from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect + + data = {"messages": [], "tools": [{"type": "function", "function": {"name": "f"}}], "tool_choice": "auto"} + assert _normalize_tool_dialect(data, to_chat=True) is data + + def test_absent_tool_fields_are_not_invented(self): + from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect + + data = {"messages": [{"role": "user", "content": "hi"}]} + result = _normalize_tool_dialect(data, to_chat=True) + assert result == data + assert "tools" not in result and "tool_choice" not in result class TestCursorInputArmFlattening: From 4139f548da9f3a72b7c9dc327aaaad730472fdbe Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Sat, 25 Jul 2026 10:45:41 -0700 Subject: [PATCH 18/22] fix(bridge): resolve the effective OpenAI base once, shared by gate and chat handler The gpt-5.4+ responses-bridge gate classified the endpoint from the call-level api_base alone, while the OpenAI chat handler resolves arg > global > env > default. A custom base configured via litellm.api_base or OPENAI_BASE_URL/ OPENAI_API_BASE was therefore invisible to the gate: it read blank as the default OpenAI endpoint and bridged a request the custom backend has no /responses route for. Extract that resolution into one _resolve_openai_api_base() and have both the gate and _complete_custom_openai() call it, so the gate can never classify an endpoint the request won't hit. The gate compares the resolved base against the default (import litellm seeds OPENAI_BASE_URL to the default, so "override is non-None" is not a safe custom-endpoint signal); whitespace collapses to the default as before. reasoning_effort="none" remains the escape hatch. --- litellm/main.py | 39 ++++++++++++++++++------- tests/test_litellm/test_main.py | 52 +++++++++++++++++++++++++++++++++ 2 files changed, 80 insertions(+), 11 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 4aa6bf9a19b..fbb43dd41fa 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -979,6 +979,23 @@ def mock_completion( raise Exception("Mock completion response failed - {}".format(e)) +_OPENAI_DEFAULT_API_BASE = "https://api.openai.com/v1" + + +def _resolve_openai_api_base(api_base: str | None) -> str: + """Effective OpenAI base a chat request will hit: arg > global > env > default. The bridge gate + and the ``_complete_custom_openai`` chat handler MUST resolve this identically, or a custom base + set via ``litellm.api_base`` or ``OPENAI_BASE_URL``/``OPENAI_API_BASE`` is invisible to the gate, + which then misreads it as the default OpenAI endpoint and bridges a request the backend can't serve.""" + return ( + api_base + or litellm.api_base + or get_secret_str("OPENAI_BASE_URL") + or get_secret_str("OPENAI_API_BASE") + or _OPENAI_DEFAULT_API_BASE + ) + + def responses_api_bridge_check( model: str, custom_llm_provider: str, @@ -1047,10 +1064,15 @@ def responses_api_bridge_check( reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None else: reasoning_active = reasoning_effort != "none" - # A blank api_base (None, "", or whitespace) is not a custom endpoint: it resolves - # to the default OpenAI base downstream, which does enforce the reasoning+tools - # constraint. Azure always targets an OpenAI-constraint endpoint regardless. - on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or not (api_base and api_base.strip()) + # The reasoning+tools constraint is enforced only by the real OpenAI endpoint (and Azure OpenAI). + # Resolve the effective base arg>global>env>default exactly as the chat handler does, so a custom + # base set via litellm.api_base or OPENAI_BASE_URL/OPENAI_API_BASE isn't misread as the default and + # bridged to a /responses route it lacks. A whitespace-only base collapses to the default too. + resolved_api_base = _resolve_openai_api_base(api_base) + on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or resolved_api_base.strip() in ( + "", + _OPENAI_DEFAULT_API_BASE, + ) if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" @@ -2396,13 +2418,8 @@ def _complete_custom_openai( stream = ctx.stream timeout = ctx.timeout - api_base = ( - api_base # for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there - or litellm.api_base - or get_secret("OPENAI_BASE_URL") - or get_secret("OPENAI_API_BASE") - or "https://api.openai.com/v1" - ) + # for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there + api_base = _resolve_openai_api_base(api_base) organization = ( organization or litellm.organization diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 057d11e1ecd..9e160370048 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -1043,6 +1043,58 @@ def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat assert model_info.get("mode") != "responses" +def test_responses_api_bridge_check_custom_api_base_via_global_with_unset_effort_stays_chat(monkeypatch): + """ + A custom base set through the litellm.api_base global (not the call arg) is resolved the + same way the chat handler resolves it, so the unset-effort arm must not reroute a chat-only + backend to a /responses route it lacks. Regression guard: the gate previously inspected only + the call-level api_base and bridged these requests. + """ + import litellm + from litellm.main import responses_api_bridge_check + + monkeypatch.setattr(litellm, "api_base", "http://vllm.internal:8000/v1") + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base=None, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + +@pytest.mark.parametrize("env_var", ["OPENAI_BASE_URL", "OPENAI_API_BASE"]) +def test_responses_api_bridge_check_custom_api_base_via_env_with_unset_effort_stays_chat(monkeypatch, env_var): + """ + A custom base set via OPENAI_BASE_URL/OPENAI_API_BASE env is resolved identically to the chat + handler, so the unset-effort arm leaves the request on chat instead of bridging it. + """ + import litellm + from litellm.main import responses_api_bridge_check + + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + monkeypatch.setenv(env_var, "http://vllm.internal:8000/v1") + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base=None, + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes(): """Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base.""" from litellm.main import responses_api_bridge_check From c27f1b7b6d274d6dfe6bbc123ee3c5c1b163e83a Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Mon, 27 Jul 2026 11:59:34 -0700 Subject: [PATCH 19/22] fix(tools): classify custom tool calls by one shared rule and make envelope payload extraction total A chat tool-call dict was classified custom-vs-function with four different spellings: the non-streaming parser required type == "custom", the streaming Delta coercion also accepted a custom payload without type, and the stream assembler required a type that later chunks never carry. The same payload could be a custom tool call mid-stream, a TypeError on the completed message, and silently dropped from the assembled message. is_custom_tool_call_dict() is now the single discriminator (explicit custom type, or a custom payload present) used by both parsers, and the assembler classifies from the accumulated custom payload, matching how the deltas it consumes were classified. The tool envelope converter picked one exclusive payload source: the nested dict when present, else the top level. An empty nested envelope therefore shadowed top-level fields and the normalized tool lost its name. Payload extraction is now total over both locations, nested first, and an envelope with no name anywhere passes through unchanged instead of being emitted stripped. --- .../streaming_chunk_builder_utils.py | 2 +- .../proxy/response_api_endpoints/endpoints.py | 10 +++-- litellm/types/utils.py | 10 +++-- .../test_streaming_chunk_builder_utils.py | 37 ++++++++++++++++ .../response_api_endpoints/test_endpoints.py | 24 +++++++++++ tests/test_litellm/types/test_types_utils.py | 42 +++++++++++++++++++ 6 files changed, 118 insertions(+), 7 deletions(-) diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index 09bd55096e8..f4f1b6fca0d 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -322,7 +322,7 @@ class ChunkProcessor: # Convert the map to a list of tool calls for index in sorted(tool_call_map.keys()): tool_call_data = tool_call_map[index] - if tool_call_data["type"] == "custom" and tool_call_data["id"] and tool_call_data["custom_name"]: + if tool_call_data["id"] and tool_call_data["custom_name"]: tool_calls_list.append( ChatCompletionMessageCustomToolCall( id=tool_call_data["id"], diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index b67352f2057..3a2c57aa2ba 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -43,10 +43,14 @@ def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object: if payload_keys is None: return obj nested = obj.get(tool_type) - source = nested if isinstance(nested, dict) else obj - if source is obj and "name" not in obj: + nested_source = nested if isinstance(nested, dict) else {} + payload = { + key: nested_source[key] if key in nested_source else obj[key] + for key in payload_keys + if key in nested_source or key in obj + } + if "name" not in payload: return obj - payload = {key: source[key] for key in payload_keys if key in source} if isinstance(payload.get("format"), dict): convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape payload = {**payload, "format": convert(payload["format"])} diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 7ff12b617e3..404725ec61b 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1162,12 +1162,16 @@ class ChatCompletionMessageToolCall(OpenAIObject): setattr(self, key, value) +def is_custom_tool_call_dict(tool_call: dict) -> bool: + return tool_call.get("type") == "custom" or tool_call.get("custom") is not None + + def chat_completion_tool_call_from_dict( tool_call: dict, ) -> "ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall": - if tool_call.get("type") == "custom": + if is_custom_tool_call_dict(tool_call): return ChatCompletionMessageCustomToolCall( - **{k: v for k, v in tool_call.items() if not (k == "function" and v is None)} + **{k: v for k, v in tool_call.items() if not (k in ("function", "type") and v is None)} ) return ChatCompletionMessageToolCall(**tool_call) @@ -1393,7 +1397,7 @@ class Delta(SafeAttributeModel, OpenAIObject): if tool_call.get("index", None) is None: tool_call["index"] = current_index current_index += 1 - if tool_call.get("type") == "custom" or "custom" in tool_call: + if is_custom_tool_call_dict(tool_call): coerced_tool_calls.append( ChatCompletionDeltaCustomToolCall( **{k: v for k, v in tool_call.items() if not (k == "function" and v is None)} diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py index 197adf80f03..2db5461702a 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py @@ -1027,3 +1027,40 @@ def test_get_combined_tool_content_custom_tool_call(): "type": "custom", "custom": {"name": "ApplyPatch", "input": "*** Begin Patch\n*** End Patch\n"}, } + + +def test_get_combined_tool_content_custom_tool_call_without_type_field(): + """Delta coercion classifies a tool-call chunk as custom from its ``custom`` payload + alone (``type`` may never arrive on any chunk). The assembler must use the same + evidence; requiring ``type == "custom"`` dropped the whole tool call from the + combined message (it matched neither the custom nor the function branch).""" + from litellm.litellm_core_utils.streaming_chunk_builder_utils import ChunkProcessor + from litellm.types.utils import ChatCompletionMessageCustomToolCall + + processor = ChunkProcessor.__new__(ChunkProcessor) + tool_call_chunks = [ + { + "choices": [ + { + "delta": { + "tool_calls": [ + { + "index": 0, + "id": "call_TBs", + "custom": {"name": "ApplyPatch", "input": "*** Begin"}, + } + ] + } + } + ] + }, + {"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": " Patch"}}]}}]}, + ] + combined = processor.get_combined_tool_content(tool_call_chunks) + assert len(combined) == 1 + assert isinstance(combined[0], ChatCompletionMessageCustomToolCall) + assert combined[0].model_dump() == { + "id": "call_TBs", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}, + } diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index fa5a16a9f3b..00ac8ca386a 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1121,6 +1121,30 @@ class TestToolEnvelopeConversionMatrix: entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}, 42, {"type": "auto"}] assert [_convert_tool_envelope(entry, to_chat=to_chat) for entry in entries] == entries + @pytest.mark.parametrize("to_chat", [True, False]) + def test_empty_nested_envelope_falls_back_to_top_level_payload(self, to_chat): + """An empty nested envelope must not shadow payload fields that sit at the top + level; treating the empty dict as the sole payload source dropped the name.""" + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope + + hybrid = {"type": "custom", "custom": {}, "name": "ApplyPatch", "format": self.TEXT} + expected_payload = {"name": "ApplyPatch", "format": self.TEXT} + expected = {"type": "custom", "custom": expected_payload} if to_chat else {"type": "custom", **expected_payload} + assert _convert_tool_envelope(hybrid, to_chat=to_chat) == expected + + def test_nested_payload_wins_over_stray_top_level_fields(self): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope + + tool = {"type": "custom", "custom": {"name": "NestedName"}, "name": "TopName"} + assert _convert_tool_envelope(tool, to_chat=False) == {"type": "custom", "name": "NestedName"} + + @pytest.mark.parametrize("to_chat", [True, False]) + def test_nameless_envelope_passes_through_unchanged(self, to_chat): + from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope + + nameless = {"type": "custom", "custom": {}, "description": "no name anywhere"} + assert _convert_tool_envelope(nameless, to_chat=to_chat) == nameless + class TestToolChoiceSharesTheToolEnvelopeRule: """ diff --git a/tests/test_litellm/types/test_types_utils.py b/tests/test_litellm/types/test_types_utils.py index 4d08239360f..a446f820870 100644 --- a/tests/test_litellm/types/test_types_utils.py +++ b/tests/test_litellm/types/test_types_utils.py @@ -638,6 +638,48 @@ def test_chat_completion_tool_call_from_dict_custom_strips_null_function(): assert "function" not in parsed.model_dump() +def test_chat_completion_tool_call_from_dict_typeless_custom_payload(): + """A tool-call dict can carry a ``custom`` payload with ``type`` absent or None + (e.g. rebuilt from streaming deltas, where only the first chunk has ``type``). + Classifying on ``type == "custom"`` alone sent these to the function branch, + which raised TypeError (missing ``function``) on a payload the streaming path + accepts as custom.""" + from litellm.types.utils import ChatCompletionMessageCustomToolCall, chat_completion_tool_call_from_dict + + typeless = {"id": "call_1", "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}} + parsed = chat_completion_tool_call_from_dict(typeless) + assert isinstance(parsed, ChatCompletionMessageCustomToolCall) + assert parsed.type == "custom" + assert parsed.custom.name == "ApplyPatch" + + null_typed = {"id": "call_2", "type": None, "custom": {"name": "f", "input": "{}"}} + assert isinstance(chat_completion_tool_call_from_dict(null_typed), ChatCompletionMessageCustomToolCall) + + +def test_custom_tool_call_classification_agrees_across_streaming_and_non_streaming(): + """The streaming Delta coercion and the non-streaming from_dict parser must + classify the same tool-call dict identically, or a provider payload becomes a + custom tool call mid-stream and something else on the completed message.""" + from litellm.types.utils import ( + ChatCompletionDeltaCustomToolCall, + ChatCompletionMessageCustomToolCall, + Delta, + chat_completion_tool_call_from_dict, + ) + + tool_calls = [ + {"id": "c1", "type": "custom", "custom": {"name": "ApplyPatch", "input": ""}}, + {"id": "c2", "custom": {"name": "ApplyPatch", "input": "x"}}, + {"id": "c3", "type": "function", "function": {"name": "g", "arguments": "{}"}}, + ] + for tool_call in tool_calls: + message_parsed = chat_completion_tool_call_from_dict(dict(tool_call)) + delta_parsed = Delta(tool_calls=[dict(tool_call, index=0)]).tool_calls[0] + assert isinstance(message_parsed, ChatCompletionMessageCustomToolCall) == isinstance( + delta_parsed, ChatCompletionDeltaCustomToolCall + ) + + def test_message_with_mixed_function_and_custom_tool_calls(): from litellm.types.utils import ( ChatCompletionMessageCustomToolCall, From 075babd00fbc4ccba28506f0e700323b22440814 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 1 Aug 2026 16:22:05 -0700 Subject: [PATCH 20/22] chore(lint): clear the new LIT001/LIT002 violations and ratchet the lint budgets The type-discipline gate flagged 17 new mutable-collection annotations and 31 new mutable-collection constructions added by this branch. Replace raw dict literals with the OpenAI SDK's TypedDict call forms, annotate read-only params as Mapping/Sequence, precompute the custom tool call id set as a frozenset, and accumulate streamed arguments as tuples. The few places where a plain list/dict is a hard contract (pydantic response fields, fastapi route tags, parsed request bodies, in-place tool call patching) carry reasoned mutable-ok suppressions instead. Ratchet the ruff, type-discipline, and basedpyright budgets down by the violations this branch now fixes on net --- basedpyright-code-budget.json | 6 +- .../transformation.py | 123 +++++++++++------- litellm/integrations/helicone.py | 31 ++--- litellm/integrations/lunary.py | 19 +-- .../convert_dict_to_response.py | 7 +- .../prompt_templates/common_utils.py | 37 ++++-- .../streaming_chunk_builder_utils.py | 25 ++-- .../llms/openai/chat/gpt_transformation.py | 6 +- litellm/main.py | 2 +- .../proxy/response_api_endpoints/endpoints.py | 69 ++++++---- .../transformation.py | 14 +- litellm/types/utils.py | 25 ++-- ruff-strict-budget.json | 10 +- type-discipline-budget.json | 6 +- 14 files changed, 227 insertions(+), 153 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index f6dd90077b1..73b9d0c8192 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -42,7 +42,7 @@ "limit": 18 }, "reportIndexIssue": { - "limit": 37 + "limit": 36 }, "reportInvalidTypeForm": { "limit": 35 @@ -114,10 +114,10 @@ "limit": 31978 }, "reportUnnecessaryCast": { - "limit": 177 + "limit": 175 }, "reportUnnecessaryComparison": { - "limit": 1021 + "limit": 1019 }, "reportUnnecessaryContains": { "limit": 7 diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 768bf6c3e66..d999c9f4e60 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -4,6 +4,7 @@ Handler for transforming /chat/completions api requests to litellm.responses req import json import os +from collections.abc import Mapping from typing import ( TYPE_CHECKING, Any, @@ -21,6 +22,13 @@ from typing import ( ) from openai.types.responses.custom_tool_param import CustomToolParam +from openai.types.responses.response_input_param import ( + FunctionCallOutput, + ResponseCustomToolCallOutputParam, + ResponseCustomToolCallParam, +) +from openai.types.responses.tool_choice_custom_param import ToolChoiceCustomParam +from openai.types.responses.tool_choice_function_param import ToolChoiceFunctionParam from openai.types.responses.tool_param import FunctionToolParam from pydantic import BaseModel @@ -40,6 +48,8 @@ from litellm.responses.utils import normalize_responses_api_stream_options from litellm.types.llms.openai import ( ChatCompletionAnnotation, ChatCompletionReasoningItem, + ChatCompletionToolCallChunk, + ChatCompletionToolCallFunctionChunk, ChatCompletionToolParamFunctionChunk, Reasoning, ResponsesAPIOptionalRequestParams, @@ -101,7 +111,11 @@ def _build_reasoning_item( } -def _tool_call_dict_from_output_item(item: dict[str, Any]) -> dict[str, Any]: +class _ChatToolCallDict(ChatCompletionToolCallChunk, total=False): + provider_specific_fields: Mapping[str, Any] + + +def _tool_call_dict_from_output_item(item: Mapping[str, Any], index: int) -> _ChatToolCallDict: """Convert a ``function_call`` or ``custom_tool_call`` output item dict to a chat completions tool_call dict. Custom (grammar/freeform) tool calls carry their raw string payload in ``input`` rather than ``arguments``; both map to @@ -115,22 +129,32 @@ def _tool_call_dict_from_output_item(item: dict[str, Any]) -> dict[str, Any]: is_custom = item.get("type") == "custom_tool_call" arguments = (item.get("input") if is_custom else item.get("arguments")) or "" name = item.get("name") or ("custom_tool" if is_custom else "") - tool_call_dict: dict[str, Any] = { - "id": LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item(item.get("id"), item.get("call_id")), - "function": {"name": name, "arguments": arguments}, - "type": "function", - } - provider_specific_fields = item.get("provider_specific_fields") - if provider_specific_fields and not isinstance(provider_specific_fields, dict): - provider_specific_fields = ( - dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else None - ) + function_chunk = ChatCompletionToolCallFunctionChunk(name=name, arguments=arguments) + tool_call_dict = _ChatToolCallDict( + id=LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item(item.get("id"), item.get("call_id")), + type="function", + function=function_chunk, + index=index, + ) + raw_provider_fields = item.get("provider_specific_fields") + if isinstance(raw_provider_fields, dict): + provider_specific_fields = raw_provider_fields + elif raw_provider_fields and hasattr(raw_provider_fields, "__dict__"): + provider_specific_fields = vars(raw_provider_fields) + else: + provider_specific_fields = None if provider_specific_fields: tool_call_dict["provider_specific_fields"] = provider_specific_fields - tool_call_dict["function"]["provider_specific_fields"] = provider_specific_fields + function_chunk["provider_specific_fields"] = provider_specific_fields return tool_call_dict +def _flat_responses_tool_choice(choice_type: str, name: str) -> Union[ToolChoiceFunctionParam, ToolChoiceCustomParam]: + if choice_type == "custom": + return ToolChoiceCustomParam(type="custom", name=name) + return ToolChoiceFunctionParam(type="function", name=name) + + def _reasoning_item_to_response_input( r_item: Union[ChatCompletionReasoningItem, Dict[str, Any]], ) -> Dict[str, Any]: @@ -163,12 +187,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return tool_choice if isinstance(tool_choice.get("name"), str) and tool_choice.get("name"): # Return only Responses shape so stray chat ``function``/``custom`` keys are not sent upstream. - return {"type": choice_type, "name": tool_choice["name"]} + return _flat_responses_tool_choice(choice_type, tool_choice["name"]) nested = tool_choice.get(choice_type) if isinstance(nested, dict): nested_name = nested.get("name") if isinstance(nested_name, str) and nested_name: - return {"type": choice_type, "name": nested_name} + return _flat_responses_tool_choice(choice_type, nested_name) return tool_choice def _handle_raw_dict_response_item(self, item: Dict[str, Any], index: int) -> Tuple[Optional[Any], int]: @@ -221,7 +245,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) -> Tuple[List[Any], Optional[str]]: input_items: List[Any] = [] instructions: Optional[str] = None - custom_tool_call_ids: set = set() + custom_tool_call_ids = frozenset( + tool_call["id"] + for msg in messages + if msg.get("role") == "assistant" and isinstance(msg.get("tool_calls"), list) + for tool_call in msg.get("tool_calls") or () + if isinstance(tool_call, dict) + and not tool_call.get("function") + and isinstance(tool_call.get("custom"), dict) + ) for msg in messages: role = msg.get("role") @@ -269,19 +301,19 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): tool_output = [{"type": "input_text", "text": str(content)}] if tool_call_id in custom_tool_call_ids: input_items.append( - { - "type": "custom_tool_call_output", - "call_id": tool_call_id, - "output": content if isinstance(content, str) else tool_output, - } + ResponseCustomToolCallOutputParam( + type="custom_tool_call_output", + call_id=tool_call_id, + output=content if isinstance(content, str) else tool_output, + ) ) else: input_items.append( - { - "type": "function_call_output", - "call_id": tool_call_id, - "output": tool_output, - } + FunctionCallOutput( + type="function_call_output", + call_id=tool_call_id, + output=tool_output, + ) ) elif role == "assistant" and tool_calls and isinstance(tool_calls, list): for r_item in _get_reasoning_items(msg): @@ -300,14 +332,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): input_tool_call["arguments"] = function["arguments"] input_items.append(input_tool_call) elif isinstance(custom, dict): - custom_tool_call_ids.add(tool_call["id"]) input_items.append( - { - "type": "custom_tool_call", - "call_id": tool_call["id"], - "name": custom.get("name", ""), - "input": custom.get("input", ""), - } + ResponseCustomToolCallParam( + type="custom_tool_call", + call_id=tool_call["id"], + name=custom.get("name", ""), + input=custom.get("input", ""), + ) ) else: raise ValueError(f"tool call not supported: {tool_call}") @@ -598,7 +629,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): # Tool calls accumulate into the single trailing tool_calls choice # like the typed branches above; a choice per call would hide every # call after choices[0] from chat clients - accumulated_tool_calls.append(_tool_call_dict_from_output_item(raw_item)) + accumulated_tool_calls.append(_tool_call_dict_from_output_item(raw_item, tool_call_index)) tool_call_index += 1 elif handle_raw_dict_callback is not None: choice, index = handle_raw_dict_callback(item=raw_item, index=index) @@ -925,10 +956,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) custom_payload = tool["custom"] - flat_custom: CustomToolParam = { - "type": "custom", - "name": custom_payload.get("name", ""), - } + flat_custom = CustomToolParam(type="custom", name=custom_payload.get("name", "")) if custom_payload.get("description") is not None: flat_custom["description"] = custom_payload["description"] if isinstance(custom_payload.get("format"), dict): @@ -1130,7 +1158,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): def __init__(self, streaming_response, sync_stream: bool, json_mode: Optional[bool] = False): super().__init__(streaming_response, sync_stream, json_mode) self._chat_completion_id: str | None = None - self._tool_call_index_map: dict[int, int] = {} + self._tool_call_index_map: dict[int, int] = {} # mutable-ok: per-stream accumulator state def _handle_string_chunk( self, str_line: Union[str, "BaseModel"] @@ -1151,7 +1179,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): @staticmethod def _sequential_tool_call_index( - tool_call_index_map: dict[int, int] | None, + tool_call_index_map: dict[int, int] | None, # mutable-ok: per-stream state, remapped in place output_index: int, ) -> int: """Chat-completions tool_call indices must be 0-based and sequential, but @@ -1170,7 +1198,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): @staticmethod def translate_responses_chunk_to_openai_stream( parsed_chunk: Union[dict, BaseModel], - tool_call_index_map: dict[int, int] | None = None, + tool_call_index_map: dict[int, int] | None = None, # mutable-ok: per-stream state, remapped in place ) -> "ModelResponseStream": """ Translate a Responses API streaming chunk to OpenAI chat completion streaming format. @@ -1229,7 +1257,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): # New output item added output_item = parsed_chunk.get("item", {}) if output_item.get("type") in ("function_call", "custom_tool_call"): - converted = _tool_call_dict_from_output_item(output_item) + converted = _tool_call_dict_from_output_item(output_item, parsed_chunk.get("output_index", 0)) provider_specific_fields = converted.get("provider_specific_fields") function_chunk = ChatCompletionToolCallFunctionChunk( @@ -1299,16 +1327,15 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): # tool call; per-stream callers already received it via # output_item.added and the argument delta events return ModelResponseStream( - choices=[ + choices=[ # mutable-ok: ModelResponseStream coerces only list choices StreamingChoices( index=0, delta=Delta( - tool_calls=[ - { - **_tool_call_dict_from_output_item(dict(output_item)), - "index": parsed_chunk.get("output_index", 0), - } - ] + tool_calls=( + _tool_call_dict_from_output_item( + output_item, parsed_chunk.get("output_index", 0) + ), + ) ), finish_reason=None, ) diff --git a/litellm/integrations/helicone.py b/litellm/integrations/helicone.py index c9346f7e6cf..5d072ad873c 100644 --- a/litellm/integrations/helicone.py +++ b/litellm/integrations/helicone.py @@ -61,24 +61,19 @@ class HeliconeLogger: for tool_call in message["tool_calls"]: function = tool_call.get("function") custom = tool_call.get("custom") - if function: - content.append( - { - "type": "tool_use", - "id": tool_call["id"], - "name": function["name"], - "input": function["arguments"], - } - ) - elif custom: - content.append( - { - "type": "tool_use", - "id": tool_call["id"], - "name": custom["name"], - "input": custom["input"], - } - ) + if not function and not custom: + continue + name, tool_input = ( + (function["name"], function["arguments"]) if function else (custom["name"], custom["input"]) + ) + content.append( + { + "type": "tool_use", + "id": tool_call["id"], + "name": name, + "input": tool_input, + } + ) elif "content" in message and message["content"]: content = [{"type": "text", "text": message["content"]}] diff --git a/litellm/integrations/lunary.py b/litellm/integrations/lunary.py index 94cb5bab8fe..02a035bc445 100644 --- a/litellm/integrations/lunary.py +++ b/litellm/integrations/lunary.py @@ -22,25 +22,18 @@ def parse_tool_calls(tool_calls): def clean_tool_call(tool_call): custom = getattr(tool_call, "custom", None) if custom is not None: - return { - "type": tool_call.type, - "id": tool_call.id, - "function": { - "name": custom.name, - "arguments": custom.input, - }, - } - serialized = { + name, arguments = custom.name, custom.input + else: + name, arguments = tool_call.function.name, tool_call.function.arguments + return { "type": tool_call.type, "id": tool_call.id, "function": { - "name": tool_call.function.name, - "arguments": tool_call.function.arguments, + "name": name, + "arguments": arguments, }, } - return serialized - return [ clean_tool_call(tool_call) for tool_call in tool_calls diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 1b23db87264..cf3937072c2 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -3,6 +3,7 @@ import json import re import time import traceback +from collections.abc import Sequence from typing import Dict, Iterable, List, Literal, Optional, Tuple, Union, cast import litellm @@ -371,7 +372,9 @@ from collections import defaultdict def _handle_invalid_parallel_tool_calls( - tool_calls: List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]], + tool_calls: List[ + Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall] + ], # mutable-ok: patched in place via slice assignment ): """ Handle hallucinated parallel tool call from openai - https://community.openai.com/t/model-tries-to-call-unknown-function-multi-tool-use-parallel/490653 @@ -532,7 +535,7 @@ class LiteLLMResponseObjectHandler: def _should_convert_tool_call_to_json_mode( tool_calls: ( - list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | list[DatabricksTool] | None + Sequence[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | Sequence[DatabricksTool] | None ) = None, convert_tool_call_to_json_mode: Optional[bool] = None, ) -> bool: diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index 3a7a710c6a9..52974d42b96 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -21,6 +21,14 @@ from typing import ( cast, ) +from openai.types.chat.chat_completion_custom_tool_param import ( + CustomFormatGrammar, + CustomFormatGrammarGrammar, +) +from openai.types.shared_params.custom_tool_input_format import ( + Grammar as ResponsesGrammarFormat, +) + import litellm from litellm import verbose_logger from litellm.router_utils.batch_utils import InMemoryFile @@ -1252,29 +1260,36 @@ def is_function_call(optional_params: dict) -> bool: return False -def convert_custom_tool_format_to_chat_shape(format_obj: dict) -> dict: +def convert_custom_tool_format_to_chat_shape(format_obj: Mapping[str, Any]) -> Mapping[str, Any]: """ Responses API grammar formats are flat ({"type": "grammar", "definition", "syntax"}); Chat Completions wraps the same fields in a "grammar" object. Text formats are identical on both surfaces and pass through, as does anything unrecognized. """ - if format_obj.get("type") == "grammar" and "grammar" not in format_obj: - return { - "type": "grammar", - "grammar": {k: format_obj[k] for k in ("definition", "syntax") if k in format_obj}, - } - return format_obj + if format_obj.get("type") != "grammar" or "grammar" in format_obj: + return format_obj + grammar = CustomFormatGrammarGrammar() + if "definition" in format_obj: + grammar["definition"] = format_obj["definition"] + if "syntax" in format_obj: + grammar["syntax"] = format_obj["syntax"] + return CustomFormatGrammar(type="grammar", grammar=grammar) -def convert_custom_tool_format_to_responses_shape(format_obj: dict) -> dict: +def convert_custom_tool_format_to_responses_shape(format_obj: Mapping[str, Any]) -> Mapping[str, Any]: """ Inverse of convert_custom_tool_format_to_chat_shape: unwrap the Chat Completions "grammar" object into the flat Responses API grammar shape. """ grammar = format_obj.get("grammar") - if format_obj.get("type") == "grammar" and isinstance(grammar, dict): - return {"type": "grammar", **{k: grammar[k] for k in ("definition", "syntax") if k in grammar}} - return format_obj + if format_obj.get("type") != "grammar" or not isinstance(grammar, dict): + return format_obj + flat = ResponsesGrammarFormat(type="grammar") + if "definition" in grammar: + flat["definition"] = grammar["definition"] + if "syntax" in grammar: + flat["syntax"] = grammar["syntax"] + return flat def get_file_ids_from_messages(messages: List[AllMessageValues]) -> List[str]: diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index f4f1b6fca0d..6d013718668 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -1,5 +1,6 @@ import base64 import time +from collections.abc import Mapping, Sequence from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union, cast from litellm.types.llms.openai import ( @@ -205,9 +206,13 @@ class ChunkProcessor: return response def get_combined_tool_content( - self, tool_call_chunks: List[Dict[str, Any]] - ) -> List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]]: - tool_calls_list: List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] = [] + self, tool_call_chunks: Sequence[Mapping[str, Any]] + ) -> List[ + Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall] + ]: # mutable-ok: assigned verbatim to Message.tool_calls, a List field + tool_calls_list: List[ + Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall] + ] = [] # mutable-ok: see return type tool_call_map: Dict[int, Dict[str, Any]] = {} # Map to store tool calls by index for chunk in tool_call_chunks: @@ -245,9 +250,9 @@ class ChunkProcessor: "id": None, "name": None, "type": None, - "arguments": [], + "arguments": (), "custom_name": None, - "custom_input": [], + "custom_input": (), "provider_specific_fields": None, } @@ -263,20 +268,20 @@ class ChunkProcessor: if function.get("name"): tool_call_map[index]["name"] = function["name"] if function.get("arguments"): - tool_call_map[index]["arguments"].append(function["arguments"]) + tool_call_map[index]["arguments"] += (function["arguments"],) else: # function is an object if hasattr(function, "name") and function.name: tool_call_map[index]["name"] = function.name if hasattr(function, "arguments") and function.arguments: - tool_call_map[index]["arguments"].append(function.arguments) + tool_call_map[index]["arguments"] += (function.arguments,) custom = tool_call.get("custom") if isinstance(custom, dict): if custom.get("name"): tool_call_map[index]["custom_name"] = custom["name"] if custom.get("input"): - tool_call_map[index]["custom_input"].append(custom["input"]) + tool_call_map[index]["custom_input"] += (custom["input"],) else: # tool_call is an object if hasattr(tool_call, "id") and tool_call.id: @@ -287,14 +292,14 @@ class ChunkProcessor: if hasattr(tool_call.function, "name") and tool_call.function.name: tool_call_map[index]["name"] = tool_call.function.name if hasattr(tool_call.function, "arguments") and tool_call.function.arguments: - tool_call_map[index]["arguments"].append(tool_call.function.arguments) + tool_call_map[index]["arguments"] += (tool_call.function.arguments,) custom = getattr(tool_call, "custom", None) if custom is not None: if getattr(custom, "name", None): tool_call_map[index]["custom_name"] = custom.name if getattr(custom, "input", None): - tool_call_map[index]["custom_input"].append(custom.input) + tool_call_map[index]["custom_input"] += (custom.input,) # Preserve provider_specific_fields from streaming chunks provider_fields = None diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index e4492a8aba6..b6c8b7f2a07 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -533,12 +533,14 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): for choice in choices: ## HANDLE JSON MODE - anthropic returns single function call] tool_calls = choice["message"].get("tool_calls", None) - new_tool_calls: list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | None = None + new_tool_calls: list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | None = ( + None # mutable-ok: holds _handle_invalid_parallel_tool_calls' list; Message.__init__ expects list + ) message_content = choice["message"].get("content", None) if tool_calls is not None: _openai_tool_calls = [] for _tc in tool_calls: - _openai_tc = chat_completion_tool_call_from_dict(dict(_tc)) + _openai_tc = chat_completion_tool_call_from_dict(_tc) _openai_tool_calls.append(_openai_tc) fixed_tool_calls = _handle_invalid_parallel_tool_calls(_openai_tool_calls) diff --git a/litellm/main.py b/litellm/main.py index fbb43dd41fa..cadf65c3e50 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1058,7 +1058,7 @@ def responses_api_bridge_check( # summary alias is present with ``reasoning_effort`` (tools alone stay on chat). has_function_tool = any( (tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function") - for tool in (tools or []) + for tool in (tools or ()) ) if isinstance(reasoning_effort, dict): reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 3a2c57aa2ba..2068de17785 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -1,6 +1,8 @@ import asyncio import json import time +from collections.abc import Mapping +from types import MappingProxyType from typing import Any, AsyncIterator, Dict, Optional, cast from uuid import uuid4 @@ -23,19 +25,30 @@ from litellm.types.responses.main import DeleteResponseResult router = APIRouter() _user_api_key_auth_dep = Depends(user_api_key_auth) +_RESPONSES_TAGS = ["responses"] # mutable-ok: fastapi's route signature requires List[str] tags -_TOOL_PAYLOAD_KEYS = { - "custom": ("name", "description", "format"), - "function": ("name", "description", "parameters", "strict"), -} +_TOOL_PAYLOAD_KEYS: Mapping[str, tuple[str, ...]] = MappingProxyType( + { + "custom": ("name", "description", "format"), + "function": ("name", "description", "parameters", "strict"), + } +) +_EMPTY_TOOL_PAYLOAD: Mapping[str, Any] = MappingProxyType({}) -def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object: +def _convert_tool_payload_value(key: str, value: object, *, to_chat: bool) -> object: + if key != "format" or not isinstance(value, dict): + return value from litellm.litellm_core_utils.prompt_templates.common_utils import ( convert_custom_tool_format_to_chat_shape, convert_custom_tool_format_to_responses_shape, ) + convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape + return convert(value) + + +def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object: if not isinstance(obj, dict): return obj tool_type = obj.get("type") @@ -43,35 +56,37 @@ def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object: if payload_keys is None: return obj nested = obj.get(tool_type) - nested_source = nested if isinstance(nested, dict) else {} - payload = { - key: nested_source[key] if key in nested_source else obj[key] + nested_source = nested if isinstance(nested, dict) else _EMPTY_TOOL_PAYLOAD + payload = { # mutable-ok: tool entries are embedded verbatim in the JSON request body + key: _convert_tool_payload_value(key, nested_source[key] if key in nested_source else obj[key], to_chat=to_chat) for key in payload_keys if key in nested_source or key in obj } if "name" not in payload: return obj - if isinstance(payload.get("format"), dict): - convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape - payload = {**payload, "format": convert(payload["format"])} - return {"type": tool_type, tool_type: payload} if to_chat else {"type": tool_type, **payload} + return {"type": tool_type, tool_type: payload} if to_chat else {"type": tool_type, **payload} # mutable-ok: same -def _normalize_tool_dialect(data: dict, *, to_chat: bool) -> dict: - converted: dict = {} +def _normalize_tool_dialect( + data: dict, *, to_chat: bool +) -> dict: # mutable-ok: the parsed request body contract is a plain dict tools = data.get("tools") - if isinstance(tools, list): - normalized_tools = [_convert_tool_envelope(tool, to_chat=to_chat) for tool in tools] - if normalized_tools != tools: - converted["tools"] = normalized_tools tool_choice = data.get("tool_choice") + normalized_tools = ( + [ + _convert_tool_envelope(tool, to_chat=to_chat) for tool in tools + ] # mutable-ok: body's tools stays a plain JSON list + if isinstance(tools, list) + else tools + ) normalized_choice = _convert_tool_envelope(tool_choice, to_chat=to_chat) - if normalized_choice != tool_choice: - converted["tool_choice"] = normalized_choice - return {**data, **converted} if converted else data + if normalized_tools == tools and normalized_choice == tool_choice: + return data + replaceable = (("tools", normalized_tools), ("tool_choice", normalized_choice)) + return {**data, **{key: value for key, value in replaceable if key in data}} # mutable-ok: plain body dict -def _is_chat_completions_body(data: dict) -> bool: +def _is_chat_completions_body(data: Mapping[str, Any]) -> bool: messages = data.get("messages") if isinstance(messages, list) and len(messages) > 0: return True @@ -340,13 +355,13 @@ async def responses_api( @router.get( "/cursor/models", - dependencies=[Depends(user_api_key_auth)], - tags=["responses"], + dependencies=(_user_api_key_auth_dep,), + tags=_RESPONSES_TAGS, ) @router.get( "/cursor/v1/models", - dependencies=[Depends(user_api_key_auth)], - tags=["responses"], + dependencies=(_user_api_key_auth_dep,), + tags=_RESPONSES_TAGS, ) async def cursor_model_list( user_api_key_dict: UserAPIKeyAuth = _user_api_key_auth_dep, @@ -447,7 +462,7 @@ async def cursor_chat_completions( # Rebuild rather than pop: _read_request_body can return the request-scope # cached parsed-body dict itself, and removing keys from it corrupts the # cache's key snapshot so later readers get an empty body - data = {key: value for key, value in data.items() if key != "stream_options"} + data = {key: value for key, value in data.items() if key != "stream_options"} # mutable-ok: plain body dict data = _normalize_tool_dialect(data, to_chat=False) diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 176274d236f..090723edb85 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -7,6 +7,12 @@ import re from collections.abc import Sequence from typing import Any, Literal, cast +from openai.types.chat.chat_completion_named_tool_choice_param import ( + ChatCompletionNamedToolChoiceParam, +) +from openai.types.chat.chat_completion_named_tool_choice_param import ( + Function as NamedToolChoiceFunction, +) from openai.types.responses import ResponseFunctionToolCall from openai.types.responses.response_create_params import ResponseInputParam from openai.types.responses.tool_param import FunctionToolParam @@ -160,13 +166,17 @@ class LiteLLMCompletionResponsesConfig: elif tool_choice_type == "function": function_name = tool_choice.get("name") if function_name: - return {"type": "function", "function": {"name": function_name}} + return ChatCompletionNamedToolChoiceParam( + type="function", function=NamedToolChoiceFunction(name=function_name) + ) return "required" elif tool_choice_type == "custom": custom = tool_choice.get("custom") custom_name = tool_choice.get("name") or (custom.get("name") if isinstance(custom, dict) else None) if custom_name: - return {"type": "function", "function": {"name": custom_name}} + return ChatCompletionNamedToolChoiceParam( + type="function", function=NamedToolChoiceFunction(name=custom_name) + ) return "required" # Return as-is for unknown formats diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 404725ec61b..6d051b70432 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1,6 +1,7 @@ import json import time from enum import Enum +from types import MappingProxyType from typing import ( TYPE_CHECKING, Any, @@ -1162,16 +1163,16 @@ class ChatCompletionMessageToolCall(OpenAIObject): setattr(self, key, value) -def is_custom_tool_call_dict(tool_call: dict) -> bool: +def is_custom_tool_call_dict(tool_call: Mapping[str, Any]) -> bool: return tool_call.get("type") == "custom" or tool_call.get("custom") is not None def chat_completion_tool_call_from_dict( - tool_call: dict, + tool_call: Mapping[str, Any], ) -> "ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall": if is_custom_tool_call_dict(tool_call): return ChatCompletionMessageCustomToolCall( - **{k: v for k, v in tool_call.items() if not (k in ("function", "type") and v is None)} + **MappingProxyType({k: v for k, v in tool_call.items() if not (k in ("function", "type") and v is None)}) ) return ChatCompletionMessageToolCall(**tool_call) @@ -1228,7 +1229,9 @@ def add_provider_specific_fields(object: BaseModel, provider_specific_fields: Op class Message(SafeAttributeModel, OpenAIObject): content: Optional[str] role: Literal["assistant", "user", "system", "tool", "function"] - tool_calls: Optional[List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]]] + tool_calls: Optional[ + List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] + ] # mutable-ok: public pydantic response field; only the union member is new function_call: Optional[FunctionCall] audio: Optional[ChatCompletionAudioResponse] = None images: Optional[List[ImageURLListItem]] = None @@ -1352,7 +1355,9 @@ class Delta(SafeAttributeModel, OpenAIObject): content: Optional[str] role: Optional[str] function_call: Optional[FunctionCall] - tool_calls: Optional[List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]]] + tool_calls: Optional[ + List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]] + ] # mutable-ok: public pydantic response field; only the union member is new audio: Optional[ChatCompletionAudioResponse] images: Optional[List[ImageURLListItem]] annotations: Optional[List[ChatCompletionAnnotation]] @@ -1389,8 +1394,10 @@ class Delta(SafeAttributeModel, OpenAIObject): if function_call is not None and isinstance(function_call, dict): function_call = FunctionCall(**function_call) - if tool_calls is not None and isinstance(tool_calls, list): - coerced_tool_calls: List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]] = [] + if tool_calls is not None and isinstance(tool_calls, (list, tuple)): + coerced_tool_calls: List[ + Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall] + ] = [] # mutable-ok: public Delta.tool_calls contract is a list current_index = 0 for tool_call in tool_calls: if isinstance(tool_call, dict): @@ -1400,7 +1407,9 @@ class Delta(SafeAttributeModel, OpenAIObject): if is_custom_tool_call_dict(tool_call): coerced_tool_calls.append( ChatCompletionDeltaCustomToolCall( - **{k: v for k, v in tool_call.items() if not (k == "function" and v is None)} + **MappingProxyType( + {k: v for k, v in tool_call.items() if not (k == "function" and v is None)} + ) ) ) else: diff --git a/ruff-strict-budget.json b/ruff-strict-budget.json index b8650eea7aa..4c8132ff859 100644 --- a/ruff-strict-budget.json +++ b/ruff-strict-budget.json @@ -135,7 +135,7 @@ "limit": 30 }, "PERF401": { - "limit": 142 + "limit": 141 }, "PERF402": { "limit": 9 @@ -222,7 +222,7 @@ "limit": 38 }, "RET504": { - "limit": 702 + "limit": 701 }, "RUF010": { "limit": 874 @@ -267,7 +267,7 @@ "limit": 324 }, "SIM103": { - "limit": 129 + "limit": 128 }, "SIM113": { "limit": 6 @@ -324,7 +324,7 @@ "limit": 879 }, "UP006": { - "limit": 12050 + "limit": 12045 }, "UP007": { "limit": 2526 @@ -363,6 +363,6 @@ "limit": 104 }, "UP045": { - "limit": 17793 + "limit": 17791 } } diff --git a/type-discipline-budget.json b/type-discipline-budget.json index ff037a2872e..bc28630a4e5 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -1,9 +1,9 @@ { "LIT001": { - "limit": 23191 + "limit": 23180 }, "LIT002": { - "limit": 27276 + "limit": 27259 }, "LIT003": { "limit": 292 @@ -24,6 +24,6 @@ "limit": 1004 }, "LIT009": { - "limit": 2467 + "limit": 2465 } } From eea9bb1497d69ed16fcc022a51bc81d6048da735 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:13:15 -0700 Subject: [PATCH 21/22] chore(lint): use a PEP 604 union for the flat tool_choice helper (UP007) --- .../litellm_responses_transformation/transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 1cc1e88f1fa..79a4620dc80 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -141,7 +141,7 @@ def _tool_call_dict_from_output_item(item: Mapping[str, Any], index: int) -> _Ch return tool_call_dict -def _flat_responses_tool_choice(choice_type: str, name: str) -> Union[ToolChoiceFunctionParam, ToolChoiceCustomParam]: +def _flat_responses_tool_choice(choice_type: str, name: str) -> ToolChoiceFunctionParam | ToolChoiceCustomParam: if choice_type == "custom": return ToolChoiceCustomParam(type="custom", name=name) return ToolChoiceFunctionParam(type="function", name=name) From f25f1d292164487941290f3038d4489b8d80078c Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:42:03 -0700 Subject: [PATCH 22/22] feat(proxy): resolve Cursor thinking/fast model-name suffixes on /cursor/chat/completions Cursor appends -thinking- and -fast to custom model names when the user picks a thinking level or fast mode, so a model configured as claude-opus-5 arrives as claude-opus-5-thinking-xhigh-fast and fails routing with no healthy deployments. When the raw name is not servable by the router but the suffix-stripped base name is, rewrite the body to the base model and carry the thinking level into reasoning_effort (chat bodies) or reasoning.effort (Responses bodies), never clobbering an effort the client already sent. Explicitly configured aliases keep winning because the raw-name servability check runs first. --- .../proxy/response_api_endpoints/endpoints.py | 64 ++++- .../response_api_endpoints/test_endpoints.py | 228 ++++++++++++++++++ 2 files changed, 288 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 7dcb01d3e59..6b0b4db1b18 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -3,7 +3,7 @@ import json import time from collections.abc import AsyncIterator, Mapping from types import MappingProxyType -from typing import Any, cast +from typing import TYPE_CHECKING, Any, NamedTuple, cast, get_args from uuid import uuid4 import fastapi @@ -19,9 +19,12 @@ from litellm.proxy.auth.user_api_key_auth import ( user_api_key_auth_websocket, ) from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing -from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse +from litellm.types.llms.openai import REASONING_EFFORT, ResponseAPIUsage, ResponsesAPIResponse from litellm.types.responses.main import DeleteResponseResult +if TYPE_CHECKING: + from litellm.router import Router + router = APIRouter() _user_api_key_auth_dep = Depends(user_api_key_auth) @@ -93,6 +96,58 @@ def _is_chat_completions_body(data: Mapping[str, Any]) -> bool: return "messages" in data and "input" not in data +_CURSOR_THINKING_SEPARATOR = "-thinking-" +_CURSOR_FAST_SUFFIX = "-fast" +_CURSOR_THINKING_LEVELS: frozenset[str] = frozenset(get_args(REASONING_EFFORT)) + + +class _CursorModelVariant(NamedTuple): + base_model: str + reasoning_effort: str | None + + +def _parse_cursor_model_variant(model: str) -> _CursorModelVariant: + stripped = model.removesuffix(_CURSOR_FAST_SUFFIX) + base, separator, level = stripped.rpartition(_CURSOR_THINKING_SEPARATOR) + if separator and base and level in _CURSOR_THINKING_LEVELS: + return _CursorModelVariant(base, level) + return _CursorModelVariant(stripped, None) + + +def _router_can_serve(model: str, llm_router: "Router | None") -> bool: + if llm_router is None: + return False + if model in llm_router.model_names or model in llm_router.model_group_alias: + return True + if model in llm_router.team_public_model_names: + return True + return bool(llm_router.pattern_router.get_pattern(model)) + + +def _resolve_cursor_model_variant( + data: dict, llm_router: "Router | None" +) -> dict: # mutable-ok: the parsed request body contract is a plain dict + model = data.get("model") + if not isinstance(model, str) or _router_can_serve(model, llm_router): + return data + variant = _parse_cursor_model_variant(model) + if variant.base_model == model or not _router_can_serve(variant.base_model, llm_router): + return data + resolved = {**data, "model": variant.base_model} # mutable-ok: plain body dict + if variant.reasoning_effort is None: + return resolved + if _is_chat_completions_body(data): + if "reasoning_effort" in data: + return resolved + return {**resolved, "reasoning_effort": variant.reasoning_effort} # mutable-ok: plain body dict + reasoning = data.get("reasoning") + if isinstance(reasoning, dict): + if reasoning.get("effort"): + return resolved + return {**resolved, "reasoning": {**reasoning, "effort": variant.reasoning_effort}} # mutable-ok: same + return {**resolved, "reasoning": {"effort": variant.reasoning_effort}} # mutable-ok: plain body dict + + @router.post( "/v1/responses", dependencies=[Depends(user_api_key_auth)], @@ -440,7 +495,8 @@ async def cursor_chat_completions( from litellm.types.llms.openai import ResponsesAPIResponse from litellm.types.utils import ModelResponse - data = await _read_request_body(request=request) + raw_body = await _read_request_body(request=request) + data = _resolve_cursor_model_variant(raw_body, llm_router) if _is_chat_completions_body(data): # Genuine chat completions body (Cursor sends these for models whose BYOK it @@ -448,7 +504,7 @@ async def cursor_chat_completions( # Keyed on messages CONTENT, not key presence: Cursor can send a null or # empty messages stub alongside a real agent-mode input array normalized = _normalize_tool_dialect(data, to_chat=True) - if normalized is not data: + if normalized is not raw_body: _safe_set_request_parsed_body(request=request, parsed_body=normalized) return await chat_completion( request=request, diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 00ac8ca386a..60168e7f912 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1340,3 +1340,231 @@ class TestChatCompletionsBodyDetection: assert response.status_code == 200 assert mock_router.aresponses.call_args is not None assert mock_router.aresponses.call_args.kwargs["input"] == [{"role": "user", "content": "hello"}] + + +class TestParseCursorModelVariant: + @pytest.mark.parametrize( + "model,expected_base,expected_effort", + [ + ("claude-opus-5-thinking-high", "claude-opus-5", "high"), + ("claude-opus-5-thinking-xhigh-fast", "claude-opus-5", "xhigh"), + ("gemini-3.0-pro-thinking-low", "gemini-3.0-pro", "low"), + ("claude-opus-5-fast", "claude-opus-5", None), + ("gpt-5.6-sol", "gpt-5.6-sol", None), + ("foo-thinking-ultra-fast", "foo-thinking-ultra", None), + ("-thinking-high", "-thinking-high", None), + ], + ) + def test_parse_matrix(self, model, expected_base, expected_effort): + from litellm.proxy.response_api_endpoints.endpoints import _parse_cursor_model_variant + + variant = _parse_cursor_model_variant(model) + assert variant.base_model == expected_base + assert variant.reasoning_effort == expected_effort + + +class TestResolveCursorModelVariant: + @pytest.fixture(scope="class") + def wildcard_router(self): + from litellm import Router + + return Router( + model_list=[ + {"model_name": "anthropic/*", "litellm_params": {"model": "anthropic/*", "api_key": "fake"}}, + {"model_name": "openai/*", "litellm_params": {"model": "openai/*", "api_key": "fake"}}, + { + "model_name": "explicit-alias-thinking-high", + "litellm_params": {"model": "anthropic/claude-opus-5", "api_key": "fake"}, + }, + ] + ) + + def test_chat_body_suffix_stripped_into_reasoning_effort(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = { + "model": "claude-opus-5-thinking-xhigh-fast", + "messages": [{"role": "user", "content": "hi"}], + } + resolved = _resolve_cursor_model_variant(body, wildcard_router) + assert resolved["model"] == "claude-opus-5" + assert resolved["reasoning_effort"] == "xhigh" + assert resolved["messages"] == body["messages"] + assert body["model"] == "claude-opus-5-thinking-xhigh-fast" + + def test_responses_body_suffix_stripped_into_reasoning_dict(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "claude-opus-5-thinking-high", "input": [{"role": "user", "content": "hi"}]} + resolved = _resolve_cursor_model_variant(body, wildcard_router) + assert resolved["model"] == "claude-opus-5" + assert resolved["reasoning"] == {"effort": "high"} + + def test_responses_body_merges_effort_into_existing_reasoning(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = { + "model": "claude-opus-5-thinking-high", + "input": [{"role": "user", "content": "hi"}], + "reasoning": {"summary": "auto"}, + } + resolved = _resolve_cursor_model_variant(body, wildcard_router) + assert resolved["model"] == "claude-opus-5" + assert resolved["reasoning"] == {"summary": "auto", "effort": "high"} + + def test_existing_reasoning_effort_wins_but_model_still_rewritten(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + chat_body = { + "model": "claude-opus-5-thinking-high", + "messages": [{"role": "user", "content": "hi"}], + "reasoning_effort": "low", + } + resolved_chat = _resolve_cursor_model_variant(chat_body, wildcard_router) + assert resolved_chat["model"] == "claude-opus-5" + assert resolved_chat["reasoning_effort"] == "low" + + responses_body = { + "model": "claude-opus-5-thinking-high", + "input": [{"role": "user", "content": "hi"}], + "reasoning": {"effort": "low"}, + } + resolved_responses = _resolve_cursor_model_variant(responses_body, wildcard_router) + assert resolved_responses["model"] == "claude-opus-5" + assert resolved_responses["reasoning"] == {"effort": "low"} + + def test_fast_only_suffix_strips_without_reasoning(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "claude-opus-5-fast", "messages": [{"role": "user", "content": "hi"}]} + resolved = _resolve_cursor_model_variant(body, wildcard_router) + assert resolved["model"] == "claude-opus-5" + assert "reasoning_effort" not in resolved + + def test_explicitly_configured_suffixed_name_untouched(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "explicit-alias-thinking-high", "messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(body, wildcard_router) is body + + def test_provider_inferable_bare_name_untouched(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "gpt-4o-mini", "messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(body, wildcard_router) is body + + def test_unservable_base_untouched(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "totally-unknown-thinking-high", "messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(body, wildcard_router) is body + + def test_no_router_untouched(self): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + body = {"model": "claude-opus-5-thinking-high", "messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(body, None) is body + + def test_missing_or_non_string_model_untouched(self, wildcard_router): + from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant + + no_model = {"messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(no_model, wildcard_router) is no_model + null_model = {"model": None, "messages": [{"role": "user", "content": "hi"}]} + assert _resolve_cursor_model_variant(null_model, wildcard_router) is null_model + + +def _router_serving_only(base_model: str) -> MagicMock: + mock_router = MagicMock() + mock_router.model_names = set() + mock_router.model_group_alias = {} + mock_router.team_public_model_names = frozenset() + mock_router.pattern_router.get_pattern.side_effect = ( + lambda model: [{"model_name": "anthropic/*"}] if model == base_model else None + ) + return mock_router + + +class TestCursorModelSuffixResolutionEndToEnd: + @pytest.mark.asyncio + async def test_chat_arm_rewrites_suffixed_model_before_delegation(self): + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + + seen = {} + + async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict): + from litellm.proxy.common_utils.http_parsing_utils import _read_request_body + + seen["body"] = await _read_request_body(request=request) + return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []} + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with ( + patch("litellm.proxy.proxy_server.llm_router", new=_router_serving_only("claude-opus-5")), + patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion), + ): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "claude-opus-5-thinking-xhigh-fast", + "messages": [{"role": "user", "content": "hi"}], + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert seen["body"]["model"] == "claude-opus-5" + assert seen["body"]["reasoning_effort"] == "xhigh" + assert seen["body"]["messages"] == [{"role": "user", "content": "hi"}] + + @pytest.mark.asyncio + async def test_responses_arm_rewrites_suffixed_model_before_routing(self): + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.types.llms.openai import ResponsesAPIResponse + + mock_response = ResponsesAPIResponse( + id="resp_suffix1", + created_at=1234567890, + model="claude-opus-5", + object="response", + output=[ + ResponseOutputMessage( + id="msg_suffix1", + type="message", + role="assistant", + status="completed", + content=[ResponseOutputText(type="output_text", text="ok", annotations=[])], + ) + ], + ) + + mock_router = _router_serving_only("claude-opus-5") + mock_router.aresponses = AsyncMock(return_value=mock_response) + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.llm_router", new=mock_router): + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "claude-opus-5-thinking-high", + "input": [{"role": "user", "content": "hello"}], + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert mock_router.aresponses.call_args is not None + assert mock_router.aresponses.call_args.kwargs["model"] == "claude-opus-5" + assert mock_router.aresponses.call_args.kwargs["reasoning"] == {"effort": "high"}