mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #34029 from BerriAI/litellm_lit4395_cursor_agent
fix(proxy): make /cursor/chat/completions work with Cursor agent mode
This commit is contained in:
commit
cbeaf86c8d
29 changed files with 3013 additions and 203 deletions
|
|
@ -1,6 +1,6 @@
|
|||
{
|
||||
"reportAny": {
|
||||
"limit": 29682
|
||||
"limit": 29813
|
||||
},
|
||||
"reportArgumentType": {
|
||||
"limit": 2645
|
||||
|
|
@ -21,10 +21,10 @@
|
|||
"limit": 325
|
||||
},
|
||||
"reportDuplicateImport": {
|
||||
"limit": 42
|
||||
"limit": 38
|
||||
},
|
||||
"reportExplicitAny": {
|
||||
"limit": 9440
|
||||
"limit": 9473
|
||||
},
|
||||
"reportFunctionMemberAccess": {
|
||||
"limit": 11
|
||||
|
|
@ -42,7 +42,7 @@
|
|||
"limit": 18
|
||||
},
|
||||
"reportIndexIssue": {
|
||||
"limit": 37
|
||||
"limit": 35
|
||||
},
|
||||
"reportInvalidTypeForm": {
|
||||
"limit": 35
|
||||
|
|
@ -54,10 +54,10 @@
|
|||
"limit": 0
|
||||
},
|
||||
"reportMissingParameterType": {
|
||||
"limit": 5848
|
||||
"limit": 5855
|
||||
},
|
||||
"reportMissingTypeArgument": {
|
||||
"limit": 15850
|
||||
"limit": 15849
|
||||
},
|
||||
"reportMissingTypeStubs": {
|
||||
"limit": 41
|
||||
|
|
@ -84,7 +84,7 @@
|
|||
"limit": 77
|
||||
},
|
||||
"reportPrivateUsage": {
|
||||
"limit": 2437
|
||||
"limit": 2436
|
||||
},
|
||||
"reportRedeclaration": {
|
||||
"limit": 12
|
||||
|
|
@ -99,31 +99,31 @@
|
|||
"limit": 0
|
||||
},
|
||||
"reportUnknownArgumentType": {
|
||||
"limit": 45297
|
||||
"limit": 45269
|
||||
},
|
||||
"reportUnknownLambdaType": {
|
||||
"limit": 113
|
||||
},
|
||||
"reportUnknownMemberType": {
|
||||
"limit": 40411
|
||||
"limit": 40452
|
||||
},
|
||||
"reportUnknownParameterType": {
|
||||
"limit": 20301
|
||||
"limit": 20309
|
||||
},
|
||||
"reportUnknownVariableType": {
|
||||
"limit": 31968
|
||||
"limit": 31978
|
||||
},
|
||||
"reportUnnecessaryCast": {
|
||||
"limit": 177
|
||||
"limit": 173
|
||||
},
|
||||
"reportUnnecessaryComparison": {
|
||||
"limit": 1021
|
||||
"limit": 1017
|
||||
},
|
||||
"reportUnnecessaryContains": {
|
||||
"limit": 7
|
||||
},
|
||||
"reportUnnecessaryIsInstance": {
|
||||
"limit": 1204
|
||||
"limit": 1203
|
||||
},
|
||||
"reportUntypedBaseClass": {
|
||||
"limit": 165
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ Handler for transforming /chat/completions api requests to litellm.responses req
|
|||
|
||||
import json
|
||||
import os
|
||||
from collections.abc import AsyncIterator, Callable, Iterable, Iterator
|
||||
from collections.abc import AsyncIterator, Callable, Iterable, Iterator, Mapping
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
|
|
@ -13,6 +13,14 @@ from typing import (
|
|||
cast,
|
||||
)
|
||||
|
||||
from openai.types.responses.custom_tool_param import CustomToolParam
|
||||
from openai.types.responses.response_input_param import (
|
||||
FunctionCallOutput,
|
||||
ResponseCustomToolCallOutputParam,
|
||||
ResponseCustomToolCallParam,
|
||||
)
|
||||
from openai.types.responses.tool_choice_custom_param import ToolChoiceCustomParam
|
||||
from openai.types.responses.tool_choice_function_param import ToolChoiceFunctionParam
|
||||
from openai.types.responses.tool_param import FunctionToolParam
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
|
@ -32,6 +40,8 @@ from litellm.responses.utils import normalize_responses_api_stream_options
|
|||
from litellm.types.llms.openai import (
|
||||
ChatCompletionAnnotation,
|
||||
ChatCompletionReasoningItem,
|
||||
ChatCompletionToolCallChunk,
|
||||
ChatCompletionToolCallFunctionChunk,
|
||||
ChatCompletionToolParamFunctionChunk,
|
||||
Reasoning,
|
||||
ResponsesAPIOptionalRequestParams,
|
||||
|
|
@ -93,6 +103,50 @@ def _build_reasoning_item(
|
|||
}
|
||||
|
||||
|
||||
class _ChatToolCallDict(ChatCompletionToolCallChunk, total=False):
|
||||
provider_specific_fields: Mapping[str, Any]
|
||||
|
||||
|
||||
def _tool_call_dict_from_output_item(item: Mapping[str, Any], index: int) -> _ChatToolCallDict:
|
||||
"""Convert a ``function_call`` or ``custom_tool_call`` output item dict to a chat
|
||||
completions tool_call dict. Custom (grammar/freeform) tool calls carry their raw
|
||||
string payload in ``input`` rather than ``arguments``; both map to
|
||||
``function.arguments`` so chat clients (e.g. Cursor agent mode) receive them like
|
||||
any other tool call. The single conversion rule shared by the non-streaming
|
||||
accumulator and the streaming ``output_item.added`` branch."""
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
is_custom = item.get("type") == "custom_tool_call"
|
||||
arguments = (item.get("input") if is_custom else item.get("arguments")) or ""
|
||||
name = item.get("name") or ("custom_tool" if is_custom else "")
|
||||
function_chunk = ChatCompletionToolCallFunctionChunk(name=name, arguments=arguments)
|
||||
tool_call_dict = _ChatToolCallDict(
|
||||
id=LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item(item.get("id"), item.get("call_id")),
|
||||
type="function",
|
||||
function=function_chunk,
|
||||
index=index,
|
||||
)
|
||||
raw_provider_fields = item.get("provider_specific_fields")
|
||||
if isinstance(raw_provider_fields, dict):
|
||||
provider_specific_fields = raw_provider_fields
|
||||
elif raw_provider_fields and hasattr(raw_provider_fields, "__dict__"):
|
||||
provider_specific_fields = vars(raw_provider_fields)
|
||||
else:
|
||||
provider_specific_fields = None
|
||||
if provider_specific_fields:
|
||||
tool_call_dict["provider_specific_fields"] = provider_specific_fields
|
||||
function_chunk["provider_specific_fields"] = provider_specific_fields
|
||||
return tool_call_dict
|
||||
|
||||
|
||||
def _flat_responses_tool_choice(choice_type: str, name: str) -> ToolChoiceFunctionParam | ToolChoiceCustomParam:
|
||||
if choice_type == "custom":
|
||||
return ToolChoiceCustomParam(type="custom", name=name)
|
||||
return ToolChoiceFunctionParam(type="function", name=name)
|
||||
|
||||
|
||||
def _reasoning_item_to_response_input(
|
||||
r_item: ChatCompletionReasoningItem | dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
|
|
@ -117,17 +171,20 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
pass
|
||||
|
||||
def _normalize_tool_choice_for_responses_api(self, tool_choice: Any) -> Any:
|
||||
"""Chat tool_choice uses function.name; Responses API expects top-level name."""
|
||||
if not isinstance(tool_choice, dict) or tool_choice.get("type") != "function":
|
||||
"""Chat tool_choice nests the name under function/custom; Responses API expects top-level name."""
|
||||
if not isinstance(tool_choice, dict):
|
||||
return tool_choice
|
||||
choice_type = tool_choice.get("type")
|
||||
if choice_type not in ("function", "custom"):
|
||||
return tool_choice
|
||||
if isinstance(tool_choice.get("name"), str) and tool_choice.get("name"):
|
||||
# Return only Responses shape so stray chat ``function`` key is not sent upstream.
|
||||
return {"type": "function", "name": tool_choice["name"]}
|
||||
fn = tool_choice.get("function")
|
||||
if isinstance(fn, dict):
|
||||
fn_name = fn.get("name")
|
||||
if isinstance(fn_name, str) and fn_name:
|
||||
return {"type": "function", "name": fn_name}
|
||||
# Return only Responses shape so stray chat ``function``/``custom`` keys are not sent upstream.
|
||||
return _flat_responses_tool_choice(choice_type, tool_choice["name"])
|
||||
nested = tool_choice.get(choice_type)
|
||||
if isinstance(nested, dict):
|
||||
nested_name = nested.get("name")
|
||||
if isinstance(nested_name, str) and nested_name:
|
||||
return _flat_responses_tool_choice(choice_type, nested_name)
|
||||
return tool_choice
|
||||
|
||||
def _handle_raw_dict_response_item(self, item: dict[str, Any], index: int) -> tuple[Any | None, int]:
|
||||
|
|
@ -169,36 +226,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
choice = Choices(message=msg, finish_reason="stop", index=index)
|
||||
return choice, index + 1
|
||||
|
||||
# Handle function_call items (e.g., from GPT-5 Codex format)
|
||||
if item_type == "function_call":
|
||||
# Extract provider_specific_fields if present and pass through as-is
|
||||
provider_specific_fields = item.get("provider_specific_fields")
|
||||
if provider_specific_fields and not isinstance(provider_specific_fields, dict):
|
||||
provider_specific_fields = (
|
||||
dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {}
|
||||
)
|
||||
|
||||
tool_call_dict = {
|
||||
"id": item.get("call_id") or item.get("id", ""),
|
||||
"function": {
|
||||
"name": item.get("name", ""),
|
||||
"arguments": item.get("arguments", ""),
|
||||
},
|
||||
"type": "function",
|
||||
}
|
||||
|
||||
# Pass through provider_specific_fields as-is if present
|
||||
if provider_specific_fields:
|
||||
tool_call_dict["provider_specific_fields"] = provider_specific_fields
|
||||
# Also add to function's provider_specific_fields for consistency
|
||||
tool_call_dict["function"]["provider_specific_fields"] = provider_specific_fields
|
||||
|
||||
msg = Message(
|
||||
content=None,
|
||||
tool_calls=[tool_call_dict],
|
||||
)
|
||||
choice = Choices(message=msg, finish_reason="tool_calls", index=index)
|
||||
return choice, index + 1
|
||||
# function_call / custom_tool_call dicts are intercepted and accumulated by
|
||||
# _convert_response_output_to_choices before this callback is reached
|
||||
|
||||
# Unknown or unsupported type
|
||||
return None, index
|
||||
|
|
@ -208,6 +237,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
) -> tuple[list[Any], str | None]:
|
||||
input_items: list[Any] = []
|
||||
instructions: str | None = None
|
||||
custom_tool_call_ids = frozenset(
|
||||
tool_call["id"]
|
||||
for msg in messages
|
||||
if msg.get("role") == "assistant" and isinstance(msg.get("tool_calls"), list)
|
||||
for tool_call in msg.get("tool_calls") or ()
|
||||
if isinstance(tool_call, dict)
|
||||
and not tool_call.get("function")
|
||||
and isinstance(tool_call.get("custom"), dict)
|
||||
)
|
||||
|
||||
for msg in messages:
|
||||
role = msg.get("role")
|
||||
|
|
@ -253,18 +291,28 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
else:
|
||||
# Fallback: convert unexpected types to input_text
|
||||
tool_output = [{"type": "input_text", "text": str(content)}]
|
||||
input_items.append(
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": tool_call_id,
|
||||
"output": tool_output,
|
||||
}
|
||||
)
|
||||
if tool_call_id in custom_tool_call_ids:
|
||||
input_items.append(
|
||||
ResponseCustomToolCallOutputParam(
|
||||
type="custom_tool_call_output",
|
||||
call_id=tool_call_id,
|
||||
output=content if isinstance(content, str) else tool_output,
|
||||
)
|
||||
)
|
||||
else:
|
||||
input_items.append(
|
||||
FunctionCallOutput(
|
||||
type="function_call_output",
|
||||
call_id=tool_call_id,
|
||||
output=tool_output,
|
||||
)
|
||||
)
|
||||
elif role == "assistant" and tool_calls and isinstance(tool_calls, list):
|
||||
for r_item in _get_reasoning_items(msg):
|
||||
input_items.append(_reasoning_item_to_response_input(r_item))
|
||||
for tool_call in tool_calls:
|
||||
function = tool_call.get("function")
|
||||
custom = tool_call.get("custom")
|
||||
if function:
|
||||
input_tool_call: dict[str, Any] = {
|
||||
"type": "function_call",
|
||||
|
|
@ -275,6 +323,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
if "arguments" in function:
|
||||
input_tool_call["arguments"] = function["arguments"]
|
||||
input_items.append(input_tool_call)
|
||||
elif isinstance(custom, dict):
|
||||
input_items.append(
|
||||
ResponseCustomToolCallParam(
|
||||
type="custom_tool_call",
|
||||
call_id=tool_call["id"],
|
||||
name=custom.get("name", ""),
|
||||
input=custom.get("input", ""),
|
||||
)
|
||||
)
|
||||
else:
|
||||
raise ValueError(f"tool call not supported: {tool_call}")
|
||||
elif content is not None:
|
||||
|
|
@ -555,11 +612,21 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
accumulated_tool_calls.append(tool_call_dict)
|
||||
tool_call_index += 1
|
||||
|
||||
elif isinstance(item, dict) and handle_raw_dict_callback is not None:
|
||||
# Handle raw dict responses (e.g., from GPT-5 Codex)
|
||||
choice, index = handle_raw_dict_callback(item=item, index=index)
|
||||
if choice is not None:
|
||||
choices.append(choice)
|
||||
elif isinstance(item, (dict, BaseModel)):
|
||||
# Raw dict items (e.g., from GPT-5 Codex) and pydantic items matching no
|
||||
# openai SDK class above: typed ResponseCustomToolCall and litellm's own
|
||||
# GenericResponseOutputItem from the completion bridge both land here
|
||||
raw_item = item if isinstance(item, dict) else item.model_dump()
|
||||
if raw_item.get("type") in ("function_call", "custom_tool_call"):
|
||||
# Tool calls accumulate into the single trailing tool_calls choice
|
||||
# like the typed branches above; a choice per call would hide every
|
||||
# call after choices[0] from chat clients
|
||||
accumulated_tool_calls.append(_tool_call_dict_from_output_item(raw_item, tool_call_index))
|
||||
tool_call_index += 1
|
||||
elif handle_raw_dict_callback is not None:
|
||||
choice, index = handle_raw_dict_callback(item=raw_item, index=index)
|
||||
if choice is not None:
|
||||
choices.append(choice)
|
||||
else:
|
||||
pass # don't fail request if item in list is not supported
|
||||
|
||||
|
|
@ -868,6 +935,18 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
description=function_tool.get("description"),
|
||||
)
|
||||
)
|
||||
elif tool.get("type") == "custom" and isinstance(tool.get("custom"), dict):
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_responses_shape,
|
||||
)
|
||||
|
||||
custom_payload = tool["custom"]
|
||||
flat_custom = CustomToolParam(type="custom", name=custom_payload.get("name", ""))
|
||||
if custom_payload.get("description") is not None:
|
||||
flat_custom["description"] = custom_payload["description"]
|
||||
if isinstance(custom_payload.get("format"), dict):
|
||||
flat_custom["format"] = convert_custom_tool_format_to_responses_shape(custom_payload["format"])
|
||||
responses_tools.append(flat_custom)
|
||||
else:
|
||||
responses_tools.append(tool) # type: ignore
|
||||
|
||||
|
|
@ -1062,6 +1141,7 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
def __init__(self, streaming_response, sync_stream: bool, json_mode: bool | None = False):
|
||||
super().__init__(streaming_response, sync_stream, json_mode)
|
||||
self._chat_completion_id: str | None = None
|
||||
self._tool_call_index_map: dict[int, int] = {} # mutable-ok: per-stream accumulator state
|
||||
|
||||
def _handle_string_chunk(
|
||||
self, str_line: Union[str, "BaseModel"]
|
||||
|
|
@ -1080,15 +1160,35 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
|
||||
return self.chunk_parser(json.loads(str_line))
|
||||
|
||||
@staticmethod
|
||||
def _sequential_tool_call_index(
|
||||
tool_call_index_map: dict[int, int] | None, # mutable-ok: per-stream state, remapped in place
|
||||
output_index: int,
|
||||
) -> int:
|
||||
"""Chat-completions tool_call indices must be 0-based and sequential, but
|
||||
Responses API ``output_index`` counts every output item (reasoning,
|
||||
message, ...), so the first tool call of a reasoning model arrives at
|
||||
output_index >= 1 and strict SSE accumulators (e.g. Cursor agent mode)
|
||||
misplace it. When a per-stream map is provided, remap each distinct
|
||||
output_index to the next sequential slot; without a map (stateless
|
||||
callers), fall back to the raw output_index."""
|
||||
if tool_call_index_map is None:
|
||||
return output_index
|
||||
if output_index not in tool_call_index_map:
|
||||
tool_call_index_map[output_index] = len(tool_call_index_map) # mutable-ok: per-stream accumulator state
|
||||
return tool_call_index_map[output_index]
|
||||
|
||||
@staticmethod
|
||||
def translate_responses_chunk_to_openai_stream(
|
||||
parsed_chunk: dict | BaseModel,
|
||||
tool_call_index_map: dict[int, int] | None = None, # mutable-ok: per-stream state, remapped in place
|
||||
) -> "ModelResponseStream":
|
||||
"""
|
||||
Translate a Responses API streaming chunk to OpenAI chat completion streaming format.
|
||||
|
||||
Args:
|
||||
parsed_chunk: Dict containing the Responses API event chunk
|
||||
tool_call_index_map: Per-stream output_index -> sequential tool_call index map
|
||||
|
||||
Returns:
|
||||
ModelResponseStream: OpenAI-formatted streaming chunk
|
||||
|
|
@ -1139,37 +1239,26 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
elif event_type == "response.output_item.added":
|
||||
# New output item added
|
||||
output_item = parsed_chunk.get("item", {})
|
||||
if output_item.get("type") == "function_call":
|
||||
# Extract provider_specific_fields if present
|
||||
provider_specific_fields = output_item.get("provider_specific_fields")
|
||||
if provider_specific_fields and not isinstance(provider_specific_fields, dict):
|
||||
provider_specific_fields = (
|
||||
dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {}
|
||||
)
|
||||
if output_item.get("type") in ("function_call", "custom_tool_call"):
|
||||
converted = _tool_call_dict_from_output_item(output_item, parsed_chunk.get("output_index", 0))
|
||||
provider_specific_fields = converted.get("provider_specific_fields")
|
||||
|
||||
function_chunk = ChatCompletionToolCallFunctionChunk(
|
||||
name=output_item.get("name", None),
|
||||
arguments=parsed_chunk.get("arguments", ""),
|
||||
name=converted["function"]["name"] or None,
|
||||
arguments=converted["function"]["arguments"] or parsed_chunk.get("arguments") or "",
|
||||
)
|
||||
|
||||
if provider_specific_fields:
|
||||
function_chunk["provider_specific_fields"] = provider_specific_fields
|
||||
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index(
|
||||
tool_call_index_map, parsed_chunk.get("output_index", 0)
|
||||
)
|
||||
|
||||
tool_call_index = parsed_chunk.get("output_index", 0)
|
||||
tool_call_chunk = ChatCompletionToolCallChunk(
|
||||
id=LiteLLMCompletionResponsesConfig._tool_call_id_from_responses_item(
|
||||
output_item.get("id"), output_item.get("call_id")
|
||||
),
|
||||
id=converted["id"],
|
||||
index=tool_call_index,
|
||||
type="function",
|
||||
function=function_chunk,
|
||||
)
|
||||
|
||||
# Add provider_specific_fields if present
|
||||
if provider_specific_fields:
|
||||
tool_call_chunk.provider_specific_fields = provider_specific_fields # type: ignore
|
||||
|
||||
|
|
@ -1182,10 +1271,15 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
)
|
||||
]
|
||||
)
|
||||
elif event_type == "response.function_call_arguments.delta":
|
||||
elif event_type in (
|
||||
ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA,
|
||||
ResponsesAPIStreamEvents.CUSTOM_TOOL_CALL_INPUT_DELTA,
|
||||
):
|
||||
content_part: str | None = parsed_chunk.get("delta", None)
|
||||
if content_part:
|
||||
tool_call_index = parsed_chunk.get("output_index", 0)
|
||||
tool_call_index = OpenAiResponsesToChatCompletionStreamIterator._sequential_tool_call_index(
|
||||
tool_call_index_map, parsed_chunk.get("output_index", 0)
|
||||
)
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
|
|
@ -1209,39 +1303,32 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
elif event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE:
|
||||
# New output item added
|
||||
output_item = parsed_chunk.get("item", {})
|
||||
if output_item.get("type") == "function_call":
|
||||
# Extract provider_specific_fields if present
|
||||
provider_specific_fields = output_item.get("provider_specific_fields")
|
||||
if provider_specific_fields and not isinstance(provider_specific_fields, dict):
|
||||
provider_specific_fields = (
|
||||
dict(provider_specific_fields) if hasattr(provider_specific_fields, "__dict__") else {}
|
||||
if output_item.get("type") in ("function_call", "custom_tool_call"):
|
||||
if tool_call_index_map is None:
|
||||
# Stateless callers (the responses guardrail handler extracting
|
||||
# tool calls from a buffered output_item.done) get the complete
|
||||
# tool call; per-stream callers already received it via
|
||||
# output_item.added and the argument delta events
|
||||
return ModelResponseStream(
|
||||
choices=[ # mutable-ok: ModelResponseStream coerces only list choices
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(
|
||||
tool_calls=(
|
||||
_tool_call_dict_from_output_item(
|
||||
output_item, parsed_chunk.get("output_index", 0)
|
||||
),
|
||||
)
|
||||
),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
function_chunk = ChatCompletionToolCallFunctionChunk(
|
||||
name=output_item.get("name", None),
|
||||
arguments="", # responses API sends everything again, we don't
|
||||
)
|
||||
|
||||
# Add provider_specific_fields to function if present
|
||||
if provider_specific_fields:
|
||||
function_chunk["provider_specific_fields"] = provider_specific_fields
|
||||
|
||||
tool_call_index = parsed_chunk.get("output_index", 0)
|
||||
tool_call_chunk = ChatCompletionToolCallChunk(
|
||||
id=output_item.get("call_id"),
|
||||
index=tool_call_index,
|
||||
type="function",
|
||||
function=function_chunk,
|
||||
)
|
||||
|
||||
# Add provider_specific_fields if present
|
||||
if provider_specific_fields:
|
||||
tool_call_chunk.provider_specific_fields = provider_specific_fields # type: ignore
|
||||
|
||||
# Do NOT emit finish_reason here — response.completed handles the terminal
|
||||
# finish_reason. Emitting "tool_calls" here would prematurely terminate
|
||||
# the stream before subsequent tool calls arrive (same fix as #17246 for
|
||||
# the message-type branch).
|
||||
# the message-type branch). The item's fields were already streamed via
|
||||
# output_item.added and the argument delta events.
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
|
|
@ -1300,7 +1387,9 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
output_items = response_data.get("output", []) if response_data else []
|
||||
|
||||
has_function_calls = any(
|
||||
item.get("type") == "function_call" for item in output_items if isinstance(item, dict)
|
||||
item.get("type") in ("function_call", "custom_tool_call")
|
||||
for item in output_items
|
||||
if isinstance(item, dict)
|
||||
)
|
||||
|
||||
finish_reason = "tool_calls" if has_function_calls else "stop"
|
||||
|
|
@ -1370,7 +1459,9 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
"""
|
||||
verbose_logger.debug("Chat provider: transform_streaming_response called with chunk: %s", chunk)
|
||||
return self._with_stream_scoped_id(
|
||||
OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk)
|
||||
OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
|
||||
chunk, tool_call_index_map=self._tool_call_index_map
|
||||
)
|
||||
)
|
||||
|
||||
def _with_stream_scoped_id(self, chunk: "ModelResponseStream") -> "ModelResponseStream":
|
||||
|
|
|
|||
|
|
@ -58,12 +58,19 @@ class HeliconeLogger:
|
|||
content = []
|
||||
if "tool_calls" in message and message["tool_calls"]:
|
||||
for tool_call in message["tool_calls"]:
|
||||
function = tool_call.get("function")
|
||||
custom = tool_call.get("custom")
|
||||
if not function and not custom:
|
||||
continue
|
||||
name, tool_input = (
|
||||
(function["name"], function["arguments"]) if function else (custom["name"], custom["input"])
|
||||
)
|
||||
content.append(
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": tool_call["id"],
|
||||
"name": tool_call["function"]["name"],
|
||||
"input": tool_call["function"]["arguments"],
|
||||
"name": name,
|
||||
"input": tool_input,
|
||||
}
|
||||
)
|
||||
elif "content" in message and message["content"]:
|
||||
|
|
|
|||
|
|
@ -20,18 +20,25 @@ def parse_tool_calls(tool_calls):
|
|||
return None
|
||||
|
||||
def clean_tool_call(tool_call):
|
||||
serialized = {
|
||||
custom = getattr(tool_call, "custom", None)
|
||||
if custom is not None:
|
||||
name, arguments = custom.name, custom.input
|
||||
else:
|
||||
name, arguments = tool_call.function.name, tool_call.function.arguments
|
||||
return {
|
||||
"type": tool_call.type,
|
||||
"id": tool_call.id,
|
||||
"function": {
|
||||
"name": tool_call.function.name,
|
||||
"arguments": tool_call.function.arguments,
|
||||
"name": name,
|
||||
"arguments": arguments,
|
||||
},
|
||||
}
|
||||
|
||||
return serialized
|
||||
|
||||
return [clean_tool_call(tool_call) for tool_call in tool_calls]
|
||||
return [
|
||||
clean_tool_call(tool_call)
|
||||
for tool_call in tool_calls
|
||||
if getattr(tool_call, "function", None) is not None or getattr(tool_call, "custom", None) is not None
|
||||
]
|
||||
|
||||
|
||||
def parse_messages(input):
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ import json
|
|||
import re
|
||||
import time
|
||||
import traceback
|
||||
from collections.abc import Iterable
|
||||
from collections.abc import Iterable, Sequence
|
||||
from typing import Literal, cast
|
||||
|
||||
import litellm
|
||||
|
|
@ -20,6 +20,7 @@ from litellm.types.llms.openai import (
|
|||
)
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionDeltaToolCall,
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
ChatCompletionRedactedThinkingBlock,
|
||||
Choices,
|
||||
|
|
@ -41,6 +42,7 @@ from litellm.types.utils import (
|
|||
TranscriptionUsageDurationObject,
|
||||
TranscriptionUsageTokensObject,
|
||||
Usage,
|
||||
chat_completion_tool_call_from_dict,
|
||||
)
|
||||
from litellm.types.utils import Logprobs as TextCompletionLogprobs
|
||||
|
||||
|
|
@ -368,7 +370,9 @@ from collections import defaultdict
|
|||
|
||||
|
||||
def _handle_invalid_parallel_tool_calls(
|
||||
tool_calls: list[ChatCompletionMessageToolCall],
|
||||
tool_calls: list[
|
||||
ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall
|
||||
], # mutable-ok: patched in place via slice assignment
|
||||
):
|
||||
"""
|
||||
Handle hallucinated parallel tool call from openai - https://community.openai.com/t/model-tries-to-call-unknown-function-multi-tool-use-parallel/490653
|
||||
|
|
@ -381,6 +385,8 @@ def _handle_invalid_parallel_tool_calls(
|
|||
try:
|
||||
replacements: dict[int, list[ChatCompletionMessageToolCall]] = defaultdict(list)
|
||||
for i, tool_call in enumerate(tool_calls):
|
||||
if isinstance(tool_call, ChatCompletionMessageCustomToolCall):
|
||||
continue
|
||||
current_function = tool_call.function.name
|
||||
function_args = json.loads(tool_call.function.arguments)
|
||||
if current_function == "multi_tool_use.parallel":
|
||||
|
|
@ -525,19 +531,17 @@ class LiteLLMResponseObjectHandler:
|
|||
|
||||
|
||||
def _should_convert_tool_call_to_json_mode(
|
||||
tool_calls: list[ChatCompletionMessageToolCall] | list[DatabricksTool] | None = None,
|
||||
tool_calls: (
|
||||
Sequence[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | Sequence[DatabricksTool] | None
|
||||
) = None,
|
||||
convert_tool_call_to_json_mode: bool | None = None,
|
||||
) -> bool:
|
||||
"""
|
||||
Determine if tool calls should be converted to JSON mode
|
||||
"""
|
||||
if (
|
||||
convert_tool_call_to_json_mode
|
||||
and tool_calls is not None
|
||||
and len(tool_calls) == 1
|
||||
and tool_calls[0]["function"]["name"] == RESPONSE_FORMAT_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
if convert_tool_call_to_json_mode and tool_calls is not None and len(tool_calls) == 1:
|
||||
function = tool_calls[0].get("function")
|
||||
return function is not None and function["name"] == RESPONSE_FORMAT_TOOL_NAME
|
||||
return False
|
||||
|
||||
|
||||
|
|
@ -642,7 +646,7 @@ def convert_to_model_response_object(
|
|||
if tool_calls is not None:
|
||||
_openai_tool_calls = []
|
||||
for _tc in tool_calls:
|
||||
_openai_tc = ChatCompletionMessageToolCall(**_tc)
|
||||
_openai_tc = chat_completion_tool_call_from_dict(_tc)
|
||||
_openai_tool_calls.append(_openai_tc)
|
||||
fixed_tool_calls = _handle_invalid_parallel_tool_calls(_openai_tool_calls)
|
||||
|
||||
|
|
|
|||
|
|
@ -16,6 +16,14 @@ from typing import (
|
|||
cast,
|
||||
)
|
||||
|
||||
from openai.types.chat.chat_completion_custom_tool_param import (
|
||||
CustomFormatGrammar,
|
||||
CustomFormatGrammarGrammar,
|
||||
)
|
||||
from openai.types.shared_params.custom_tool_input_format import (
|
||||
Grammar as ResponsesGrammarFormat,
|
||||
)
|
||||
|
||||
import litellm
|
||||
from litellm import verbose_logger
|
||||
from litellm.router_utils.batch_utils import InMemoryFile
|
||||
|
|
@ -1250,6 +1258,38 @@ def is_function_call(optional_params: dict) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def convert_custom_tool_format_to_chat_shape(format_obj: Mapping[str, Any]) -> Mapping[str, Any]:
|
||||
"""
|
||||
Responses API grammar formats are flat ({"type": "grammar", "definition", "syntax"});
|
||||
Chat Completions wraps the same fields in a "grammar" object. Text formats are
|
||||
identical on both surfaces and pass through, as does anything unrecognized.
|
||||
"""
|
||||
if format_obj.get("type") != "grammar" or "grammar" in format_obj:
|
||||
return format_obj
|
||||
grammar = CustomFormatGrammarGrammar()
|
||||
if "definition" in format_obj:
|
||||
grammar["definition"] = format_obj["definition"]
|
||||
if "syntax" in format_obj:
|
||||
grammar["syntax"] = format_obj["syntax"]
|
||||
return CustomFormatGrammar(type="grammar", grammar=grammar)
|
||||
|
||||
|
||||
def convert_custom_tool_format_to_responses_shape(format_obj: Mapping[str, Any]) -> Mapping[str, Any]:
|
||||
"""
|
||||
Inverse of convert_custom_tool_format_to_chat_shape: unwrap the Chat Completions
|
||||
"grammar" object into the flat Responses API grammar shape.
|
||||
"""
|
||||
grammar = format_obj.get("grammar")
|
||||
if format_obj.get("type") != "grammar" or not isinstance(grammar, dict):
|
||||
return format_obj
|
||||
flat = ResponsesGrammarFormat(type="grammar")
|
||||
if "definition" in grammar:
|
||||
flat["definition"] = grammar["definition"]
|
||||
if "syntax" in grammar:
|
||||
flat["syntax"] = grammar["syntax"]
|
||||
return flat
|
||||
|
||||
|
||||
def get_file_ids_from_messages(messages: list[AllMessageValues]) -> list[str]:
|
||||
"""
|
||||
Gets file ids from messages
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import base64
|
||||
import time
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import TYPE_CHECKING, Any, Union, cast
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
|
|
@ -10,6 +11,8 @@ from litellm.types.llms.openai import (
|
|||
from litellm.types.utils import (
|
||||
CacheCreationTokenDetails,
|
||||
ChatCompletionAudioResponse,
|
||||
ChatCompletionCustomToolCallPayload,
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
Choices,
|
||||
CompletionTokensDetails,
|
||||
|
|
@ -202,8 +205,14 @@ class ChunkProcessor:
|
|||
response = self.update_model_response_with_hidden_params(model_response=response, chunk=chunk)
|
||||
return response
|
||||
|
||||
def get_combined_tool_content(self, tool_call_chunks: list[dict[str, Any]]) -> list[ChatCompletionMessageToolCall]:
|
||||
tool_calls_list: list[ChatCompletionMessageToolCall] = []
|
||||
def get_combined_tool_content(
|
||||
self, tool_call_chunks: Sequence[Mapping[str, Any]]
|
||||
) -> list[
|
||||
ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall
|
||||
]: # mutable-ok: assigned verbatim to Message.tool_calls, a list field
|
||||
tool_calls_list: list[
|
||||
ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall
|
||||
] = [] # mutable-ok: see return type
|
||||
tool_call_map: dict[int, dict[str, Any]] = {} # Map to store tool calls by index
|
||||
|
||||
for chunk in tool_call_chunks:
|
||||
|
|
@ -219,12 +228,15 @@ class ChunkProcessor:
|
|||
|
||||
# Check if tool_call has function (either as attribute or dict key)
|
||||
has_function = False
|
||||
has_custom = False
|
||||
if isinstance(tool_call, dict):
|
||||
has_function = "function" in tool_call and tool_call["function"] is not None
|
||||
has_custom = "custom" in tool_call and tool_call["custom"] is not None
|
||||
else:
|
||||
has_function = hasattr(tool_call, "function") and tool_call.function is not None
|
||||
has_custom = getattr(tool_call, "custom", None) is not None
|
||||
|
||||
if not has_function:
|
||||
if not has_function and not has_custom:
|
||||
continue
|
||||
|
||||
# Get index (handle both dict and object)
|
||||
|
|
@ -238,7 +250,9 @@ class ChunkProcessor:
|
|||
"id": None,
|
||||
"name": None,
|
||||
"type": None,
|
||||
"arguments": [],
|
||||
"arguments": (),
|
||||
"custom_name": None,
|
||||
"custom_input": (),
|
||||
"provider_specific_fields": None,
|
||||
}
|
||||
|
||||
|
|
@ -254,13 +268,20 @@ class ChunkProcessor:
|
|||
if function.get("name"):
|
||||
tool_call_map[index]["name"] = function["name"]
|
||||
if function.get("arguments"):
|
||||
tool_call_map[index]["arguments"].append(function["arguments"])
|
||||
tool_call_map[index]["arguments"] += (function["arguments"],)
|
||||
else:
|
||||
# function is an object
|
||||
if hasattr(function, "name") and function.name:
|
||||
tool_call_map[index]["name"] = function.name
|
||||
if hasattr(function, "arguments") and function.arguments:
|
||||
tool_call_map[index]["arguments"].append(function.arguments)
|
||||
tool_call_map[index]["arguments"] += (function.arguments,)
|
||||
|
||||
custom = tool_call.get("custom")
|
||||
if isinstance(custom, dict):
|
||||
if custom.get("name"):
|
||||
tool_call_map[index]["custom_name"] = custom["name"]
|
||||
if custom.get("input"):
|
||||
tool_call_map[index]["custom_input"] += (custom["input"],)
|
||||
else:
|
||||
# tool_call is an object
|
||||
if hasattr(tool_call, "id") and tool_call.id:
|
||||
|
|
@ -271,7 +292,14 @@ class ChunkProcessor:
|
|||
if hasattr(tool_call.function, "name") and tool_call.function.name:
|
||||
tool_call_map[index]["name"] = tool_call.function.name
|
||||
if hasattr(tool_call.function, "arguments") and tool_call.function.arguments:
|
||||
tool_call_map[index]["arguments"].append(tool_call.function.arguments)
|
||||
tool_call_map[index]["arguments"] += (tool_call.function.arguments,)
|
||||
|
||||
custom = getattr(tool_call, "custom", None)
|
||||
if custom is not None:
|
||||
if getattr(custom, "name", None):
|
||||
tool_call_map[index]["custom_name"] = custom.name
|
||||
if getattr(custom, "input", None):
|
||||
tool_call_map[index]["custom_input"] += (custom.input,)
|
||||
|
||||
# Preserve provider_specific_fields from streaming chunks
|
||||
provider_fields = None
|
||||
|
|
@ -299,7 +327,17 @@ class ChunkProcessor:
|
|||
# Convert the map to a list of tool calls
|
||||
for index in sorted(tool_call_map.keys()):
|
||||
tool_call_data = tool_call_map[index]
|
||||
if tool_call_data["id"] and tool_call_data["name"]:
|
||||
if tool_call_data["id"] and tool_call_data["custom_name"]:
|
||||
tool_calls_list.append(
|
||||
ChatCompletionMessageCustomToolCall(
|
||||
id=tool_call_data["id"],
|
||||
custom=ChatCompletionCustomToolCallPayload(
|
||||
name=tool_call_data["custom_name"],
|
||||
input="".join(tool_call_data["custom_input"]),
|
||||
),
|
||||
)
|
||||
)
|
||||
elif tool_call_data["id"] and tool_call_data["name"]:
|
||||
combined_arguments = "".join(tool_call_data["arguments"]) or "{}"
|
||||
|
||||
# Build function - provider_specific_fields should be on tool_call level, not function level
|
||||
|
|
|
|||
|
|
@ -43,12 +43,14 @@ from litellm.types.llms.openai import (
|
|||
OpenAIMessageContentListBlock,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
Choices,
|
||||
Function,
|
||||
Message,
|
||||
ModelResponse,
|
||||
ModelResponseStream,
|
||||
chat_completion_tool_call_from_dict,
|
||||
)
|
||||
from litellm.utils import convert_to_model_response_object
|
||||
|
||||
|
|
@ -524,12 +526,14 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig):
|
|||
for choice in choices:
|
||||
## HANDLE JSON MODE - anthropic returns single function call]
|
||||
tool_calls = choice["message"].get("tool_calls", None)
|
||||
new_tool_calls: list[ChatCompletionMessageToolCall] | None = None
|
||||
new_tool_calls: list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | None = (
|
||||
None # mutable-ok: holds _handle_invalid_parallel_tool_calls' list; Message.__init__ expects list
|
||||
)
|
||||
message_content = choice["message"].get("content", None)
|
||||
if tool_calls is not None:
|
||||
_openai_tool_calls = []
|
||||
for _tc in tool_calls:
|
||||
_openai_tc = ChatCompletionMessageToolCall(**_tc) # type: ignore
|
||||
_openai_tc = chat_completion_tool_call_from_dict(_tc)
|
||||
_openai_tool_calls.append(_openai_tc)
|
||||
fixed_tool_calls = _handle_invalid_parallel_tool_calls(_openai_tool_calls)
|
||||
|
||||
|
|
|
|||
|
|
@ -970,6 +970,23 @@ def mock_completion(
|
|||
raise Exception(f"Mock completion response failed - {e}")
|
||||
|
||||
|
||||
_OPENAI_DEFAULT_API_BASE = "https://api.openai.com/v1"
|
||||
|
||||
|
||||
def _resolve_openai_api_base(api_base: str | None) -> str:
|
||||
"""Effective OpenAI base a chat request will hit: arg > global > env > default. The bridge gate
|
||||
and the ``_complete_custom_openai`` chat handler MUST resolve this identically, or a custom base
|
||||
set via ``litellm.api_base`` or ``OPENAI_BASE_URL``/``OPENAI_API_BASE`` is invisible to the gate,
|
||||
which then misreads it as the default OpenAI endpoint and bridges a request the backend can't serve."""
|
||||
return (
|
||||
api_base
|
||||
or litellm.api_base
|
||||
or get_secret_str("OPENAI_BASE_URL")
|
||||
or get_secret_str("OPENAI_API_BASE")
|
||||
or _OPENAI_DEFAULT_API_BASE
|
||||
)
|
||||
|
||||
|
||||
def responses_api_bridge_check(
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
|
|
@ -977,6 +994,7 @@ def responses_api_bridge_check(
|
|||
tools: list[Any] | None = None,
|
||||
reasoning_effort: Any | None = None,
|
||||
reasoning_summary: Any | None = None,
|
||||
api_base: str | None = None,
|
||||
) -> tuple[dict, str]:
|
||||
model_info: dict[str, Any] = {}
|
||||
|
||||
|
|
@ -1013,16 +1031,53 @@ def responses_api_bridge_check(
|
|||
# ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects
|
||||
# those keys.
|
||||
#
|
||||
# - gpt-5.4+: tools + reasoning_effort (original) or any reasoning-summary alias.
|
||||
# - gpt-5.4+: FUNCTION tools with reasoning active must be bridged. OpenAI enables
|
||||
# reasoning by default for these models (unset reasoning_effort means medium
|
||||
# server-side), and Chat Completions rejects function tools whenever reasoning is
|
||||
# on ("Function tools with reasoning_effort are not supported ... use
|
||||
# /v1/responses or set reasoning_effort to 'none'"), so only an explicit
|
||||
# ``"none"`` keeps the request chat-servable. Custom (grammar) tools are served
|
||||
# natively by Chat Completions with reasoning on, so custom-only requests stay on
|
||||
# chat and keep their native custom tool_call response shape.
|
||||
# - The UNSET-effort arm only fires against endpoints known to enforce that
|
||||
# constraint (the default OpenAI endpoint, or Azure OpenAI where api_base is
|
||||
# always set): chat-only OpenAI-compatible backends registered under the openai
|
||||
# provider with a custom api_base and gpt-5.4+ model names serve tools without
|
||||
# reasoning fine and have no /responses route, so they keep pre-existing
|
||||
# behavior (bridge only on an explicit reasoning_effort).
|
||||
# - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning
|
||||
# summary alias is present with ``reasoning_effort`` (tools alone stay on chat).
|
||||
has_function_tool = any(
|
||||
(tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function")
|
||||
for tool in (tools or ())
|
||||
)
|
||||
if isinstance(reasoning_effort, dict):
|
||||
reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None
|
||||
else:
|
||||
reasoning_active = reasoning_effort != "none"
|
||||
# The reasoning+tools constraint is enforced only by the real OpenAI endpoint (and Azure OpenAI).
|
||||
# Resolve the effective base arg>global>env>default exactly as the chat handler does, so a custom
|
||||
# base set via litellm.api_base or OPENAI_BASE_URL/OPENAI_API_BASE isn't misread as the default and
|
||||
# bridged to a /responses route it lacks. A whitespace-only base collapses to the default too.
|
||||
resolved_api_base = _resolve_openai_api_base(api_base)
|
||||
on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or resolved_api_base.strip() in (
|
||||
"",
|
||||
_OPENAI_DEFAULT_API_BASE,
|
||||
)
|
||||
if (
|
||||
custom_llm_provider in ("openai", "azure")
|
||||
and model_info.get("mode") != "responses"
|
||||
and OpenAIGPT5Config.is_model_gpt_5_model(model)
|
||||
and not OpenAIGPT5Config.is_model_gpt_5_search_model(model)
|
||||
and reasoning_effort is not None
|
||||
and (reasoning_summary is not None or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and tools))
|
||||
and (
|
||||
(reasoning_effort is not None and reasoning_summary is not None)
|
||||
or (
|
||||
OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model)
|
||||
and has_function_tool
|
||||
and reasoning_active
|
||||
and (reasoning_effort is not None or on_constraint_enforcing_endpoint)
|
||||
)
|
||||
)
|
||||
):
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
|
|
@ -2354,13 +2409,8 @@ def _complete_custom_openai(
|
|||
stream = ctx.stream
|
||||
timeout = ctx.timeout
|
||||
|
||||
api_base = (
|
||||
api_base # for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there
|
||||
or litellm.api_base
|
||||
or get_secret("OPENAI_BASE_URL")
|
||||
or get_secret("OPENAI_API_BASE")
|
||||
or "https://api.openai.com/v1"
|
||||
)
|
||||
# for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there
|
||||
api_base = _resolve_openai_api_base(api_base)
|
||||
organization = (
|
||||
organization
|
||||
or litellm.organization
|
||||
|
|
@ -5139,6 +5189,7 @@ def completion( # type: ignore
|
|||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
web_search_options=web_search_options,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model):
|
||||
|
|
@ -5378,6 +5429,7 @@ def completion( # type: ignore
|
|||
tools=tools,
|
||||
reasoning_effort=reasoning_effort,
|
||||
reasoning_summary=_reasoning_summary_for_bridge,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
# Use base_model (the true underlying model) for Azure model-type
|
||||
|
|
|
|||
|
|
@ -288,6 +288,8 @@ class LiteLLMRoutes(enum.Enum):
|
|||
"/chat/completions",
|
||||
"/v1/chat/completions",
|
||||
"/cursor/chat/completions",
|
||||
"/cursor/models",
|
||||
"/cursor/v1/models",
|
||||
# completions
|
||||
"/engines/{model}/completions",
|
||||
"/openai/deployments/{model}/completions",
|
||||
|
|
|
|||
|
|
@ -1,8 +1,9 @@
|
|||
import asyncio
|
||||
import json
|
||||
import time
|
||||
from collections.abc import AsyncIterator
|
||||
from typing import Any, cast
|
||||
from collections.abc import AsyncIterator, Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Any, NamedTuple, cast, get_args
|
||||
from uuid import uuid4
|
||||
|
||||
import fastapi
|
||||
|
|
@ -18,11 +19,134 @@ from litellm.proxy.auth.user_api_key_auth import (
|
|||
user_api_key_auth_websocket,
|
||||
)
|
||||
from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing
|
||||
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
|
||||
from litellm.types.llms.openai import REASONING_EFFORT, ResponseAPIUsage, ResponsesAPIResponse
|
||||
from litellm.types.responses.main import DeleteResponseResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.router import Router
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
_user_api_key_auth_dep = Depends(user_api_key_auth)
|
||||
_RESPONSES_TAGS = ["responses"] # mutable-ok: fastapi's route signature requires List[str] tags
|
||||
|
||||
_TOOL_PAYLOAD_KEYS: Mapping[str, tuple[str, ...]] = MappingProxyType(
|
||||
{
|
||||
"custom": ("name", "description", "format"),
|
||||
"function": ("name", "description", "parameters", "strict"),
|
||||
}
|
||||
)
|
||||
_EMPTY_TOOL_PAYLOAD: Mapping[str, Any] = MappingProxyType({})
|
||||
|
||||
|
||||
def _convert_tool_payload_value(key: str, value: object, *, to_chat: bool) -> object:
|
||||
if key != "format" or not isinstance(value, dict):
|
||||
return value
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_chat_shape,
|
||||
convert_custom_tool_format_to_responses_shape,
|
||||
)
|
||||
|
||||
convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape
|
||||
return convert(value)
|
||||
|
||||
|
||||
def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object:
|
||||
if not isinstance(obj, dict):
|
||||
return obj
|
||||
tool_type = obj.get("type")
|
||||
payload_keys = _TOOL_PAYLOAD_KEYS.get(tool_type)
|
||||
if payload_keys is None:
|
||||
return obj
|
||||
nested = obj.get(tool_type)
|
||||
nested_source = nested if isinstance(nested, dict) else _EMPTY_TOOL_PAYLOAD
|
||||
payload = { # mutable-ok: tool entries are embedded verbatim in the JSON request body
|
||||
key: _convert_tool_payload_value(key, nested_source[key] if key in nested_source else obj[key], to_chat=to_chat)
|
||||
for key in payload_keys
|
||||
if key in nested_source or key in obj
|
||||
}
|
||||
if "name" not in payload:
|
||||
return obj
|
||||
return {"type": tool_type, tool_type: payload} if to_chat else {"type": tool_type, **payload} # mutable-ok: same
|
||||
|
||||
|
||||
def _normalize_tool_dialect(
|
||||
data: dict, *, to_chat: bool
|
||||
) -> dict: # mutable-ok: the parsed request body contract is a plain dict
|
||||
tools = data.get("tools")
|
||||
tool_choice = data.get("tool_choice")
|
||||
normalized_tools = (
|
||||
[
|
||||
_convert_tool_envelope(tool, to_chat=to_chat) for tool in tools
|
||||
] # mutable-ok: body's tools stays a plain JSON list
|
||||
if isinstance(tools, list)
|
||||
else tools
|
||||
)
|
||||
normalized_choice = _convert_tool_envelope(tool_choice, to_chat=to_chat)
|
||||
if normalized_tools == tools and normalized_choice == tool_choice:
|
||||
return data
|
||||
replaceable = (("tools", normalized_tools), ("tool_choice", normalized_choice))
|
||||
return {**data, **{key: value for key, value in replaceable if key in data}} # mutable-ok: plain body dict
|
||||
|
||||
|
||||
def _is_chat_completions_body(data: Mapping[str, Any]) -> bool:
|
||||
messages = data.get("messages")
|
||||
if isinstance(messages, list) and len(messages) > 0:
|
||||
return True
|
||||
return "messages" in data and "input" not in data
|
||||
|
||||
|
||||
_CURSOR_THINKING_SEPARATOR = "-thinking-"
|
||||
_CURSOR_FAST_SUFFIX = "-fast"
|
||||
_CURSOR_THINKING_LEVELS: frozenset[str] = frozenset(get_args(REASONING_EFFORT))
|
||||
|
||||
|
||||
class _CursorModelVariant(NamedTuple):
|
||||
base_model: str
|
||||
reasoning_effort: str | None
|
||||
|
||||
|
||||
def _parse_cursor_model_variant(model: str) -> _CursorModelVariant:
|
||||
stripped = model.removesuffix(_CURSOR_FAST_SUFFIX)
|
||||
base, separator, level = stripped.rpartition(_CURSOR_THINKING_SEPARATOR)
|
||||
if separator and base and level in _CURSOR_THINKING_LEVELS:
|
||||
return _CursorModelVariant(base, level)
|
||||
return _CursorModelVariant(stripped, None)
|
||||
|
||||
|
||||
def _router_can_serve(model: str, llm_router: "Router | None") -> bool:
|
||||
if llm_router is None:
|
||||
return False
|
||||
if model in llm_router.model_names or model in llm_router.model_group_alias:
|
||||
return True
|
||||
if model in llm_router.team_public_model_names:
|
||||
return True
|
||||
return bool(llm_router.pattern_router.get_pattern(model))
|
||||
|
||||
|
||||
def _resolve_cursor_model_variant(
|
||||
data: dict, llm_router: "Router | None"
|
||||
) -> dict: # mutable-ok: the parsed request body contract is a plain dict
|
||||
model = data.get("model")
|
||||
if not isinstance(model, str) or _router_can_serve(model, llm_router):
|
||||
return data
|
||||
variant = _parse_cursor_model_variant(model)
|
||||
if variant.base_model == model or not _router_can_serve(variant.base_model, llm_router):
|
||||
return data
|
||||
resolved = {**data, "model": variant.base_model} # mutable-ok: plain body dict
|
||||
if variant.reasoning_effort is None:
|
||||
return resolved
|
||||
if _is_chat_completions_body(data):
|
||||
if "reasoning_effort" in data:
|
||||
return resolved
|
||||
return {**resolved, "reasoning_effort": variant.reasoning_effort} # mutable-ok: plain body dict
|
||||
reasoning = data.get("reasoning")
|
||||
if isinstance(reasoning, dict):
|
||||
if reasoning.get("effort"):
|
||||
return resolved
|
||||
return {**resolved, "reasoning": {**reasoning, "effort": variant.reasoning_effort}} # mutable-ok: same
|
||||
return {**resolved, "reasoning": {"effort": variant.reasoning_effort}} # mutable-ok: plain body dict
|
||||
|
||||
|
||||
@router.post(
|
||||
"/v1/responses",
|
||||
|
|
@ -287,6 +411,33 @@ async def responses_api(
|
|||
)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/cursor/models",
|
||||
dependencies=(_user_api_key_auth_dep,),
|
||||
tags=_RESPONSES_TAGS,
|
||||
)
|
||||
@router.get(
|
||||
"/cursor/v1/models",
|
||||
dependencies=(_user_api_key_auth_dep,),
|
||||
tags=_RESPONSES_TAGS,
|
||||
)
|
||||
async def cursor_model_list(
|
||||
user_api_key_dict: UserAPIKeyAuth = _user_api_key_auth_dep,
|
||||
):
|
||||
"""
|
||||
OpenAI-compatible model listing for the Cursor BYOK base URL.
|
||||
|
||||
Clients pointed at `<proxy>/cursor` as an OpenAI-compatible base URL resolve and
|
||||
verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import model_list
|
||||
|
||||
return await model_list(user_api_key_dict=user_api_key_dict)
|
||||
|
||||
|
||||
@router.post(
|
||||
"/cursor/chat/completions",
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
|
|
@ -298,11 +449,21 @@ async def cursor_chat_completions(
|
|||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
):
|
||||
"""
|
||||
Cursor-specific endpoint that accepts Responses API input format but returns chat completions format.
|
||||
|
||||
This endpoint handles requests from Cursor IDE which sends Responses API format (`input` field)
|
||||
but expects chat completions format response (`choices`, `messages`, etc.).
|
||||
|
||||
Cursor BYOK endpoint. Accepts both request shapes Cursor sends to its OpenAI-compatible
|
||||
base URL and always answers in chat completions format.
|
||||
|
||||
Cursor agent mode sends Responses API format bodies (`input`, flat tool defs, `reasoning`,
|
||||
custom tools) to the chat/completions path while expecting chat completions responses;
|
||||
those are routed through the Responses API pipeline and converted back. Genuine chat
|
||||
completions bodies (`messages` present) are routed through the standard chat completions
|
||||
pipeline, after normalizing each level of the `tools` array and `tool_choice` to the chat
|
||||
completions shapes OpenAI requires. Cursor mixes Responses API shapes into chat bodies
|
||||
per level, independently: a flat tool def (`{"type": "custom", "name": "ApplyPatch", ...}`)
|
||||
gets nested under `custom`, and a flat grammar format
|
||||
(`{"type": "grammar", "definition", "syntax"}`) gets wrapped as
|
||||
`{"type": "grammar", "grammar": {...}}` wherever it appears, including inside tool defs
|
||||
Cursor already sent pre-nested.
|
||||
|
||||
```bash
|
||||
curl -X POST http://localhost:4000/cursor/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
|
|
@ -318,9 +479,11 @@ async def cursor_chat_completions(
|
|||
responses_api_bridge,
|
||||
)
|
||||
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
|
||||
from litellm.proxy.common_utils.http_parsing_utils import _safe_set_request_parsed_body
|
||||
from litellm.proxy.proxy_server import (
|
||||
_read_request_body,
|
||||
async_data_generator,
|
||||
chat_completion,
|
||||
general_settings,
|
||||
llm_router,
|
||||
proxy_config,
|
||||
|
|
@ -332,20 +495,39 @@ async def cursor_chat_completions(
|
|||
user_temperature,
|
||||
version,
|
||||
)
|
||||
from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
||||
data = await _read_request_body(request=request)
|
||||
raw_body = await _read_request_body(request=request)
|
||||
data = _resolve_cursor_model_variant(raw_body, llm_router)
|
||||
|
||||
# Convert 'messages' to 'input' for Responses API compatibility
|
||||
# Cursor sends 'messages' but Responses API expects 'input'
|
||||
if "messages" in data and "input" not in data:
|
||||
data["input"] = data.pop("messages")
|
||||
if _is_chat_completions_body(data):
|
||||
# Genuine chat completions body (Cursor sends these for models whose BYOK it
|
||||
# already fixed); delegate so behavior matches /chat/completions exactly.
|
||||
# Keyed on messages CONTENT, not key presence: Cursor can send a null or
|
||||
# empty messages stub alongside a real agent-mode input array
|
||||
normalized = _normalize_tool_dialect(data, to_chat=True)
|
||||
if normalized is not raw_body:
|
||||
_safe_set_request_parsed_body(request=request, parsed_body=normalized)
|
||||
return await chat_completion(
|
||||
request=request,
|
||||
fastapi_response=fastapi_response,
|
||||
model=None,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
)
|
||||
|
||||
# OpenAI's Responses API rejects chat-completions-only stream_options
|
||||
# (Cursor sends include_usage); usage arrives via response.completed anyway.
|
||||
# Rebuild rather than pop: _read_request_body can return the request-scope
|
||||
# cached parsed-body dict itself, and removing keys from it corrupts the
|
||||
# cache's key snapshot so later readers get an empty body
|
||||
data = {key: value for key, value in data.items() if key != "stream_options"} # mutable-ok: plain body dict
|
||||
|
||||
data = _normalize_tool_dialect(data, to_chat=False)
|
||||
|
||||
processor = ProxyBaseLLMRequestProcessing(data=data)
|
||||
|
||||
def cursor_data_generator(response, user_api_key_dict, request_data):
|
||||
def cursor_data_generator(response, user_api_key_dict, request_data, request=None):
|
||||
"""
|
||||
Custom generator that transforms Responses API streaming chunks to chat completion chunks.
|
||||
|
||||
|
|
@ -353,17 +535,21 @@ async def cursor_chat_completions(
|
|||
to chat completion format that Cursor IDE expects.
|
||||
|
||||
Args:
|
||||
response: The streaming response (BaseResponsesAPIStreamingIterator or other)
|
||||
response: The streaming Responses API event iterator (router-wrapped or not)
|
||||
user_api_key_dict: User API key authentication dict
|
||||
request_data: Request data containing model, logging_obj, etc.
|
||||
request: The originating FastAPI request, forwarded for disconnect handling
|
||||
|
||||
Returns:
|
||||
Async generator that yields SSE-formatted chat completion chunks
|
||||
"""
|
||||
# If response is a BaseResponsesAPIStreamingIterator, transform it first
|
||||
if isinstance(response, BaseResponsesAPIStreamingIterator):
|
||||
# Any async-iterable here is a Responses API event stream needing conversion.
|
||||
# Class-identity checks miss router-wrapped streams (e.g.
|
||||
# HiddenParamsAsyncIteratorWrapper around LiteLLMCompletionStreamingIterator),
|
||||
# which previously leaked raw Responses events to the client.
|
||||
if hasattr(response, "__anext__"):
|
||||
# Transform Responses API iterator to chat completion iterator
|
||||
# Cast to AsyncIterator[str] since BaseResponsesAPIStreamingIterator implements __aiter__/__anext__
|
||||
# Cast to AsyncIterator[str] since the stream implements __aiter__/__anext__
|
||||
completion_stream = responses_api_bridge.transformation_handler.get_model_response_iterator(
|
||||
streaming_response=cast(AsyncIterator[str], response),
|
||||
sync_stream=False,
|
||||
|
|
@ -382,12 +568,14 @@ async def cursor_chat_completions(
|
|||
response=streamwrapper,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
request_data=request_data,
|
||||
request=request,
|
||||
)
|
||||
# Otherwise, use the default generator
|
||||
return async_data_generator(
|
||||
response=response,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
request_data=request_data,
|
||||
request=request,
|
||||
)
|
||||
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -7,6 +7,12 @@ import re
|
|||
from collections.abc import Sequence
|
||||
from typing import Any, Literal, cast
|
||||
|
||||
from openai.types.chat.chat_completion_named_tool_choice_param import (
|
||||
ChatCompletionNamedToolChoiceParam,
|
||||
)
|
||||
from openai.types.chat.chat_completion_named_tool_choice_param import (
|
||||
Function as NamedToolChoiceFunction,
|
||||
)
|
||||
from openai.types.responses import ResponseFunctionToolCall
|
||||
from openai.types.responses.response_create_params import ResponseInputParam
|
||||
from openai.types.responses.tool_param import FunctionToolParam
|
||||
|
|
@ -160,7 +166,17 @@ class LiteLLMCompletionResponsesConfig:
|
|||
elif tool_choice_type == "function":
|
||||
function_name = tool_choice.get("name")
|
||||
if function_name:
|
||||
return {"type": "function", "function": {"name": function_name}}
|
||||
return ChatCompletionNamedToolChoiceParam(
|
||||
type="function", function=NamedToolChoiceFunction(name=function_name)
|
||||
)
|
||||
return "required"
|
||||
elif tool_choice_type == "custom":
|
||||
custom = tool_choice.get("custom")
|
||||
custom_name = tool_choice.get("name") or (custom.get("name") if isinstance(custom, dict) else None)
|
||||
if custom_name:
|
||||
return ChatCompletionNamedToolChoiceParam(
|
||||
type="function", function=NamedToolChoiceFunction(name=custom_name)
|
||||
)
|
||||
return "required"
|
||||
|
||||
# Return as-is for unknown formats
|
||||
|
|
|
|||
|
|
@ -1405,6 +1405,10 @@ class ResponsesAPIStreamEvents(str, Enum):
|
|||
FUNCTION_CALL_ARGUMENTS_DELTA = "response.function_call_arguments.delta"
|
||||
FUNCTION_CALL_ARGUMENTS_DONE = "response.function_call_arguments.done"
|
||||
|
||||
# Custom tool call events (grammar/freeform tools, e.g. Cursor agent tools)
|
||||
CUSTOM_TOOL_CALL_INPUT_DELTA = "response.custom_tool_call_input.delta"
|
||||
CUSTOM_TOOL_CALL_INPUT_DONE = "response.custom_tool_call_input.done"
|
||||
|
||||
# File search events
|
||||
FILE_SEARCH_CALL_IN_PROGRESS = "response.file_search_call.in_progress"
|
||||
FILE_SEARCH_CALL_SEARCHING = "response.file_search_call.searching"
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import json
|
||||
import time
|
||||
from enum import Enum
|
||||
from types import MappingProxyType
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
|
|
@ -1084,6 +1085,43 @@ class ChatCompletionDeltaToolCall(OpenAIObject):
|
|||
setattr(self, key, value)
|
||||
|
||||
|
||||
class _CustomToolCallAccess(OpenAIObject):
|
||||
def __contains__(self, key):
|
||||
return hasattr(self, key)
|
||||
|
||||
def get(self, key, default=None):
|
||||
return getattr(self, key, default)
|
||||
|
||||
def __getitem__(self, key):
|
||||
return getattr(self, key)
|
||||
|
||||
def __setitem__(self, key, value):
|
||||
setattr(self, key, value)
|
||||
|
||||
|
||||
class ChatCompletionCustomToolCallPayload(_CustomToolCallAccess):
|
||||
name: str
|
||||
input: str
|
||||
|
||||
|
||||
class ChatCompletionDeltaCustomToolCallPayload(_CustomToolCallAccess):
|
||||
name: str | None = None
|
||||
input: str | None = None
|
||||
|
||||
|
||||
class ChatCompletionMessageCustomToolCall(_CustomToolCallAccess):
|
||||
id: str
|
||||
type: Literal["custom"] = "custom"
|
||||
custom: ChatCompletionCustomToolCallPayload
|
||||
|
||||
|
||||
class ChatCompletionDeltaCustomToolCall(_CustomToolCallAccess):
|
||||
id: str | None = None
|
||||
type: str | None = None
|
||||
custom: ChatCompletionDeltaCustomToolCallPayload
|
||||
index: int
|
||||
|
||||
|
||||
class ChatCompletionMessageToolCall(OpenAIObject):
|
||||
def __init__(
|
||||
self,
|
||||
|
|
@ -1125,6 +1163,20 @@ class ChatCompletionMessageToolCall(OpenAIObject):
|
|||
setattr(self, key, value)
|
||||
|
||||
|
||||
def is_custom_tool_call_dict(tool_call: Mapping[str, Any]) -> bool:
|
||||
return tool_call.get("type") == "custom" or tool_call.get("custom") is not None
|
||||
|
||||
|
||||
def chat_completion_tool_call_from_dict(
|
||||
tool_call: Mapping[str, Any],
|
||||
) -> "ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall":
|
||||
if is_custom_tool_call_dict(tool_call):
|
||||
return ChatCompletionMessageCustomToolCall(
|
||||
**MappingProxyType({k: v for k, v in tool_call.items() if not (k in ("function", "type") and v is None)})
|
||||
)
|
||||
return ChatCompletionMessageToolCall(**tool_call)
|
||||
|
||||
|
||||
from openai.types.chat.chat_completion_audio import ChatCompletionAudio
|
||||
|
||||
|
||||
|
|
@ -1177,7 +1229,9 @@ def add_provider_specific_fields(object: BaseModel, provider_specific_fields: Op
|
|||
class Message(SafeAttributeModel, OpenAIObject):
|
||||
content: Optional[str]
|
||||
role: Literal["assistant", "user", "system", "tool", "function"]
|
||||
tool_calls: Optional[List[ChatCompletionMessageToolCall]]
|
||||
tool_calls: Optional[
|
||||
List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]]
|
||||
] # mutable-ok: public pydantic response field; only the union member is new
|
||||
function_call: Optional[FunctionCall]
|
||||
audio: Optional[ChatCompletionAudioResponse] = None
|
||||
images: Optional[List[ImageURLListItem]] = None
|
||||
|
|
@ -1208,7 +1262,7 @@ class Message(SafeAttributeModel, OpenAIObject):
|
|||
"function_call": (FunctionCall(**function_call) if function_call is not None else None),
|
||||
"tool_calls": (
|
||||
[
|
||||
(ChatCompletionMessageToolCall(**tool_call) if isinstance(tool_call, dict) else tool_call)
|
||||
(chat_completion_tool_call_from_dict(tool_call) if isinstance(tool_call, dict) else tool_call)
|
||||
for tool_call in tool_calls
|
||||
]
|
||||
if tool_calls is not None and len(tool_calls) > 0
|
||||
|
|
@ -1301,7 +1355,9 @@ class Delta(SafeAttributeModel, OpenAIObject):
|
|||
content: Optional[str]
|
||||
role: Optional[str]
|
||||
function_call: Optional[FunctionCall]
|
||||
tool_calls: Optional[List[ChatCompletionDeltaToolCall]]
|
||||
tool_calls: Optional[
|
||||
List[Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]]
|
||||
] # mutable-ok: public pydantic response field; only the union member is new
|
||||
audio: Optional[ChatCompletionAudioResponse]
|
||||
images: Optional[List[ImageURLListItem]]
|
||||
annotations: Optional[List[ChatCompletionAnnotation]]
|
||||
|
|
@ -1338,18 +1394,29 @@ class Delta(SafeAttributeModel, OpenAIObject):
|
|||
if function_call is not None and isinstance(function_call, dict):
|
||||
function_call = FunctionCall(**function_call)
|
||||
|
||||
if tool_calls is not None and isinstance(tool_calls, list):
|
||||
coerced_tool_calls: List[ChatCompletionDeltaToolCall] = []
|
||||
if tool_calls is not None and isinstance(tool_calls, (list, tuple)):
|
||||
coerced_tool_calls: List[
|
||||
Union[ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall]
|
||||
] = [] # mutable-ok: public Delta.tool_calls contract is a list
|
||||
current_index = 0
|
||||
for tool_call in tool_calls:
|
||||
if isinstance(tool_call, dict):
|
||||
if tool_call.get("index", None) is None:
|
||||
tool_call["index"] = current_index
|
||||
current_index += 1
|
||||
if tool_call.get("type", None) is None:
|
||||
tool_call["type"] = "function"
|
||||
coerced_tool_calls.append(ChatCompletionDeltaToolCall(**tool_call))
|
||||
elif isinstance(tool_call, ChatCompletionDeltaToolCall):
|
||||
if is_custom_tool_call_dict(tool_call):
|
||||
coerced_tool_calls.append(
|
||||
ChatCompletionDeltaCustomToolCall(
|
||||
**MappingProxyType(
|
||||
{k: v for k, v in tool_call.items() if not (k == "function" and v is None)}
|
||||
)
|
||||
)
|
||||
)
|
||||
else:
|
||||
if tool_call.get("type", None) is None:
|
||||
tool_call["type"] = "function"
|
||||
coerced_tool_calls.append(ChatCompletionDeltaToolCall(**tool_call))
|
||||
elif isinstance(tool_call, (ChatCompletionDeltaToolCall, ChatCompletionDeltaCustomToolCall)):
|
||||
coerced_tool_calls.append(tool_call)
|
||||
tool_calls = coerced_tool_calls
|
||||
|
||||
|
|
|
|||
|
|
@ -69,7 +69,7 @@
|
|||
"limit": 4
|
||||
},
|
||||
"C405": {
|
||||
"limit": 22
|
||||
"limit": 21
|
||||
},
|
||||
"C408": {
|
||||
"limit": 14
|
||||
|
|
@ -81,7 +81,7 @@
|
|||
"limit": 4
|
||||
},
|
||||
"C901": {
|
||||
"limit": 312
|
||||
"limit": 311
|
||||
},
|
||||
"D419": {
|
||||
"limit": 9
|
||||
|
|
@ -243,7 +243,7 @@
|
|||
"limit": 0
|
||||
},
|
||||
"RUF046": {
|
||||
"limit": 6
|
||||
"limit": 5
|
||||
},
|
||||
"RUF051": {
|
||||
"limit": 0
|
||||
|
|
@ -312,7 +312,7 @@
|
|||
"limit": 547
|
||||
},
|
||||
"TRY004": {
|
||||
"limit": 98
|
||||
"limit": 97
|
||||
},
|
||||
"TRY201": {
|
||||
"limit": 420
|
||||
|
|
|
|||
|
|
@ -2475,6 +2475,15 @@ def test_map_optional_params_tool_choice_chat_nested_to_responses_api():
|
|||
{"type": "function", "name": "foo"},
|
||||
),
|
||||
({"type": "required"}, {"type": "required"}),
|
||||
(
|
||||
{"type": "custom", "custom": {"name": "ApplyPatch"}},
|
||||
{"type": "custom", "name": "ApplyPatch"},
|
||||
),
|
||||
(
|
||||
{"type": "custom", "name": "ApplyPatch"},
|
||||
{"type": "custom", "name": "ApplyPatch"},
|
||||
),
|
||||
({"type": "custom"}, {"type": "custom"}),
|
||||
],
|
||||
)
|
||||
def test_normalize_tool_choice_for_responses_api(tool_choice, expected):
|
||||
|
|
@ -2962,3 +2971,432 @@ async def test_acompletion_bridge_normalizes_stream_options_on_the_wire(
|
|||
assert "stream_options" not in request_body
|
||||
else:
|
||||
assert request_body["stream_options"] == expected_wire_stream_options
|
||||
|
||||
|
||||
def test_chunk_parser_custom_tool_call_stream_sequence():
|
||||
"""Cursor agent mode drives grammar/freeform ``custom_tool_call`` items (e.g. its
|
||||
ApplyPatch tool). The stream converter must surface them as chat-completions
|
||||
tool_call deltas: the added event opens the call (id from ``call_id``, name, empty
|
||||
arguments), each ``custom_tool_call_input.delta`` streams arguments, the done event
|
||||
must NOT finish the stream, and ``response.completed`` must report
|
||||
finish_reason="tool_calls". Before the fix every one of these events fell through
|
||||
to an empty-content chunk and the completed event said "stop", so Cursor never saw
|
||||
the tool call and agent mode stalled."""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
OpenAiResponsesToChatCompletionStreamIterator,
|
||||
)
|
||||
|
||||
iterator = OpenAiResponsesToChatCompletionStreamIterator(
|
||||
streaming_response=None, sync_stream=True
|
||||
)
|
||||
|
||||
added = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.output_item.added",
|
||||
"output_index": 1,
|
||||
"item": {
|
||||
"type": "custom_tool_call",
|
||||
"id": "ctc_1",
|
||||
"call_id": "call_patch1",
|
||||
"name": "ApplyPatch",
|
||||
"input": "",
|
||||
},
|
||||
}
|
||||
)
|
||||
tool_call = added.choices[0].delta.tool_calls[0]
|
||||
assert tool_call.id == "call_patch1"
|
||||
assert tool_call.type == "function"
|
||||
assert tool_call.function.name == "ApplyPatch"
|
||||
assert tool_call.function.arguments == ""
|
||||
assert tool_call.index == 0
|
||||
assert added.choices[0].finish_reason is None
|
||||
|
||||
delta = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.custom_tool_call_input.delta",
|
||||
"output_index": 1,
|
||||
"delta": "*** Begin Patch",
|
||||
}
|
||||
)
|
||||
delta_tool_call = delta.choices[0].delta.tool_calls[0]
|
||||
assert delta_tool_call.function.arguments == "*** Begin Patch"
|
||||
assert delta_tool_call.index == 0
|
||||
assert delta.choices[0].finish_reason is None
|
||||
|
||||
done = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.output_item.done",
|
||||
"output_index": 1,
|
||||
"item": {
|
||||
"type": "custom_tool_call",
|
||||
"call_id": "call_patch1",
|
||||
"name": "ApplyPatch",
|
||||
"input": "*** Begin Patch",
|
||||
},
|
||||
}
|
||||
)
|
||||
assert done.choices[0].finish_reason is None
|
||||
|
||||
completed = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.completed",
|
||||
"response": {
|
||||
"output": [
|
||||
{"type": "reasoning", "id": "rs_1"},
|
||||
{"type": "custom_tool_call", "call_id": "call_patch1"},
|
||||
],
|
||||
"usage": {"input_tokens": 7, "output_tokens": 3, "total_tokens": 10},
|
||||
},
|
||||
}
|
||||
)
|
||||
assert completed.choices[0].finish_reason == "tool_calls"
|
||||
assert completed.usage is not None
|
||||
assert completed.usage.total_tokens == 10
|
||||
|
||||
|
||||
def test_chunk_parser_remaps_tool_call_indices_sequentially():
|
||||
"""Responses API output_index counts every output item, so a reasoning model's
|
||||
first tool call arrives at output_index >= 1. Chat-completions clients accumulate
|
||||
streamed tool_calls by index and expect the first call at 0; Cursor agent mode
|
||||
misplaces calls when indices start above 0 (the community BYOK bridge assigns its
|
||||
own sequential indices for the same reason). The iterator must remap each distinct
|
||||
output_index to the next sequential slot and route argument deltas to the mapped
|
||||
slot."""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
OpenAiResponsesToChatCompletionStreamIterator,
|
||||
)
|
||||
|
||||
iterator = OpenAiResponsesToChatCompletionStreamIterator(
|
||||
streaming_response=None, sync_stream=True
|
||||
)
|
||||
|
||||
first = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.output_item.added",
|
||||
"output_index": 2,
|
||||
"item": {
|
||||
"type": "function_call",
|
||||
"id": "fc_1",
|
||||
"call_id": "call_read1",
|
||||
"name": "read_file",
|
||||
"arguments": "",
|
||||
},
|
||||
}
|
||||
)
|
||||
assert first.choices[0].delta.tool_calls[0].index == 0
|
||||
|
||||
first_args = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.function_call_arguments.delta",
|
||||
"output_index": 2,
|
||||
"delta": '{"path":',
|
||||
}
|
||||
)
|
||||
assert first_args.choices[0].delta.tool_calls[0].index == 0
|
||||
|
||||
second = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.output_item.added",
|
||||
"output_index": 4,
|
||||
"item": {
|
||||
"type": "function_call",
|
||||
"id": "fc_2",
|
||||
"call_id": "call_grep1",
|
||||
"name": "grep",
|
||||
"arguments": "",
|
||||
},
|
||||
}
|
||||
)
|
||||
assert second.choices[0].delta.tool_calls[0].index == 1
|
||||
|
||||
second_args = iterator.chunk_parser(
|
||||
{
|
||||
"type": "response.function_call_arguments.delta",
|
||||
"output_index": 4,
|
||||
"delta": '{"pattern":',
|
||||
}
|
||||
)
|
||||
assert second_args.choices[0].delta.tool_calls[0].index == 1
|
||||
|
||||
|
||||
def test_convert_response_output_custom_tool_call_to_tool_calls_choice():
|
||||
"""Non-streaming twin of the custom_tool_call fix: a typed ResponseCustomToolCall
|
||||
output item must become a chat tool_call (arguments = the raw custom input string,
|
||||
id = call_id) in a finish_reason="tool_calls" choice instead of being silently
|
||||
dropped, which left Cursor agent mode with an empty assistant message."""
|
||||
from openai.types.responses import ResponseCustomToolCall
|
||||
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
item = ResponseCustomToolCall(
|
||||
type="custom_tool_call",
|
||||
id="ctc_9",
|
||||
call_id="call_custom9",
|
||||
name="ApplyPatch",
|
||||
input="*** Begin Patch\n*** End Patch",
|
||||
)
|
||||
|
||||
choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices([item])
|
||||
|
||||
assert len(choices) == 1
|
||||
choice = choices[0]
|
||||
assert choice.finish_reason == "tool_calls"
|
||||
tool_call = choice.message.tool_calls[0]
|
||||
assert tool_call.id == "call_custom9"
|
||||
assert tool_call.function.name == "ApplyPatch"
|
||||
assert tool_call.function.arguments == "*** Begin Patch\n*** End Patch"
|
||||
|
||||
|
||||
def test_convert_response_output_accumulates_raw_tool_calls_into_one_choice():
|
||||
"""Raw dict and generic-pydantic tool-call items must accumulate into the single
|
||||
trailing tool_calls choice exactly like typed items. Emitting one choice per tool
|
||||
call (the old raw-dict behavior) hid every call after choices[0] from chat
|
||||
clients, which read only the first choice; a multi-tool agent turn through the
|
||||
completion bridge lost all but one call."""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
items = [
|
||||
{
|
||||
"type": "function_call",
|
||||
"id": "fc_1",
|
||||
"call_id": "call_read42",
|
||||
"name": "read_file",
|
||||
"arguments": '{"path": "a.py"}',
|
||||
},
|
||||
{
|
||||
"type": "custom_tool_call",
|
||||
"id": "ctc_1",
|
||||
"call_id": "call_patch42",
|
||||
"name": "ApplyPatch",
|
||||
"input": "*** Begin Patch",
|
||||
},
|
||||
]
|
||||
|
||||
choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
|
||||
items,
|
||||
handle_raw_dict_callback=handler._handle_raw_dict_response_item,
|
||||
)
|
||||
|
||||
assert len(choices) == 1
|
||||
choice = choices[0]
|
||||
assert choice.finish_reason == "tool_calls"
|
||||
tool_calls = choice.message.tool_calls
|
||||
assert len(tool_calls) == 2
|
||||
assert tool_calls[0].id == "call_read42"
|
||||
assert tool_calls[0].function.name == "read_file"
|
||||
assert tool_calls[0].function.arguments == '{"path": "a.py"}'
|
||||
assert tool_calls[1].id == "call_patch42"
|
||||
assert tool_calls[1].function.name == "ApplyPatch"
|
||||
assert tool_calls[1].function.arguments == "*** Begin Patch"
|
||||
|
||||
|
||||
def test_convert_response_output_generic_pydantic_message_item():
|
||||
"""litellm's completion bridge (used for non-Responses-native providers behind the
|
||||
router) emits GenericResponseOutputItem pydantic models rather than openai SDK
|
||||
classes. The converter must normalize unrecognized pydantic items through the
|
||||
raw-dict handler instead of dropping them; dropping them made transform_response
|
||||
raise 'Unknown items in responses API response' on an otherwise-successful
|
||||
completion (hit live via /cursor/chat/completions multi-turn tool round trips)."""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
from litellm.types.responses.main import GenericResponseOutputItem, OutputText
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
item = GenericResponseOutputItem(
|
||||
type="message",
|
||||
id="msg_generic1",
|
||||
status="completed",
|
||||
role="assistant",
|
||||
content=[OutputText(type="output_text", text="42", annotations=[])],
|
||||
)
|
||||
|
||||
choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
|
||||
[item],
|
||||
handle_raw_dict_callback=handler._handle_raw_dict_response_item,
|
||||
)
|
||||
|
||||
assert len(choices) == 1
|
||||
assert choices[0].message.content == "42"
|
||||
assert choices[0].finish_reason == "stop"
|
||||
|
||||
|
||||
def test_convert_tools_to_responses_format_flattens_nested_custom_tool():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
tools = [
|
||||
{
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}},
|
||||
},
|
||||
{"type": "function", "function": {"name": "f", "parameters": {"type": "object"}}},
|
||||
]
|
||||
converted = handler._convert_tools_to_responses_format(tools)
|
||||
assert converted[0] == {
|
||||
"type": "custom",
|
||||
"name": "ApplyPatch",
|
||||
"description": "V4A patch",
|
||||
"format": {"type": "text"},
|
||||
}
|
||||
assert converted[1]["type"] == "function"
|
||||
assert converted[1]["name"] == "f"
|
||||
|
||||
|
||||
def test_convert_tools_to_responses_format_flattens_custom_tool_without_optional_keys():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
converted = handler._convert_tools_to_responses_format([{"type": "custom", "custom": {"name": "Minimal"}}])
|
||||
assert converted[0] == {"type": "custom", "name": "Minimal"}
|
||||
|
||||
|
||||
def test_convert_tools_to_responses_format_unwraps_nested_grammar_format():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
converted = handler._convert_tools_to_responses_format(
|
||||
[
|
||||
{
|
||||
"type": "custom",
|
||||
"custom": {
|
||||
"name": "ApplyPatch",
|
||||
"format": {
|
||||
"type": "grammar",
|
||||
"grammar": {"definition": "start: patch", "syntax": "lark"},
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
assert converted[0] == {
|
||||
"type": "custom",
|
||||
"name": "ApplyPatch",
|
||||
"format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"},
|
||||
}
|
||||
|
||||
|
||||
def test_convert_tools_to_responses_format_text_format_passes_through():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
converted = handler._convert_tools_to_responses_format(
|
||||
[{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}]
|
||||
)
|
||||
assert converted[0] == {"type": "custom", "name": "A", "format": {"type": "text"}}
|
||||
|
||||
|
||||
def test_convert_chat_completion_messages_maps_custom_tool_call_history():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
|
||||
[
|
||||
{"role": "user", "content": "use ApplyPatch"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_c",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch"},
|
||||
},
|
||||
{
|
||||
"id": "call_f",
|
||||
"type": "function",
|
||||
"function": {"name": "shell", "arguments": '{"cmd": "ls"}'},
|
||||
},
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "call_c", "content": "patch applied"},
|
||||
{"role": "tool", "tool_call_id": "call_f", "content": "a.py"},
|
||||
]
|
||||
)
|
||||
assert {
|
||||
"type": "custom_tool_call",
|
||||
"call_id": "call_c",
|
||||
"name": "ApplyPatch",
|
||||
"input": "*** Begin Patch",
|
||||
} in input_items
|
||||
assert {"type": "custom_tool_call_output", "call_id": "call_c", "output": "patch applied"} in input_items
|
||||
assert {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'} in input_items
|
||||
assert {
|
||||
"type": "function_call_output",
|
||||
"call_id": "call_f",
|
||||
"output": [{"type": "input_text", "text": "a.py"}],
|
||||
} in input_items
|
||||
|
||||
|
||||
def test_convert_chat_completion_messages_still_rejects_unknown_tool_call_shape():
|
||||
import pytest
|
||||
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
with pytest.raises(ValueError, match="tool call not supported"):
|
||||
handler.convert_chat_completion_messages_to_responses_api(
|
||||
[{"role": "assistant", "tool_calls": [{"id": "call_x", "type": "mystery"}]}]
|
||||
)
|
||||
|
||||
|
||||
def test_output_item_done_stateless_emits_complete_tool_call():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
OpenAiResponsesToChatCompletionStreamIterator,
|
||||
)
|
||||
|
||||
for item, expected_name, expected_args in (
|
||||
(
|
||||
{"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'},
|
||||
"shell",
|
||||
'{"cmd": "ls"}',
|
||||
),
|
||||
(
|
||||
{"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"},
|
||||
"ApplyPatch",
|
||||
"*** Begin Patch",
|
||||
),
|
||||
):
|
||||
chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
|
||||
{"type": "response.output_item.done", "output_index": 2, "item": item}
|
||||
)
|
||||
tool_calls = chunk.choices[0].delta.tool_calls
|
||||
assert tool_calls is not None and len(tool_calls) == 1
|
||||
assert tool_calls[0].id == item["call_id"]
|
||||
assert tool_calls[0].function.name == expected_name
|
||||
assert tool_calls[0].function.arguments == expected_args
|
||||
assert tool_calls[0].index == 2
|
||||
assert chunk.choices[0].finish_reason is None
|
||||
|
||||
|
||||
def test_output_item_done_with_stream_map_keeps_empty_delta():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
OpenAiResponsesToChatCompletionStreamIterator,
|
||||
)
|
||||
|
||||
chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
|
||||
{
|
||||
"type": "response.output_item.done",
|
||||
"output_index": 0,
|
||||
"item": {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "x"},
|
||||
},
|
||||
tool_call_index_map={0: 0},
|
||||
)
|
||||
assert chunk.choices[0].delta.tool_calls is None
|
||||
assert chunk.choices[0].finish_reason is None
|
||||
|
|
|
|||
|
|
@ -257,3 +257,4 @@ def test_translate_responses_chunk_passthrough_chat_completion_chunk():
|
|||
|
||||
assert result.choices[0].delta.content == "Hi! How can I help?"
|
||||
assert result.choices[0].finish_reason is None
|
||||
|
||||
|
|
|
|||
53
tests/test_litellm/integrations/test_helicone.py
Normal file
53
tests/test_litellm/integrations/test_helicone.py
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
import os
|
||||
import sys
|
||||
import types
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../.."))
|
||||
|
||||
from litellm.integrations.helicone import HeliconeLogger
|
||||
|
||||
|
||||
def _claude_mapping(messages, response_obj):
|
||||
logger = HeliconeLogger.__new__(HeliconeLogger)
|
||||
return logger.claude_mapping(model="gpt-5.6", messages=messages, response_obj=response_obj)
|
||||
|
||||
|
||||
def test_claude_mapping_serializes_custom_tool_calls(monkeypatch):
|
||||
"""
|
||||
Stub the anthropic module unconditionally: the SDK may be absent (it lives in the
|
||||
proxy-runtime extra), and the tests/test_litellm/llms/anthropic test package can
|
||||
shadow it on sys.path, so an import probe proves nothing about the real SDK.
|
||||
"""
|
||||
stub = types.ModuleType("anthropic")
|
||||
stub.HUMAN_PROMPT = "\n\nHuman:"
|
||||
stub.AI_PROMPT = "\n\nAssistant:"
|
||||
monkeypatch.setitem(sys.modules, "anthropic", stub)
|
||||
response_obj = {
|
||||
"id": "chatcmpl-1",
|
||||
"choices": [
|
||||
{
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_c",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch"},
|
||||
},
|
||||
{
|
||||
"id": "call_f",
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "arguments": '{"path": "a.py"}'},
|
||||
},
|
||||
],
|
||||
},
|
||||
"finish_reason": "tool_calls",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 2},
|
||||
}
|
||||
mapped = _claude_mapping([{"role": "user", "content": "hi"}], response_obj)
|
||||
tool_use_blocks = [b for b in mapped["content"] if b["type"] == "tool_use"]
|
||||
assert {"type": "tool_use", "id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"} in tool_use_blocks
|
||||
assert {"type": "tool_use", "id": "call_f", "name": "read_file", "input": '{"path": "a.py"}'} in tool_use_blocks
|
||||
40
tests/test_litellm/integrations/test_lunary.py
Normal file
40
tests/test_litellm/integrations/test_lunary.py
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../.."))
|
||||
|
||||
from litellm.integrations.lunary import parse_tool_calls
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
Function,
|
||||
)
|
||||
|
||||
|
||||
def test_parse_tool_calls_serializes_custom_tool_calls():
|
||||
custom_call = ChatCompletionMessageCustomToolCall(
|
||||
id="call_c",
|
||||
custom={"name": "ApplyPatch", "input": "*** Begin Patch"},
|
||||
)
|
||||
function_call = ChatCompletionMessageToolCall(
|
||||
id="call_f",
|
||||
type="function",
|
||||
function=Function(name="read_file", arguments='{"path": "a.py"}'),
|
||||
)
|
||||
parsed = parse_tool_calls([custom_call, function_call])
|
||||
assert parsed == [
|
||||
{
|
||||
"type": "custom",
|
||||
"id": "call_c",
|
||||
"function": {"name": "ApplyPatch", "arguments": "*** Begin Patch"},
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"id": "call_f",
|
||||
"function": {"name": "read_file", "arguments": '{"path": "a.py"}'},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def test_parse_tool_calls_none_passthrough():
|
||||
assert parse_tool_calls(None) is None
|
||||
|
|
@ -0,0 +1,104 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../../.."))
|
||||
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
|
||||
_handle_invalid_parallel_tool_calls,
|
||||
_should_convert_tool_call_to_json_mode,
|
||||
convert_to_model_response_object,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
Function,
|
||||
ModelResponse,
|
||||
)
|
||||
|
||||
OPENAI_CUSTOM_TOOL_CALL_RESPONSE = {
|
||||
"id": "chatcmpl-abc",
|
||||
"created": 1784657740,
|
||||
"model": "gpt-5.6",
|
||||
"object": "chat.completion",
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "tool_calls",
|
||||
"index": 0,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_njxQ",
|
||||
"type": "custom",
|
||||
"custom": {
|
||||
"name": "ApplyPatch",
|
||||
"input": "*** Begin Patch\n*** Update File: main.py\n@@\n+def hello():\n+ print(\"Hello\")\n*** End Patch\n",
|
||||
},
|
||||
}
|
||||
],
|
||||
"refusal": None,
|
||||
"annotations": [],
|
||||
},
|
||||
}
|
||||
],
|
||||
"usage": {"completion_tokens": 10, "prompt_tokens": 5, "total_tokens": 15},
|
||||
}
|
||||
|
||||
|
||||
def test_convert_openai_custom_tool_call_response():
|
||||
result = convert_to_model_response_object(
|
||||
response_object=OPENAI_CUSTOM_TOOL_CALL_RESPONSE,
|
||||
model_response_object=ModelResponse(),
|
||||
response_type="completion",
|
||||
)
|
||||
tool_calls = result.choices[0].message.tool_calls
|
||||
assert len(tool_calls) == 1
|
||||
assert isinstance(tool_calls[0], ChatCompletionMessageCustomToolCall)
|
||||
dumped = tool_calls[0].model_dump()
|
||||
assert dumped == OPENAI_CUSTOM_TOOL_CALL_RESPONSE["choices"][0]["message"]["tool_calls"][0]
|
||||
assert result.choices[0].finish_reason == "tool_calls"
|
||||
|
||||
|
||||
def test_should_convert_tool_call_to_json_mode_ignores_custom_tool_call():
|
||||
custom_tool_call = ChatCompletionMessageCustomToolCall(
|
||||
id="call_c",
|
||||
custom={"name": "ApplyPatch", "input": "patch"},
|
||||
)
|
||||
assert (
|
||||
_should_convert_tool_call_to_json_mode(
|
||||
tool_calls=[custom_tool_call],
|
||||
convert_tool_call_to_json_mode=True,
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
def test_should_convert_tool_call_to_json_mode_still_matches_response_format_tool():
|
||||
response_format_call = ChatCompletionMessageToolCall(
|
||||
id="call_f",
|
||||
type="function",
|
||||
function=Function(name=RESPONSE_FORMAT_TOOL_NAME, arguments='{"answer": 4}'),
|
||||
)
|
||||
assert (
|
||||
_should_convert_tool_call_to_json_mode(
|
||||
tool_calls=[response_format_call],
|
||||
convert_tool_call_to_json_mode=True,
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
|
||||
def test_handle_invalid_parallel_tool_calls_skips_custom_tool_calls():
|
||||
custom_tool_call = ChatCompletionMessageCustomToolCall(
|
||||
id="call_c",
|
||||
custom={"name": "ApplyPatch", "input": "patch"},
|
||||
)
|
||||
function_tool_call = ChatCompletionMessageToolCall(
|
||||
id="call_f",
|
||||
type="function",
|
||||
function=Function(name="get_weather", arguments='{"city": "SF"}'),
|
||||
)
|
||||
result = _handle_invalid_parallel_tool_calls([custom_tool_call, function_tool_call])
|
||||
assert result == [custom_tool_call, function_tool_call]
|
||||
|
|
@ -721,3 +721,48 @@ class TestUnpackLegacyDefs:
|
|||
out = unpack_legacy_defs(schema)
|
||||
assert "components" not in out
|
||||
assert out["properties"]["r0"]["properties"]["p0"] == {"type": "string"}
|
||||
|
||||
|
||||
class TestCustomToolFormatShapeConversion:
|
||||
def test_flat_grammar_to_chat_shape(self):
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_chat_shape,
|
||||
)
|
||||
|
||||
assert convert_custom_tool_format_to_chat_shape(
|
||||
{"type": "grammar", "definition": "start: patch", "syntax": "lark"}
|
||||
) == {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}}
|
||||
|
||||
def test_nested_grammar_to_responses_shape(self):
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_responses_shape,
|
||||
)
|
||||
|
||||
assert convert_custom_tool_format_to_responses_shape(
|
||||
{"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "regex"}}
|
||||
) == {"type": "grammar", "definition": "start: patch", "syntax": "regex"}
|
||||
|
||||
def test_both_directions_are_idempotent_and_pass_text_through(self):
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_chat_shape,
|
||||
convert_custom_tool_format_to_responses_shape,
|
||||
)
|
||||
|
||||
flat = {"type": "grammar", "definition": "d", "syntax": "lark"}
|
||||
nested = {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}}
|
||||
text = {"type": "text"}
|
||||
assert convert_custom_tool_format_to_chat_shape(nested) == nested
|
||||
assert convert_custom_tool_format_to_responses_shape(flat) == flat
|
||||
assert convert_custom_tool_format_to_chat_shape(text) == text
|
||||
assert convert_custom_tool_format_to_responses_shape(text) == text
|
||||
assert convert_custom_tool_format_to_chat_shape(convert_custom_tool_format_to_responses_shape(nested)) == nested
|
||||
|
||||
def test_unrecognized_formats_pass_through(self):
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_custom_tool_format_to_chat_shape,
|
||||
convert_custom_tool_format_to_responses_shape,
|
||||
)
|
||||
|
||||
for weird in ({}, {"type": "grammar"}, {"type": "future_format", "x": 1}):
|
||||
assert convert_custom_tool_format_to_chat_shape(dict(weird)) in (weird, {"type": "grammar", "grammar": {}})
|
||||
assert convert_custom_tool_format_to_responses_shape(dict(weird)) == weird
|
||||
|
|
|
|||
|
|
@ -992,3 +992,75 @@ def test_cost_field_in_usage_chunks():
|
|||
assert usage.cost == 0.00025
|
||||
assert usage.prompt_tokens == 10
|
||||
assert usage.completion_tokens == 5
|
||||
|
||||
|
||||
def test_get_combined_tool_content_custom_tool_call():
|
||||
from litellm.litellm_core_utils.streaming_chunk_builder_utils import ChunkProcessor
|
||||
from litellm.types.utils import ChatCompletionMessageCustomToolCall
|
||||
|
||||
processor = ChunkProcessor.__new__(ChunkProcessor)
|
||||
tool_call_chunks = [
|
||||
{
|
||||
"choices": [
|
||||
{
|
||||
"delta": {
|
||||
"tool_calls": [
|
||||
{
|
||||
"index": 0,
|
||||
"id": "call_TBs",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": ""},
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
{"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** Begin Patch\n"}}]}}]},
|
||||
{"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** End Patch\n"}}]}}]},
|
||||
]
|
||||
combined = processor.get_combined_tool_content(tool_call_chunks)
|
||||
assert len(combined) == 1
|
||||
assert isinstance(combined[0], ChatCompletionMessageCustomToolCall)
|
||||
assert combined[0].model_dump() == {
|
||||
"id": "call_TBs",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch\n*** End Patch\n"},
|
||||
}
|
||||
|
||||
|
||||
def test_get_combined_tool_content_custom_tool_call_without_type_field():
|
||||
"""Delta coercion classifies a tool-call chunk as custom from its ``custom`` payload
|
||||
alone (``type`` may never arrive on any chunk). The assembler must use the same
|
||||
evidence; requiring ``type == "custom"`` dropped the whole tool call from the
|
||||
combined message (it matched neither the custom nor the function branch)."""
|
||||
from litellm.litellm_core_utils.streaming_chunk_builder_utils import ChunkProcessor
|
||||
from litellm.types.utils import ChatCompletionMessageCustomToolCall
|
||||
|
||||
processor = ChunkProcessor.__new__(ChunkProcessor)
|
||||
tool_call_chunks = [
|
||||
{
|
||||
"choices": [
|
||||
{
|
||||
"delta": {
|
||||
"tool_calls": [
|
||||
{
|
||||
"index": 0,
|
||||
"id": "call_TBs",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin"},
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
{"choices": [{"delta": {"tool_calls": [{"index": 0, "custom": {"input": " Patch"}}]}}]},
|
||||
]
|
||||
combined = processor.get_combined_tool_content(tool_call_chunks)
|
||||
assert len(combined) == 1
|
||||
assert isinstance(combined[0], ChatCompletionMessageCustomToolCall)
|
||||
assert combined[0].model_dump() == {
|
||||
"id": "call_TBs",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch"},
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3355,3 +3355,102 @@ async def test_transport_read_error_before_finish_reason_raises(logging_obj: Log
|
|||
if chunk.choices and chunk.choices[0].finish_reason
|
||||
]
|
||||
assert fabricated_finish_reasons == []
|
||||
|
||||
|
||||
def test_openai_custom_tool_call_stream_deltas_survive_conversion(logging_obj: Logging):
|
||||
"""
|
||||
Regression test: OpenAI chat completions custom tool calls stream as
|
||||
delta.tool_calls entries with a `custom` payload and NO `function` key.
|
||||
Delta() used to raise on those dicts and chunk_creator's except branch
|
||||
replaced the choice with an empty Delta, silently dropping the entire
|
||||
tool call from the client stream.
|
||||
"""
|
||||
from openai.types.chat.chat_completion_chunk import ChatCompletionChunk
|
||||
|
||||
from litellm.types.utils import ChatCompletionDeltaCustomToolCall
|
||||
|
||||
raw_chunks = [
|
||||
{
|
||||
"id": "chatcmpl-custom",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1784657671,
|
||||
"model": "gpt-5.6",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"index": 0,
|
||||
"id": "call_TBs",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": ""},
|
||||
}
|
||||
],
|
||||
},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-custom",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1784657671,
|
||||
"model": "gpt-5.6",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** Begin Patch\n"}}]},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-custom",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1784657671,
|
||||
"model": "gpt-5.6",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"tool_calls": [{"index": 0, "custom": {"input": "*** End Patch\n"}}]},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"id": "chatcmpl-custom",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1784657671,
|
||||
"model": "gpt-5.6",
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
|
||||
},
|
||||
]
|
||||
sdk_chunks = [ChatCompletionChunk.construct(**raw) for raw in raw_chunks]
|
||||
first_dumped = sdk_chunks[0].choices[0].model_dump()
|
||||
assert first_dumped["delta"]["tool_calls"][0]["custom"] == {"name": "ApplyPatch", "input": ""}
|
||||
|
||||
wrapper = CustomStreamWrapper(
|
||||
completion_stream=iter(sdk_chunks),
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
emitted = list(wrapper)
|
||||
tool_call_deltas = [
|
||||
chunk.choices[0].delta.tool_calls[0]
|
||||
for chunk in emitted
|
||||
if chunk.choices and chunk.choices[0].delta and chunk.choices[0].delta.tool_calls
|
||||
]
|
||||
assert len(tool_call_deltas) == 3
|
||||
assert isinstance(tool_call_deltas[0], ChatCompletionDeltaCustomToolCall)
|
||||
assert tool_call_deltas[0].id == "call_TBs"
|
||||
assert tool_call_deltas[0].type == "custom"
|
||||
assert tool_call_deltas[0].custom.name == "ApplyPatch"
|
||||
combined_input = "".join(tc.custom.input or "" for tc in tool_call_deltas)
|
||||
assert combined_input == "*** Begin Patch\n*** End Patch\n"
|
||||
finish_reasons = [chunk.choices[0].finish_reason for chunk in emitted if chunk.choices]
|
||||
assert "tool_calls" in finish_reasons
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from unittest.mock import AsyncMock, MagicMock, patch
|
|||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
import litellm
|
||||
from litellm.proxy.proxy_server import app
|
||||
|
||||
|
||||
|
|
@ -711,3 +712,859 @@ class TestManagedResponsesSameProvider:
|
|||
call_kwargs: dict = {}
|
||||
handler._inject_credentials(call_kwargs, model="vertex_ai/gemini-2.0-flash")
|
||||
assert "custom_llm_provider" not in call_kwargs
|
||||
|
||||
|
||||
def _auth_override():
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
|
||||
return UserAPIKeyAuth(api_key="sk-test-cursor", user_id="cursor-user")
|
||||
|
||||
|
||||
def test_cursor_chat_completions_messages_body_uses_chat_pipeline():
|
||||
"""A genuine chat-completions body (``messages`` present; what Cursor sends for
|
||||
models whose BYOK it already fixed) must run through the standard chat pipeline
|
||||
untouched: multi-turn tool history (assistant tool_calls + role="tool" results)
|
||||
and nested chat-format tool defs are valid there, while blindly renaming
|
||||
``messages`` to ``input`` (the pre-fix behavior) produced items the Responses API
|
||||
rejects. Asserts acompletion is called with the exact messages and aresponses is
|
||||
never touched."""
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
|
||||
import litellm.proxy.proxy_server as ps
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "read a file"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_hist1",
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "arguments": '{"path": "a.py"}'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "call_hist1", "content": "file contents"},
|
||||
{"role": "user", "content": "now summarize"},
|
||||
]
|
||||
|
||||
mock_router = MagicMock()
|
||||
mock_router.acompletion = AsyncMock(
|
||||
return_value=litellm.ModelResponse(
|
||||
id="chatcmpl-cursor-1",
|
||||
choices=[
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "summary"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
model="gpt-4o",
|
||||
)
|
||||
)
|
||||
mock_router.aresponses = AsyncMock()
|
||||
mock_router.get_available_deployment = MagicMock(return_value=None)
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = _auth_override
|
||||
try:
|
||||
with patch.object(ps, "llm_router", mock_router):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "gpt-4o",
|
||||
"messages": messages,
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "parameters": {"type": "object"}},
|
||||
}
|
||||
],
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-test-cursor"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
body = response.json()
|
||||
assert body["choices"][0]["message"]["content"] == "summary"
|
||||
assert "output" not in body
|
||||
|
||||
mock_router.acompletion.assert_called_once()
|
||||
called_kwargs = mock_router.acompletion.call_args.kwargs
|
||||
assert called_kwargs["messages"] == messages
|
||||
assert "input" not in called_kwargs
|
||||
mock_router.aresponses.assert_not_called()
|
||||
|
||||
|
||||
def test_cursor_chat_completions_input_body_uses_responses_pipeline_and_strips_stream_options():
|
||||
"""A Responses-shaped body (``input``, no ``messages``; what Cursor agent mode
|
||||
sends) must run through the Responses pipeline with chat-completions output, and
|
||||
``stream_options`` (chat-completions-only; Cursor sends include_usage) must be
|
||||
stripped before the Responses call since OpenAI's Responses API rejects it.
|
||||
Stripping must not mutate the dict _read_request_body returned: that can be the
|
||||
request-scope cached parsed body itself, and removing a key from it corrupts the
|
||||
cache's key snapshot so any later _read_request_body caller (spend tracking,
|
||||
logging hooks) silently gets an empty body; a follow-up read must still see the
|
||||
full original body."""
|
||||
import asyncio
|
||||
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
|
||||
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.common_utils.http_parsing_utils import (
|
||||
_read_request_body as real_read_request_body,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
import litellm.proxy.proxy_server as ps
|
||||
|
||||
captured_requests = []
|
||||
|
||||
async def capturing_read_request_body(request):
|
||||
captured_requests.append(request)
|
||||
return await real_read_request_body(request=request)
|
||||
|
||||
mock_router = MagicMock()
|
||||
mock_router.aresponses = AsyncMock(
|
||||
return_value=ResponsesAPIResponse(
|
||||
id="resp_cursor_agent1",
|
||||
created_at=1234567890,
|
||||
model="gpt-4o",
|
||||
object="response",
|
||||
output=[
|
||||
ResponseOutputMessage(
|
||||
id="msg_agent1",
|
||||
type="message",
|
||||
role="assistant",
|
||||
status="completed",
|
||||
content=[
|
||||
ResponseOutputText(type="output_text", text="agent reply", annotations=[])
|
||||
],
|
||||
)
|
||||
],
|
||||
)
|
||||
)
|
||||
mock_router.acompletion = AsyncMock()
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = _auth_override
|
||||
try:
|
||||
with patch.object(ps, "llm_router", mock_router), patch.object(
|
||||
ps, "_read_request_body", side_effect=capturing_read_request_body
|
||||
):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "gpt-4o",
|
||||
"input": [{"role": "user", "content": "hello"}],
|
||||
"stream_options": {"include_usage": True},
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-test-cursor"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
body = response.json()
|
||||
assert body["choices"][0]["message"]["content"] == "agent reply"
|
||||
assert "output" not in body
|
||||
|
||||
mock_router.aresponses.assert_called_once()
|
||||
called_kwargs = mock_router.aresponses.call_args.kwargs
|
||||
assert "stream_options" not in called_kwargs
|
||||
mock_router.acompletion.assert_not_called()
|
||||
|
||||
assert captured_requests
|
||||
followup_body = asyncio.run(real_read_request_body(request=captured_requests[0]))
|
||||
assert followup_body.get("stream_options") == {"include_usage": True}
|
||||
assert followup_body.get("input") == [{"role": "user", "content": "hello"}]
|
||||
|
||||
|
||||
def test_cursor_models_route_delegates_to_model_list():
|
||||
"""Clients pointed at <proxy>/cursor as an OpenAI-compatible base URL resolve and
|
||||
verify keys via GET {base}/models (the OpenAI SDK contract). Without a dedicated
|
||||
route those requests fall through to the Cursor Cloud Agents passthrough and 401
|
||||
for lack of a Cursor API key, so BYOK verification fails before any chat request
|
||||
is sent. Both /cursor/models and /cursor/v1/models must serve the standard model
|
||||
list instead."""
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
|
||||
import litellm.proxy.proxy_server as ps
|
||||
|
||||
model_payload = {"data": [{"id": "gpt-5.6", "object": "model"}], "object": "list"}
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = _auth_override
|
||||
try:
|
||||
with patch.object(ps, "model_list", AsyncMock(return_value=model_payload)) as mock_model_list:
|
||||
client = TestClient(app)
|
||||
for path in ("/cursor/models", "/cursor/v1/models"):
|
||||
response = client.get(path, headers={"Authorization": "Bearer sk-test-cursor"})
|
||||
assert response.status_code == 200, f"{path}: {response.text}"
|
||||
assert response.json() == model_payload
|
||||
assert mock_model_list.call_count == 2
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
|
||||
class TestNestFlatChatTools:
|
||||
def test_flat_custom_tool_is_nested(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
result = _convert_tool_envelope(
|
||||
{"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}},
|
||||
to_chat=True,
|
||||
)
|
||||
assert result == {
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}},
|
||||
}
|
||||
|
||||
def test_flat_function_tool_is_nested(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
result = _convert_tool_envelope(
|
||||
{"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}},
|
||||
to_chat=True,
|
||||
)
|
||||
assert result == {
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "description": "d", "parameters": {"type": "object"}},
|
||||
}
|
||||
|
||||
def test_already_nested_and_unrecognized_tools_pass_through_unchanged(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
tools = [
|
||||
{"type": "custom", "custom": {"name": "already_nested"}},
|
||||
{"type": "function", "function": {"name": "f", "parameters": {}}},
|
||||
{"type": "web_search"},
|
||||
{"type": "custom"},
|
||||
{"name": "typeless"},
|
||||
{},
|
||||
"junk",
|
||||
None,
|
||||
42,
|
||||
]
|
||||
assert [_convert_tool_envelope(tool, to_chat=True) for tool in tools] == tools
|
||||
|
||||
|
||||
class TestCursorMessagesArmToolNormalization:
|
||||
@pytest.mark.asyncio
|
||||
async def test_flat_custom_tool_nested_before_chat_completion_delegation(self):
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
|
||||
seen = {}
|
||||
|
||||
async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict):
|
||||
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
|
||||
|
||||
seen["body"] = await _read_request_body(request=request)
|
||||
return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []}
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "gpt-5.6",
|
||||
"messages": [{"role": "user", "content": "use ApplyPatch"}],
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "parameters": {"type": "object"}},
|
||||
},
|
||||
{
|
||||
"type": "custom",
|
||||
"name": "ApplyPatch",
|
||||
"description": "V4A patch",
|
||||
"format": {
|
||||
"type": "grammar",
|
||||
"definition": "start: patch",
|
||||
"syntax": "lark",
|
||||
},
|
||||
},
|
||||
],
|
||||
"tool_choice": {"type": "custom", "name": "ApplyPatch"},
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
assert seen["body"]["tools"] == [
|
||||
{"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}},
|
||||
{
|
||||
"type": "custom",
|
||||
"custom": {
|
||||
"name": "ApplyPatch",
|
||||
"description": "V4A patch",
|
||||
"format": {
|
||||
"type": "grammar",
|
||||
"grammar": {"definition": "start: patch", "syntax": "lark"},
|
||||
},
|
||||
},
|
||||
},
|
||||
]
|
||||
assert seen["body"]["tool_choice"] == {"type": "custom", "custom": {"name": "ApplyPatch"}}
|
||||
assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_messages_body_without_flat_tools_leaves_parsed_body_cache_untouched(self):
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
|
||||
seen = {}
|
||||
|
||||
async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict):
|
||||
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
|
||||
|
||||
seen["body"] = await _read_request_body(request=request)
|
||||
return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []}
|
||||
|
||||
body = {
|
||||
"model": "gpt-5.6",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"tools": [{"type": "function", "function": {"name": "f", "parameters": {}}}],
|
||||
}
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json=body,
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
assert seen["body"]["tools"] == body["tools"]
|
||||
assert seen["body"]["messages"] == body["messages"]
|
||||
|
||||
|
||||
class TestToolEnvelopeConversionMatrix:
|
||||
"""
|
||||
Cursor mixes Responses API shapes into chat bodies PER LEVEL, independently
|
||||
(live-captured: a pre-nested custom envelope carrying a flat grammar format).
|
||||
Tool definitions and tool_choice share one envelope rule, so every cell of
|
||||
direction x envelope x format must land on that direction's canonical shape.
|
||||
"""
|
||||
|
||||
FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"}
|
||||
NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}}
|
||||
TEXT = {"type": "text"}
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
@pytest.mark.parametrize("envelope", ["flat", "nested"])
|
||||
@pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"])
|
||||
def test_every_direction_envelope_and_format_lands_canonical(self, to_chat, envelope, format_shape):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
format_value = {
|
||||
"absent": None,
|
||||
"text": self.TEXT,
|
||||
"flat_grammar": self.FLAT_GRAMMAR,
|
||||
"nested_grammar": self.NESTED_GRAMMAR,
|
||||
}[format_shape]
|
||||
payload = {"name": "ApplyPatch", "description": "V4A patch"}
|
||||
if format_value is not None:
|
||||
payload["format"] = format_value
|
||||
tool = {"type": "custom", "custom": payload} if envelope == "nested" else {"type": "custom", **payload}
|
||||
|
||||
canonical_payload = {"name": "ApplyPatch", "description": "V4A patch"}
|
||||
if format_shape in ("flat_grammar", "nested_grammar"):
|
||||
canonical_payload["format"] = self.NESTED_GRAMMAR if to_chat else self.FLAT_GRAMMAR
|
||||
elif format_shape == "text":
|
||||
canonical_payload["format"] = self.TEXT
|
||||
expected = (
|
||||
{"type": "custom", "custom": canonical_payload} if to_chat else {"type": "custom", **canonical_payload}
|
||||
)
|
||||
|
||||
assert _convert_tool_envelope(tool, to_chat=to_chat) == expected
|
||||
|
||||
def test_nested_envelope_with_flat_grammar_matches_live_cursor_capture(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
cursor_tool = {"type": "custom", "custom": {"name": "ApplyPatch", "format": self.FLAT_GRAMMAR}}
|
||||
assert _convert_tool_envelope(cursor_tool, to_chat=True) == {
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "format": self.NESTED_GRAMMAR},
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_conversion_is_idempotent(self, to_chat):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
once = _convert_tool_envelope({"type": "custom", "name": "A", "format": self.FLAT_GRAMMAR}, to_chat=to_chat)
|
||||
assert _convert_tool_envelope(once, to_chat=to_chat) == once
|
||||
|
||||
def test_nested_function_tool_flattens_and_flat_passes_through(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
nested = {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}}
|
||||
flat = {"type": "function", "name": "read_file", "parameters": {"type": "object"}}
|
||||
assert _convert_tool_envelope(nested, to_chat=False) == flat
|
||||
assert _convert_tool_envelope(flat, to_chat=False) == flat
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_unrecognized_entries_pass_through(self, to_chat):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}, 42, {"type": "auto"}]
|
||||
assert [_convert_tool_envelope(entry, to_chat=to_chat) for entry in entries] == entries
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_empty_nested_envelope_falls_back_to_top_level_payload(self, to_chat):
|
||||
"""An empty nested envelope must not shadow payload fields that sit at the top
|
||||
level; treating the empty dict as the sole payload source dropped the name."""
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
hybrid = {"type": "custom", "custom": {}, "name": "ApplyPatch", "format": self.TEXT}
|
||||
expected_payload = {"name": "ApplyPatch", "format": self.TEXT}
|
||||
expected = {"type": "custom", "custom": expected_payload} if to_chat else {"type": "custom", **expected_payload}
|
||||
assert _convert_tool_envelope(hybrid, to_chat=to_chat) == expected
|
||||
|
||||
def test_nested_payload_wins_over_stray_top_level_fields(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
tool = {"type": "custom", "custom": {"name": "NestedName"}, "name": "TopName"}
|
||||
assert _convert_tool_envelope(tool, to_chat=False) == {"type": "custom", "name": "NestedName"}
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_nameless_envelope_passes_through_unchanged(self, to_chat):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
nameless = {"type": "custom", "custom": {}, "description": "no name anywhere"}
|
||||
assert _convert_tool_envelope(nameless, to_chat=to_chat) == nameless
|
||||
|
||||
|
||||
class TestToolChoiceSharesTheToolEnvelopeRule:
|
||||
"""
|
||||
tool_choice carries the same {"type": T, T: {...}} chat envelope as a tool
|
||||
definition, so it converts through the same function in both directions.
|
||||
OpenAI requires the nested key on chat (SDK ChatCompletionNamedToolChoiceParam
|
||||
and ChatCompletionNamedToolChoiceCustomParam both mark it Required).
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize("choice_type", ["custom", "function"])
|
||||
def test_flat_tool_choice_is_nested_for_chat(self, choice_type):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
assert _convert_tool_envelope({"type": choice_type, "name": "ApplyPatch"}, to_chat=True) == {
|
||||
"type": choice_type,
|
||||
choice_type: {"name": "ApplyPatch"},
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize("choice_type", ["custom", "function"])
|
||||
def test_nested_tool_choice_is_flattened_for_responses(self, choice_type):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
assert _convert_tool_envelope({"type": choice_type, choice_type: {"name": "ApplyPatch"}}, to_chat=False) == {
|
||||
"type": choice_type,
|
||||
"name": "ApplyPatch",
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_sentinel_and_malformed_tool_choice_pass_through(self, to_chat):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
|
||||
|
||||
for unchanged in ("auto", "required", "none", None, {"type": "auto"}, 42):
|
||||
assert _convert_tool_envelope(unchanged, to_chat=to_chat) == unchanged
|
||||
|
||||
|
||||
class TestNormalizeToolDialectCoversBothFields:
|
||||
"""
|
||||
The regression that motivated one normalizer: tools were converted while
|
||||
tool_choice was left flat, so OpenAI rejected the request. Both fields move
|
||||
together in a single call, on both arms.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize("to_chat", [True, False])
|
||||
def test_tools_and_tool_choice_convert_together(self, to_chat):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
|
||||
|
||||
flat = {"type": "custom", "name": "ApplyPatch"}
|
||||
nested = {"type": "custom", "custom": {"name": "ApplyPatch"}}
|
||||
source = flat if to_chat else nested
|
||||
expected = nested if to_chat else flat
|
||||
|
||||
out = _normalize_tool_dialect({"messages": [], "tools": [source], "tool_choice": source}, to_chat=to_chat)
|
||||
assert out["tools"] == [expected]
|
||||
assert out["tool_choice"] == expected
|
||||
|
||||
def test_body_needing_no_conversion_is_returned_by_identity(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
|
||||
|
||||
data = {"messages": [], "tools": [{"type": "function", "function": {"name": "f"}}], "tool_choice": "auto"}
|
||||
assert _normalize_tool_dialect(data, to_chat=True) is data
|
||||
|
||||
def test_absent_tool_fields_are_not_invented(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
|
||||
|
||||
data = {"messages": [{"role": "user", "content": "hi"}]}
|
||||
result = _normalize_tool_dialect(data, to_chat=True)
|
||||
assert result == data
|
||||
assert "tools" not in result and "tool_choice" not in result
|
||||
|
||||
|
||||
class TestCursorInputArmFlattening:
|
||||
@pytest.mark.asyncio
|
||||
async def test_nested_chat_shapes_in_input_body_reach_aresponses_flattened(self):
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
|
||||
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
mock_response = ResponsesAPIResponse(
|
||||
id="resp_flat123",
|
||||
created_at=1234567890,
|
||||
model="gpt-5.6",
|
||||
object="response",
|
||||
output=[
|
||||
ResponseOutputMessage(
|
||||
id="msg_flat123",
|
||||
type="message",
|
||||
role="assistant",
|
||||
status="completed",
|
||||
content=[ResponseOutputText(type="output_text", text="ok", annotations=[])],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with patch("litellm.proxy.proxy_server.llm_router") as mock_router:
|
||||
mock_router.aresponses = AsyncMock(return_value=mock_response)
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "gpt-5.6",
|
||||
"input": [{"role": "user", "content": "use ApplyPatch"}],
|
||||
"tools": [
|
||||
{
|
||||
"type": "custom",
|
||||
"custom": {
|
||||
"name": "ApplyPatch",
|
||||
"format": {
|
||||
"type": "grammar",
|
||||
"grammar": {"definition": "start: patch", "syntax": "lark"},
|
||||
},
|
||||
},
|
||||
},
|
||||
{"type": "function", "name": "read_file", "parameters": {"type": "object"}},
|
||||
],
|
||||
"tool_choice": {"type": "custom", "custom": {"name": "ApplyPatch"}},
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
call_kwargs = mock_router.aresponses.call_args.kwargs
|
||||
assert call_kwargs["tools"] == [
|
||||
{
|
||||
"type": "custom",
|
||||
"name": "ApplyPatch",
|
||||
"format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"},
|
||||
},
|
||||
{"type": "function", "name": "read_file", "parameters": {"type": "object"}},
|
||||
]
|
||||
assert call_kwargs["tool_choice"] == {"type": "custom", "name": "ApplyPatch"}
|
||||
|
||||
|
||||
class TestChatCompletionsBodyDetection:
|
||||
def test_routing_matrix(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _is_chat_completions_body
|
||||
|
||||
assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}]}) is True
|
||||
assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}], "input": []}) is True
|
||||
assert _is_chat_completions_body({"messages": None, "input": [{"role": "user", "content": "hi"}]}) is False
|
||||
assert _is_chat_completions_body({"messages": [], "input": [{"role": "user", "content": "hi"}]}) is False
|
||||
assert _is_chat_completions_body({"messages": None}) is True
|
||||
assert _is_chat_completions_body({"messages": []}) is True
|
||||
assert _is_chat_completions_body({"input": [{"role": "user", "content": "hi"}]}) is False
|
||||
assert _is_chat_completions_body({}) is False
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_null_messages_stub_with_input_reaches_responses_arm(self):
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
|
||||
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
mock_response = ResponsesAPIResponse(
|
||||
id="resp_stub1",
|
||||
created_at=1234567890,
|
||||
model="gpt-5.6",
|
||||
object="response",
|
||||
output=[
|
||||
ResponseOutputMessage(
|
||||
id="msg_stub1",
|
||||
type="message",
|
||||
role="assistant",
|
||||
status="completed",
|
||||
content=[ResponseOutputText(type="output_text", text="ok", annotations=[])],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with patch("litellm.proxy.proxy_server.llm_router") as mock_router:
|
||||
mock_router.aresponses = AsyncMock(return_value=mock_response)
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "gpt-5.6",
|
||||
"messages": None,
|
||||
"input": [{"role": "user", "content": "hello"}],
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
assert mock_router.aresponses.call_args is not None
|
||||
assert mock_router.aresponses.call_args.kwargs["input"] == [{"role": "user", "content": "hello"}]
|
||||
|
||||
|
||||
class TestParseCursorModelVariant:
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_base,expected_effort",
|
||||
[
|
||||
("claude-opus-5-thinking-high", "claude-opus-5", "high"),
|
||||
("claude-opus-5-thinking-xhigh-fast", "claude-opus-5", "xhigh"),
|
||||
("gemini-3.0-pro-thinking-low", "gemini-3.0-pro", "low"),
|
||||
("claude-opus-5-fast", "claude-opus-5", None),
|
||||
("gpt-5.6-sol", "gpt-5.6-sol", None),
|
||||
("foo-thinking-ultra-fast", "foo-thinking-ultra", None),
|
||||
("-thinking-high", "-thinking-high", None),
|
||||
],
|
||||
)
|
||||
def test_parse_matrix(self, model, expected_base, expected_effort):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _parse_cursor_model_variant
|
||||
|
||||
variant = _parse_cursor_model_variant(model)
|
||||
assert variant.base_model == expected_base
|
||||
assert variant.reasoning_effort == expected_effort
|
||||
|
||||
|
||||
class TestResolveCursorModelVariant:
|
||||
@pytest.fixture(scope="class")
|
||||
def wildcard_router(self):
|
||||
from litellm import Router
|
||||
|
||||
return Router(
|
||||
model_list=[
|
||||
{"model_name": "anthropic/*", "litellm_params": {"model": "anthropic/*", "api_key": "fake"}},
|
||||
{"model_name": "openai/*", "litellm_params": {"model": "openai/*", "api_key": "fake"}},
|
||||
{
|
||||
"model_name": "explicit-alias-thinking-high",
|
||||
"litellm_params": {"model": "anthropic/claude-opus-5", "api_key": "fake"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
def test_chat_body_suffix_stripped_into_reasoning_effort(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {
|
||||
"model": "claude-opus-5-thinking-xhigh-fast",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
}
|
||||
resolved = _resolve_cursor_model_variant(body, wildcard_router)
|
||||
assert resolved["model"] == "claude-opus-5"
|
||||
assert resolved["reasoning_effort"] == "xhigh"
|
||||
assert resolved["messages"] == body["messages"]
|
||||
assert body["model"] == "claude-opus-5-thinking-xhigh-fast"
|
||||
|
||||
def test_responses_body_suffix_stripped_into_reasoning_dict(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "claude-opus-5-thinking-high", "input": [{"role": "user", "content": "hi"}]}
|
||||
resolved = _resolve_cursor_model_variant(body, wildcard_router)
|
||||
assert resolved["model"] == "claude-opus-5"
|
||||
assert resolved["reasoning"] == {"effort": "high"}
|
||||
|
||||
def test_responses_body_merges_effort_into_existing_reasoning(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {
|
||||
"model": "claude-opus-5-thinking-high",
|
||||
"input": [{"role": "user", "content": "hi"}],
|
||||
"reasoning": {"summary": "auto"},
|
||||
}
|
||||
resolved = _resolve_cursor_model_variant(body, wildcard_router)
|
||||
assert resolved["model"] == "claude-opus-5"
|
||||
assert resolved["reasoning"] == {"summary": "auto", "effort": "high"}
|
||||
|
||||
def test_existing_reasoning_effort_wins_but_model_still_rewritten(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
chat_body = {
|
||||
"model": "claude-opus-5-thinking-high",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"reasoning_effort": "low",
|
||||
}
|
||||
resolved_chat = _resolve_cursor_model_variant(chat_body, wildcard_router)
|
||||
assert resolved_chat["model"] == "claude-opus-5"
|
||||
assert resolved_chat["reasoning_effort"] == "low"
|
||||
|
||||
responses_body = {
|
||||
"model": "claude-opus-5-thinking-high",
|
||||
"input": [{"role": "user", "content": "hi"}],
|
||||
"reasoning": {"effort": "low"},
|
||||
}
|
||||
resolved_responses = _resolve_cursor_model_variant(responses_body, wildcard_router)
|
||||
assert resolved_responses["model"] == "claude-opus-5"
|
||||
assert resolved_responses["reasoning"] == {"effort": "low"}
|
||||
|
||||
def test_fast_only_suffix_strips_without_reasoning(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "claude-opus-5-fast", "messages": [{"role": "user", "content": "hi"}]}
|
||||
resolved = _resolve_cursor_model_variant(body, wildcard_router)
|
||||
assert resolved["model"] == "claude-opus-5"
|
||||
assert "reasoning_effort" not in resolved
|
||||
|
||||
def test_explicitly_configured_suffixed_name_untouched(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "explicit-alias-thinking-high", "messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(body, wildcard_router) is body
|
||||
|
||||
def test_provider_inferable_bare_name_untouched(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "gpt-4o-mini", "messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(body, wildcard_router) is body
|
||||
|
||||
def test_unservable_base_untouched(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "totally-unknown-thinking-high", "messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(body, wildcard_router) is body
|
||||
|
||||
def test_no_router_untouched(self):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
body = {"model": "claude-opus-5-thinking-high", "messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(body, None) is body
|
||||
|
||||
def test_missing_or_non_string_model_untouched(self, wildcard_router):
|
||||
from litellm.proxy.response_api_endpoints.endpoints import _resolve_cursor_model_variant
|
||||
|
||||
no_model = {"messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(no_model, wildcard_router) is no_model
|
||||
null_model = {"model": None, "messages": [{"role": "user", "content": "hi"}]}
|
||||
assert _resolve_cursor_model_variant(null_model, wildcard_router) is null_model
|
||||
|
||||
|
||||
def _router_serving_only(base_model: str) -> MagicMock:
|
||||
mock_router = MagicMock()
|
||||
mock_router.model_names = set()
|
||||
mock_router.model_group_alias = {}
|
||||
mock_router.team_public_model_names = frozenset()
|
||||
mock_router.pattern_router.get_pattern.side_effect = (
|
||||
lambda model: [{"model_name": "anthropic/*"}] if model == base_model else None
|
||||
)
|
||||
return mock_router
|
||||
|
||||
|
||||
class TestCursorModelSuffixResolutionEndToEnd:
|
||||
@pytest.mark.asyncio
|
||||
async def test_chat_arm_rewrites_suffixed_model_before_delegation(self):
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
|
||||
seen = {}
|
||||
|
||||
async def fake_chat_completion(request, fastapi_response, model, user_api_key_dict):
|
||||
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
|
||||
|
||||
seen["body"] = await _read_request_body(request=request)
|
||||
return {"id": "chatcmpl-fake", "object": "chat.completion", "choices": []}
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with (
|
||||
patch("litellm.proxy.proxy_server.llm_router", new=_router_serving_only("claude-opus-5")),
|
||||
patch("litellm.proxy.proxy_server.chat_completion", new=fake_chat_completion),
|
||||
):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "claude-opus-5-thinking-xhigh-fast",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
assert seen["body"]["model"] == "claude-opus-5"
|
||||
assert seen["body"]["reasoning_effort"] == "xhigh"
|
||||
assert seen["body"]["messages"] == [{"role": "user", "content": "hi"}]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_responses_arm_rewrites_suffixed_model_before_routing(self):
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
|
||||
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
mock_response = ResponsesAPIResponse(
|
||||
id="resp_suffix1",
|
||||
created_at=1234567890,
|
||||
model="claude-opus-5",
|
||||
object="response",
|
||||
output=[
|
||||
ResponseOutputMessage(
|
||||
id="msg_suffix1",
|
||||
type="message",
|
||||
role="assistant",
|
||||
status="completed",
|
||||
content=[ResponseOutputText(type="output_text", text="ok", annotations=[])],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
mock_router = _router_serving_only("claude-opus-5")
|
||||
mock_router.aresponses = AsyncMock(return_value=mock_response)
|
||||
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
|
||||
try:
|
||||
with patch("litellm.proxy.proxy_server.llm_router", new=mock_router):
|
||||
client = TestClient(app)
|
||||
response = client.post(
|
||||
"/cursor/chat/completions",
|
||||
json={
|
||||
"model": "claude-opus-5-thinking-high",
|
||||
"input": [{"role": "user", "content": "hello"}],
|
||||
},
|
||||
headers={"Authorization": "Bearer sk-1234"},
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
assert response.status_code == 200
|
||||
assert mock_router.aresponses.call_args is not None
|
||||
assert mock_router.aresponses.call_args.kwargs["model"] == "claude-opus-5"
|
||||
assert mock_router.aresponses.call_args.kwargs["reasoning"] == {"effort": "high"}
|
||||
|
|
|
|||
|
|
@ -959,6 +959,27 @@ class TestToolChoiceTransformation:
|
|||
)
|
||||
assert result == {"type": "function", "function": {"name": "get_weather"}}
|
||||
|
||||
def test_transform_tool_choice_custom_follows_function_downgrade(self):
|
||||
"""
|
||||
This bridge downgrades custom tools to function tools
|
||||
(convert_custom_tool_to_function_tool), so a custom tool_choice must become a
|
||||
function tool_choice naming the same tool or it references a tool type absent
|
||||
from the converted request.
|
||||
"""
|
||||
flat = LiteLLMCompletionResponsesConfig._transform_tool_choice(
|
||||
{"type": "custom", "name": "ApplyPatch"}
|
||||
)
|
||||
assert flat == {"type": "function", "function": {"name": "ApplyPatch"}}
|
||||
|
||||
nested = LiteLLMCompletionResponsesConfig._transform_tool_choice(
|
||||
{"type": "custom", "custom": {"name": "ApplyPatch"}}
|
||||
)
|
||||
assert nested == {"type": "function", "function": {"name": "ApplyPatch"}}
|
||||
|
||||
def test_transform_tool_choice_custom_without_name_falls_back_to_required(self):
|
||||
result = LiteLLMCompletionResponsesConfig._transform_tool_choice({"type": "custom"})
|
||||
assert result == "required"
|
||||
|
||||
def test_transform_tool_choice_function_without_name_falls_back_to_required(self):
|
||||
"""A function-type dict with no name still falls back to required"""
|
||||
result = LiteLLMCompletionResponsesConfig._transform_tool_choice(
|
||||
|
|
|
|||
|
|
@ -810,8 +810,12 @@ def test_responses_api_bridge_check_azure_gpt_5_4_tools_plus_reasoning_routes_to
|
|||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_azure_gpt_5_4_tools_without_reasoning_stays_chat():
|
||||
"""Azure gpt-5.4 with tools only should not be force-routed to Responses API."""
|
||||
def test_responses_api_bridge_check_azure_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
||||
"""
|
||||
Azure gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables
|
||||
reasoning by default for gpt-5.4+, and Chat Completions rejects function tools
|
||||
whenever reasoning is on.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
|
|
@ -824,11 +828,15 @@ def test_responses_api_bridge_check_azure_gpt_5_4_tools_without_reasoning_stays_
|
|||
)
|
||||
|
||||
assert model == "gpt-5.4"
|
||||
assert model_info.get("mode") != "responses"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_gpt_5_4_tools_without_reasoning_stays_chat():
|
||||
"""gpt-5.4 with tools only should not be force-routed to Responses API."""
|
||||
def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_to_responses():
|
||||
"""
|
||||
gpt-5.4 with tools and UNSET reasoning_effort must bridge: OpenAI enables reasoning
|
||||
by default for gpt-5.4+, and Chat Completions rejects function tools whenever
|
||||
reasoning is on ("use /v1/responses or set reasoning_effort to 'none'").
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
|
|
@ -841,6 +849,302 @@ def test_responses_api_bridge_check_gpt_5_4_tools_without_reasoning_stays_chat()
|
|||
)
|
||||
|
||||
assert model == "gpt-5.4"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_gpt_5_4_tools_with_reasoning_none_stays_chat():
|
||||
"""
|
||||
Explicit reasoning_effort "none" is OpenAI's documented escape hatch that keeps
|
||||
function tools servable on Chat Completions; the bridge must not fire.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.4",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort="none",
|
||||
)
|
||||
|
||||
assert model == "gpt-5.4"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_responses():
|
||||
"""A reasoning summary is Responses-only regardless of effort value."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.4",
|
||||
custom_llm_provider="openai",
|
||||
reasoning_effort="none",
|
||||
reasoning_summary="detailed",
|
||||
)
|
||||
|
||||
assert model == "gpt-5.4"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat():
|
||||
"""
|
||||
Chat Completions serves custom (grammar) tools natively with reasoning on; only
|
||||
FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat
|
||||
so responses keep the native custom tool_call shape instead of the bridge's
|
||||
function-shaped mapping.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}],
|
||||
reasoning_effort=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses():
|
||||
"""One function tool in the mix is enough to make chat unservable with reasoning on."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[
|
||||
{"type": "custom", "custom": {"name": "ApplyPatch"}},
|
||||
{"type": "function", "function": {"name": "shell"}},
|
||||
],
|
||||
reasoning_effort=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses():
|
||||
"""Responses-style flat function tool defs still count as function tools."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}],
|
||||
reasoning_effort=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_dict_effort_none_stays_chat():
|
||||
"""The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort={"effort": "none"},
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_dict_effort_active_routes_to_responses():
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort={"effort": "low"},
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_responses():
|
||||
"""A summary inside the dict form is Responses-only even when effort is none."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort={"effort": "none", "summary": "concise"},
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("blank_api_base", [None, "", " ", "\t"])
|
||||
def test_responses_api_bridge_check_blank_api_base_is_default_openai(blank_api_base):
|
||||
"""
|
||||
A blank api_base (None, empty, or whitespace) resolves to the default OpenAI
|
||||
endpoint downstream, which enforces the reasoning+tools constraint, so gpt-5.4+
|
||||
function-tool requests with unset reasoning_effort must still auto-bridge.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
api_base=blank_api_base,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat():
|
||||
"""
|
||||
Chat-only OpenAI-compatible backends registered under the openai provider with a
|
||||
custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and
|
||||
have no /responses route; the unset-effort arm must not reroute them.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
api_base="http://vllm.internal:8000/v1",
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_custom_api_base_via_global_with_unset_effort_stays_chat(monkeypatch):
|
||||
"""
|
||||
A custom base set through the litellm.api_base global (not the call arg) is resolved the
|
||||
same way the chat handler resolves it, so the unset-effort arm must not reroute a chat-only
|
||||
backend to a /responses route it lacks. Regression guard: the gate previously inspected only
|
||||
the call-level api_base and bridged these requests.
|
||||
"""
|
||||
import litellm
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
monkeypatch.setattr(litellm, "api_base", "http://vllm.internal:8000/v1")
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("env_var", ["OPENAI_BASE_URL", "OPENAI_API_BASE"])
|
||||
def test_responses_api_bridge_check_custom_api_base_via_env_with_unset_effort_stays_chat(monkeypatch, env_var):
|
||||
"""
|
||||
A custom base set via OPENAI_BASE_URL/OPENAI_API_BASE env is resolved identically to the chat
|
||||
handler, so the unset-effort arm leaves the request on chat instead of bridging it.
|
||||
"""
|
||||
import litellm
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
monkeypatch.setenv(env_var, "http://vllm.internal:8000/v1")
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes():
|
||||
"""Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort="high",
|
||||
api_base="http://vllm.internal:8000/v1",
|
||||
)
|
||||
|
||||
assert model == "gpt-5.6"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes():
|
||||
"""Azure OpenAI always sets api_base and does enforce the constraint; keep bridging."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.4",
|
||||
custom_llm_provider="azure",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
api_base="https://myresource.openai.azure.com",
|
||||
)
|
||||
|
||||
assert model == "gpt-5.4"
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat():
|
||||
"""Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {"max_tokens": 128000}
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.1",
|
||||
custom_llm_provider="openai",
|
||||
tools=[{"type": "function", "function": {"name": "get_capital"}}],
|
||||
reasoning_effort=None,
|
||||
)
|
||||
|
||||
assert model == "gpt-5.1"
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -603,3 +603,134 @@ def test_delattr_fast_path_missing_attribute_is_noop():
|
|||
|
||||
del racy.x
|
||||
del racy.x
|
||||
def test_chat_completion_tool_call_from_dict_custom():
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
chat_completion_tool_call_from_dict,
|
||||
)
|
||||
|
||||
custom_tc = {
|
||||
"id": "call_njxQ",
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch\n*** End Patch\n"},
|
||||
}
|
||||
parsed = chat_completion_tool_call_from_dict(custom_tc)
|
||||
assert isinstance(parsed, ChatCompletionMessageCustomToolCall)
|
||||
assert parsed.model_dump() == custom_tc
|
||||
|
||||
func_tc = {"id": "call_1", "type": "function", "function": {"name": "f", "arguments": "{}"}}
|
||||
parsed_func = chat_completion_tool_call_from_dict(func_tc)
|
||||
assert isinstance(parsed_func, ChatCompletionMessageToolCall)
|
||||
assert "custom" not in parsed_func.model_dump()
|
||||
|
||||
|
||||
def test_chat_completion_tool_call_from_dict_custom_strips_null_function():
|
||||
from litellm.types.utils import chat_completion_tool_call_from_dict
|
||||
|
||||
sdk_shaped = {
|
||||
"id": "call_x",
|
||||
"type": "custom",
|
||||
"function": None,
|
||||
"custom": {"name": "ApplyPatch", "input": ""},
|
||||
}
|
||||
parsed = chat_completion_tool_call_from_dict(sdk_shaped)
|
||||
assert "function" not in parsed.model_dump()
|
||||
|
||||
|
||||
def test_chat_completion_tool_call_from_dict_typeless_custom_payload():
|
||||
"""A tool-call dict can carry a ``custom`` payload with ``type`` absent or None
|
||||
(e.g. rebuilt from streaming deltas, where only the first chunk has ``type``).
|
||||
Classifying on ``type == "custom"`` alone sent these to the function branch,
|
||||
which raised TypeError (missing ``function``) on a payload the streaming path
|
||||
accepts as custom."""
|
||||
from litellm.types.utils import ChatCompletionMessageCustomToolCall, chat_completion_tool_call_from_dict
|
||||
|
||||
typeless = {"id": "call_1", "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}}
|
||||
parsed = chat_completion_tool_call_from_dict(typeless)
|
||||
assert isinstance(parsed, ChatCompletionMessageCustomToolCall)
|
||||
assert parsed.type == "custom"
|
||||
assert parsed.custom.name == "ApplyPatch"
|
||||
|
||||
null_typed = {"id": "call_2", "type": None, "custom": {"name": "f", "input": "{}"}}
|
||||
assert isinstance(chat_completion_tool_call_from_dict(null_typed), ChatCompletionMessageCustomToolCall)
|
||||
|
||||
|
||||
def test_custom_tool_call_classification_agrees_across_streaming_and_non_streaming():
|
||||
"""The streaming Delta coercion and the non-streaming from_dict parser must
|
||||
classify the same tool-call dict identically, or a provider payload becomes a
|
||||
custom tool call mid-stream and something else on the completed message."""
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionDeltaCustomToolCall,
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
Delta,
|
||||
chat_completion_tool_call_from_dict,
|
||||
)
|
||||
|
||||
tool_calls = [
|
||||
{"id": "c1", "type": "custom", "custom": {"name": "ApplyPatch", "input": ""}},
|
||||
{"id": "c2", "custom": {"name": "ApplyPatch", "input": "x"}},
|
||||
{"id": "c3", "type": "function", "function": {"name": "g", "arguments": "{}"}},
|
||||
]
|
||||
for tool_call in tool_calls:
|
||||
message_parsed = chat_completion_tool_call_from_dict(dict(tool_call))
|
||||
delta_parsed = Delta(tool_calls=[dict(tool_call, index=0)]).tool_calls[0]
|
||||
assert isinstance(message_parsed, ChatCompletionMessageCustomToolCall) == isinstance(
|
||||
delta_parsed, ChatCompletionDeltaCustomToolCall
|
||||
)
|
||||
|
||||
|
||||
def test_message_with_mixed_function_and_custom_tool_calls():
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionMessageCustomToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
Message,
|
||||
)
|
||||
|
||||
message = Message(
|
||||
content=None,
|
||||
role="assistant",
|
||||
tool_calls=[
|
||||
{"id": "call_c", "type": "custom", "custom": {"name": "ApplyPatch", "input": "patch"}},
|
||||
{"id": "call_f", "type": "function", "function": {"name": "f", "arguments": "{}"}},
|
||||
],
|
||||
)
|
||||
assert isinstance(message.tool_calls[0], ChatCompletionMessageCustomToolCall)
|
||||
assert isinstance(message.tool_calls[1], ChatCompletionMessageToolCall)
|
||||
dumped = message.model_dump()["tool_calls"]
|
||||
assert dumped[0] == {"id": "call_c", "type": "custom", "custom": {"name": "ApplyPatch", "input": "patch"}}
|
||||
assert "custom" not in dumped[1]
|
||||
|
||||
|
||||
def test_delta_custom_tool_call_first_and_continuation_chunks():
|
||||
from litellm.types.utils import ChatCompletionDeltaCustomToolCall, Delta
|
||||
|
||||
first_chunk_tc = {
|
||||
"index": 0,
|
||||
"id": "call_TBs",
|
||||
"function": None,
|
||||
"type": "custom",
|
||||
"custom": {"name": "ApplyPatch", "input": ""},
|
||||
}
|
||||
continuation_tc = {"index": 0, "id": None, "function": None, "type": None, "custom": {"input": "***"}}
|
||||
|
||||
first_delta = Delta(role="assistant", tool_calls=[first_chunk_tc])
|
||||
assert isinstance(first_delta.tool_calls[0], ChatCompletionDeltaCustomToolCall)
|
||||
first_dump = first_delta.model_dump()["tool_calls"][0]
|
||||
assert first_dump["type"] == "custom"
|
||||
assert first_dump["custom"] == {"name": "ApplyPatch", "input": ""}
|
||||
assert "function" not in first_dump
|
||||
|
||||
continuation_delta = Delta(tool_calls=[continuation_tc])
|
||||
cont_dump = continuation_delta.model_dump()["tool_calls"][0]
|
||||
assert cont_dump["type"] is None
|
||||
assert cont_dump["custom"]["input"] == "***"
|
||||
assert "function" not in cont_dump
|
||||
|
||||
|
||||
def test_delta_function_tool_call_unchanged_by_custom_support():
|
||||
from litellm.types.utils import ChatCompletionDeltaToolCall, Delta
|
||||
|
||||
delta = Delta(tool_calls=[{"index": 0, "id": "c2", "type": "function", "function": {"name": "g", "arguments": ""}}])
|
||||
assert isinstance(delta.tool_calls[0], ChatCompletionDeltaToolCall)
|
||||
assert "custom" not in delta.model_dump()["tool_calls"][0]
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
{
|
||||
"LIT001": {
|
||||
"limit": 23349
|
||||
"limit": 23350
|
||||
},
|
||||
"LIT002": {
|
||||
"limit": 27252
|
||||
"limit": 27239
|
||||
},
|
||||
"LIT003": {
|
||||
"limit": 292
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
"limit": 0
|
||||
},
|
||||
"LIT006": {
|
||||
"limit": 1105
|
||||
"limit": 1103
|
||||
},
|
||||
"LIT007": {
|
||||
"limit": 0
|
||||
|
|
@ -24,6 +24,6 @@
|
|||
"limit": 1004
|
||||
},
|
||||
"LIT009": {
|
||||
"limit": 2465
|
||||
"limit": 2460
|
||||
}
|
||||
}
|
||||
|
|
|
|||
133
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
133
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -2621,10 +2621,20 @@ export interface paths {
|
|||
put?: never;
|
||||
/**
|
||||
* Cursor Chat Completions
|
||||
* @description Cursor-specific endpoint that accepts Responses API input format but returns chat completions format.
|
||||
* @description Cursor BYOK endpoint. Accepts both request shapes Cursor sends to its OpenAI-compatible
|
||||
* base URL and always answers in chat completions format.
|
||||
*
|
||||
* This endpoint handles requests from Cursor IDE which sends Responses API format (`input` field)
|
||||
* but expects chat completions format response (`choices`, `messages`, etc.).
|
||||
* Cursor agent mode sends Responses API format bodies (`input`, flat tool defs, `reasoning`,
|
||||
* custom tools) to the chat/completions path while expecting chat completions responses;
|
||||
* those are routed through the Responses API pipeline and converted back. Genuine chat
|
||||
* completions bodies (`messages` present) are routed through the standard chat completions
|
||||
* pipeline, after normalizing each level of the `tools` array and `tool_choice` to the chat
|
||||
* completions shapes OpenAI requires. Cursor mixes Responses API shapes into chat bodies
|
||||
* per level, independently: a flat tool def (`{"type": "custom", "name": "ApplyPatch", ...}`)
|
||||
* gets nested under `custom`, and a flat grammar format
|
||||
* (`{"type": "grammar", "definition", "syntax"}`) gets wrapped as
|
||||
* `{"type": "grammar", "grammar": {...}}` wherever it appears, including inside tool defs
|
||||
* Cursor already sent pre-nested.
|
||||
*
|
||||
* ```bash
|
||||
* curl -X POST http://localhost:4000/cursor/chat/completions -H "Content-Type: application/json" -H "Authorization: Bearer sk-1234" -d '{
|
||||
|
|
@ -2641,6 +2651,58 @@ export interface paths {
|
|||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/cursor/models": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/**
|
||||
* Cursor Model List
|
||||
* @description OpenAI-compatible model listing for the Cursor BYOK base URL.
|
||||
*
|
||||
* Clients pointed at `<proxy>/cursor` as an OpenAI-compatible base URL resolve and
|
||||
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
* route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
* demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
*/
|
||||
get: operations["cursor_model_list_cursor_models_get"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/cursor/v1/models": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/**
|
||||
* Cursor Model List
|
||||
* @description OpenAI-compatible model listing for the Cursor BYOK base URL.
|
||||
*
|
||||
* Clients pointed at `<proxy>/cursor` as an OpenAI-compatible base URL resolve and
|
||||
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
* route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
* demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
*/
|
||||
get: operations["cursor_model_list_cursor_v1_models_get"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/cursor/{endpoint}": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -22215,6 +22277,15 @@ export interface components {
|
|||
*/
|
||||
type: "ephemeral";
|
||||
};
|
||||
/** ChatCompletionCustomToolCallPayload */
|
||||
ChatCompletionCustomToolCallPayload: {
|
||||
/** Input */
|
||||
input: string;
|
||||
/** Name */
|
||||
name: string;
|
||||
} & {
|
||||
[key: string]: unknown;
|
||||
};
|
||||
/** ChatCompletionDeveloperMessage */
|
||||
ChatCompletionDeveloperMessage: {
|
||||
cache_control?: components["schemas"]["ChatCompletionCachedContent"];
|
||||
|
|
@ -22301,6 +22372,20 @@ export interface components {
|
|||
/** Url */
|
||||
url: string;
|
||||
};
|
||||
/** ChatCompletionMessageCustomToolCall */
|
||||
ChatCompletionMessageCustomToolCall: {
|
||||
custom: components["schemas"]["ChatCompletionCustomToolCallPayload"];
|
||||
/** Id */
|
||||
id: string;
|
||||
/**
|
||||
* Type
|
||||
* @default custom
|
||||
* @constant
|
||||
*/
|
||||
type: "custom";
|
||||
} & {
|
||||
[key: string]: unknown;
|
||||
};
|
||||
/** ChatCompletionMessageToolCall */
|
||||
ChatCompletionMessageToolCall: {
|
||||
[key: string]: unknown;
|
||||
|
|
@ -27939,7 +28024,7 @@ export interface components {
|
|||
/** Thinking Blocks */
|
||||
thinking_blocks?: (components["schemas"]["ChatCompletionThinkingBlock"] | components["schemas"]["ChatCompletionRedactedThinkingBlock"])[] | null;
|
||||
/** Tool Calls */
|
||||
tool_calls: components["schemas"]["ChatCompletionMessageToolCall"][] | null;
|
||||
tool_calls: (components["schemas"]["ChatCompletionMessageToolCall"] | components["schemas"]["ChatCompletionMessageCustomToolCall"])[] | null;
|
||||
} & {
|
||||
[key: string]: unknown;
|
||||
};
|
||||
|
|
@ -38670,6 +38755,46 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
cursor_model_list_cursor_models_get: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
cursor_model_list_cursor_v1_models_get: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
cursor_proxy_route_cursor__endpoint__get: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue