fix(cursor): convert tools and tool_choice through one envelope rule

Chat Completions nests a named tool_choice under its tool type while the
Responses API keeps it flat; ChatCompletionNamedToolChoiceParam and
ChatCompletionNamedToolChoiceCustomParam both mark the nested key required. The
messages arm normalized tool definitions but forwarded tool_choice at whatever
level Cursor sent it, so a flat {"type": "custom", "name": "ApplyPatch"} reached
OpenAI unchanged and was rejected while the tool defs beside it nested correctly

A tool definition and a named tool_choice carry the same envelope, so both now
convert through a single _convert_tool_envelope, and _normalize_tool_dialect
moves tools and tool_choice together on each arm. That covers all four cells of
{tool def, tool_choice} x {to chat, to responses} and removes the shape where
one field can be converted while the other is missed, replacing three helpers
with two and cutting 24 lines

Also restores the end-to-end assertion that a flat tool_choice reaches
chat_completion nested, which had been flipped to pin the passthrough behavior
This commit is contained in:
Tin Chi Lo 2026-07-22 23:51:19 -07:00
parent 56cc475c80
commit c7c656e8a9
2 changed files with 140 additions and 164 deletions

View file

@ -24,57 +24,47 @@ router = APIRouter()
_user_api_key_auth_dep = Depends(user_api_key_auth)
_FLAT_CUSTOM_TOOL_KEYS = ("name", "description", "format")
_FLAT_FUNCTION_TOOL_KEYS = ("name", "description", "parameters", "strict")
_TOOL_PAYLOAD_KEYS = {
"custom": ("name", "description", "format"),
"function": ("name", "description", "parameters", "strict"),
}
def _nest_flat_chat_tool(tool: object) -> object:
def _convert_tool_envelope(obj: object, *, to_chat: bool) -> object:
from litellm.litellm_core_utils.prompt_templates.common_utils import (
convert_custom_tool_format_to_chat_shape,
)
if not isinstance(tool, dict):
return tool
if tool.get("type") == "custom":
if isinstance(tool.get("custom"), dict):
envelope = tool
payload = tool["custom"]
elif "name" in tool:
envelope = {"type": "custom"}
payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool}
else:
return tool
if isinstance(payload.get("format"), dict):
payload = {**payload, "format": convert_custom_tool_format_to_chat_shape(payload["format"])}
return {**envelope, "custom": payload}
if tool.get("type") == "function" and "function" not in tool and "name" in tool:
return {"type": "function", "function": {k: tool[k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool}}
return tool
def _flatten_chat_tool_for_responses(tool: object) -> object:
from litellm.litellm_core_utils.prompt_templates.common_utils import (
convert_custom_tool_format_to_responses_shape,
)
if not isinstance(tool, dict):
return tool
if tool.get("type") == "custom":
if isinstance(tool.get("custom"), dict):
payload = {k: tool["custom"][k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool["custom"]}
elif "name" in tool:
payload = {k: tool[k] for k in _FLAT_CUSTOM_TOOL_KEYS if k in tool}
else:
return tool
if isinstance(payload.get("format"), dict):
payload = {**payload, "format": convert_custom_tool_format_to_responses_shape(payload["format"])}
return {"type": "custom", **payload}
if tool.get("type") == "function" and isinstance(tool.get("function"), dict):
return {
"type": "function",
**{k: tool["function"][k] for k in _FLAT_FUNCTION_TOOL_KEYS if k in tool["function"]},
}
return tool
if not isinstance(obj, dict):
return obj
tool_type = obj.get("type")
payload_keys = _TOOL_PAYLOAD_KEYS.get(tool_type)
if payload_keys is None:
return obj
nested = obj.get(tool_type)
source = nested if isinstance(nested, dict) else obj
if source is obj and "name" not in obj:
return obj
payload = {key: source[key] for key in payload_keys if key in source}
if isinstance(payload.get("format"), dict):
convert = convert_custom_tool_format_to_chat_shape if to_chat else convert_custom_tool_format_to_responses_shape
payload = {**payload, "format": convert(payload["format"])}
return {"type": tool_type, tool_type: payload} if to_chat else {"type": tool_type, **payload}
def _normalize_tool_dialect(data: dict, *, to_chat: bool) -> dict:
converted: dict = {}
tools = data.get("tools")
if isinstance(tools, list):
normalized_tools = [_convert_tool_envelope(tool, to_chat=to_chat) for tool in tools]
if normalized_tools != tools:
converted["tools"] = normalized_tools
tool_choice = data.get("tool_choice")
normalized_choice = _convert_tool_envelope(tool_choice, to_chat=to_chat)
if normalized_choice != tool_choice:
converted["tool_choice"] = normalized_choice
return {**data, **converted} if converted else data
def _is_chat_completions_body(data: dict) -> bool:
@ -84,18 +74,6 @@ def _is_chat_completions_body(data: dict) -> bool:
return "messages" in data and "input" not in data
def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object:
if not isinstance(tool_choice, dict):
return tool_choice
choice_type = tool_choice.get("type")
if choice_type not in ("custom", "function"):
return tool_choice
nested = tool_choice.get(choice_type)
if isinstance(nested, dict) and isinstance(nested.get("name"), str):
return {"type": choice_type, "name": nested["name"]}
return tool_choice
@router.post(
"/v1/responses",
dependencies=[Depends(user_api_key_auth)],
@ -450,14 +428,9 @@ async def cursor_chat_completions(
# already fixed); delegate so behavior matches /chat/completions exactly.
# Keyed on messages CONTENT, not key presence: Cursor can send a null or
# empty messages stub alongside a real agent-mode input array
tools = data.get("tools")
normalized: dict = {}
if isinstance(tools, list):
nested_tools = [_nest_flat_chat_tool(tool) for tool in tools]
if nested_tools != tools:
normalized["tools"] = nested_tools
if normalized:
_safe_set_request_parsed_body(request=request, parsed_body={**data, **normalized})
normalized = _normalize_tool_dialect(data, to_chat=True)
if normalized is not data:
_safe_set_request_parsed_body(request=request, parsed_body=normalized)
return await chat_completion(
request=request,
fastapi_response=fastapi_response,
@ -472,13 +445,7 @@ async def cursor_chat_completions(
# cache's key snapshot so later readers get an empty body
data = {key: value for key, value in data.items() if key != "stream_options"}
tools = data.get("tools")
if isinstance(tools, list):
data = {**data, "tools": [_flatten_chat_tool_for_responses(tool) for tool in tools]}
tool_choice = data.get("tool_choice")
flattened_tool_choice = _flatten_chat_tool_choice_for_responses(tool_choice)
if flattened_tool_choice != tool_choice:
data = {**data, "tool_choice": flattened_tool_choice}
data = _normalize_tool_dialect(data, to_chat=False)
processor = ProxyBaseLLMRequestProcessing(data=data)

View file

@ -911,10 +911,11 @@ def test_cursor_models_route_delegates_to_model_list():
class TestNestFlatChatTools:
def test_flat_custom_tool_is_nested(self):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
result = _nest_flat_chat_tool(
{"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}}
result = _convert_tool_envelope(
{"type": "custom", "name": "ApplyPatch", "description": "V4A patch", "format": {"type": "text"}},
to_chat=True,
)
assert result == {
"type": "custom",
@ -922,10 +923,11 @@ class TestNestFlatChatTools:
}
def test_flat_function_tool_is_nested(self):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
result = _nest_flat_chat_tool(
{"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}}
result = _convert_tool_envelope(
{"type": "function", "name": "read_file", "description": "d", "parameters": {"type": "object"}},
to_chat=True,
)
assert result == {
"type": "function",
@ -933,7 +935,7 @@ class TestNestFlatChatTools:
}
def test_already_nested_and_unrecognized_tools_pass_through_unchanged(self):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
tools = [
{"type": "custom", "custom": {"name": "already_nested"}},
@ -946,7 +948,7 @@ class TestNestFlatChatTools:
None,
42,
]
assert [_nest_flat_chat_tool(tool) for tool in tools] == tools
assert [_convert_tool_envelope(tool, to_chat=True) for tool in tools] == tools
class TestCursorMessagesArmToolNormalization:
@ -1010,7 +1012,7 @@ class TestCursorMessagesArmToolNormalization:
},
},
]
assert seen["body"]["tool_choice"] == {"type": "custom", "name": "ApplyPatch"}
assert seen["body"]["tool_choice"] == {"type": "custom", "custom": {"name": "ApplyPatch"}}
assert seen["body"]["messages"] == [{"role": "user", "content": "use ApplyPatch"}]
@pytest.mark.asyncio
@ -1048,21 +1050,23 @@ class TestCursorMessagesArmToolNormalization:
assert seen["body"]["messages"] == body["messages"]
class TestNestFlatChatToolShapeMatrix:
class TestToolEnvelopeConversionMatrix:
"""
Cursor mixes Responses API shapes into chat bodies PER LEVEL, independently
(live-captured: a pre-nested custom envelope carrying a flat grammar format).
Every cell of envelope x format must land on the canonical chat shape.
Tool definitions and tool_choice share one envelope rule, so every cell of
direction x envelope x format must land on that direction's canonical shape.
"""
FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"}
NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}}
TEXT = {"type": "text"}
@pytest.mark.parametrize("to_chat", [True, False])
@pytest.mark.parametrize("envelope", ["flat", "nested"])
@pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"])
def test_every_envelope_and_format_combination_lands_canonical(self, envelope, format_shape):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
def test_every_direction_envelope_and_format_lands_canonical(self, to_chat, envelope, format_shape):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
format_value = {
"absent": None,
@ -1077,109 +1081,114 @@ class TestNestFlatChatToolShapeMatrix:
canonical_payload = {"name": "ApplyPatch", "description": "V4A patch"}
if format_shape in ("flat_grammar", "nested_grammar"):
canonical_payload["format"] = self.NESTED_GRAMMAR
canonical_payload["format"] = self.NESTED_GRAMMAR if to_chat else self.FLAT_GRAMMAR
elif format_shape == "text":
canonical_payload["format"] = self.TEXT
expected = (
{"type": "custom", "custom": canonical_payload} if to_chat else {"type": "custom", **canonical_payload}
)
assert _nest_flat_chat_tool(tool) == {"type": "custom", "custom": canonical_payload}
assert _convert_tool_envelope(tool, to_chat=to_chat) == expected
def test_nested_envelope_with_flat_grammar_matches_live_cursor_capture(self):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
cursor_tool = {
cursor_tool = {"type": "custom", "custom": {"name": "ApplyPatch", "format": self.FLAT_GRAMMAR}}
assert _convert_tool_envelope(cursor_tool, to_chat=True) == {
"type": "custom",
"custom": {
"name": "ApplyPatch",
"format": {"type": "grammar", "definition": "start: patch", "syntax": "lark"},
},
}
assert _nest_flat_chat_tool(cursor_tool) == {
"type": "custom",
"custom": {
"name": "ApplyPatch",
"format": {
"type": "grammar",
"grammar": {"definition": "start: patch", "syntax": "lark"},
},
},
"custom": {"name": "ApplyPatch", "format": self.NESTED_GRAMMAR},
}
def test_canonical_nested_tool_is_returned_equal(self):
from litellm.proxy.response_api_endpoints.endpoints import _nest_flat_chat_tool
@pytest.mark.parametrize("to_chat", [True, False])
def test_conversion_is_idempotent(self, to_chat):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
canonical = {
"type": "custom",
"custom": {"name": "A", "format": {"type": "grammar", "grammar": {"definition": "d", "syntax": "lark"}}},
}
assert _nest_flat_chat_tool(canonical) == canonical
once = _convert_tool_envelope({"type": "custom", "name": "A", "format": self.FLAT_GRAMMAR}, to_chat=to_chat)
assert _convert_tool_envelope(once, to_chat=to_chat) == once
class TestFlattenChatToolsForResponsesInputArm:
"""
Mirror of TestNestFlatChatToolShapeMatrix for the input arm: chat-nested shapes in a
Responses-shaped body must flatten to the Responses dialect, per level, idempotently.
"""
FLAT_GRAMMAR = {"type": "grammar", "definition": "start: patch", "syntax": "lark"}
NESTED_GRAMMAR = {"type": "grammar", "grammar": {"definition": "start: patch", "syntax": "lark"}}
@pytest.mark.parametrize("envelope", ["flat", "nested"])
@pytest.mark.parametrize("format_shape", ["absent", "text", "flat_grammar", "nested_grammar"])
def test_every_envelope_and_format_combination_lands_flat(self, envelope, format_shape):
from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses
format_value = {
"absent": None,
"text": {"type": "text"},
"flat_grammar": self.FLAT_GRAMMAR,
"nested_grammar": self.NESTED_GRAMMAR,
}[format_shape]
payload = {"name": "ApplyPatch", "description": "V4A patch"}
if format_value is not None:
payload["format"] = format_value
tool = {"type": "custom", "custom": payload} if envelope == "nested" else {"type": "custom", **payload}
canonical = {"type": "custom", "name": "ApplyPatch", "description": "V4A patch"}
if format_shape in ("flat_grammar", "nested_grammar"):
canonical["format"] = self.FLAT_GRAMMAR
elif format_shape == "text":
canonical["format"] = {"type": "text"}
assert _flatten_chat_tool_for_responses(tool) == canonical
def test_nested_function_tool_is_flattened_and_flat_passes_through(self):
from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses
def test_nested_function_tool_flattens_and_flat_passes_through(self):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
nested = {"type": "function", "function": {"name": "read_file", "parameters": {"type": "object"}}}
flat = {"type": "function", "name": "read_file", "parameters": {"type": "object"}}
assert _flatten_chat_tool_for_responses(nested) == flat
assert _flatten_chat_tool_for_responses(flat) == flat
assert _convert_tool_envelope(nested, to_chat=False) == flat
assert _convert_tool_envelope(flat, to_chat=False) == flat
def test_unrecognized_entries_pass_through(self):
from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_for_responses
@pytest.mark.parametrize("to_chat", [True, False])
def test_unrecognized_entries_pass_through(self, to_chat):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}]
assert [_flatten_chat_tool_for_responses(entry) for entry in entries] == entries
entries = [{"type": "web_search"}, {"type": "custom"}, "junk", None, {}, 42, {"type": "auto"}]
assert [_convert_tool_envelope(entry, to_chat=to_chat) for entry in entries] == entries
class TestFlattenChatToolChoiceForResponsesInputArm:
def test_nested_custom_and_function_tool_choice_flatten(self):
from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses
class TestToolChoiceSharesTheToolEnvelopeRule:
"""
tool_choice carries the same {"type": T, T: {...}} chat envelope as a tool
definition, so it converts through the same function in both directions.
OpenAI requires the nested key on chat (SDK ChatCompletionNamedToolChoiceParam
and ChatCompletionNamedToolChoiceCustomParam both mark it Required).
"""
assert _flatten_chat_tool_choice_for_responses({"type": "custom", "custom": {"name": "ApplyPatch"}}) == {
"type": "custom",
@pytest.mark.parametrize("choice_type", ["custom", "function"])
def test_flat_tool_choice_is_nested_for_chat(self, choice_type):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
assert _convert_tool_envelope({"type": choice_type, "name": "ApplyPatch"}, to_chat=True) == {
"type": choice_type,
choice_type: {"name": "ApplyPatch"},
}
@pytest.mark.parametrize("choice_type", ["custom", "function"])
def test_nested_tool_choice_is_flattened_for_responses(self, choice_type):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
assert _convert_tool_envelope({"type": choice_type, choice_type: {"name": "ApplyPatch"}}, to_chat=False) == {
"type": choice_type,
"name": "ApplyPatch",
}
assert _flatten_chat_tool_choice_for_responses({"type": "function", "function": {"name": "f"}}) == {
"type": "function",
"name": "f",
}
def test_flat_and_string_tool_choice_pass_through(self):
from litellm.proxy.response_api_endpoints.endpoints import _flatten_chat_tool_choice_for_responses
@pytest.mark.parametrize("to_chat", [True, False])
def test_sentinel_and_malformed_tool_choice_pass_through(self, to_chat):
from litellm.proxy.response_api_endpoints.endpoints import _convert_tool_envelope
for unchanged in ("auto", "required", None, {"type": "custom", "name": "x"}, {"type": "auto"}, 42):
assert _flatten_chat_tool_choice_for_responses(unchanged) == unchanged
for unchanged in ("auto", "required", "none", None, {"type": "auto"}, 42):
assert _convert_tool_envelope(unchanged, to_chat=to_chat) == unchanged
class TestNormalizeToolDialectCoversBothFields:
"""
The regression that motivated one normalizer: tools were converted while
tool_choice was left flat, so OpenAI rejected the request. Both fields move
together in a single call, on both arms.
"""
@pytest.mark.parametrize("to_chat", [True, False])
def test_tools_and_tool_choice_convert_together(self, to_chat):
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
flat = {"type": "custom", "name": "ApplyPatch"}
nested = {"type": "custom", "custom": {"name": "ApplyPatch"}}
source = flat if to_chat else nested
expected = nested if to_chat else flat
out = _normalize_tool_dialect({"messages": [], "tools": [source], "tool_choice": source}, to_chat=to_chat)
assert out["tools"] == [expected]
assert out["tool_choice"] == expected
def test_body_needing_no_conversion_is_returned_by_identity(self):
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
data = {"messages": [], "tools": [{"type": "function", "function": {"name": "f"}}], "tool_choice": "auto"}
assert _normalize_tool_dialect(data, to_chat=True) is data
def test_absent_tool_fields_are_not_invented(self):
from litellm.proxy.response_api_endpoints.endpoints import _normalize_tool_dialect
data = {"messages": [{"role": "user", "content": "hi"}]}
result = _normalize_tool_dialect(data, to_chat=True)
assert result == data
assert "tools" not in result and "tool_choice" not in result
class TestCursorInputArmFlattening: