fix(github_copilot): synthesize empty choices at the provider seam (#30929)

Newer Copilot Claude models (opus-4.7, opus-4.8) return responses with
choices=[], either carrying Anthropic-native content blocks or, for the
max_tokens=1 probe Claude Code sends, no content at all. github_copilot
is dispatched through the OpenAI SDK handler, which calls
convert_to_model_response_object directly and never invokes
GithubCopilotConfig.transform_response, so the empty-choices guard there
surfaced as a 500

Instead of synthesizing choices inside the shared
convert_to_model_response_object (which would silently turn empty choices
into a fabricated success for every provider), add a no-op
transform_parsed_response_dict hook on BaseConfig. GithubCopilotConfig
overrides it to synthesize choices from Anthropic-native content, reusing
its existing parsing, and the OpenAI SDK handler routes its parsed
response through the hook before generic conversion. The core utility
keeps treating empty choices as an error for all other providers

Fixes: https://github.com/BerriAI/litellm/issues/30927

Signed-off-by: David J. M. Karlsen <david@davidkarlsen.com>
This commit is contained in:
David J. M. Karlsen 2026-06-24 13:19:38 +02:00 • committed by GitHub
parent e1187c0462
commit b302c4204a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 224 additions and 74 deletions

View file

@ -377,6 +377,17 @@ class BaseConfig(ABC):
) -> "ModelResponse":
pass
def transform_parsed_response_dict(self, parsed_response: dict) -> dict:
"""
Repair a parsed OpenAI-format response dict before generic conversion.
Providers routed through the OpenAI SDK handler bypass transform_response,
which calls convert_to_model_response_object directly on the SDK's parsed
output. Override this to normalize a malformed response (e.g. github_copilot
returning empty choices for Anthropic-native Claude responses).
"""
return parsed_response
@abstractmethod
def get_error_class(
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]

View file

@ -194,6 +194,88 @@ class GithubCopilotConfig(OpenAIConfig):
)
return text_content, tool_calls, thinking_blocks
@staticmethod
def _normalize_anthropic_usage(usage: dict) -> dict:
normalized = dict(usage)
if "input_tokens" in usage and "prompt_tokens" not in usage:
normalized["prompt_tokens"] = usage["input_tokens"]
if "output_tokens" in usage and "completion_tokens" not in usage:
normalized["completion_tokens"] = usage["output_tokens"]
if "total_tokens" not in normalized:
normalized["total_tokens"] = normalized.get(
"prompt_tokens", 0
) + normalized.get("completion_tokens", 0)
return normalized
@classmethod
def _synthesize_choices_for_anthropic_native(cls, response_json: dict) -> dict:
"""
Synthesize a `choices` array from an Anthropic-native Copilot response.
Newer Copilot Claude models (e.g. opus-4.7, opus-4.8) return content
blocks and `stop_reason` without an OpenAI-style `choices` array, and the
max_tokens=1 probe returns no content at all. Returns the response
unchanged when it already carries choices.
See: https://github.com/BerriAI/litellm/issues/29391
"""
if response_json.get("choices"):
return response_json
content = ""
tool_calls: List[ChatCompletionToolCallChunk] = []
thinking_blocks: Optional[List[Any]] = None
raw_content = response_json.get("content")
if isinstance(raw_content, list):
content, tool_calls, thinking_blocks = cls._parse_anthropic_native_content(
raw_content
)
elif isinstance(raw_content, str):
content = raw_content
stop_reason = response_json.get("stop_reason")
finish_reason_map = {
"end_turn": "stop",
"max_tokens": "length",
"stop_sequence": "stop",
"tool_use": "tool_calls",
}
if tool_calls:
finish_reason = "tool_calls"
elif stop_reason in finish_reason_map:
finish_reason = finish_reason_map[stop_reason]
elif content:
finish_reason = "stop"
else:
finish_reason = "length"
message: dict = {
"role": "assistant",
"content": content if content or not tool_calls else None,
}
if tool_calls:
message["tool_calls"] = tool_calls
if thinking_blocks:
message["thinking_blocks"] = thinking_blocks
synthesized = {
**response_json,
"choices": [
{"index": 0, "message": message, "finish_reason": finish_reason}
],
}
usage = response_json.get("usage")
if isinstance(usage, dict):
synthesized["usage"] = cls._normalize_anthropic_usage(usage)
return synthesized
def transform_parsed_response_dict(self, parsed_response: dict) -> dict:
"""
Repair the OpenAI-SDK-parsed response on the handler path that bypasses
transform_response. See: https://github.com/BerriAI/litellm/issues/30927
"""
return self._synthesize_choices_for_anthropic_native(parsed_response)
def transform_response(
self,
model: str,
@ -208,15 +290,6 @@ class GithubCopilotConfig(OpenAIConfig):
api_key: Optional[str] = None,
json_mode: Optional[bool] = None,
) -> "ModelResponse":
"""
Handle newer Copilot models (e.g. claude-opus-4.7, claude-opus-4.8) that
return Anthropic-native format responses without a `choices` array.
Synthesizes the missing `choices` from Anthropic-native fields, then
delegates to the parent so all standard post-processing applies.
See: https://github.com/BerriAI/litellm/issues/29391
"""
try:
response_json = raw_response.json()
except Exception:
@ -235,70 +308,12 @@ class GithubCopilotConfig(OpenAIConfig):
)
if not response_json.get("choices"):
content = ""
tool_calls: List[ChatCompletionToolCallChunk] = []
thinking_blocks: Optional[List[Any]] = None
if "content" in response_json and isinstance(
response_json["content"], list
):
content, tool_calls, thinking_blocks = (
self._parse_anthropic_native_content(response_json["content"])
)
elif isinstance(response_json.get("content"), str):
content = response_json["content"]
stop_reason = response_json.get("stop_reason")
finish_reason_map = {
"end_turn": "stop",
"max_tokens": "length",
"stop_sequence": "stop",
"tool_use": "tool_calls",
}
# Prefer tool_calls when blocks were extracted; otherwise map stop_reason.
if tool_calls:
finish_reason = "tool_calls"
elif stop_reason in finish_reason_map:
finish_reason = finish_reason_map[stop_reason]
elif content:
finish_reason = "stop"
else:
finish_reason = "length"
message: dict = {
"role": "assistant",
"content": content if content or not tool_calls else None,
}
if tool_calls:
message["tool_calls"] = tool_calls
if thinking_blocks:
message["thinking_blocks"] = thinking_blocks
response_json["choices"] = [
{
"index": 0,
"message": message,
"finish_reason": finish_reason,
}
]
if "usage" in response_json:
usage = response_json["usage"]
if "input_tokens" in usage and "prompt_tokens" not in usage:
usage["prompt_tokens"] = usage["input_tokens"]
if "output_tokens" in usage and "completion_tokens" not in usage:
usage["completion_tokens"] = usage["output_tokens"]
if "total_tokens" not in usage:
usage["total_tokens"] = usage.get("prompt_tokens", 0) + usage.get(
"completion_tokens", 0
)
# Build a patched response so super() sees valid JSON with choices
patched = httpx.Response(
response_json = self._synthesize_choices_for_anthropic_native(response_json)
raw_response = httpx.Response(
status_code=raw_response.status_code,
headers=raw_response.headers,
content=json.dumps(response_json).encode(),
)
raw_response = patched
return super().transform_response(
model=model,

View file

@ -785,7 +785,11 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
)
logging_obj.model_call_details["response_headers"] = headers
stringified_response = response.model_dump()
stringified_response = (
provider_config.transform_parsed_response_dict(
response.model_dump()
)
)
logging_obj.post_call(
input=messages,
api_key=api_key,
@ -933,7 +937,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM):
timeout=timeout,
logging_obj=logging_obj,
)
stringified_response = response.model_dump()
stringified_response = provider_config.transform_parsed_response_dict(
response.model_dump()
)
logging_obj.post_call(
input=data["messages"],
api_key=api_key,

View file

@ -1627,7 +1627,12 @@ class TestMissingChoicesGuard:
assert "no 'choices'" in exc_info.value.message
def test_convert_to_model_response_object_empty_choices_raises_api_error(self):
"""Empty choices list raises APIError."""
"""Empty choices list raises APIError, same as missing/null choices.
Provider-specific repair (e.g. github_copilot synthesizing choices for
Anthropic-native responses) happens before this guard, in the provider
config; the core utility keeps treating empty choices as an error.
"""
from litellm.exceptions import APIError
response_object = {
@ -1683,7 +1688,9 @@ class TestMissingChoicesGuard:
assert "no 'choices'" in exc_info.value.message
def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error(self):
def test_convert_to_model_response_object_stream_true_no_choices_raises_api_error(
self,
):
"""Missing choices via stream=True path raises APIError when generator is consumed."""
from litellm.exceptions import APIError
@ -2471,6 +2478,13 @@ class TestConvertToModelResponseObjectCompletion:
def test_model_response_none_raises(self):
with pytest.raises(Exception):
convert_to_model_response_object(
response_object={"choices": [{"message": {"content": "hi", "role": "assistant"}, "finish_reason": "stop"}]},
response_object={
"choices": [
{
"message": {"content": "hi", "role": "assistant"},
"finish_reason": "stop",
}
]
},
model_response_object=None,
)

View file

@ -878,3 +878,107 @@ class TestGithubCopilotTransformResponse:
litellm_params={},
encoding=None,
)
class TestGithubCopilotTransformParsedResponseDict:
"""
Tests for GithubCopilotConfig.transform_parsed_response_dict, the hook the
OpenAI SDK handler calls on its parsed response. That handler bypasses
transform_response, so this is the seam that repairs empty-choices responses
from newer Copilot Claude models on the live completion path.
See: https://github.com/BerriAI/litellm/issues/30927
"""
def test_synthesizes_choices_from_anthropic_content(self):
config = GithubCopilotConfig()
parsed = {
"id": "msg_vrtx_01",
"model": "claude-opus-4.8",
"object": "chat.completion",
"choices": [],
"content": [{"type": "text", "text": "Hello!"}],
"stop_reason": "end_turn",
"usage": {"input_tokens": 10, "output_tokens": 5},
}
repaired = config.transform_parsed_response_dict(parsed)
assert len(repaired["choices"]) == 1
choice = repaired["choices"][0]
assert choice["message"]["content"] == "Hello!"
assert choice["finish_reason"] == "stop"
assert repaired["usage"]["prompt_tokens"] == 10
assert repaired["usage"]["completion_tokens"] == 5
assert repaired["usage"]["total_tokens"] == 15
def test_passthrough_when_choices_present(self):
config = GithubCopilotConfig()
parsed = {
"id": "chatcmpl-1",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok"},
"finish_reason": "stop",
}
],
}
assert config.transform_parsed_response_dict(parsed) is parsed
@patch("litellm.llms.openai.openai.OpenAIChatCompletion._get_openai_client")
@patch(
"litellm.llms.openai.openai.OpenAIChatCompletion.make_sync_openai_chat_completion_request"
)
def test_openai_handler_repairs_github_copilot_empty_choices(
mock_request, mock_get_client
):
"""
The OpenAI SDK handler calls convert_to_model_response_object directly on the
SDK's parsed output, bypassing transform_response. convert raises APIError on
empty choices, so the handler must route github_copilot responses through
transform_parsed_response_dict first. Removing that wiring (or resolving a
config without the override) fails this test with APIError.
See: https://github.com/BerriAI/litellm/issues/30927
"""
from litellm.llms.openai.openai import OpenAIChatCompletion
mock_get_client.return_value = MagicMock()
class _FakeSDKResponse:
def model_dump(self):
return {
"id": "msg_vrtx_01",
"model": "claude-opus-4.8",
"object": "chat.completion",
"choices": [],
"content": [{"type": "text", "text": "Hi there"}],
"stop_reason": "end_turn",
"usage": {"input_tokens": 12, "output_tokens": 3},
}
mock_request.return_value = ({}, _FakeSDKResponse())
result = OpenAIChatCompletion().completion(
model="claude-opus-4.8",
messages=[{"role": "user", "content": "Hi"}],
model_response=ModelResponse(),
timeout=60.0,
optional_params={},
litellm_params={},
logging_obj=MagicMock(),
custom_llm_provider="github_copilot",
client=MagicMock(),
api_key="gh.test-key-123456789",
acompletion=False,
)
assert isinstance(result, ModelResponse)
assert result.choices[0].message.content == "Hi there"
assert result.choices[0].finish_reason == "stop"
mock_request.assert_called_once()