mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-21 00:21:49 +00:00
fix(gemini): preserve candidates with finishReason and no content (#40477)
This commit is contained in:
parent
1c61c2606e
commit
97dbd2dfbf
5 changed files with 184 additions and 15 deletions
|
|
@ -224,6 +224,7 @@ _FINISH_REASON_MAP: Final[dict[str, OpenAIChatCompletionFinishReason]] = {
|
|||
"IMAGE_PROHIBITED_CONTENT": "content_filter",
|
||||
"TOO_MANY_TOOL_CALLS": "stop",
|
||||
"MALFORMED_RESPONSE": "stop",
|
||||
"NO_IMAGE": "content_filter",
|
||||
# Zhipu GLM
|
||||
"network_error": "stop",
|
||||
"sensitive": "content_filter",
|
||||
|
|
|
|||
|
|
@ -1367,6 +1367,8 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
return "max_tokens"
|
||||
elif openai_finish_reason == "tool_calls":
|
||||
return "tool_use"
|
||||
elif openai_finish_reason in ["content_filter", "refusal"]:
|
||||
return "refusal"
|
||||
return "end_turn"
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -1340,6 +1340,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
"IMAGE_PROHIBITED_CONTENT",
|
||||
"TOO_MANY_TOOL_CALLS",
|
||||
"MALFORMED_RESPONSE",
|
||||
"NO_IMAGE",
|
||||
}
|
||||
)
|
||||
|
||||
|
|
@ -2224,22 +2225,23 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
|
||||
grounding_metadata: Final[list[dict]] = []
|
||||
url_context_metadata: Final[list[dict]] = []
|
||||
image_response: list[ImageURLListItem] | None = None
|
||||
safety_ratings: Final[list] = []
|
||||
citation_metadata: Final[list] = []
|
||||
chat_completion_message: Final[ChatCompletionResponseMessage] = {"role": "assistant"}
|
||||
chat_completion_logprobs: ChoiceLogprobs | None = None
|
||||
tools: list[ChatCompletionToolCallChunk] | None = []
|
||||
functions: ChatCompletionToolCallFunctionChunk | None = None
|
||||
thinking_blocks: list[ChatCompletionThinkingBlock] | None = None
|
||||
reasoning_content: str | None = None
|
||||
thought_signatures: Sequence[str] | None = None
|
||||
server_side_tool_invocations: list[dict[str, object]] | None = None
|
||||
|
||||
for idx, candidate in enumerate(_candidates):
|
||||
if "content" not in candidate:
|
||||
if "content" not in candidate and "finishReason" not in candidate:
|
||||
continue
|
||||
|
||||
image_response: list[ImageURLListItem] | None = None
|
||||
chat_completion_message: ChatCompletionResponseMessage = {"role": "assistant"}
|
||||
chat_completion_logprobs: ChoiceLogprobs | None = None
|
||||
tools: list[ChatCompletionToolCallChunk] | None = []
|
||||
functions: ChatCompletionToolCallFunctionChunk | None = None
|
||||
thinking_blocks: list[ChatCompletionThinkingBlock] | None = None
|
||||
reasoning_content: str | None = None
|
||||
thought_signatures: Sequence[str] | None = None
|
||||
server_side_tool_invocations: list[dict[str, object]] | None = None
|
||||
|
||||
# Extract metadata using helper function
|
||||
(
|
||||
candidate_grounding_metadata,
|
||||
|
|
@ -2253,7 +2255,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
safety_ratings.extend(candidate_safety_ratings)
|
||||
citation_metadata.extend(candidate_citation_metadata)
|
||||
|
||||
if "parts" in candidate["content"]:
|
||||
if "content" in candidate and candidate["content"] and "parts" in candidate["content"]:
|
||||
(
|
||||
content,
|
||||
reasoning_content,
|
||||
|
|
@ -2348,6 +2350,11 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
tool_invocation_fields["server_side_tool_invocations"] = server_side_tool_invocations
|
||||
chat_completion_message["provider_specific_fields"] = tool_invocation_fields
|
||||
|
||||
if candidate.get("finishReason"):
|
||||
finish_reason_fields = chat_completion_message.get("provider_specific_fields") or {}
|
||||
finish_reason_fields["native_finish_reason"] = candidate.get("finishReason")
|
||||
chat_completion_message["provider_specific_fields"] = finish_reason_fields
|
||||
|
||||
if isinstance(model_response, ModelResponseStream):
|
||||
choice = VertexGeminiConfig._create_streaming_choice(
|
||||
chat_completion_message=chat_completion_message,
|
||||
|
|
@ -2368,6 +2375,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
message=chat_completion_message,
|
||||
logprobs=chat_completion_logprobs,
|
||||
enhancements=None,
|
||||
provider_specific_fields=chat_completion_message.get("provider_specific_fields"),
|
||||
)
|
||||
model_response.choices.append(choice)
|
||||
|
||||
|
|
|
|||
|
|
@ -2272,13 +2272,27 @@ class LiteLLMCompletionResponsesConfig:
|
|||
if choices and len(choices) > 0:
|
||||
finish_reason = choices[0].finish_reason
|
||||
|
||||
status: Final[ResponsesAPIStatus] = (
|
||||
LiteLLMCompletionResponsesConfig._map_chat_completion_finish_reason_to_responses_status(
|
||||
finish_reason
|
||||
)
|
||||
)
|
||||
incomplete_details = getattr(chat_completion_response, "incomplete_details", None)
|
||||
if incomplete_details is None and status == "incomplete":
|
||||
from openai.types.responses.response import IncompleteDetails
|
||||
|
||||
if finish_reason == "length":
|
||||
incomplete_details = IncompleteDetails(reason="max_output_tokens")
|
||||
elif finish_reason in ["content_filter", "refusal"]:
|
||||
incomplete_details = IncompleteDetails(reason="content_filter")
|
||||
|
||||
responses_api_response: Final[ResponsesAPIResponse] = ResponsesAPIResponse(
|
||||
id=chat_completion_response.id,
|
||||
created_at=chat_completion_response.created,
|
||||
model=chat_completion_response.model,
|
||||
object="response",
|
||||
error=getattr(chat_completion_response, "error", None),
|
||||
incomplete_details=getattr(chat_completion_response, "incomplete_details", None),
|
||||
incomplete_details=incomplete_details,
|
||||
instructions=getattr(chat_completion_response, "instructions", None),
|
||||
metadata=getattr(chat_completion_response, "metadata", {}),
|
||||
output=LiteLLMCompletionResponsesConfig._transform_chat_completion_choices_to_responses_output(
|
||||
|
|
@ -2296,9 +2310,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
max_output_tokens=getattr(chat_completion_response, "max_output_tokens", None),
|
||||
previous_response_id=getattr(chat_completion_response, "previous_response_id", None),
|
||||
reasoning=None,
|
||||
status=LiteLLMCompletionResponsesConfig._map_chat_completion_finish_reason_to_responses_status(
|
||||
finish_reason
|
||||
),
|
||||
status=status,
|
||||
text={},
|
||||
truncation=getattr(chat_completion_response, "truncation", None),
|
||||
usage=LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
|
|
|
|||
|
|
@ -5836,3 +5836,149 @@ def test_supported_reasoning_efforts_still_map(model):
|
|||
drop_params=False,
|
||||
)
|
||||
assert "thinkingConfig" in result
|
||||
|
||||
|
||||
def test_gemini_candidate_with_finish_reason_no_content_chat_completion():
|
||||
config = VertexGeminiConfig()
|
||||
completion_response = {
|
||||
"candidates": [
|
||||
{
|
||||
"finishReason": "NO_IMAGE",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 19,
|
||||
"candidatesTokenCount": 0,
|
||||
"totalTokenCount": 19,
|
||||
},
|
||||
}
|
||||
model_response = ModelResponse()
|
||||
logging_obj = MagicMock()
|
||||
raw_response = MagicMock()
|
||||
raw_response.headers = {}
|
||||
|
||||
resp = config._transform_google_generate_content_to_openai_model_response(
|
||||
completion_response=completion_response,
|
||||
model_response=model_response,
|
||||
model="gemini-2.5-flash-image",
|
||||
logging_obj=logging_obj,
|
||||
raw_response=raw_response,
|
||||
)
|
||||
assert len(resp.choices) == 1
|
||||
assert resp.choices[0].finish_reason == "content_filter"
|
||||
assert resp.choices[0].message.content is None
|
||||
assert resp.choices[0].provider_specific_fields["native_finish_reason"] == "NO_IMAGE"
|
||||
|
||||
|
||||
def test_gemini_candidate_with_finish_reason_no_content_anthropic_messages():
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
|
||||
config = VertexGeminiConfig()
|
||||
completion_response = {
|
||||
"candidates": [
|
||||
{
|
||||
"finishReason": "NO_IMAGE",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 19,
|
||||
"candidatesTokenCount": 0,
|
||||
"totalTokenCount": 19,
|
||||
},
|
||||
}
|
||||
resp = config._transform_google_generate_content_to_openai_model_response(
|
||||
completion_response=completion_response,
|
||||
model_response=ModelResponse(),
|
||||
model="gemini-2.5-flash-image",
|
||||
logging_obj=MagicMock(),
|
||||
raw_response=MagicMock(headers={}),
|
||||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
anthropic_resp = adapter.translate_openai_response_to_anthropic(
|
||||
response=resp,
|
||||
tool_name_mapping={},
|
||||
)
|
||||
assert anthropic_resp["stop_reason"] == "refusal"
|
||||
assert anthropic_resp["content"] == []
|
||||
|
||||
|
||||
def test_gemini_candidate_with_finish_reason_no_content_responses_api():
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
config = VertexGeminiConfig()
|
||||
completion_response = {
|
||||
"candidates": [
|
||||
{
|
||||
"finishReason": "NO_IMAGE",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 19,
|
||||
"candidatesTokenCount": 0,
|
||||
"totalTokenCount": 19,
|
||||
},
|
||||
}
|
||||
resp = config._transform_google_generate_content_to_openai_model_response(
|
||||
completion_response=completion_response,
|
||||
model_response=ModelResponse(),
|
||||
model="gemini-2.5-flash-image",
|
||||
logging_obj=MagicMock(),
|
||||
raw_response=MagicMock(headers={}),
|
||||
)
|
||||
|
||||
responses_resp = LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
|
||||
request_input="Generate picture",
|
||||
responses_api_request={},
|
||||
chat_completion_response=resp,
|
||||
)
|
||||
assert responses_resp.status == "incomplete"
|
||||
assert responses_resp.incomplete_details is not None
|
||||
assert responses_resp.incomplete_details.reason == "content_filter"
|
||||
|
||||
|
||||
def test_gemini_candidate_other_finish_reasons_no_content():
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
config = VertexGeminiConfig()
|
||||
max_tokens_response = {
|
||||
"candidates": [{"finishReason": "MAX_TOKENS", "index": 0}],
|
||||
"usageMetadata": {"promptTokenCount": 10, "candidatesTokenCount": 50, "totalTokenCount": 60},
|
||||
}
|
||||
resp_length = config._transform_google_generate_content_to_openai_model_response(
|
||||
completion_response=max_tokens_response,
|
||||
model_response=ModelResponse(),
|
||||
model="gemini-2.5-flash",
|
||||
logging_obj=MagicMock(),
|
||||
raw_response=MagicMock(headers={}),
|
||||
)
|
||||
assert len(resp_length.choices) == 1
|
||||
assert resp_length.choices[0].finish_reason == "length"
|
||||
assert resp_length.choices[0].provider_specific_fields["native_finish_reason"] == "MAX_TOKENS"
|
||||
|
||||
anthropic_length = LiteLLMAnthropicMessagesAdapter().translate_openai_response_to_anthropic(
|
||||
response=resp_length,
|
||||
tool_name_mapping={},
|
||||
)
|
||||
assert anthropic_length["stop_reason"] == "max_tokens"
|
||||
|
||||
responses_length = LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
|
||||
request_input="thinking request",
|
||||
responses_api_request={},
|
||||
chat_completion_response=resp_length,
|
||||
)
|
||||
assert responses_length.status == "incomplete"
|
||||
assert responses_length.incomplete_details.reason == "max_output_tokens"
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue