mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge pull request #18135 from BerriAI/litellm_gemini_flash_day_0
feat: gemini-3-flash-preview day 0 support
This commit is contained in:
commit
25f2213262
10 changed files with 930 additions and 22 deletions
222
docs/my-website/blog/gemini_3_flash/index.md
Normal file
222
docs/my-website/blog/gemini_3_flash/index.md
Normal file
|
|
@ -0,0 +1,222 @@
|
|||
---
|
||||
slug: gemini_3_flash
|
||||
title: "DAY 0 Support: Gemini 3 Flash on LiteLLM"
|
||||
date: 2025-12-17T10:00:00
|
||||
authors:
|
||||
- name: Sameer Kankute
|
||||
title: SWE @ LiteLLM (LLM Translation)
|
||||
url: https://www.linkedin.com/in/sameer-kankute/
|
||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQHB_loQYd5gjg/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1719137160975?e=1765411200&v=beta&t=c8396f--_lH6Fb_pVvx_jGholPfcl0bvwmNynbNdnII
|
||||
- name: Krrish Dholakia
|
||||
title: "CEO, LiteLLM"
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||
- name: Ishaan Jaff
|
||||
title: "CTO, LiteLLM"
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
tags: [gemini, day 0 support, llms]
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
|
||||
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Gemini 3 Flash Day 0 Support
|
||||
|
||||
LiteLLM now supports `gemini-3-flash-preview` and all the new API changes along with it.
|
||||
|
||||
## What's New
|
||||
|
||||
### 1. New Thinking Levels: `thinkingLevel` with MINIMAL & MEDIUM
|
||||
|
||||
Gemini 3 Flash introduces granular thinking control with `thinkingLevel` instead of `thinkingBudget`.
|
||||
- **MINIMAL**: Ultra-lightweight thinking for fast responses
|
||||
- **MEDIUM**: Balanced thinking for complex reasoning
|
||||
- **HIGH**: Maximum reasoning depth
|
||||
|
||||
LiteLLM automatically maps the OpenAI `reasoning_effort` parameter to Gemini's `thinkingLevel`, so you can use familiar `reasoning_effort` values (`minimal`, `low`, `medium`, `high`) without changing your code!
|
||||
|
||||
### 2. Thought Signatures
|
||||
|
||||
Like `gemini-3-pro`, this model also includes thought signatures for tool calls. LiteLLM handles signature extraction and embedding internally. [Learn more about thought signatures](../gemini_3/index.md#thought-signatures).
|
||||
|
||||
**Edge Case Handling**: If thought signatures are missing in the request, LiteLLM adds a dummy signature ensuring the API call doesn't break
|
||||
|
||||
---
|
||||
## Supported Endpoints
|
||||
|
||||
LiteLLM provides **full end-to-end support** for Gemini 3 Flash on:
|
||||
|
||||
- ✅ `/v1/chat/completions` - OpenAI-compatible chat completions endpoint
|
||||
- ✅ `/v1/responses` - OpenAI Responses API endpoint (streaming and non-streaming)
|
||||
- ✅ [`/v1/messages`](../../docs/anthropic_unified) - Anthropic-compatible messages endpoint
|
||||
- ✅ `/v1/generateContent` – [Google Gemini API](../../docs/generateContent.md) compatible endpoint
|
||||
All endpoints support:
|
||||
- Streaming and non-streaming responses
|
||||
- Function calling with thought signatures
|
||||
- Multi-turn conversations
|
||||
- All Gemini 3-specific features
|
||||
- Converstion of provider specific thinking related param to thinkingLevel
|
||||
|
||||
## Quick Start
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
**Basic Usage with MEDIUM thinking (NEW)**
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
# No need to make any changes to your code as we map openai reasoning param to thinkingLevel
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "Solve this complex math problem: 25 * 4 + 10"}],
|
||||
reasoning_effort="medium", # NEW: MEDIUM thinking level
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
**1. Setup config.yaml**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-3-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-3-flash-preview
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
**2. Start proxy**
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
**3. Call with MEDIUM thinking**
|
||||
|
||||
```bash
|
||||
curl -X POST http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-d '{
|
||||
"model": "gemini-3-flash",
|
||||
"messages": [{"role": "user", "content": "Complex reasoning task"}],
|
||||
"reasoning_effort": "medium"
|
||||
}'
|
||||
``'
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
---
|
||||
|
||||
## All `reasoning_effort` Levels
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="minimal" label="MINIMAL">
|
||||
|
||||
**Ultra-fast, minimal reasoning**
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "What's 2+2?"}],
|
||||
reasoning_effort="minimal",
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="low" label="LOW">
|
||||
|
||||
**Simple instruction following**
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "Write a haiku about coding"}],
|
||||
reasoning_effort="low",
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="medium" label="MEDIUM (NEW)">
|
||||
|
||||
**Balanced reasoning for complex tasks** ✨
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "Analyze this dataset and find patterns"}],
|
||||
reasoning_effort="medium", # NEW!
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="high" label="HIGH">
|
||||
|
||||
**Maximum reasoning depth**
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "Prove this mathematical theorem"}],
|
||||
reasoning_effort="high",
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
---
|
||||
|
||||
## Key Features
|
||||
|
||||
✅ **Thinking Levels**: MINIMAL, LOW, MEDIUM, HIGH
|
||||
✅ **Thought Signatures**: Track reasoning with unique identifiers
|
||||
✅ **Seamless Integration**: Works with existing OpenAI-compatible client
|
||||
✅ **Backward Compatible**: Gemini 2.5 models continue using `thinkingBudget`
|
||||
|
||||
---
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install litellm --upgrade
|
||||
```
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-3-flash-preview",
|
||||
messages=[{"role": "user", "content": "Your question here"}],
|
||||
reasoning_effort="medium", # Use MEDIUM thinking
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## `reasoning_effort` Mapping for Gemini 3+
|
||||
|
||||
| reasoning_effort | thinking_level |
|
||||
|------------------|----------------|
|
||||
| `minimal` | `minimal` |
|
||||
| `low` | `low` |
|
||||
| `medium` | `medium` |
|
||||
| `high` | `high` |
|
||||
| `disable` | `minimal` |
|
||||
| `none` | `minimal` |
|
||||
|
||||
|
|
@ -75,6 +75,7 @@ class GoogleGenAIConfig(BaseGoogleGenAIGenerateContentConfig, VertexLLM):
|
|||
"seed",
|
||||
"response_mime_type",
|
||||
"response_schema",
|
||||
"response_json_schema",
|
||||
"routing_config",
|
||||
"model_selection_config",
|
||||
"safety_settings",
|
||||
|
|
@ -105,13 +106,37 @@ class GoogleGenAIConfig(BaseGoogleGenAIGenerateContentConfig, VertexLLM):
|
|||
Returns:
|
||||
Mapped parameters for the provider
|
||||
"""
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_camel_to_snake,
|
||||
_snake_to_camel,
|
||||
)
|
||||
|
||||
_generate_content_config_dict: Dict[str, Any] = {}
|
||||
supported_google_genai_params = (
|
||||
self.get_supported_generate_content_optional_params(model)
|
||||
)
|
||||
# Create a set with both camelCase and snake_case versions for faster lookup
|
||||
supported_params_set = set(supported_google_genai_params)
|
||||
supported_params_set.update(_snake_to_camel(p) for p in supported_google_genai_params)
|
||||
supported_params_set.update(_camel_to_snake(p) for p in supported_google_genai_params if "_" not in p)
|
||||
|
||||
for param, value in generate_content_config_dict.items():
|
||||
if param in supported_google_genai_params:
|
||||
_generate_content_config_dict[param] = value
|
||||
# Google GenAI API expects camelCase, so we'll always output in camelCase
|
||||
# Check if param (or its variants) is supported
|
||||
param_snake = _camel_to_snake(param)
|
||||
param_camel = _snake_to_camel(param)
|
||||
|
||||
# Check if param is supported in any format
|
||||
is_supported = (
|
||||
param in supported_google_genai_params or
|
||||
param_snake in supported_google_genai_params or
|
||||
param_camel in supported_google_genai_params
|
||||
)
|
||||
|
||||
if is_supported:
|
||||
# Always output in camelCase for Google GenAI API
|
||||
output_key = param_camel if param != param_camel else param
|
||||
_generate_content_config_dict[output_key] = value
|
||||
return _generate_content_config_dict
|
||||
|
||||
def validate_environment(
|
||||
|
|
|
|||
|
|
@ -228,12 +228,13 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
|
||||
Gemini 3 models include:
|
||||
- gemini-3-pro-preview
|
||||
- gemini-3-flash
|
||||
- gemini-3-flash-preview (Gemini 3 Flash)
|
||||
- Any future Gemini 3.x models
|
||||
"""
|
||||
# Check for Gemini 3 models
|
||||
if "gemini-3" in model:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _supports_penalty_parameters(self, model: str) -> bool:
|
||||
|
|
@ -685,22 +686,40 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
Returns:
|
||||
GeminiThinkingConfig with thinkingLevel and includeThoughts
|
||||
"""
|
||||
# Check if this is gemini-3-flash which supports MINIMAL thinking level
|
||||
is_gemini3flash= model and (
|
||||
"gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
|
||||
)
|
||||
if reasoning_effort == "minimal":
|
||||
return {"thinkingLevel": "low", "includeThoughts": True}
|
||||
if is_gemini3flash:
|
||||
return {"thinkingLevel": "minimal", "includeThoughts": True}
|
||||
else:
|
||||
return {"thinkingLevel": "low", "includeThoughts": True}
|
||||
elif reasoning_effort == "low":
|
||||
return {"thinkingLevel": "low", "includeThoughts": True}
|
||||
elif reasoning_effort == "medium":
|
||||
return {
|
||||
"thinkingLevel": "high",
|
||||
"includeThoughts": True,
|
||||
} # medium is not out yet
|
||||
# For gemini-3-flash-preview, medium maps to "medium", otherwise "high"
|
||||
if is_gemini3flash:
|
||||
return {"thinkingLevel": "medium", "includeThoughts": True}
|
||||
else:
|
||||
return {
|
||||
"thinkingLevel": "high",
|
||||
"includeThoughts": True,
|
||||
} # medium is not out yet for other models
|
||||
elif reasoning_effort == "high":
|
||||
return {"thinkingLevel": "high", "includeThoughts": True}
|
||||
elif reasoning_effort == "disable":
|
||||
# Gemini 3 cannot fully disable thinking, so we use "low" but hide thoughts
|
||||
return {"thinkingLevel": "low", "includeThoughts": False}
|
||||
# Gemini 3 cannot fully disable thinking, so we use "minimal" for gemini-3-flash-preview, "low" for others
|
||||
if is_gemini3flash:
|
||||
return {"thinkingLevel": "minimal", "includeThoughts": False}
|
||||
else:
|
||||
return {"thinkingLevel": "low", "includeThoughts": False}
|
||||
elif reasoning_effort == "none":
|
||||
return {"thinkingLevel": "low", "includeThoughts": False}
|
||||
# For gemini-3-flash-preview, use "minimal" instead of "low"
|
||||
if is_gemini3flash:
|
||||
return {"thinkingLevel": "minimal", "includeThoughts": False}
|
||||
else:
|
||||
return {"thinkingLevel": "low", "includeThoughts": False}
|
||||
else:
|
||||
raise ValueError(f"Invalid reasoning effort: {reasoning_effort}")
|
||||
|
||||
|
|
@ -751,17 +770,38 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
@staticmethod
|
||||
def _map_thinking_param(
|
||||
thinking_param: AnthropicThinkingParam,
|
||||
model: Optional[str] = None,
|
||||
) -> GeminiThinkingConfig:
|
||||
thinking_enabled = thinking_param.get("type") == "enabled"
|
||||
thinking_budget = thinking_param.get("budget_tokens")
|
||||
|
||||
params: GeminiThinkingConfig = {}
|
||||
if thinking_enabled and not VertexGeminiConfig._is_thinking_budget_zero(
|
||||
thinking_budget
|
||||
):
|
||||
params["includeThoughts"] = True
|
||||
if thinking_budget is not None and isinstance(thinking_budget, int):
|
||||
params["thinkingBudget"] = thinking_budget
|
||||
|
||||
# For Gemini 3+ models, use thinkingLevel instead of thinkingBudget
|
||||
if model and VertexGeminiConfig._is_gemini_3_or_newer(model):
|
||||
if thinking_enabled:
|
||||
if thinking_budget is None or thinking_budget == 0:
|
||||
params["includeThoughts"] = False
|
||||
else:
|
||||
params["includeThoughts"] = True
|
||||
if thinking_budget >= 10000:
|
||||
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
|
||||
params["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
|
||||
else:
|
||||
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
|
||||
params["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
|
||||
else:
|
||||
# Thinking disabled
|
||||
params["includeThoughts"] = False
|
||||
else:
|
||||
# For older Gemini models, use thinkingBudget
|
||||
if thinking_enabled and not VertexGeminiConfig._is_thinking_budget_zero(
|
||||
thinking_budget
|
||||
):
|
||||
params["includeThoughts"] = True
|
||||
if thinking_budget is not None and isinstance(thinking_budget, int):
|
||||
params["thinkingBudget"] = thinking_budget
|
||||
|
||||
return params
|
||||
|
||||
def map_response_modalities(self, value: list) -> list:
|
||||
|
|
@ -938,7 +978,8 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
optional_params[
|
||||
"thinkingConfig"
|
||||
] = VertexGeminiConfig._map_thinking_param(
|
||||
cast(AnthropicThinkingParam, value)
|
||||
cast(AnthropicThinkingParam, value),
|
||||
model=model,
|
||||
)
|
||||
elif param == "modalities" and isinstance(value, list):
|
||||
response_modalities = self.map_response_modalities(value)
|
||||
|
|
@ -970,7 +1011,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
"thinkingLevel" not in thinking_config
|
||||
and "thinkingBudget" not in thinking_config
|
||||
):
|
||||
thinking_config["thinkingLevel"] = "low"
|
||||
# For gemini-3-flash-preview, default to "minimal" to match Gemini 2.5 Flash behavior
|
||||
# For other Gemini 3 models, default to "low"
|
||||
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
|
||||
thinking_config["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
|
||||
optional_params["thinkingConfig"] = thinking_config
|
||||
|
||||
return optional_params
|
||||
|
|
|
|||
|
|
@ -14732,6 +14732,98 @@
|
|||
"supports_web_search": true,
|
||||
"tpm": 800000
|
||||
},
|
||||
"gemini/gemini-3-flash-preview": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
"max_audio_length_hours": 8.4,
|
||||
"max_audio_per_prompt": 1,
|
||||
"max_images_per_prompt": 3000,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
"max_pdf_size_mb": 30,
|
||||
"max_tokens": 65535,
|
||||
"max_video_length": 1,
|
||||
"max_videos_per_prompt": 10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_reasoning_token": 3e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"rpm": 2000,
|
||||
"source": "https://ai.google.dev/pricing/gemini-3",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/completions",
|
||||
"/v1/batch"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_output": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_url_context": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"tpm": 800000
|
||||
},
|
||||
"gemini-3-flash-preview": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
"max_audio_length_hours": 8.4,
|
||||
"max_audio_per_prompt": 1,
|
||||
"max_images_per_prompt": 3000,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
"max_pdf_size_mb": 30,
|
||||
"max_tokens": 65535,
|
||||
"max_video_length": 1,
|
||||
"max_videos_per_prompt": 10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_reasoning_token": 3e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"source": "https://ai.google.dev/pricing/gemini-3",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/completions",
|
||||
"/v1/batch"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_output": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_url_context": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini/gemini-2.5-pro-exp-03-25": {
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_token": 0.0,
|
||||
|
|
|
|||
|
|
@ -90,6 +90,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
"metadata",
|
||||
"parallel_tool_calls",
|
||||
"previous_response_id",
|
||||
"reasoning",
|
||||
"stream",
|
||||
"temperature",
|
||||
"text",
|
||||
|
|
@ -178,6 +179,17 @@ class LiteLLMCompletionResponsesConfig:
|
|||
text_param
|
||||
)
|
||||
|
||||
# Extract reasoning_effort from reasoning parameter
|
||||
reasoning_effort = None
|
||||
reasoning_param = responses_api_request.get("reasoning")
|
||||
if reasoning_param:
|
||||
if isinstance(reasoning_param, dict):
|
||||
# reasoning can be {"effort": "low|medium|high"}
|
||||
reasoning_effort = reasoning_param.get("effort")
|
||||
elif isinstance(reasoning_param, str):
|
||||
# reasoning could be a string directly
|
||||
reasoning_effort = reasoning_param
|
||||
|
||||
litellm_completion_request: dict = {
|
||||
"messages": LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
|
||||
input=input,
|
||||
|
|
@ -198,6 +210,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
"service_tier": kwargs.get("service_tier"),
|
||||
"web_search_options": web_search_options,
|
||||
"response_format": response_format,
|
||||
"reasoning_effort": reasoning_effort,
|
||||
# litellm specific params
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
"extra_headers": extra_headers,
|
||||
|
|
@ -219,7 +232,6 @@ class LiteLLMCompletionResponsesConfig:
|
|||
litellm_completion_request = {
|
||||
k: v for k, v in litellm_completion_request.items() if v is not None
|
||||
}
|
||||
|
||||
return litellm_completion_request
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -169,7 +169,7 @@ class SafetSettingsConfig(TypedDict, total=False):
|
|||
class GeminiThinkingConfig(TypedDict, total=False):
|
||||
includeThoughts: bool
|
||||
thinkingBudget: int
|
||||
thinkingLevel: Literal["low", "medium", "high"]
|
||||
thinkingLevel: Literal["minimal", "low", "medium", "high"]
|
||||
|
||||
|
||||
GeminiResponseModalities = Literal["TEXT", "IMAGE", "AUDIO", "VIDEO"]
|
||||
|
|
|
|||
|
|
@ -14732,6 +14732,98 @@
|
|||
"supports_web_search": true,
|
||||
"tpm": 800000
|
||||
},
|
||||
"gemini/gemini-3-flash-preview": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
"max_audio_length_hours": 8.4,
|
||||
"max_audio_per_prompt": 1,
|
||||
"max_images_per_prompt": 3000,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
"max_pdf_size_mb": 30,
|
||||
"max_tokens": 65535,
|
||||
"max_video_length": 1,
|
||||
"max_videos_per_prompt": 10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_reasoning_token": 3e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"rpm": 2000,
|
||||
"source": "https://ai.google.dev/pricing/gemini-3",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/completions",
|
||||
"/v1/batch"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_output": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_url_context": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"tpm": 800000
|
||||
},
|
||||
"gemini-3-flash-preview": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"input_cost_per_audio_token": 1e-06,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
"max_audio_length_hours": 8.4,
|
||||
"max_audio_per_prompt": 1,
|
||||
"max_images_per_prompt": 3000,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
"max_pdf_size_mb": 30,
|
||||
"max_tokens": 65535,
|
||||
"max_video_length": 1,
|
||||
"max_videos_per_prompt": 10,
|
||||
"mode": "chat",
|
||||
"output_cost_per_reasoning_token": 3e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"source": "https://ai.google.dev/pricing/gemini-3",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/completions",
|
||||
"/v1/batch"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"video"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_output": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_url_context": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"gemini/gemini-2.5-pro-exp-03-25": {
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"input_cost_per_token": 0.0,
|
||||
|
|
|
|||
|
|
@ -228,4 +228,4 @@ general_settings:
|
|||
# settings for using redis caching
|
||||
# REDIS_HOST: redis-16337.c322.us-east-1-2.ec2.cloud.redislabs.com
|
||||
# REDIS_PORT: "16337"
|
||||
# REDIS_PASSWORD:
|
||||
# REDIS_PASSWORD:
|
||||
|
|
@ -1229,3 +1229,175 @@ def test_gemini_function_args_preserve_unicode():
|
|||
assert parsed_args["recipient"] == "José"
|
||||
assert "\\u" not in arguments_str
|
||||
assert "José" in arguments_str
|
||||
|
||||
|
||||
def test_anthropic_thinking_param_to_gemini_3_thinkingLevel():
|
||||
"""
|
||||
Test that Anthropic thinking parameters are correctly transformed to Gemini 3 thinkingLevel
|
||||
instead of thinkingBudget.
|
||||
|
||||
For Gemini 3+ models (gemini-3-flash, gemini-3-pro, gemini-3-flash-preview):
|
||||
- Should use thinkingLevel instead of thinkingBudget
|
||||
- budget_tokens should map to thinkingLevel
|
||||
|
||||
Related issue: https://github.com/BerriAI/litellm/issues/XXXX
|
||||
"""
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
VertexGeminiConfig,
|
||||
)
|
||||
from litellm.types.llms.anthropic import AnthropicThinkingParam
|
||||
|
||||
# Test 1: Anthropic thinking enabled with budget_tokens for Gemini 3 model
|
||||
thinking_param: AnthropicThinkingParam = {
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
}
|
||||
|
||||
result = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param,
|
||||
model="gemini-3-flash",
|
||||
)
|
||||
|
||||
# For Gemini 3, should use thinkingLevel, not thinkingBudget
|
||||
assert "thinkingLevel" in result, "Should have thinkingLevel for Gemini 3"
|
||||
assert "thinkingBudget" not in result, "Should NOT have thinkingBudget for Gemini 3"
|
||||
assert result["includeThoughts"] is True
|
||||
assert result["thinkingLevel"] in ["minimal", "low"], "thinkingLevel should be 'minimal' or 'low'"
|
||||
|
||||
# Test 2: Anthropic thinking disabled for Gemini 3
|
||||
thinking_param_disabled: AnthropicThinkingParam = {
|
||||
"type": "disabled",
|
||||
"budget_tokens": None,
|
||||
}
|
||||
|
||||
result_disabled = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param_disabled,
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert result_disabled.get("includeThoughts") is False
|
||||
assert "thinkingLevel" not in result_disabled or result_disabled.get("thinkingLevel") is None
|
||||
|
||||
# Test 3: Budget tokens = 0 for Gemini 3
|
||||
thinking_param_zero: AnthropicThinkingParam = {
|
||||
"type": "enabled",
|
||||
"budget_tokens": 0,
|
||||
}
|
||||
|
||||
result_zero = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param_zero,
|
||||
model="gemini-3-flash",
|
||||
)
|
||||
|
||||
assert result_zero["includeThoughts"] is False
|
||||
assert "thinkingLevel" not in result_zero or result_zero.get("thinkingLevel") is None
|
||||
|
||||
# Test 4: Fiercefalcon model (Gemini 3 Flash checkpoint) should use thinkingLevel
|
||||
result_gemini3flashpreview = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param,
|
||||
model="gemini-3-flash-preview",
|
||||
)
|
||||
|
||||
assert "thinkingLevel" in result_gemini3flashpreview, "Should have thinkingLevel for gemini-3-flash-preview"
|
||||
assert "thinkingBudget" not in result_gemini3flashpreview, "Should NOT have thinkingBudget for gemini-3-flash-preview"
|
||||
assert result_gemini3flashpreview["includeThoughts"] is True
|
||||
|
||||
|
||||
def test_anthropic_thinking_param_to_gemini_2_thinkingBudget():
|
||||
"""
|
||||
Test that Anthropic thinking parameters are correctly transformed to Gemini 2 thinkingBudget
|
||||
(not thinkingLevel).
|
||||
|
||||
For Gemini 2.x models (gemini-2.5-flash, gemini-2.0-flash):
|
||||
- Should continue using thinkingBudget
|
||||
- thinkingLevel should NOT be used
|
||||
|
||||
Related issue: https://github.com/BerriAI/litellm/issues/XXXX
|
||||
"""
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
VertexGeminiConfig,
|
||||
)
|
||||
from litellm.types.llms.anthropic import AnthropicThinkingParam
|
||||
|
||||
# Test 1: Anthropic thinking enabled with budget_tokens for Gemini 2 model
|
||||
thinking_param: AnthropicThinkingParam = {
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
}
|
||||
|
||||
result = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param,
|
||||
model="gemini-2.5-flash",
|
||||
)
|
||||
|
||||
# For Gemini 2, should use thinkingBudget, not thinkingLevel
|
||||
assert "thinkingBudget" in result, "Should have thinkingBudget for Gemini 2"
|
||||
assert "thinkingLevel" not in result, "Should NOT have thinkingLevel for Gemini 2"
|
||||
assert result["includeThoughts"] is True
|
||||
assert result["thinkingBudget"] == 10000
|
||||
|
||||
# Test 2: Anthropic thinking enabled for gemini-2.0-flash model
|
||||
result_gemini2 = VertexGeminiConfig._map_thinking_param(
|
||||
thinking_param=thinking_param,
|
||||
model="gemini-2.0-flash-thinking-exp-01-21",
|
||||
)
|
||||
|
||||
assert "thinkingBudget" in result_gemini2, "Should have thinkingBudget for Gemini 2"
|
||||
assert "thinkingLevel" not in result_gemini2, "Should NOT have thinkingLevel for Gemini 2"
|
||||
assert result_gemini2["includeThoughts"] is True
|
||||
assert result_gemini2["thinkingBudget"] == 10000
|
||||
|
||||
|
||||
def test_anthropic_thinking_param_via_map_openai_params():
|
||||
"""
|
||||
Test that the thinking parameter is correctly transformed through the full map_openai_params flow
|
||||
for Gemini 3 models, resulting in thinkingConfig with thinkingLevel.
|
||||
|
||||
This tests the full integration from Anthropic API format to Gemini format.
|
||||
"""
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
VertexGeminiConfig,
|
||||
)
|
||||
from litellm.types.llms.anthropic import AnthropicThinkingParam
|
||||
|
||||
config = VertexGeminiConfig()
|
||||
|
||||
# Test with Gemini 3 model
|
||||
non_default_params = {
|
||||
"thinking": {
|
||||
"type": "enabled",
|
||||
"budget_tokens": 10000,
|
||||
}
|
||||
}
|
||||
optional_params: dict = {}
|
||||
|
||||
result = config.map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model="gemini-3-flash",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
# Check that thinkingConfig was created with thinkingLevel
|
||||
assert "thinkingConfig" in result, "Should have thinkingConfig in optional_params"
|
||||
thinking_config = result["thinkingConfig"]
|
||||
assert "thinkingLevel" in thinking_config, "Should have thinkingLevel for Gemini 3"
|
||||
assert "thinkingBudget" not in thinking_config, "Should NOT have thinkingBudget for Gemini 3"
|
||||
assert thinking_config["includeThoughts"] is True
|
||||
|
||||
# Test with Gemini 2 model
|
||||
optional_params_2 = {}
|
||||
result_2 = config.map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params_2,
|
||||
model="gemini-2.5-flash",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
# Check that thinkingConfig was created with thinkingBudget
|
||||
assert "thinkingConfig" in result_2, "Should have thinkingConfig in optional_params"
|
||||
thinking_config_2 = result_2["thinkingConfig"]
|
||||
assert "thinkingBudget" in thinking_config_2, "Should have thinkingBudget for Gemini 2"
|
||||
assert "thinkingLevel" not in thinking_config_2, "Should NOT have thinkingLevel for Gemini 2"
|
||||
assert thinking_config_2["includeThoughts"] is True
|
||||
assert thinking_config_2["thinkingBudget"] == 10000
|
||||
|
|
|
|||
|
|
@ -0,0 +1,249 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test to verify the Google GenAI transformation logic for generateContent parameters
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.gemini.google_genai.transformation import GoogleGenAIConfig
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_response_json_schema_camelcase():
|
||||
"""Test that responseJsonSchema (camelCase) is passed through correctly"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"responseJsonSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"recipe_name": {"type": "string"}
|
||||
}
|
||||
},
|
||||
"temperature": 1.0
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# responseJsonSchema should be in the result (camelCase format for Google GenAI API)
|
||||
assert "responseJsonSchema" in result
|
||||
assert result["responseJsonSchema"] == generate_content_config_dict["responseJsonSchema"]
|
||||
assert "temperature" in result
|
||||
assert result["temperature"] == 1.0
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_response_schema_snakecase():
|
||||
"""Test that response_schema (snake_case) is converted to responseJsonSchema (camelCase)"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"response_json_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"recipe_name": {"type": "string"}
|
||||
}
|
||||
},
|
||||
"temperature": 1.0
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# response_schema should be converted to responseJsonSchema (camelCase)
|
||||
assert "responseJsonSchema" in result
|
||||
assert result["responseJsonSchema"] == generate_content_config_dict["response_json_schema"]
|
||||
assert "temperature" in result
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_thinking_config_camelcase():
|
||||
"""Test that thinkingConfig (camelCase) is passed through correctly"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"thinkingConfig": {
|
||||
"thinkingLevel": "minimal",
|
||||
"includeThoughts": True
|
||||
},
|
||||
"temperature": 1.0
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# thinkingConfig should be in the result (camelCase format for Google GenAI API)
|
||||
assert "thinkingConfig" in result
|
||||
assert result["thinkingConfig"]["thinkingLevel"] == "minimal"
|
||||
assert result["thinkingConfig"]["includeThoughts"] is True
|
||||
assert "temperature" in result
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_thinking_config_snakecase():
|
||||
"""Test that thinking_config (snake_case) is converted to thinkingConfig (camelCase)"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"thinking_config": {
|
||||
"thinkingLevel": "medium",
|
||||
"includeThoughts": True
|
||||
},
|
||||
"temperature": 1.0
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# thinking_config should be converted to thinkingConfig (camelCase)
|
||||
assert "thinkingConfig" in result
|
||||
assert result["thinkingConfig"]["thinkingLevel"] == "medium"
|
||||
assert result["thinkingConfig"]["includeThoughts"] is True
|
||||
assert "thinking_config" not in result # Should not be in snake_case format
|
||||
assert "temperature" in result
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_mixed_formats():
|
||||
"""Test that both camelCase and snake_case parameters work together"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"responseJsonSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"recipe_name": {"type": "string"}
|
||||
}
|
||||
},
|
||||
"thinking_config": {
|
||||
"thinkingLevel": "low",
|
||||
"includeThoughts": True
|
||||
},
|
||||
"temperature": 1.0,
|
||||
"max_output_tokens": 100
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# All parameters should be converted to camelCase
|
||||
assert "responseJsonSchema" in result
|
||||
assert "thinkingConfig" in result
|
||||
assert result["thinkingConfig"]["thinkingLevel"] == "low"
|
||||
assert "temperature" in result
|
||||
assert "maxOutputTokens" in result # This one stays as-is if it's in supported list
|
||||
|
||||
|
||||
def test_map_generate_content_optional_params_response_mime_type():
|
||||
"""Test that responseMimeType is handled correctly"""
|
||||
config = GoogleGenAIConfig()
|
||||
|
||||
generate_content_config_dict = {
|
||||
"responseMimeType": "application/json",
|
||||
"responseJsonSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"recipe_name": {"type": "string"}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = config.map_generate_content_optional_params(
|
||||
generate_content_config_dict=generate_content_config_dict,
|
||||
model="gemini/gemini-3-flash-preview"
|
||||
)
|
||||
|
||||
# responseMimeType should be passed through (it's already camelCase)
|
||||
assert "responseMimeType" in result or "response_mime_type" in result
|
||||
assert "responseJsonSchema" in result
|
||||
|
||||
|
||||
def test_responses_api_reasoning_dict_format():
|
||||
"""Test that reasoning parameter with dict format is mapped to reasoning_effort"""
|
||||
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
|
||||
|
||||
responses_api_request: ResponsesAPIOptionalRequestParams = {
|
||||
"reasoning": {"effort": "high"},
|
||||
"temperature": 1.0,
|
||||
}
|
||||
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gemini/2.5-pro",
|
||||
input="Hello, what is the capital of France?",
|
||||
responses_api_request=responses_api_request,
|
||||
)
|
||||
|
||||
# reasoning_effort should be extracted from reasoning dict
|
||||
assert "reasoning_effort" in result
|
||||
assert result["reasoning_effort"] == "high"
|
||||
|
||||
|
||||
def test_responses_api_reasoning_string_format():
|
||||
"""Test that reasoning parameter with string format is mapped to reasoning_effort"""
|
||||
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
|
||||
|
||||
responses_api_request: ResponsesAPIOptionalRequestParams = {
|
||||
"reasoning": "medium", # Could be a string directly
|
||||
"temperature": 1.0,
|
||||
}
|
||||
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gemini/2.5-pro",
|
||||
input="Hello, what is the capital of France?",
|
||||
responses_api_request=responses_api_request,
|
||||
)
|
||||
|
||||
# reasoning_effort should be extracted from reasoning string
|
||||
assert "reasoning_effort" in result
|
||||
assert result["reasoning_effort"] == "medium"
|
||||
|
||||
|
||||
def test_responses_api_reasoning_low_effort():
|
||||
"""Test that low reasoning effort is correctly mapped"""
|
||||
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
|
||||
|
||||
responses_api_request: ResponsesAPIOptionalRequestParams = {
|
||||
"reasoning": {"effort": "low"},
|
||||
}
|
||||
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gemini/2.5-pro",
|
||||
input="Test",
|
||||
responses_api_request=responses_api_request,
|
||||
)
|
||||
|
||||
assert "reasoning_effort" in result
|
||||
assert result["reasoning_effort"] == "low"
|
||||
|
||||
|
||||
def test_responses_api_no_reasoning():
|
||||
"""Test that no reasoning_effort is included when reasoning is not provided"""
|
||||
from litellm.types.llms.openai import ResponsesAPIOptionalRequestParams
|
||||
|
||||
responses_api_request: ResponsesAPIOptionalRequestParams = {
|
||||
"temperature": 1.0,
|
||||
}
|
||||
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gemini/2.5-pro",
|
||||
input="Test",
|
||||
responses_api_request=responses_api_request,
|
||||
)
|
||||
|
||||
# reasoning_effort should not be in result if not provided (filtered out as None)
|
||||
assert "reasoning_effort" not in result or result.get("reasoning_effort") is None
|
||||
Loading…
Add table
Reference in a new issue