Rename: gemini-3-flash-preview

This commit is contained in:
Sameer Kankute 2025-12-17 21:48:15 +05:30
parent e7b8d2290f
commit e3cf0110bb
6 changed files with 67 additions and 67 deletions

View file

@ -1,7 +1,7 @@
---
slug: gemini_3_flash
title: "DAY 0 Support: Gemini 3 Flash on LiteLLM"
date: 2025-11-19T10:00:00
date: 2025-12-17T10:00:00
authors:
- name: Sameer Kankute
title: SWE @ LiteLLM (LLM Translation)
@ -25,7 +25,7 @@ import TabItem from '@theme/TabItem';
# Gemini 3 Flash Day 0 Support
LiteLLM now supports `gemini-3-flash` and all the new API changes along with it.
LiteLLM now supports `gemini-3-flash-preview` and all the new API changes along with it.
## What's New
@ -73,7 +73,7 @@ from litellm import completion
# No need to make any changes to your code as we map openai reasoning param to thinkingLevel
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "Solve this complex math problem: 25 * 4 + 10"}],
reasoning_effort="medium", # NEW: MEDIUM thinking level
)
@ -91,7 +91,7 @@ print(response.choices[0].message.content)
model_list:
- model_name: gemini-3-flash
litellm_params:
model: gemini/gemini-3-flash
model: gemini/gemini-3-flash-preview
api_key: os.environ/GEMINI_API_KEY
```
@ -104,15 +104,15 @@ litellm --config /path/to/config.yaml
**3. Call with MEDIUM thinking**
```bash
curl http://localhost:4000/v1/chat/completions \
curl -X POST http://localhost:4000/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
-d '{
"model": "gemini-3-flash-reasoning",
"model": "gemini-3-flash",
"messages": [{"role": "user", "content": "Complex reasoning task"}],
"reasoning_effort": "medium"
}'
```
``'
</TabItem>
</Tabs>
@ -130,7 +130,7 @@ curl http://localhost:4000/v1/chat/completions \
from litellm import completion
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "What's 2+2?"}],
reasoning_effort="minimal",
)
@ -144,7 +144,7 @@ response = completion(
```python
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "Write a haiku about coding"}],
reasoning_effort="low",
)
@ -158,7 +158,7 @@ response = completion(
```python
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "Analyze this dataset and find patterns"}],
reasoning_effort="medium", # NEW!
)
@ -172,7 +172,7 @@ response = completion(
```python
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "Prove this mathematical theorem"}],
reasoning_effort="high",
)
@ -203,7 +203,7 @@ import litellm
from litellm import completion
response = completion(
model="gemini-3-flash",
model="gemini/gemini-3-flash-preview",
messages=[{"role": "user", "content": "Your question here"}],
reasoning_effort="medium", # Use MEDIUM thinking
)

View file

@ -229,15 +229,15 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
Gemini 3 models include:
- gemini-3-pro-preview
- gemini-3-flash
- fiercefalcon (Gemini 3 Flash checkpoint)
- gemini-3-flash-preview (Gemini 3 Flash)
- Any future Gemini 3.x models
"""
# Check for Gemini 3 models
if "gemini-3" in model:
return True
# Check for fiercefalcon (Gemini 3 Flash checkpoint) # TODO: remove
if "fiercefalcon" in model.lower(): # TODO : Remove this once we have the official name of the model
# Check for gemini-3-flash-preview
if "gemini-3-flash-preview" in model.lower():
return True
return False
@ -692,19 +692,19 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
GeminiThinkingConfig with thinkingLevel and includeThoughts
"""
# Check if this is gemini-3-flash which supports MINIMAL thinking level
is_fiercefalcon= model and (
"fiercefalcon" in model.lower() or "gemini-3-flash" in model.lower()
is_gemini3flash= model and (
"gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
)
if reasoning_effort == "minimal":
if is_fiercefalcon:
if is_gemini3flash:
return {"thinkingLevel": "minimal", "includeThoughts": True}
else:
return {"thinkingLevel": "low", "includeThoughts": True}
elif reasoning_effort == "low":
return {"thinkingLevel": "low", "includeThoughts": True}
elif reasoning_effort == "medium":
# For fiercefalcon, medium maps to "medium", otherwise "high"
if is_fiercefalcon:
# For gemini-3-flash-preview, medium maps to "medium", otherwise "high"
if is_gemini3flash:
return {"thinkingLevel": "medium", "includeThoughts": True}
else:
return {
@ -714,14 +714,14 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
elif reasoning_effort == "high":
return {"thinkingLevel": "high", "includeThoughts": True}
elif reasoning_effort == "disable":
# Gemini 3 cannot fully disable thinking, so we use "minimal" for fiercefalcon, "low" for others
if is_fiercefalcon:
# Gemini 3 cannot fully disable thinking, so we use "minimal" for gemini-3-flash-preview, "low" for others
if is_gemini3flash:
return {"thinkingLevel": "minimal", "includeThoughts": False}
else:
return {"thinkingLevel": "low", "includeThoughts": False}
elif reasoning_effort == "none":
# For fiercefalcon, use "minimal" instead of "low"
if is_fiercefalcon:
# For gemini-3-flash-preview, use "minimal" instead of "low"
if is_gemini3flash:
return {"thinkingLevel": "minimal", "includeThoughts": False}
else:
return {"thinkingLevel": "low", "includeThoughts": False}
@ -790,11 +790,11 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
else:
params["includeThoughts"] = True
if thinking_budget >= 10000:
is_fiercefalcon = "fiercefalcon" in model.lower() or "gemini-3-flash" in model.lower()
params["thinkingLevel"] = "minimal" if is_fiercefalcon else "low"
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
params["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
else:
is_fiercefalcon = "fiercefalcon" in model.lower() or "gemini-3-flash" in model.lower()
params["thinkingLevel"] = "minimal" if is_fiercefalcon else "low"
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
params["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
else:
# Thinking disabled
params["includeThoughts"] = False
@ -1016,10 +1016,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
"thinkingLevel" not in thinking_config
and "thinkingBudget" not in thinking_config
):
# For fiercefalcon, default to "minimal" to match Gemini 2.5 Flash behavior
# For gemini-3-flash-preview, default to "minimal" to match Gemini 2.5 Flash behavior
# For other Gemini 3 models, default to "low"
is_fiercefalcon = "fiercefalcon" in model.lower() or "gemini-3-flash" in model.lower()
thinking_config["thinkingLevel"] = "minimal" if is_fiercefalcon else "low"
is_gemini3flash = "gemini-3-flash-preview" in model.lower() or "gemini-3-flash" in model.lower()
thinking_config["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
optional_params["thinkingConfig"] = thinking_config
return optional_params

View file

@ -14732,10 +14732,10 @@
"supports_web_search": true,
"tpm": 800000
},
"gemini/fiercefalcon": {
"cache_read_input_token_cost": 3e-08,
"gemini/gemini-3-flash-preview": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_audio_token": 1e-06,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 5e-07,
"litellm_provider": "gemini",
"max_audio_length_hours": 8.4,
"max_audio_per_prompt": 1,
@ -14747,10 +14747,10 @@
"max_video_length": 1,
"max_videos_per_prompt": 10,
"mode": "chat",
"output_cost_per_reasoning_token": 2.5e-06,
"output_cost_per_token": 2.5e-06,
"output_cost_per_reasoning_token": 3e-06,
"output_cost_per_token": 3e-06,
"rpm": 2000,
"source": "https://ai.google.dev/gemini-api/docs/models",
"source": "https://ai.google.dev/pricing/gemini-3",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/completions",
@ -14779,10 +14779,10 @@
"supports_web_search": true,
"tpm": 800000
},
"fiercefalcon": {
"cache_read_input_token_cost": 3e-08,
"gemini-3-flash-preview": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_audio_token": 1e-06,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
"max_audio_length_hours": 8.4,
"max_audio_per_prompt": 1,
@ -14794,9 +14794,9 @@
"max_video_length": 1,
"max_videos_per_prompt": 10,
"mode": "chat",
"output_cost_per_reasoning_token": 2.5e-06,
"output_cost_per_token": 2.5e-06,
"source": "https://ai.google.dev/gemini-api/docs/models",
"output_cost_per_reasoning_token": 3e-06,
"output_cost_per_token": 3e-06,
"source": "https://ai.google.dev/pricing/gemini-3",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/completions",

View file

@ -14732,10 +14732,10 @@
"supports_web_search": true,
"tpm": 800000
},
"gemini/fiercefalcon": {
"cache_read_input_token_cost": 3e-08,
"gemini/gemini-3-flash-preview": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_audio_token": 1e-06,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 5e-07,
"litellm_provider": "gemini",
"max_audio_length_hours": 8.4,
"max_audio_per_prompt": 1,
@ -14747,10 +14747,10 @@
"max_video_length": 1,
"max_videos_per_prompt": 10,
"mode": "chat",
"output_cost_per_reasoning_token": 2.5e-06,
"output_cost_per_token": 2.5e-06,
"output_cost_per_reasoning_token": 3e-06,
"output_cost_per_token": 3e-06,
"rpm": 2000,
"source": "https://ai.google.dev/gemini-api/docs/models",
"source": "https://ai.google.dev/pricing/gemini-3",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/completions",
@ -14779,10 +14779,10 @@
"supports_web_search": true,
"tpm": 800000
},
"fiercefalcon": {
"cache_read_input_token_cost": 3e-08,
"gemini-3-flash-preview": {
"cache_read_input_token_cost": 5e-08,
"input_cost_per_audio_token": 1e-06,
"input_cost_per_token": 3e-07,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
"max_audio_length_hours": 8.4,
"max_audio_per_prompt": 1,
@ -14794,9 +14794,9 @@
"max_video_length": 1,
"max_videos_per_prompt": 10,
"mode": "chat",
"output_cost_per_reasoning_token": 2.5e-06,
"output_cost_per_token": 2.5e-06,
"source": "https://ai.google.dev/gemini-api/docs/models",
"output_cost_per_reasoning_token": 3e-06,
"output_cost_per_token": 3e-06,
"source": "https://ai.google.dev/pricing/gemini-3",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/completions",

View file

@ -1236,7 +1236,7 @@ def test_anthropic_thinking_param_to_gemini_3_thinkingLevel():
Test that Anthropic thinking parameters are correctly transformed to Gemini 3 thinkingLevel
instead of thinkingBudget.
For Gemini 3+ models (gemini-3-flash, gemini-3-pro, fiercefalcon):
For Gemini 3+ models (gemini-3-flash, gemini-3-pro, gemini-3-flash-preview):
- Should use thinkingLevel instead of thinkingBudget
- budget_tokens should map to thinkingLevel
@ -1293,14 +1293,14 @@ def test_anthropic_thinking_param_to_gemini_3_thinkingLevel():
assert "thinkingLevel" not in result_zero or result_zero.get("thinkingLevel") is None
# Test 4: Fiercefalcon model (Gemini 3 Flash checkpoint) should use thinkingLevel
result_fiercefalcon = VertexGeminiConfig._map_thinking_param(
result_gemini3flashpreview = VertexGeminiConfig._map_thinking_param(
thinking_param=thinking_param,
model="fiercefalcon",
model="gemini-3-flash-preview",
)
assert "thinkingLevel" in result_fiercefalcon, "Should have thinkingLevel for fiercefalcon"
assert "thinkingBudget" not in result_fiercefalcon, "Should NOT have thinkingBudget for fiercefalcon"
assert result_fiercefalcon["includeThoughts"] is True
assert "thinkingLevel" in result_gemini3flashpreview, "Should have thinkingLevel for gemini-3-flash-preview"
assert "thinkingBudget" not in result_gemini3flashpreview, "Should NOT have thinkingBudget for gemini-3-flash-preview"
assert result_gemini3flashpreview["includeThoughts"] is True
def test_anthropic_thinking_param_to_gemini_2_thinkingBudget():

View file

@ -33,7 +33,7 @@ def test_map_generate_content_optional_params_response_json_schema_camelcase():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# responseJsonSchema should be in the result (camelCase format for Google GenAI API)
@ -59,7 +59,7 @@ def test_map_generate_content_optional_params_response_schema_snakecase():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# response_schema should be converted to responseJsonSchema (camelCase)
@ -82,7 +82,7 @@ def test_map_generate_content_optional_params_thinking_config_camelcase():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# thinkingConfig should be in the result (camelCase format for Google GenAI API)
@ -106,7 +106,7 @@ def test_map_generate_content_optional_params_thinking_config_snakecase():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# thinking_config should be converted to thinkingConfig (camelCase)
@ -138,7 +138,7 @@ def test_map_generate_content_optional_params_mixed_formats():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# All parameters should be converted to camelCase
@ -165,7 +165,7 @@ def test_map_generate_content_optional_params_response_mime_type():
result = config.map_generate_content_optional_params(
generate_content_config_dict=generate_content_config_dict,
model="gemini/fiercefalcon"
model="gemini/gemini-3-flash-preview"
)
# responseMimeType should be passed through (it's already camelCase)