Fix PR review issues: gpt-4-0314 prompt caching, case-insensitive data URL check, test I/O mocking

- Remove incorrect supports_prompt_caching from gpt-4-0314 (predates the feature)
- Make data-URL detection case-insensitive in Gemini tool call result conversion
- Mock show_banner/generate_feedback_box in max_budget tests to prevent real I/O

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Krrish Dholakia 2026-03-21 11:30:29 -07:00
parent a5b7e49713
commit 509d2e9ac3
4 changed files with 54 additions and 22 deletions

View file

@ -1506,7 +1506,7 @@ def convert_to_gemini_tool_call_result( # noqa: PLR0915
# Detect data-URL images (e.g. from Anthropic tool_result with a single image block
# that was serialised as a plain string by translate_anthropic_messages_to_openai)
# and promote them to inline_data so Gemini receives actual image bytes.
if content_str.startswith("data:") and ";base64," in content_str:
if content_str[:5].lower() == "data:" and ";base64," in content_str:
try:
mime_rest = content_str[5:].split(";base64,", 1)
if len(mime_rest) == 2 and mime_rest[0].startswith("image/"):

View file

@ -6152,7 +6152,8 @@
"max_query_tokens": 4096,
"max_tokens": 32768,
"mode": "rerank",
"output_cost_per_token": 0.0
"output_cost_per_token": 0.0,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076"
},
"azure_ai/cohere-rerank-v4.0-fast": {
"input_cost_per_query": 0.002,
@ -6163,7 +6164,8 @@
"max_query_tokens": 4096,
"max_tokens": 32768,
"mode": "rerank",
"output_cost_per_token": 0.0
"output_cost_per_token": 0.0,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076"
},
"azure_ai/deepseek-v3.2": {
"input_cost_per_token": 5.8e-07,
@ -6173,6 +6175,7 @@
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.68e-06,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
@ -6187,6 +6190,7 @@
"max_tokens": 163840,
"mode": "chat",
"output_cost_per_token": 1.68e-06,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549",
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
@ -16936,6 +16940,18 @@
"supports_system_messages": true,
"supports_tool_choice": true
},
"gpt-4-0314": {
"deprecation_date": "2026-03-26",
"input_cost_per_token": 3e-05,
"litellm_provider": "openai",
"max_input_tokens": 8192,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 6e-05,
"supports_system_messages": true,
"supports_tool_choice": true
},
"gpt-4-0613": {
"deprecation_date": "2025-06-06",
"input_cost_per_token": 3e-05,
@ -30506,7 +30522,7 @@
"output_cost_per_token": 5.4e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supported_regions": [
"us-west2"
"us-central1"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
@ -30526,7 +30542,7 @@
"output_cost_per_token_batches": 8.4e-07,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supported_regions": [
"us-west2"
"global"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
@ -30543,6 +30559,9 @@
"mode": "chat",
"output_cost_per_token": 5.4e-06,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
"supported_regions": [
"us-central1"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
@ -31167,7 +31186,10 @@
"input_cost_per_token": 3e-07,
"output_cost_per_token": 1.2e-06,
"ocr_cost_per_page": 0.0003,
"source": "https://cloud.google.com/vertex-ai/pricing"
"source": "https://cloud.google.com/vertex-ai/pricing",
"supported_regions": [
"us-central1"
]
},
"vertex_ai/openai/gpt-oss-120b-maas": {
"input_cost_per_token": 1.5e-07,

View file

@ -16949,7 +16949,6 @@
"max_tokens": 4096,
"mode": "chat",
"output_cost_per_token": 6e-05,
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_tool_choice": true
},

View file

@ -4,10 +4,11 @@ converted to float.
GitHub Issue: #23843
"""
from unittest.mock import patch
import pytest
import litellm
from litellm.proxy.proxy_server import initialize
@pytest.mark.asyncio
@ -17,22 +18,32 @@ async def test_max_budget_string_converted_to_float():
string. initialize() should convert it to float so the comparison
`litellm.max_budget > 0` doesn't raise TypeError.
"""
original = litellm.max_budget
try:
await initialize(max_budget="100.5")
assert isinstance(litellm.max_budget, float)
assert litellm.max_budget == 100.5
finally:
litellm.max_budget = original
with patch("litellm.proxy.common_utils.banner.show_banner"), patch(
"litellm.proxy.proxy_server.generate_feedback_box"
):
from litellm.proxy.proxy_server import initialize
original = litellm.max_budget
try:
await initialize(max_budget="100.5")
assert isinstance(litellm.max_budget, float)
assert litellm.max_budget == 100.5
finally:
litellm.max_budget = original
@pytest.mark.asyncio
async def test_max_budget_float_stays_float():
"""max_budget as float should still work."""
original = litellm.max_budget
try:
await initialize(max_budget=200.0)
assert isinstance(litellm.max_budget, float)
assert litellm.max_budget == 200.0
finally:
litellm.max_budget = original
with patch("litellm.proxy.common_utils.banner.show_banner"), patch(
"litellm.proxy.proxy_server.generate_feedback_box"
):
from litellm.proxy.proxy_server import initialize
original = litellm.max_budget
try:
await initialize(max_budget=200.0)
assert isinstance(litellm.max_budget, float)
assert litellm.max_budget == 200.0
finally:
litellm.max_budget = original