mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Fix PR review issues: gpt-4-0314 prompt caching, case-insensitive data URL check, test I/O mocking
- Remove incorrect supports_prompt_caching from gpt-4-0314 (predates the feature) - Make data-URL detection case-insensitive in Gemini tool call result conversion - Mock show_banner/generate_feedback_box in max_budget tests to prevent real I/O Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
a5b7e49713
commit
509d2e9ac3
4 changed files with 54 additions and 22 deletions
|
|
@ -1506,7 +1506,7 @@ def convert_to_gemini_tool_call_result( # noqa: PLR0915
|
|||
# Detect data-URL images (e.g. from Anthropic tool_result with a single image block
|
||||
# that was serialised as a plain string by translate_anthropic_messages_to_openai)
|
||||
# and promote them to inline_data so Gemini receives actual image bytes.
|
||||
if content_str.startswith("data:") and ";base64," in content_str:
|
||||
if content_str[:5].lower() == "data:" and ";base64," in content_str:
|
||||
try:
|
||||
mime_rest = content_str[5:].split(";base64,", 1)
|
||||
if len(mime_rest) == 2 and mime_rest[0].startswith("image/"):
|
||||
|
|
|
|||
|
|
@ -6152,7 +6152,8 @@
|
|||
"max_query_tokens": 4096,
|
||||
"max_tokens": 32768,
|
||||
"mode": "rerank",
|
||||
"output_cost_per_token": 0.0
|
||||
"output_cost_per_token": 0.0,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076"
|
||||
},
|
||||
"azure_ai/cohere-rerank-v4.0-fast": {
|
||||
"input_cost_per_query": 0.002,
|
||||
|
|
@ -6163,7 +6164,8 @@
|
|||
"max_query_tokens": 4096,
|
||||
"max_tokens": 32768,
|
||||
"mode": "rerank",
|
||||
"output_cost_per_token": 0.0
|
||||
"output_cost_per_token": 0.0,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076"
|
||||
},
|
||||
"azure_ai/deepseek-v3.2": {
|
||||
"input_cost_per_token": 5.8e-07,
|
||||
|
|
@ -6173,6 +6175,7 @@
|
|||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.68e-06,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
|
|
@ -6187,6 +6190,7 @@
|
|||
"max_tokens": 163840,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.68e-06,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549",
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
|
|
@ -16936,6 +16940,18 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-4-0314": {
|
||||
"deprecation_date": "2026-03-26",
|
||||
"input_cost_per_token": 3e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-05,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-4-0613": {
|
||||
"deprecation_date": "2025-06-06",
|
||||
"input_cost_per_token": 3e-05,
|
||||
|
|
@ -30506,7 +30522,7 @@
|
|||
"output_cost_per_token": 5.4e-06,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
|
||||
"supported_regions": [
|
||||
"us-west2"
|
||||
"us-central1"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -30526,7 +30542,7 @@
|
|||
"output_cost_per_token_batches": 8.4e-07,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
|
||||
"supported_regions": [
|
||||
"us-west2"
|
||||
"global"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
|
|
@ -30543,6 +30559,9 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 5.4e-06,
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models",
|
||||
"supported_regions": [
|
||||
"us-central1"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
|
|
@ -31167,7 +31186,10 @@
|
|||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"ocr_cost_per_page": 0.0003,
|
||||
"source": "https://cloud.google.com/vertex-ai/pricing"
|
||||
"source": "https://cloud.google.com/vertex-ai/pricing",
|
||||
"supported_regions": [
|
||||
"us-central1"
|
||||
]
|
||||
},
|
||||
"vertex_ai/openai/gpt-oss-120b-maas": {
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
|
|
|
|||
|
|
@ -16949,7 +16949,6 @@
|
|||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-05,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
|
|
|
|||
|
|
@ -4,10 +4,11 @@ converted to float.
|
|||
GitHub Issue: #23843
|
||||
"""
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.proxy.proxy_server import initialize
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -17,22 +18,32 @@ async def test_max_budget_string_converted_to_float():
|
|||
string. initialize() should convert it to float so the comparison
|
||||
`litellm.max_budget > 0` doesn't raise TypeError.
|
||||
"""
|
||||
original = litellm.max_budget
|
||||
try:
|
||||
await initialize(max_budget="100.5")
|
||||
assert isinstance(litellm.max_budget, float)
|
||||
assert litellm.max_budget == 100.5
|
||||
finally:
|
||||
litellm.max_budget = original
|
||||
with patch("litellm.proxy.common_utils.banner.show_banner"), patch(
|
||||
"litellm.proxy.proxy_server.generate_feedback_box"
|
||||
):
|
||||
from litellm.proxy.proxy_server import initialize
|
||||
|
||||
original = litellm.max_budget
|
||||
try:
|
||||
await initialize(max_budget="100.5")
|
||||
assert isinstance(litellm.max_budget, float)
|
||||
assert litellm.max_budget == 100.5
|
||||
finally:
|
||||
litellm.max_budget = original
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_max_budget_float_stays_float():
|
||||
"""max_budget as float should still work."""
|
||||
original = litellm.max_budget
|
||||
try:
|
||||
await initialize(max_budget=200.0)
|
||||
assert isinstance(litellm.max_budget, float)
|
||||
assert litellm.max_budget == 200.0
|
||||
finally:
|
||||
litellm.max_budget = original
|
||||
with patch("litellm.proxy.common_utils.banner.show_banner"), patch(
|
||||
"litellm.proxy.proxy_server.generate_feedback_box"
|
||||
):
|
||||
from litellm.proxy.proxy_server import initialize
|
||||
|
||||
original = litellm.max_budget
|
||||
try:
|
||||
await initialize(max_budget=200.0)
|
||||
assert isinstance(litellm.max_budget, float)
|
||||
assert litellm.max_budget == 200.0
|
||||
finally:
|
||||
litellm.max_budget = original
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue