From 509d2e9ac3b3bbdc4a2c40840029d067857a66cc Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Sat, 21 Mar 2026 11:30:29 -0700 Subject: [PATCH] Fix PR review issues: gpt-4-0314 prompt caching, case-insensitive data URL check, test I/O mocking - Remove incorrect supports_prompt_caching from gpt-4-0314 (predates the feature) - Make data-URL detection case-insensitive in Gemini tool call result conversion - Mock show_banner/generate_feedback_box in max_budget tests to prevent real I/O Co-Authored-By: Claude Opus 4.6 --- .../prompt_templates/factory.py | 2 +- ...odel_prices_and_context_window_backup.json | 32 ++++++++++++--- model_prices_and_context_window.json | 1 - .../proxy/test_max_budget_env_var.py | 41 ++++++++++++------- 4 files changed, 54 insertions(+), 22 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 0e1fa05b9d4..d29ca1649ff 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -1506,7 +1506,7 @@ def convert_to_gemini_tool_call_result( # noqa: PLR0915 # Detect data-URL images (e.g. from Anthropic tool_result with a single image block # that was serialised as a plain string by translate_anthropic_messages_to_openai) # and promote them to inline_data so Gemini receives actual image bytes. - if content_str.startswith("data:") and ";base64," in content_str: + if content_str[:5].lower() == "data:" and ";base64," in content_str: try: mime_rest = content_str[5:].split(";base64,", 1) if len(mime_rest) == 2 and mime_rest[0].startswith("image/"): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 879dd42be47..b2fabb4936f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6152,7 +6152,8 @@ "max_query_tokens": 4096, "max_tokens": 32768, "mode": "rerank", - "output_cost_per_token": 0.0 + "output_cost_per_token": 0.0, + "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076" }, "azure_ai/cohere-rerank-v4.0-fast": { "input_cost_per_query": 0.002, @@ -6163,7 +6164,8 @@ "max_query_tokens": 4096, "max_tokens": 32768, "mode": "rerank", - "output_cost_per_token": 0.0 + "output_cost_per_token": 0.0, + "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076" }, "azure_ai/deepseek-v3.2": { "input_cost_per_token": 5.8e-07, @@ -6173,6 +6175,7 @@ "max_tokens": 163840, "mode": "chat", "output_cost_per_token": 1.68e-06, + "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549", "supports_assistant_prefill": true, "supports_function_calling": true, "supports_prompt_caching": true, @@ -6187,6 +6190,7 @@ "max_tokens": 163840, "mode": "chat", "output_cost_per_token": 1.68e-06, + "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-deepseek-v3-2-and-deepseek-v3-2-speciale-in-microsoft-foundry/4477549", "supports_assistant_prefill": true, "supports_function_calling": true, "supports_prompt_caching": true, @@ -16936,6 +16940,18 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "gpt-4-0314": { + "deprecation_date": "2026-03-26", + "input_cost_per_token": 3e-05, + "litellm_provider": "openai", + "max_input_tokens": 8192, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "chat", + "output_cost_per_token": 6e-05, + "supports_system_messages": true, + "supports_tool_choice": true + }, "gpt-4-0613": { "deprecation_date": "2025-06-06", "input_cost_per_token": 3e-05, @@ -30506,7 +30522,7 @@ "output_cost_per_token": 5.4e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models", "supported_regions": [ - "us-west2" + "us-central1" ], "supports_assistant_prefill": true, "supports_function_calling": true, @@ -30526,7 +30542,7 @@ "output_cost_per_token_batches": 8.4e-07, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models", "supported_regions": [ - "us-west2" + "global" ], "supports_assistant_prefill": true, "supports_function_calling": true, @@ -30543,6 +30559,9 @@ "mode": "chat", "output_cost_per_token": 5.4e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models", + "supported_regions": [ + "us-central1" + ], "supports_assistant_prefill": true, "supports_function_calling": true, "supports_prompt_caching": true, @@ -31167,7 +31186,10 @@ "input_cost_per_token": 3e-07, "output_cost_per_token": 1.2e-06, "ocr_cost_per_page": 0.0003, - "source": "https://cloud.google.com/vertex-ai/pricing" + "source": "https://cloud.google.com/vertex-ai/pricing", + "supported_regions": [ + "us-central1" + ] }, "vertex_ai/openai/gpt-oss-120b-maas": { "input_cost_per_token": 1.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ec96dcb2152..b2fabb4936f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -16949,7 +16949,6 @@ "max_tokens": 4096, "mode": "chat", "output_cost_per_token": 6e-05, - "supports_prompt_caching": true, "supports_system_messages": true, "supports_tool_choice": true }, diff --git a/tests/test_litellm/proxy/test_max_budget_env_var.py b/tests/test_litellm/proxy/test_max_budget_env_var.py index 4b965392823..90dfb81f3ae 100644 --- a/tests/test_litellm/proxy/test_max_budget_env_var.py +++ b/tests/test_litellm/proxy/test_max_budget_env_var.py @@ -4,10 +4,11 @@ converted to float. GitHub Issue: #23843 """ +from unittest.mock import patch + import pytest import litellm -from litellm.proxy.proxy_server import initialize @pytest.mark.asyncio @@ -17,22 +18,32 @@ async def test_max_budget_string_converted_to_float(): string. initialize() should convert it to float so the comparison `litellm.max_budget > 0` doesn't raise TypeError. """ - original = litellm.max_budget - try: - await initialize(max_budget="100.5") - assert isinstance(litellm.max_budget, float) - assert litellm.max_budget == 100.5 - finally: - litellm.max_budget = original + with patch("litellm.proxy.common_utils.banner.show_banner"), patch( + "litellm.proxy.proxy_server.generate_feedback_box" + ): + from litellm.proxy.proxy_server import initialize + + original = litellm.max_budget + try: + await initialize(max_budget="100.5") + assert isinstance(litellm.max_budget, float) + assert litellm.max_budget == 100.5 + finally: + litellm.max_budget = original @pytest.mark.asyncio async def test_max_budget_float_stays_float(): """max_budget as float should still work.""" - original = litellm.max_budget - try: - await initialize(max_budget=200.0) - assert isinstance(litellm.max_budget, float) - assert litellm.max_budget == 200.0 - finally: - litellm.max_budget = original + with patch("litellm.proxy.common_utils.banner.show_banner"), patch( + "litellm.proxy.proxy_server.generate_feedback_box" + ): + from litellm.proxy.proxy_server import initialize + + original = litellm.max_budget + try: + await initialize(max_budget=200.0) + assert isinstance(litellm.max_budget, float) + assert litellm.max_budget == 200.0 + finally: + litellm.max_budget = original