From 0a7c9a7231b3d7f2172b28a048278e482dd4d2ce Mon Sep 17 00:00:00 2001 From: Eric Chen Date: Sun, 20 Jul 2025 22:10:02 -0500 Subject: [PATCH 01/66] Fix anyof corner cases for BerriAI/litellm#11164 --- litellm/llms/vertex_ai/common_utils.py | 4 +-- ...test_vertex_and_google_ai_studio_gemini.py | 30 +++++++++++++++++++ 2 files changed, 31 insertions(+), 3 deletions(-) diff --git a/litellm/llms/vertex_ai/common_utils.py b/litellm/llms/vertex_ai/common_utils.py index cceac0ea794..41b2b030740 100644 --- a/litellm/llms/vertex_ai/common_utils.py +++ b/litellm/llms/vertex_ai/common_utils.py @@ -238,9 +238,7 @@ def _filter_anyof_fields(schema_dict: Dict[str, Any]) -> Dict[str, Any]: item["title"] = title if description: item["description"] = description - return {"anyOf": any_of} - else: - return schema_dict + return {"anyOf": any_of} return schema_dict diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 21395f59e5c..1a8780c6c16 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -495,6 +495,36 @@ def test_vertex_ai_map_tool_with_anyof(): "anyOf": [{"type": "string", "nullable": True, "title": "Base Branch"}] }, f"Expected only anyOf field and its contents to be kept, but got {tools[0]['function_declarations'][0]['parameters']['properties']['base_branch']}" + new_value = [ + { + "type": "function", + "function": { + "name": "git_create_branch", + "description": "Creates a new branch from an optional base branch", + "parameters": { + "type": "object", + "properties": { + "repo_path": {"title": "Repo Path", "type": "string"}, + "branch_name": {"title": "Branch Name", "type": "string"}, + "base_branch": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "default": None, + }, + }, + "required": ["repo_path", "branch_name"], + "title": "GitCreateBranch", + }, + }, + } + ] + new_tools = v._map_function(value=new_value) + + assert new_tools[0]["function_declarations"][0]["parameters"]["properties"][ + "base_branch" + ] == { + "anyOf": [{"type": "string", "nullable": True}] + }, f"Expected only anyOf field and its contents to be kept, but got {new_tools[0]['function_declarations'][0]['parameters']['properties']['base_branch']}" + def test_vertex_ai_streaming_usage_calculation(): """ From 2ce03d97356b0d61f50ee4564685ed29b06265e7 Mon Sep 17 00:00:00 2001 From: Adam Holmberg Date: Tue, 22 Jul 2025 10:41:29 -0500 Subject: [PATCH 02/66] fix: make gemini and openai responses return reasoning by default This aligns the proxy experience with other models that think automatically (e.g. Deepseek R1 and grok3). It does so by setting the necessary request input to return thinking, but not specifying a budget or effort (thus defaulting to the internal automatic level). --- .../transformation.py | 8 +- .../vertex_and_google_ai_studio_gemini.py | 14 +- ...responses_transformation_transformation.py | 191 ++++++++++++------ ...test_vertex_and_google_ai_studio_gemini.py | 76 +++++++ 4 files changed, 222 insertions(+), 67 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index f35510e41ba..97f27d6db12 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -157,8 +157,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): responses_api_request["metadata"] = value elif key in ("previous_response_id"): responses_api_request["previous_response_id"] = value - elif key == "reasoning_effort": - responses_api_request["reasoning"] = self._map_reasoning_effort(value) + + responses_api_request["reasoning"] = self._map_reasoning_effort(optional_params.get("reasoning_effort")) # Get stream parameter from litellm_params if not in optional_params stream = optional_params.get("stream") or litellm_params.get("stream", False) @@ -452,7 +452,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): responses_tools.append(tool) return cast(List["ALL_RESPONSES_API_TOOL_PARAMS"], responses_tools) - def _map_reasoning_effort(self, reasoning_effort: str) -> Optional[Reasoning]: + def _map_reasoning_effort(self, reasoning_effort: Optional[str]) -> Reasoning: if reasoning_effort == "high": return Reasoning(effort="high", summary="detailed") elif reasoning_effort == "medium": @@ -460,7 +460,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return Reasoning(effort="medium", summary="auto") elif reasoning_effort == "low": return Reasoning(effort="low", summary="auto") - return None + return Reasoning(summary="auto") def _map_responses_status_to_finish_reason(self, status: Optional[str]) -> str: """Map responses API status to chat completion finish_reason""" diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index d09599e8789..bb0a203d19b 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -417,8 +417,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): @staticmethod def _map_reasoning_effort_to_thinking_budget( - reasoning_effort: str, + reasoning_effort: Optional[str], ) -> GeminiThinkingConfig: + if not reasoning_effort: + return { "includeThoughts": True } if reasoning_effort == "low": return { "thinkingBudget": DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, @@ -595,10 +597,6 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): optional_params["parallel_tool_calls"] = value elif param == "seed": optional_params["seed"] = value - elif param == "reasoning_effort" and isinstance(value, str): - optional_params["thinkingConfig"] = ( - VertexGeminiConfig._map_reasoning_effort_to_thinking_budget(value) - ) elif param == "thinking": optional_params["thinkingConfig"] = ( VertexGeminiConfig._map_thinking_param( @@ -613,6 +611,12 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): optional_params = self._add_tools_to_optional_params( optional_params, [_tools] ) + if supports_reasoning(model): + optional_params["thinkingConfig"] = ( + VertexGeminiConfig._map_reasoning_effort_to_thinking_budget( + non_default_params.get("reasoning_effort") + ) + ) if litellm.vertex_ai_safety_settings is not None: optional_params["safety_settings"] = litellm.vertex_ai_safety_settings diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index ef76cfa02d1..e29e7509f9b 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -10,77 +10,152 @@ import httpx import pytest sys.path.insert( - 0, os.path.abspath("../../..") -) # Adds the parent directory to the system-path + 0, os.path.abspath("../../../../..") +) # Adds the parent directory to the system path import litellm +from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + OpenAiResponsesToChatCompletionStreamIterator, +) +from litellm.types.llms.openai import Reasoning +from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices -def test_convert_chat_completion_messages_to_responses_api_image_input(): - from litellm.completion_extras.litellm_responses_transformation.transformation import ( - LiteLLMResponsesTransformationHandler, - ) +class TestLiteLLMResponsesTransformation: + def setup_method(self): + self.handler = LiteLLMResponsesTransformationHandler() + self.model = "responses-api-model" + self.logging_obj = MagicMock() - handler = LiteLLMResponsesTransformationHandler() + def test_transform_request_reasoning_effort(self): + """ + Test that reasoning_effort is mapped to reasoning parameter correctly. + """ + # Case 1: reasoning_effort = "high" + optional_params_high = {"reasoning_effort": "high"} + result_high = self.handler.transform_request( + model=self.model, + messages=[], + optional_params=optional_params_high, + litellm_params={}, + headers={}, + litellm_logging_obj=self.logging_obj, + ) + assert "reasoning" in result_high + assert result_high["reasoning"] == Reasoning(effort="high", summary="detailed") - user_content = "What's in this image?" - user_image = "https://w7.pngwing.com/pngs/666/274/png-transparent-image-pictures-icon-photo-thumbnail.png" + # Case 2: reasoning_effort = "medium" + optional_params_medium = {"reasoning_effort": "medium"} + result_medium = self.handler.transform_request( + model=self.model, + messages=[], + optional_params=optional_params_medium, + litellm_params={}, + headers={}, + litellm_logging_obj=self.logging_obj, + ) + assert "reasoning" in result_medium + assert result_medium["reasoning"] == Reasoning(effort="medium", summary="auto") - messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": user_content, - }, - { - "type": "image_url", - "image_url": {"url": user_image}, - }, - ], - }, - ] + # Case 3: reasoning_effort = "low" + optional_params_low = {"reasoning_effort": "low"} + result_low = self.handler.transform_request( + model=self.model, + messages=[], + optional_params=optional_params_low, + litellm_params={}, + headers={}, + litellm_logging_obj=self.logging_obj, + ) + assert "reasoning" in result_low + assert result_low["reasoning"] == Reasoning(effort="low", summary="auto") - response, _ = handler.convert_chat_completion_messages_to_responses_api(messages) + # Case 4: no reasoning_effort + optional_params_none = {} + result_none = self.handler.transform_request( + model=self.model, + messages=[], + optional_params=optional_params_none, + litellm_params={}, + headers={}, + litellm_logging_obj=self.logging_obj, + ) + assert "reasoning" in result_none + assert result_none["reasoning"] == Reasoning(summary="auto") - response_str = json.dumps(response) + # Case 5: reasoning_effort = None + optional_params_explicit_none = {"reasoning_effort": None} + result_explicit_none = self.handler.transform_request( + model=self.model, + messages=[], + optional_params=optional_params_explicit_none, + litellm_params={}, + headers={}, + litellm_logging_obj=self.logging_obj, + ) + assert "reasoning" in result_explicit_none + assert result_explicit_none["reasoning"] == Reasoning(summary="auto") - assert user_content in response_str - assert user_image in response_str + def test_convert_chat_completion_messages_to_responses_api_image_input(self): + """ + Test that chat completion messages with image inputs are converted correctly. + """ + user_content = "What's in this image?" + user_image = "https://w7.pngwing.com/pngs/666/274/png-transparent-image-pictures-icon-photo-thumbnail.png" - print("response: ", response) - assert response[0]["content"][1]["image_url"] == user_image + messages = [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": user_content, + }, + { + "type": "image_url", + "image_url": {"url": user_image}, + }, + ], + }, + ] + response, _ = self.handler.convert_chat_completion_messages_to_responses_api(messages) -def test_openai_responses_chunk_parser_reasoning_summary(): - from litellm.completion_extras.litellm_responses_transformation.transformation import ( - OpenAiResponsesToChatCompletionStreamIterator, - ) - from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices + response_str = json.dumps(response) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + assert user_content in response_str + assert user_image in response_str - chunk = { - "delta": "**Compar", - "item_id": "rs_686d544208748198b6912e27b7c299c00e24bd875d35bade", - "output_index": 0, - "sequence_number": 4, - "summary_index": 0, - "type": "response.reasoning_summary_text.delta", - } + print("response: ", response) + assert response[0]["content"][1]["image_url"] == user_image - result = iterator.chunk_parser(chunk) + def test_openai_responses_chunk_parser_reasoning_summary(self): + """ + Test that OpenAI responses chunk parser handles reasoning summary correctly. + """ + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) - assert isinstance(result, ModelResponseStream) - assert len(result.choices) == 1 - choice = result.choices[0] - assert isinstance(choice, StreamingChoices) - assert choice.index == 0 - delta = choice.delta - assert isinstance(delta, Delta) - assert delta.content is None - assert delta.reasoning_content == "**Compar" - assert delta.tool_calls is None - assert delta.function_call is None + chunk = { + "delta": "**Compar", + "item_id": "rs_686d544208748198b6912e27b7c299c00e24bd875d35bade", + "output_index": 0, + "sequence_number": 4, + "summary_index": 0, + "type": "response.reasoning_summary_text.delta", + } + + result = iterator.chunk_parser(chunk) + + assert isinstance(result, ModelResponseStream) + assert len(result.choices) == 1 + choice = result.choices[0] + assert isinstance(choice, StreamingChoices) + assert choice.index == 0 + delta = choice.delta + assert isinstance(delta, Delta) + assert delta.content is None + assert delta.reasoning_content == "**Compar" + assert delta.tool_calls is None + assert delta.function_call is None diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 21395f59e5c..74a8e859d43 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -441,6 +441,82 @@ def test_vertex_ai_map_thinking_param_with_budget_tokens_0(): } +def test_vertex_ai_reasoning_effort_mapping(): + """ + Test that reasoning_effort is mapped to thinkingConfig correctly for models that support it. + - A default thinking config is applied if reasoning_effort is not specified. + - reasoning_effort correctly maps to thinkingConfig. + - No thinkingConfig is applied for models that do not support reasoning. + - reasoning_effort is prioritized over thinking param. + """ + v = VertexGeminiConfig() + optional_params = {} + + # Case 1: Model supports reasoning, no reasoning_effort provided + # Should apply default thinkingConfig + with patch( + "litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning", + return_value=True, + ): + result_params = v.map_openai_params( + non_default_params={}, + optional_params=deepcopy(optional_params), + model="gemini-2.5-pro", + drop_params=False, + ) + assert "thinkingConfig" in result_params + assert result_params["thinkingConfig"] == {"includeThoughts": True} + + # Case 2: Model supports reasoning, reasoning_effort is 'low' + # Should apply thinkingConfig with budget + with patch( + "litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning", + return_value=True, + ): + result_params_with_effort = v.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params=deepcopy(optional_params), + model="gemini-2.5-pro", + drop_params=False, + ) + assert "thinkingConfig" in result_params_with_effort + assert result_params_with_effort["thinkingConfig"]["includeThoughts"] is True + assert "thinkingBudget" in result_params_with_effort["thinkingConfig"] + + # Case 3: Model does not support reasoning + # Should not apply thinkingConfig + with patch( + "litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning", + return_value=False, + ): + result_params_no_support = v.map_openai_params( + non_default_params={}, + optional_params=deepcopy(optional_params), + model="gemini-pro", + drop_params=False, + ) + assert "thinkingConfig" not in result_params_no_support + + # Case 4: Model supports reasoning, but reasoning_effort is set, should be prioritized over thinking + with patch( + "litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning", + return_value=True, + ): + result_params_with_effort = v.map_openai_params( + non_default_params={ + "reasoning_effort": "low", + "thinking": {"type": "enabled", "budget_tokens": 1000}, + }, + optional_params=deepcopy(optional_params), + model="gemini-2.5-pro", + drop_params=False, + ) + assert "thinkingConfig" in result_params_with_effort + assert result_params_with_effort["thinkingConfig"]["includeThoughts"] is True + assert "thinkingBudget" in result_params_with_effort["thinkingConfig"] + assert result_params_with_effort["thinkingConfig"]["thinkingBudget"] != 1000 + + def test_vertex_ai_map_tools(): v = VertexGeminiConfig() tools = v._map_function(value=[{"code_execution": {}}]) From f81161a05cbe62eee96bfd3c7b6beb16819bf716 Mon Sep 17 00:00:00 2001 From: xywei Date: Tue, 22 Jul 2025 21:16:39 -0500 Subject: [PATCH 03/66] Remove vector store methods from global scope --- litellm/__init__.py | 1 - 1 file changed, 1 deletion(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index c056e667269..a8fe5063709 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1195,7 +1195,6 @@ from .router import Router from .assistants.main import * from .batches.main import * from .images.main import * -from .vector_stores import * from .batch_completion.main import * # type: ignore from .rerank_api.main import * from .llms.anthropic.experimental_pass_through.messages.handler import * From ed0ad6fd599812b16b0b012aabe35328e31e4054 Mon Sep 17 00:00:00 2001 From: xywei Date: Tue, 22 Jul 2025 22:18:02 -0500 Subject: [PATCH 04/66] Help mypy with typing asearch --- .../vector_store_integrations/vector_store_pre_call_hook.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py index 59c378f8204..8ef160dd783 100644 --- a/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py +++ b/litellm/integrations/vector_store_integrations/vector_store_pre_call_hook.py @@ -8,6 +8,7 @@ It searches the vector store for relevant context and appends it to the messages from typing import TYPE_CHECKING, Dict, List, Optional, Tuple, cast import litellm +import litellm.vector_stores from litellm._logging import verbose_logger from litellm.integrations.custom_logger import CustomLogger from litellm.types.llms.openai import AllMessageValues, ChatCompletionUserMessage @@ -192,4 +193,4 @@ class VectorStorePreCallHook(CustomLogger): modified_messages.insert(-1, cast(AllMessageValues, context_message)) return modified_messages - return messages \ No newline at end of file + return messages From eca86cd93b4c8c6a0b3cbe94ebdb19226fea1208 Mon Sep 17 00:00:00 2001 From: Viktor Nagy Date: Sun, 3 Aug 2025 17:45:34 +0200 Subject: [PATCH 05/66] Ensure that `function_call_prompt` extends system messages following its current schema Fixes #11267 --- .../prompt_templates/factory.py | 9 ++- .../test_system_message_format_bug.py | 72 +++++++++++++++++++ 2 files changed, 80 insertions(+), 1 deletion(-) create mode 100644 tests/test_litellm/test_system_message_format_bug.py diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index b4ace1545d2..2f3ffb3b89c 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -3711,7 +3711,14 @@ def function_call_prompt(messages: list, functions: list): function_added_to_prompt = False for message in messages: if "system" in message["role"]: - message["content"] += f""" {function_prompt}""" + if isinstance(message["content"], str): + message["content"] += f""" {function_prompt}""" + else: + message["content"].append({ + "type": "text", + "text": f""" {function_prompt}""", + "cache_control": {"type": "ephemeral"} + }) function_added_to_prompt = True if function_added_to_prompt is False: diff --git a/tests/test_litellm/test_system_message_format_bug.py b/tests/test_litellm/test_system_message_format_bug.py new file mode 100644 index 00000000000..a733b1be998 --- /dev/null +++ b/tests/test_litellm/test_system_message_format_bug.py @@ -0,0 +1,72 @@ +""" +Test for GitHub issue #11267 - System message format issue with Ollama + tools +""" + +from unittest.mock import patch + +@patch("litellm.add_function_to_prompt", True) +def test_system_message_format_issue_reproduction(): + """ + Reproduces the system message format bug from GitHub issue #11267. + """ + from litellm import completion + + # Define test data directly from data.jsonl content + model = "ollama/custom_model_name" # Use explicit Ollama model + messages = [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What is the capital of France?" + } + ] + }, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "You are Claude Code, Anthropic's official CLI for Claude.", + "cache_control": {"type": "ephemeral"} + } + ] + } + ] + + temperature = 1 + + # Add tools to trigger the bug - this is what causes the issue + tools = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get weather for a location", + "parameters": { + "type": "object", + "properties": { + "location": {"type": "string"} + }, + "required": ["location"] + } + } + } + ] + + response = completion( + model=model, + messages=messages, + tools=tools, + temperature=temperature, + mock_response=True + ) + + assert len(messages[1]["content"]) == 2 + + +if __name__ == "__main__": + print("Testing system message format issue...") + test_system_message_format_issue_reproduction() + print("Tests completed!") \ No newline at end of file From 5702e5ee1f5a6064e56992014fcbc03b3831046a Mon Sep 17 00:00:00 2001 From: Viktor Nagy <126671+nagyv@users.noreply.github.com> Date: Wed, 6 Aug 2025 06:18:12 +0200 Subject: [PATCH 06/66] Removed cache control --- litellm/litellm_core_utils/prompt_templates/factory.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 2f3ffb3b89c..7be6988d658 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -3716,8 +3716,7 @@ def function_call_prompt(messages: list, functions: list): else: message["content"].append({ "type": "text", - "text": f""" {function_prompt}""", - "cache_control": {"type": "ephemeral"} + "text": f""" {function_prompt}""" }) function_added_to_prompt = True From 4fdeff8e1a71fdacc444dc35cfbef882fa25fe2e Mon Sep 17 00:00:00 2001 From: Yikai Zhao Date: Thu, 7 Aug 2025 22:58:07 +0800 Subject: [PATCH 07/66] Fix token_counter with special token input --- litellm/litellm_core_utils/token_counter.py | 2 +- tests/test_litellm/litellm_core_utils/test_token_counter.py | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/token_counter.py b/litellm/litellm_core_utils/token_counter.py index 4df944edbaa..fab2c1e76ee 100644 --- a/litellm/litellm_core_utils/token_counter.py +++ b/litellm/litellm_core_utils/token_counter.py @@ -529,7 +529,7 @@ def _get_count_function( encoding = tiktoken.get_encoding("cl100k_base") def count_tokens(text: str) -> int: - return len(encoding.encode(text)) + return len(encoding.encode(text, disallowed_special=())) else: raise ValueError("Unsupported tokenizer type") diff --git a/tests/test_litellm/litellm_core_utils/test_token_counter.py b/tests/test_litellm/litellm_core_utils/test_token_counter.py index 71ee367bdec..5d17ea3dc3c 100644 --- a/tests/test_litellm/litellm_core_utils/test_token_counter.py +++ b/tests/test_litellm/litellm_core_utils/test_token_counter.py @@ -451,6 +451,7 @@ def test_img_url_token_counter(img_url): def test_token_encode_disallowed_special(): encode(model="gpt-3.5-turbo", text="Hello, world! <|endoftext|>") + token_counter(model="gpt-3.5-turbo", text="Hello, world! <|endoftext|>") def test_token_counter(): From 81f25633381039b099ebc03df19a6f0c56e40722 Mon Sep 17 00:00:00 2001 From: Davide Pugliese Date: Thu, 7 Aug 2025 14:01:14 +0200 Subject: [PATCH 08/66] Enhance logging for containers --- .dockerignore | 1 + .gitignore | 3 +- litellm/_logging.py | 51 +- tests/test_litellm/conftest.py | 74 +++ tests/test_litellm/test_logging_behavior.py | 638 ++++++++++++++++++++ 5 files changed, 760 insertions(+), 7 deletions(-) create mode 100644 tests/test_litellm/test_logging_behavior.py diff --git a/.dockerignore b/.dockerignore index 89c3c34bd71..766b7a1db67 100644 --- a/.dockerignore +++ b/.dockerignore @@ -10,3 +10,4 @@ tests *.tgz log.txt docker/Dockerfile.* +*.whl diff --git a/.gitignore b/.gitignore index f8d028ff47b..9613ef77d90 100644 --- a/.gitignore +++ b/.gitignore @@ -93,4 +93,5 @@ test.py litellm_config.yaml .cursor -.vscode/launch.json \ No newline at end of file +.vscode/launch.json +*.whl \ No newline at end of file diff --git a/litellm/_logging.py b/litellm/_logging.py index 8c23994f92a..9a8c8251ad4 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -4,21 +4,41 @@ import os import sys from datetime import datetime from logging import Formatter - set_verbose = False +def __strtobool(val: str) -> bool: + """Convert a string representation of truth to true (1) or false (0). + + True values are 'y', 'yes', 't', 'true', 'on', and '1'; false values + are 'n', 'no', 'f', 'false', 'off', and '0'. Raises ValueError if + 'val' is anything else. + """ + val = val.lower() + if val in ('y', 'yes', 't', 'true', 'on', '1'): + return True + elif val in ('n', 'no', 'f', 'false', 'off', '0'): + return False + else: + raise ValueError(f"invalid truth value {val!r}") + if set_verbose is True: logging.warning( "`litellm.set_verbose` is deprecated. Please set `os.environ['LITELLM_LOG'] = 'DEBUG'` for debug logs." ) -json_logs = bool(os.getenv("JSON_LOGS", False)) + +json_logs = __strtobool(os.getenv("JSON_LOGS", "False")) # Create a handler for the logger (you may need to adapt this based on your needs) log_level = os.getenv("LITELLM_LOG", "DEBUG") numeric_level: str = getattr(logging, log_level.upper()) handler = logging.StreamHandler() handler.setLevel(numeric_level) +log_file = os.getenv("LITELLM_LOG_FILE", "") +file_handler = None +if log_file: + file_handler = logging.FileHandler(log_file) + file_handler.setLevel(numeric_level) class JsonFormatter(Formatter): def __init__(self): super(JsonFormatter, self).__init__() @@ -40,6 +60,7 @@ class JsonFormatter(Formatter): return json.dumps(json_record) +json_formatter = JsonFormatter() # Function to set up exception handlers for JSON logging def _setup_json_exception_handlers(formatter): @@ -89,8 +110,10 @@ def _setup_json_exception_handlers(formatter): # Create a formatter and set it for the handler if json_logs: - handler.setFormatter(JsonFormatter()) - _setup_json_exception_handlers(JsonFormatter()) + handler.setFormatter(json_formatter) + if file_handler: + file_handler.setFormatter(json_formatter) + _setup_json_exception_handlers(json_formatter) else: formatter = logging.Formatter( "\033[92m%(asctime)s - %(name)s:%(levelname)s\033[0m: %(filename)s:%(lineno)s - %(message)s", @@ -98,11 +121,18 @@ else: ) handler.setFormatter(formatter) + if file_handler: + file_handler.setFormatter(formatter) verbose_proxy_logger = logging.getLogger("LiteLLM Proxy") verbose_router_logger = logging.getLogger("LiteLLM Router") verbose_logger = logging.getLogger("LiteLLM") +# Set logger levels +verbose_proxy_logger.setLevel(numeric_level) +verbose_router_logger.setLevel(numeric_level) +verbose_logger.setLevel(numeric_level) + # Add the handler to the logger verbose_router_logger.addHandler(handler) verbose_proxy_logger.addHandler(handler) @@ -123,6 +153,13 @@ def _suppress_loggers(): # Call the suppression function _suppress_loggers() +if file_handler: + verbose_router_logger.addHandler(file_handler) + verbose_proxy_logger.addHandler(file_handler) + verbose_logger.addHandler(file_handler) + + + ALL_LOGGERS = [ logging.getLogger(), verbose_logger, @@ -151,10 +188,10 @@ def _turn_on_json(): - Adds a JSON formatter to all loggers """ handler = logging.StreamHandler() - handler.setFormatter(JsonFormatter()) + handler.setFormatter(json_formatter) _initialize_loggers_with_handler(handler) # Set up exception handlers - _setup_json_exception_handlers(JsonFormatter()) + _setup_json_exception_handlers(json_formatter) def _turn_on_debug(): @@ -190,3 +227,5 @@ def _is_debugging_on() -> bool: if verbose_logger.isEnabledFor(logging.DEBUG) or set_verbose is True: return True return False + + diff --git a/tests/test_litellm/conftest.py b/tests/test_litellm/conftest.py index a88148f9d11..db4224ca6e4 100644 --- a/tests/test_litellm/conftest.py +++ b/tests/test_litellm/conftest.py @@ -3,15 +3,88 @@ import importlib import os import sys +import tempfile +import random +import string import pytest +# Set up a temporary log directory and file BEFORE importing litellm +temp_dir = tempfile.mkdtemp(prefix="litellm_test_") +test_log_file = os.path.join(temp_dir, "test_litellm.log") + +# Store original log file for cleanup +orig_log_file = os.getenv("LITELLM_LOG_FILE") + +# Set environment variables to use temporary files BEFORE importing litellm +os.environ["LITELLM_LOG_FILE"] = test_log_file + +# Import litellm after setting up the environment sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path import litellm + + +@pytest.fixture(scope="function") +def temp_log_file(): + """ + Creates a temporary log file in /tmp/litellm.log for testing. + Returns the path to the temporary log file and cleans it up after the test. + """ + # Generate a random number for the log file + random_number = ''.join(random.choices(string.digits, k=8)) + log_file_path = f"/tmp/litellm{random_number}.log" + + # Set the environment variable for litellm to use this temporary log file + original_log_file = os.environ.get("LITELLM_LOG_FILE") + os.environ["LITELLM_LOG_FILE"] = log_file_path + + yield log_file_path + + # Cleanup: Restore original environment variable and remove the temporary file + if original_log_file is not None: + os.environ["LITELLM_LOG_FILE"] = original_log_file + else: + os.environ.pop("LITELLM_LOG_FILE", None) + + # Remove the temporary log file if it exists + if os.path.exists(log_file_path): + try: + os.remove(log_file_path) + except OSError: + pass # Ignore errors if file can't be removed + + +@pytest.fixture(scope="session", autouse=True) +def cleanup_temp_log_dir(): + """ + Cleans up the temporary log directory created at module import time. + This runs once per test session after all tests are complete. + """ + yield + + if orig_log_file is not None: + os.environ["LITELLM_LOG_FILE"] = orig_log_file + else: + os.environ.pop("LITELLM_LOG_FILE", None) + + # Cleanup: Remove the temporary directory created at module import time + if os.path.exists(temp_dir): + try: + # Remove the test log file first + if os.path.exists(test_log_file): + os.remove(test_log_file) + + # Remove the temporary directory + import shutil + shutil.rmtree(temp_dir, ignore_errors=True) + except OSError: + pass # Ignore errors if cleanup fails + + @pytest.fixture(scope="function", autouse=True) def setup_and_teardown(): """ @@ -63,3 +136,4 @@ def pytest_collection_modifyitems(config, items): # Reorder the items list items[:] = custom_logger_tests + other_tests + diff --git a/tests/test_litellm/test_logging_behavior.py b/tests/test_litellm/test_logging_behavior.py new file mode 100644 index 00000000000..24f92838acc --- /dev/null +++ b/tests/test_litellm/test_logging_behavior.py @@ -0,0 +1,638 @@ +import os +import tempfile +import re +import json +from pathlib import Path +from datetime import datetime + +import pytest + +# Import the loggers from litellm._logging +from litellm._logging import verbose_logger, verbose_proxy_logger, verbose_router_logger + + +class TestLoggingBehavior: + """Test suite to verify logging behavior for all LiteLLM loggers.""" + + def read_log_file_contents(self, log_file_path): + """Helper method to read and return contents of log file.""" + if not os.path.exists(log_file_path): + return "" + + with open(log_file_path, 'r') as f: + return f.read() + + @pytest.fixture(autouse=True) + def setup_log_file(self, temp_log_file): + """Use the temp_log_file fixture to ensure proper isolation.""" + self.temp_log_path = temp_log_file + + # Set environment variable before importing/reloading + original_log_file = os.environ.get("LITELLM_LOG_FILE") + os.environ["LITELLM_LOG_FILE"] = temp_log_file + + # Force reload of the logging module to pick up new environment variable + import importlib + import litellm._logging + importlib.reload(litellm._logging) + + yield + + # Cleanup: Restore original environment variable + if original_log_file is not None: + os.environ["LITELLM_LOG_FILE"] = original_log_file + else: + os.environ.pop("LITELLM_LOG_FILE", None) + + # Reload again to restore original state + importlib.reload(litellm._logging) + + def test_verbose_logger_info_level(self): + """Test that verbose_logger writes to file with INFO level.""" + test_message = "INFO level test message from verbose_logger" + + # Log at INFO level + verbose_logger.info(test_message) + + # Force flush all handlers to ensure they write to disk + for handler in verbose_logger.handlers: + if hasattr(handler, 'flush'): + handler.flush() + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_verbose_logger_debug_level(self): + """Test that verbose_logger writes to file with DEBUG level.""" + test_message = "DEBUG level test message from verbose_logger" + + # Log at DEBUG level + verbose_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_verbose_proxy_logger_info_level(self): + """Test that verbose_proxy_logger writes to file with INFO level.""" + test_message = "INFO level test message from verbose_proxy_logger" + + # Log at INFO level + verbose_proxy_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_verbose_proxy_logger_debug_level(self): + """Test that verbose_proxy_logger writes to file with DEBUG level.""" + test_message = "DEBUG level test message from verbose_proxy_logger" + + # Log at DEBUG level + verbose_proxy_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_verbose_router_logger_info_level(self): + """Test that verbose_router_logger writes to file with INFO level.""" + test_message = "INFO level test message from verbose_router_logger" + + # Log at INFO level + verbose_router_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_verbose_router_logger_debug_level(self): + """Test that verbose_router_logger writes to file with DEBUG level.""" + test_message = "DEBUG level test message from verbose_router_logger" + + # Log at DEBUG level + verbose_router_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert test_message in log_contents, f"Message '{test_message}' should be found in log file" + + def test_log_format_includes_timestamp_and_level(self): + """Test that log entries include timestamp and level information.""" + test_message = "Format test message" + + # Log at INFO level + verbose_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + + # Check for timestamp format (should be in HH:MM:SS format based on _logging.py) + assert re.search(r'\d{2}:\d{2}:\d{2}', log_contents), "Log should contain timestamp in HH:MM:SS format" + + # Check for level information + assert 'INFO' in log_contents, "Log should contain INFO level indicator" + + # Check for logger name + assert 'LiteLLM' in log_contents, "Log should contain LiteLLM logger name" + + def test_multiple_loggers_write_to_same_file(self): + """Test that all loggers write to the same file.""" + messages = { + 'verbose_logger': "Message from verbose_logger", + 'verbose_proxy_logger': "Message from verbose_proxy_logger", + 'verbose_router_logger': "Message from verbose_router_logger" + } + + # Log messages from different loggers + verbose_logger.info(messages['verbose_logger']) + verbose_proxy_logger.info(messages['verbose_proxy_logger']) + verbose_router_logger.info(messages['verbose_router_logger']) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + + # Verify all messages are in the same file + for message in messages.values(): + assert message in log_contents, f"Message '{message}' should be found in log file" + + def test_log_file_is_not_empty(self): + """Test that the log file is not empty after logging.""" + # Log a message + verbose_logger.info("Test message to ensure file is not empty") + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + + # Verify file is not empty + assert len(log_contents.strip()) > 0, "Log file should not be empty after logging" + + +class TestJSONLoggingBehavior: + """Test suite to verify JSON logging behavior for all LiteLLM loggers.""" + + def read_log_file_contents(self, log_file_path): + """Helper method to read and return contents of log file.""" + if not os.path.exists(log_file_path): + return "" + + with open(log_file_path, 'r') as f: + return f.read() + + @pytest.fixture(autouse=True) + def setup_json_logging(self, temp_log_file): + """Set up JSON logging environment and ensure proper isolation.""" + self.temp_log_path = temp_log_file + + # Store original environment variables + original_log_file = os.environ.get("LITELLM_LOG_FILE") + original_json_logs = os.environ.get("JSON_LOGS") + + # Set environment variables for JSON logging + os.environ["LITELLM_LOG_FILE"] = temp_log_file + os.environ["JSON_LOGS"] = "True" + + # Force reload of the logging module to pick up new environment variables + import importlib + import litellm._logging + importlib.reload(litellm._logging) + + yield + + # Cleanup: Restore original environment variables + if original_log_file is not None: + os.environ["LITELLM_LOG_FILE"] = original_log_file + else: + os.environ.pop("LITELLM_LOG_FILE", None) + + if original_json_logs is not None: + os.environ["JSON_LOGS"] = original_json_logs + else: + os.environ.pop("JSON_LOGS", None) + + # Reload again to restore original state + importlib.reload(litellm._logging) + + def test_verbose_logger_json_info_level(self): + """Test that verbose_logger writes JSON formatted logs at INFO level.""" + test_message = "JSON INFO level test message from verbose_logger" + + # Log at INFO level + verbose_logger.info(test_message) + + # Force flush all handlers to ensure they write to disk + for handler in verbose_logger.handlers: + if hasattr(handler, 'flush'): + handler.flush() + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + assert len(log_lines) > 0, "Should have at least one log line" + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + + # Verify JSON structure + assert "message" in target_log, "JSON log should contain 'message' field" + assert "level" in target_log, "JSON log should contain 'level' field" + assert "timestamp" in target_log, "JSON log should contain 'timestamp' field" + + # Verify content + assert target_log["message"] == test_message + assert target_log["level"] == "INFO" + + # Verify timestamp is in ISO 8601 format + timestamp_str = target_log["timestamp"] + try: + datetime.fromisoformat(timestamp_str) + except ValueError: + pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") + + def test_verbose_logger_json_debug_level(self): + """Test that verbose_logger writes JSON formatted logs at DEBUG level.""" + test_message = "JSON DEBUG level test message from verbose_logger" + + # Log at DEBUG level + verbose_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + assert target_log["level"] == "DEBUG" + + def test_verbose_proxy_logger_json_info_level(self): + """Test that verbose_proxy_logger writes JSON formatted logs at INFO level.""" + test_message = "JSON INFO level test message from verbose_proxy_logger" + + # Log at INFO level + verbose_proxy_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + + # Verify JSON structure and content + assert target_log["message"] == test_message + assert target_log["level"] == "INFO" + + # Verify timestamp is in ISO 8601 format + timestamp_str = target_log["timestamp"] + try: + datetime.fromisoformat(timestamp_str) + except ValueError: + pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") + + def test_verbose_proxy_logger_json_debug_level(self): + """Test that verbose_proxy_logger writes JSON formatted logs at DEBUG level.""" + test_message = "JSON DEBUG level test message from verbose_proxy_logger" + + # Log at DEBUG level + verbose_proxy_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + assert target_log["level"] == "DEBUG" + + def test_verbose_router_logger_json_info_level(self): + """Test that verbose_router_logger writes JSON formatted logs at INFO level.""" + test_message = "JSON INFO level test message from verbose_router_logger" + + # Log at INFO level + verbose_router_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + + # Verify JSON structure and content + assert target_log["message"] == test_message + assert target_log["level"] == "INFO" + + # Verify timestamp is in ISO 8601 format + timestamp_str = target_log["timestamp"] + try: + datetime.fromisoformat(timestamp_str) + except ValueError: + pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") + + def test_verbose_router_logger_json_debug_level(self): + """Test that verbose_router_logger writes JSON formatted logs at DEBUG level.""" + test_message = "JSON DEBUG level test message from verbose_router_logger" + + # Log at DEBUG level + verbose_router_logger.debug(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify structure + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + assert target_log["level"] == "DEBUG" + + def test_json_output_is_valid_json(self): + """Test that all JSON log output can be parsed as valid JSON.""" + test_messages = [ + "JSON test message 1", + "JSON test message 2", + "JSON test message 3" + ] + + # Log messages from all loggers + verbose_logger.info(test_messages[0]) + verbose_proxy_logger.info(test_messages[1]) + verbose_router_logger.info(test_messages[2]) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse each line as JSON + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + parsed_logs = [] + + for line in log_lines: + try: + parsed = json.loads(line) + parsed_logs.append(parsed) + except json.JSONDecodeError as e: + pytest.fail(f"Failed to parse JSON log line: {line}. Error: {e}") + + assert len(parsed_logs) >= len(test_messages), f"Should have at least {len(test_messages)} parsed log entries" + + # Verify each parsed log has required fields + for parsed_log in parsed_logs: + assert isinstance(parsed_log, dict), "Parsed log should be a dictionary" + assert "message" in parsed_log, "Each log should have a 'message' field" + assert "level" in parsed_log, "Each log should have a 'level' field" + assert "timestamp" in parsed_log, "Each log should have a 'timestamp' field" + + def test_json_timestamp_iso8601_format(self): + """Test that JSON log timestamps are in ISO 8601 format.""" + test_message = "Timestamp format test message" + + # Log a message + verbose_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify timestamp format + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + + timestamp_str = target_log["timestamp"] + + # Verify timestamp can be parsed as ISO 8601 + try: + parsed_timestamp = datetime.fromisoformat(timestamp_str) + assert isinstance(parsed_timestamp, datetime), "Parsed timestamp should be a datetime object" + except ValueError as e: + pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format. Error: {e}") + + # Verify timestamp format matches expected pattern (YYYY-MM-DDTHH:MM:SS.ffffff) + import re + iso8601_pattern = r'^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?$' + assert re.match(iso8601_pattern, timestamp_str), f"Timestamp '{timestamp_str}' does not match ISO 8601 pattern" + + def test_json_logs_contain_expected_fields(self): + """Test that JSON logs contain all expected fields with correct types.""" + test_message = "Field validation test message" + + # Log a message + verbose_logger.info(test_message) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse JSON and verify fields + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + + # Find the line containing our test message + target_log = None + for line in log_lines: + try: + parsed = json.loads(line) + if parsed.get("message") == test_message: + target_log = parsed + break + except json.JSONDecodeError: + continue + + assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" + + # Verify required fields exist and have correct types + assert "message" in target_log, "JSON log should contain 'message' field" + assert "level" in target_log, "JSON log should contain 'level' field" + assert "timestamp" in target_log, "JSON log should contain 'timestamp' field" + + assert isinstance(target_log["message"], str), "'message' field should be a string" + assert isinstance(target_log["level"], str), "'level' field should be a string" + assert isinstance(target_log["timestamp"], str), "'timestamp' field should be a string" + + # Verify field values + assert target_log["message"] == test_message + assert target_log["level"] in ["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"], "Level should be a valid log level" + + def test_multiple_json_loggers_write_to_same_file(self): + """Test that all loggers write JSON formatted logs to the same file.""" + messages = { + 'verbose_logger': "JSON message from verbose_logger", + 'verbose_proxy_logger': "JSON message from verbose_proxy_logger", + 'verbose_router_logger': "JSON message from verbose_router_logger" + } + + # Log messages from different loggers + verbose_logger.info(messages['verbose_logger']) + verbose_proxy_logger.info(messages['verbose_proxy_logger']) + verbose_router_logger.info(messages['verbose_router_logger']) + + # Read log file contents + log_file_path = os.environ.get("LITELLM_LOG_FILE") + assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" + + log_contents = self.read_log_file_contents(log_file_path) + assert log_contents.strip(), "Log file should not be empty" + + # Parse all JSON logs + log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] + parsed_logs = [] + + for line in log_lines: + try: + parsed = json.loads(line) + parsed_logs.append(parsed) + except json.JSONDecodeError: + continue + + # Find logs for each message + found_messages = set() + for parsed_log in parsed_logs: + message = parsed_log.get("message", "") + if message in messages.values(): + found_messages.add(message) + + # Verify all messages are found in JSON format + for message in messages.values(): + assert message in found_messages, f"Message '{message}' should be found in JSON logs" \ No newline at end of file From c9b334fcd11e5200ec518fc5fc523fcadf53d9b7 Mon Sep 17 00:00:00 2001 From: TensorNull Date: Sat, 9 Aug 2025 10:42:45 +0800 Subject: [PATCH 09/66] feat: add CometAPI support with config, error handling and tests --- litellm/__init__.py | 8 +- litellm/constants.py | 2 + .../get_llm_provider_logic.py | 3 + litellm/llms/cometapi/chat/transformation.py | 207 ++++++++++++ litellm/llms/cometapi/common_utils.py | 6 + litellm/main.py | 39 +++ litellm/types/utils.py | 1 + litellm/utils.py | 2 + .../chat/test_cometapi_chat_transformation.py | 318 ++++++++++++++++++ 9 files changed, 585 insertions(+), 1 deletion(-) create mode 100644 litellm/llms/cometapi/chat/transformation.py create mode 100644 litellm/llms/cometapi/common_utils.py create mode 100644 tests/test_litellm/llms/cometapi/chat/test_cometapi_chat_transformation.py diff --git a/litellm/__init__.py b/litellm/__init__.py index f7e1fb8f24d..727ed9866f3 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -232,6 +232,7 @@ nlp_cloud_key: Optional[str] = None novita_api_key: Optional[str] = None snowflake_key: Optional[str] = None nebius_key: Optional[str] = None +cometapi_key: Optional[str] = None common_cloud_provider_auth_params: dict = { "params": ["project", "region_name", "token"], "providers": ["vertex_ai", "bedrock", "watsonx", "azure", "vertex_ai_beta"], @@ -533,6 +534,7 @@ morph_models: List = [] lambda_ai_models: List = [] hyperbolic_models: List = [] recraft_models: List = [] +cometapi_models: List = [] oci_models: List = [] @@ -723,6 +725,8 @@ def add_known_models(): hyperbolic_models.append(key) elif value.get("litellm_provider") == "recraft": recraft_models.append(key) + elif value.get("litellm_provider") == "cometapi": + cometapi_models.append(key) elif value.get("litellm_provider") == "oci": oci_models.append(key) @@ -813,6 +817,7 @@ model_list = ( + morph_models + lambda_ai_models + recraft_models + + cometapi_models + oci_models ) @@ -887,6 +892,7 @@ models_by_provider: dict = { "lambda_ai": lambda_ai_models, "hyperbolic": hyperbolic_models, "recraft": recraft_models, + "cometapi": cometapi_models, "oci": oci_models, } @@ -1191,7 +1197,7 @@ from .llms.azure.azure import ( AzureOpenAIError, AzureOpenAIAssistantsAPIConfig, ) - +from .llms.cometapi.chat.transformation import CometAPIConfig from .llms.azure.chat.gpt_transformation import AzureOpenAIConfig from .llms.azure.completion.transformation import AzureOpenAITextConfig from .llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig diff --git a/litellm/constants.py b/litellm/constants.py index c7404f10a78..c526ee1065c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -224,6 +224,7 @@ LITELLM_CHAT_PROVIDERS = [ "together_ai", "datarobot", "openrouter", + "cometapi", "vertex_ai", "vertex_ai_beta", "gemini", @@ -877,6 +878,7 @@ SENTRY_DENYLIST = [ "CLOUDFLARE_API_KEY", "BASETEN_KEY", "OPENROUTER_KEY", + "COMETAPI_KEY", "DATAROBOT_API_TOKEN", "FIREWORKS_API_KEY", "FIREWORKS_AI_API_KEY", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 702196a7f05..7a2ec0b523d 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -356,6 +356,9 @@ def get_llm_provider( # noqa: PLR0915 # bytez models elif model.startswith("bytez/"): custom_llm_provider = "bytez" + # cometapi models + elif model.startswith("cometapi/"): + custom_llm_provider = "cometapi" elif model.startswith("oci/"): custom_llm_provider = "oci" if not custom_llm_provider: diff --git a/litellm/llms/cometapi/chat/transformation.py b/litellm/llms/cometapi/chat/transformation.py new file mode 100644 index 00000000000..391f9626d37 --- /dev/null +++ b/litellm/llms/cometapi/chat/transformation.py @@ -0,0 +1,207 @@ +""" +Support for CometAPI's `/v1/chat/completions` endpoint. + +Based on OpenAI-compatible API interface implementation +Documentation: [CometAPI Documentation Link] +""" + +from typing import Any, AsyncIterator, Iterator, List, Optional, Tuple, Union + +import httpx + +from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.types.llms.openai import AllMessageValues, ChatCompletionToolParam +from litellm.types.utils import ModelResponse, ModelResponseStream + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig +from ..common_utils import CometAPIException + + +class CometAPIConfig(OpenAIGPTConfig): + """ + CometAPI configuration class, inherits from OpenAIGPTConfig + + Since CometAPI is OpenAI-compatible API, we inherit from OpenAIGPTConfig + and only need to override necessary methods to handle CometAPI-specific features + """ + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + """ + Map OpenAI format parameters to CometAPI format + """ + mapped_openai_params = super().map_openai_params( + non_default_params, optional_params, model, drop_params + ) + + # CometAPI-specific parameters (if any) + extra_body = {} + # TODO: Add CometAPI-specific parameter handling here + # Example: + # custom_param = non_default_params.pop("custom_param", None) + # if custom_param is not None: + # extra_body["custom_param"] = custom_param + + if extra_body: + mapped_openai_params["extra_body"] = extra_body + + return mapped_openai_params + + def remove_cache_control_flag_from_messages_and_tools( + self, + model: str, + messages: List[AllMessageValues], + tools: Optional[List["ChatCompletionToolParam"]] = None, + ) -> Tuple[List[AllMessageValues], Optional[List["ChatCompletionToolParam"]]]: + """ + Remove cache control flags from messages and tools if not supported + """ + # For CometAPI, use default behavior (remove cache control) + return super().remove_cache_control_flag_from_messages_and_tools( + model, messages, tools + ) + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + """ + Transform the overall request to be sent to the API. + + Returns: + dict: The transformed request. Sent as the body of the API call. + """ + extra_body = optional_params.pop("extra_body", {}) + response = super().transform_request( + model, messages, optional_params, litellm_params, headers + ) + response.update(extra_body) + return response + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + Get the complete URL for the CometAPI call. + + Returns: + str: The complete URL for the API call. + """ + # Default base + if api_base is None: + api_base = "https://api.cometapi.com/v1" + endpoint = "chat/completions" + + # Normalize + api_base = api_base.rstrip("/") + + # If endpoint already present, return as-is + if endpoint in api_base: + return api_base + + # Ensure we include /v1 prefix when missing + if api_base.endswith("/v1"): + return f"{api_base}/{endpoint}" + if api_base.endswith("/v1/"): + return f"{api_base}{endpoint}" + # If user provided https://api.cometapi.com, add /v1 + if api_base == "https://api.cometapi.com": + return f"{api_base}/v1/{endpoint}" + # Generic fallback: if '/v1' not in path, add it + if "/v1" not in api_base.split("//", 1)[-1]: + return f"{api_base}/v1/{endpoint}" + return f"{api_base}/{endpoint}" + + def get_error_class( + self, + error_message: str, + status_code: int, + headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + """ + Return CometAPI-specific error class + """ + return CometAPIException( + message=error_message, + status_code=status_code, + headers=headers, + ) + + def get_model_response_iterator( + self, + streaming_response: Union[Iterator[str], AsyncIterator[str], ModelResponse], + sync_stream: bool, + json_mode: Optional[bool] = False, + ) -> Any: + """ + Get model response iterator for streaming responses + """ + return CometAPIChatCompletionStreamingHandler( + streaming_response=streaming_response, + sync_stream=sync_stream, + json_mode=json_mode, + ) + + +class CometAPIChatCompletionStreamingHandler(BaseModelResponseIterator): + """ + Handler for CometAPI streaming chat completion responses + """ + + def chunk_parser(self, chunk: dict) -> ModelResponseStream: + """ + Parse individual chunks from streaming response + """ + try: + # Handle error in chunk + if "error" in chunk: + error_chunk = chunk["error"] + error_message = "CometAPI Error: {}".format( + error_chunk.get("message", "Unknown error") + ) + raise CometAPIException( + message=error_message, + status_code=error_chunk.get("code", 400), + headers={"Content-Type": "application/json"}, + ) + + # Process choices + new_choices = [] + for choice in chunk["choices"]: + # Handle reasoning content if present + if "delta" in choice and "reasoning" in choice["delta"]: + choice["delta"]["reasoning_content"] = choice["delta"].get("reasoning") + new_choices.append(choice) + + return ModelResponseStream( + id=chunk["id"], + object="chat.completion.chunk", + created=chunk["created"], + usage=chunk.get("usage"), + model=chunk["model"], + choices=new_choices, + ) + except KeyError as e: + raise CometAPIException( + message=f"KeyError: {e}, Got unexpected response from CometAPI: {chunk}", + status_code=400, + headers={"Content-Type": "application/json"}, + ) + except Exception as e: + raise e diff --git a/litellm/llms/cometapi/common_utils.py b/litellm/llms/cometapi/common_utils.py new file mode 100644 index 00000000000..2e5e3e5fab7 --- /dev/null +++ b/litellm/llms/cometapi/common_utils.py @@ -0,0 +1,6 @@ +from litellm.llms.base_llm.chat.transformation import BaseLLMException + + +class CometAPIException(BaseLLMException): + """CometAPI exception handling class""" + pass diff --git a/litellm/main.py b/litellm/main.py index 6bedf8f7ea5..8f9eaf621ab 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1883,6 +1883,45 @@ def completion( # type: ignore # noqa: PLR0915 encoding=encoding, stream=stream, ) + elif custom_llm_provider == "cometapi": + api_key = ( + api_key + or litellm.cometapi_key + or get_secret_str("COMETAPI_KEY") + or litellm.api_key + ) + + api_base = ( + api_base + or litellm.api_base + or get_secret_str("COMETAPI_API_BASE") + or "https://api.cometapi.com/v1" + ) + + ## COMPLETION CALL + response = base_llm_http_handler.completion( + model=model, + messages=messages, + headers=headers, + model_response=model_response, + api_key=api_key, + api_base=api_base, + acompletion=acompletion, + logging_obj=logging, + optional_params=optional_params, + litellm_params=litellm_params, + timeout=timeout, + client=client, + custom_llm_provider=custom_llm_provider, + encoding=encoding, + stream=stream, + provider_config=provider_config, + ) + + ## LOGGING + logging.post_call( + input=messages, api_key=api_key, original_response=response + ) elif ( model in litellm.open_ai_chat_completion_models or custom_llm_provider == "custom_openai" diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 75c7d28460b..438bfb175b3 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2329,6 +2329,7 @@ class LlmProviders(str, Enum): PG_VECTOR = "pg_vector" HYPERBOLIC = "hyperbolic" RECRAFT = "recraft" + COMETAPI = "cometapi" OCI = "oci" AUTO_ROUTER = "auto_router" DOTPROMPT = "dotprompt" diff --git a/litellm/utils.py b/litellm/utils.py index 64d5f04a971..79ee94d5ad9 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -6843,6 +6843,8 @@ class ProviderConfigManager: return litellm.TogetherAIConfig() elif litellm.LlmProviders.OPENROUTER == provider: return litellm.OpenrouterConfig() + elif litellm.LlmProviders.COMETAPI == provider: + return litellm.CometAPIConfig() elif litellm.LlmProviders.DATAROBOT == provider: return litellm.DataRobotConfig() elif litellm.LlmProviders.GEMINI == provider: diff --git a/tests/test_litellm/llms/cometapi/chat/test_cometapi_chat_transformation.py b/tests/test_litellm/llms/cometapi/chat/test_cometapi_chat_transformation.py new file mode 100644 index 00000000000..c7723fa4142 --- /dev/null +++ b/tests/test_litellm/llms/cometapi/chat/test_cometapi_chat_transformation.py @@ -0,0 +1,318 @@ +""" +Unit tests for CometAPI Chat Configuration + +Tests the CometAPIChatConfig class methods using mocks +""" + +import os +import sys + +import pytest + +sys.path.insert( + 0, os.path.abspath("../../../../..") +) # Adds the parent directory to the system path + +from litellm.llms.cometapi.chat.transformation import ( + CometAPIChatCompletionStreamingHandler, + CometAPIConfig, +) +from litellm.llms.cometapi.common_utils import CometAPIException + + +class TestCometAPIChatCompletionStreamingHandler: + def test_chunk_parser_successful(self): + handler = CometAPIChatCompletionStreamingHandler( + streaming_response=None, sync_stream=True + ) + + # Test input chunk + chunk = { + "id": "test_id", + "created": 1234567890, + "model": "gpt-3.5-turbo", + "usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}, + "choices": [ + {"delta": {"content": "test content", "reasoning": "test reasoning"}} + ], + } + + # Parse chunk + result = handler.chunk_parser(chunk) + + # Verify response + assert result.id == "test_id" + assert result.object == "chat.completion.chunk" + assert result.created == 1234567890 + assert result.model == "gpt-3.5-turbo" + assert result.usage.prompt_tokens == chunk["usage"]["prompt_tokens"] + assert result.usage.completion_tokens == chunk["usage"]["completion_tokens"] + assert result.usage.total_tokens == chunk["usage"]["total_tokens"] + assert len(result.choices) == 1 + assert result.choices[0]["delta"]["reasoning_content"] == "test reasoning" + + def test_chunk_parser_error_response(self): + handler = CometAPIChatCompletionStreamingHandler( + streaming_response=None, sync_stream=True + ) + + # Test error chunk + error_chunk = { + "error": { + "message": "test error", + "code": 400, + } + } + + # Verify error handling + with pytest.raises(CometAPIException) as exc_info: + handler.chunk_parser(error_chunk) + + assert "CometAPI Error: test error" in str(exc_info.value) + assert exc_info.value.status_code == 400 + + def test_chunk_parser_key_error(self): + handler = CometAPIChatCompletionStreamingHandler( + streaming_response=None, sync_stream=True + ) + + # Test invalid chunk missing required fields + invalid_chunk = {"incomplete": "data"} + + # Verify KeyError handling + with pytest.raises(CometAPIException) as exc_info: + handler.chunk_parser(invalid_chunk) + + assert "KeyError" in str(exc_info.value) + assert exc_info.value.status_code == 400 + + +class TestCometAPIConfig: + def test_transform_request_basic(self): + """Test basic request transformation""" + config = CometAPIConfig() + + transformed_request = config.transform_request( + model="cometapi/gpt-3.5-turbo", + messages=[ + {"role": "user", "content": "Hello, world!"} + ], + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert transformed_request["model"] == "cometapi/gpt-3.5-turbo" + assert transformed_request["messages"] == [ + {"role": "user", "content": "Hello, world!"} + ] + + def test_transform_request_with_extra_body(self): + """Test request transformation with extra_body parameters""" + config = CometAPIConfig() + + transformed_request = config.transform_request( + model="cometapi/gpt-4", + messages=[{"role": "user", "content": "Hello, world!"}], + optional_params={"extra_body": {"custom_param": "custom_value"}}, + litellm_params={}, + headers={}, + ) + + # Validate that extra_body parameters are merged into the request + assert transformed_request["custom_param"] == "custom_value" + assert transformed_request["messages"] == [ + {"role": "user", "content": "Hello, world!"} + ] + + def test_cache_control_flag_removal(self): + """Test cache control flag removal from messages""" + config = CometAPIConfig() + + transformed_request = config.transform_request( + model="cometapi/gpt-3.5-turbo", + messages=[ + { + "role": "user", + "content": "Hello, world!", + "cache_control": {"type": "ephemeral"}, + } + ], + optional_params={}, + litellm_params={}, + headers={}, + ) + + # CometAPI should remove cache_control flags by default + assert transformed_request["messages"][0].get("cache_control") is None + + def test_map_openai_params(self): + """Test OpenAI parameter mapping""" + config = CometAPIConfig() + + non_default_params = { + "temperature": 0.7, + "max_tokens": 100, + "top_p": 0.9, + } + + mapped_params = config.map_openai_params( + non_default_params=non_default_params, + optional_params={}, + model="cometapi/gpt-3.5-turbo", + drop_params=False, + ) + + assert mapped_params["temperature"] == 0.7 + assert mapped_params["max_tokens"] == 100 + assert mapped_params["top_p"] == 0.9 + + def test_get_error_class(self): + """Test error class creation""" + config = CometAPIConfig() + + error = config.get_error_class( + error_message="Test error", + status_code=400, + headers={"Content-Type": "application/json"} + ) + + assert isinstance(error, CometAPIException) + assert error.message == "Test error" + assert error.status_code == 400 + + +# Integration test example (requires real API key) +@pytest.mark.skip(reason="Skipping integration test") +def test_cometapi_integration(): + """ + Integration test - requires real API key + Run with: pytest -k test_cometapi_integration -s + """ + import os + from litellm import completion + + # Try to get API key from multiple environment variables + api_key = ( + os.getenv("COMETAPI_API_KEY") + or os.getenv("COMETAPI_KEY") + or os.getenv("COMET_API_KEY") + ) + + if not api_key: + pytest.skip("COMETAPI_API_KEY not set - skipping integration test") + + response = completion( + model="cometapi/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Say hello in one word"}], + api_key=api_key, + max_tokens=10, + temperature=0.7 + ) + + # Verify response structure + assert response.choices[0].message.content + assert len(response.choices[0].message.content.strip()) > 0 + assert response.model + assert response.usage + assert response.usage.total_tokens > 0 + + +def test_cometapi_streaming_integration(): + """ + Integration test for streaming - requires real API key + Run with: pytest -k test_cometapi_streaming_integration -s + """ + import os + from litellm import completion + + # Try to get API key from multiple environment variables + api_key = ( + os.getenv("COMETAPI_API_KEY") + or os.getenv("COMETAPI_KEY") + or os.getenv("COMET_API_KEY") + ) + + if not api_key: + pytest.skip("COMETAPI_API_KEY not set - skipping streaming integration test") + + try: + print(f"🔍 Testing streaming with API key: {api_key[:6]}...{api_key[-4:]} (length: {len(api_key)})") + print(f"🔍 API base URL: {os.getenv('COMETAPI_API_BASE', 'default')}") + + # test streaming API call + response = completion( + model="cometapi/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Count from 1 to 5"}], + api_key=api_key, + max_tokens=50, + stream=True + ) + + # collect streaming response + chunks = [] + content_parts = [] + + for chunk in response: + chunks.append(chunk) + if chunk.choices[0].delta.content: + content_parts.append(chunk.choices[0].delta.content) + + # Verify we received at least one chunk and content + assert len(chunks) > 0, "Should receive at least one chunk" + assert len(content_parts) > 0, "Should receive content in chunks" + + full_content = "".join(content_parts) + assert len(full_content.strip()) > 0, "Should have non-empty content" + + print(f"✅ Received {len(chunks)} chunks") + print(f"✅ Full content: {full_content}") + + except Exception as e: + print(f"❌ Streaming integration test error details:") + print(f" Error type: {type(e).__name__}") + print(f" Error message: {str(e)}") + if hasattr(e, 'status_code'): + print(f" Status code: {e.status_code}") + if hasattr(e, 'response'): + print(f" Response: {e.response}") + + # Re-raise with more context for pytest + pytest.fail(f"Streaming integration test failed: {type(e).__name__}: {str(e)}") +def test_cometapi_with_custom_base_url(): + """ + Test CometAPI with custom base URL + """ + import os + from litellm import completion + + api_key = ( + os.getenv("COMETAPI_API_KEY") + or os.getenv("COMETAPI_KEY") + or os.getenv("COMET_API_KEY") + ) + + custom_base_url = os.getenv("COMETAPI_API_BASE", "https://api.cometapi.com/v1") + + if not api_key: + pytest.skip("COMETAPI_API_KEY not set - skipping custom base URL test") + + try: + response = completion( + model="cometapi/gpt-3.5-turbo", + messages=[{"role": "user", "content": "Hello"}], + api_key=api_key, + api_base=custom_base_url, + max_tokens=5 + ) + + assert response.choices[0].message.content + print(f"✅ Custom base URL test passed: {response.choices[0].message.content}") + + except Exception as e: + pytest.fail(f"Custom base URL test failed: {str(e)}") + + +if __name__ == "__main__": + # Quick test runner + pytest.main([__file__, "-v"]) \ No newline at end of file From ea0f76812276908ec0a56babcecbc2542409ea45 Mon Sep 17 00:00:00 2001 From: TensorNull Date: Sat, 9 Aug 2025 10:59:23 +0800 Subject: [PATCH 10/66] fix: specify type for extra_body in CometAPIConfig --- litellm/llms/cometapi/chat/transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/cometapi/chat/transformation.py b/litellm/llms/cometapi/chat/transformation.py index 391f9626d37..fedb8f61e5b 100644 --- a/litellm/llms/cometapi/chat/transformation.py +++ b/litellm/llms/cometapi/chat/transformation.py @@ -41,7 +41,7 @@ class CometAPIConfig(OpenAIGPTConfig): ) # CometAPI-specific parameters (if any) - extra_body = {} + extra_body: dict[str, Any] = {} # TODO: Add CometAPI-specific parameter handling here # Example: # custom_param = non_default_params.pop("custom_param", None) From 8b602f9507d38a6524f9e50fa550425f43bd6e2a Mon Sep 17 00:00:00 2001 From: TensorNull Date: Tue, 12 Aug 2025 17:18:33 +0800 Subject: [PATCH 11/66] [Feat] - Add CometAPI documentation with authentication, usage examples, and error handling --- docs/my-website/docs/providers/cometapi.md | 147 +++++++++++++++++++++ 1 file changed, 147 insertions(+) create mode 100644 docs/my-website/docs/providers/cometapi.md diff --git a/docs/my-website/docs/providers/cometapi.md b/docs/my-website/docs/providers/cometapi.md new file mode 100644 index 00000000000..cb05e048121 --- /dev/null +++ b/docs/my-website/docs/providers/cometapi.md @@ -0,0 +1,147 @@ +# CometAPI +LiteLLM supports all AI models from [CometAPI](https://www.cometapi.com/). CometAPI provides access to 500+ AI models through a unified API interface, including cutting-edge models like GPT-5, Claude Opus 4.1, and various other state-of-the-art language models. + +## Authentication + +To use CometAPI models, you need to obtain an API key from [CometAPI Token Console](https://api.cometapi.com/console/token). CometAPI offers free tokens for new users - you can get your free API key instantly by registering. + +## Usage + +Set your CometAPI key as an environment variable and use the completion function: + +```python +import os +from litellm import completion + +# Set API key +os.environ["COMETAPI_KEY"] = "your_comet_api_key_here" + +# Define messages +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Method 1: Using environment variable (recommended) +response = completion( + model="cometapi/gpt-5", + messages=messages +) + +print(response.choices[0].message.content) +``` + +### Alternative Usage - Explicit API Key + +You can also pass the API key explicitly: + +```python +import os +from litellm import completion + +# Define messages +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Method 2: Explicitly passing API key +response = completion( + model="cometapi/gpt-4o", + messages=messages, + api_key="your_comet_api_key_here" +) + +print(response.choices[0].message.content) +``` + +## Usage - Streaming + +Just set `stream=True` when calling completion: + +```python +import os +from litellm import completion + +os.environ["COMETAPI_KEY"] = "your_comet_api_key_here" + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +response = completion( + model="cometapi/gpt-5", + messages=messages, + stream=True +) + +for chunk in response: + print(chunk.choices[0].delta.content or "", end="") +``` + +## Usage - Async Streaming + +For async streaming, use `acompletion`: + +```python +from litellm import acompletion +import asyncio, os, traceback + +async def completion_call(): + try: + os.environ["COMETAPI_KEY"] = "your_comet_api_key_here" + + print("test acompletion + streaming") + response = await acompletion( + model="cometapi/chatgpt-4o-latest", + messages=[{"content": "Hello, how are you?", "role": "user"}], + stream=True + ) + print(f"response: {response}") + async for chunk in response: + print(chunk) + except: + print(f"error occurred: {traceback.format_exc()}") + pass + +# Run the async function +await completion_call() +``` + +## CometAPI Models + +CometAPI offers access to 500+ AI models through a unified API. Some popular models include: + +| Model Name | Function Call | +|------------|---------------| +| cometapi/gpt-5 | `completion('cometapi/gpt-5', messages)` | +| cometapi/gpt-5-mini | `completion('cometapi/gpt-5-mini', messages)` | +| cometapi/gpt-5-nano | `completion('cometapi/gpt-5-nano', messages)` | +| cometapi/claude-opus-4.1 | `completion('cometapi/claude-opus-4.1', messages)` | +| cometapi/o4-mini-deep-research | `completion('cometapi/o4-mini-deep-research', messages)` | +| cometapi/o3-deep-research | `completion('cometapi/o3-deep-research', messages)` | +| cometapi/gpt-oss-20b | `completion('cometapi/gpt-oss-20b', messages)` | +| cometapi/gpt-oss-120b | `completion('cometapi/gpt-oss-120b', messages)` | +| cometapi/chatgpt-4o-latest | `completion('cometapi/chatgpt-4o-latest', messages)` | + +For a complete list of available models, visit the [CometAPI Models page](https://www.cometapi.com/model/). + +## Environment Variables + +| Variable | Description | Required | +|----------|-------------|----------| +| `COMETAPI_KEY` | Your CometAPI API key | Yes | + +## Error Handling + +```python +import os +from litellm import completion + +try: + os.environ["COMETAPI_KEY"] = "your_comet_api_key_here" + + messages = [{"content": "Hello, how are you?", "role": "user"}] + + response = completion( + model="cometapi/gpt-5", + messages=messages + ) + + print(response.choices[0].message.content) + +except Exception as e: + print(f"Error: {e}") +``` From fa81c20df682a6ec1ac0f3aab5366ffd1e95b409 Mon Sep 17 00:00:00 2001 From: TensorNull Date: Tue, 12 Aug 2025 17:28:10 +0800 Subject: [PATCH 12/66] fix: Remove outdated models from the model list in the CometAPI document --- docs/my-website/docs/providers/cometapi.md | 3 --- 1 file changed, 3 deletions(-) diff --git a/docs/my-website/docs/providers/cometapi.md b/docs/my-website/docs/providers/cometapi.md index cb05e048121..1245bacfad4 100644 --- a/docs/my-website/docs/providers/cometapi.md +++ b/docs/my-website/docs/providers/cometapi.md @@ -109,9 +109,6 @@ CometAPI offers access to 500+ AI models through a unified API. Some popular mod | cometapi/gpt-5 | `completion('cometapi/gpt-5', messages)` | | cometapi/gpt-5-mini | `completion('cometapi/gpt-5-mini', messages)` | | cometapi/gpt-5-nano | `completion('cometapi/gpt-5-nano', messages)` | -| cometapi/claude-opus-4.1 | `completion('cometapi/claude-opus-4.1', messages)` | -| cometapi/o4-mini-deep-research | `completion('cometapi/o4-mini-deep-research', messages)` | -| cometapi/o3-deep-research | `completion('cometapi/o3-deep-research', messages)` | | cometapi/gpt-oss-20b | `completion('cometapi/gpt-oss-20b', messages)` | | cometapi/gpt-oss-120b | `completion('cometapi/gpt-oss-120b', messages)` | | cometapi/chatgpt-4o-latest | `completion('cometapi/chatgpt-4o-latest', messages)` | From d898f9e0ddc394582c445dd4eec2bf0166c92410 Mon Sep 17 00:00:00 2001 From: Edward Samuel Pasaribu Date: Tue, 12 Aug 2025 18:51:56 +0800 Subject: [PATCH 13/66] Add openrouter gpt-5 family models pricing --- model_prices_and_context_window.json | 65 ++++++++++++++++++++++++++++ 1 file changed, 65 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 28dec7cce90..4db8f43d0c7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11688,6 +11688,71 @@ "mode": "chat", "supports_tool_choice": true }, + "openrouter/openai/gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "openrouter/openai/gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_tool_choice": true, + "supports_reasoning": true + }, + "openrouter/openai/gpt-5-chat": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_tool_choice": false, + "supports_reasoning": true + }, "openrouter/openai/gpt-oss-20b": { "max_tokens": 32768, "max_input_tokens": 131072, From 36f160b582e8b9f081697e0d3477742e4c048163 Mon Sep 17 00:00:00 2001 From: Edward Samuel Pasaribu Date: Tue, 12 Aug 2025 18:54:06 +0800 Subject: [PATCH 14/66] Update openrouter/openai/gpt-5-mini pricing --- model_prices_and_context_window.json | 8 -------- 1 file changed, 8 deletions(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 4db8f43d0c7..caa98164f00 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11704,15 +11704,7 @@ "supported_output_modalities": [ "text" ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, "supports_tool_choice": true, - "supports_native_streaming": true, "supports_reasoning": true }, "openrouter/openai/gpt-5-nano": { From 902e5e71e115322d8d8c38ca69dc2cd28b207272 Mon Sep 17 00:00:00 2001 From: sudu Date: Wed, 13 Aug 2025 12:40:34 +0800 Subject: [PATCH 15/66] Update simple_shuffle.py, choose weights by 'weight', 'rpm', 'tpm' in one loop --- litellm/router_strategy/simple_shuffle.py | 68 ++++++----------------- 1 file changed, 17 insertions(+), 51 deletions(-) diff --git a/litellm/router_strategy/simple_shuffle.py b/litellm/router_strategy/simple_shuffle.py index da24c02f2e3..e35ecfdcd22 100644 --- a/litellm/router_strategy/simple_shuffle.py +++ b/litellm/router_strategy/simple_shuffle.py @@ -39,57 +39,23 @@ def simple_shuffle( Dict: A single healthy deployment """ - ############## Check if 'weight' param set for a weighted pick ################# - weight = healthy_deployments[0].get("litellm_params").get("weight", None) - if weight is not None: - # use weight-random pick if rpms provided - weights = [m["litellm_params"].get("weight", 0) for m in healthy_deployments] - verbose_router_logger.debug(f"\nweight {weights}") - total_weight = sum(weights) - weights = [weight / total_weight for weight in weights] - verbose_router_logger.debug(f"\n weights {weights}") - # Perform weighted random pick - selected_index = random.choices(range(len(weights)), weights=weights)[0] - verbose_router_logger.debug(f"\n selected index, {selected_index}") - deployment = healthy_deployments[selected_index] - verbose_router_logger.info( - f"get_available_deployment for model: {model}, Selected deployment: {llm_router_instance.print_deployment(deployment) or deployment[0]} for model: {model}" - ) - return deployment or deployment[0] - ############## Check if we can do a RPM/TPM based weighted pick ################# - rpm = healthy_deployments[0].get("litellm_params").get("rpm", None) - if rpm is not None: - # use weight-random pick if rpms provided - rpms = [m["litellm_params"].get("rpm", 0) for m in healthy_deployments] - verbose_router_logger.debug(f"\nrpms {rpms}") - total_rpm = sum(rpms) - weights = [rpm / total_rpm for rpm in rpms] - verbose_router_logger.debug(f"\n weights {weights}") - # Perform weighted random pick - selected_index = random.choices(range(len(rpms)), weights=weights)[0] - verbose_router_logger.debug(f"\n selected index, {selected_index}") - deployment = healthy_deployments[selected_index] - verbose_router_logger.info( - f"get_available_deployment for model: {model}, Selected deployment: {llm_router_instance.print_deployment(deployment) or deployment[0]} for model: {model}" - ) - return deployment or deployment[0] - ############## Check if we can do a RPM/TPM based weighted pick ################# - tpm = healthy_deployments[0].get("litellm_params").get("tpm", None) - if tpm is not None: - # use weight-random pick if rpms provided - tpms = [m["litellm_params"].get("tpm", 0) for m in healthy_deployments] - verbose_router_logger.debug(f"\ntpms {tpms}") - total_tpm = sum(tpms) - weights = [tpm / total_tpm for tpm in tpms] - verbose_router_logger.debug(f"\n weights {weights}") - # Perform weighted random pick - selected_index = random.choices(range(len(tpms)), weights=weights)[0] - verbose_router_logger.debug(f"\n selected index, {selected_index}") - deployment = healthy_deployments[selected_index] - verbose_router_logger.info( - f"get_available_deployment for model: {model}, Selected deployment: {llm_router_instance.print_deployment(deployment) or deployment[0]} for model: {model}" - ) - return deployment or deployment[0] + ############## Check if 'weight' or 'rpm' or 'tpm' param set for a weighted pick ################# + for weight_by in ["weight", "rpm", "tpm"]: + weight = healthy_deployments[0].get("litellm_params").get(weight_by, None) + if weight is not None: + weights = [m["litellm_params"].get(weight_by, 0) for m in healthy_deployments] + verbose_router_logger.debug(f"\nweight {weights}") + total_weight = sum(weights) + weights = [weight / total_weight for weight in weights] + verbose_router_logger.debug(f"\n weights {weights} by {weight_by}") + # Perform weighted random pick + selected_index = random.choices(range(len(weights)), weights=weights)[0] + verbose_router_logger.debug(f"\n selected index, {selected_index}") + deployment = healthy_deployments[selected_index] + verbose_router_logger.info( + f"get_available_deployment for model: {model}, Selected deployment: {llm_router_instance.print_deployment(deployment) or deployment[0]} for model: {model}" + ) + return deployment or deployment[0] ############## No RPM/TPM passed, we do a random pick ################# item = random.choice(healthy_deployments) From 3d0f417829a0f113900c31e0fa6ca8ac185f267f Mon Sep 17 00:00:00 2001 From: iamkankute Date: Wed, 13 Aug 2025 11:13:56 +0530 Subject: [PATCH 16/66] fix: remove incorrect web search support for azure/gpt-4.1 family --- model_prices_and_context_window.json | 28 ++++------------------------ 1 file changed, 4 insertions(+), 24 deletions(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 28dec7cce90..d866d14617c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -2658,12 +2658,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.03, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.05 - } + "supports_web_search": false }, "azure/gpt-4.1-2025-04-14": { "max_tokens": 32768, @@ -2696,12 +2691,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.03, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.05 - } + "supports_web_search": false }, "azure/gpt-4.1-mini": { "max_tokens": 32768, @@ -2734,12 +2724,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.025, - "search_context_size_medium": 0.0275, - "search_context_size_high": 0.03 - } + "supports_web_search": false }, "azure/gpt-4.1-mini-2025-04-14": { "max_tokens": 32768, @@ -2772,12 +2757,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.025, - "search_context_size_medium": 0.0275, - "search_context_size_high": 0.03 - } + "supports_web_search": false }, "azure/gpt-4.1-nano": { "max_tokens": 32768, From 86d8fcf5fda939bcb34b69c770948f317b024e4f Mon Sep 17 00:00:00 2001 From: Yuki Imajuku Date: Wed, 13 Aug 2025 14:56:32 +0900 Subject: [PATCH 17/66] update model prices and context window --- ...odel_prices_and_context_window_backup.json | 34 +++++++++++++++++++ model_prices_and_context_window.json | 34 +++++++++++++++++++ 2 files changed, 68 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1001aca9c09..1abc6519603 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11447,6 +11447,40 @@ "supports_assistant_prefill": true, "supports_tool_choice": true }, + "openrouter/anthropic/claude-opus-4": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "input_cost_per_image": 0.0048, + "litellm_provider": "openrouter", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, + "openrouter/anthropic/claude-opus-4.1": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "input_cost_per_image": 0.0048, + "litellm_provider": "openrouter", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, "openrouter/mistralai/mistral-large": { "max_tokens": 32000, "input_cost_per_token": 8e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1001aca9c09..1abc6519603 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11447,6 +11447,40 @@ "supports_assistant_prefill": true, "supports_tool_choice": true }, + "openrouter/anthropic/claude-opus-4": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "input_cost_per_image": 0.0048, + "litellm_provider": "openrouter", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, + "openrouter/anthropic/claude-opus-4.1": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "input_cost_per_image": 0.0048, + "litellm_provider": "openrouter", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, "openrouter/mistralai/mistral-large": { "max_tokens": 32000, "input_cost_per_token": 8e-06, From 38635b9e070b627ec222cdf6be72a353d8dfc146 Mon Sep 17 00:00:00 2001 From: iamkankute Date: Wed, 13 Aug 2025 11:41:50 +0530 Subject: [PATCH 18/66] fix: remove incorrect web search support for azure/gpt-4.1 family --- ...odel_prices_and_context_window_backup.json | 32 +++---------------- 1 file changed, 4 insertions(+), 28 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 28dec7cce90..764f71d1334 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -2658,12 +2658,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.03, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.05 - } + "supports_web_search": false }, "azure/gpt-4.1-2025-04-14": { "max_tokens": 32768, @@ -2696,12 +2691,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.03, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.05 - } + "supports_web_search": false }, "azure/gpt-4.1-mini": { "max_tokens": 32768, @@ -2734,12 +2724,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.025, - "search_context_size_medium": 0.0275, - "search_context_size_high": 0.03 - } + "supports_web_search": false }, "azure/gpt-4.1-mini-2025-04-14": { "max_tokens": 32768, @@ -2772,12 +2757,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_native_streaming": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.025, - "search_context_size_medium": 0.0275, - "search_context_size_high": 0.03 - } + "supports_web_search": false }, "azure/gpt-4.1-nano": { "max_tokens": 32768, @@ -12588,8 +12568,6 @@ "output_cost_per_token": 3e-07, "litellm_provider": "bedrock_converse", "mode": "chat", - "supports_function_calling": true, - "supports_vision": true, "supports_response_schema": true, "supports_tool_choice": true, "supports_reasoning": true @@ -12602,8 +12580,6 @@ "output_cost_per_token": 6e-07, "litellm_provider": "bedrock_converse", "mode": "chat", - "supports_function_calling": true, - "supports_vision": true, "supports_response_schema": true, "supports_tool_choice": true, "supports_reasoning": true From f2b0b6124c070d51e02e092f45dfc639d78ff4d4 Mon Sep 17 00:00:00 2001 From: nielsbosma Date: Wed, 13 Aug 2025 12:18:17 +0200 Subject: [PATCH 19/66] feat(logging): add support for custom span names in Braintrust logging --- .../docs/observability/braintrust.md | 19 ++++++++++++++++--- litellm/integrations/braintrust_logging.py | 10 ++++++++-- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/docs/my-website/docs/observability/braintrust.md b/docs/my-website/docs/observability/braintrust.md index eb26680b18a..e6b4fe769bc 100644 --- a/docs/my-website/docs/observability/braintrust.md +++ b/docs/my-website/docs/observability/braintrust.md @@ -71,6 +71,10 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ It is recommended that you include the `project_id` or `project_name` to ensure your traces are being written out to the correct Braintrust project. +### Custom Span Names + +You can customize the span name in Braintrust logging by passing `span_name` in the metadata. By default, the span name is set to "Chat Completion". + @@ -84,7 +88,9 @@ response = litellm.completion( "project_id": "1234", # passing project_name will try to find a project with that name, or create one if it doesn't exist # if both project_id and project_name are passed, project_id will be used - # "project_name": "my-special-project" + # "project_name": "my-special-project", + # custom span name for this operation (default: "Chat Completion") + "span_name": "User Greeting Handler" } ) ``` @@ -99,6 +105,7 @@ response = litellm.completion( ], metadata={ "project_id": "1234", + "span_name": "Custom Operation", "item1": "an item", "item2": "another item" } @@ -121,7 +128,8 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ { "role": "user", "content": "What time is it now? Use your tool"} ], "metadata": { - "project_id": "my-special-project" + "project_id": "my-special-project", + "span_name": "Tool Usage Request" } }' ``` @@ -146,7 +154,8 @@ response = client.chat.completions.create( ], extra_body={ # pass in any provider-specific param, if not supported by openai, https://docs.litellm.ai/docs/completion/input#provider-specific-params "metadata": { # 👈 use for logging additional params (e.g. to braintrust) - "project_id": "my-special-project" + "project_id": "my-special-project", + "span_name": "Poetry Generation" } } ) @@ -168,3 +177,7 @@ Here's everything you can pass in metadata for a braintrust request `braintrust_*` - If you are adding metadata from _proxy request headers_, any metadata field starting with `braintrust_` will be passed as metadata to the logging request. If you are using the SDK, just pass your metadata like normal (e.g., `metadata={"project_name": "my-test-project", "item1": "an item", "item2": "another item"}`) `project_id` - Set the project id for a braintrust call. Default is `litellm`. + +`project_name` - Set the project name for a braintrust call. Will try to find a project with that name, or create one if it doesn't exist. If both `project_id` and `project_name` are passed, `project_id` will be used. + +`span_name` - Set a custom span name for the operation. Default is `"Chat Completion"`. Use this to provide more descriptive names for different types of operations in your application (e.g., "User Query", "Document Summary", "Code Generation"). diff --git a/litellm/integrations/braintrust_logging.py b/litellm/integrations/braintrust_logging.py index 8149a6131e8..39b5334e93f 100644 --- a/litellm/integrations/braintrust_logging.py +++ b/litellm/integrations/braintrust_logging.py @@ -274,12 +274,15 @@ class BraintrustLogger(CustomLogger): "end": end_time.timestamp(), } + # Allow metadata override for span name + span_name = metadata.get("span_name", "Chat Completion") + request_data = { "id": litellm_call_id, "input": prompt["messages"], "metadata": clean_metadata, "tags": tags, - "span_attributes": {"name": "Chat Completion", "type": "llm"}, + "span_attributes": {"name": span_name, "type": "llm"}, } if choices is not None: request_data["output"] = [choice.dict() for choice in choices] @@ -426,13 +429,16 @@ class BraintrustLogger(CustomLogger): - api_call_start_time.timestamp() ) + # Allow metadata override for span name + span_name = metadata.get("span_name", "Chat Completion") + request_data = { "id": litellm_call_id, "input": prompt["messages"], "output": output, "metadata": clean_metadata, "tags": tags, - "span_attributes": {"name": "Chat Completion", "type": "llm"}, + "span_attributes": {"name": span_name, "type": "llm"}, } if choices is not None: request_data["output"] = [choice.dict() for choice in choices] From fe54da79a135c3fc31d7d87291f83db0cb35a645 Mon Sep 17 00:00:00 2001 From: nielsbosma Date: Wed, 13 Aug 2025 12:26:14 +0200 Subject: [PATCH 20/66] test(braintrust-logging): add span_name tests for events Add tests to verify custom and default span_name in BraintrustLogger, including async, metadata merging, and span name behavior. --- .../integrations/test_braintrust_logging.py | 246 +++++++++++++++++- .../integrations/test_braintrust_span_name.py | 199 ++++++++++++++ 2 files changed, 443 insertions(+), 2 deletions(-) create mode 100644 tests/test_litellm/integrations/test_braintrust_span_name.py diff --git a/tests/test_litellm/integrations/test_braintrust_logging.py b/tests/test_litellm/integrations/test_braintrust_logging.py index 5ae40e82760..cca13b4e9e6 100644 --- a/tests/test_litellm/integrations/test_braintrust_logging.py +++ b/tests/test_litellm/integrations/test_braintrust_logging.py @@ -1,7 +1,9 @@ import os import unittest -from unittest.mock import patch +from datetime import datetime +from unittest.mock import MagicMock, Mock, patch +import litellm from litellm.integrations.braintrust_logging import BraintrustLogger class TestBraintrustLogger(unittest.TestCase): @@ -40,4 +42,244 @@ class TestBraintrustLogger(unittest.TestCase): with patch.dict(os.environ, {}, clear=True): with self.assertRaises(Exception) as context: BraintrustLogger(api_key=None) - self.assertIn("Missing keys=['BRAINTRUST_API_KEY']", str(context.exception)) \ No newline at end of file + self.assertIn("Missing keys=['BRAINTRUST_API_KEY']", str(context.exception)) + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_log_success_event_with_default_span_name(self, mock_http_handler): + """Test log_success_event uses default span name when not provided.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler.post.return_value = mock_response + + # Create a mock response object + message_mock = Mock() + message_mock.json = Mock(return_value={"content": "test"}) + + choice_mock = Mock() + choice_mock.message = message_mock + choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + + response_obj = Mock(spec=litellm.ModelResponse) + response_obj.choices = [choice_mock] + # Mock the __getitem__ to support response_obj["choices"] + response_obj.__getitem__ = Mock(return_value=[choice_mock]) + response_obj.usage = litellm.Usage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30 + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_log_success_event_with_custom_span_name(self, mock_http_handler): + """Test log_success_event uses custom span name when provided.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler.post.return_value = mock_response + + # Create a mock response object + message_mock = Mock() + message_mock.json = Mock(return_value={"content": "test"}) + + choice_mock = Mock() + choice_mock.message = message_mock + choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + + response_obj = Mock(spec=litellm.ModelResponse) + response_obj.choices = [choice_mock] + response_obj.__getitem__ = Mock(return_value=[choice_mock]) + response_obj.usage = litellm.Usage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30 + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {"span_name": "Custom Operation"}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Custom Operation') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') + async def test_async_log_success_event_with_default_span_name(self, mock_http_handler): + """Test async_log_success_event uses default span name when not provided.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler.post = MagicMock(return_value=mock_response) + + # Create a mock response object + message_mock = Mock() + message_mock.json = Mock(return_value={"content": "test"}) + + choice_mock = Mock() + choice_mock.message = message_mock + choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + + response_obj = Mock(spec=litellm.ModelResponse) + response_obj.choices = [choice_mock] + response_obj.__getitem__ = Mock(return_value=[choice_mock]) + response_obj.usage = litellm.Usage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30 + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + await logger.async_log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') + async def test_async_log_success_event_with_custom_span_name(self, mock_http_handler): + """Test async_log_success_event uses custom span name when provided.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler.post = MagicMock(return_value=mock_response) + + # Create a mock response object + message_mock = Mock() + message_mock.json = Mock(return_value={"content": "test"}) + + choice_mock = Mock() + choice_mock.message = message_mock + choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + + response_obj = Mock(spec=litellm.ModelResponse) + response_obj.choices = [choice_mock] + response_obj.__getitem__ = Mock(return_value=[choice_mock]) + response_obj.usage = litellm.Usage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30 + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {"span_name": "Async Custom Operation"}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + await logger.async_log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Async Custom Operation') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_span_name_with_multiple_metadata_fields(self, mock_http_handler): + """Test that span_name works correctly alongside other metadata fields.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler.post.return_value = mock_response + + # Create a mock response object + message_mock = Mock() + message_mock.json = Mock(return_value={"content": "test"}) + + choice_mock = Mock() + choice_mock.message = message_mock + choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + + response_obj = Mock(spec=litellm.ModelResponse) + response_obj.choices = [choice_mock] + response_obj.__getitem__ = Mock(return_value=[choice_mock]) + response_obj.usage = litellm.Usage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30 + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": { + "metadata": { + "span_name": "Multi Metadata Test", + "project_id": "custom-project", + "user_id": "user123", + "session_id": "session456" + } + }, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + + # Check span name + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Multi Metadata Test') + + # Check that other metadata is preserved + event_metadata = json_data['events'][0]['metadata'] + self.assertEqual(event_metadata['user_id'], 'user123') + self.assertEqual(event_metadata['session_id'], 'session456') \ No newline at end of file diff --git a/tests/test_litellm/integrations/test_braintrust_span_name.py b/tests/test_litellm/integrations/test_braintrust_span_name.py new file mode 100644 index 00000000000..d3d98ea70af --- /dev/null +++ b/tests/test_litellm/integrations/test_braintrust_span_name.py @@ -0,0 +1,199 @@ +import json +import os +import unittest +from datetime import datetime +from unittest.mock import MagicMock, Mock, patch + +import litellm +from litellm.integrations.braintrust_logging import BraintrustLogger + + +class TestBraintrustSpanName(unittest.TestCase): + """Test custom span_name functionality in Braintrust logging.""" + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_default_span_name(self, mock_http_handler): + """Test that default span name is 'Chat Completion' when not provided.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + # Mock HTTP response + mock_http_handler.post.return_value = Mock() + + # Create a properly structured mock response + response_obj = litellm.ModelResponse( + id="test-id", + object="chat.completion", + created=1234567890, + model="gpt-3.5-turbo", + choices=[{ + "index": 0, + "message": {"role": "assistant", "content": "test response"}, + "finish_reason": "stop" + }], + usage={"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30} + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_custom_span_name(self, mock_http_handler): + """Test that custom span name is used when provided in metadata.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + # Mock HTTP response + mock_http_handler.post.return_value = Mock() + + # Create a properly structured mock response + response_obj = litellm.ModelResponse( + id="test-id", + object="chat.completion", + created=1234567890, + model="gpt-3.5-turbo", + choices=[{ + "index": 0, + "message": {"role": "assistant", "content": "test response"}, + "finish_reason": "stop" + }], + usage={"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30} + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {"span_name": "Custom Operation"}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Custom Operation') + + @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') + def test_span_name_with_other_metadata(self, mock_http_handler): + """Test that span_name works alongside other metadata fields.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + # Mock HTTP response + mock_http_handler.post.return_value = Mock() + + # Create a properly structured mock response + response_obj = litellm.ModelResponse( + id="test-id", + object="chat.completion", + created=1234567890, + model="gpt-3.5-turbo", + choices=[{ + "index": 0, + "message": {"role": "assistant", "content": "test response"}, + "finish_reason": "stop" + }], + usage={"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30} + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": { + "metadata": { + "span_name": "Multi Metadata Test", + "project_id": "custom-project", + "user_id": "user123", + "session_id": "session456", + "environment": "production" + } + }, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + logger.log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + + # Check span name + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Multi Metadata Test') + + # Check that other metadata is preserved (except for filtered keys) + event_metadata = json_data['events'][0]['metadata'] + self.assertEqual(event_metadata['user_id'], 'user123') + self.assertEqual(event_metadata['session_id'], 'session456') + self.assertEqual(event_metadata['environment'], 'production') + + # Span name should be in span_attributes, not in metadata + self.assertIn('span_name', event_metadata) # span_name is also kept in metadata + + @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') + async def test_async_custom_span_name(self, mock_http_handler): + """Test async logging with custom span name.""" + # Setup + logger = BraintrustLogger(api_key="test-key") + logger.default_project_id = "test-project-id" + + # Mock async HTTP response + mock_http_handler.post = MagicMock(return_value=Mock()) + + # Create a properly structured mock response + response_obj = litellm.ModelResponse( + id="test-id", + object="chat.completion", + created=1234567890, + model="gpt-3.5-turbo", + choices=[{ + "index": 0, + "message": {"role": "assistant", "content": "test response"}, + "finish_reason": "stop" + }], + usage={"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30} + ) + + kwargs = { + "litellm_call_id": "test-call-id", + "messages": [{"role": "user", "content": "test"}], + "litellm_params": {"metadata": {"span_name": "Async Custom Operation"}}, + "model": "gpt-3.5-turbo", + "response_cost": 0.001 + } + + # Execute + await logger.async_log_success_event(kwargs, response_obj, datetime.now(), datetime.now()) + + # Verify + call_args = mock_http_handler.post.call_args + self.assertIsNotNone(call_args) + json_data = call_args.kwargs['json'] + self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Async Custom Operation') + + +if __name__ == "__main__": + unittest.main() \ No newline at end of file From 6a1f5bdc2f1be4deccc0db0280c1789014c6224c Mon Sep 17 00:00:00 2001 From: Dor Zion Date: Mon, 11 Aug 2025 12:05:28 +0300 Subject: [PATCH 21/66] Add Noma Security guardrail support --- .../docs/proxy/guardrails/noma_security.md | 299 +++++++++++ docs/my-website/sidebars.js | 1 + .../guardrail_hooks/noma/__init__.py | 36 ++ .../guardrails/guardrail_hooks/noma/noma.py | 403 ++++++++++++++ litellm/types/guardrails.py | 19 + .../guardrails/guardrail_hooks/test_noma.py | 498 ++++++++++++++++++ 6 files changed, 1256 insertions(+) create mode 100644 docs/my-website/docs/proxy/guardrails/noma_security.md create mode 100644 litellm/proxy/guardrails/guardrail_hooks/noma/__init__.py create mode 100644 litellm/proxy/guardrails/guardrail_hooks/noma/noma.py create mode 100644 tests/test_litellm/proxy/guardrails/guardrail_hooks/test_noma.py diff --git a/docs/my-website/docs/proxy/guardrails/noma_security.md b/docs/my-website/docs/proxy/guardrails/noma_security.md new file mode 100644 index 00000000000..3a50841d65e --- /dev/null +++ b/docs/my-website/docs/proxy/guardrails/noma_security.md @@ -0,0 +1,299 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Noma Security + +Use [Noma Security](https://noma.security/) to protect your LLM applications with comprehensive AI content moderation and safety guardrails. + +## Quick Start + +### 1. Define Guardrails on your LiteLLM config.yaml + +Define your guardrails under the `guardrails` section: + +```yaml showLineNumbers title="litellm config.yaml" +model_list: + - model_name: gpt-4o-mini + litellm_params: + model: openai/gpt-4o-mini + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "noma-guard" + litellm_params: + guardrail: noma + mode: "during_call" + api_key: os.environ/NOMA_API_KEY + api_base: os.environ/NOMA_API_BASE + - guardrail_name: "noma-pre-guard" + litellm_params: + guardrail: noma + mode: "pre_call" + api_key: os.environ/NOMA_API_KEY + api_base: os.environ/NOMA_API_BASE +``` + +#### Supported values for `mode` + +- `pre_call` Run **before** LLM call, on **input** +- `post_call` Run **after** LLM call, on **input & output** +- `during_call` Run **during** LLM call, on **input**. Same as `pre_call` but runs in parallel with the LLM call. Response not returned until guardrail check completes + +### 2. Start LiteLLM Gateway + +```shell +litellm --config config.yaml --detailed_debug +``` + +### 3. Test request + + + + +Expect this to fail since the request contains harmful content: + +```shell showLineNumbers title="Curl Request" +curl -i http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "gpt-4o-mini", + "messages": [ + {"role": "user", "content": "Tell me how to hack into someone's email account"} + ] + }' +``` + +Expected response on failure: + +```json +{ + "error": { + "message": "{\n \"error\": \"Request blocked by Noma guardrail\",\n \"details\": {\n \"prompt\": {\n \"harmfulContent\": {\n \"result\": true,\n \"confidence\": 0.95\n }\n }\n }\n }", + "type": "None", + "param": "None", + "code": "400" + } +} +``` + + + + + +```shell showLineNumbers title="Curl Request" +curl -i http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "gpt-4o-mini", + "messages": [ + {"role": "user", "content": "What is the capital of France?"} + ] + }' +``` + +Expected response: + +```json +{ + "id": "chatcmpl-123", + "object": "chat.completion", + "created": 1677652288, + "model": "gpt-4o-mini", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "The capital of France is Paris." + }, + "finish_reason": "stop" + } + ], + "usage": { + "prompt_tokens": 9, + "completion_tokens": 12, + "total_tokens": 21 + } +} +``` + + + + +## Supported Params + +```yaml +guardrails: + - guardrail_name: "noma-guard" + litellm_params: + guardrail: noma + mode: "pre_call" + api_key: os.environ/NOMA_API_KEY + api_base: os.environ/NOMA_API_BASE + ### OPTIONAL ### + # application_id: "my-app" + # monitor_mode: false + # block_failures: true +``` + +### Required Parameters + +- **`api_key`**: Your Noma Security API key (set as `os.environ/NOMA_API_KEY` in YAML config) + +### Optional Parameters + +- **`api_base`**: Noma API base URL (defaults to `https://api.noma.security/`) +- **`application_id`**: Your application identifier (defaults to `"litellm"`) +- **`monitor_mode`**: If `true`, logs violations without blocking (defaults to `false`) +- **`block_failures`**: If `true`, blocks requests when guardrail API failures occur (defaults to `true`) + +## Environment Variables + +You can set these environment variables instead of hardcoding values in your config: + +```shell +export NOMA_API_KEY="your-api-key-here" +export NOMA_API_BASE="https://api.noma.security/" # Optional +export NOMA_APPLICATION_ID="my-app" # Optional +export NOMA_MONITOR_MODE="false" # Optional +export NOMA_BLOCK_FAILURES="true" # Optional +``` + +## Advanced Configuration + +### Monitor Mode + +Use monitor mode to test your guardrails without blocking requests: + +```yaml +guardrails: + - guardrail_name: "noma-monitor" + litellm_params: + guardrail: noma + mode: "pre_call" + api_key: os.environ/NOMA_API_KEY + monitor_mode: true # Log violations but don't block +``` + +### Handling API Failures + +Control behavior when the Noma API is unavailable: + +```yaml +guardrails: + - guardrail_name: "noma-failopen" + litellm_params: + guardrail: noma + mode: "pre_call" + api_key: os.environ/NOMA_API_KEY + block_failures: false # Allow requests to proceed if guardrail API fails +``` + +### Multiple Guardrails + +Apply different configurations for input and output: + +```yaml +guardrails: + - guardrail_name: "noma-strict-input" + litellm_params: + guardrail: noma + mode: "pre_call" + api_key: os.environ/NOMA_API_KEY + block_failures: true + + - guardrail_name: "noma-monitor-output" + litellm_params: + guardrail: noma + mode: "post_call" + api_key: os.environ/NOMA_API_KEY + monitor_mode: true +``` + +## ✨ Pass Additional Parameters + +Use `extra_body` to pass additional parameters to the Noma Security API call, such as dynamically setting the application ID for specific requests. + + + + +```python +import openai +client = openai.OpenAI( + api_key="your-api-key", + base_url="http://0.0.0.0:4000" +) + +response = client.chat.completions.create( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "Hello, how are you?"}], + extra_body={ + "guardrails": { + "noma-guard": { + "extra_body": { + "application_id": "my-specific-app-id" + } + } + } + } +) +``` + + + + +```shell +curl 'http://0.0.0.0:4000/v1/chat/completions' \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "gpt-4o-mini", + "messages": [ + { + "role": "user", + "content": "Hello, how are you?" + } + ], + "guardrails": { + "noma-guard": { + "extra_body": { + "application_id": "my-specific-app-id" + } + } + } +}' +``` + + + +This allows you to override the default `application_id` parameter for specific requests, which is useful for tracking usage across different applications or components. + +## Response Details + +When content is blocked, Noma provides detailed information about the violations as JSON inside the `message` field, with the following structure: + +```json +{ + "error": "Request blocked by Noma guardrail", + "details": { + "prompt": { + "harmfulContent": { + "result": true, + "confidence": 0.95 + }, + "sensitiveData": { + "email": { + "result": true, + "entities": ["user@example.com"] + } + }, + "bannedTopics": { + "violence": { + "result": true, + "confidence": 0.88 + } + } + } + } +} +``` diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 419afcd5466..7d55525919f 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -40,6 +40,7 @@ const sidebars = { "proxy/guardrails/guardrails_ai", "proxy/guardrails/lakera_ai", "proxy/guardrails/model_armor", + "proxy/guardrails/noma_security", "proxy/guardrails/openai_moderation", "proxy/guardrails/pangea", "proxy/guardrails/pillar_security", diff --git a/litellm/proxy/guardrails/guardrail_hooks/noma/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/noma/__init__.py new file mode 100644 index 00000000000..dc3e4d9768e --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/noma/__init__.py @@ -0,0 +1,36 @@ +from typing import TYPE_CHECKING + +from litellm.types.guardrails import SupportedGuardrailIntegrations + +from .noma import NomaGuardrail + +if TYPE_CHECKING: + from litellm.types.guardrails import Guardrail, LitellmParams + + +def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"): + import litellm + + _noma_callback = NomaGuardrail( + guardrail_name=guardrail.get("guardrail_name", ""), + api_key=litellm_params.api_key, + api_base=litellm_params.api_base, + application_id=litellm_params.application_id, + monitor_mode=litellm_params.monitor_mode, + block_failures=litellm_params.block_failures, + event_hook=litellm_params.mode, + default_on=litellm_params.default_on, + ) + litellm.logging_callback_manager.add_litellm_callback(_noma_callback) + + return _noma_callback + + +guardrail_initializer_registry = { + SupportedGuardrailIntegrations.NOMA.value: initialize_guardrail, +} + + +guardrail_class_registry = { + SupportedGuardrailIntegrations.NOMA.value: NomaGuardrail, +} diff --git a/litellm/proxy/guardrails/guardrail_hooks/noma/noma.py b/litellm/proxy/guardrails/guardrail_hooks/noma/noma.py new file mode 100644 index 00000000000..ed5929f0564 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/noma/noma.py @@ -0,0 +1,403 @@ +# +-------------------------------------------------------------+ +# +# Noma Security Guardrail Integration for LiteLLM +# https://noma.security +# +# +-------------------------------------------------------------+ + +import copy +import os +from typing import Any, Dict, Literal, Optional, Union +from urllib.parse import urljoin + +from fastapi import HTTPException + +import litellm +from litellm import DualCache, ModelResponse +from litellm._logging import verbose_proxy_logger +from litellm.integrations.custom_guardrail import CustomGuardrail +from litellm.llms.custom_httpx.http_handler import ( + get_async_httpx_client, + httpxSpecialProvider, +) +from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.guardrails import GuardrailEventHooks +from litellm.types.utils import EmbeddingResponse, ImageResponse + + +class NomaBlockedMessage(HTTPException): + """Exception raised when Noma guardrail blocks a message""" + + def __init__(self, classification_response: dict): + classification = self._filter_triggered_classifications(classification_response) + super().__init__( + status_code=400, + detail={ + "error": "Request blocked by Noma guardrail", + "details": classification, + }, + ) + + def _filter_triggered_classifications( + self, + response_dict: dict, + ) -> dict: + """Filter and return only triggered classifications""" + filtered_response = copy.deepcopy(response_dict) + + # Filter prompt classifications if present + if filtered_response.get("prompt"): + filtered_response["prompt"] = self.filter_classification_object( + filtered_response["prompt"] + ) + + # Filter response classifications if present + if filtered_response.get("response"): + filtered_response["response"] = self.filter_classification_object( + filtered_response["response"] + ) + + return filtered_response + + def filter_classification_object( + self, + classification_obj: dict, + ) -> dict: + """Filter classification object to only include triggered items""" + if not classification_obj: + return {} + + result = {} + + for key, value in classification_obj.items(): + if value is None: + continue + + if key in [ + "allowedTopics", + "bannedTopics", + "topicGuardrails", + ] and isinstance(value, dict): + filtered_topics = {} + for topic, topic_result in value.items(): + if self._is_result_true(topic_result): + filtered_topics[topic] = topic_result + + if filtered_topics: + result[key] = filtered_topics + + elif key == "sensitiveData" and isinstance(value, dict): + filtered_sensitive = {} + for data_type, data_result in value.items(): + if self._is_result_true(data_result): + filtered_sensitive[data_type] = data_result + + if filtered_sensitive: + result[key] = filtered_sensitive + + elif isinstance(value, dict) and "result" in value: + if self._is_result_true(value): + result[key] = value + + return result + + def _is_result_true(self, result_obj: Optional[Dict[str, Any]]) -> bool: + """ + Check if a result object has a "result" field that is True. + + Args: + result_obj: A dictionary that may contain a "result" field + + Returns: + True if the "result" field exists and is True, False otherwise + """ + if not result_obj or not isinstance(result_obj, dict): + return False + + return result_obj.get("result") is True + + +class NomaGuardrail(CustomGuardrail): + """ + Noma Security Guardrail for LiteLLM + + This guardrail integrates with Noma Security's AI-DR API to provide + content moderation and safety checks for LLM inputs and outputs. + """ + + _DEFAULT_API_BASE = "https://api.noma.security/" + _AIDR_ENDPOINT = "/ai-dr/v1/prompt/scan/aggregate" + + def __init__( + self, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + application_id: Optional[str] = None, + monitor_mode: Optional[bool] = None, + block_failures: Optional[bool] = None, + **kwargs, + ): + self.async_handler = get_async_httpx_client( + llm_provider=httpxSpecialProvider.GuardrailCallback + ) + self.api_key = api_key or os.environ.get("NOMA_API_KEY") + self.api_base = api_base or os.environ.get( + "NOMA_API_BASE", NomaGuardrail._DEFAULT_API_BASE + ) + self.application_id = application_id or os.environ.get( + "NOMA_APPLICATION_ID", "litellm" + ) + + if monitor_mode is None: + self.monitor_mode = ( + os.environ.get("NOMA_MONITOR_MODE", "false").lower() == "true" + ) + else: + self.monitor_mode = monitor_mode + + if block_failures is None: + self.block_failures = ( + os.environ.get("NOMA_BLOCK_FAILURES", "true").lower() == "true" + ) + else: + self.block_failures = block_failures + + super().__init__(**kwargs) + + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: DualCache, + data: dict, + call_type: Literal[ + "completion", + "text_completion", + "embeddings", + "image_generation", + "moderation", + "audio_transcription", + "pass_through_endpoint", + "rerank", + "mcp_call", + ], + ) -> Optional[Union[Exception, str, dict]]: + verbose_proxy_logger.debug("Running Noma pre-call hook") + + if ( + self.should_run_guardrail( + data=data, event_type=GuardrailEventHooks.pre_call + ) + is False + ): + return data + + try: + return await self._check_user_message(data, user_api_key_dict) + except NomaBlockedMessage: + raise + except Exception as e: + verbose_proxy_logger.error(f"Noma pre-call hook failed: {str(e)}") + + if self.block_failures and not self.monitor_mode: + raise + return data + + async def async_moderation_hook( + self, + data: dict, + user_api_key_dict: UserAPIKeyAuth, + call_type: Literal[ + "completion", + "embeddings", + "image_generation", + "moderation", + "audio_transcription", + "responses", + "mcp_call", + ], + ) -> Union[Exception, str, dict, None]: + event_type: GuardrailEventHooks = GuardrailEventHooks.during_call + if self.should_run_guardrail(data=data, event_type=event_type) is not True: + return data + + try: + return await self._check_user_message(data, user_api_key_dict) + except NomaBlockedMessage: + raise + except Exception as e: + verbose_proxy_logger.error(f"Noma moderation hook failed: {str(e)}") + + if self.block_failures and not self.monitor_mode: + raise + return data + + async def async_post_call_success_hook( + self, + data: dict, + user_api_key_dict: UserAPIKeyAuth, + response: Union[Any, ModelResponse, EmbeddingResponse, ImageResponse], + ): + event_type: GuardrailEventHooks = GuardrailEventHooks.post_call + if self.should_run_guardrail(data=data, event_type=event_type) is not True: + return response + + try: + return await self._check_llm_response(data, response, user_api_key_dict) + except NomaBlockedMessage: + raise + except Exception as e: + verbose_proxy_logger.error(f"Noma post-call hook failed: {str(e)}") + if self.block_failures and not self.monitor_mode: + raise + return response + + async def _check_user_message( + self, + request_data: dict, + user_auth: UserAPIKeyAuth, + ) -> Union[Exception, str, dict, None]: + """Check user message for policy violations""" + extra_data = self.get_guardrail_dynamic_request_body_params(request_data) + + user_message = await self._extract_user_message(request_data) + if not user_message: + return request_data + + payload = {"request": {"text": user_message}} + response_json = await self._call_noma_api( + payload=payload, + llm_request_id=None, + request_data=request_data, + user_auth=user_auth, + extra_data=extra_data, + ) + await self._check_verdict("user", user_message, response_json) + + return request_data + + async def _check_llm_response( + self, + request_data: dict, + response: Union[Any, ModelResponse, EmbeddingResponse, ImageResponse], + user_auth: UserAPIKeyAuth, + ) -> Union[Exception, ModelResponse, Any]: + """Check LLM response for policy violations""" + extra_data = self.get_guardrail_dynamic_request_body_params(request_data) + + if not isinstance(response, litellm.ModelResponse): + return response + + content = None + for choice in response.choices: + if isinstance(choice, litellm.Choices) and choice.message.content: + content = choice.message.content + break + + if not content or not isinstance(content, str): + return response + + payload = {"response": {"text": content}} + + response_json = await self._call_noma_api( + payload=payload, + llm_request_id=response.id, + request_data=request_data, + user_auth=user_auth, + extra_data=extra_data, + ) + await self._check_verdict("assistant", content, response_json) + + return response + + async def _extract_user_message(self, data: dict) -> Optional[str]: + """Extract the last user message from request data""" + messages = data.get("messages", []) + if not messages: + return None + + # Get the last user message + user_messages = [msg for msg in messages if msg.get("role") == "user"] + if not user_messages: + return None + + last_user_message = user_messages[-1].get("content", "") + if not last_user_message or not isinstance(last_user_message, str): + return None + + return last_user_message + + async def _call_noma_api( + self, + payload: dict, + llm_request_id: Optional[str], + request_data: dict, + user_auth: UserAPIKeyAuth, + extra_data: dict, + ) -> dict: + call_id = request_data.get("litellm_call_id") + headers = { + "X-Noma-AIDR-Application-ID": self.application_id, + **({"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}), + **({"X-Noma-Request-ID": call_id} if call_id else {}), + } + endpoint = urljoin( + self.api_base or "https://api.noma.security/", NomaGuardrail._AIDR_ENDPOINT + ) + + response = await self.async_handler.post( + endpoint, + headers=headers, + json={ + **payload, + "context": { + "applicationId": extra_data.get("application_id") + or request_data.get("metadata", {}) + .get("headers", {}) + .get("x-noma-application-id"), + "ipAddress": request_data.get("metadata", {}).get( + "requester_ip_address", None + ), + "userId": user_auth.user_email + if user_auth.user_email + else user_auth.user_id, + "sessionId": call_id, + "requestId": llm_request_id, + }, + }, + ) + response.raise_for_status() + + return response.json() + + async def _check_verdict( + self, + type: Literal["user", "assistant"], + message: str, + response_json: dict, + ) -> None: + """ + Check the verdict from the Noma API and raise an exception if needed + """ + if not response_json.get("verdict", True): + msg = str.format( + "Noma guardrail blocked {type} message: {message}", + type=type, + message=message, + ) + + if self.monitor_mode: + verbose_proxy_logger.warning(msg) + else: + verbose_proxy_logger.debug(msg) + original_response = response_json.get("originalResponse", {}) + raise NomaBlockedMessage(original_response) + else: + msg = str.format( + "Noma guardrail allowed {type} message: {message}", + type=type, + message=message, + ) + if self.monitor_mode: + verbose_proxy_logger.info(msg) + else: + verbose_proxy_logger.debug(msg) diff --git a/litellm/types/guardrails.py b/litellm/types/guardrails.py index fd18484a898..f31f304bda9 100644 --- a/litellm/types/guardrails.py +++ b/litellm/types/guardrails.py @@ -40,6 +40,7 @@ class SupportedGuardrailIntegrations(Enum): AZURE_TEXT_MODERATIONS = "azure/text_moderations" MODEL_ARMOR = "model_armor" OPENAI_MODERATION = "openai_moderation" + NOMA = "noma" class Role(Enum): SYSTEM = "system" @@ -359,6 +360,23 @@ class PillarGuardrailConfigModel(BaseModel): ) +class NomaGuardrailConfigModel(BaseModel): + """Configuration parameters for the Noma Security guardrail""" + + application_id: Optional[str] = Field( + default=None, + description="Application ID for Noma Security. Defaults to 'litellm' if not provided", + ) + monitor_mode: Optional[bool] = Field( + default=None, + description="If True, logs violations without blocking. Defaults to False if not provided", + ) + block_failures: Optional[bool] = Field( + default=None, + description="If True, blocks requests on API failures. Defaults to True if not provided", + ) + + class BaseLitellmParams(BaseModel): # works for new and patch update guardrails api_key: Optional[str] = Field( default=None, description="API key for the guardrail service" @@ -445,6 +463,7 @@ class LitellmParams( LakeraV2GuardrailConfigModel, LassoGuardrailConfigModel, PillarGuardrailConfigModel, + NomaGuardrailConfigModel, BaseLitellmParams, ): guardrail: str = Field(description="The type of guardrail integration to use") diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_noma.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_noma.py new file mode 100644 index 00000000000..aeea5f81b10 --- /dev/null +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_noma.py @@ -0,0 +1,498 @@ +import os +from unittest.mock import MagicMock, patch + +import httpx +import pytest + +import litellm +from litellm import ModelResponse +from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.guardrails.guardrail_hooks.noma import ( + NomaGuardrail, + initialize_guardrail, +) +from litellm.proxy.guardrails.guardrail_hooks.noma.noma import NomaBlockedMessage +from litellm.proxy.guardrails.init_guardrails import init_guardrails_v2 +from litellm.types.utils import Choices, Message + + +@pytest.fixture +def noma_guardrail(): + """Create a NomaGuardrail instance for testing""" + return NomaGuardrail( + api_key="test-api-key", + api_base="https://api.test.noma.security/", + application_id="test-app", + monitor_mode=False, + block_failures=True, + guardrail_name="test-noma-guardrail", + event_hook="pre_call", + default_on=True, + ) + + +@pytest.fixture +def mock_user_api_key_dict(): + """Create a mock UserAPIKeyAuth object""" + return UserAPIKeyAuth( + user_id="test-user-id", + user_email="test@example.com", + key_name="test-key", + key_alias=None, + team_id=None, + team_alias=None, + user_role=None, + api_key="test-api-key", + permissions={}, + models=[], + spend=0.0, + max_budget=None, + soft_budget=None, + tpm_limit=None, + rpm_limit=None, + parallel_request_limit=None, + metadata={}, + max_parallel_requests=None, + allowed_cache_controls=[], + model_spend={}, + model_max_budget={}, + ) + + +@pytest.fixture +def mock_request_data(): + """Create mock request data""" + return { + "messages": [ + {"role": "system", "content": "You are a helpful assistant"}, + {"role": "user", "content": "Hello, how are you?"}, + ], + "litellm_call_id": "test-call-id", + "metadata": {"requester_ip_address": "192.168.1.1"}, + } + + +class TestNomaGuardrailConfiguration: + """Test configuration and initialization of Noma guardrail""" + + def test_init_with_config(self): + """Test initializing Noma guardrail via init_guardrails_v2""" + with patch.dict( + os.environ, + { + "NOMA_API_KEY": "test-api-key", + "NOMA_API_BASE": "https://api.test.noma.security/", + }, + ): + init_guardrails_v2( + all_guardrails=[ + { + "guardrail_name": "noma-pre-guard", + "litellm_params": { + "guardrail": "noma", + "mode": "pre_call", + "application_id": "test-app", + "monitor_mode": False, + "block_failures": True, + }, + } + ], + config_file_path="", + ) + + def test_init_with_env_vars(self): + """Test initialization with environment variables""" + with patch.dict( + os.environ, + { + "NOMA_API_KEY": "env-api-key", + "NOMA_API_BASE": "https://env.api.noma.security/", + "NOMA_APPLICATION_ID": "env-app-id", + "NOMA_MONITOR_MODE": "true", + "NOMA_BLOCK_FAILURES": "false", + }, + ): + guardrail = NomaGuardrail() + assert guardrail.api_key == "env-api-key" + assert guardrail.api_base == "https://env.api.noma.security/" + assert guardrail.application_id == "env-app-id" + assert guardrail.monitor_mode is True + assert guardrail.block_failures is False + + def test_init_with_params_override_env(self): + """Test that constructor params override environment variables""" + with patch.dict( + os.environ, + { + "NOMA_API_KEY": "env-api-key", + "NOMA_MONITOR_MODE": "true", + }, + ): + guardrail = NomaGuardrail( + api_key="param-api-key", + monitor_mode=False, + ) + assert guardrail.api_key == "param-api-key" + assert guardrail.monitor_mode is False + + def test_initialize_guardrail_function(self): + """Test the initialize_guardrail function""" + from litellm.types.guardrails import Guardrail, LitellmParams + + litellm_params = LitellmParams( + guardrail="noma", + mode="pre_call", + api_key="test-key", + api_base="https://test.api/", + application_id="test-app", + monitor_mode=True, + block_failures=False, + ) + + guardrail = Guardrail( + guardrail_name="test-guardrail", + litellm_params=litellm_params, + ) + + with patch("litellm.logging_callback_manager.add_litellm_callback") as mock_add: + result = initialize_guardrail(litellm_params, guardrail) + + assert isinstance(result, NomaGuardrail) + assert result.api_key == "test-key" + assert result.api_base == "https://test.api/" + assert result.application_id == "test-app" + assert result.monitor_mode is True + assert result.block_failures is False + mock_add.assert_called_once_with(result) + + +class TestNomaBlockedMessage: + """Test the NomaBlockedMessage exception class""" + + def test_blocked_message_basic(self): + """Test basic blocked message creation""" + response = { + "verdict": False, + "prompt": { + "harmfulContent": {"result": True, "confidence": 0.9}, + "code": {"result": False, "confidence": 0.1}, + }, + } + + exception = NomaBlockedMessage(response) + assert exception.status_code == 400 + assert exception.detail["error"] == "Request blocked by Noma guardrail" + assert "harmfulContent" in exception.detail["details"]["prompt"] + assert "code" not in exception.detail["details"]["prompt"] + + def test_blocked_message_with_sensitive_data(self): + """Test blocked message with sensitive data detection""" + response = { + "verdict": False, + "prompt": { + "sensitiveData": { + "email": {"result": True, "entities": ["test@example.com"]}, + "phone": {"result": False}, + }, + }, + } + + exception = NomaBlockedMessage(response) + assert "email" in exception.detail["details"]["prompt"]["sensitiveData"] + assert "phone" not in exception.detail["details"]["prompt"]["sensitiveData"] + + def test_blocked_message_with_topics(self): + """Test blocked message with topic guardrails""" + response = { + "verdict": False, + "prompt": { + "bannedTopics": { + "violence": {"result": True, "confidence": 0.95}, + "politics": {"result": False, "confidence": 0.2}, + }, + }, + } + + exception = NomaBlockedMessage(response) + assert "violence" in exception.detail["details"]["prompt"]["bannedTopics"] + assert "politics" not in exception.detail["details"]["prompt"]["bannedTopics"] + + +class TestNomaGuardrailHooks: + """Test the guardrail hook methods""" + + @pytest.mark.asyncio + async def test_pre_call_hook_allowed( + self, noma_guardrail, mock_user_api_key_dict, mock_request_data + ): + """Test pre-call hook when content is allowed""" + mock_response = MagicMock() + mock_response.json.return_value = {"verdict": True} + mock_response.raise_for_status = MagicMock() + + with patch.object( + noma_guardrail.async_handler, "post", return_value=mock_response + ) as mock_post: + result = await noma_guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key_dict, + cache=MagicMock(), + data=mock_request_data, + call_type="completion", + ) + + assert result == mock_request_data + mock_post.assert_called_once() + + # Verify API call details + call_args = mock_post.call_args + assert call_args[0][0].endswith("/ai-dr/v1/prompt/scan/aggregate") + assert call_args[1]["headers"]["X-Noma-AIDR-Application-ID"] == "test-app" + assert call_args[1]["headers"]["Authorization"] == "Bearer test-api-key" + assert call_args[1]["json"]["request"]["text"] == "Hello, how are you?" + + @pytest.mark.asyncio + async def test_pre_call_hook_blocked( + self, noma_guardrail, mock_user_api_key_dict, mock_request_data + ): + """Test pre-call hook when content is blocked""" + mock_response = MagicMock() + mock_response.json.return_value = { + "verdict": False, + "originalResponse": { + "prompt": {"harmfulContent": {"result": True, "confidence": 0.9}} + }, + } + mock_response.raise_for_status = MagicMock() + + with patch.object( + noma_guardrail.async_handler, "post", return_value=mock_response + ): + with pytest.raises(NomaBlockedMessage) as exc_info: + await noma_guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key_dict, + cache=MagicMock(), + data=mock_request_data, + call_type="completion", + ) + + assert exc_info.value.status_code == 400 + assert "harmfulContent" in exc_info.value.detail["details"]["prompt"] + + @pytest.mark.asyncio + async def test_pre_call_hook_monitor_mode( + self, mock_user_api_key_dict, mock_request_data + ): + """Test pre-call hook in monitor mode (logs but doesn't block)""" + guardrail = NomaGuardrail( + api_key="test-key", + monitor_mode=True, + guardrail_name="test-guardrail", + event_hook="pre_call", + default_on=True, + ) + + mock_response = MagicMock() + mock_response.json.return_value = { + "verdict": False, + "originalResponse": {"prompt": {"harmfulContent": {"result": True}}}, + } + mock_response.raise_for_status = MagicMock() + + with patch.object(guardrail.async_handler, "post", return_value=mock_response): + # Should not raise exception in monitor mode + result = await guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key_dict, + cache=MagicMock(), + data=mock_request_data, + call_type="completion", + ) + + assert result == mock_request_data + + @pytest.mark.asyncio + async def test_post_call_success_hook( + self, noma_guardrail, mock_user_api_key_dict, mock_request_data + ): + """Test post-call success hook""" + # Create a mock ModelResponse + response = ModelResponse( + id="test-response-id", + choices=[ + Choices( + finish_reason="stop", + index=0, + message=Message( + content="I'm doing well, thank you!", role="assistant" + ), + ) + ], + created=1234567890, + model="gpt-3.5-turbo", + object="chat.completion", + system_fingerprint=None, + usage={"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}, + ) + + mock_api_response = MagicMock() + mock_api_response.json.return_value = {"verdict": True} + mock_api_response.raise_for_status = MagicMock() + + # Update guardrail to use post_call event hook + noma_guardrail.event_hook = "post_call" + + with patch.object( + noma_guardrail.async_handler, "post", return_value=mock_api_response + ) as mock_post: + result = await noma_guardrail.async_post_call_success_hook( + data=mock_request_data, + user_api_key_dict=mock_user_api_key_dict, + response=response, + ) + + assert result == response + mock_post.assert_called_once() + + # Verify API call details + call_args = mock_post.call_args + assert ( + call_args[1]["json"]["response"]["text"] == "I'm doing well, thank you!" + ) + assert call_args[1]["json"]["context"]["requestId"] == "test-response-id" + + @pytest.mark.asyncio + async def test_moderation_hook( + self, noma_guardrail, mock_user_api_key_dict, mock_request_data + ): + """Test moderation hook (during_call)""" + # Update guardrail to use during_call event hook + noma_guardrail.event_hook = "during_call" + + mock_response = MagicMock() + mock_response.json.return_value = {"verdict": True} + mock_response.raise_for_status = MagicMock() + + with patch.object( + noma_guardrail.async_handler, "post", return_value=mock_response + ): + result = await noma_guardrail.async_moderation_hook( + data=mock_request_data, + user_api_key_dict=mock_user_api_key_dict, + call_type="completion", + ) + + assert result == mock_request_data + + @pytest.mark.asyncio + async def test_api_failure_handling( + self, noma_guardrail, mock_user_api_key_dict, mock_request_data + ): + with patch.object( + noma_guardrail.async_handler, + "post", + side_effect=httpx.HTTPStatusError( + "API Error", request=MagicMock(), response=MagicMock(status_code=500) + ), + ): + with pytest.raises(httpx.HTTPStatusError): + await noma_guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key_dict, + cache=MagicMock(), + data=mock_request_data, + call_type="completion", + ) + + @pytest.mark.asyncio + async def test_api_failure_no_block( + self, mock_user_api_key_dict, mock_request_data + ): + guardrail = NomaGuardrail( + api_key="test-key", + block_failures=False, + guardrail_name="test-guardrail", + event_hook="pre_call", + default_on=True, + ) + + with patch.object( + guardrail.async_handler, + "post", + side_effect=httpx.HTTPStatusError( + "API Error", request=MagicMock(), response=MagicMock(status_code=500) + ), + ): + result = await guardrail.async_pre_call_hook( + user_api_key_dict=mock_user_api_key_dict, + cache=MagicMock(), + data=mock_request_data, + call_type="completion", + ) + + assert result == mock_request_data + + def test_extract_user_message(self, noma_guardrail): + data = { + "messages": [ + {"role": "system", "content": "System prompt"}, + {"role": "user", "content": "First user message"}, + {"role": "assistant", "content": "Assistant response"}, + {"role": "user", "content": "Second user message"}, + ] + } + + import asyncio + + message = asyncio.run(noma_guardrail._extract_user_message(data)) + assert message == "Second user message" + + data = {"messages": [{"role": "system", "content": "System prompt"}]} + message = asyncio.run(noma_guardrail._extract_user_message(data)) + assert message is None + + data = {"messages": []} + message = asyncio.run(noma_guardrail._extract_user_message(data)) + assert message is None + + data = {} + message = asyncio.run(noma_guardrail._extract_user_message(data)) + assert message is None + + +class TestIntegration: + @pytest.mark.asyncio + async def test_full_guardrail_flow(self): + """Test full guardrail flow with multiple hooks""" + with patch.dict( + os.environ, + { + "NOMA_API_KEY": "test-api-key", + "NOMA_API_BASE": "https://api.test.noma.security/", + }, + ): + init_guardrails_v2( + all_guardrails=[ + { + "guardrail_name": "noma-pre-guard", + "litellm_params": { + "guardrail": "noma", + "mode": "pre_call", + "application_id": "test-app", + }, + }, + { + "guardrail_name": "noma-post-guard", + "litellm_params": { + "guardrail": "noma", + "mode": "post_call", + "application_id": "test-app", + }, + }, + ], + config_file_path="", + ) + + custom_loggers = ( + litellm.logging_callback_manager.get_custom_loggers_for_type( + callback_type=litellm.integrations.custom_guardrail.CustomGuardrail + ) + ) + assert len(custom_loggers) >= 2 From b75961fb2092d63004b10e6759e646ea2370da18 Mon Sep 17 00:00:00 2001 From: Yuki Imajuku Date: Wed, 13 Aug 2025 21:05:14 +0900 Subject: [PATCH 22/66] update openrouter claude sonnet --- litellm/model_prices_and_context_window_backup.json | 12 ++++++------ model_prices_and_context_window.json | 12 ++++++------ 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1abc6519603..eeb0f782f9a 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11388,9 +11388,9 @@ }, "openrouter/anthropic/claude-3.7-sonnet": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 128000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 128000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, @@ -11405,9 +11405,9 @@ }, "openrouter/anthropic/claude-3.7-sonnet:beta": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 128000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 128000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, @@ -11432,9 +11432,9 @@ }, "openrouter/anthropic/claude-sonnet-4": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 64000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 64000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1abc6519603..eeb0f782f9a 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11388,9 +11388,9 @@ }, "openrouter/anthropic/claude-3.7-sonnet": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 128000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 128000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, @@ -11405,9 +11405,9 @@ }, "openrouter/anthropic/claude-3.7-sonnet:beta": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 128000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 128000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, @@ -11432,9 +11432,9 @@ }, "openrouter/anthropic/claude-sonnet-4": { "supports_computer_use": true, - "max_tokens": 8192, + "max_tokens": 64000, "max_input_tokens": 200000, - "max_output_tokens": 8192, + "max_output_tokens": 64000, "input_cost_per_token": 3e-06, "output_cost_per_token": 1.5e-05, "input_cost_per_image": 0.0048, From 1e81a1bd7c13116d2dfc8429192f7cf38ece823f Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Sun, 17 Aug 2025 17:13:29 +0200 Subject: [PATCH 23/66] feat: Add thinking and reasoning_effort parameter support for GitHub Copilot provider - Add github_copilot case to get_supported_openai_params function - Implement get_supported_openai_params method in GithubCopilotConfig - Dynamically add thinking and reasoning_effort params for Anthropic models - Add comprehensive tests for parameter support validation - Ensure case-insensitive model detection for parameter inclusion Fixes UnsupportedParamsError when using advanced reasoning parameters with Anthropic models through GitHub Copilot proxy. --- .../get_supported_openai_params.py | 2 + .../github_copilot/chat/transformation.py | 16 ++++++++ .../test_github_copilot_transformation.py | 40 +++++++++++++++++++ 3 files changed, 58 insertions(+) diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index 5fcd2ddb70a..e17a0a88a0e 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -271,6 +271,8 @@ def get_supported_openai_params( # noqa: PLR0915 model=model ) ) + elif custom_llm_provider == "github_copilot": + return litellm.GithubCopilotConfig().get_supported_openai_params(model=model) elif custom_llm_provider in litellm._custom_providers: if request_type == "chat_completion": provider_config = litellm.ProviderConfigManager.get_provider_chat_config( diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index 4526e6247b4..de363654feb 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -77,6 +77,22 @@ class GithubCopilotConfig(OpenAIConfig): return validated_headers + def get_supported_openai_params(self, model: str) -> list: + """ + Get supported OpenAI parameters for GitHub Copilot. + + For Anthropic models (like claude-sonnet-4), includes thinking and reasoning parameters. + For other models, returns standard OpenAI parameters. + """ + # Get base OpenAI parameters + base_params = super().get_supported_openai_params(model) + + # Add thinking and reasoning parameters for Anthropic Claude models + if "claude" in model.lower(): + base_params.extend(["thinking", "reasoning_effort"]) + + return base_params + def _determine_initiator(self, messages: List[AllMessageValues]) -> str: """ Determine if request is user or agent initiated based on message roles. diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py index f21c123579d..ac7046f8ced 100644 --- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py +++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py @@ -362,3 +362,43 @@ def test_x_initiator_header_system_only_messages(): ) assert headers["X-Initiator"] == "user" + + +def test_get_supported_openai_params_claude_model(): + """Test that Claude models support thinking and reasoning parameters.""" + config = GithubCopilotConfig() + + # Test Claude model supports thinking and reasoning_effort parameters + supported_params = config.get_supported_openai_params("claude-sonnet-4") + assert "thinking" in supported_params + assert "reasoning_effort" in supported_params + + # Test Claude model with different naming + supported_params_claude = config.get_supported_openai_params("claude-3.5-sonnet") + assert "thinking" in supported_params_claude + assert "reasoning_effort" in supported_params_claude + + # Test non-Claude model doesn't include thinking/reasoning parameters + supported_params_gpt = config.get_supported_openai_params("gpt-4o") + assert "thinking" not in supported_params_gpt + assert "reasoning_effort" not in supported_params_gpt + + # Test with other non-Claude models + supported_params_o3 = config.get_supported_openai_params("o3-mini") + assert "thinking" not in supported_params_o3 + assert "reasoning_effort" not in supported_params_o3 + + +def test_get_supported_openai_params_case_insensitive(): + """Test that Claude model detection is case-insensitive.""" + config = GithubCopilotConfig() + + # Test uppercase + supported_params_upper = config.get_supported_openai_params("CLAUDE-SONNET-4") + assert "thinking" in supported_params_upper + assert "reasoning_effort" in supported_params_upper + + # Test mixed case + supported_params_mixed = config.get_supported_openai_params("Claude-3.5-Sonnet") + assert "thinking" in supported_params_mixed + assert "reasoning_effort" in supported_params_mixed From d92092f04003dfeaee0cff5de7eb8ed4969fd570 Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Sun, 17 Aug 2025 17:26:59 +0200 Subject: [PATCH 24/66] formatting --- litellm/litellm_core_utils/get_supported_openai_params.py | 7 +++---- litellm/llms/github_copilot/chat/transformation.py | 6 +++--- litellm/llms/github_copilot/common_utils.py | 1 - 3 files changed, 6 insertions(+), 8 deletions(-) diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index e17a0a88a0e..b71b609cc50 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -266,10 +266,9 @@ def get_supported_openai_params( # noqa: PLR0915 from litellm.llms.elevenlabs.audio_transcription.transformation import ( ElevenLabsAudioTranscriptionConfig, ) - return ( - ElevenLabsAudioTranscriptionConfig().get_supported_openai_params( - model=model - ) + + return ElevenLabsAudioTranscriptionConfig().get_supported_openai_params( + model=model ) elif custom_llm_provider == "github_copilot": return litellm.GithubCopilotConfig().get_supported_openai_params(model=model) diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index de363654feb..8ef2a54c62d 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -80,17 +80,17 @@ class GithubCopilotConfig(OpenAIConfig): def get_supported_openai_params(self, model: str) -> list: """ Get supported OpenAI parameters for GitHub Copilot. - + For Anthropic models (like claude-sonnet-4), includes thinking and reasoning parameters. For other models, returns standard OpenAI parameters. """ # Get base OpenAI parameters base_params = super().get_supported_openai_params(model) - + # Add thinking and reasoning parameters for Anthropic Claude models if "claude" in model.lower(): base_params.extend(["thinking", "reasoning_effort"]) - + return base_params def _determine_initiator(self, messages: List[AllMessageValues]) -> str: diff --git a/litellm/llms/github_copilot/common_utils.py b/litellm/llms/github_copilot/common_utils.py index 4c9a4b6dad0..86fbb706e52 100644 --- a/litellm/llms/github_copilot/common_utils.py +++ b/litellm/llms/github_copilot/common_utils.py @@ -28,7 +28,6 @@ class GithubCopilotError(BaseLLMException): ) - class GetDeviceCodeError(GithubCopilotError): pass From 0febdf8c1ce8e4360fcd3032110a9590f1b3fcd2 Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Sun, 17 Aug 2025 17:51:58 +0200 Subject: [PATCH 25/66] feat: add thinking and reasoning parameter support for GitHub Copilot provider - Add dynamic parameter support for anthropic models through GitHub Copilot - Include thinking parameter for anthropic model compatibility - Support reasoning_effort parameter for both anthropic and reasoning models - Update test coverage for parameter validation logic - Ensure proper parameter filtering based on model type --- litellm/llms/github_copilot/chat/transformation.py | 10 +++++++--- .../test_github_copilot_transformation.py | 8 +++++--- 2 files changed, 12 insertions(+), 6 deletions(-) diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index 8ef2a54c62d..e71e9b610ba 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -82,14 +82,18 @@ class GithubCopilotConfig(OpenAIConfig): Get supported OpenAI parameters for GitHub Copilot. For Anthropic models (like claude-sonnet-4), includes thinking and reasoning parameters. - For other models, returns standard OpenAI parameters. + For other models, returns standard OpenAI parameters (which may include reasoning_effort for o-series models). """ # Get base OpenAI parameters base_params = super().get_supported_openai_params(model) - # Add thinking and reasoning parameters for Anthropic Claude models + # Add Claude-specific parameters for Anthropic models if "claude" in model.lower(): - base_params.extend(["thinking", "reasoning_effort"]) + if "thinking" not in base_params: + base_params.append("thinking") + # reasoning_effort is not included by parent for Claude models, so add it + if "reasoning_effort" not in base_params: + base_params.append("reasoning_effort") return base_params diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py index ac7046f8ced..d389c445526 100644 --- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py +++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py @@ -378,15 +378,17 @@ def test_get_supported_openai_params_claude_model(): assert "thinking" in supported_params_claude assert "reasoning_effort" in supported_params_claude - # Test non-Claude model doesn't include thinking/reasoning parameters + # Test non-Claude model doesn't include thinking parameters but may include reasoning_effort supported_params_gpt = config.get_supported_openai_params("gpt-4o") assert "thinking" not in supported_params_gpt + # gpt-4o should NOT have reasoning_effort (not a reasoning model) assert "reasoning_effort" not in supported_params_gpt - # Test with other non-Claude models + # Test O-series reasoning models include reasoning_effort but not thinking supported_params_o3 = config.get_supported_openai_params("o3-mini") assert "thinking" not in supported_params_o3 - assert "reasoning_effort" not in supported_params_o3 + # o3-mini should have reasoning_effort (it's an O-series reasoning model) + assert "reasoning_effort" in supported_params_o3 def test_get_supported_openai_params_case_insensitive(): From 223587179f4281a62656aca5418b2a8fdfa7fec1 Mon Sep 17 00:00:00 2001 From: Ryan Means Date: Wed, 30 Jul 2025 17:03:02 -0700 Subject: [PATCH 26/66] Update Pangea Guardrail to support new AIDR endpoint --- .../guardrail_hooks/pangea/pangea.py | 361 +++++++----------- 1 file changed, 139 insertions(+), 222 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py b/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py index 07e9fa760fd..be4052e4eca 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py +++ b/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py @@ -1,6 +1,6 @@ # litellm/proxy/guardrails/guardrail_hooks/pangea.py import os -from typing import TYPE_CHECKING, Any, Optional, Protocol, Type +from typing import TYPE_CHECKING, Any, Optional, Type from fastapi import HTTPException @@ -19,7 +19,7 @@ from litellm.proxy.common_utils.callback_utils import ( add_guardrail_to_applied_guardrails_header, ) from litellm.types.guardrails import GuardrailEventHooks -from litellm.types.utils import LLMResponseTypes, ModelResponse, TextCompletionResponse +from litellm.types.utils import Choices, LLMResponseTypes, ModelResponse, TextCompletionResponse if TYPE_CHECKING: from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel @@ -31,14 +31,6 @@ class PangeaGuardrailMissingSecrets(Exception): pass -class _Transformer(Protocol): - def get_messages(self) -> list[dict]: # noqa: E704 - ... - - def update_original_body(self, prompt_messages: list[dict]) -> Any: # noqa: E704 - ... - - class _TextCompletionRequest: def __init__(self, body): self.body = body @@ -53,109 +45,6 @@ class _TextCompletionRequest: return self.body -class _TextCompletionResponse: - def __init__(self, body): - self.body = body - - def get_messages(self) -> list[dict]: - messages = [] - for choice in self.body["choices"]: - messages.append({"role": "assistant", "content": choice["text"]}) - - return messages - - def update_original_body(self, prompt_messages: list[dict]) -> Any: - assert len(prompt_messages) == len(self.body["choices"]) - - for choice, prompt_message in zip(self.body["choices"], prompt_messages): - choice["text"] = prompt_message["content"] - - return self.body - - -class _ChatCompletionRequest: - def __init__(self, body): - self.body = body - - def get_messages(self) -> list[dict]: - messages = [] - - for message in self.body["messages"]: - role = message["role"] - content = message["content"] - if isinstance(content, str): - messages.append({"role": role, "content": content}) - if isinstance(content, list): - for content_part in content: - if content_part["type"] == "text": - messages.append({"role": role, "content": content_part["text"]}) - - return messages - - def update_original_body(self, prompt_messages: list[dict]) -> Any: - count = 0 - - for message in self.body["messages"]: - content = message["content"] - if isinstance(content, str): - message["content"] = prompt_messages[count]["content"] - count += 1 - if isinstance(content, list): - for content_part in content: - if content_part["type"] == "text": - content_part["text"] = prompt_messages[count]["content"] - count += 1 - - assert len(prompt_messages) == count - return self.body - - -class _ChatCompletionResponse: - def __init__(self, body): - self.body = body - - def get_messages(self) -> list[dict]: - messages = [] - - for choice in self.body["choices"]: - messages.append( - { - "role": choice["message"]["role"], - "content": choice["message"]["content"], - } - ) - - return messages - - def update_original_body(self, prompt_messages: list[dict]) -> Any: - assert len(prompt_messages) == len(self.body["choices"]) - - for choice, prompt_message in zip(self.body["choices"], prompt_messages): - choice["message"]["content"] = prompt_message["content"] - - return self.body - - -def _get_transformer_for_request(body, call_type) -> Optional[_Transformer]: - match call_type: - case "text_completion" | "atext_completion": - return _TextCompletionRequest(body) - case "completion" | "acompletion": - return _ChatCompletionRequest(body) - - return None - - -def _get_transformer_for_response(body) -> Optional[_Transformer]: - match body: - case TextCompletionResponse(): - return _TextCompletionResponse(body) - case ModelResponse(): - return _ChatCompletionResponse(body) - - return None - - class PangeaHandler(CustomGuardrail): """ Pangea AI Guardrail handler to interact with the Pangea AI Guard service. @@ -200,7 +89,6 @@ class PangeaHandler(CustomGuardrail): ) self.pangea_input_recipe = pangea_input_recipe self.pangea_output_recipe = pangea_output_recipe - self.guardrail_endpoint = f"{self.api_base}/v1/text/guard" # Pass relevant kwargs to the parent class super().__init__(guardrail_name=guardrail_name, **kwargs) @@ -208,7 +96,9 @@ class PangeaHandler(CustomGuardrail): f"Initialized Pangea Guardrail: name={guardrail_name}, recipe={pangea_input_recipe}, api_base={self.api_base}" ) - async def _call_pangea_guard(self, payload: dict, hook_name: str) -> dict: + async def _call_pangea_ai_guard( + self, api: str, payload: dict, hook_name: str + ) -> dict: """ Makes the API call to the Pangea AI Guard endpoint. The function itself will raise an error in the case that a response @@ -216,6 +106,7 @@ class PangeaHandler(CustomGuardrail): should act on. Args: + api (str): Which API to use (text/guard or v1beta/guard) payload (dict): The request payload. request_data (dict): Original request data (used for logging/headers). hook_name (str): Name of the hook calling this function (for logging). @@ -227,62 +118,84 @@ class PangeaHandler(CustomGuardrail): Returns: list[dict]: The original response body """ + endpoint = f"{self.api_base}/{api}" + headers = { "Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json", } - try: - verbose_proxy_logger.debug( - f"Pangea Guardrail ({hook_name}): Calling endpoint {self.guardrail_endpoint} with payload: {payload}" - ) - response = await self.async_handler.post( - url=self.guardrail_endpoint, json=payload, headers=headers - ) - response.raise_for_status() # Raise HTTPError for bad responses (4xx or 5xx) - result = response.json() - verbose_proxy_logger.debug( - f"Pangea Guardrail ({hook_name}): Received response: {result}" + verbose_proxy_logger.debug( + f"Pangea Guardrail ({hook_name}): Calling endpoint {endpoint} with payload: {payload}" + ) + + response = await self.async_handler.post( + url=endpoint, json=payload, headers=headers + ) + response.raise_for_status() + + result = response.json() + + if result.get("result", {}).get("blocked"): + verbose_proxy_logger.warning( + f"Pangea Guardrail ({hook_name}): Request blocked. Response: {result}" ) - - # Check if the request was blocked - if result.get("result", {}).get("blocked") is True: - verbose_proxy_logger.warning( - f"Pangea Guardrail ({hook_name}): Request blocked. Response: {result}" - ) - raise HTTPException( - status_code=400, # Bad Request, indicating violation - detail={ - "error": "Violated Pangea guardrail policy", - "guardrail_name": self.guardrail_name, - "pangea_response": result.get("result"), - }, - ) - else: - verbose_proxy_logger.info( - f"Pangea Guardrail ({hook_name}): Request passed. Response: {result.get('result', {}).get('detectors')}" - ) - - return result - - except HTTPException as e: - # Re-raise HTTPException if it's the one we raised for blocking - raise e - except Exception as e: - verbose_proxy_logger.error( - f"Pangea Guardrail ({hook_name}): Error calling API: {e}. Response text: {getattr(e, 'response', None) and getattr(e.response, 'text', None)}" # type: ignore - ) - # Decide if you want to block by default on error, or allow through - # Raising an exception here will block the request. - # To allow through on error, you might just log and return. raise HTTPException( - status_code=500, + status_code=400, # Bad Request, indicating violation detail={ - "error": "Error communicating with Pangea Guardrail", + "error": "Violated Pangea guardrail policy", "guardrail_name": self.guardrail_name, - "exception": str(e), }, - ) from e + ) + verbose_proxy_logger.info( + f"Pangea Guardrail ({hook_name}): Request passed. Response: {result.get('result', {}).get('detectors')}" + ) + + return result + + async def _async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: DualCache, + data: dict, + call_type: str + ): + transformer = None + messages: Any = None + if call_type == "text_completion" or call_type == "atext_completion": + transformer = _TextCompletionRequest(data) + messages = transformer.get_messages() + else: + messages = data.get("messages") + + ai_guard_payload = { + "debug": False, + "input": { + "messages": messages, # type: ignore + "tools": data.get("tools") + }, + "event_type": "input", + } + if self.pangea_input_recipe: + ai_guard_payload["recipe"] = self.pangea_input_recipe + + ai_guard_response = await self._call_pangea_ai_guard( + "v1beta/guard", ai_guard_payload, "async_pre_call_hook" + ) + add_guardrail_to_applied_guardrails_header( + request_data=data, guardrail_name=self.guardrail_name + ) + + if not ai_guard_response.get("result", {}).get("transformed"): + return + + output = ai_guard_response.get("result", {}).get("output", {}) + if call_type == "text_completion" or call_type == "atext_completion": + data = transformer.update_original_body(output["messages"]) # type: ignore + else: + data["messages"] = output["messages"] + return data + @log_guardrail_information async def async_pre_call_hook( @@ -299,50 +212,75 @@ class PangeaHandler(CustomGuardrail): ) return data - transformer = _get_transformer_for_request(data, call_type) - if not transformer: - verbose_proxy_logger.warning( - f"Pangea Guardrail (async_pre_call_hook): Skipping guardrail {self.guardrail_name}" - f" because we cannot determine type of request: call_type '{call_type}'" - ) - return - - messages = transformer.get_messages() - if not messages: - verbose_proxy_logger.warning( - f"Pangea Guardrail (async_pre_call_hook): Skipping guardrail {self.guardrail_name}" - " because messages is empty." - ) - return - - ai_guard_payload = { - "debug": False, # Or make this configurable if needed - "messages": messages, - } - if self.pangea_input_recipe: - ai_guard_payload["recipe"] = self.pangea_input_recipe - - ai_guard_response = await self._call_pangea_guard( - ai_guard_payload, "async_pre_call_hook" - ) - # Add guardrail name to header if passed - add_guardrail_to_applied_guardrails_header( - request_data=data, guardrail_name=self.guardrail_name - ) - prompt_messages = ai_guard_response.get("result", {}).get("prompt_messages", []) - try: - return transformer.update_original_body(prompt_messages) + return await self._async_pre_call_hook(user_api_key_dict, cache, data, call_type) + except HTTPException: + raise except Exception as e: raise HTTPException( status_code=500, detail={ - "error": "Failed to update original request body", + "error": "Error in Pangea Guardrail", "guardrail_name": self.guardrail_name, "exceptions": str(e), - }, + } ) from e + async def _async_post_call_success_hook( + self, + data: dict, + user_api_key_dict: UserAPIKeyAuth, + # This union isn't actually correct -- it can get other response types depending on the API called + response: LLMResponseTypes, + ): + if isinstance(response, TextCompletionResponse): + # Assume the earlier call type as well + input_messages = _TextCompletionRequest(data).get_messages() + if not isinstance(response, ModelResponse): + return + else: + input_messages = data.get("messages") + + if choices := response.get("choices"): + if isinstance(choices, list): + serialized_choices = [] + for c in choices: + if isinstance(c, Choices): + try: + serialized_choices.append(c.model_dump()) + except Exception: + serialized_choices.append(c.dict()) + else: + serialized_choices.append(c) + choices = serialized_choices + + ai_guard_payload = { + "debug": False, + "input": { + "messages": input_messages, + "tools": data.get("tools"), + "choices": choices, + }, + "event_type": "output", + } + + if self.pangea_output_recipe: + ai_guard_payload["recipe"] = self.pangea_output_recipe + + ai_guard_response = await self._call_pangea_ai_guard( + "v1beta/guard", ai_guard_payload, "async_pre_call_hook" + ) + add_guardrail_to_applied_guardrails_header( + request_data=data, guardrail_name=self.guardrail_name + ) + + if not ai_guard_response.get("result", {}).get("transformed"): + return + + output = ai_guard_response.get("result", {}).get("output", {}) + response.choices = output["choices"] + return data + @log_guardrail_information async def async_post_call_success_hook( self, @@ -365,39 +303,18 @@ class PangeaHandler(CustomGuardrail): f"Pangea Guardrail (async_pre_call_hook): Guardrail is disabled {self.guardrail_name}." ) return data - - transformer = _get_transformer_for_response(response) - if not transformer: - verbose_proxy_logger.warning( - f"Pangea Guardrail (async_post_call_success_hook): Skipping guardrail {self.guardrail_name}" - " because we cannot determine type of request" - ) - return - - messages = transformer.get_messages() - verbose_proxy_logger.warning(f"GOT MESSAGES: {messages}") - ai_guard_payload = { - "debug": False, # Or make this configurable if needed - "messages": messages, - } - if self.pangea_output_recipe: - ai_guard_payload["recipe"] = self.pangea_output_recipe - - ai_guard_response = await self._call_pangea_guard( - ai_guard_payload, "post_call_success_hook" - ) - prompt_messages = ai_guard_response.get("result", {}).get("prompt_messages", []) - try: - return transformer.update_original_body(prompt_messages) + return await self._async_post_call_success_hook(data, user_api_key_dict, response) + except HTTPException: + raise except Exception as e: raise HTTPException( status_code=500, detail={ - "error": "Failed to update original response body", + "error": "Error in Pangea Guardrail", "guardrail_name": self.guardrail_name, "exceptions": str(e), - }, + } ) from e @staticmethod From a05330fcd81dbedde61754092df0ace207270117 Mon Sep 17 00:00:00 2001 From: Ryan Means Date: Thu, 14 Aug 2025 13:19:02 -0700 Subject: [PATCH 27/66] Fix unit tests --- .../guardrail_hooks/pangea/pangea.py | 2 +- .../guardrails/guardrail_hooks/test_pangea.py | 138 +++++++++++++++--- 2 files changed, 117 insertions(+), 23 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py b/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py index be4052e4eca..c3649c712b2 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py +++ b/litellm/proxy/guardrails/guardrail_hooks/pangea/pangea.py @@ -279,7 +279,7 @@ class PangeaHandler(CustomGuardrail): output = ai_guard_response.get("result", {}).get("output", {}) response.choices = output["choices"] - return data + return response @log_guardrail_information async def async_post_call_success_hook( diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_pangea.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_pangea.py index 78a686f6724..9d5d6fd54c4 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_pangea.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_pangea.py @@ -75,6 +75,7 @@ async def test_pangea_ai_guard_request_blocked(pangea_guardrail): }, ] } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" with pytest.raises(HTTPException, match="Violated Pangea guardrail policy"): with patch( @@ -82,9 +83,9 @@ async def test_pangea_ai_guard_request_blocked(pangea_guardrail): return_value=httpx.Response( status_code=200, # Mock only tested part of response - json={"result": {"blocked": True, "prompt_messages": data["messages"]}}, + json={"result": {"blocked": True, "transformed": False}}, request=httpx.Request( - method="POST", url=pangea_guardrail.guardrail_endpoint + method="POST", url=guardrail_endpoint, ), ), ) as mock_method: @@ -94,7 +95,52 @@ async def test_pangea_ai_guard_request_blocked(pangea_guardrail): called_kwargs = mock_method.call_args.kwargs assert called_kwargs["json"]["recipe"] == "guard_llm_request" - assert called_kwargs["json"]["messages"] == data["messages"] + assert called_kwargs["json"]["input"]["messages"] == data["messages"] + +@pytest.mark.asyncio +async def test_pangea_ai_guard_request_transformed(pangea_guardrail): + data = { + "messages": [ + {"role": "system", "content": "You are a helpful assistant"}, + { + "role": "user", + "content": "Here is an SSN for one my employees: 078-05-1120", + }, + ] + } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + return_value=httpx.Response( + status_code=200, + # Mock only tested part of response + json={ + "result": { + "blocked": False, + "transformed": True, + "output": { + "messages": [ + {"role": "system", "content": "You are a helpful assistant"}, + { + "role": "user", + "content": "Here is an SSN for one my employees: ", + }, + ] + }, + }, + }, + request=httpx.Request( + method="POST", url=guardrail_endpoint, + ), + ), + ): + request = await pangea_guardrail.async_pre_call_hook( + user_api_key_dict=None, cache=None, data=data, call_type="completion" + ) + + assert request["messages"][1]["content"] == "Here is an SSN for one my employees: " + @pytest.mark.asyncio @@ -109,15 +155,16 @@ async def test_pangea_ai_guard_request_ok(pangea_guardrail): }, ] } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" with patch( "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", return_value=httpx.Response( status_code=200, # Mock only tested part of response - json={"result": {"blocked": False, "prompt_messages": data["messages"]}}, + json={"result": {"blocked": False, "transformed": False}}, request=httpx.Request( - method="POST", url=pangea_guardrail.guardrail_endpoint + method="POST", url=guardrail_endpoint, ), ), ) as mock_method: @@ -127,7 +174,7 @@ async def test_pangea_ai_guard_request_ok(pangea_guardrail): called_kwargs = mock_method.call_args.kwargs assert called_kwargs["json"]["recipe"] == "guard_llm_request" - assert called_kwargs["json"]["messages"] == data["messages"] + assert called_kwargs["json"]["input"]["messages"] == data["messages"] @pytest.mark.asyncio @@ -139,6 +186,7 @@ async def test_pangea_ai_guard_response_blocked(pangea_guardrail): {"role": "user", "content": "Hello"}, ] } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" with pytest.raises(HTTPException, match="Violated Pangea guardrail policy"): with patch( @@ -149,16 +197,11 @@ async def test_pangea_ai_guard_response_blocked(pangea_guardrail): json={ "result": { "blocked": True, - "prompt_messages": [ - { - "role": "assistant", - "content": "Yes, I will leak all my PII for you", - } - ], + "transformed": False, } }, request=httpx.Request( - method="POST", url=pangea_guardrail.guardrail_endpoint + method="POST", url=guardrail_endpoint, ), ), ) as mock_method: @@ -180,7 +223,7 @@ async def test_pangea_ai_guard_response_blocked(pangea_guardrail): called_kwargs = mock_method.call_args.kwargs assert called_kwargs["json"]["recipe"] == "guard_llm_response" assert ( - called_kwargs["json"]["messages"][0]["content"] + called_kwargs["json"]["input"]["choices"][0]["message"]["content"] == "Yes, I will leak all my PII for you" ) @@ -194,6 +237,7 @@ async def test_pangea_ai_guard_response_ok(pangea_guardrail): {"role": "user", "content": "Hello"}, ] } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" with patch( "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", @@ -203,16 +247,11 @@ async def test_pangea_ai_guard_response_ok(pangea_guardrail): json={ "result": { "blocked": False, - "prompt_messages": [ - { - "role": "assistant", - "content": "Yes, I will leak all my PII for you", - } - ], + "transformed": False, } }, request=httpx.Request( - method="POST", url=pangea_guardrail.guardrail_endpoint + method="POST", url=guardrail_endpoint, ), ), ) as mock_method: @@ -234,6 +273,61 @@ async def test_pangea_ai_guard_response_ok(pangea_guardrail): called_kwargs = mock_method.call_args.kwargs assert called_kwargs["json"]["recipe"] == "guard_llm_response" assert ( - called_kwargs["json"]["messages"][0]["content"] + called_kwargs["json"]["input"]["choices"][0]["message"]["content"] == "Yes, I will leak all my PII for you" ) + +@pytest.mark.asyncio +async def test_pangea_ai_guard_response_transformed(pangea_guardrail): + # Content of data isn't that import since its mocked + data = { + "messages": [ + {"role": "system", "content": "You are a helpful assistant"}, + {"role": "user", "content": "Hello"}, + ] + } + guardrail_endpoint = f"{pangea_guardrail.api_base}/v1beta/guard" + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + return_value=httpx.Response( + status_code=200, + # Mock only tested part of response + json={ + "result": { + "blocked": False, + "transformed": True, + "output": { + "messages": data["messages"], + "choices": [ + { + "message": { + "role": "assistant", + "content": "Yes, here is an SSN: ", + }, + }, + ], + }, + }, + }, + request=httpx.Request( + method="POST", url=guardrail_endpoint, + ), + ), + ): + response = await pangea_guardrail.async_post_call_success_hook( + data=data, + user_api_key_dict=None, + response=ModelResponse( + choices=[ + { + "message": { + "role": "assistant", + "content": "Yes, here is an SSN: 078-05-1120", + } + } + ] + ), + ) + + assert response.choices[0]["message"]["content"] == "Yes, here is an SSN: " From 8b66b50c31809eae4f5efe80d3dddaab94951841 Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Tue, 19 Aug 2025 08:15:50 +0200 Subject: [PATCH 28/66] fix: restrict thinking/reasoning_effort parameters to models with extended thinking support Only models in the 4 family and 3-7 family support extended thinking features. Previously all models would incorrectly receive these parameters. Now uses supports_reasoning() to check model registry for actual capability. --- litellm/llms/github_copilot/chat/transformation.py | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index e71e9b610ba..2aab0207ca1 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -81,14 +81,19 @@ class GithubCopilotConfig(OpenAIConfig): """ Get supported OpenAI parameters for GitHub Copilot. - For Anthropic models (like claude-sonnet-4), includes thinking and reasoning parameters. + For Claude models that support extended thinking (Claude 4 family and Claude 3-7), includes thinking and reasoning_effort parameters. For other models, returns standard OpenAI parameters (which may include reasoning_effort for o-series models). """ + from litellm.utils import supports_reasoning + # Get base OpenAI parameters base_params = super().get_supported_openai_params(model) - # Add Claude-specific parameters for Anthropic models - if "claude" in model.lower(): + # Add Claude-specific parameters for models that support extended thinking + if "claude" in model.lower() and supports_reasoning( + model=model, + custom_llm_provider="github_copilot", + ): if "thinking" not in base_params: base_params.append("thinking") # reasoning_effort is not included by parent for Claude models, so add it From 9f82b89051bb2ef5d9938493e8a93b45b1346c23 Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Tue, 19 Aug 2025 08:21:18 +0200 Subject: [PATCH 29/66] fix: remove redundant github_copilot check in get_supported_openai_params The provider_config_manager already handles github_copilot provider through LlmProviders.GITHUB_COPILOT mapping, making the explicit check unnecessary. --- litellm/litellm_core_utils/get_supported_openai_params.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index b71b609cc50..35a3af0939b 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -270,8 +270,6 @@ def get_supported_openai_params( # noqa: PLR0915 return ElevenLabsAudioTranscriptionConfig().get_supported_openai_params( model=model ) - elif custom_llm_provider == "github_copilot": - return litellm.GithubCopilotConfig().get_supported_openai_params(model=model) elif custom_llm_provider in litellm._custom_providers: if request_type == "chat_completion": provider_config = litellm.ProviderConfigManager.get_provider_chat_config( From 9b0fda7b14c3c6237b78e201d5537bce93efb25b Mon Sep 17 00:00:00 2001 From: Tim Elfrink Date: Tue, 19 Aug 2025 08:40:10 +0200 Subject: [PATCH 30/66] fix: resolve case sensitivity and test failures for extended thinking support - Fix supports_reasoning() call to use lowercase model names for proper lookup - Remove custom_llm_provider parameter as model registry entries are provider-agnostic - Update tests to use full model names with date stamps (required for supports_reasoning) - Add test coverage for models without extended thinking support --- .../github_copilot/chat/transformation.py | 3 +- .../test_github_copilot_transformation.py | 34 ++++++++++++------- 2 files changed, 23 insertions(+), 14 deletions(-) diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index 2aab0207ca1..65f83c178a3 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -91,8 +91,7 @@ class GithubCopilotConfig(OpenAIConfig): # Add Claude-specific parameters for models that support extended thinking if "claude" in model.lower() and supports_reasoning( - model=model, - custom_llm_provider="github_copilot", + model=model.lower(), ): if "thinking" not in base_params: base_params.append("thinking") diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py index d389c445526..96da5fb9423 100644 --- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py +++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py @@ -365,18 +365,23 @@ def test_x_initiator_header_system_only_messages(): def test_get_supported_openai_params_claude_model(): - """Test that Claude models support thinking and reasoning parameters.""" + """Test that Claude models with extended thinking support have thinking and reasoning parameters.""" config = GithubCopilotConfig() - # Test Claude model supports thinking and reasoning_effort parameters - supported_params = config.get_supported_openai_params("claude-sonnet-4") + # Test Claude 4 model supports thinking and reasoning_effort parameters + supported_params = config.get_supported_openai_params("claude-sonnet-4-20250514") assert "thinking" in supported_params assert "reasoning_effort" in supported_params - # Test Claude model with different naming - supported_params_claude = config.get_supported_openai_params("claude-3.5-sonnet") - assert "thinking" in supported_params_claude - assert "reasoning_effort" in supported_params_claude + # Test Claude 3-7 model supports thinking and reasoning_effort parameters + supported_params_claude37 = config.get_supported_openai_params("claude-3-7-sonnet-20250219") + assert "thinking" in supported_params_claude37 + assert "reasoning_effort" in supported_params_claude37 + + # Test Claude 3.5 model does NOT support thinking parameters (no extended thinking) + supported_params_claude35 = config.get_supported_openai_params("claude-3.5-sonnet") + assert "thinking" not in supported_params_claude35 + assert "reasoning_effort" not in supported_params_claude35 # Test non-Claude model doesn't include thinking parameters but may include reasoning_effort supported_params_gpt = config.get_supported_openai_params("gpt-4o") @@ -392,15 +397,20 @@ def test_get_supported_openai_params_claude_model(): def test_get_supported_openai_params_case_insensitive(): - """Test that Claude model detection is case-insensitive.""" + """Test that Claude model detection is case-insensitive for models with extended thinking.""" config = GithubCopilotConfig() - # Test uppercase - supported_params_upper = config.get_supported_openai_params("CLAUDE-SONNET-4") + # Test uppercase Claude 4 model with full model name + supported_params_upper = config.get_supported_openai_params("CLAUDE-SONNET-4-20250514") assert "thinking" in supported_params_upper assert "reasoning_effort" in supported_params_upper - # Test mixed case - supported_params_mixed = config.get_supported_openai_params("Claude-3.5-Sonnet") + # Test mixed case Claude 3-7 model (has extended thinking) with full model name + supported_params_mixed = config.get_supported_openai_params("Claude-3-7-Sonnet-20250219") assert "thinking" in supported_params_mixed assert "reasoning_effort" in supported_params_mixed + + # Test that Claude 3.5 models don't have thinking support (case insensitive) + supported_params_35 = config.get_supported_openai_params("CLAUDE-3.5-SONNET") + assert "thinking" not in supported_params_35 + assert "reasoning_effort" not in supported_params_35 From 2fa8f971e04a4b978d5de194dd9968749616ff80 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Sat, 23 Aug 2025 19:13:32 -0400 Subject: [PATCH 31/66] feat: multiple images in openai images/edits endpoint --- docs/my-website/docs/image_edits.md | 62 +++- litellm/images/main.py | 15 +- tests/image_gen_tests/test_image_edits.py | 280 ++++++++++++++++++ .../src/components/chat_ui.tsx | 91 ++++-- .../chat_ui/llm_calls/image_edits.tsx | 72 +++-- 5 files changed, 460 insertions(+), 60 deletions(-) diff --git a/docs/my-website/docs/image_edits.md b/docs/my-website/docs/image_edits.md index f0254032964..246e1c70f0e 100644 --- a/docs/my-website/docs/image_edits.md +++ b/docs/my-website/docs/image_edits.md @@ -4,7 +4,7 @@ import TabItem from '@theme/TabItem'; # /images/edits -LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint. +LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint. Now supports both single and multiple image editing. | Feature | Supported | Notes | |---------|-----------|--------| @@ -13,7 +13,7 @@ LiteLLM provides image editing functionality that maps to OpenAI's `/images/edit | End-user Tracking | ✅ | | | Fallbacks | ✅ | Works between supported models | | Loadbalancing | ✅ | Works between supported models | -| Supported operations | Create image edits | | +| Supported operations | Create image edits | Single and multiple images supported | | Supported LiteLLM SDK Versions | 1.63.8+ | | | Supported LiteLLM Proxy Versions | 1.71.1+ | | | Supported LLM providers | **OpenAI** | Currently only `openai` is supported | @@ -41,6 +41,26 @@ response = litellm.image_edit( print(response) ``` +#### Multiple Images Edit +```python showLineNumbers title="OpenAI Multiple Images Edit" +import litellm + +# Edit multiple images with a prompt +response = litellm.image_edit( + model="gpt-image-1", + image=[ + open("image1.png", "rb"), + open("image2.png", "rb"), + open("image3.png", "rb") + ], + prompt="Apply vintage filter to all images", + n=1, + size="1024x1024" +) + +print(response) +``` + #### Image Edit with Mask ```python showLineNumbers title="OpenAI Image Edit with Mask" import litellm @@ -80,6 +100,30 @@ response = asyncio.run(edit_image()) print(response) ``` +#### Async Multiple Images Edit +```python showLineNumbers title="Async OpenAI Multiple Images Edit" +import litellm +import asyncio + +async def edit_multiple_images(): + response = await litellm.aimage_edit( + model="gpt-image-1", + image=[ + open("portrait1.png", "rb"), + open("portrait2.png", "rb") + ], + prompt="Add professional lighting to the portraits", + n=1, + size="1024x1024", + response_format="url" + ) + return response + +# Run the async function +response = asyncio.run(edit_multiple_images()) +print(response) +``` + #### Image Edit with Custom Parameters ```python showLineNumbers title="OpenAI Image Edit with Custom Parameters" import litellm @@ -163,6 +207,20 @@ curl -X POST "http://localhost:4000/v1/images/edits" \ -F "response_format=url" ``` +#### cURL Multiple Images Example +```bash showLineNumbers title="cURL Multiple Images Edit Request" +curl -X POST "http://localhost:4000/v1/images/edits" \ + -H "Authorization: Bearer your-api-key" \ + -F "model=gpt-image-1" \ + -F "image=@image1.png" \ + -F "image=@image2.png" \ + -F "image=@image3.png" \ + -F "prompt=Apply artistic filter to all images" \ + -F "n=1" \ + -F "size=1024x1024" \ + -F "response_format=url" +``` + diff --git a/litellm/images/main.py b/litellm/images/main.py index 70d9eb41ddd..d6405904c5d 100644 --- a/litellm/images/main.py +++ b/litellm/images/main.py @@ -1,7 +1,7 @@ import asyncio import contextvars from functools import partial -from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast, overload +from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast, overload, List import httpx @@ -675,7 +675,7 @@ def image_variation( @client def image_edit( - image: FileTypes, + image: Union[FileTypes, List[FileTypes]], prompt: str, model: Optional[str] = None, mask: Optional[str] = None, @@ -703,6 +703,9 @@ def image_edit( litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None) _is_async = kwargs.pop("async_call", False) is True + #add images / or return a single image + images = image if isinstance(image, list) else [image] + # get llm provider logic litellm_params = GenericLiteLLMParams(**kwargs) model, custom_llm_provider, _, _ = get_llm_provider( @@ -751,7 +754,7 @@ def image_edit( # Call the handler with _is_async flag instead of directly calling the async handler return base_llm_http_handler.image_edit_handler( model=model, - image=image, + image=images, prompt=prompt, image_edit_provider_config=image_edit_provider_config, image_edit_optional_request_params=image_edit_request_params, @@ -777,7 +780,7 @@ def image_edit( @client async def aimage_edit( - image: FileTypes, + image: Union[FileTypes, List[FileTypes]], model: str, prompt: str, mask: Optional[str] = None, @@ -817,9 +820,11 @@ async def aimage_edit( model=model, api_base=local_vars.get("base_url", None) ) + images = image if isinstance(image, list) else [image] + func = partial( image_edit, - image=image, + image=images, prompt=prompt, mask=mask, model=model, diff --git a/tests/image_gen_tests/test_image_edits.py b/tests/image_gen_tests/test_image_edits.py index c69dc635131..745993499d2 100644 --- a/tests/image_gen_tests/test_image_edits.py +++ b/tests/image_gen_tests/test_image_edits.py @@ -19,6 +19,9 @@ from litellm.utils import ImageResponse from litellm.integrations.custom_logger import CustomLogger from litellm.types.utils import StandardLoggingPayload +# Configure pytest marks to avoid warnings +pytestmark = pytest.mark.asyncio + class TestCustomLogger(CustomLogger): def __init__(self): self.standard_logging_payload: Optional[StandardLoggingPayload] = None @@ -35,6 +38,8 @@ TEST_IMAGES = [ open(os.path.join(pwd, "litellm_site.png"), "rb"), ] +SINGLE_TEST_IMAGE = open(os.path.join(pwd, "ishaan_github.png"), "rb") + def get_test_images_as_bytesio(): """Helper function to get test images as BytesIO objects""" bytesio_images = [] @@ -501,3 +506,278 @@ def test_recraft_image_edit_config(): assert files[0][0] == "image" # Field name (not image[] like OpenAI) assert files[0][1][1] == mock_image # Image data assert files[0][1][2] == "image/png" # Content type + + +@pytest.mark.parametrize("sync_mode", [True, False]) +@pytest.mark.flaky(retries=3, delay=2) +@pytest.mark.asyncio +async def test_multiple_vs_single_image_edit(sync_mode): + """Test that both single and multiple image editing work correctly""" + from litellm import image_edit, aimage_edit + litellm._turn_on_debug() + + try: + prompt = "Add a soft blue tint to the image(s)" + + # Test single image + if sync_mode: + single_result = image_edit( + prompt=prompt, + model="gpt-image-1", + image=SINGLE_TEST_IMAGE, + ) + else: + single_result = await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=SINGLE_TEST_IMAGE, + ) + + print("Single image result:", single_result) + ImageResponse.model_validate(single_result) + + # Test multiple images + if sync_mode: + multiple_result = image_edit( + prompt=prompt, + model="gpt-image-1", + image=TEST_IMAGES, + ) + else: + multiple_result = await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=TEST_IMAGES, + ) + + print("Multiple images result:", multiple_result) + ImageResponse.model_validate(multiple_result) + + # Both should return valid responses + assert single_result is not None + assert multiple_result is not None + assert single_result.data is not None + assert multiple_result.data is not None + assert len(single_result.data) > 0 + assert len(multiple_result.data) > 0 + + except litellm.ContentPolicyViolationError as e: + pytest.skip(f"Content policy violation: {e}") + + +@pytest.mark.flaky(retries=3, delay=2) +@pytest.mark.asyncio +async def test_multiple_image_edit_with_different_formats(): + """Test multiple images editing with different file formats and types""" + from litellm import aimage_edit + litellm._turn_on_debug() + + try: + prompt = "Create a cohesive artistic style across all images" + + # Test with mixed BytesIO and file objects + mixed_images = [ + SINGLE_TEST_IMAGE, # File object + get_test_images_as_bytesio()[1] # BytesIO object + ] + + result = await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=mixed_images, + ) + + print("Mixed format images result:", result) + ImageResponse.model_validate(result) + + assert result is not None + assert result.data is not None + assert len(result.data) > 0 + + # Save result if available + if result.data and result.data[0].b64_json: + image_bytes = base64.b64decode(result.data[0].b64_json) + with open("test_multiple_image_edit_mixed.png", "wb") as f: + f.write(image_bytes) + + except litellm.ContentPolicyViolationError as e: + pytest.skip(f"Content policy violation: {e}") + + +@pytest.mark.flaky(retries=3, delay=2) +@pytest.mark.asyncio +async def test_image_edit_array_handling(): + """Test that the image parameter correctly handles both single items and arrays""" + from litellm import aimage_edit + + # Mock response + mock_response = { + "created": 1589478378, + "data": [ + { + "b64_json": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==" + } + ] + } + + class MockResponse: + def __init__(self, json_data, status_code): + self._json_data = json_data + self.status_code = status_code + self.text = json.dumps(json_data) + + def json(self): + return self._json_data + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + mock_post.return_value = MockResponse(mock_response, 200) + + prompt = "Test prompt" + + # Test 1: Single image (should be converted to list internally) + result1 = await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=SINGLE_TEST_IMAGE, + ) + + # Test 2: Multiple images (already a list) + result2 = await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=TEST_IMAGES, + ) + + # Test 3: Empty list (should fail validation) + with pytest.raises(Exception): + await aimage_edit( + prompt=prompt, + model="gpt-image-1", + image=[], + ) + + # Both valid calls should succeed + ImageResponse.model_validate(result1) + ImageResponse.model_validate(result2) + + # Verify that both calls were made to the API + assert mock_post.call_count == 2 + + +@pytest.mark.asyncio +async def test_openai_transformation_handles_multiple_images(): + """Test that OpenAI transformation correctly handles multiple images in request""" + from litellm.llms.openai.image_edit.transformation import OpenAIImageEditConfig + from litellm.types.router import GenericLiteLLMParams + + config = OpenAIImageEditConfig() + + # Test with multiple images + prompt = "Edit these images" + images = [b"fake_image_1", b"fake_image_2", b"fake_image_3"] + litellm_params = GenericLiteLLMParams(api_key="test_key") + + data, files = config.transform_image_edit_request( + model="gpt-image-1", + prompt=prompt, + image=images, + image_edit_optional_request_params={"n": 1}, + litellm_params=litellm_params, + headers={} + ) + + # Check that data contains the prompt and parameters + assert data["prompt"] == prompt + assert data["model"] == "gpt-image-1" + assert data["n"] == 1 + + # Check that files contains all images with correct field names + assert len(files) == len(images) + for i, file_entry in enumerate(files): + assert file_entry[0] == "image[]" # OpenAI uses image[] for multiple files + assert file_entry[1][1] == images[i] # Image data + assert file_entry[1][2] == "image/png" # Content type + + print(f"Successfully processed {len(images)} images in transformation") + + +@pytest.mark.asyncio +async def test_multiple_image_edit_parameter_validation(): + """Test parameter validation with multiple images""" + from litellm import aimage_edit + + # Mock response + mock_response = { + "created": 1589478378, + "data": [ + { + "b64_json": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==" + } + ] + } + + class MockResponse: + def __init__(self, json_data, status_code): + self._json_data = json_data + self.status_code = status_code + self.text = json.dumps(json_data) + + def json(self): + return self._json_data + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + mock_post.return_value = MockResponse(mock_response, 200) + + # Test with valid parameters + result = await aimage_edit( + prompt="Test prompt", + model="gpt-image-1", + image=TEST_IMAGES, + n=1, + size="1024x1024", + response_format="b64_json" + ) + + ImageResponse.model_validate(result) + + # Verify the request was made with correct parameters + mock_post.assert_called_once() + call_args = mock_post.call_args + + # Check that the request contains the expected data + if 'data' in call_args.kwargs: + form_data = call_args.kwargs['data'] + assert 'model' in form_data + assert 'prompt' in form_data + assert 'n' in form_data + assert form_data['n'] == 1 # Could be int or string depending on implementation print("Parameter validation passed for multiple image edit") + + +@pytest.mark.asyncio +async def test_multiple_image_edit_error_handling(): + """Test error handling with multiple images""" + from litellm import aimage_edit + + # Test with None image (should raise error) + with pytest.raises(Exception): + await aimage_edit( + prompt="Test prompt", + model="gpt-image-1", + image=None, + ) + + # Test with invalid model (should raise error) + with pytest.raises(Exception): + await aimage_edit( + prompt="Test prompt", + model="invalid-model", + image=TEST_IMAGES, + ) + + print("Error handling tests passed for multiple image edit") diff --git a/ui/litellm-dashboard/src/components/chat_ui.tsx b/ui/litellm-dashboard/src/components/chat_ui.tsx index 19029a01f13..b478336f9df 100644 --- a/ui/litellm-dashboard/src/components/chat_ui.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui.tsx @@ -170,8 +170,8 @@ const ChatUI: React.FC = ({ const saved = sessionStorage.getItem('useApiSessionManagement'); return saved ? JSON.parse(saved) : true; // Default to API session management }); - const [uploadedImage, setUploadedImage] = useState(null); - const [imagePreviewUrl, setImagePreviewUrl] = useState(null); + const [uploadedImages, setUploadedImages] = useState([]); + const [imagePreviewUrls, setImagePreviewUrls] = useState([]); const [responsesUploadedImage, setResponsesUploadedImage] = useState(null); const [responsesImagePreviewUrl, setResponsesImagePreviewUrl] = useState(null); const [chatUploadedImage, setChatUploadedImage] = useState(null); @@ -468,18 +468,26 @@ const ChatUI: React.FC = ({ }; const handleImageUpload = (file: File) => { - setUploadedImage(file); + setUploadedImages(prev => [...prev, file]); const previewUrl = URL.createObjectURL(file); - setImagePreviewUrl(previewUrl); + setImagePreviewUrls(prev => [...prev, previewUrl]); return false; // Prevent default upload behavior }; - const handleRemoveImage = () => { - if (imagePreviewUrl) { - URL.revokeObjectURL(imagePreviewUrl); + const handleRemoveImage = (index: number) => { + if (imagePreviewUrls[index]) { + URL.revokeObjectURL(imagePreviewUrls[index]); } - setUploadedImage(null); - setImagePreviewUrl(null); + setUploadedImages(prev => prev.filter((_, i) => i !== index)); + setImagePreviewUrls(prev => prev.filter((_, i) => i !== index)); + }; + + const handleRemoveAllImages = () => { + imagePreviewUrls.forEach(url => { + URL.revokeObjectURL(url); + }); + setUploadedImages([]); + setImagePreviewUrls([]); }; const handleResponsesImageUpload = (file: File): false => { @@ -516,8 +524,8 @@ const ChatUI: React.FC = ({ if (inputMessage.trim() === "") return; // For image edits, require both image and prompt - if (endpointType === EndpointType.IMAGE_EDITS && !uploadedImage) { - NotificationsManager.fromBackend("Please upload an image for editing"); + if (endpointType === EndpointType.IMAGE_EDITS && uploadedImages.length === 0) { + NotificationsManager.fromBackend("Please upload at least one image for editing"); return; } @@ -617,9 +625,9 @@ const ChatUI: React.FC = ({ ); } else if (endpointType === EndpointType.IMAGE_EDITS) { // For image edits - if (uploadedImage) { + if (uploadedImages.length > 0) { await makeOpenAIImageEditsRequest( - uploadedImage, + uploadedImages.length === 1 ? uploadedImages[0] : uploadedImages, inputMessage, (imageUrl, model) => updateImageUI(imageUrl, model), selectedModel, @@ -689,7 +697,7 @@ const ChatUI: React.FC = ({ abortControllerRef.current = null; // Clear image after successful request for image edits if (endpointType === EndpointType.IMAGE_EDITS) { - handleRemoveImage(); + handleRemoveAllImages(); } // Clear image after successful request for responses API if (endpointType === EndpointType.RESPONSES && responsesUploadedImage) { @@ -708,7 +716,7 @@ const ChatUI: React.FC = ({ setChatHistory([]); setMessageTraceId(null); setResponsesSessionId(null); // Clear responses session ID - handleRemoveImage(); // Clear any uploaded images for image edits + handleRemoveAllImages(); // Clear any uploaded images for image edits handleRemoveResponsesImage(); // Clear any uploaded images for responses handleRemoveChatImage(); // Clear any uploaded images for chat completions sessionStorage.removeItem('chatHistory'); @@ -1075,7 +1083,7 @@ const ChatUI: React.FC = ({ {/* Image Upload Section for Image Edits */} {endpointType === EndpointType.IMAGE_EDITS && (
- {!uploadedImage ? ( + {uploadedImages.length === 0 ? ( = ({

-

Click or drag image to upload

+

Click or drag images to upload

- Support for PNG, JPG, JPEG formats + Support for PNG, JPG, JPEG formats. Multiple images supported.

) : ( -
- Upload preview - +
+ {uploadedImages.map((file, index) => ( +
+ {`Upload + +
+ ))} + {/* Add more images button */} +
document.getElementById('additional-image-upload')?.click()}> +
+ +

Add more

+
+ { + const files = Array.from(e.target.files || []); + files.forEach(file => handleImageUpload(file)); + }} + /> +
)}
diff --git a/ui/litellm-dashboard/src/components/chat_ui/llm_calls/image_edits.tsx b/ui/litellm-dashboard/src/components/chat_ui/llm_calls/image_edits.tsx index dd3cf96c3ec..2b100d71b6b 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/llm_calls/image_edits.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/llm_calls/image_edits.tsx @@ -4,7 +4,7 @@ import { getProxyBaseUrl } from "@/components/networking"; import NotificationManager from "@/components/molecules/notifications_manager"; export async function makeOpenAIImageEditsRequest( - imageFile: File, + imageFiles: File | File[], prompt: string, updateUI: (imageUrl: string, model: string) => void, selectedModel: string, @@ -28,34 +28,60 @@ export async function makeOpenAIImageEditsRequest( }); try { - const response = await client.images.edit({ - model: selectedModel, - image: imageFile, - prompt: prompt, - }, { signal }); - - console.log(response.data); + // handle single and multiple images + const imagesToProcess = Array.isArray(imageFiles) ? imageFiles : [imageFiles]; - if (response.data && response.data[0]) { - // Handle either URL or base64 data from response - if (response.data[0].url) { - // Use the URL directly - updateUI(response.data[0].url, selectedModel); - } else if (response.data[0].b64_json) { - // Convert base64 to data URL format - const base64Data = response.data[0].b64_json; - updateUI(`data:image/png;base64,${base64Data}`, selectedModel); - } else { - throw new Error("No image data found in response"); + // For multiple images, we'll make separate API calls for each image + // since OpenAI's edit endpoint processes one image at a time + const results = []; + + for (let i = 0; i < imagesToProcess.length; i++) { + const image = imagesToProcess[i]; + console.log(`Processing image ${i + 1} of ${imagesToProcess.length}`); + + const response = await client.images.edit({ + model: selectedModel, + image: image, + prompt: prompt, + }, { signal }); + + console.log(`Response for image ${i + 1}:`, response.data); + + if (response.data && response.data[0]) { + // Handle either URL or base64 data from response + if (response.data[0].url) { + // Use the URL directly + updateUI(response.data[0].url, selectedModel); + results.push(response.data[0].url); + } else if (response.data[0].b64_json) { + // Convert base64 to data URL format + const base64Data = response.data[0].b64_json; + const dataUrl = `data:image/png;base64,${base64Data}`; + updateUI(dataUrl, selectedModel); + results.push(dataUrl); + } } - } else { - throw new Error("Invalid response format"); } - } catch (error) { + + if (results.length > 1) { + NotificationManager.success(`Successfully processed ${results.length} images`); + } + + } catch (error: any) { + console.error("Error making image edit request:", error); + if (signal?.aborted) { console.log("Image edits request was cancelled"); } else { - NotificationManager.fromBackend(`Error occurred while editing image. Please try again. Error: ${error}`); + let errorMessage = "Failed to edit image(s)"; + + if (error?.error?.message) { + errorMessage = error.error.message; + } else if (error?.message) { + errorMessage = error.message; + } + + NotificationManager.fromBackend(`Image edit failed: ${errorMessage}`); } throw error; // Re-throw to allow the caller to handle the error } From f2bb1ce31ec51d67ae61b89940c77f82e1114fcd Mon Sep 17 00:00:00 2001 From: Toy-97 Date: Sun, 24 Aug 2025 08:30:25 +0800 Subject: [PATCH 32/66] Deepinfra Metadata Update 24082025 Deepseek v3.1 Price Dropped --- model_prices_and_context_window.json | 1706 +++++++++++++------------- 1 file changed, 853 insertions(+), 853 deletions(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a29d03f2c6c..4484afa86aa 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15095,236 +15095,16 @@ "litellm_provider": "ollama", "mode": "completion" }, - "deepinfra/deepseek-ai/DeepSeek-V3": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 3.8e-07, - "output_cost_per_token": 8.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Phind/Phind-CodeLlama-34B-v2": { + "deepinfra/Austism/chronos-hermes-13b-v2": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 6e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 1.5e-08, - "output_cost_per_token": 2e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemma-2-9b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2-7B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/QVQ-72B-Preview": { - "max_tokens": 32000, - "max_input_tokens": 32000, - "max_output_tokens": 32000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-3.3-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2.3e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/Phi-4-multimodal-instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mistralai/Devstral-Small-2507": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 2.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/WizardLM-2-7B": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-V3-0324": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 2.8e-07, - "output_cost_per_token": 8.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 8e-08, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/anthropic/claude-3-7-sonnet-latest": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 3.3e-06, - "output_cost_per_token": 1.65e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-235B-A22B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, + "output_cost_per_token": 1.3e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/WizardLM-2-8x22B": { - "max_tokens": 65536, - "max_input_tokens": 65536, - "max_output_tokens": 65536, - "input_cost_per_token": 4.8e-07, - "output_cost_per_token": 4.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-Guard-4-12B": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1.8e-07, - "output_cost_per_token": 1.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, "deepinfra/Gryphe/MythoMax-L2-13b": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -15335,72 +15115,32 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Llama-3.2-1B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-09, - "output_cost_per_token": 1e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemma-2-27b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 2.7e-07, + "deepinfra/Gryphe/MythoMax-L2-13b-turbo": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 1.3e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 2e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": { - "max_tokens": 65536, - "max_input_tokens": 65536, - "max_output_tokens": 65536, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 6.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2.5-7B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 4e-08, + "deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1e-07, "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", - "supports_tool_choice": false + "supports_tool_choice": true }, - "deepinfra/google/gemini-1.5-flash-8b": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 3.75e-08, - "output_cost_per_token": 1.5e-07, + "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15415,6 +15155,376 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/NovaSky-AI/Sky-T1-32B-Preview": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Phind/Phind-CodeLlama-34B-v2": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/QVQ-72B-Preview": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "max_output_tokens": 32000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/QwQ-32B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/QwQ-32B-Preview": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2-72B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2-7B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2.5-72B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 3.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2.5-7B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 4e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-Coder-7B": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.5e-08, + "output_cost_per_token": 5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-14B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-30B-A3B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 8e-08, + "output_cost_per_token": 2.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-32B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 1.6e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Sao10K/L3-70B-Euryale-v2.1": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3-8B-Lunaris-v1": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/allenai/olmOCR-7B-0725-FP8": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.5e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/anthropic/claude-3-7-sonnet-latest": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 3.3e-06, + "output_cost_per_token": 1.65e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/anthropic/claude-4-opus": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 1.65e-05, + "output_cost_per_token": 8.25e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/anthropic/claude-4-sonnet": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 3.3e-06, + "output_cost_per_token": 1.65e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/bigcode/starcoder2-15b-instruct-v0.1": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.4e-07, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/deepinfra/airoboros-70b": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.18e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/deepseek-ai/DeepSeek-R1": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 4.5e-07, + "output_cost_per_token": 2.15e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-0528": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.15e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { "max_tokens": 131072, "max_input_tokens": 131072, @@ -15425,22 +15535,182 @@ "mode": "chat", "supports_tool_choice": false }, - "deepinfra/meta-llama/Llama-Guard-3-8B": { + "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-Turbo": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 8.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3-0324": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 2.8e-07, + "output_cost_per_token": 8.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3.1": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1e-06, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, + "deepinfra/google/codegemma-7b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 7e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemini-1.5-flash": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-1.5-flash-8b": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 3.75e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.0-flash-001": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.5-flash": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 2.1e-07, + "output_cost_per_token": 1.75e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.5-pro": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 8.75e-07, + "output_cost_per_token": 7e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-1.1-7b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 7e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-2-27b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 2.7e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemma-2-9b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemma-3-12b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 5e-08, - "output_cost_per_token": 8e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-3-27b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 1.7e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-3-4b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 4e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15455,6 +15725,16 @@ "mode": "chat", "supports_tool_choice": false }, + "deepinfra/mattshumer/Reflection-Llama-3.1-70B": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, "deepinfra/meta-llama/Llama-2-13b-chat-hf": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -15465,182 +15745,32 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/anthropic/claude-4-opus": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 1.65e-05, - "output_cost_per_token": 8.25e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openchat/openchat-3.6-8b": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-3-27b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 1.7e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Austism/chronos-hermes-13b-v2": { + "deepinfra/meta-llama/Llama-2-70b-chat-hf": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 1.3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 7.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/QwQ-32B-Preview": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 1.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/anthropic/claude-4-sonnet": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 3.3e-06, - "output_cost_per_token": 1.65e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/Phi-3-medium-4k-instruct": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 1.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mattshumer/Reflection-Llama-3.1-70B": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/openchat/openchat_3.5": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 7.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2.3e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-V3.1": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 3.2e-07, - "output_cost_per_token": 1.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen2.5-Coder-7B": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.5e-08, - "output_cost_per_token": 5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.4e-07, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 8e-07, + "input_cost_per_token": 6.4e-07, "output_cost_per_token": 8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2.18e-06, + "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 4.9e-08, + "output_cost_per_token": 4.9e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/zai-org/GLM-4.5": { + "deepinfra/meta-llama/Llama-3.2-1B-Instruct": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 5.5e-07, - "output_cost_per_token": 2e-06, + "input_cost_per_token": 5e-09, + "output_cost_per_token": 1e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15655,276 +15785,36 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-07, + "deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Llama-3.3-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2.3e-07, "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemini-1.5-flash": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemini-2.5-pro": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 8.75e-07, - "output_cost_per_token": 7e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen3-30B-A3B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 8e-08, - "output_cost_per_token": 2.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/QwQ-32B": { + "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 1.5e-07, + "input_cost_per_token": 3.8e-08, + "output_cost_per_token": 1.2e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/moonshotai/Kimi-K2-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3-70B-Euryale-v2.1": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/microsoft/phi-4-reasoning-plus": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 3.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-3-12b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemini-2.5-flash": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 2.1e-07, - "output_cost_per_token": 1.75e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 4.5e-07, - "output_cost_per_token": 2.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-7B-Instruct-v0.3": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.8e-08, - "output_cost_per_token": 5.4e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2.5-72B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 3.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen3-14B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/allenai/olmOCR-7B-0725-FP8": { - "max_tokens": 16384, - "max_input_tokens": 16384, - "max_output_tokens": 16384, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 1.5e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 4e-07, - "output_cost_per_token": 1.6e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/phi-4": { - "max_tokens": 16384, - "max_input_tokens": 16384, - "max_output_tokens": 16384, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 1.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/zai-org/GLM-4.5-Air": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-07, - "output_cost_per_token": 1.1e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openai/gpt-oss-120b": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 4.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/codegemma-7b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 7e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 3e-07, - "output_cost_per_token": 1.2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-Nemo-Instruct-2407": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 4e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openbmb/MiniCPM-Llama3-V-2_5": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.4e-07, - "output_cost_per_token": 3.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/bigcode/starcoder2-15b-instruct-v0.1": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.5e-07, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "max_tokens": 1048576, "max_input_tokens": 1048576, @@ -15935,6 +15825,16 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, "deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": { "max_tokens": 327680, "max_input_tokens": 327680, @@ -15945,32 +15845,62 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemini-2.0-flash-001": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 1e-07, + "deepinfra/meta-llama/Llama-Guard-3-8B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Llama-Guard-4-12B": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-07, "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Gryphe/MythoMax-L2-13b-turbo": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 1.3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-1.1-7b-it": { + "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 7e-08, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 8e-07, + "output_cost_per_token": 8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2.3e-07, + "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15995,97 +15925,37 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen3-32B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 3e-07, + "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 1.5e-08, + "output_cost_per_token": 2e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Llama-2-70b-chat-hf": { + "deepinfra/microsoft/Phi-3-medium-4k-instruct": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 6.4e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/nvidia/Nemotron-4-340B-Instruct": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 4.2e-06, - "output_cost_per_token": 4.2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-0528": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-Turbo": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/NovaSky-AI/Sky-T1-32B-Preview": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 1.8e-07, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 1.4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, + "deepinfra/microsoft/Phi-4-multimodal-instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 5e-08, "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/mistralai/Mistral-7B-Instruct-v0.1": { + "deepinfra/microsoft/WizardLM-2-7B": { "max_tokens": 32768, "max_input_tokens": 32768, "max_output_tokens": 32768, @@ -16093,64 +15963,64 @@ "output_cost_per_token": 5.5e-08, "litellm_provider": "deepinfra", "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/microsoft/WizardLM-2-8x22B": { + "max_tokens": 65536, + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "input_cost_per_token": 4.8e-07, + "output_cost_per_token": 4.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/microsoft/phi-4": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 1.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen2-72B-Instruct": { + "deepinfra/microsoft/phi-4-reasoning-plus": { "max_tokens": 32768, "max_input_tokens": 32768, "max_output_tokens": 32768, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 5e-07, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 3.5e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Sao10K/L3-8B-Lunaris-v1": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/deepinfra/airoboros-70b": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 9e-07, + "deepinfra/mistralai/Devstral-Small-2505": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 1.2e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemma-3-4b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 4e-08, + "deepinfra/mistralai/Devstral-Small-2507": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 2.8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, + "deepinfra/mistralai/Mistral-7B-Instruct-v0.1": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -16165,35 +16035,115 @@ "mode": "chat", "supports_tool_choice": false }, - "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 3.8e-08, - "output_cost_per_token": 1.2e-07, + "deepinfra/mistralai/Mistral-7B-Instruct-v0.3": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.8e-08, + "output_cost_per_token": 5.4e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/mistralai/Devstral-Small-2505": { + "deepinfra/mistralai/Mistral-Nemo-Instruct-2407": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 4e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": { "max_tokens": 128000, "max_input_tokens": 128000, "max_output_tokens": 128000, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 1.2e-07, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct": { + "deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": { + "max_tokens": 65536, + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 6.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 8e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/moonshotai/Kimi-K2-Instruct": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 4.9e-08, - "output_cost_per_token": 4.9e-08, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, "litellm_provider": "deepinfra", "mode": "chat", - "supports_tool_choice": false + "supports_tool_choice": true + }, + "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/nvidia/Nemotron-4-340B-Instruct": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 4.2e-06, + "output_cost_per_token": 4.2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/openai/gpt-oss-120b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 4.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true }, "deepinfra/openai/gpt-oss-20b": { "max_tokens": 131072, @@ -16205,6 +16155,56 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/openbmb/MiniCPM-Llama3-V-2_5": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.4e-07, + "output_cost_per_token": 3.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/openchat/openchat-3.6-8b": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/openchat/openchat_3.5": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/zai-org/GLM-4.5": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.5e-07, + "output_cost_per_token": 2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/zai-org/GLM-4.5-Air": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 1.1e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, "perplexity/codellama-34b-instruct": { "max_tokens": 16384, "max_input_tokens": 16384, From c444263e7dae340eedc0b9a3234f0950d1653e70 Mon Sep 17 00:00:00 2001 From: Michal Otmianowski Date: Mon, 25 Aug 2025 10:42:12 +0200 Subject: [PATCH 33/66] verify expires field prior to serving cache entry --- litellm/caching/s3_cache.py | 11 ++++- tests/test_litellm/caching/test_s3_cache.py | 51 ++++++++++++++++++++- 2 files changed, 58 insertions(+), 4 deletions(-) diff --git a/litellm/caching/s3_cache.py b/litellm/caching/s3_cache.py index 15f7a5c1e16..e3142ea1359 100644 --- a/litellm/caching/s3_cache.py +++ b/litellm/caching/s3_cache.py @@ -13,6 +13,7 @@ import asyncio import json from functools import partial from typing import Optional +from datetime import datetime from litellm._logging import print_verbose, verbose_logger @@ -72,8 +73,7 @@ class S3Cache(BaseCache): import datetime # Calculate expiration time - expiration_time = datetime.datetime.now() + datetime.timedelta(seconds=ttl) - + expiration_time = datetime.datetime.now(datetime.timezone.utc) + datetime.timedelta(seconds=ttl) # Upload the data to S3 with the calculated expiration time self.s3_client.put_object( Bucket=self.bucket_name, @@ -126,6 +126,13 @@ class S3Cache(BaseCache): ) if cached_response is not None: + if "Expires" in cached_response: + expires_time = cached_response['Expires'] + current_time = datetime.now(expires_time.tzinfo) + + if current_time > expires_time: + return None + # cached_response is in `b{} convert it to ModelResponse cached_response = ( cached_response["Body"].read().decode("utf-8") diff --git a/tests/test_litellm/caching/test_s3_cache.py b/tests/test_litellm/caching/test_s3_cache.py index dce3f7d585d..9c902768bfc 100644 --- a/tests/test_litellm/caching/test_s3_cache.py +++ b/tests/test_litellm/caching/test_s3_cache.py @@ -56,7 +56,7 @@ def test_s3_cache_set_cache_with_ttl(mock_s3_dependencies): assert "max-age=3600" in call_args[1]["CacheControl"] -def test_s3_cache_get_cache(mock_s3_dependencies): +def test_s3_cache_get_cache_no_expires_info_in_response(mock_s3_dependencies): """Test basic get_cache functionality""" cache = S3Cache("test-bucket") @@ -75,6 +75,54 @@ def test_s3_cache_get_cache(mock_s3_dependencies): assert result == {"key": "value", "number": 42} +def test_s3_cache_get_cache_with_expires_valid(mock_s3_dependencies): + """Test get_cache when response contains Expires and cache entry is still valid""" + cache = S3Cache("test-bucket") + + # Create a future expiration time (1 hour from now) + future_time = datetime.datetime.now(datetime.timezone.utc) + datetime.timedelta(hours=1) + + mock_response = { + "Body": MagicMock(), + "Expires": future_time + } + mock_response["Body"].read.return_value = b'{"key": "value", "number": 42}' + cache.s3_client.get_object.return_value = mock_response + + result = cache.get_cache("test_key") + + cache.s3_client.get_object.assert_called_once_with( + Bucket="test-bucket", + Key="test_key" + ) + + # Should return the cached value since it's not expired + assert result == {"key": "value", "number": 42} + + +def test_s3_cache_get_cache_with_expires_expired(mock_s3_dependencies): + """Test get_cache when response contains Expires and cache entry is no longer valid""" + cache = S3Cache("test-bucket") + + # Create a past expiration time (1 hour ago) + past_time = datetime.datetime.now(datetime.timezone.utc) - datetime.timedelta(hours=1) + + mock_response = { + "Body": MagicMock(), + "Expires": past_time + } + mock_response["Body"].read.return_value = b'{"key": "value", "number": 42}' + cache.s3_client.get_object.return_value = mock_response + + result = cache.get_cache("test_key") + + cache.s3_client.get_object.assert_called_once_with( + Bucket="test-bucket", + Key="test_key" + ) + + # Should return None since the cache entry is expired + assert result is None def test_s3_cache_get_cache_not_found(mock_s3_dependencies): """Test get_cache when key is not found""" @@ -126,7 +174,6 @@ def test_s3_cache_initialization(): cache_with_path = S3Cache("test-bucket", s3_path="my/cache/path") assert cache_with_path.key_prefix == "my/cache/path/" - # ============================================================================ # ASYNC TESTS # ============================================================================ From 3b6462236c5b97bb6260acb367a3e41e4b987ebb Mon Sep 17 00:00:00 2001 From: Michal Otmianowski Date: Mon, 25 Aug 2025 11:09:06 +0200 Subject: [PATCH 34/66] clean imports --- litellm/caching/s3_cache.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/litellm/caching/s3_cache.py b/litellm/caching/s3_cache.py index e3142ea1359..180964605f6 100644 --- a/litellm/caching/s3_cache.py +++ b/litellm/caching/s3_cache.py @@ -13,7 +13,7 @@ import asyncio import json from functools import partial from typing import Optional -from datetime import datetime +from datetime import datetime, timezone, timedelta from litellm._logging import print_verbose, verbose_logger @@ -70,10 +70,9 @@ class S3Cache(BaseCache): if ttl is not None: cache_control = f"immutable, max-age={ttl}, s-maxage={ttl}" - import datetime # Calculate expiration time - expiration_time = datetime.datetime.now(datetime.timezone.utc) + datetime.timedelta(seconds=ttl) + expiration_time = datetime.now(timezone.utc) + timedelta(seconds=ttl) # Upload the data to S3 with the calculated expiration time self.s3_client.put_object( Bucket=self.bucket_name, From c1ee8c26af210e8bc7b770b3e5b1128317248e9e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 25 Aug 2025 09:12:50 -0700 Subject: [PATCH 35/66] [UI QA] - Allow setting Team Member RPM/TPM limits when creating a team (#13943) * allow setting team_member_rpm_limit on creating * create_team_member_rate_limits * docs fix * fix img --- docs/my-website/docs/proxy/self_serve.md | 31 ++++++++++++++++++ .../img/create_team_member_rate_limits.png | Bin 0 -> 101843 bytes .../my-website/release_notes/v1.75.8/index.md | 2 +- ui/litellm-dashboard/src/components/teams.tsx | 14 ++++++++ 4 files changed, 46 insertions(+), 1 deletion(-) create mode 100644 docs/my-website/img/create_team_member_rate_limits.png diff --git a/docs/my-website/docs/proxy/self_serve.md b/docs/my-website/docs/proxy/self_serve.md index 815231b59a2..dff55a8ac04 100644 --- a/docs/my-website/docs/proxy/self_serve.md +++ b/docs/my-website/docs/proxy/self_serve.md @@ -309,6 +309,37 @@ curl -X POST '/team/new' \ +### Team Member Rate Limits + +Set a default tpm/rpm limit for an individual team member. + +You can do this when creating a new team, or by updating an existing team. + + + + + + + + + + +```bash +curl -X POST '/team/new' \ +-H 'Authorization: Bearer ' \ +-H 'Content-Type: application/json' \ +-D '{ + "team_alias": "team_1", + "team_member_rpm_limit": 100, + "team_member_tpm_limit": 1000 +}' +``` + + + + + + ### Set default params for new teams When you connect litellm to your SSO provider, litellm can auto-create teams. Use this to set the default `models`, `max_budget`, `budget_duration` for these auto-created teams. diff --git a/docs/my-website/img/create_team_member_rate_limits.png b/docs/my-website/img/create_team_member_rate_limits.png new file mode 100644 index 0000000000000000000000000000000000000000..0c5eba04461423d1ae7b1b329a08f87a0c2d0afb GIT binary patch literal 101843 zcmeFZc|4SB_&+WsL}-yhmI{?+$i6G5RhCl;*&53r%h>lZDN7=h?E5K8cE-MzHOv^< z8BF$M84SiU%lC0kopU~Qn&0xr@1Ng?m)AU=dG7nUmur1r*L8>JXk9;ii1QE)4b9=3 zH-5WALqjh`L$m)Z!$IIqlc(SS@NvNEvgTzPn&Jp1BA5>NpU?cp9Zebhh@+w5a7?QHT?zOiz*7GvL{pRIB5=(>L$~iJ&3@o&AMi`F zkCSGvH{goq);_MkukY+T|LqxC8k%4mngidSc>sLw{X_zPdu{&u+@H4ZpB~d_zdubc zmA3!;_5QPa&2~ttUBCyU;|&8B8k!RVdw=_G-Z{5ILqpqTb5Gw@U-PztIn-VZYyma1 z6!Wxq+-pUnb9#LuW4;^X5}a<+gd-1+U=x8}fq%4e)yT^$ubAP)}@F%L;GsIwJFLS9}TBz_5W z>5?e$gs6*`gDcom)WPNKU!8pK=Qm3ib7vbzR~x7U-(J69GpL)Z@|iPxFZ%oEuW`EC zK>m4?gUh#R0TTr6eFKsZ69@fQGfPjKUo_kM=C5Xd&g-w&DeZNpU~tFM1#0KEH!Kwi zDJi8tU-3Wg{xiV8TKv;M*TKeB<*`p>U^Y^n~mgF4@J1e;r`NPKJZuUkJp z{pX7e{&|sv%q5wBzvy3IeQT-&+S~DeQNUm1_2*rHzEln=f&Qjdl|#9TeHa=VHJY2h zUB2hJZ*h!1$w>Eu>}Q1X@bRNZ83bGzluLD8VmPCyW56=<(rk3VXc1QOO?54Un zFdW)Pdw_wBPwg?y|6cs>BK$8U{J-FWZWmfzWsjU!Denr)Yu++kJWGX+y2IZS-3JI& z{r#|l7S7w$xK6F%$obfm@R^=)KbI-v^Hc`rujF5{^yytNWZd|Y2GuUT7}s6V9}7|n z5hMkSiAl!;bt;diStawIw*9n^~2#97K@2~k@) zp#vpp&f|>(kjs#PEjyob6~nG_jyabW41yFhBE`fKJHq(mO8&_6Z zTHD>%x84|bN>b3stzVbbcShLM%XXr`KyA3zQJzT#K_i8!cXUB(mpm@MXSJMHFY*!j zq)+5y9(?5keVaq=YU^&_;2>Ot5hQz?v(Lad;}OD}8po{*`JbCgqV$bcx!^n}Lxc)2 z&YhyX+F~jP(~2!VKsukaMA9WPad5HDvFt+}~nOGD3@Ik5DX zBap|m(_x^<9HJVble|Z&Nd3?(&W< zaf$9_XzpDj?GqL!)<53%He>PJ=+{$~_T5^+2@2^uIBA=~sVbi0Fc1`63GH@|$O z|0WtG&x9T@i9qE^Xoz+rpR_IhW}MtgBT2dIYg8_JFk$sBeCma4c)X1L$l;rDUQQ(< zNRbl=YJ8ZcbRZ@)-UPX6)*35ypKbEF{)EP~*;sq8QT+1V4o9 zRY$I%R#GMUVs4(Z$t6|Sn_9CcsSe-!>cgG4H>2uHzUJhb#3m_kHrhQHdL9$l>&2AS z67_0h$PN@Iq5>`-9VTyIhMZLLTzTIfC6qeS-R)>}-=8|?!8Fp|y_+Y#JbOK5`vW~M z(qw3D@*$#*>8pED{6@AN?trl|??OmwUak!(#?V3Dv*Yyz4hG4ZTh^-7hCFkDO|l`5 z)WR%o117VT8faDaagG~AbS^e0FX>e{whe8GyDtubBb3%YT1rbx*I2U09i?VobXU+l zFf@dY`{GfoaN=rPmRRkpDyA${veEESLf@iDHnIyd?q|S?7U|}ev-lYKl`!suYpkxW z-e2-mlaZR`lFjTtJ3HGDc2Z(8?vx~?nP3G`?}qq8+OhBA>0T|(x9}nA#3gy6-1t$M zWo^80qD>=mQ!byX5UQXvJF&G~gG-4qx@y1H?%ZfMGT!1Aqgd`?d5j;>d zq>dmrAJZOHJ_XsVWC=%vP~{F4F*fWwd8f(dv8c{xg>S=2Tb$?>_AS$oW#?H=8A9xV0JsVu^k+wnA6ukHPeRmyUe3ux_n z8^)n#$yq&^Wrp~B3qV8>}CoHC509T9e z3Kq-{IsAXo4V<=fskB#1LNZTV#UMQH!tuGVQ z&pub-?l5|f`_Hv??&|4Pm|M{gYf&*t9;;D~=*RKTvPv5!NtOqHTJ}rc=tf>dUc8Yt z)%t=7mSR#`;CGH0(u{E#g-8-+QI03;}qezf@a9W+3 zq~hiK9s>eWTDGm>V|GRwwzj$&owbpE4-C~yxF2w6A)el*E`5CcYx9a8y?N0Zu2CSj z*@Hx%lnj{UzGR49mI^n;L?0Cn7l8TCW&eJ5kVh(80tST`ooZ`BWe)pRH&UpHH9nnH zTJ?jnK{+`&>vxUIMn~G|1_e?FrlV%UH0-(!(hPi`NOjv;TkD&Lpueb5-Hwa*AA)1z zC_E_7e1ql6w5(|}BbbjrA0$NLE^?z}G5nU_#gS<5lIhZVVMR*_3;4NE0##E47@){qWTuA5E4qXw-M;Z10hRRq zREhA52d%-f{-`9PZbLq;1r`ihR+FU44Q%=&v<9rGuC6G7H%EKWPn_w;KB~`4S)wwPKR9JrnF@sVauYYKF8D=8NuKmFJYDFpt9AsBSac`Krfii0QW%sqfi6 zT3Xbdn>p0Eyb<>zR+JFV%0rQXRegzk~YUmHPjjrut&)J34ZUGcu%+HVIs_ zUG^-ZqN3?#ZYFW2Ua|AfsStO}6E)pCR+^e_+4=d0H#XeBYU;2{rikNR_u_RANhwwq zYTkS%?=rC1Jvz#HkU5+6#eS|U#$#zZ#>U5TCS|qxsR;b;koUy5S);XH`8Q|#jwZZf zam8uI$5?1dgLS6fcSi;!-f9ClM#gtm2%BTmtIFUK1?vBNd4E$$6MboMtS&X_rNHs* z^vt72k9yFL_Zm?EC^(Ju!Tco1jeyugS_tQ%Bg&F7~^Il)s=MS0u)F&m2w0Dp| z+D&*g9sRW)Q)4~UJYi)HVMhqhr~?hnB!E2^_w0Vps*9wis=v_r#~HLN-rCkl!_p^+ zisq+RRY^;+tdApE8&1qpp*wGNKFEEEn&+otO4Rn6y1Wvr*?+KaSB09Le4fca&bWnH z!+@_Fw;xsIk{HF2H?aNxph-+jjEc@qw0xQPG=f4IFINo4bWTESAc!W`qH?iS>D7N#&UXJ9}yq3#X%@ zx~HZ9xowYgiBiSbalpt)WMP_LQ@+|`we3R;Qk{fx>r=m`a~gj?5MTEq!E_%DwUKw| zsihgz6RveqpMWCdm7I?*(Uq)M4_+~zGjjGslDj(%0Y z@Kt{?N^EBTTHTZ{TqN4lbax8oiu%RUq{-4f^1)Ac)n&gYo@2I4P?R0(2ILC>fCr9@ zjOfJrR^P4PLGcFclJPeRo-dF+Z|XY*&bV|Yqg-8ysRi{CMs6h4o{M7S8Fuhfx4}b( z4#hXEbWhlUuAZi2kNMEJl}tdFnQawWHfj!%1uI7T<4*?EE39s;jD~`O=OrOQ(t2E{ z2aL@7ixyYF#`%Bf5ptV%M{?7&euXdWq^l9`l=v}}mx8umUQcCiOLF!8U{kKu_yO-{ zR9Wya*u=zhyZl9F9#QgRR8_-g8cTH4)R^D_pg8GB0U7RhOsuX zqN~o0k_&f7YDZ2+rmV~~HHz_K>QPG?)t5F4lD9t~M<)RPI=$HGp@TdeGMi@rHd&}0 z&H3<@J`nKiN1cy_ zi$8tSE)#ff|RFQLR@*(x|P_eo*MqXoB7ft24;H;pf@)4y9ZYzb{HxK6ynR;$V zhfKVMCbN%w;+NbD*Gcbs8*Qvoj-|w}_{z;bLxYa8Rrov<;u?p;9}SW22G4AcJRB=) zDg>!0RwSvqto$w%e6z6brJ>s{wt(R&Czng7Q!7mLbI~t_jEMhZH5bP2m;rv5sB{Rl zjzk&BBNzDaFzhbX875-9eJmGWc;U+$4r{9tuAHoxzt%2ZwGle@_D)~%kdSE8ezU^1 zrf_)yK%3X3xk;SoQL2=v#V`BNElfogv{dr);5dthGqDJ)-fI%8aTEes(oa%INK9;+ zCS(R|rF>cjw0%UgXIt1^b+} z$vh>)n<$#8UY!zGtB79}N?|;f0>5BP8HzeK!u?8afZZ5@)4AU+W9Y!CQitWCYa-Wg z?j%l3-$frndEuJFc2!A>OcQD9;To)CGgG{W>EVzD$7v@lvjiWC*Fk34aZ3VwgJ9^D z!QK?n0KcVx{RAVm+Nd5nCRCTQ%t4u3>+E8rRL&GQF4%hn2lM(|GUmlCsrW2&vTQH< zr56;4RX}BqVOHmC$tN6Y=Ql?^%-(FY1*oeM-mTaBE{6(P)C3XTErrd~NV!;J_7v}( z;Uufk^7&S-094d?YpSYss(4yDKW6Ope@pv|YJ3OdDKKLEXtbS*``Bo0{MvCq@VCR_ zm3_1-XEtkG@)?+-))aRQrgg%zA_)|K{`1j5>uA%Q2e+ip#m!3k=C6 z_Qe&`?C>g;X0{^qK3*?fktsh3PZWAImzaz;s(Bf@&-gD)5Q>NGy3Q64k z%?2W;cV3`j$9!vvw?IsvgJZI&XqA+?ivMyrpfI`xWPw-T=# zxh@kLlQJ_$%PoUhJ5g=%bFYf<`0IXU4~@$mDNiGUwnCOja7A2PL5gC9*O4eY3KCS` zCLJ?p>RXn^VYBMDoo;=dARQEMFjgGS6{SXAOeLMk6-#dQ*l|S^xl2}ELLS}Jn*oIU4^EPjJL_iD23-Stc3ck8l$PvHX%t{)D#KprMnQt|? zPR`pOyf^Fzub*vw5-Dt$stI?WfR+rq?%X>XUd9eNvG`hK6o!Mp8h<0pUTCs7Q_OXo zd}uo7ZQByTZymofG^HIZLVOV|BdBa-{wyB0rhVMUkjIT!f!8aOhqoHc1ygDWA7m=Z z%1VR@Ir&r7)dg}$r0Q}#Md|3Ww#xR*Xe!0~rMM(~cRQ9S122B4!XoT5LV(|2F%GHo zF~@ZgkM18#rA&Hk=OX_b5>?}6KhGkG=zNx*IbkLFVlC-wx`7~JCiQ9)suldZt*ve2 z^$M)QXsP(eF!lC&Vn=gP>(|^9r8_=Wrus$P!Sh6{MRt$${nMv;c5ufgIu^(#MRlI( zHEkS&v7)Jt$n=TqCYD>l>&OCrhuyDE!NEiO%Q7$G-Fdxv`KQeuduf{S`h0-Qq*GK*l@C9{yRHVtnlKKM z%T~slard~*s?yWNi1*0QC2h|v;$X^lwG-a+Ij7697lBY{q`Qx!J7J56 zKY@-oDWPO6ZLyN}L{m6lDyLK3v(>|gyp*YUD4a6TZpxwvbH%uhxnshZu!?Rva%;la zjwKh2F(C+=nY&fU;4<0;V@)m9732x(cxv@-2;XT!0x#fJ6bC$liP|H{JaQjqI&VG89k2$xp znu1sircEqyPNMZ)C!e!CvJ;W>b2<^OxD)BvP4>wQZtgBmujePOt+59Go)Q<(Z{W4s z5;AP3@PdLQuGE;K>u}DP>AFl53dQfoF>*0#+_UDP8w>`^&Uw&nC>K-`YvOo6P6l?( zA@wp7c00ik{0nS!Crr%{4DoY}4~!p{f*Yn*eLMus=C|tnd=P6awKx~9D8jH{)>}Sq zW_Ov1;hmVWcZ9X+g_l201Jv-q z@rG~ew6yH#E0$oc;N};pR)R{VMQBNM-MykwLnaIqkw&m|CyUC0YC*kO@JRbZ$4t)K z;mNzpFPS8#v?P+09Kmu4iszval7#K7XyE$w{i#YTA04~U@aWpmjg->svmL_uY$e*p zM*RLb>8A}`D3(Kq9b0(bl z{Hw}5n40>LbEPPdv@}x?E6TLu9DZ=j({}4qRjf#Rx`4c3eWzs53xm%m8$g5Pje!-w z95B$K9Me|iavyTY757*keX?O}@DZ+UskaFVhCRnx$xA~$hw@aCE4wJhw;NNnSm zbDzU;aiYs~+=Nh7nyulh4O-*ao!gi4Rzv0$?eP(^+W$S#8R71Jamkt|BDKgyq+LQj zdQ)=gw14CAY>$!dLbUfFMY(8bkW!I|W4I1qm7MoEGI%5CG71DL|HLE-=%#f9Xc`xj zh+AH*dLGkr+jq{=&x_o0)hE7nsYN|M+#t*G#wVgewF2*$SmpHElQGpg+%h=PDe)Xm z&RB){)_9rcIwO{)c+l=bN3F&A3F59^YCvypzCv}{*OzBw4zJdwL44JE^)>r;y1Xq1)Y z_Jro5=866Jnr#jz2lVS4^-&m#o2lH+hHTT9Xgws$VOi{bxwrNX>X&-Qrtk}2zY_c2 z*|XAP239*oWzR^laGxwO32@>pR8}=gkelClGT~~@!1=UcIcDw^SQCM7t)5gA)h4ci zs~*isNk@wyZwqVEqRvL_-v1yE-NI*p`_@9qq0aH?>ds5gBn={3Ci#O+y)Dh2)hXdM zm>*?LAD+Q<>!dbRJtU(lu$tImPVCKLgAjt7j=8y2O|->sT$vsydyW9F7Gq41J!~lA z%ln9^l>}SPpxd1Wm6nzY4sJ~BYEAj*PYDIosV@$DjS#vne32*JyCec7dx$lN;s0A6 z>6p+;JPzZBFrTa8gh2Enc^|c!2gb+W#%a+d)P14sa6towQoRb&=sd8*IUUmm>-Wc> zwsrdL&W+YnYVq|_-C-Y${XTPe$Zhu_8h;Ol$>i1cWLs1s87KB^Y7e8;+a3|1g2Mcn z$@hyv4P%@Gya#(f=PR>`9 z5!!%%ig(6LB2Oeb+$*Ym%Y>gpf#lZX=S-L-nz_5X+4dao2)Mh!hs|ZNW}c@$xMG*B zUR^j!ccUNbr|P>=+0+!om6yokrEu0OYu2=2+vnK_H$MX?7>)Gbe=uO<)%d$T7jcXm zySts03uUHXthk=7b){YI+!9?ZQEU%(at==QUMmJijJe;6U0C?)yfAf}8BKA=Cxf$2 zmopZ(S(80ny5H}k+a+&*ZOEuQP0z~nnv>m5WVJD9A%d|}2<+jDv%DWJKfhiy`{+%t z3}i6XUO1%{901*O0l!X#E68!L&)4PwmZtggVb-xXNAf6GR)&<8A-n$*rq~P!l%man}Xoi7j&$Mu`o~6HNkC;-AQQ5t6xq# zg`M)8Ud_s^nW+$zId1iYWLbtz)Q0*U6Wc|lHpP_h!qelZ|E^|!+ac~Cqrr+JYD8V*;Cd1>oIkq;&^^aKO11L_SU zz46i8Z}xt|e}|#`OE|u0;X0PN4L8MMMgIM;A{2Y?bfQDmo6isS6v2}jVxYv$?pG9* zcDdtUpGGL9ESC%lk%fGnBaGKPKyju*^X(_q3%@yQcBJcVP~GlLJD6Tp=L_y5V%x*< z#yL}exOCkJ`ea;|tLsr|s;Q~D7T$8T>5~XDjBs}IsEDP3?7Xqph-|nmqJuaAMVxMh zBCc+=6B^=0GkZ}{JUMg3G)H44nIP8|wdlcRk%kSw#^VaC|G?T_4!f zz!Oh+Q^#kOpNUx|ua>!Ej`-9r*US~rVWwbpo2`b28j+Pv?Lyo~zCQOZIKgf0|KHdG z1Dk#hfJbL^VEksBvipeR7Eb8=ijmT>{gaOAyc*t)v#~zQF!2h{$ynCIsmhQ;I$flK_0ylrcx)DL zoUE1XCFsvqC?;YdotglqxbwR{XkIzqbW*%vJEOc@5H68ic;P`T6kkOlS!~VfMcni` zJ{ylT34RnPP3D#@n_t_)n!&R@zmy2u_Ts8XXm@`Ak27ysOP@mAd_6 zB^H~)>$me18W?0@{h4@2@9vhd+;+g(ChZ1FKHGlEbBBDFOgG;XyigLr@AJ-lG5o&IY`j$_$S+A1{z35|KFf)`-m8uKpz&q=-MFW% zv%XE>%dyRwGPNmuVchac%^td&zlCT?5dFL+ZPP6sG)_j>ew5o( zJFS#m+?kcSM($f#p1zQtks;7LUT`)Fca`~L&4WpgGs`C2YA1MkF`Sg>?)%n8wNQ`P zH|-4+ymdiBv>p$8Ipz@_K)FM3>qNj>>dFt@XUVL9xrhou*#aoCU)D9z2BXPRyI`EV zeJ;zwe@n*4Y8U8s9t63yp0aw29zb+yWhJ{mSaj=uc;oz&_f8X6 zE?qET2{cYq8U19$%l-^}XAxETHlwD^W2Uxv5FqUK0`Ax)$?Rf@!y6?tDG6edt~Z{g zrie)V7KY0SOBE39WZ^wAX?0UWZ+}M@Em}YsZrIl1VNG6t(s@Yc^qx-*NwkX$nqgI1 z_uQI_mOXCAo4WqMzOdQjQ_Q36$;~dX@~`4dMHT>)3Anf%-$z?_o^AgKj4&c8%{b$c|cv5S|Bmj^Z(ztSTd)%f-% zG+^I;&Q}xWdW^P(#Sip-^82f$58!(q)p-B1+X5g~RpVf(O?#k2=iz#XaM0H8zt~C| z|BmC?1;|cksv<8w|8Ki_K#HtTc3z4w{ z>yp>awNUb_Vef70{|*XR=l@-)|3A=_n!SKgz(36Q<&^(6evW_)NecXt)qw00pwdd& z@kG{_0)S_0o`ip|thiUHcEYT@v@|F)vnt)j!NuCf{xJT*9Nj7;5bGOz{eODO7jN^m z#`a71rIWH=&xq2>QG*}>bGFDPE(eUiM3`!oO9$NM&b$!3z-c0mu@_Q)DQ+nB;R{#AwxA0|AKP^C}8qvvJvqw?V?;x`^ zLnrfL@W7%xcxRpU6|duqT-?*Sll!v$0t@O{h+w5t&t za?>5aM#R(m$mnN8Sg0Hf&PI|5HePy;9AW=V)qZ^W520`4AmM8}xFb&~fYXb~Ax* zS}+aHXaP^sPL+^k&MxGDY!+g0kA6@hKvm<(N3DB3+!PshiK;b5ZWaM_6#OAM{M)>A z3TQPoU3w?(F-sa2c-*M@K}7qf9DRnqK3RlV^$O|P1vJ8V1O0>V(daL_{#(LLTk-O? zT9@A5GA}G|Y5(E<`)Fs*u%$*d&j}18qMU8?;C8`k2-R`7!S6Zb`sX7_msKjNv>`ejT0r{q`)CvzxQ~1YwdX92_UGOeb{%IH3&dd65)t*z4 zZNEjvzqU)B?*Wo^#9-!D-&N=9d^O@(CBFv`W&^9MM@sCof_{QvfXpiuPzYMH#Oi$A zr-O6Mm(%7Azb_#l+bk-P17NKiKZn+*8G&_MtqSR>*&FnRhx^}yW_sHoe*3}iy8R29 zpMGS<$7%+Dwp)t-kNN({UVjZk;Ru_xJFSJ}ce;8-N31S-g)6ZAF+=Fqh0pT0E=Y3} ziuun@dJQN3cXxkn0AP0mSR-Hg-JjOu-y2lqNm~|LQB$XJLEL9^v8A8@h$;c;Euc*& z*!*K4c?`&cb@D$e?MY{go}KmX-<_B|fXy3LD5fp?MB3 z(cjMkfNfrWdN436#rsil#DWI>sB5q01W-$)nMmwbXAJ~u{ALpxI>l-NLXHW|#^9~~ zKmjF)N8LwkvWn0uy8#mBt`V}sgZ}g)5FoVJ8mOp`E}$%X zG;qlbTSPs9O^WB;VyIcJw%zK9t$+RI8FQc~B;%<@Tf0)boOL|iRVP%{Mv|Z;pr|VJaRAPXr>$s)Fn@7!*}jhl6MxG zcqvQOFAynVPa+*2MnRfCbqTX`ja}PK1*eK3Z_I=1O#=+ zD#_{E4jBd17V|GwPKCGB=nMYE;MM%%ZX@fCUp>&qd7SlY8K3~Sfd{bS@Z#?*|BuL3 zvpjT9Z`M+GQ&%(ok^r~w%7-d({1h+6Q*QokR1e`6M2{Sg_7z+zix+L1-=0fN1tL6- z`PX@2u5z<+=FT7H0vL}qf`d=Tz;Ar;W!{}d#vfx*T6w=xRtOUfMZi-#_Id;8)jh4X?8XW;dN;RDf&Z@M0@9~_A8jt=e6;P4%HWyKArNZPI2zI z;AM#XmPRaaB(7nrtnc2gq}ysg_fIM#`}RI%sol>oW%AaPI`a|ps%PNBI?>+`)X>TS!_3k4H@go+&kGRo%5E3p*~q@@*r9#xdKr=!4Y z)9v#sW%I3nP(td&5gvLxUehma_&9xR_UbR^i%JGfeC!=Ohbu|X&W1C1j5=Mf#MSR! z(a_My&R?o3G+O*UAzJPn)JLG<`XlYlGxmkb1Ap?2|E**Cka+%yEYsKI{Mu@ATNY!7SQmlTZQQT1;!^eV-D&~(9jRY^0exGij~V;) z&E_J{2JEBaXj88e~gLq-MY1iFC zxQCnPba7TAkYjQP2&0}QPJrNtC9OoaY9OB<2NkUbEd-G^$ECLuvz=b^ax<>P`kF1CpG~T&fXUU5!qJ0 z?vtzB_3Z=;SL27o@~OrR*>O&uZCunu8i~Nr0>3*->3(uVKnCmj=IX^{AlGwZ*@qHU z5OH~Sv^&%)Z?{xHSs%j!z7Fon}3dHq`Daw`T1jMu!aX|ta*T1k{b8o3W$nrNP4 zIlPx)5&p!f2@7TQ7>B_t*Q((Gy4MQX2J~Uki^u7=d@!!pb z7{vsyt%+b4@a%P!wU;Y1Ah>*9*}wz-&FU~*LKt`!(BDJ9v%?%PfCh-Q#RJ8{k%_Y3 zjthC+@c#3Mr;Q4>`6p7~tNkl#N3YG&C$OrnS$OZ0cx(hD%Z=2=75dII*#KPH>JbdB z*?>%xojX7pRAvnb=J8VEu-3Xw+x4zDGdRJ*lMLYBZOsi4lPjg5hD#S}F9Aus*tXOo z*4Iw{>{K$*_SZRTWi1B1KgeYlskjqwmC}y#kZ}Q2`<(msrwG5}<>TH9Qp_({7tBJ` zd+*<=noIF%^k~=}8x>dr6ib!Yw*p8tEZY#chOVi$6>K$!PEECSl^I>!oWk%Nbl;x2 z762qnMq^YHhuw2EdxuZ)BAGQcnZ}n$J-LNh3!G74$W&1|-~*oD3N!I)arI%AVgCt4 z@*h6}BxHJw#%|@kxgNnXyiz0)Hwv6`&^tps#OwLl-*MP7HY2wlpNtFeM3&n0I@d6t z6z7xWwc4|)$$*K#k+;yvv|1ORN?NpXiFE{QIZ?CMm`cZ4lq{BcPlt1^L-?VqD&OE^ z{}lF=ytiqMUD0%gsnhR*DR}S%~DKa>2TU7;fgU^7a*OYbBQvt0L|Q}D-q8Q zP@**)|hB z8lc0nqmJtD5{Axh+)j|XFwJ^F(RLtj1+WkT0pn13)E#4ACF8qj&dX%F)yU!8X^+~o zMi4auh9zF@Gm!<$n-?;j7~T5K}7*l4fKYr_?Q(_6SE ztWQ%ET_agP;#N_Y!a~4YDqO_~uxjRoFtS=6SQuf2&_Gd! z17f|6-xVMH$v(IEPix~(a|BXWrfkT8y_ygWSijPV(%))?I@CU&#n`_*nBX)t{B#q0 zmQjXR)3B^*%W0{+pl(pC z7-^+$vLzojH%4gWg4P??DiDEG;>K=$Q+0J%HEg5*dRFfs)>H9zvJC8{xnP5=z+t6S z_-gdka)X}?D87~S_mLl7w7Vp+s`5I_laqK7#$-io4jbAPU-g`c)>Fur4Dt)+@wS%R z`uw!123OA|sGY(c#LCNa?FFYU+l^Vh>yB*OPBma*`Ns!f<|j&3*!S{*(b}^ z!x8x#26aG?!aW!x3-~iE6;>P5+It5(u<dcX?ZuyWO^0h%xoscHCX62+qvR6nFgm zzRSHL`7hi23)p|Df{VkUdc$-=uc1Lf84pQxIQjLFB3W{?HdKsNh{wnJWEt^e_sB>B zi@|Q856_JV-iqzkDq&!&Z|#-Ci38U zt~6jID;x7tf~vois0Qc(5#YrNf&(})Uk24rKn8F$LMe{+CuQuc<>rT)uE0o7c(dJH zZ!bUmkLqyyL$1Vg*(miq5h4GtZ-cKeOH5_u3T##z*_ZWKSNk*!ffNo&?zZH+zx?J% zmSemTQr*aedQ!Qe1;Sn{SX5GK$$qmAgEXT9}FzB=hF&?M*od;Y3O}!^0?5~=SmcsUjcXSfOy)k6{RKf<0_Pl)p(rRA6 zQTC>(oBydxa9(BO8b8mm&aHf=rha>l&4JsPs)698#V>J2E&mbLA2|-F2F(+~e?v|& znQDutw(+Sc;~96GCmvX#jJge-YwQp0Z{EB(WHGg&PHZNa$w;F#gH!CG*h;^C-TQ~G zpLTy(t>FuWaF4@>Cz^U}R-Gi+lu576o086LaYr7=*tQ&5kK_J2>V5wpmYau+ivqXOOZGyRibw#* z{pzUtAK|T?7@!;I7%2aH3FbPN{~H!%ao{u{85x$v9%=aOIBEnp}3-H+I|hc?6ZGw=sy$r*VKw!fyy+EK|ShcD!8fx2K8a3 zU@BFoY|CaQe8_w&gL>v3v-f{^FyfKt3gW4P6KD1JRhO}GDvYI|;O zZX-~8TQ_n-==)cHkNb7vUyYkzDPO-(A+s603Dlo88N0vuwp9LF2Y~9@D=sb?LO#Ra z!O`y+Bik2K+EuTcz7Bt^??nMTNXg&CAYV6tN)HB>l}Yb~017Lv?*D1YX3qTuetKJN zE*l@Q-YjT9d`D|KuB4^ciy`BC@13y!#PFl;R;yz875sRp*_5MI8z^B`@a@POcKL7I zl*4~3=^TqJ*Pp0uM}|Eq|1>R%?A~S!`sV*VOjjf8TyXMJQ#S^5h=K1 z1-~HLtT|xUm7&XF-`A_;=tn=Wekyb7QMi*FYzJBLD7E$P)&73r{eVw}Hw1li68~y( z_@;Knhiw%Al=QO(e@DN&L3Aam9aFAAh(o;Q-34Tl<(FZ=do}>-ZP4dnZ16aJCo?h^WcHRX^mADIiiyJj>_Wjt8Hd8aNSKHi~ z)$JR;TCbEPLzX_KyjgcmOacWw_@^qBWXQK`u0(#xqhV}pfAW&`!>xz2dEFyDoIxz7 z`x^X>evE^6gAM@o&c#*`U$LX95zSTtE$NVk$LE_uf%7z&q0a@sL$Cl6z1sp#0#X+O z(w}mvyJ)z0&W&9F57?=9Z?-BsY5hP059gVh=YY)X>%J^UQ-ho5&TOM^#`rNMrAq?n zisG@7zp0S*z$EYdo`R?7qrD&9*F|Y#*(E$2T;hNrQk;Z21ORXOfw+?>G1`ncRnPmm z^0yz}lxB9|RU0pOymkNpe1Xb!D;}3wsf;>tda1s^o$2`N17{@@TbUxzny<%!n#}(} z#Mk&J5%{g^KFo6FHrCnN`;jpNTWx-}Ydj&d)2tM)b}K*qqU*y(PlCP^#ou(MaQxXP z3pY1iPNRXod%>-(K{|$a5jP@%Ga!S*l`8!Om6GXe0bZMZCI{6TiIy4P*mTBpE zD&eQsm^cTU!dBZ$i{Pki6t z?{m`-j8rCkOr1r84D!PtBmlZ8<6W-9w?$Xu<@nf$ltYn=K+DkvwWDny1N;v%*Z&av z#}s>u*ui1v@>Cn#-rFfW*89G3ZdxAQ8l^H~!jkN4=eg=^U>er?;{Xn<9Z(wl*mQo% zz#no7%dPo6lJGk5vFcjqQn)qik0?;u8eNVs<)|_Lk6-+QO#-ur=>l?H50~WrU5I_3 z@xK}E3q3GlKgTHJGgSKWuLrUc7IQtcYJQmeK3WZfhvT(g8r#e>rZAv%W>~fdc5&D6 zt^a|YS~b%x4{+*rcYv{7&+A|&?*F4MccRsZs_$0)^#Qja``x<>m7BoHc~55S_?I~d zpx{dY?71=>r5#i;?ky}OEgc9PwRq94xaGu{!^ocb>`*lmQ7HNRjF+jKF)Mn{}K90c^7{fD2kHv3E+@9-JP! zfo%W|?+RKdY*0a_s;VxQbc%tuXJTVbIg=F~>Zu**An35t$RwiVrk}hmu0jIL*fo*=YB5A1t z%d>atpsVrdfI*0xvDZxWUJ}#s!T#am@?+K-X%%Er_kF{>1*0@0vhmP3MQGs*)+5<8 zS($TSbe2n~M_VKugKU1Whyr z2fGaz2;~9GIR6wq35(fu_Wk0Gk{&#s1nEkWr|Z8L?6Z?O0(KGc{TXzD*d(MQo@^TF}?DW zimkk|Ei!HI-`30k8>7*L)JMa~_|y_+sVSHjg+%DpctH?mk}{`xx!vaT(iHH3lE}uA z&eI=>{YzbK)sqoN%q|q-JZJl6KRd3J&-At5p$eis~rZ1D_gj3Y8*>z)Cy? zRzk(ISL!=*P*h4c30mRELMBKHH*5P2xeXe7WGG9>jC)SCy!!*QtFrUuO_hQ>fEDcd zh5#BIpf zs8No7Syq( zep}_$RueBdGU)PZgJrP?7%sX{P~+gaTs3zEsEpAEoSB`|O#rcR0Zw3a5WEFG__|Xy zc(uNH_Za3FF7`@VF6|fz>ZOP7jU_B-Upr0mD0~Cw*qqtT1+2MsW&br2oE3(?6Z)Ma zNmm{cw^r#I$Po>%lT~2eapK@Bo=atecb@LK`=^)fYnI9vB%*Fc31-ZVJbZe%S{i9* z$U5b-cqZoaZ_2|9mW&B=k$!wd=P1Z#)1J0Jva|^ee6R`ZKHywKt>r6ItXgIfc8= zl2mu`*&lu;b3Qa8B};CoLi^RCkEnme0jwY5kvy*QinC)kGGafA{B)#Q2Tg#cChZcN^_;=D zmQvGp!l%aUuT_I~hH^=rgQXg3v`tSRKbc)QFLd|9KE@krv_~T!AE*wZIc?Pmpys~w z;FAiog<(~Kg#e7)m}nlzem@R3s)c(v{(5W9M|^U@b?2})(B}OO!N>HfK|`04nvpM! z1oNluOG7P~M2gf3%^<=e^j}9motI%fuF(|z_&^RLb8vZ)!pt&oeeLKzI@HU13A0!K zeD68FeT>W!{?Y^bZ%%q|0!R()<96Lf>6&cD#dx$jo!$0)6HT z8HdqIZ_nN=ZEa?0u5lCYO5iL~rikAM;?anT?esVfe2UNVJG63(^q?smmZ0S2%&F*Z zx3{^^1$JG0cILB=xwuch>=+yYpm5)p^?LPd(c3o41FOqhou=eF%-|c<@gA?i|A(-* zj*GI}!iFCdWsn#`1*Abh1Zim)1q2lV5do1HLP{Ej7#gKhq-y|?lI|Kna+FT#L1Gvh zhHk!lPI%7yo^#&sujl#cc-P*0t+m$`dtD3awQt;r`bj2=?=Zu5d&OuRI=THrf_Sy! zptNW>HZ7#iZaeTJ<~dJM>~{T1(bbz#+JO|_fO{%QnH$h`Q2~rXUl`Z^aKyUU-$ep# zHi5Z_4hnreeA6z<$YBN4V%WDG({mh9O8{Ad_I6i`m(7Fc_aBbwX)ru~12v7)yx;e{ zW81g=V-XOmEa%%p_IK}*NQW{5$Wd!dqRH2fnI}KYC!X?1-`05-ZJgJ$<4~O+6wBlM z8-U7eM!EOBVu|M=odYTUDfH97iTIJJA9YjPwxa|%dq>g*#_hv}OoG)`1)RgB1qq!d z*cg_2vON=BBWvp}4O<&4e!aEYU;YtQ9(gax_a_`fji9q)_#i12af=w$!&%pCwBJq z(#+{v?K(AXG1&nYeWQ;P*A-IFb%tbBQbdb38`BfS(6)B&!zm}E2slYyWOa~eaJASZ zXCP57<8RrFCb4oP68F_=@5S6nF#n|H=6POLrRLWxu`FIYAREoc0&zVzZwzo!)gBC_ zFl+CX(H_07bk;6Ea}2<62Gc1$rsn_(US(mzc44*^g`8P`0Im#aS4xp@kT7cX|Drl% z(3)_-$e%X-WhPUP&8PAsd@42_YLLr&c{0|_Ir#*EiRJG9*c4cUksRu9o~{@n3-nJD zv3w;F>TS0>jX|@kl(>%;gzjslcfi>UJ5H}+ngF+rjGwRRRzx03txHhlYdhu@9{ zqP6uliwbxU9cfr~B$nV%7^CW*FBTClIW_B=={Vlbyd{B*8T_>)t1$J5F;hR=ATEfM zGgWJty1!Hlk%HN)eueGcAU5qy%^zW&>_~H&%tuM-RBYMe^sF#C=;!t9{pU8!Y^gg` ze7(}b3-wcW&5ai6cdC3wQh~!6G7sDyige`PQT@^Q);oc{9NCCX z<0&@3fT9U%(@pM>q6L%K6AmGh&o{+dduJ)xnt64NZmuk^)oI(3>>j!W)v!dD3^hY_F2M4+HUBc{TAC(E=XAY7t+DO1&rudNf1eU)K8VuJ?()&_ zPt7KTPkY;PI3HM0343;pM(-u;ZU;O#Z(M5jxUZ{Eso2Qij%2!H7WE-eeW(&K*gtCX zDc{{)xpyy@rt?fda%<;oXu=6t$e71B*Gj+c8UsyMM9D(5cyaA~=ocq}=1=5r!ux#n zU1zEcuCL4chQJ2PsCJ_rNwao8_kGx?p5jXbdf)pybF{6$01kl`%y&Rgd$gq=ZUoC= zR~uBDjT&puCIX z-U{iQ8}tLGmy}heD!&?hdZUeBU$qV!G?S% z1KiK`f140&W)~FK-*2vb6Z8|XMK+*OpR6pj)f=kA4jQPQ zC+X*{T^qZSOp8 zVe(iPgnFo?7D;P}L*`8w**BjGuh=LKgzv6=%85yhB6ZAC`AR-?J>lA3a60)KPYJTz z!e{>N=5VRjCa^*mWvya~=5sbOi^5!DsyTP$Q(}A=KD==NqBzZl!`Cp5HbHVcUSa>xZP${EKRsWIlJ^!s9?GMY*J6O^!5Ypc!CokHycw~f<9q(YsB zIUM?%Y0iz0O@m0zFxT6tE+fP+CP+K7fqH827mlGXqF%lZ?`kfCrkqbzbgmwC%>+ep zY;;RtQ)9NCBTHT;9J2&FE;N!0*7I6a#~|DSt9GV)sK?%?DaD4?%hm0q_RVro4wh=o z&b;-CuSvEY8ks%t6ZlpbXKtn!RjT5Zt)Kt?a>wJdD<8V|t=U)su*+={#$ZL<6{1fY)d?O4+V}Q-H_4rF?u<5iNvYH5iT-zTYtc{=zke&Bo|r^jx+>TOUA6B>LRrjqZK#OyK`Ky&-U{V8?S5B}r% zJA~IavsJ`qMELUgpIwEtOLhh94MrNg$mr)CIOUMDMf=xSFaTc|Uz@+a}U9R}bt8URR6Iyx1x*EQDZ9M4P zM)*N1aU)inL|tUj1=BhctHJ?ClskEcgDuMNcHzQN*4`|#NRP17l=|B?G|tArZpUu9 zPTt)@?WZvFSZQRHlaE5fGZ}0W$}5qcaY-G!UC!f*#9}*7;vTldRnvBF<+>Schg-he zj_?qaRb^Ez=7^9E`mN>7xt*yxa;>#G zoL|Tf@2#-N2qw#?rt_bsAb>1LY1iN^TigunJi|26@E=_LX>QGy`!hvx=J1Ax`7<4+DU$WX9ogzH^27@mXk4tIdy z48m}{Um6$OJbiykwJSkR22aj^k!M}V9tI2hG)H1ZzSv7^=Ya+0+Qv$?)@<~~);}NK zM|erA{jS2q>-HU5XH6enQiYj&Rm^!H2=5Ar5Z)9C3T!BQ{`WE)Sbmf$zN)?gk_IV) zl58uVI+xemrZ^>EA1kH!D6X7g0FB&H4d9U=modj|B1tVbpuq6$&X=P5WH8C?tSnXq zX+i^#*@w5u$e$RIp|wLM-+s4kNHkwO-5g(C5iL1szypZsa5 ztQKsJ0(UiSJ#jBkR}_hknoc)$nSb4khp7&hJ!;a4nv1LD3w-`n!!(G|CC{~Tp^g@z z3S;|hCMn#lNF-vQG&d27ngMpZZ(z3vem>;_BN7LRQ4WN6Tf3#+ch4$3|8dM3MKHb)a1WwwQ&EI#*!o8ajBd)vGte6I zKvwU-Gb_0JXQwNcZUm+A@vJ@}4N^IFgaqa(o<~af&Z(iag1kBGYMWvWmrSZ*OtGofkvteT32Os=V6`BBM^9n7GV(dINKg`; zpS#^XQF#a4=7c{tWWPHvmn+@V5vS*Ciq@Iv6AtigqZb9v0okpESS}@faD~wx-m{@h zLk2sIG?ZiCq$;2OIL2h&n6m8d+A!0(*E976ITKzPn*bXLV$tz7bAqsPb~t_Wd|)Hf z&l2w5E6j>GoQA-D4;$e0a&5kBv@DB&qZ6wazF!3wh9O7DHe=ilZqbkqq?@I8M0~-z zCkEV$_Zg96ax{^$qnW4sE5@yF2@C!u*{_D6gVzbp6WXvczh-^?65J|z%y-jh$(VDC zQ6d*P6S2E%$~82+?iAxTRkRCF)-nGYhHNS=Npr0+OUG>FG(a}4!r#9uk`<;881A_+ zqf;M)Yqxdv-X+g&VY85+I=q4}LmDxR$uA+C;b9zGQjTig1;hoIYZ?{~^Rx48eUC z_2t>4D>G__FAli|Z~j>>!0u4&3f$c1j2*4b)Y|>g=N{$Iml)pS1(K=lZ%< z^{(W%$BZ2~G)(j^e#c7T>*W$S_yO@L%6(x4YNKbKfQ;-?_^fq)4BzJq^N+~!WF*E5s}Emm;?lZIo!n)ug>v28f$L%Ut3e4APsyB4;8bi5dnOM5ax z+P#%+s0as-3I{~7MbtiVJrfP^O{HY0YbDs?UoHN9cM!Nbx zbpo;DNQ0|6B2mYn7J1WHdehf@PUtAP*C9fj5m#}1?L=(2L6R+P%t4HZ&`t9==le26 z(qbQ#3Y=xtAjaAkvbRvR+Vc`an+>}Dng-P=o3(aRmdst2KWd0plaf$_tM1b=$gsi9 zxCn5!8*|5B7zX#eE53- z^pwR{Ots*)91A0$(lD0@KjNZO_pOr%YHii{%HZm@D;NuX!v~y7K-j&*59V(cZM87} z`N`P`_99{7os{II9Ls{I7y$ilI`yQc9=*k1c_FR%PS zcqaYAoxqAv@~~;AhmtSFOVpJF=E?fz)6U88}&7?bNV~b!Tvam?DUs(&h z{Z~zIVLBy@N)6vMB6$nCB2FLhXVuEij;#j~JJ|TdU)9uNZEdUK^;dAGOinDhf0!3$ z!dn;B-FZ%VNXclfd8o@rfjQ(le62a5uIiQ*&_Q&gY-UU-Dl1Mjqr^UFN8Gx=8rMNn zvX#>mRh+CnTwscHr$ZUjp-iOW&+8yxX4=;zeb6swjosSL+TwyM-Y4yv zCT^DFq+5oYyV&}r`G0FPnJ>SNj8fmJ7B3l#Xa_<=UT96qL4bI)&qynZXAm9a%AT5K zQ1NxWwATCHj>7bDkaF~O#KV?!m`3sVBP0pN+330GnfuoO4jZVZ?rg|j<;k>lO&|8y z@-yr=}j&oKLb?q=}|mM$rhOyNfv+&;6R0&bm39!y&|9C+h3`^BW| z+D_bSmc00_OX)EEUlna+SFkgo&_z?0HJehyXVSRKzvfLVl^L=Sok5hn;V-^FOa{{R z9wCFH%X0VkoBwEw;q$c;%NFl_4tIPZ%IA36vyWf0!m%aWbMujnSg3G&zBC%y z7QxDODkOQ$z1s&mhQcjBH9pPDiVU5J>!t|W^~7D1Xgg;`%#DtP&RrWx(R4l-m>Brh zK{E!jIWDdED$~xqba11)o2^92K{}o^;YE-`mTIneNwc!}@65K+7?`232v^#ux=XMv z%r~06%P+DVEm^UW@%Ynkz1A@DWV4}(U?okRhtvi>{dB?Q^Xy?3COfQUx++%a7TU3; z)`7UHxw)f#v${uZa8KUvrCB5Tka<53Gm5f-s19mTj5Hf%d~~Ijy>lLaOi|=02NIjtz8eeM7eX9Re+7`c;eGq3#UFd6FfAWfo96o4JPB zbKBAEeJ}OM^HD@dS^*)R7K0ay4-VJgbEXEb|FnxSl4 zjU74|$|>C=dFOC2eA{gC_!e_j`!>S!`-if3%>5w3UW$Y}@DF5K2JQ>#32SND5*7A#CjBmejUVa zg&{Ahsi4AGa?va^y4Ydam;2QW;3-ev??D2d>mB#_SS>p3D|#beyu4L?3M|eRU@t3i z)Dxlpr$adKUES>3&c|ehv&%lLOdNJGVs&i$nQ}w z#tB@O=7DkWX16QZ=Cj>b-dLXIDmHr{`8sb5^it6EZBLORqmth5FmL2v+!~|+pS)ln zU`!T-CE?+YFDBQcbR`GGqx!am`^2**-jN-{uc~Va`g0hVHdYC4ypuhAYG~W1C||4@ z-~|-Gqa804tX2u-JVBbChL0HSMkj5qzKzJ$?ca(wzj0_Y-EXR;4yG)20w3?dF&h_X z=0RI3$VyhzlU!TUXS$IUNHQpbzfwOsYDdCan{naoCS9aTnq*ZXSLN#DH?PGs7lBKQ zZ=MDd^d+9il+)sa8S@+Agnstiqj#2v-^hERe_YPlT z>Sip&w~t=l8$vn$KL+^R%gm8kQ%Cll)hJH!bfNV}L*WmJbg*rpg<`#dJ)6Qm=UmPIGr({@g0MS`EQzW{XLl zEL+lrIsw^{J@!ICp3tXF(G<(+dib$kpQVKYe?62Z->M6 zO|Y_?^J|1)?Dc?B&^S+8+Rpf*XStmm%+#96E+tvfedQVK1NE=ly4LE!HNgX{ z?uv+$=V}e8g2YPt+hhB=h3;WuU9qZ_ifPHPk=@6NJcm4AoZPO~wBG5jShj`&?=dX7 zR;U3ZRPzJh%CfXv1up zg!BOuGYe~B{8T)8=;p4#;X#irjJcy~I;`l8w1#CEJ)BkS{j6?CZkkiUW?NHh2)L{G1&I6jy#dS-`tyZ-R zW?4~>ud~TuYP+X4ejPiCYfh!PyB2(F104Q}t#ZBcy4X)oc6-9HtAHSe=Ob3W#;mXh zq}J_raro;%p%IISa;-3%x7ts=5a?QXkEu(TRU1^q7?nlV@b};G#v#!8KvXUmbn&!S zJoBWbW~UxNRPU&(XKl~$1G@kwASkz(*u21lRe!774BM%{YlKH;m*Sf+TBmx7xqKg5 z(=YaNkNdFQ=EyjKJ`BmWst&r2g2_D%O(6 zsKhx;YrrT9PqOYn26QMMV55A_Od7K|wlaQ7skSK#HSo>=9oAzr3bL>%o?208jB!~PL0fVR5JGIy02yemr-Dt zFIZhRyZp(Oo~jBt--7`lT`d2Wvy`MApse}Tx>n>VG~W#n>~*J|SoN?Qk7XJ3^^|pK z*<9AcO#7N+Y##Zm=@;A(iv$Om`lNHW)QaJ9_{rTqET&aE$TH@aK~_x})~qH%PT40_ zmS%dVdbO4;!$d#3KX2rxcfKcm(T)0IO(FGEeu9E*TWFrn8IaVCfU!1$sBT3MSAY|c z;0hjh8KGpezX6kmJ5ZsT?r}yYhG=*Nov3v9l9I^FHM5SPM1yCeG)o7SmvWm#3bb zJ~zoY!!VFvGV%Fk&(!e7zqSgcY>#poFBdzad~V=jCm7Im%$_~pblpd?m{kv{pVp5x z7H`7klsJrr)I;CMEB(cog45jMOtF3yl}pW2=j;Eeg{B<`cJ*mPnUH5WB}fm1sO0iM zLuS2CH#*MXZsUW|Bi39Tmq$V3D2f6YXt&=L_HJN>{xV~@?kN#}8Z|?1R#c=qEwvpV zX^=U~O18#_l3u#NFrntk*;6|!T5D~DhUYTFZ z&qYkeA6m4}T1?D(a7y0+F&|7bAIpr{>O6eTf11xT$jNE|1z@)hFD3eqxs$-8*S0f3 z@8yW?-yZQqiH;k7G-kHB$>=g2GrcgvA4rwMV1cT|wl<`m%1FcVJs5Ire6JopoPWI_{Lij)#JywL9<>15iXBOqy)cZweQGbYN$f-hATS2-kR6--K)~IW6_{9u<`j zHW}U z$bQIjU|cy%spPY2L-~f@XXi4)t=a^0Xp102F?m>@^xApH@HHLHiyGsoh@To}5W_Yr z4*XQ}=KxZ`KLac58UbO9q5)0xcRa>bCCN<8^c=4f9t`P+B)JXZc%}eMEA7k263@aJ zn~S>0?X6Mf8P?MQ<`+U{&AkaYxf!`TCY{fX4&Lx?_+(v)myN ztk5D?BugxF>Am(IdG$zjr;yb)A1c+&d?h0(Yn~7hNL}ue+DgR>KIT7vGZ#gEVV43Y zlurOa;|+pP@@>492cief-NbBBxq%(PtY2`catuoe^l~@SR&WHby~Cr_Te}f1jnu!! zyvDT|V5_^<#JWZSw~VXa#Md}cW}reo)>7`*tJmSkX%Sx0g??@{^Af$g$x$m^Tb%ZC zM_8|ii1g^&@mHe}(q}vF4bDvrs-rGG=$RAu_`mPLHR0i28?wUi_*YzRTuUa^wA`$% z8yh-PH0)@4OA!1SPsY60Og94}sp^0?0S}tbRLA!6s8SHANmwOQ@YajNEH)Rl7A%2#hzOIWu5raPNVVeqYW@vZoJX&bra z>PY_XwGPOZTOIL_n~neU0suHm2pK|M$>IY*eJyv78>*leYxia`VWa725!<<+rniK6 z5^H3)GGX6FMh*RpHGo9lmK3EJoqH*|qb3nw(kAse@_0Mhy6S?p9K&WhYX5G|?^$wf ze^+2)H6Z=XaYTtZe~iTFed_9FgYpN_O$`oe^wEqRxb(O0N5sP@UR~}WeJP=!L0kVuX&9!e` z?^!1Q8u~7yP`9gVRaCX}uBwrVw0l1<>95mzPKd$M$fj@rOkF8axt6y_BxXsRh+lnG zq0Vz6XSzVP>Blbh-0KIY9zFx;0E~?SP-w*JTVs`xV~2nw{g`%8)wcQkKxS1<0-D-Z zHRq5NMBnd$kiZ<}CkPbPM`GVzq~1q`P=u=IL?F`OlE#gCCB~5Mrp>mq>2LeU#@{;M zg;{HMguk3ol49fOI`$(DoRBrrz<=Yv`kAX9!>?&K<0ieXrTQ4GObnWP?tRfDumVRQE#NC%Z!Cr3Gox&-uJ}SV> zmAyhT@;Npv+1jMcZhg8}#2f9HKk3-Q9wl9#n93>Xw!^UosLPL#!wo*T?JTvgm;~MW zQ|kBs2GibS|32X4COJs=Bo_u!yc$8ECe<{JNk==PjC~fC^poa}wx7Q|zhtStG4X++GLzd@vPetn&|}q3@)@G&&pK~=N%%kG9juxP2p$i z7mP24#q>w#Q{EG$ALgQ~BhUgZ>V|A)#o3WVlYRT~xoXTGerlqQ zRwb99x6u1Hzyo%R;3C9YF>w|4uHZEdaxg@_*>R+F4OZBUVynQfhCRJR}YExr< zFTpc%;ajN*)N{SXMRw7dVQZ^A>9n-j0k*KeQK?(H+TK9nRMR>Ah3HoxPk*`>NKNpC z^z`;;t9NP7)v~q=4r+9<1+86&(FrcIGs*COBoO$;&c4g4fnGZJoar=KN3<;uvhtpj ztSEIjy|VIjI*Yve1TmfNK6L}UWD%LUNQcQ7Y#?9BEViQh^vIR!<-mHMqoC$LtCqj@ zYaj6wCSs%DwwnOJIj?N5I%AO);a=A&|F<6rye0nqgjVYz?>h!32~QBq({??tX26ml zbWWCA7g>sDderB#wVh~@jY@esz6TD!^t+&V)ZWwaqJ0t__;yQ8&wvNmhe`wcP(O2P z(*N_uLB!MUYeg$SDkuu-V#q!R8a+&(c>aFhc@W+K;t67V(atpZCnb?Fp=ogWJH)aw ze`96heP31!6WYlioK79KRij!eU~cBII-u8~NKO!#L?+)e4OG+n-$_~cI0Q(E?^%$y z$qE1b_gm9$5azr{dPQp4G4P(JuvBg?NpLbL(FZ8*mWc%_=cUlNNH*4<-fzxC1YO24 zDx%MT%E8$6?vK|n|L<$i%M)Yq2|$H;4^P08O~(p#^J`GsW~zCkD^sbiD$whY#%gKddAGZ@@wjRKC_H z_hj5S_?68frJ?OxhyU}=fBy8lJqNJS)Kl(*{aYpe zKdx_hlj4p2&Ba09e~%f#-vK)7bmk{g#sBspNoD-5-F{=ZOl@@I|a!c&5*HXND<-QWErX6BpEJlyU5LYqs>1wp+9fY7V*~(Cb7(9=z|F=hj;;% zO-$XHdyR%c_no{$g@>}n;|h=N&R77>{l|x3Ao@jae);=fDS8$ylBeR3O-*cqWK z5J4v_dQ7?Si6)x(U(fT8e?-8t{f{i@{+Aq+!Q)5;a#Goc=6&lTPD_Pf`b{Qd`fq+% zVhiJa4N477p3zadCDKr#_ag4kr>1>%1(bA#@LsB$B)RAju^ei(oD1M2w@eP#`)}W9 zFX1m{^Qa=Y`A&+sgWOwd$LIac+s5uoNh?+0TGVJYO}^Ln#z(j{J3WJ0i7O_OG_kCI z-jIM8U@?M{+)QKzH&#SkdJ-$wg_Q4`Mf!hu2>$og`j3}G*Fi^yyIzMMk>+#ph+Im3UZ0C+wa@Walff{y)!Q$mHk7 z>SMXoBVzSyy>eTwFU1n`f6v9gE%m=$hU!G+3;2nUygab7Eo$it{yfd{oh!&s{fUCY zDyia4RzvUP{>w;xdL%}X(d_R{eWF&G$quY{*m{J{pQ%X7?RWOX^H-scW$TfHd@>U| z5JUg-A~TA<7r;Zb?p(x$2B31y&-OA&pWF!-=D4O}^{)^6=~1$LRp1}jhyHcW(Svw)86pb9k7;D1;QxQfdvwK% z0!a8*KL!1nYJmLU_gDaX9jrIyzkH^&yM~bGT$a_he|-Rsj+^-V6Z*|5{l2`|x*P)R z1a$AF?w4veu4;xbAKf3oJ<`_?P-TF^ zg|iX!z)pR#{yB0vJ}rm(Z)y;6*yUsS*`WcWNCL zhCYypLRC(kHVlQ>((pQnA;91wQTCuLDjN`dy%)%`lj0kaG3g8kXm9-4fYICm$lPzT z=Lg11MwQk-0&gO=0+>)e$6F(uOpY~)f&|G8ZUT&G*C9*-6yXB+VZeU-5#6Q~MiMxK ztdt^e4QV72BB@a*oUrFha5!KUiF0S)WdRrk#3; zK5P<`0SHr1Ce#DZtAzuXH;2BiSRX!G;4i=4K#KjZbEUpK>t zFj+hFgeFQ`o8r-3i&4Cc0ftR;t#+R%m1KUr3J8VlqnBI(bCwmrSnApE%H<*#BnfEd zrOuz(`J7G#91ZVvd~{>GG5{a|uiK9RwP2TnZewoCBvUaMEbQ`{?fYKwtO`|y&d$bT$f(lx$Jzh$onVROB=5L$;{9W`w z&Z%u)uWpl37Bug~m3)PsuWHA&0BR=DmCN(y%LD-EDPux0d+iMxG~P*g`#!%CP||S~ z&jMJ%Fi=SiFPw~z1Ey@lG(edN8#fBduIQ62$ozc=EB2P{C7W3w>pQlW@kPA$LU$h+ zb9U~#Tg8Xe{qM5E&;sG`~tj9-qR^q*1UsC>>16*SI$X0}zXw%PzoQ5y&4E*?AREP=)-Z z)&)4_4O?c%X+0-(I^S*s=bA)HaHavae;q{ELBHHyYE5v-;zBAX%u;RzQJAE=ujcOo zrcJk)=@n!t+$fvrY$e136VSjayTSibVH%JzHermyodl`X8#UXkVwmUrbIqmZD({qF z{~y8akqJD2*}lWc&BTHO26y*d1pi1m)i z_$1ynONQ?a(Ep=C$Hj1}+>e|AwkyO0rw>cuDp9-R-`mQ+*-QauIm33|yD4(_FvzS( z$@?|{=`jGsIb}4$8d#Nmq33hH@9ZEFd074K$M`mI*vR<(FsGf}?H*qG(|LyCe92tT zh^arh;NP+h{(m2t6HA|zaIF&X9foGFtrdsM;_LI;o+J-kxNzS+^v9fhw%_A==QO*oA&u$F64m=-ZaPZ zV0~f*@R-s+0o(@Ptm~SmIa8e`O;I;2L`vOID+Jv5>^h_a@$#afso9GIrdDkSV-g?Q z;vGP5xaREr$iKoYD90v_e!dTYFR6D=8#kRxcy=y>AZM)jXm_mvush1#US?B{Z7cCP zUo&P>LJn`gY}bu&T&^E^;pp~furqL z!6gzyQ&9Ou|4}?hk{%_{de>(J5~$(@P@owbJa{Q{=AO=c*^J$|INjn$V)IP!4pVOd zStONw10IFNXke#uWZAz7eqEtflU4^g_8f*N%>%O!!3W%RK3UCcRU?0$EMi%S?)>7g znctsp^$hSJ{k)y52f&mb?YEql@Im?@#$1|3aEaDyY6`7Yc(@;QmgaMD++iuTUMdgG zt?^>$-`-~PLP?=d4pT`+`f!?AYVni&5FQ>^m~t3a5xZsVE(;Sh67J`8_Vw@;<>3!g zbYQ;|qzTt>WFVl97NX&rh_d)uKCKv#VCME-@d(3l$R_V^LK+vA@Qw3tRnArU-$1}o zgIrAf3~^18N7txrw3R%&O0E7R&piFIVQ-)P3X1P9UC8&eoPy`_Gy!qk`HjcatR#aOUF_jUlrn+}Lddmmh5ELXzW}pK z{X(I;T&|ui)an{gIiMP~{hHye%eXK~U}fJVQ8_YU&SenRUEqx-8-cD%FpK zA89`)L98}TDQb{T&G{k^IyQ@BX*F`wV4LIxs*kuYmNKokwJ3-w9m`bRZ_U@;2rJ!g zZ5vLMR6h34wNIs1_J?<|Y|^bl5Ebhv9%^+F|MObTcsCA>L-`&(ATv`Lq`48ssC5NP zqDl){zY1b`90E#1h|GI@Lf(uO3NF5B)Doy9yplOxuzAWv^N!wBU5Gs39-ZmU(+NPIwNTcKETocNNM$}>R$2iJ0(eD@ zjsb&J8FGAgO6baPj!Yc+>EvO#%xFOvFY@Kdk~T!ctMq4M<;+=PN`fn+%_2qZdB2a! z;pJ{7n|ux0MSBt!vy@(Np1qyWu=NZKc}1VO5K%x@mvIcZ3}-h0hk%L{YoTt5oa&Yv z0ECbKv-#2Hhi|*w%YJnS0+hlW3eu!@)nuaM8Qz`vIxYArhyXve*LI)F?)C`pXPQth zF8eYAx?cxItk;S%^f?#8&6qWGJ|DGRr5SjS&cKz3+0$OIhNG2mo_>BD@GsP1Hi%zf zKKc%jV<}sFS*2FqSa)ovvg@^HX{gCO#Me>jTaHL}LBgbRC8)Qf)b!4p4od!3^g~jk zn7V6o5p&$3{oddV*3!y{fSa}K6H*DcXm*=yW z`Xk-qSfHECgp5(R_@^_6Ik&HuPPfS%;HSe9m77H4 z^vzIJ&?g9FC44V;YU{Zj+Hj_F%Rox>RtXmw_)0;uQC#Jbam;WxfB!S9LZD%OI2V2O z-vsD?Qo|!D!nYa;lOZiulFxlJrQ2diDYVzt!f)8ngCs5OotQPZ` zIvf^k?M#4$^=>v}2?W`?P5N&0CxwVqG|dwCJBywR+7H_7CU4%fH>=ViV%=CwF886A z4`;UR6KiNH*{_x6~gw{ha4^s~h0t6U5L>;d#m{z`Rb%K$Z%MjXW2NuI!9$9-{eH&K zs34cJFCL`vr5d*|m+>QeTVH}Lk7${%f84!H=RZ&xZbK=LC zFU(_Xw>L2oiY*46R2gV8 z#c-(=2R%EkGA;^k4z6;GT4%P(5hZQcZSrIhJ6izj6Ofk%8rZ)Zc!#Y#BjjVsVPpmc;zZs;0$V zO{9JSb9`;vufrs+nt=Mme~p#%tD8>OI_zFk=g{hLeZ?$Adl7$3MAvuVPpz8Q|Kx{h z^BIOetbp#06HvWa(Z2GjqqK(}_dl6?`c(M1|5{lsU;HsLq=&W3g=^;(M}}f*FCX1o zQ5x$y^1*ncCb(Td{zut?EckOO_l6q`dLFo%*;!6zxTX$+Fx25>7 zOo{$8hF`mY!QvZFitnO`+PMgNRChOCzn<7!jclIT*rwem*sP-W9(8tm?ww8!^N%dP?}xu5q%hjvaFf8#gr zvrgU3r)Q=7o+qF7x-juY%nkc$t~os3K3?@(7neK-5k+2S`^j<8xN7wK;l27%0}V)2 z+=36>y&h97HJn3l_cFLN0SZ@1)~8O3f9r^YwbRH>E==!SpC939K9xV&Fl|;7wwkvb zH;&=T_Bt5POpFpd*styxE(HQaHVI$7bmR7X_z6c!vvXVJ?t$+9k^P3Q(s||YDlFer zpV-)4xmhKM(>n(lYZ|EMCh9I+8dy6P^+%rzR#!NigOGOVl?oB3`Q`?RwzVtO?q@w- zCl`n7ni4gOAHg!8`H+jNu(6rLUu|zM;<|HctoL#%k2x=sB>lec98orQ7p514w5H3f z>pnrwiqC6U4n7MT8TLIoo>KNnw?*zp$}LA+?dENKH)-C_IP|R~jjAEkG`A~9DWfWa z#=uR`>m+QJ*M(Wnq{P&KhYx*pHLdWBPS4wl3{Fmndj>5`}m_=kWBx+)$-I=6j>nYi8W9LNf0o8!KX zqR+0ggWxCGH&7#QczMR<+MA6f5$ED+z zcKiG8ZoSy3%(uIbmS2-~cl3^HSCN+CfGt{~q+VSKouF{w*CNvIX@J>E2GvpDAc|&w zT8`RE9y_MMWId^&mb-QF8*6zY4Q|P{`ar+S)9Dazz1uAVxjai~&R$TJZsap#3uS$Y z-aRB58iYs)o(of}?7l$1rGCiYcn0pMTX@gT{h}~Cn_ubxo13%i?zeU{&;ABZ2mfj_b}Ml z;wh3QMG(2qfz$8{c+ak;A zvxNE{@SA||X)@1v{SfSJn&qI@QxzyNM=|&+w*O z$3PC~TuaO}NO%9{{-aLsyXods6SDf4g@u=<+XQN-+}_LYIY-iwnjA}M^SmJtW&UP# zI<6mb>B(SQ9}z}Ed%OC*+p>7yXD7M8dJhk)(W}Re0T@Bzr!1p^!=7w{?dEZF#DV4p z6Nq9m_(=%3F?Z=n-7D7cIiurljXplY^zM=-*^Szbc*`uZU)bzA0 z0++4jK8&VJsXth?oNaXBo%d)GRXYYp=7%JKWnTqDp_e;c?@v8$#P|R5AJ&APuEM^{ zHrzA_r8LXpU~&!49p7xQjuD-5J2zh51zL|e6spfit)hls}cdImH&^uw~UK=>)ysK6afWALP7)ul$4fM5m6fH?(P_RKtM`B zLZ*vd@n)ZZyc_lhQ(4Cg=nw6iI8wym$oILpWK_j^?6 ziwpf0_iiRar=zI9V=k1-Pt;6yQ(v_#x|r%cr~d3=;U0uaEDu(e<$uC;n3B^ zRCNgsA(e@PR=y!(4f=@z!e|v(JqC@rGv6=F(kC^fS66*|XSAD8=!#A%MzzEEHg1f{Q9)EgyKjs}EVPGF%kwG)Ofk4wg-0yi z%RmQt9a<<-$+VrM;$}9VRQBm97b%@dx*OZ}BO=DzynL|l;ZewYLO5NVw=3y;MK4T> z&c`e*Pl{x2xr`_yRW8?$$!d}rFkIPuKdx6--Ve9zg0j0WPD?Ns=8U>I%HXyvhv~m0 z9JfWnUEy}t-&P5|H{Q@$3H6`apZRTTzeiFmJo;(D#p~AhYKe=+!yB_M2Q4BxOrth- zMY>p3+uvx0o_!YlY0Wilf8G4?;u~S~ove;JgV?gOiL=AD-C*xZ-hI1t*Be87x#CCnxw3*5z(@f(eB-Xl!s3anoIo{Xf4_6N+~kTS_!m3ny`@ggx~N^x^SIk3yY=3kP*w7vlJEU; z0@+yuJOQI!tMK^w`XlSI1~?>uU|CWds8(#>_;_QfNo>!Sht}=x%06;j=~B)2cqDTG z1A|T|(AcIKu2gGF=ceJT9$upPyGj0W%l~@!=I5hH^xYR`N+FKLgz z);9xP_t4!8nI?1eNf9z)+`e)GvfpdEjhCm<)Xu}Hccesi=CFw3$w{*>|30&UGb~@T zjtdOALN&w63qcQHQh~MI-^wrT&J76x?O@DT)@Y?$?HYFd%A)`Q>)vF3;ex>rtPa>G zq^az)N6$Z58gbm`c*X=UGuuG2Ts3ttU$2PtF8+Ay@@NkvdgxFdwmsKh?TLb- z%hHq|A=_GI2m3nzfUXnnghR);tV*|SOXJ0wO2X@OwysS-7`RJ78XEQ)KXP{6b3u^U z(Kx^w#M0uLzMrK>ct8faCVygn%4aM$5%39?uvE&E60BC>?ye_2S*BwjuV_|n5;DRh zAoyBp36*TU%wzjI``WgxZC4KE;dP12SU*j2;hL}dplw8)sYX;m1^UyWGw9*O{0{1B z%WHY|na^}S&(eTG=ZzL6BE~|jR2KFHzvj#*a2|i_TA}V2wv8^AB7?TZa$5ZxN73GM z<4!Zh2P>uh{f0a}og<|Sw^Z-CmK(Towr2 z-?T5T=s1cU$^zZaOMVqm=<^_6NLGk*^KuKU&`mv{PAz_)J0KqQv+K+aVd$lb#!HGN zU*HI2F%VuJ-sV{!_IV^ES{$WY*XU{7T$`g$2qfbxwEu>T!G`(witMc6?_`fjbN$m_ z>2U-247ZWP{ke$l6d*At1fAzfemMWp^XzC+H|Qajp=WIqPj=I{JwoD+deC*|g0j~r zd`m2LD(EVC+YLK%hdoe!)Wm8=^n+q#|L$iZha2e|okol51RRVs&ZPULFjJhTbVOKH#;bkH z$J;%HZO>5n9f9vp#ur3XgETw)zJ18;EO!Gn#FiZ4XeT^uafr>!RgLcM>%h%gBejWs zIng?a1YtIqHCC&$Wup=ZmxRBA6VvH=O@IVX3|rNducvD9b&sv4wu{&b8DT4?ek8n_V!pvUaaSZE;oC6Tl5 zewK~XX>|!4XQ}itYF71$_)$QrZT;;#>iwSWyR}Qvf;SN_oF^}Cd&tV>mZug_G<77o zP0+V|C?G(N>T0^@&@=mer7ViPyj zq1Mp_MJ5y7L;Mc^!P6$2L@$~qgEbw?`xZ%$9e_CkuiL# zG}@!(M^2%>!M481k$}skbWTeiY$2_PqWAn6wu}ktcDWO7y4{!4aAsZ*Jgfl?Gms11 zph*}PF=Ct|nE#L;Ra6fY6Ulr{GK`asd_=Br=b65p5jLQ@=+tW??ZQ{;J9=rmsu|3=}8Su{<$Sx_K-r0si$S(i7}; zxplAonq>poqySIh?w%Oh&+|u^nmgN4%KoK5$B}*>CDz->Oj4W^F|Dprp;B!RRetPN zg)5{ib~eATQz0t?Tw!?Q5s_{EquTL}=9V#7`(-+I<;|Z*?#z_kV!chXAKgSY-x;m- zluT2)ZR&y$u_VLN^lUL)N1 z6xeJg_5j=zFvRy}-@;S%a<-Ly8IVoX3Xw-O7Hsm}4IM=5D_VLc-sBS>L4rb`wqpKH zg?$3OzURrZuAE|~^miAsxe-GNb3dWpJJYFfK#1Byr?yiic>@*OcSHbN>+*%dPkt;o>9|RAg2^Z-vhmU2SJ-rxfLOEV0G<{bbDIQ(>X4Qyt>q`Xj&UZ687%yF_YPX~0by?=YTCjZ@POh*B1dLxJ90bY73|@-HGraq0f{^?=V!hsv7{wHDrQ5$#NQJML5w8I{ZT5D2|mJuyV?o7~1CGNU*%S=xH8=A!}ip^o05MtHy0Hz&|8 zJ@LsUf5&t}@}L^6-%Ie z(%PoGcMKQ=*Q>T|TZ(;-OP#FM`j0FR-G279&4Qq$gl?uV$wM7iqebZgl>MPX0+tNvsU5$IVPV+NV-=ZFcH zsiYRIC_iOh+Yvj|v)EfRdC3=Ywu@g-LU#5Ec5CPI8ee0AA_f zruhx3ymfgnk?m?!X-I3)T262=ME-l*)#JvakMyE@_O;X;Tr+Qa^f=x<)|S+JO#E4! zlJpa8E9D6$zEXSvD&%0>MY%TFpR{sge4u<%-b!@u0MT!cIF@M?WLwG9w2_;{O!WLH z78NaLC=Mz9VAHP0Qy0p_v#kAFmrC3n-fWE3J6&M6Y8DH$i_^rXRjVqh>*q zPo=CL(us&1e3KA)oe^985P-Roqr1uZs>5@oXahQLgL`vl<~p2NJrjLy+nfteYbhoYkH`G z=;&gGIDl?;KbU#CD5gjGQq!e?FJqg%R_MlAkDNd$Z`tBV%W-0mXR~N8uh(&v?-3HQ zJKHG6)$U7ZO1GY2*tS}+;_{kT%R6yrBRxE;16G?!6mlbUI+3uTW&t)fAj(3~Suz6n zH~4ZRu{=H>Qtgpivk>8}kj3Dy?_%yZ_3LcaOd>_%C{#_GKeT+(=NgiYwN;UtY~o=b z71eQZ6j&agBo(bSPli_T!kcUPd0ocmt%IF@w--VEXkI>oPY1w;dvT{yU4d@w7iCrL z@}Nlmhp|oDSnOm-ECEkQmPtB4!xh$Ik{i^Z>uWGSw#Dra%<)H(tZPTwXWOD@ia)18 zRA(_JSW0l-{$r~9rjmGmKIfieLaN0fk3b<>V;_{&x`@e3pEwpK25IXni@j+p_WVxu zrn;a8Bo1o|DzihJ*$19C^KZYs-K|;4duhezXbmK!uBCo2mlNfxBf5PfzQV1|WJ&DG z`lRE*rrW)UaOyFxB%$uyU1R|!>9HUQX6U`)dl87Oc>--BV~Dn|W>FckZ9ZL|;S> zBlZ~rdRmCGUN&tNYVksM5EI(es_dJgu`y?A^XD(cJE#-ww~6i~m9o#KM;V-LSSfEV)uE zw?b><+r4lWhLd3Z!2g+U=;KAKcPAIy`~h1ixvCI(y#)&dZXqfGx3KX{ltpKsD`+S{ zvSe&C8es;oV7Bk3iLj!T+-?p%ZbL5&yt*PU!F|rHTjN5H@0{)T-&;!zI_?5!f0+d7 za<{9DN_33t1fn%|GC~yJ4$t3qh_0rJ(K0BG4qz#icA^w?yL{>~s^}j>?MBUHI3e9! z1@3UQS!UzYmlW3IM0TBM=T)u-xjBhx^SWug^VkeF_Ya)E2?};J7H)MY&OlTdW3ptk zXz@HMAK23`Por0MQ(o}1=9XC3bX8+jrWSH4b%_OJgCe6)l%bis7tbU$> zW6iFf8xm9sMOQ_DASU_tJC4flhy{NVF`}-dN@3H95}}?~Zi4(jSA<@(0dRPP+#7!O z*!YM^JFe%Q5llue)K>r%xfuN#3wxBn-PjtBx!WX43lexR30(;~Yoc!%qZ?Vr0qw4r z>m90%1DP=HwNmUnNX@c(!#64n(&Up))z={-KXJ3e3Bty&T2^-IBdCOw4>y5;YMsw}s{|TD;HTA!(!jlX4d;ydu=)nsv+J9F4|0~AxiFgHg z4o1;=)qi7P|Ci_f6XgFj=q_M1(7o%E{{J6@o)Mt$QN_+#z(UNC@2H$ojYFZmxY z;I}LBo8`qw==qG!0eTkVQLV=Bo)F6YKQby_US%ozICQAS?M>#ySf3H_QiiM z!=)3)|I?h~{NGpo&t*u7|ASN#JP|9V$pkK=9h1Yq(Em?ViLhu1_sy2u+7YD0zcoF$ zKVXLJ#G8ppVaYXzlr_KRNe-w!+TUzfhB;a`9s*X@%!kaj{{z`)+GL3@SmcwJdUJFL81F`sCmConFz^UMwBzguc&(bJ+Y* z{TU#%924Rrf9-&;+4qUq=UB+pT@Bpz0l#B>`?|0o6SH6Jf2M=}_roI z)w%i&!{wxU5K}~DVhRkce>ExjYpuQ#r%d{t5@Mb%5~k|jUuV31aEn|_7sQY&As^a) zS>4wMcfU$u08c=xHa{?H{sMljRwv2k4=JvG$DXb^sq#|xxU!`pv4O@S9v(-O&xYHCgVDG{yb z%^f;FZy)f+*inp8h%k0e5POAZI+In&Ix{VTXc)q*Mucd z^Nvj6UuN;kp3)=FolvXQ8wsDywS`2UaQ(#Mw`+0#W^h3MVDq(HyP9g}NJuFXvu zY4JDw@#_OmDsODiBgQ0~cmDad{)hwgI=}U#@&Dy>e>2H{|78F(@G8?94w?TKyyx^c z;J|+Hz^L@WFDLeENdGl_pJ_?p4633NmVVAWh@HPSIa(I)L*0&khd5Hf$UrL);oJ1;P_Ke746G-vVY#_M!H=+q ze;tovvgbI6&u>>R=)A;@BW8Rr7s4XM5k^_52YBLETlZRN+39}A_kMzZiQLnoJ7EM6 z?A67S+rZWQ;iWEdQuS=2$0R(OZ}@| ziaKBA_!2kt_D8iarN17NVtMth(M;H3b6HmZFo)rfk`9)f&m-kuj2r*!@WGyoi|pq` zW-`zGyofMj*!cVT@*Qv@p%h6opZ|Iy0sncOq1z6|qmgpRfN1RpGB9XV_ODR)*Xo04 zq3_9?Uk_3!^^nvCq8aUtpMD*{KD6{8c1_@S{GnJ*tHK~NgEK*E5?J=+cj6U3PZ^(h zjry-I4Lz5} zaB^=aQm@bUW8w_@E-^Uvjiz2O{u((v>TBrjv*fy?H<4-I#vY%V-h1`@$y3>gln1`1 zvZi}23EQ`26>d2S>*vhh|NP%iyPtA&AFi8r#Bh2Ck8=~3$gFj? zTj$R&^GUeTOY@V2M0u##P`!H~_}#Ao{xy^8yD3NXaND#a#!m;AaJgqoO?&?Q&nNj8 zrbU}UBk+RSeC{Dkk>h_(06T)a$V*O$=le3AS108*`(16I=Pz^(@tF|*dPFzgVz`|! zm_k8W*?rUHS1tds27gS#=kdRuaV-4lG5V`W7cO@--d~@}uc#P$_wHT9Z{|*fdv(z( zcr9LjhU9K?eLHL2o^TU)3D;EFxa(Kc9*i=f%^x05Nk|0E@r{+l!}m-p$h zNui3C*7%xO5E^Zfmz(Rr`{m0QmDe9njYmI1;-Q=xMZV}mpieMD%A%$R8MyiR_}I3@BV#K+zJve<(V69*`(arlfGt@Ah4r$$ zEfq~`W4>`avC%;qVfQjD;o9bq!>|QIv#@qcvIc|BLm6P%(;M*O-I;_+U~FTckDiHXf_;GS z*0+0I*Mp!a5_zW19r4L4cn{L)DeQ8kH2h1|Jd)mnH=0rtXw2Em^HIV)*kNP?h&Gwx^GD1mNbo_ z-;JinvXZ?h%1i_o5*#U<;7(njMwE`tF0%N39cDg*khf?478 zkAr$cj0*M!a5fNUC${k%h5@^aefP?o*6M2k^h{>qg}$#?z}7|~Pe!%4F@Hqlp^Y$1wU|;c+Um|6vdHy zpl^5b_H12l$GA~)6X-}ecwKPi5%QgE8@9h>Y~!fiLC(^l2Dge~{o7aynk^uI=Qtq1 zeyFiSUA~kV57$#^D*~FLaJE;D<^=tKr`|DeLFJ2SPZBJu#7ixmX7s1j2rwJ%l$>+N zKun+lXr|m>&CTmjn@oF+YGGjZirsLqh`r4kHy_g{@I5*h*5vU&yh&Ez?nW6F7R**! zfUMj%NZ~$M@~TaO!Kk00738$di!6b2C@nzPcG1q`3TxnXT9nrWJ-<<(eY9(w3o7F@ z?6H)0n^B;DV2cMJIny{EKyCz8HmIW6#jQPiJ=b&VJ0q+{A*{c$`KHuxoF&X0h_Lvx zf=Z)}7sJ^xX1Q<=NHN4|^nYQ(?#xl`eqN-Ec6ctOTv$&9fEL4Ny#ZU)3N&HuwgK=r zdsYV}zQim3?0Dz6W#>Vo&^4RvBQ_DmLER24lQ?rVmr~MShJsOk!2t-!OO?l!|5Uqt zjpFBfaTJ|w!s9;3wLFZ;YaIYjNKuZ_=tBXMXyA#k@}ThE2K~|=L9BDw9v|&4n_@x9 z)a12} ztZbkR@)J9TKxB77acvs|EU2dEBf0|E_83rvm?0|#dgEQ?@2{Obc!sh$1(;KtSRh#4 zr;Tt=XOr8le$70(JEX}z9N?Y0EXn#!6tj5G#hL{x9c?%D1UVyf^qv&D1~~ybMtvs- zQ}MR-_Q;Y8^iWeXt8bti?ROOo^Dm3JGWEy`64{(={3)d`)yX&rYUso%;E=~!8E3)U zC7)JmEae{QYbUd=CaEH;--}H~#pt-_)J#Ao;+gQC@i-TV7y|BmQ9bl1uHI!A^gB9tfUoQ+LimksKZic4}BPF!4JnjHhxo9Fv6!SQf;<%4t; zYgH|K|2lAPHJBJ@eO2@cvKeOHz$5cV#ArDL;~xwNFsmz6reFmmEOE$_3LPJM_^VGl z&cv91FMgoaSFKbv74j@Vx*fPyE6`f^jiz!#IqZu6NX|RB?*N39RUig_}Ec$kDv0yey2IE-bjR8m$l6)I#qclKL(8xvtky#d{wBXOSXHK63WNmv^ zWsNsug*$+|_ z?-r&-yW(x#iWa{p;~#;Sl{U$$A&10VWlYeMpoV;!xGXNJw?T?GGl+SlwZq>F2a%1J zpWF1@$GYn-J-O>CM)Z+Az7^YGGmHV0_G{ZnjP$Z+_LwrxnuS~b6d#pYcRXKDx5$wd znX^T;Qu69X!0t)v+`v2aR3icL1ydk;Y`K9-Ng4IXEKPKZ>040Q?ca|$BjByDIS@GA zY@^X8>-TcQajS31(vU%O-MmAnKAA3X8N!&9A2KEl6GPbIe8Fff2`VL{oZh#b**zP3 zITZJLn?#Q(gCH<64rVy$xP!2-kIJ;oSz0E84tdYA)vR#tU-ZYa7;XeW-mx_vQIQ?Y z*U+pa6AIZVX0l-bLe+CVhNdhbY}f_aOsEfM5caK{IQ+Kez4$6m(sKy40 zE8`7G-b&BIE| z)DN~XS~fvwJkE)Kh9iP*S7d#6YwN!jO_)r7mnUjvu*=q;)PkOrbrcb; zpiq?gm0tb(K$q<@aMsT3B>B1ipEPI>?MVlPb1uk`psG~uG7h~GloHtwb8$4w5`gTL z&Bqn)`FIV?!*d_AR?4i)F)Zxtgeulc#SdP}E1b)TU&G`4=%3keU9ndF!Mr9hdQ?6e zl_i&aOA1*|u%n-;@j7Q02ERFU*uTm2Q}NN_MJ^oT+;m?e^$tc2ETdWPqi>{DjNMdL zQOA|Ik>PL=8uxFq$r98xw&!y;Ng=3gvwE0d`5h%@*|Xcq`zcuHs1`F{q8PM$pk2P9 z=3PDIfTRXJ)hGDXSEmXu^W-N2V{;>6g#N7-^w~?8KHu)*#swqsN#WeW?!l=BKkbg! zqHNQkC4&V(659JsF^)JNb3?b?8Mk2{ryvp4x6)adQ3O9xbMa+jP&zX{i*fay>j181hzi z>RR&XpvimxR8*zcRIL>g@kXO$S>i4{X~!=HG_>z6f4p~OvjkbxY|gubc`vy)c@0sx z2TQ3=y@$#MeHRgco7X&56^jsM?W$G1c z;LgfIIJi;yE5oPrhF*k)`7NouJ5Jwdx}t$cbnDLZ5S!4v#Q?;G5+!=#%EgL-Tm{-Z zc0684lMY|HveT>%-vU%2PZb!&7Km9dJA2i8583BZ%CD#ENfulmjLOweUWH=3`+J@N6AVEX6^ir-sMy{>nLa+2 zBY7i^G+*=ZKNdMgy42QtoGge>;s~^z3h}v&6%^OPW0W7~^yIe#mo}q;L6^f~5g=A@ z?&N4&EMV5qIV*UrfNaHV`m(duT_5y^vt!oP1$;6I?<*$C$?PZdBUYV~%HLP4Rwi3f zLv1l<0W5$h_?=W=<2~Nudfd^?>GbwqTIS)W-Xp@Bv+11sUdQ{QfXatj^6iSzfwt|9 zTK5s!%+tG@RPF1A*UVv9OUtM^4GNnHLr5!cbD54H}Z%$sZ4TT!ym4TiLz$J zTQlML43tO_LayDuR@{hsh@N*SNjrC!ifM7)-Bq}2Uq5pQ=^1uqF+}4vZ>1S!y`ZAM zzrd-Tf+aVO50i!VPuqjQn7(CcRO6V!(C2 z4D0>oVaf7Jx5S>sf;tDLn~wCElo03Kj^hpp$Tw&e}Mre~FfZv|uOr{3yh6J0H7Gol*q}vZ_1<3A7;c2ebQAF z9>ub(>?9+i6YvvtPlCi}mGf{cX-Q7PIPHJ1+px?<-`Owh_vRAbsq0gjo0Y3Ci+fJC zhGy97^*f~kQ$2fg+QFcqHWWq+_%Wh|s$Wz=zr}*A;4Amx8Oon?en2;EoB?mS3W-^0 ziCNY#yF&bo-6^3Yp3Bt6jEVi9xh5Z19Ma{{k;ZVVp#ajjkpU||>zt@$c8k85i&CO8 zJtCfc%YZR81o6(L>=I%5;iOwuZ*w;$YOwr4?~a+-Vocb?4svHh%}&H~yJK0sA(lNV zD(cj@G>L>Fw>%A<1G_W$B@a9;Yl&&vZm?&+Cl-FVyPJ>?k*!irY`q>z>fNN%uSkt} z)_X_T{kbZ{Ct0K4B2-e?33dx+Jgde#W>tFm$a3itPKHG}S^a6_*v`}H`Mjg?#qvB$ zc~n|%h)kFtIz&U&oVr{o%vo@`CVsFSMbY+iIfX+)Gd0~lM~IU;J?9qo&*Z|tt`>rk)0CUj(Uv*a#aMNn`iEnXy56W;7jkw@ z95p1(PmK8TPn>tPG&Cr@mUT?i)r;Xbl!J2CtOjumANTHPqjPc8a`j%4nCHY-^bKqV zis`c!e(c_4e*8+gQQO^uK%uUuwfH{1yy9SL@e&22T!A-KYS52RWcU`RL}g+0eUN2< zjeTlmBygFHvpPa2@oL>tVeSnD*3tH85L}Ese{Aq_LJOt_1Er?l2=uS|Bk?Qms409ch&`Z5&2T1_4&87ye>Q z2(8Ry{|$G%+1DgE_Gr4wf{+VNG~|H@MEb6om5$TMkmuw8B!T6#DzuSR0D4klPaS{= z(Ol+o1VWQ=rXzb3=!u(+##Db#eTn9p`y(~GVwkp-Go9x#W{l=$c+HA!8@0hn^eSbi z-0rh8?b{yP!Y~gN)^zzY#|2p%O^SX_id&hZvQnA*a&N!5-)F|tB$3P|P%l1pL1%nd z9N&5B7wy>7=dD6M+*2TEEUE3X=XH8~h`+LBY3tIKZH{NF?3F0e1f%efm_f8b&gaH| z6a7cD{pN`vF^FSIiE}r<#wnWgB(0LFK5zuRyQ*w%Mivy3+SGqBEIG87!)kYe^o`yt zR@W`n%}bCVKTmouY{fL@ zV^TDIpZyjEHU&Y>&R6@=(o+XS&39JP6tiR-mKX88WddRd*&}6xXmsj}k<)kcb8~YW z(qg?x8+~eE9ZJa!LwHRTR4-bhzWpeB@Q^viA_jcjFJ_q4D>-S1JkaFEr@fJ2+t8Mna z=UdN%dp`!w?uPeXFhnpFK>p$;h+oiR}g2RDlk_BY_ z4=pw`IVP33c_$PksEsZBzo-PF@#Oh>0r7YOhXi&~gWAlS<7En0Osd?r}53Yl_l_%^lt4N{1n};oweMS0EZr{fx(}wWX?A0Yf`jIMs%#9$p===1=?X-2C z@-*9(BU-A8Hie#{VN+gMlO12@(kb5$nh@AY=BY=T@sB0RAzQ_sE^U!o8xQvrue3b% zF#c+z!aKYM2yH(Vcp|ZA+G9-7$?CePwgpKr&DYQ~FW-V&=7Z#rgGB_yu+XMxA}+X? z33=VblB(VC=9s{2Ei&%{zK1==vgh85P&;KfNN$W*Sz6#^ zr@hI~QPGcuS5ud5iF#RmH>y#vR3gDqukRSs5+a3AyMYd|^frxqPP=u-TOF&QtN|%J z5fyPdp7MjU9USjCB}3n!YZ6cP5z97IKrKH>Ch5#+|AzJ&(9GlV53WBT&2gt$o8eWK zdI}* ziXx`yXa_&_gw$eMsAd0#$zZO^biq49z8wM8aB-BY4yZonrZEF@!DrAepbx=A^Cl-= zO?1!r5!dE3{a6ipG`tQ7us4>DFM|;HQPm!gbJn^IZ(^|#r4>uqRS)8} zH?P@=hE$ONN1bh~%{@>M-@k*H?%4&xdP92th4mPwC6RJHSFL9>L43ncK3;_$?Ol5> z$C}bKdE3iAjB;_A)MjpRp|W-RwG`_dZ@g9KB#6cF0(jW+-uccF%RX*^7#Ey70D=K7 zthyvlya0y~XGAfkgtWH2MYFOb@ME~4FU^tiv{rm!Vfn~PE3n?Ul%TP|lalM|mG?_d z#LIr|tra<|2XjsZnN08iC)TAni}kr98uTfH+q$3J%0MZ*g`{XO5S2!z@^p_Bw;IohNA0X4Mq<?&ax|6sjdyiH zmp&n6SDFTWj(-ox@Yxay?|ev3EbXz#ne&r|5a*;7b5T+ncGE~_+IXD|c@2?&!dB99 z#FM*cLE|rj^cP`x&K?KTfmaG057mZrJgobRvmYD?fSRFSd=*~C;A(z5Nb#<|SJe-L z8xNM}pW6XC!wyIo!UoldI7=UD7SW^PG6{``cMB|kh9`8DJ+P7|U!uW`kAor$o6@x_ zb09Zgjn&8AbxvPaK>6lkb#d7;201`Skhaz^$^{MOAuPH3=ZoBQ`a7ZwJDP-itM4)-&GPI80GDyb^#=e67NO z&jqThD-~OyX3IOQAtizjoEU5gqu>VkFj_IK=m57au_0Rc&%;2?ZgSv_1($5R56tz z!|Riit2S%3hkLtxlUHiilFYhtICqf&tUWxaD{&!^)tjNou?O4XzWVnJ|{ax{@PGq z1{x=(j4)ChgZky+#oEjX8vEuOPt_Ru%@j5=Iiw;<_;VnI7lwd{Z7Zt-w(JTD>#mUH$|@1-zn$-?~T` zS3b@AZ!EDst?YX(;dAIo(`9vSY3Al}i@btyhCvIJaJ$?54jCwr?w@|uunf}k6DRUd zCXcKkFyF+=4RNVLF-U)2$&iL+tj7B+&07zc002JKIG5;0WrZ}4?hy!dt#U0#K^5s? zcACcg5slX+;<3JvIm~W(;mwAL=(EZQpink{ns^04e6>imCi^FU6ZEMIQC%`Z+kH^HEZaiPItT9o*S|Lv=EI8-xUd^WH zC<7%6#aTacb!-#Ds?@~FcEZ6#Rgj-AIV$Pc{z;rw^`ye}#XkeV(U5bj+>R^Kq-11Q zc$ipaCcb)^|K(Ewo`KF<`)dH}H(hGhIk4aKqA?QNvS@ zB5~s5z6B&IH@iIAmZu!lagc(oivr(mX84dYW6~9g8}(`DrCfCc3L#V$w3!Qbebvg9 zjX7{7>oCKs>!Kset=(a|A_7U77tCmC%~|Bjt+r^xbR~K$BC=K9U`4dika!9mlW!kr zpLU4L*fOQy7sjty4k|ni5+TDA9Gu}Egw%g$h_70EmF`n|;VNXd7NmpVD1a$JjzJZZ znOSQ=zuBrMU6lhnl)aENHV;*GV-miHBS0(S&`9-cV_)oG{ENUHYxx2T65bA4ja%F~ zNBw4eRns;$>B|TEr;eZU>%C%)dazOGBmTjIyjV}tG6jf4ZZRhbU5;`YOa|w(#lw6y ze`Y2oUudB2B7A#69Qi7f-YQ}x!7N{H@S!_&X&eaC@W7OlCtVH@Wat;;E`0euEd>vgH-`zH3+j>k6kjF0nC9^)Az? zf|!`8{D?V~xB}0_wlB-*D2oZch(VTTpDiMitJl`j+jg76MaR5FPZ}9yl5QprPMu7V zpf$do9W9Y{g?HunbI-3(?+bk-k9U;ygOmtQr_0qBNt~YK3uBt`U-1c_RNiLm9O(49 z7=0~z38AG;*1Yb0kmFy|Gk5{f_1U}-v0JZM;f)a2COJ#BUDrEtnJmO#^}8BO=8*Tw zbMN7Aq8JYnd52rqd7vMEW*s_p@wN0#3DF0$Ng-|q*-(fl^K`icqfO=yM2Q*YO6#E% zr@10+dDKdBe6T^BX>8%N>u{pZgVFjGY5oV-)M$JPO~nOCR+*2emxSz-cXUl>hDfiY zClnW66ujeh)>HKHL4V3?`|$}`>$YvR_47MSP+B_>6!onwNEG?L1;cirakNJad%y8N;}33Ld!r0UD@{S^avDya3`DA>B>2z&M+ zo{%5}{SyQbr7$ghdeOm-;=ns<;Gse^ioU;95?wH9hrrhV>F+1yGx%yO!7`;cD9?pJ2!mc zt7=k%L`7KvWy7*=gm=Ki(GVkj&ff}1&K5y?}}oj=Z@PuY&PmexEH!sD<(QCMhyDGpN>B}_WQ?a$9hvfy3DAj1$B0`>}9Em zYWrSEAl3Bhd+X7Ciur4@UsW0J)LZ7XPs%;-TJJd}8x3h?%8kDPwRPe@s!^*KS@KJd zACo~%U8hNYR$|1r(bQ44kBqG<_YZEpC}S$!Yp3n7tJVmS-qQfRovGAZxOhj7jrs>1 zWEId583;`-jNr9}aFAe%7SHfw;s#f)7l8yW+PzJkHvs3Wa@2cBgS~W7Y zHB7mWzhh@RoAfifU5%Ahd0_Ol{+D$F716J~mXq#^qR;v)&vceED19x|qbh2n(R<`Q z2=B$Qwil|pocQkMrHSat}kqb zCa~b=JW#FcJ9XcA$|iqoEIHFg)}L7uXaYlYql4FBI(x&!Jy*sy*^}JYMA;5J?)Bk0 zAgL=}UiuDUT|~&e^_3IZ{-`PKVNywo5w@l{c)hBw2VR`MV1nZkZjqv zQ1*=_0_cE(nS%AVaABKz2P!*I{^{OEbgbDw)(_pf_j_rFmy z-|sn}bI#|S^ZC5r@6S_{B`k$R&X|$%!IYy$UT&h&i& z`47>Ju}a3G*^?-ZZNC-s!uFTqxPs0;-&7PXFsk>%IhK|DJf7lS`;BP&SvPEt4SFh{ z$@QM7+301Jq`fY@ssOr)jnyDO;=jUr#j@#a7egAFUaKg9V^W^4_qgSTh)oFFWguoR zT0E~K;5I^79dHCU-y_XU;{|Ab1_7H_20F0tbj|@H@~7amA1M9u@QO1iMRgoa-eje~ zoB3{DUQ19V{|eWu1m(Jn2qc}T_6Vo|<=z~%^`rE)R*)K#)uRLTnwSnRo5qRiX3ct< zg{2N1dK(rS+TQx?a(l@weqJk6WDeV7#|nRYai8Sxv_4My@YoJ26jhzUo9jo4ea=rE zdlLubse$z2w^3I8Vu12-cPJmB;$nQGd;6aEOdvYzF(G?D!3)P)rRq9?zMWLB&tzfK z|E?zcvX+;NahY`3xW$Ea1&oBr01O4+HlJRnHFh>ldZ0x^`kZdnX}jFrZP3PLS&~%- zO4yuBRETr)e>dx;jhoJG(=MENguAYDZ&`dR?Es9n$fjgc_IsQ-nq9wO#v)XF{Xxb< zzavu8dz>s{05aNJH2EwA$+WT*s7SJgro|$Mg8O zL93eLhG+QkPC;=F&I~(nu|A1)ep7Pl%^;o))BWhFy7P3$;~q6hwcIm&7VP@EZYlu@ zzF7R1h%WwuVNKeCNe*2%h*zoG^w25El_w`=qToB};a`AX|CNhU2|bL>9CZOwqEza4 zQ+Vh3-QHC-fNE=BgD1K}$Y03bl$GLg0+op5VdHn4KDM&04^D`T)pzqqEpE5JH4h- zl|=a)I}Pdmruja6TkPIL6Zaf)(K#^I6W48&=Wd3lb{ow4*~PQ!UrEo5zL0RNsKJGd zb*t5j18%5SEpkWu>sLiCx&*-jRtC(%0bG%oy$wU7j&84LD~g-+SMPQ{U8DwaM;YO^ zv+N|E9PdHxI1DMRVOVNGYHtsCRo7`5d+YX=U|lW%E6UK8SPmhQ=R$M<@><-u_$K}%Iq zn!Wn{w`T71PXlXGGhp@4Gul5}62=D_A?(bNcI3~z@-Bkxr?b6IdW1jue3`<|`iH0| z$PPu)a$o*%U@YvHHFi4VFaH5+d%+NM5U*rpQc3ak(r{7`1$*z ztVUoKdJuX^I9vXD-g2BMp6Rk7Jw`vgKO|;hWJ~QZA~%kjK$}p z{tLMp2JaKjvro#f?(}u%f2H;j=^g|stWaVNr&Pk3Y)S83bKRzC$tGLA$j$A9O>Ky& zWd0YXh0Z~j`0Lol$%?0yk6&`Sa^`!1j`yA!^Vse90BIF0W!z*st7JzDz_o5c17v&| zCH5NZqx-S!t%R53;?#G0jlc#u-PpsjV~a2-$rkM+^QSY=rJ!oPywOojLfWyN|NJ%j zEuBOX43U8{G#i`m%FsP<9vD^bRGG}~T+sNieJg8>J65!J^BE*YG&IYSY~l9zgq=q3 z8t4O-wv})K%YJk~$e&FDJY{XFXO{~DhGy%c&OH1rEe(xcR}B*mpc+`)pA)9kcYnY0 zLP{&!{e;g3t8y;z>#H|6#ZPur4=7mUL~Gqnb+548z0>N{PE$tKnwCX>hyq}d0KPs zJN#bS*vprmZI$?G?- zx3)Ot2=|QlT|0aJkHN~P_ph0z$UvcT($j4+ZuBkczE7wssG8C>vN3<^%~&CafOq=i zD*jv-z!3~eHZ3cZlBm93Shz}SqTMwfN{FtxMy2&sa?|!?XrOk{^`^5VdwCu@8g+g> z&GwI^Z$5pJ>nw3Dc}-0#bmXdaV$VghnC1&DcFs<2Pj+lKoOPccLtfcnbao|X}SH_H($$Q$v2MaYRB%P->cy46`NKd=YPJ8G!%w-Q7>q#tCTzrt=EB*5Cb~-z>Gx0bummmLXod7XIxE z|NB#YAJ{KQ@4ghizSIBRUi^Lg*?(|B>%J79ar4>bJwj1CW+5)WgnfXIM)v)J^OL<_ zxa)?7Gl1bM;)Eyn@b-W2mDGL^RzLOO<|yUPvif6{|MmXQI>7(g=Lel_GEh8vk!K?A(naDPf!Sxj+s=?Va}e!q@T7s`+xcx zX7+;%T-6*XWG*>_Nb%!+Lfm;FDHZ~_2z~bQc-}6Xx@aw&tFuxkiArG9L%%bQE z8(pYIl!9Hodz3(3ra@QtN^{76^!W*7*x*sYjLGe7Hv-a_=dgRCsg_$gUTEwRbj5X zO)LK7>V|=b#!VE#H`rk%sxkN3+>(nC-y;hV#mj$Q=YN|3IYgf0KTcYIpEtjz9Ww(M zN0(Q+hyG9P3%-B*1mM90#>W1~Fy5CE0dsufV$L30;Xfxp&3UlNlZlGT{c6mfpZs0G z9B;dI`^)kFG-_rch99A_wz3eJo1hOIFt~o`rRh~(YC*SDRmhbJLi2K^uKI# zLjl-^fi>bkPrkYb6>s<0nu)^fVR_P=F*HRg9iz=Ljae=(~fx?OT$L^hiNK=ncmGXWk_q^CuAzZiMxgQN6a<5wMEvUsg`X3&%#ezgQ z+_W*)-mPcgC_gTCec^gok1H>qc#$-6+R(It74z8-1H8<9|H&NDc7%|DSoFB3;ZEvA@ltDiUy#I#xL%@TV2Rx=0vATt4IQ zw2C7~`N>9WwLXjSK z3t+c~lU+M15L0&fw%%{vW4Q4C17&fGrs$Zp1C?LVZqs(LqOJ|oicrzNndyHkKH=zo z!nM+Cts>CS5QcjDc?XNdQJ(kUQgFfV11#EOfOY=B4EaoJ4I&czh(yl5n8z(}*GYvP zCV^b-pmXkBiqbaJyraa`^Y98EMH9W&_1~A6f5X8g76n+BYG+0>C2gEXOUk50iCV@=@wqpV z3ha6aaOYy0C@3fpWNH1R*p-Dp5XWs{CVp<}&z-o}P-<1y#rXQFf=hgWdS0WN^fv}7 zD>ie}-4iw-b#H|(>gA2;*X3TVYN!$+oaNdX*j*nOaFE(&5?ma@)7CcF66HL4e!t@T z8`7kurMX~{Esb_rapxv~*>L9>_N)mTER$~9t4bH=0VJdrPM)@Z=RK4e_x;S%>iRs% zs9@$!p@>^-`5&Ym%QiGtx8B=voo?+kML1 zW`4H?bj*viR@p`7?6FMOg`22Tq-={*L3_~V{3~_ zOFMn^&%W>T5BvAn!A78I(kTX!8Y3Ex-NyWyn!~TaFRhl8v66{3?7+3gi+lgaNok)y z+riTpkvob5{M2g`sqs^QXY9ha^oMc)^JCIAS5xx%v+0j^Y`ObxVK8qlGK>##L^#Xr z9!eIV9}Y(iQu6s+_`jwd0DOVJ1(O0m(=bexy0W3qcd{Ew59(-wW3 zOLxfsFyoObMSHehD(9+zqYk)Mb?8T+`8{*y^1uU*_Z&Mj_IDde?T6$y#;%YAsEmb; zHOI#{-CmBk#Lexn+>+E!Zsq}n!R)oPv>cj&OhF?gKtI>PfHD0~g#s@3`1y*eiq>Fi z5Ge5U^mK75Uxlo@G@w^R0XsCo+Amc#c9M(9_hdiZ07jplJ&AowP>JUMveIM86m!-5 z84Q4-3IU5-j)FB~wl0LhmKbKXuUH>-12sHjn&bq7NE8F3DlUH3lsd!JdLIh=1`;-{ z`25c2&*uM(8T`+I{_hy7Zyj|`_7T5#Jq+!Vvki0y8jX13wLS0?sndY%b0)Uq#u%@@dd0g(;B;UYhmE7aYxUObE zycX~JG!E!0-%MVBG7CbzgmIwKKV2mmTwdI!Dk^Yfv}(gK!`2K?7xS6mo2~L8XRQIb z-Rq&5qTnDXE0j+Z7tjL$GUJTH(OpK~C`?!Jy&qE<*%i63we>IQ6N~|>$%VhrR9a3GLq{qG%s_rs>b_$3F{OGx->ri$9DJ zbQ>D}xfGW=NTAsTA&jkA)&j!whm>nxb{(h$uQ6a)qy*W-C${4AyP!1~3uK)sD?^I77EG-`} zjxiQ7_MHg=JV`$xHpssd z_~jmc=QJOA*{}v^>MCYbHVrkJ0PkgVv(&bC6pj4!9mpWr0dG)5qhr|WvI)?o(JE0_ zKlMxlF$`&;w;~jDe78*B)_yVcUGA_4g_?L{-_$;f@CB71`rHIolB$qE)}uPM{?jRg z%OmMr!V%IB50#kR`T)c~uXnK`Co$iWPCW&p5dmc9U1wax4jejm!#fYc)1l}ICXO3! zJX%9NI`hs(?~TGxWdu@SxI4!Tn=V2g=+1#!_VfZ!-s$8}iQJJ0gC^3ay>BWV*p(v_ ziX6_?D{+_9ly?>^!$Fzt8P|09lK7xfZs)5CbA-&-yVsBtXvES~Y1DB?g~U_}X^=hx z;R0}3;Naw(#yQp+m}?cF_&wBz6u9j-9=@oZto{3O`mb4M1@mW*hjVp*{7i5oH(E@Z z12>JurMGb4R0=_6RX3j=R3Z1M;x3z$p$)MXE5Cco!xK!j5=wqCm}$bZ2%Vr}0P#4S zM-z!8yG0Y+2&U&)l*|QV{*XlUA>XA|h3WY1&D7vCe6`I^&6C=zpbWgrV)CGTIVrU4 z){|=veb-9COrSr`FJNq5zTC-2YI2LVa$UL4p<>;_y+KY+j#%Wh8+f!yAFo^(EmWt=@B#aww%Vr?8MihICm|FzxN9yb`%tp|7PE+_-l zEanAPRyLTrM2kDL^?I6wr#%d8zO+mBrx%ep{^1^gqgVi;e?AJe!KH~d`wq^*Q} z*6wx*p!7)rgSrPWIp`0Xf@+{zlosiwY6Y1ZH&Q!8ArwpMEs4xrdcvUC5ll&WU}%ac zZ9>${Rvp+)S zI;q)#!{m1Fi&wH+pd+7*j_|%lT zBuZ^LSS`nW4cp$kVbd0c-{bE_zW(A??Y{)dg+=AF^^^&|oJn}iUbr%L^GDOy*26KJ zneQKz3cBjrW|zB#D~vUprb|KFl1(kb=~XT#xaNaijEjI;&G}g1VnE82Q+w{F^gjO; z7IhfROlaU(K%4e-UyJ;r`l*)QajH1eG+GpLE*W9k1>4XKbt| z52y$COaODH7+3naft^OMn0!;>o2NEy2_&5$i^cVW>Zn<^Fwzzd1xC+1w=?rpC+xU& zetH!b?OwoGB-E>m&3f$e8tJC8OHQ_%+nC8o=j)FnIDkx;OF6F?d}CT@0|<1$y2|_* zT!S%*vWK?|KROF$1s|ffPd<9jH%=cn$6dapQDj}vpuD1Y)4ft;06hcXku4IW{6?cs z0jCc}h0Um3Z%VrNwpu~dwKuzP{t!<0*3gDYpfsTxa4z-}CecwHr(=L58W;Ya>ce2Z z{eNC)rI^`4=E!*=|6SHAeLZ|q4o3_JUEM913Vs2I5Mi8~#mDqLLzpD{)l_jM-m5dg z+&cRDy(WHUrkW1}p~~v3MG)SO*lD1-bBm{?Y_Ulct51Z1FHf|we>&ArpuEh-cdret z!T`TUy|SS=w$`!0r*%fm6_uHp^Y^@PcwNI~Pm=85r&sP1{evVYfdzPCu%@P!9T<^)-4x_e(n}$KJx$sQ9&| zOAHp>5qt(QL3db-RVPI5qh>K_I{Ip-03wN5JXQ+&qx{iNp)woKGEgy07u+&Ew7+(7 zl?!S(M4nNVO0ODMk+zCZ4a4NB?AyM`*A`ht>aclLhh+D8M@!Z(EJ{Iq2^D*_R$%p%F2i_=S zdxuOa!Aba4>8&>(D=tn!zjTxFDxJcpLK7H1ap;ww_XPJqdqorw7z<=c$ik=zFH*(4Ij!bxa{esVJ{3ha z_FY)xpP>|gH`XhJqNtT|==PVdyw)GGK=Bh{PK*sMSa}OjjMq19t!GyvN}23kdR}-= zwv0>j@=;!d7|oyOIj9X-eF<@Nj#seYoS%I#$jGHtt`Gp4&{s4~5zqA7N4##CAc$fT z-S2LOPXUyZ^DHPl{6W4>uU+PUv^6H*rCVRc(*)9=>C51`6`hxL+rO>`oK(HLtR>~w zrXp0*Wk0W`loecKODB&>%^r2_Ho3!FO6p)`>D?&FGebNs!Ra~XHUFUdRuLZ zz6O$bXiGv{ju~Ne48$mb^LND-k&=ykCo5Q6lskvu0fjBY;ZY01(4toWzx*5!H?COO z)AeWScW{(x>6rx!JkTrgxePEi z)75X&T%o)ABh!{!ygmYEVQunzH$U6%cRd*stiXHDhP#&gS z+dTnrl$>BqLM2$Xx1mVh|(RVb`E-Dasag^!{tZ3$-1l92$|JB&r#wj&!<_{Et-fh zq@U1M#ct!JxO+q(^a1|T6rgRX)iwgz!HdlP1~bZ=CPRbl*b~NXgVu!$ zik#FftAowJr^1C{%%%ykej(+6^)%MxF%mfc z9*~=QYo4B*e#V(~cfV3@nB4x*Dl(=lJD~DbSOwvXthMz4ye=$DSrCx3Xo~a6ov|?W z@X9JD5_unHS!fl3s0{lwhs99DO0B{B@8ASIRy9hDfGjREI(J6~DhyxUg zHbCEpo$HmQ5FeHqw<6#elq34{hkI|NLLAGNBPLpPJ!eYB?B>3I4CeYay7-8KiE-0J z8iigy4cIiXe)tLN4%V|a=`@Ibg@;PzKw#75>$J5vAQ?F>N7m0~@4zHwlFV zJ|FAi4dbDrjd#u^SjWnVpQ0{2rm%;UA$5cl%s3#>wL#c7|I??S+J_umTq=59V0Lr@ zYT_{*kavA(4%Fs&yusnfZoDP&V;Q)AHv~q0L0gK0JIplL#3>?|cVak5eG-q5ghXOUX= zXy!|3z0GDl5}HcXvojQWS?;}Bh>!^fq2RXeXc71%N3MX(m$x#C8d1u9FYVZH;ryM* zCY4Y|aXVy6gw29sI!TGwWxYDnhg_#3f%NCZ;sA(xx1``?<9uUV1`D_bK|;(8*q8L> zkvSEi_lr$w17A0S-6h^v#u;`k;|J8PeJjK|wY^JCeOI%MjCK(``TM^6S(kYr@8NR_ zvE18z`xX%vAF zcgd~#W8UQ)X&oClJd5ls93JlRtU?I=78VfxCVT1QttUtYw4DDKu{`I@(6m#Da$hfr z%nyGAADdM%H?YN^$byK_LFzje`qaAV26{kEMyRXx!_^MA0;6098z@&c2+bVjqd$3@ zk!0Gyq;pDGIugl3sEf44vh>^vEuT7TU{-*~=CT`vSpViee*U>mRKXk=3+*MQdF(k7 z?rKnw#vo->0wcZFB4X_IrQJZrz>`J2Juyo*%u4M9kZrMMl67@;?XAyc-s_yTMZAX2 zmL!jY$4l9fNOV6(kxZQ?1`@u1sGM@}x~`NRCq9L`N1?#JxaU3gKIx`z6shuu#NEWx zUhO76JqrTDvO3MjmUrQVzi%|eoDM|SWY|xXcCOWV+!R1?9|?H;GgrEs2>JCIVOC%i zXzIhSd;H(8ZxNfI3!)>LANST%{##={UjpEVw;~Lh{|Pmz0*5)%<3|74ch#f7%P5UHhc6 z{u1#9dC`j{+I2E%6#f?H-kOJ;*y)dzOhR#SH_+>d(S6pl$L~%5T?Bl*E0C4-llCn= z8gE$IuUGDV3RC(k6`9}kd^PhoEb?2*i#(hKfS=-&#H~rP)0kg&A|_LLT76F{mVp7C z$LG24oN<}8_Y=_Sp+iIGyL`Wvt&`pUmprDN)auQ4Z01bNn)k&LjUt(h!~PbDvG+oD zvE^$p=02@wM(&q+3IKL|a4%F;80Ak1`KSG|`%|aE4G`1+R(vs1x02+>?6mhK zi{oB}%shw4|D17s#I${@@5?#glU74OD&{$e(eF<59J29HCz|a#&s(=%xT06+b4`~298`_iRC2bidVmg{SkC=u*vV=7tiwWgyReDYplxq+wN z#B8NTJJTxZ{pvqKg;J#_ZuD)v@9m`nitUBZYQ#f{{3*_ckdYMah~lw9gFARjhT@q$ z1o4GgD7StX!t*)5YhL2vao%s?8obq=ee*fXt=DVSm4frSR<26J4d>KP>&Ruh za&I9&)1EV4sJ;IzwaO_P*SjCh@7;UhI%DVnur0?QCsXV%&NhUKmM zaX@AJtz+E{-#;FcaD~*RQkHcyvujYnqrxB(ew!$4O`!e-4`*IIxXHAf9%XdWVzgg4 zN&XK<5*Bs8a`uEDd1aGbMrKDeSkx$Fec{fV{KaVJOc8Merj=IhsMb|azt5VxuBeXt$we|v{~cPHDODF1l1zYnNh(a^=GP`)$S{@H`w_5%Mw z(nxac%O2SN-DUT)P5(b)`(G>w&Nlxy5ZnL%MEn;s{y(3HbKBF{!yacldRFEY8Y4Ix z@AEdAFh|lUG6%Gr3uvM8q9|WgRVHINcC@4Y{)7M`0RJt<8nKp);iY^QoRf2pCwJI! zhD8$BYNb=Euhjc`1)gOq&S%-w@OB;bIm6u{mhZQ<$x8GVB*s^Z{%di2gSBXvJ#d!r zAK^A&vEd}peAw)QpW4H0t%mzH6ut80Na6qI&ajJ$Hs=?FiNpP?dAm1!K>Jj2>@ zxoI}gXtzWUDUEs@+f47(tj+=@SLuzKPk#=iKPfW4<|T9utwo%?=vV^1$F{r4_=hkjFhi4h>(nCe(2R@} z)(izvS-1u8M0o$P3;5@g;8`5Rw_BVk#VJ~yPlz)mQWJ5N^FFjo2X~g>#p3&ZdR?zL zPqL{#Lqts97Ej(A`I^FpEd84*+giFKu_vRaE)_7mgP*kJfUZ zydJ5iZ!GElC!dvfe_*tiMAFy=j;+26odF$!;KFKWc9(mX$wZy;AyY4}+J!@z&mePv zlJLiE-@lNmW$}cHXvP@pt16J@IkhF4aCoii`5K%dS`x{RzDO@7}bmYo*FzP7VQemMelI}$IWq%k(5p=M}Vh&c|}qOgKjXLm|H z)0V0X!0}nC#saM5DSOF@(|bJQ$bNC2R_BxOM;cEg?yB&SsQCT8+*Vdr)7N^=u-!-g z?X2?T=+DNhLjZ$SPZyrGOU|V>WZn!kZ4b)Lec&EQzU!%WdbW>5jx_<8%^MLHuMPp| zSG{ZAS-Y6uJ-+_)hkC%n=IzVb^>X8%Nj6ofVD17<=XvG)VJV zG+3=X<>27x<-cI4MRZ&s72G!Dft))G>Wrk+1E0pSTmI%a`003rBLloaM z5X8&}$O(}#GqR2=4w$zRE;D^SU+Zy}MV+hZ@!rSl)gJkl_;f8SoFf2CskN$5l`G-C zVd%0qcThjx5C)rmYzZT$T}Q_f6vascwQ7`zh=Ly8 zy0hVRdbvysK&vyQO0hsSq-Jw#W48L{%S6P3Mo%34~wZp`MPi}qp>H@g|1gek-ryc7f5Ch|k@Xh&_Es)L2 z2U^@*V`iA#M41xP>PLY}s?q#NE}q!c0*6Y1Tj}sirD)$xfHSfIIScn|GazJN1U?D< z5cUI^q(>)wMQ4hWQN=^gf_sUQZi4o``Rdv6;fCAvbRh9S%+w$keZgyy%Lwfu$sl|M z5aH?2Tx_qPcPKuW$$952$bS+miy?iv%99pxA{-D61CbU@Ddd&5pa;sNufDuIc1?g4 zRi#YB&}`NC?lk7t^#1Kr38bAO=4*k#G=%%Rsp?13Z}JQt#FR)wB(E@xSQZU=g)jAh zRXZC=G(@0l0iYW&lH2QQKvz{VKzyUkk6@WX@iC#9peTx zq*0uxd=2i$!&vz-W02~rng?YcEFUhGfrMyhk>kMJI(Q!aRJ|(;qXoxjWm*e_=$rmT zpfZNM+m@QF(iJE2@00)P(JL~1;mY*w8#k)fQY6-gPtsp8kxD~k02aj*Tk5`=o?k!FmTDl@V%OB3 zW33h`J0d*x_N2TbKn_-+M6j9Yq)Sop9z#brz@zLy`JKjYQI#%C9~A|k0KTL|E?IE7 z)zjylrNo%^jH~q?$)dj;PD>8(FJx=JY}Jac=AD^(NJe5ox?qyV_9iO*5&)Qzxz^rZ z%1`qI$&j*qFv8I3W*fVnPx~-{RmUzidWu2bt26JUKrlJN1&8@AVW!kcv*3N?EMD z1zXT$S#ZeBE;i3&|2Dlogqqc*QhM=3VK_f(9De^T5p)AH--UGN+1eP`E*z|Wi~*3) z>J0+_^6*UC!EEG?7C`B&Ll9S?ZmlRZVGgAIdO&?wYbZkb=*MqFG%P4wC70alK)39Q zLlr%8CTcWl8yMLBgv^2ha)~otm0JbK{?z$R(6{RRY_IHr^FfzydyOUb!y>I;8P%%R zx^03b`Lp0ve`*wxSh!4BO$B3E1Z2}%5vs8^Ko={8FsV}sS=|PxvTgvvpq`>mTi-aa zEN;c0Ab)vMJPHSpNk_(g6kC)DIv7wzs?()EyK+k;01C##VT~K$b6!kbF$nXX=*l)O zrd1(LZ+$^GowjufZ3obVZlGdWFUS+q_kZlxOz;Fn3-27~KTN^cH(Im=T806jojU+P z?DO!Fz5vHEKUcrDKl^b~7*Ouy6?E&I5EO5fcB{+8n+ZGXNOg z!^bBG2>dLNBN5~+f4sC!sNA-U&Jy0rQZMgv28ejxV5bgm!5_)6@UgbS%I&Ldr98gv zDZP_p#{MZGt?UvVnX_Gxksb+4lA)gI;(WYtKCneoF}3-f%{1PA_d*hq8`1d zse%rOZIxUv2CmqCFirq60}9&|5=Gn90A}|4I!K+zkfZ^$x@3VGCj@?&{HztELvX-n z@z913NS>c51Q5`eex@@#cy9CT6OZO+YcQ8Tm$!Fx1kP%hM$raZeA3_-(ep==+&rwdrZm$WrUBa9_Lv7XKM@7wP|s?1Zi4=o)g-5Nj%Jy*Mj zl9aKsxt_W;{!t0t`by#|^qjfaTg~zr+4^Dltb}5lf(?E>T#2x{6$;<1?gZu$UF~y5 zDZSzT!e$4+J5P^}!x~t&HwqfFqRcX~zVYA(0w*&`@RiQn8!O)Q8*Ls2o+&h70nvW( zdQPVtb+x17H1Aihl_^t#wyp%eQj$h5w0>285|idwQ7{r=^fh9pt@MI1XBf?!L%(eu z;Li_YAk5>omVCvu>Y(V{hXc?=ETDHVAn=-Zlfx9u0TR%jIsoQ9U4Xm-ZS&@Mu6`r6 zHPqN=UT*EI)8})h(akQGA-DLcpo1|o#{&5v;nWRNKxp}#;kVB%pcvYv=cfQJT}Sp} z$_q0(Lh6US5_7P0$ovsiXGuLcvx;vgL&xzDY0WKsS7k3K?9-bT%(39Q7!K4>T}`!2 zqva7cnz8-!X-pm|F+8d!)4Y{3}~Xd>YmmrB_h7NvS6~5>F-tD zM2h5n-DdiP^V=lJym%nBn*2`M(&6%L2r?DfWD-!Z1|}z|_5~EcgdBtem;$BtaDFK9;J13L3$(IMRD=Y*1IY2o+(<5_r&(EBcxm#YO$pWW|=1l8i zIgGxiE5@XuIo`~Vb?b-3+jmfdve+B4jA z?CN8oeQV3n89k(y@pqIi_j?FF14(cI{f75!m#?8!LbpslE5)Zo|oE!PPgQHA&Rs)}!^6VI3rz3hj*QIDB?YaJ>#&c&?NziSsRm4}zz ztNFOm4Jtasrg?zF?8mKO30-Yf~uIl7}qYCEOUa%hXGDu z&xPUk_8bVYjKLimn&o8jcGcu6{;x{s9>`BFPrml!wjsT~HPa00mYi3HO|Xnmhk!b1 z7X->PSBOyCb9S1aUpTyr#rJ9UfSQQa>(n!1pa7;D0PqXgWF;ub4Ze{dH0M|^W^GP6 zkq*%-zu2w0FGWgGjzx|}(6jYOIdHsUb~ew9ixvqL?Y=g>>J+oo^49OBRfMqAh_hO! zQL|r^l=|wW4>R0>KnY_XnqlK9GNQqDTCE%Wd^CYIknCkXA%I?!Z@X-ZzC&n8OGC-G zP(xdnav|)?3ZxwH^?GCFww83O{E;v0@qOP3i8k5WgXutySe!2p0=$pteevIAdWUn!18J9gAs>pNXW=);+g8&_<L@Y&5nD#bOpMOlZs#)boPXMm>Db6>iSktqN|wOw-l-gU9vX{hF@ z8#q3tl5nZSMHkY5RQl4h=`9b66 zy|i9uZi#_2R&;b4CHv8B**aG6`Q2DEd*jTFS4hM7VYsH(*=KV7Y2(dT<+@7N%J7mG zyx?o>D=VT*{MWeSZG-W6IP%J(V=AMW>`ZxUR`Hy&H-4Z9!&urL+EBK!rI5T$e=X-@ zLHF)0^>3H*D=aN`mXf4&BV%$b`P8~>lHd##B9wD0Sw7H}lpS(&>FHrh5$)M8&(%^Q zjvXi5foQ%=QHx)vU=jBTD!O%dRBrae*HjOyv=`n$X?$)l1wpyIYF=C2G3>hF8Whit zeUcw|O8;WV9Z(}t1-0?)uzGKk%Qi&1t-M)%0J!k+fM@0+etT^?`wc+L?Y;dC!oa~l(V1&h>bCCrWN&2 z6&X)o8Yd&wq~)cp0i;m7W-UjQUql&9iA$isB}*>(lK_>$YS_fJ%1T5Na@!(AWHRka zFB(KGk%%eTeg1B2cYoNg_F5Wmb@5-FUy**J0Bv-LULogywm8x?xh-{DieDvDUe9Zy zbn~j)aN@?JMx6nlx{T_DSWgx{3P~$R0|C0&i4+6=*Bh7W80Z8mH~jL0*XP*>%vt39 zUPni-fD!8WPT^bFYsZ3d`EgtZQ#2Zjp=cP>pY@%)cg-g}*w0LqV9hD1<>5YhRDt32 z1AQ&>m@e*)`!svqxy-)yxUI-0+=?)AI^&9XVaWJkycpCoY1^9J-r{+_(4G%~%o8M$ zYx5wMGvD~WpUOV~QQ6^rNuXdHRHZ!ANsy69$&$w{&tNeb^jx&Rsvi6vpxJkD*% z6cIL$a+g%bSW zZxodmt=$&;7%>TxOj!=N8D3q0N4@u%6q|pjqn`svb)%)NrO)d=WhAe{gtqk|HZ)-l zc_VDcn^KNrtDpJ4DNvJOGBv9UcFd25{At{Tb|M&}N9}&UW6#^G=3ig67$$EV0;g4A zU+3Ox0jD+v5rn<^_pg9jD|9fL4w<&6 zao$R!E% ziO~uEuJvfh^S0P<5whgV0@G9AG#>9})8isv5)dW9YLEVb1QR!e=We&Zca*%Y1t)}F zfW{Vabs++yG!`8DM8Nd{1*j#a?TLs&&Bzh})g5z6 ze~{nmQiOI7j3nMLU_SGoR=_dqQH2%0$u7BS!PPb>hC$frPHp|#tB*GCff{9?+b18M zGmaN>WT7WE(kg1QEF34XNC>|GVGaZas;j3#b+xN+723Kr7HHMeqNXdk3ci?g^my<@ zV1(yE7!B$XPtN{?(Nb%Q4ET6 zbocvg5L(V_RV^zmBH|Nl*>z|x{#vOI5^8#-Zn0-F51rfhN!F60-`&L>N`X9rDrQS zxqwxgB7z4<`_#@n6!H3tWCM5X5_`Z|K6h++b;;VM_cjlss`XMV;&g}iNiA}b@r1c8 zBmy6w4b-gdkiJA@dM(tNKmA!h=Qy9Xk_LobpWiB3i^xg53 z)w{7Q(QRH>p1iA2zguXzWZBksT%5;}bnr@Q#%Xbj%viS-bJH$vQ9`RWEX5dY5j?q2 z(%Dk<7!(}re4Nj&FpF<@DWv{$x$X0R1OlDM#h>}v60gFf0vr&wHkFeZtA`l@vU@Q3 z5{nk7)j=JBPer3e3bJTO87Rcmx0z8NL(XHm@6|Ct#WkKxm;?8-((%0B2YWT{J*hhp zuZ(O(%(vNSqSRu6Xz5~VHRq`J)745;C(YN5=7;^M6SJf3@BI@N2mlJZ4-kqLhv3`&EXjDQ`?u5zEQy0N_-zT#g{_*L+;T{DhxCN}#C8wT8 ztRB!@ecWqD!r0^TN@wBxQb}U1tHhtW4LgB0_ariVE#6^hf_eCLQeBmfwR&nZ$OO>nlTDz&2BV z_n2SZp3Q-lZ%(S04o%|SAIE!bnihUyqsM1GuLp9vy|zKjJz?aJ3{-TGdXkf!>nMEtg}r*%xg?YywuonW?6?tg01bC%Ux zuh9DTkG2L2TZ{~Lh+huq9qGL@R)%6V)RWywjld-t`QcoT(cnO^Q2*&*qqOa3LVzVf zd2HM_SoYDeqw#CJfU{994$e38=}CqIz}G{ejp+}GH9?vP4Ru^&;XG2+gcS@=KqIjv z#zgpKqO1Zcr2kBhNu z>lEEsnYQ(+n*uQydr;a~h>(f2iZ+X{x(1EspU#xmoJ?-H3OzEB%*8j|wf-Yn%vS`X z+~o@d5ul~~icwO0pFToJSUKmiOBr(?KWg6&SqQCfD(uj1+v%ISe)nE z*KhdlDQer%{nLvA+e>|c%niT$T!CMsegRG|XT063joyAGfQVh!XEGEo)y zN#qYA=T4@+Ydjz_#4i}x5wb61!kL!orY+!qS6`;sK%R~zL=~iEgc=Bm) zTE0WZ>qx?&1E``e=20{Hpo$QDvgH8zHRggD&aJ%Y>5Zio{ZY(-GDQlO!!LNv%Nc)4 zvT2sfDAl#Y7Ge~9C*+OcKMd3Lp$JrBl~% zL?S@kHm=Ym~paG2NX1vbn%--8`4UdvaZH{>{x5oPbVVJ+~MOZk8@8 zOHKUGw39jqiDJl5B7*&<(HIZ=BtONS94kFf5F-z6S?L@1m9NOmiZaI>_x{#-T0pFQ zgwQP9dL$J~A&Bc^orUs`eQ1>oDUvAn78nalvdK7n7f^BXKok^1jP%kWBe?jHD3n0C z0dycZ^Vkq-Af;IMj$SLfybqp~ZHhkASerL<6?lp;&l z%~m2~DPo4AZgykImXu}el_iEDNy?JSzRpC3v6g+Ov{=S&jGdw|L&I(COTRPSYpTn) zUcdhOn4!;Ep7WgNoag<1KV=<%1%Ot$y7Z2CbDXTIOt^;aM)S@;6m{`v%;dVhw`V$Un6U;3`rIB0C&EdGP+E} zG}!xpjA~?xZZ5#8qaEKR+^j?$sE~tGM$4^+N;h0{J9BC%4?=c zt>f^fdh_Ww2bdtoZD6X#@@2>KeMqB-9`?M3q){1`{@C<-YfGzymB|70#HX`@#xAe4 zNS{Q>Ti$tawf<4!pmti0H`O$I9D3M4dU3>v-Tgps+3L+nPt}5kcUdXf{T&`ug%*>6 zot)6~F9uZ0Lo+!p!+eTc`p}{?M*QMZZ(gVvfolzrlS5za9LyO{mf2rx*I(qSmVG3x zCtXtTdKUDK4;u`VymJ?7&QCXbp$H!OV8Ca=I1Z9p*y^CKzc@G$1CBANQyTT-t-Q+i zxvfq~w{&=}6cnt=O)sB4Cc?Q2j_AQO>SGwo# zFLH9Hw#M^#TiKqcSuZO`^clx{_vRRnI;l7#0Qed0y+3uG?npQ25w^phA?L1mZTl;PzsDcfNpgPn^?R zw)J2An61q^e!bS)$MNH_>x;90XhDevp5vk`bWq0+mkU2$IQTvRxbWk+oVd8(h*|

e+J z9#}&L{Mdu`#QbcIb2}dR2Od8x|rGySP(9-hF&0RFB1L77y zl&}zkdjH|4rhl7o*o%KK8)@^-;aADmgsB0rJoZY<&vZ;`?H^j&+U8PH&bj$lH^AI# z8T|t|34cX9A9kWM=qMpZjutZxeSQ5d00?8GFmrD((1Cx!npLTNcUxopSM))$mE25N z8yw#x*0DZZy;W1Z9y{vkzKt6G4^n0Wu!%0r$23h-S=l36!E!M;hH;XGV*|Qdpfara zFUkJQ#JG7#_|9SAucPwDXF9(kT96ejqD&?vWZ08B#^={9CNO~K7`_31>a2x^ z@+-jj_r^IZuT2{!ZR@hwE;;4))G@I=f|`l?x+}^5#T|lv!Hk)|0jL0U1}bMmR=8fV z0IdFt=q0+`=tZ4sA7NsyG0`^eq+O(XTvJEh(S@37b^1`RjSkHHMA9ffQe&@=BZK;T zTKd|$If6c-cbLbb0Kej6d3kwX>G`Tm%^2Abt-PVZJ?p22#;@S@AeY=v3>`cD01DBD zE}y<=JokVcqj!hC==ijH7*-pKz5g=6rKjif^p|#Z*tBQ4X=KezbZ6!JuE+hc3hmsi z4#{{NgZ$P^{U^JmR2h3#{{N`IWa5J;xRT68*m6(O#p&^1WG#m83VTAguOXg|7X+76ZfLQ4|0;zK?R{ z#Tn{L%ak9xIy=D`C+%<#C^R*uXUG|kYUq7gy{KK)7qw~624GNb6+UqA_y z&`@~G$1rqoCTx1=yx0sWrgJAGh$5GCe(eqPZRZjCHgxX3*f4n^AtKobj+tj@1S)pO zEp~PT&VJ5Oje*%xb}D|d(%*L<5+hY6hzU}V`E}{dLNcWPzIN=1mwSJ5rZlG$Xjpw# zm2$1n!O0-aven`eB>6C~(x(Ssxdw@~EDtpzAJuC(Ftj`Tt(%wF-#cPtLdR4hLK%20 z(kZR4^Ke(%DMu>@$(Ppkd=tZ@B44b1A)?Gs@=)ZqZ4VbSF&b|;5DIfn5L8bZ)jpYg zoi5&t;ored$(iR?j9QQK+`RP@^CR=dA|Z5DJzb&wKgRI=M=cZJB;L%RS~jQ-`qPfb zdLL|1Z^bYDW>4Pa`+>H5XH)zJ{%e8XU;57zObze^@4ckM8vpzK+KjSC0f%@4#Y_Dy zH;_I%v=n?@<<4fOrAR z!zQ&UOq3PwHgj_|9>(Xz5*{c?CrZv*lXFL|`cE(re}|;^g?TnSPb{VC00mKI-u~Jp zv0Yz1M~Mh^BoO{~Ou(uf@NiAy>Ty zcgnBBC;!KyeLaA|FAM#4p|#Rts!4;h?|z);jZE)0EDZghcdY+W$OzbG(u(ca2AE&E zS$=C1E8=g+&!242RAm~-H{WbGH$6`IU#$cH;%)!yyR`*{L4bov&}_!jYd@L)nlJg` z{V+^j^9CRP++Ox9uwGlerCc@;_x*V}?2ueI^z$d{o9decSRcoGa_&8(vjqNED}4o^ z^eyjywNfh|OuzCHq2=US?>CxL+Gl(%Bi8@xJ9?u7^)V9YL_k56KP-LAai!IHw6n1{ z(ayO-0=>`qw}lAuedIqer&`d&lZ`;rO2WCK?vGB4Y4H9Sw$ zwLN}2-PAGH85UO@0eSd1D=J~`Zbnb^kt(d+xvJMfENd%KGZ zVx>0ESP9Z!S$_}s6_)JF-r#MLq8Qd}<08lR$$ZYs^V73Hrnw7LF}I|_vKOz+O0!P2 z#$>S}qz~J659P+leBi#A?O7bNfSCQ>VD^mLuke(ruLD%(a)qpEX2gpDxecfgZ3h5P z#7E%UK|ra_6n4n`j0}FIc0$zn&lbG*A)_Du!?N4h7(y)`sA>kZ#=IFYTck5CzW-AG zz1ssRINGV=LyoC^2>E(bALHLllFm?h(eEC$C+PM-49=`;=IQr8q(7w`A8NipEx)Vd z?t5UTq_N#R;=RWh{WeXDlL07~Nn6&llL;9HHhto^9;{zC7XOel04d?3BSTiip~4H8 zm|oEqrtdRbdP6{*bhZ`PI%kZFG(pBOb(vNPp5SbqL9qolQoglX?k zLo0y7PGtT6_gT}Q#!LZv=%qr7N`f1B4nQmUvR7W_DZ4(nczC{QLN5m<#N^x4W0aAW zCKOoNvVoL1=F(Pb`MG9wA%B|h$%GngaD@HoeLxKyTj=%G5V_gg$qYoDz zw#4_pkT2PwtiamL{nN`@I7T_|wbzzLVGp;&nxA^UZmww;Xw3PD$Y@Q>X&Di?T;a)E z8(22oK)?0NgGopg?-|%VA{EDlAiY{#jz7cJyRZ`WDf}ohySRGJtQ!xx%MAECqrjh~yT*AOZ*1R@`7aaU z`6!UQda=5)lxjmLwjHT`T-Yex`^?Sj8y{h|R02swECF3brMdlx$cTsnqVdyT@&!SD zrK7Xcdw;)I5Myj3_tQf|AnQNlTIan*)N5GYb5<>3x^n5WYL>OJ2PidinBq6~O0>Lr&!3bevp1H1lcG_ zk37HaERHvQyhfj!*GvEmp=8B12y%0VXi?wIS)iKTDh<+q+bC&XBwXqQ0FUPZnZ0{8 zGnbovw`{*RLsOI5qcA@-Z!-yU{wEnT%)A#;#!-o?q6Oi!JlAa8;PP95EN)oZWIL03*Zhj~0`75Tvaco=<@9IUI-A`c45_L+i~DFQ?6ZB&&HmT5j*I}EN%Slpq$rDkn%bWeG!5LIg(K#3Ay0L%QVb2BrKPxyd5<>ce@JT@(}K=*z}FVh~Ni=_01#tQNVik;gg zX)5^vlgwh}Ewcd6x@{@{Bux(=WK1~^r*4`722!9d-Tk@{aJ*FBr70;+0OF5r)MccN z-wID9i8~2kmZNA~AoD6EIHio2502991nR`Pm<1qe*!`Dag2l}c4itw-*+g!TDJZCW zWF(jpnnOMe61IwIPfC=GHo9Y{uMw?4hSuA8GD%CFwQ4Mq^49n`el!fuoff3>QJy>G zs+zLPjZlHTQw3={#Xk`g>U~(oF`nCf$oJ`G#iTozW!MC1R}|o0HV>Apr^i+sur_X1z8C)%Cj%GbE5i+)0%m#Z2wGC zaX#_l_1@V}aZdNxQaII10mTo!RkqXt#G6%-tJJ!0sJWO znF(2P=0274A}v9e+c@ti`bV7ScGXR8x06hXnoqkxz%>KMsNI!yTY?91*||{)H#7w* zGwfNJ`hw6M2AX)*puBFOfz+h|0D(3)emMd3<*b5Elp?wmK1E+sgMYNeC9y>_S#cO4 zp0>;Zi8L&O`pnVO5jiUEiwFb%=RoT4!l}^J*82^#dcL&z5WiW7gF$%afl}!cob&Sy z0?Y5aSa@{xXgzDP=PV!GJaRO+(q8((tGZ9&^Ka~ur!w9jPH8;7&w*B(AOOI|k6;QA`j$190Kb#IYg^*pb=k&L@~oOL4*}FQ{yx-p^Gu zfmzfWJ5_unnB#10^&}%c15bFk6G$B&rYYycQGX-NL{1T;q>vpoYJ;h)ofKt7N9TI5 zdJi^bIKQh=OGqv+KO^Ok4FtyOV(Q-q@baiw+FD>$p6aYQ<_)&<5lm^@+hTxUkf++d8_gf4Ka~)SnpKDX7AGk$H#9c-v8_z$u_B{=CQNB#|TJ?O`B8 z+ci%%`)VtXq(&E|PJaOU_e>F-9^EC*)}Z#OhMy*=`GHS0T>-=;E;Qx=XM*PDq(x4# zSRMWLRwQl1%Au7Gm;{G|9lfzh`kSqC5$Rn3mo90@T?ooGb7mU?ie`@(i@KEDjSsJ% zcRZYzg71#Iqz!cuzBc`<_0A5LXJi_ihE z2xn>hDd1oaXv_fS-Q97`f~V3kfdkA7z$5zi(>PmM`E3_tV(%?}dT86Bs9oyVdj0*> z9#zI9B+&wH*^#D=>qib+VcV*FzD|9vcUf}Xqb_qSjE!IA8GvFq#KBXa8}OT%c}_~T zBlJq(znZztTVn+=c&%=8m++hCQ0$1es--W(r&kUU_ajI{xxu2OTl0)w<8`DyQ2OM| z!HTa_x6`^m7&g89e0ZzA$qnUi^_|<_4uE|D2WuZR#R-E->#l@g>o&xN5EDE=y2dmL z^PhjXt33~_zoL*m?HAs-5cM~al=;sfC zzkumx>8z8E`FwO829iD%|Hcqp9J~2E#cygwj+tXe11s@w!(W(U*f&zj5TO#-l z$x52qt7slx!d%v@;_Mf&zey2(GjS5kG_K;7h(ju(pYThsQ{!g5S~VZ_%`OHvjL1p4 z-oa{oSVG;kIw--Z{&>-zs!>w}x$Mu~xVP(Skf<6zi=JcwnLl)F1$SUzW^y~sY9YhCm#>nv$xH*=g+2vizp4g(jKD;=JCqM?(keH5Tvfdd4o$Ehf zgg63^#!CpXX9LQrkTf2J{Bu(%~Z{bCvcS+`P5BXwcpFO-A zE*1O9MMF}VRMP4Os+%>s73}RdGc-J;VnCXmQD_8EFZm0H0hSTRmm|y>az$6WC3V~= zOqmkWaX~eK0+yrwb7P`@YsU+)$(w>19Jc_trq(NHX{uub2o$B$4g3Q#L4c{j1+zu= zyKKJdRoe5pzf4AU4urL>X-&t@M0dEAI@{G7MkLC6VwLb(Y%9rAoqiTMcDA@LH1?>v zlsWG&chlCb0%kYsj*f~P3wFdq>scfIIhMpbVaB<&tvgD35c$TEabql74iuk{=i4y> z4hZHr*49BboUpdDFd$D~i+0m@kTT6+&{kK>Xbol>czEXT3+*l$G=Bm>=4CI1m#E`m zJ*%V8uZMaZ;AfzN+PqV&-P0J`VxoP;w$7ON$cOr&=Q*?&imV#nBR6s78yasOgAD-qD|(2w z{)yXcsQ=a=H*3YJ7KKaA^4^!uh@*ubn3(b_5-phzWTloOjU7Ftm>L})b_z2Ku~vjh z-nHP%9wys*6f}n945ud=aeXTc)IWaG(ot7OBB>EgW2~>rR8&IIWxE z^p4l3931LYaC26qavt}~ZrzsVXlw#O3iym5RxPZYb?ylT5vj^V`*#~iIDQ|e7VWJ` z<)c`{m7a{&75qbmdZrq_TEsY{LM7wrpUx^d%^or29qLt&#N_p6^7XiN&)X$wqh!GS z%F5vBx!d;?G2ra`_%%3pq0`%!M8e*r2228O+s!D^f)1VcD{&z9cSuGtZIW7);AH4S zq!R##d*E}`SMOdR)4;`@$`oP}e21)ZLp_f&3L7;ob;0$3!6)B9@X>^x|d5M(=>v!-;CrCnFt zTV27_$@YR$9y)~a&+l%D+}9INrxjuvCLI5M*|mPR^W7tiIW8VMxwzIaBShlkAuW4BfmgIXlLuQiQ= z$sp6~ghI}Vyv;7SgihiWp;Z~E(amZ&o)?_r$zfRSZJQEpLBB;I7`iBk=I|&%AU9pV58wH##S{SWn^+2hbSSh>C zFQ}ziZdOapFn(disP<2yQabQnfJVFWERdvorbF!nd9G=25FvNSN??l`wAv3pqEo}< z8oWKWf7IdD1Eoe1pdlr&`J0z5{uOw`FcZ4#4MG|H?*M}?-==1Ykz}vSnvH5m>2GG z?CLbv)LrS|&S0IsLm<5A%mW$(Z+JimR722Sdf=_}Ul8}Y`$>R+dQd+V{!zFcUdS;c zpdbR?$8w!&qSfGTHAWV74=oVqg%at0%!qH?5jO!G0n+{|dx%*=!jQNlOB1sgII9Jjp3yRL<< z$O6W-nMect5iaF#c}rwZr;h7tYvE1UegZD-#N{Ww)2K!xgW@#J+&{@rdKGfp$tYu0 z!b0lRfd~8J)3a>{a*+c%GqTv1tAJ?c@f6(0qR<_d|~7b%yUQ$ z)H|3m%a%~txE?R`)Bf)J@c5A8v%%z1m&ybaq^D>_R%BwcM}^qRyt{EudmDM!RkSow zw1v9I{d%-buUv#Zhf@x!GAqSr`I6|0`*u4K47IQ_7#-mHt26;{j2G4U65m*s-eQpq zbC5@WK|%pvJZCP2GUs3>K!!MYqB<=LVG#1py}Qekc_bDKYvy+^d2@dz8!aL~+VBLAitSOKlaMHhkXmYdb7&~0}a%f!`M~`%$sa88tkwCQWY0^iNy`8Kbq?918FFlUcam}nSNubbTDBfRa zmi;rdh(nv3KX{E>t2=t4UPg|qTzD%NMp+%X0j6Cgcc2J0KG`N(5hB)J@O01m;s4Q| ze0A8S#M`3O`-_UEQ;O6bt<#$e5_l;MYVQ_SRT`8+TI@R>d+cv4*<7F(-5q~T!ZZr` z8|#<71Iz9<%${U&)Omc3B{$>_KEJAD=CoUY)o;Cz3Lg<8VO*xR6lY8^#k<5)nv0J& z7YA_v?hkLnd>JJ|F!+uvlU{4vDBJ+>&_2$+9-hC8D#D+@vibkWD`?1v!clFfh;DVm zI=#UyYTCcPouwJa8-YjU`?VHT&hpt>lTTJGf4f@QYhCYJJngYu(O8ADp4qJ7DKw?O Q3Hs%`(v>^^2N`U*FaQ7m literal 0 HcmV?d00001 diff --git a/docs/my-website/release_notes/v1.75.8/index.md b/docs/my-website/release_notes/v1.75.8/index.md index 74243295d07..474a934743a 100644 --- a/docs/my-website/release_notes/v1.75.8/index.md +++ b/docs/my-website/release_notes/v1.75.8/index.md @@ -55,7 +55,7 @@ pip install litellm==1.75.8 ## Team Member Rate Limits

diff --git a/ui/litellm-dashboard/src/components/teams.tsx b/ui/litellm-dashboard/src/components/teams.tsx index 266d81e85d3..87bea9860d5 100644 --- a/ui/litellm-dashboard/src/components/teams.tsx +++ b/ui/litellm-dashboard/src/components/teams.tsx @@ -1383,6 +1383,20 @@ const Teams: React.FC = ({ > + + + + + + Date: Tue, 26 Aug 2025 01:22:51 +0800 Subject: [PATCH 36/66] feat: Add support for custom Anthropic-compatible API endpoints This commit adds support for custom Anthropic-compatible API endpoints that don't follow the standard /v1/messages or /v1/complete path convention. ## Changes - Added LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX environment variable - When set to true, prevents automatic appending of /v1/messages (for anthropic) - When set to true, prevents automatic appending of /v1/complete (for anthropic_text) - Added debug logging to indicate when suffix is being skipped - Maintained full backward compatibility - existing deployments are unaffected --- litellm/main.py | 24 ++++++- tests/test_litellm/test_main.py | 122 +++++++++++++++++++++++++++++++- 2 files changed, 143 insertions(+), 3 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index c8442d483e9..948c9a11890 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -2169,8 +2169,18 @@ def completion( # type: ignore # noqa: PLR0915 or "https://api.anthropic.com/v1/complete" ) - if api_base is not None and not api_base.endswith("/v1/complete"): + # Check if we should disable automatic URL suffix appending + disable_url_suffix = get_secret_bool("LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX") + if ( + api_base is not None + and not disable_url_suffix + and not api_base.endswith("/v1/complete") + ): api_base += "/v1/complete" + elif disable_url_suffix: + verbose_logger.debug( + "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX is set, skipping /v1/complete suffix" + ) response = base_llm_http_handler.completion( model=model, @@ -2206,8 +2216,18 @@ def completion( # type: ignore # noqa: PLR0915 or "https://api.anthropic.com/v1/messages" ) - if api_base is not None and not api_base.endswith("/v1/messages"): + # Check if we should disable automatic URL suffix appending + disable_url_suffix = get_secret_bool("LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX") + if ( + api_base is not None + and not disable_url_suffix + and not api_base.endswith("/v1/messages") + ): api_base += "/v1/messages" + elif disable_url_suffix: + verbose_logger.debug( + "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX is set, skipping /v1/messages suffix" + ) response = anthropic_chat_completions.completion( model=model, diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index b9a71e9621f..4b79486bb50 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -1121,4 +1121,124 @@ async def test_retrying() -> None: model="gpt-4o-mini", messages=[{"role": "user", "content": "Hello"}], ) - assert mock_request.call_count >= 10, "Expected retrying to be used" + + +def test_anthropic_disable_url_suffix_env_var(): + """Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/messages suffix.""" + from unittest.mock import patch, MagicMock + import os + from litellm import completion + from litellm.llms.anthropic.chat.handler import AnthropicLLM + + # Test with environment variable disabled (default behavior) + with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}): + with patch.object(AnthropicLLM, "__init__", return_value=None) as mock_init: + with patch.object(AnthropicLLM, "completion") as mock_completion: + mock_completion.return_value = MagicMock() + + # Mock the initialization to capture api_base + actual_api_base = None + def capture_init(self, **kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + + with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: + def capture_completion(**kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + return MagicMock() + + mock_anthropic.completion = capture_completion + + # This should append /v1/messages + completion( + model="anthropic/claude-3-sonnet", + messages=[{"role": "user", "content": "test"}], + api_key="test-key" + ) + + # Verify the api_base has /v1/messages appended + assert actual_api_base.endswith("/v1/messages") + assert actual_api_base == "https://api.example.com/v1/messages" + + # Test with environment variable enabled + with patch.dict(os.environ, { + "ANTHROPIC_API_BASE": "https://api.example.com/custom/path", + "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true" + }): + actual_api_base = None + + with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: + def capture_completion(**kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + return MagicMock() + + mock_anthropic.completion = capture_completion + + # This should NOT append /v1/messages + completion( + model="anthropic/claude-3-sonnet", + messages=[{"role": "user", "content": "test"}], + api_key="test-key" + ) + + # Verify the api_base does not have /v1/messages appended + assert actual_api_base == "https://api.example.com/custom/path" + assert not actual_api_base.endswith("/v1/messages") + + +def test_anthropic_text_disable_url_suffix_env_var(): + """Test that LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX prevents /v1/complete suffix for anthropic_text.""" + from unittest.mock import patch, MagicMock + import os + from litellm import completion + + # Test with environment variable disabled (default behavior) + with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}): + actual_api_base = None + + with patch("litellm.main.base_llm_http_handler") as mock_handler: + def capture_completion(**kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + return MagicMock() + + mock_handler.completion = capture_completion + + # This should append /v1/complete + completion( + model="anthropic_text/claude-instant-1", + messages=[{"role": "user", "content": "test"}], + api_key="test-key" + ) + + # Verify the api_base has /v1/complete appended + assert actual_api_base.endswith("/v1/complete") + assert actual_api_base == "https://api.example.com/v1/complete" + + # Test with environment variable enabled + with patch.dict(os.environ, { + "ANTHROPIC_API_BASE": "https://api.example.com/custom/complete", + "LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX": "true" + }): + actual_api_base = None + + with patch("litellm.main.base_llm_http_handler") as mock_handler: + def capture_completion(**kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + return MagicMock() + + mock_handler.completion = capture_completion + + # This should NOT append /v1/complete + completion( + model="anthropic_text/claude-instant-1", + messages=[{"role": "user", "content": "test"}], + api_key="test-key" + ) + + # Verify the api_base does not have /v1/complete appended + assert actual_api_base == "https://api.example.com/custom/complete" + assert not actual_api_base.endswith("/v1/complete") From 8dd04c4eeeac846fbe8da2ae75e31c144ae4c946 Mon Sep 17 00:00:00 2001 From: 0x5751 Date: Tue, 26 Aug 2025 01:58:04 +0800 Subject: [PATCH 37/66] fix: test error --- tests/test_litellm/test_main.py | 55 +++++++++++++++------------------ 1 file changed, 25 insertions(+), 30 deletions(-) diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 4b79486bb50..954597dda25 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -1128,38 +1128,31 @@ def test_anthropic_disable_url_suffix_env_var(): from unittest.mock import patch, MagicMock import os from litellm import completion - from litellm.llms.anthropic.chat.handler import AnthropicLLM # Test with environment variable disabled (default behavior) with patch.dict(os.environ, {"ANTHROPIC_API_BASE": "https://api.example.com"}): - with patch.object(AnthropicLLM, "__init__", return_value=None) as mock_init: - with patch.object(AnthropicLLM, "completion") as mock_completion: - mock_completion.return_value = MagicMock() - - # Mock the initialization to capture api_base - actual_api_base = None - def capture_init(self, **kwargs): - nonlocal actual_api_base - actual_api_base = kwargs.get("api_base") - - with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: - def capture_completion(**kwargs): - nonlocal actual_api_base - actual_api_base = kwargs.get("api_base") - return MagicMock() - - mock_anthropic.completion = capture_completion - - # This should append /v1/messages - completion( - model="anthropic/claude-3-sonnet", - messages=[{"role": "user", "content": "test"}], - api_key="test-key" - ) - - # Verify the api_base has /v1/messages appended - assert actual_api_base.endswith("/v1/messages") - assert actual_api_base == "https://api.example.com/v1/messages" + actual_api_base = None + + with patch("litellm.main.anthropic_chat_completions") as mock_anthropic: + def capture_completion(**kwargs): + nonlocal actual_api_base + actual_api_base = kwargs.get("api_base") + mock_response = MagicMock() + mock_response.choices = [MagicMock()] + return mock_response + + mock_anthropic.completion = capture_completion + + # This should append /v1/messages + completion( + model="anthropic/claude-3-sonnet", + messages=[{"role": "user", "content": "test"}], + api_key="test-key" + ) + + # Verify the api_base has /v1/messages appended + assert actual_api_base.endswith("/v1/messages") + assert actual_api_base == "https://api.example.com/v1/messages" # Test with environment variable enabled with patch.dict(os.environ, { @@ -1172,7 +1165,9 @@ def test_anthropic_disable_url_suffix_env_var(): def capture_completion(**kwargs): nonlocal actual_api_base actual_api_base = kwargs.get("api_base") - return MagicMock() + mock_response = MagicMock() + mock_response.choices = [MagicMock()] + return mock_response mock_anthropic.completion = capture_completion From 433d1a494780af26f875cfb2f9994eaf903a5c4e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 25 Aug 2025 13:44:54 -0700 Subject: [PATCH 38/66] [Bug fix] - Fix /messages fallback from Anthropic API -> Bedrock API (#13946) * use helper get_provider_specific_headers * fix get_provider_specific_headers * test_anthropic_messages_fallbacks * bedrock/us.anthropic.claude-sonnet-4 * fix: get_provider_specific_headers * TestProviderSpecificHeaderUtils * test_anthropic_messages_fallbacks --- .../get_provider_specific_headers.py | 23 ++++++++ litellm/llms/custom_httpx/llm_http_handler.py | 11 ++-- litellm/main.py | 13 +++-- litellm/proxy/proxy_config.yaml | 24 +++++---- .../test_anthropic_messages_passthrough.py | 52 +++++++++++++++++++ .../test_provider_specific_headers.py | 43 +++++++++++++++ 6 files changed, 148 insertions(+), 18 deletions(-) create mode 100644 litellm/litellm_core_utils/get_provider_specific_headers.py create mode 100644 tests/test_litellm/litellm_core_utils/test_provider_specific_headers.py diff --git a/litellm/litellm_core_utils/get_provider_specific_headers.py b/litellm/litellm_core_utils/get_provider_specific_headers.py new file mode 100644 index 00000000000..cf9165cfda9 --- /dev/null +++ b/litellm/litellm_core_utils/get_provider_specific_headers.py @@ -0,0 +1,23 @@ +from typing import Dict, Optional + +from litellm.types.utils import ProviderSpecificHeader + + +class ProviderSpecificHeaderUtils: + @staticmethod + def get_provider_specific_headers( + provider_specific_header: Optional[ProviderSpecificHeader], + custom_llm_provider: Optional[str], + ) -> Dict: + """ + Get the provider specific headers for the given custom llm provider + + Returns: + Optional[Dict]: The provider specific headers for the given custom llm provider + """ + if ( + provider_specific_header is not None + and provider_specific_header.get("custom_llm_provider") == custom_llm_provider + ): + return provider_specific_header.get("extra_headers", {}) + return {} \ No newline at end of file diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index c9d70088d04..d404077a5b6 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -1257,6 +1257,10 @@ class BaseLLMHTTPHandler: stream: Optional[bool] = False, kwargs: Optional[Dict[str, Any]] = None, ) -> Union[AnthropicMessagesResponse, AsyncIterator]: + from litellm.litellm_core_utils.get_provider_specific_headers import ( + ProviderSpecificHeaderUtils, + ) + if client is None or not isinstance(client, AsyncHTTPHandler): async_httpx_client = get_async_httpx_client( llm_provider=litellm.LlmProviders.ANTHROPIC @@ -1270,10 +1274,9 @@ class BaseLLMHTTPHandler: Optional[litellm.types.utils.ProviderSpecificHeader], kwargs.get("provider_specific_header", None), ) - extra_headers = ( - provider_specific_header.get("extra_headers", {}) - if provider_specific_header - else {} + extra_headers = ProviderSpecificHeaderUtils.get_provider_specific_headers( + provider_specific_header=provider_specific_header, + custom_llm_provider=custom_llm_provider, ) ( headers, diff --git a/litellm/main.py b/litellm/main.py index c8442d483e9..6102fe3ccce 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -61,6 +61,9 @@ from litellm.exceptions import LiteLLMUnknownProvider from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.audio_utils.utils import get_audio_file_for_health_check from litellm.litellm_core_utils.dd_tracing import tracer +from litellm.litellm_core_utils.get_provider_specific_headers import ( + ProviderSpecificHeaderUtils, +) from litellm.litellm_core_utils.health_check_utils import ( _create_health_check_response, _filter_model_params, @@ -1107,11 +1110,11 @@ def completion( # type: ignore # noqa: PLR0915 api_key=api_key, ) - if ( - provider_specific_header is not None - and provider_specific_header["custom_llm_provider"] == custom_llm_provider - ): - headers.update(provider_specific_header["extra_headers"]) + if provider_specific_header is not None: + headers.update(ProviderSpecificHeaderUtils.get_provider_specific_headers( + provider_specific_header=provider_specific_header, + custom_llm_provider=custom_llm_provider, + )) if model_response is not None and hasattr(model_response, "_hidden_params"): model_response._hidden_params["custom_llm_provider"] = custom_llm_provider diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index 28ba8cd093b..5748502a507 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -1,17 +1,23 @@ model_list: - - model_name: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0 + - model_name: anthropic/* litellm_params: - model: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0 + model: anthropic/* + api_key: os.environ/OPENAI_API_KEY_IJ - model_name: bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0 litellm_params: model: bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0 + - model_name: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0 + litellm_params: + model: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0 + +router_settings: + fallbacks: [ + {"anthropic/claude-opus-4-20250514": + { + "bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0" + } + } + ] litellm_settings: callbacks: ["datadog_llm_observability"] -guardrails: - - guardrail_name: "bedrock-pre-guard" - litellm_params: - guardrail: bedrock # supported values: "aporia", "bedrock", "lakera" - mode: "during_call" - guardrailIdentifier: ff6ujrregl1q - guardrailVersion: "DRAFT" \ No newline at end of file diff --git a/tests/pass_through_unit_tests/test_anthropic_messages_passthrough.py b/tests/pass_through_unit_tests/test_anthropic_messages_passthrough.py index 098daf78938..ee73efb4a3e 100644 --- a/tests/pass_through_unit_tests/test_anthropic_messages_passthrough.py +++ b/tests/pass_through_unit_tests/test_anthropic_messages_passthrough.py @@ -273,6 +273,58 @@ async def test_anthropic_messages_litellm_router_routing_strategy(): print(f"Non-streaming response: {json.dumps(response, indent=2)}") return response +@pytest.mark.asyncio +async def test_anthropic_messages_fallbacks(): + """ + E2E test the anthropic_messages fallbacks from Anthropic API to Bedrock + """ + litellm._turn_on_debug() + router = Router( + model_list=[ + { + "model_name": "anthropic/claude-opus-4-20250514", + "litellm_params": { + "model": "anthropic/claude-opus-4-20250514", + "api_key": "bad-key", + }, + }, + { + "model_name": "bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0", + "litellm_params": { + "model": "bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0", + }, + } + ], + fallbacks=[ + { + "anthropic/claude-opus-4-20250514": + ["bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0"] + } + ] + ) + + # Set up test parameters + messages = [{"role": "user", "content": "Hello, can you tell me a short joke?"}] + + # Call the handler + response = await router.aanthropic_messages( + messages=messages, + model="anthropic/claude-opus-4-20250514", + max_tokens=100, + metadata={ + "user_id": "hello", + }, + ) + + # Verify response + assert "id" in response + assert "content" in response + assert "model" in response + assert response["role"] == "assistant" + + print(f"Non-streaming response: {json.dumps(response, indent=2)}") + return response + @pytest.mark.asyncio async def test_anthropic_messages_litellm_router_latency_metadata_tracking(): diff --git a/tests/test_litellm/litellm_core_utils/test_provider_specific_headers.py b/tests/test_litellm/litellm_core_utils/test_provider_specific_headers.py new file mode 100644 index 00000000000..aa1d31c6166 --- /dev/null +++ b/tests/test_litellm/litellm_core_utils/test_provider_specific_headers.py @@ -0,0 +1,43 @@ +import pytest + +from litellm.litellm_core_utils.get_provider_specific_headers import ( + ProviderSpecificHeaderUtils, +) +from litellm.types.utils import ProviderSpecificHeader + + +class TestProviderSpecificHeaderUtils: + def test_get_provider_specific_headers_matching_provider(self): + """Test that the method returns extra_headers when custom_llm_provider matches.""" + provider_specific_header: ProviderSpecificHeader = { + "custom_llm_provider": "openai", + "extra_headers": {"Authorization": "Bearer token123", "Custom-Header": "value"} + } + custom_llm_provider = "openai" + + result = ProviderSpecificHeaderUtils.get_provider_specific_headers( + provider_specific_header, custom_llm_provider + ) + + expected = {"Authorization": "Bearer token123", "Custom-Header": "value"} + assert result == expected + + def test_get_provider_specific_headers_no_match_or_none(self): + """Test that the method returns empty dict when provider doesn't match or is None.""" + # Test case 1: Provider doesn't match + provider_specific_header: ProviderSpecificHeader = { + "custom_llm_provider": "anthropic", + "extra_headers": {"Authorization": "Bearer token123"} + } + custom_llm_provider = "openai" + + result = ProviderSpecificHeaderUtils.get_provider_specific_headers( + provider_specific_header, custom_llm_provider + ) + assert result == {} + + # Test case 2: provider_specific_header is None + result = ProviderSpecificHeaderUtils.get_provider_specific_headers( + None, "openai" + ) + assert result == {} From c7b0c57b1e8d2c7ef22372bbc5efc61f01afe79c Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 25 Aug 2025 14:11:17 -0700 Subject: [PATCH 39/66] [Bug Fix] Azure Passthrough request with streaming (#13831) * fix: _update_stream_param_based_on_request_body * test_update_stream_param_based_on_request_body * test_pass_through_request_stream_param_override --- .../pass_through_endpoints.py | 18 ++ .../passthrough/test_passthrough_main.py | 274 +++++++++++++++++- 2 files changed, 289 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py index fda5e941484..adedcaf781d 100644 --- a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py @@ -531,6 +531,19 @@ class HttpPassThroughEndpointHelpers(BasePassthroughUtils): subpath = subpath[1:] return base_target + subpath + + @staticmethod + def _update_stream_param_based_on_request_body( + parsed_body: dict, + stream: Optional[bool] = None, + ) -> Optional[bool]: + """ + If stream is provided in the request body, use it. + Otherwise, use the stream parameter passed to the `pass_through_request` function + """ + if "stream" in parsed_body: + return parsed_body.get("stream", stream) + return stream async def pass_through_request( # noqa: PLR0915 @@ -686,6 +699,11 @@ async def pass_through_request( # noqa: PLR0915 "headers": headers, }, ) + stream = HttpPassThroughEndpointHelpers._update_stream_param_based_on_request_body( + parsed_body=_parsed_body, + stream=stream, + ) + if stream: req = async_client.build_request( "POST", diff --git a/tests/test_litellm/passthrough/test_passthrough_main.py b/tests/test_litellm/passthrough/test_passthrough_main.py index 9b7e20159e4..a2008c2f336 100644 --- a/tests/test_litellm/passthrough/test_passthrough_main.py +++ b/tests/test_litellm/passthrough/test_passthrough_main.py @@ -1,13 +1,14 @@ import json import os import sys +from unittest.mock import MagicMock, patch +import httpx import pytest from fastapi.testclient import TestClient + from litellm.llms.custom_httpx.http_handler import HTTPHandler -from unittest.mock import MagicMock, patch -import httpx - + sys.path.insert( 0, os.path.abspath("../../..") ) # Adds the parent directory to the system path @@ -133,4 +134,271 @@ def test_bedrock_non_application_inference_profile_no_encoding(): actual_url = str(call_args.kwargs["url"]) assert "application-inference-profile%2F" not in actual_url assert "anthropic.claude-3-sonnet-20240229-v1:0" in actual_url + assert response.status_code == 200 + + +def test_update_stream_param_based_on_request_body(): + """ + Test _update_stream_param_based_on_request_body handles stream parameter correctly. + """ + from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( + HttpPassThroughEndpointHelpers, + ) + + # Test 1: stream in request body should take precedence + parsed_body = {"stream": True, "model": "test-model"} + result = HttpPassThroughEndpointHelpers._update_stream_param_based_on_request_body( + parsed_body=parsed_body, stream=False + ) + assert result is True + + # Test 2: no stream in request body should return original stream param + parsed_body = {"model": "test-model"} + result = HttpPassThroughEndpointHelpers._update_stream_param_based_on_request_body( + parsed_body=parsed_body, stream=False + ) + assert result is False + + # Test 3: stream=False in request body should return False + parsed_body = {"stream": False, "model": "test-model"} + result = HttpPassThroughEndpointHelpers._update_stream_param_based_on_request_body( + parsed_body=parsed_body, stream=True + ) + assert result is False + + # Test 4: no stream param provided, no stream in body + parsed_body = {"model": "test-model"} + result = HttpPassThroughEndpointHelpers._update_stream_param_based_on_request_body( + parsed_body=parsed_body, stream=None + ) + assert result is None + + +@pytest.fixture +def mock_request(): + """Create a mock request with headers""" + from typing import Optional + + class QueryParams: + def __init__(self): + self._dict = {} + + class MockRequest: + def __init__( + self, headers=None, method="POST", request_body: Optional[dict] = None + ): + self.headers = headers or {} + self.query_params = QueryParams() + self.method = method + self.request_body = request_body or {} + # Add url attribute that the actual code expects + self.url = "http://localhost:8000/test" + + async def body(self) -> bytes: + return bytes(json.dumps(self.request_body), "utf-8") + + return MockRequest + + +@pytest.fixture +def mock_user_api_key_dict(): + """Create a mock user API key dictionary""" + from litellm.proxy._types import UserAPIKeyAuth + return UserAPIKeyAuth( + api_key="test-key", + user_id="test-user", + team_id="test-team", + end_user_id="test-user", + ) + + +@pytest.mark.asyncio +async def test_pass_through_request_stream_param_override( + mock_request, mock_user_api_key_dict +): + """ + Test that when stream=None is passed as parameter but stream=True + is in request body, the request body value takes precedence and + the eventual POST request uses streaming. + """ + from unittest.mock import AsyncMock, Mock, patch + + from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( + pass_through_request, + ) + + # Create request body with stream=True + request_body = { + "model": "claude-3-5-sonnet-20241022", + "max_tokens": 256, + "messages": [{"role": "user", "content": "Hello, world"}], + "stream": True # This should override the function parameter + } + + # Create a mock streaming response + mock_response = AsyncMock() + mock_response.status_code = 200 + mock_response.headers = {"content-type": "text/event-stream"} + + # Mock the streaming response behavior + async def mock_aiter_bytes(): + yield b'data: {"content": "Hello"}\n\n' + yield b'data: {"content": "World"}\n\n' + yield b'data: [DONE]\n\n' + + mock_response.aiter_bytes = mock_aiter_bytes + + # Create mocks for the async client + mock_async_client = AsyncMock() + mock_request_obj = AsyncMock() + + # Mock build_request to return a request object (it's a sync method) + mock_async_client.build_request = Mock(return_value=mock_request_obj) + + # Mock send to return the streaming response + mock_async_client.send.return_value = mock_response + + # Mock get_async_httpx_client to return our mock client + mock_client_obj = Mock() + mock_client_obj.client = mock_async_client + + # Create the request + request = mock_request( + headers={}, method="POST", request_body=request_body + ) + + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client", + return_value=mock_client_obj, + ), patch( + "litellm.proxy.proxy_server.proxy_logging_obj.pre_call_hook", + return_value=request_body, # Return the request body unchanged + ), patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler", + new=AsyncMock(), # Mock the success handler + ): + # Call pass_through_request with stream=False parameter + response = await pass_through_request( + request=request, + target="https://api.anthropic.com/v1/messages", + custom_headers={"Authorization": "Bearer test-key"}, + user_api_key_dict=mock_user_api_key_dict, + stream=None, # This should be overridden by request body + ) + + # Verify that build_request was called (indicating streaming path) + mock_async_client.build_request.assert_called_once_with( + "POST", + httpx.URL("https://api.anthropic.com/v1/messages"), + json=request_body, + params=None, + headers={ + "Authorization": "Bearer test-key" + }, + ) + + # Verify that send was called with stream=True + mock_async_client.send.assert_called_once_with( + mock_request_obj, + stream=True # This proves that stream=True from request body was used + ) + + # Verify that the non-streaming request method was NOT called + mock_async_client.request.assert_not_called() + + # Verify response is a StreamingResponse + from fastapi.responses import StreamingResponse + assert isinstance(response, StreamingResponse) + assert response.status_code == 200 + + +@pytest.mark.asyncio +async def test_pass_through_request_stream_param_no_override( + mock_request, mock_user_api_key_dict +): + """ + Test that when stream=False is passed as parameter and no stream + is in request body, the function parameter is used and + the eventual request uses non-streaming. + """ + from unittest.mock import AsyncMock, Mock, patch + + from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( + pass_through_request, + ) + + # Create request body without stream parameter + request_body = { + "model": "claude-3-5-sonnet-20241022", + "max_tokens": 256, + "messages": [{"role": "user", "content": "Hello, world"}], + # No stream parameter - should use function parameter stream=False + } + + # Create a mock non-streaming response + mock_response = AsyncMock() + mock_response.status_code = 200 + mock_response.headers = {"content-type": "application/json"} + mock_response._content = b'{"response": "Hello world"}' + + async def mock_aread(): + return mock_response._content + + mock_response.aread = mock_aread + + # Create mocks for the async client + mock_async_client = AsyncMock() + + # Mock request to return the non-streaming response + mock_async_client.request.return_value = mock_response + + # Mock get_async_httpx_client to return our mock client + mock_client_obj = Mock() + mock_client_obj.client = mock_async_client + + # Create the request + request = mock_request( + headers={}, method="POST", request_body=request_body + ) + + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client", + return_value=mock_client_obj, + ), patch( + "litellm.proxy.proxy_server.proxy_logging_obj.pre_call_hook", + return_value=request_body, # Return the request body unchanged + ), patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler", + new=AsyncMock(), # Mock the success handler + ): + # Call pass_through_request with stream=False parameter + response = await pass_through_request( + request=request, + target="https://api.anthropic.com/v1/messages", + custom_headers={"Authorization": "Bearer test-key"}, + user_api_key_dict=mock_user_api_key_dict, + stream=False, # Should be used since no stream in request body + ) + + # Verify that build_request was NOT called (no streaming path) + mock_async_client.build_request.assert_not_called() + + # Verify that send was NOT called (no streaming path) + mock_async_client.send.assert_not_called() + + # Verify that the non-streaming request method WAS called + mock_async_client.request.assert_called_once_with( + method="POST", + url=httpx.URL("https://api.anthropic.com/v1/messages"), + headers={ + "Authorization": "Bearer test-key" + }, + params=None, + json=request_body, + ) + + # Verify response is a regular Response (not StreamingResponse) + from fastapi.responses import Response, StreamingResponse + assert not isinstance(response, StreamingResponse) + assert isinstance(response, Response) assert response.status_code == 200 \ No newline at end of file From 04cca1e7f383b5964398384a06849acd69fbf3f4 Mon Sep 17 00:00:00 2001 From: Christoph Koehler Date: Mon, 25 Aug 2025 17:18:40 -0600 Subject: [PATCH 40/66] feat: add image headers for Copilot Fixes #13696. See issue for details. --- .../github_copilot/chat/transformation.py | 28 ++++++ .../test_github_copilot_transformation.py | 96 +++++++++++++++++++ 2 files changed, 124 insertions(+) diff --git a/litellm/llms/github_copilot/chat/transformation.py b/litellm/llms/github_copilot/chat/transformation.py index 4526e6247b4..4aa6063570d 100644 --- a/litellm/llms/github_copilot/chat/transformation.py +++ b/litellm/llms/github_copilot/chat/transformation.py @@ -75,6 +75,10 @@ class GithubCopilotConfig(OpenAIConfig): initiator = self._determine_initiator(messages) validated_headers["X-Initiator"] = initiator + # Add Copilot-Vision-Request header if request contains images + if self._has_vision_content(messages): + validated_headers["Copilot-Vision-Request"] = "true" + return validated_headers def _determine_initiator(self, messages: List[AllMessageValues]) -> str: @@ -87,3 +91,27 @@ class GithubCopilotConfig(OpenAIConfig): if role in ["tool", "assistant"]: return "agent" return "user" + + def _has_vision_content(self, messages: List[AllMessageValues]) -> bool: + """ + Check if any message contains vision content (images). + Returns True if any message has content with vision-related types, otherwise False. + + Checks for: + - image_url content type (OpenAI format) + - Content items with type 'image_url' + """ + for message in messages: + content = message.get("content") + if isinstance(content, list): + # Check if any content item indicates vision content + for content_item in content: + if isinstance(content_item, dict): + # Check for image_url field (direct image URL) + if "image_url" in content_item: + return True + # Check for type field indicating image content + content_type = content_item.get("type") + if content_type == "image_url": + return True + return False diff --git a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py index f21c123579d..1c99b4f9f59 100644 --- a/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py +++ b/tests/test_litellm/llms/github_copilot/test_github_copilot_transformation.py @@ -362,3 +362,99 @@ def test_x_initiator_header_system_only_messages(): ) assert headers["X-Initiator"] == "user" + + +def test_copilot_vision_request_header_with_image(): + """Test that Copilot-Vision-Request header is added when messages contain images""" + config = GithubCopilotConfig() + + # Mock the authenticator + config.authenticator = MagicMock() + config.authenticator.get_api_key.return_value = "gh.test-key-123" + config.authenticator.get_api_base.return_value = None + + messages = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "What's in this image?"}, + { + "type": "image_url", + "image_url": {"url": "data:image/jpeg;base64,abc123"} + } + ] + } + ] + + headers = config.validate_environment( + headers={}, + model="github_copilot/gpt-4-vision-preview", + messages=messages, + optional_params={}, + litellm_params={}, + api_key=None, + api_base=None, + ) + + assert headers["Copilot-Vision-Request"] == "true" + assert headers["X-Initiator"] == "user" + + +def test_copilot_vision_request_header_text_only(): + """Test that Copilot-Vision-Request header is not added for text-only messages""" + config = GithubCopilotConfig() + + # Mock the authenticator + config.authenticator = MagicMock() + config.authenticator.get_api_key.return_value = "gh.test-key-123" + config.authenticator.get_api_base.return_value = None + + messages = [ + {"role": "user", "content": "Just a text message"}, + ] + + headers = config.validate_environment( + headers={}, + model="github_copilot/gpt-4", + messages=messages, + optional_params={}, + litellm_params={}, + api_key=None, + api_base=None, + ) + + assert "Copilot-Vision-Request" not in headers + assert headers["X-Initiator"] == "user" + + +def test_copilot_vision_request_header_with_type_image_url(): + """Test that Copilot-Vision-Request header is added for content with type: image_url""" + config = GithubCopilotConfig() + + # Mock the authenticator + config.authenticator = MagicMock() + config.authenticator.get_api_key.return_value = "gh.test-key-123" + config.authenticator.get_api_base.return_value = None + + messages = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "Analyze this image"}, + {"type": "image_url", "image_url": {"url": "https://example.com/image.jpg"}} + ] + } + ] + + headers = config.validate_environment( + headers={}, + model="github_copilot/gpt-4-vision-preview", + messages=messages, + optional_params={}, + litellm_params={}, + api_key=None, + api_base=None, + ) + + assert headers["Copilot-Vision-Request"] == "true" + assert headers["X-Initiator"] == "user" From 6ce1d82970490f4ba17b04fda2249fc3b266a0a8 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 25 Aug 2025 17:39:40 -0700 Subject: [PATCH 41/66] [Bug] Fix: Vertex Mistral not working for streaming (#13952) * fix OpenAI like chat handler * fix MockResponse * test_partner_models_httpx_streaming * test_partner_models_httpx_streaming --- litellm/llms/openai_like/chat/handler.py | 18 ++---------------- .../test_amazing_vertex_completion.py | 5 ++--- .../test_groq_streaming_encoding.py | 15 +++++++++++++-- 3 files changed, 17 insertions(+), 21 deletions(-) diff --git a/litellm/llms/openai_like/chat/handler.py b/litellm/llms/openai_like/chat/handler.py index ae0a3cc3551..821fc9b7f15 100644 --- a/litellm/llms/openai_like/chat/handler.py +++ b/litellm/llms/openai_like/chat/handler.py @@ -49,15 +49,8 @@ async def make_call( model_response = ModelResponse(**response.json()) completion_stream = MockResponseIterator(model_response=model_response) else: - # Use aiter_text with explicit UTF-8 encoding to avoid ASCII encoding errors - async def utf8_aiter_lines(): - async for line in response.aiter_text(encoding='utf-8'): - for line_part in line.splitlines(keepends=True): - if line_part.strip(): - yield line_part.rstrip('\r\n') - completion_stream = ModelResponseIterator( - streaming_response=utf8_aiter_lines(), sync_stream=False + streaming_response=response.aiter_lines(), sync_stream=False ) # LOGGING logging_obj.post_call( @@ -100,15 +93,8 @@ def make_sync_call( model_response = ModelResponse(**response.json()) completion_stream = MockResponseIterator(model_response=model_response) else: - # Use iter_text with explicit UTF-8 encoding to avoid ASCII encoding errors - def utf8_iter_lines(): - for line in response.iter_text(encoding='utf-8'): - for line_part in line.splitlines(keepends=True): - if line_part.strip(): - yield line_part.rstrip('\r\n') - completion_stream = ModelResponseIterator( - streaming_response=utf8_iter_lines(), sync_stream=True + streaming_response=response.iter_lines(), sync_stream=True ) # LOGGING diff --git a/tests/local_testing/test_amazing_vertex_completion.py b/tests/local_testing/test_amazing_vertex_completion.py index b846763737a..b908eabd0cf 100644 --- a/tests/local_testing/test_amazing_vertex_completion.py +++ b/tests/local_testing/test_amazing_vertex_completion.py @@ -910,6 +910,7 @@ async def test_partner_models_httpx(model, region, sync_mode): [ ("vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas", "us-east5"), ("vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas", "us-south1"), + ("vertex_ai/mistral-large-2411", "us-central1"), # critical - we had this issue: https://github.com/BerriAI/litellm/issues/13888 ], ) @pytest.mark.parametrize( @@ -920,7 +921,7 @@ async def test_partner_models_httpx(model, region, sync_mode): @pytest.mark.flaky(retries=3, delay=1) async def test_partner_models_httpx_streaming(model, region, sync_mode): try: - #load_vertex_ai_credentials() + load_vertex_ai_credentials() litellm._turn_on_debug() messages = [ @@ -955,8 +956,6 @@ async def test_partner_models_httpx_streaming(model, region, sync_mode): print(f"response: {response}") except litellm.RateLimitError as e: pass - except litellm.InternalServerError as e: - pass except Exception as e: if "429 Quota exceeded" in str(e): pass diff --git a/tests/test_litellm/test_groq_streaming_encoding.py b/tests/test_litellm/test_groq_streaming_encoding.py index cb1d69a4a10..e22e5ff6a0d 100644 --- a/tests/test_litellm/test_groq_streaming_encoding.py +++ b/tests/test_litellm/test_groq_streaming_encoding.py @@ -5,11 +5,14 @@ This test verifies that the OpenAI-like handler correctly handles UTF-8 encoded content in streaming responses, specifically fixing the ASCII encoding error described in issue #12660. """ -import pytest import asyncio -from unittest.mock import Mock, AsyncMock +from unittest.mock import AsyncMock, Mock + +import pytest + from litellm.llms.openai_like.chat.handler import make_call, make_sync_call + class MockResponse: """Mock httpx response for testing UTF-8 handling.""" @@ -25,6 +28,14 @@ class MockResponse: """Mock aiter_text that yields content with the specified encoding.""" yield self.test_content + def iter_lines(self): + """Mock iter_lines method for synchronous streaming.""" + yield self.test_content + + async def aiter_lines(self): + """Mock aiter_lines method for asynchronous streaming.""" + yield self.test_content + def json(self): return {"choices": [{"delta": {"content": "test"}}]} From 97e9502f4a8048d6e21e3e07cda7b35fe5fcf717 Mon Sep 17 00:00:00 2001 From: Teddy Amkie Date: Mon, 25 Aug 2025 18:41:49 -0700 Subject: [PATCH 42/66] Add DeepSeek-v3.1 pricing for Fireworks AI provider - Add fireworks_ai/accounts/fireworks/models/deepseek-v3p1 model configuration - Set context window: 128K input, 8K output tokens - Pricing: /bin/zsh.56/1M input tokens, .68/1M output tokens - Supports response schema and tool choice - Based on DeepSeek API unified pricing effective Sept 2025 --- model_prices_and_context_window.json | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a29d03f2c6c..14b7055091c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -16674,6 +16674,18 @@ "supports_tool_choice": false, "supports_response_schema": true }, + "fireworks_ai/accounts/fireworks/models/deepseek-v3p1": { + "max_tokens": 8192, + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.68e-06, + "litellm_provider": "fireworks_ai", + "mode": "chat", + "supports_response_schema": true, + "source": "https://fireworks.ai/pricing", + "supports_tool_choice": true + }, "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct": { "max_tokens": 131072, "max_input_tokens": 131072, From 1150b93e8fbc9d04a8532bff85599b685210cd56 Mon Sep 17 00:00:00 2001 From: manascb1344 Date: Tue, 26 Aug 2025 18:20:29 +0530 Subject: [PATCH 43/66] feat(constants): expand nebius_models set with DeepSeek, Llama, Qwen, NVIDIA, ZAI, and others; normalize model IDs (remove -fast variants, add new instruct/thinking variants) --- litellm/constants.py | 55 ++++++++++++++++++++++++++++++++++++-------- 1 file changed, 46 insertions(+), 9 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 72cc52fdf21..78d5e5760d1 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -638,18 +638,55 @@ featherless_ai_models: set = set([ ]) nebius_models: set = set([ + # deepseek models + "deepseek-ai/DeepSeek-R1-0528", + "deepseek-ai/DeepSeek-V3-0324", + "deepseek-ai/DeepSeek-V3", + "deepseek-ai/DeepSeek-R1", + "deepseek-ai/DeepSeek-R1-Distill-Llama-70B", + # google models + "google/gemma-2-2b-it", + "google/gemma-2-9b-it-fast", + # llama models + "meta-llama/Llama-3.3-70B-Instruct", + "meta-llama/Meta-Llama-3.1-70B-Instruct", + "meta-llama/Meta-Llama-3.1-8B-Instruct", + "meta-llama/Meta-Llama-3.1-405B-Instruct", + "NousResearch/Hermes-3-Llama-405B", + # microsoft models + "microsoft/phi-4", + # mistral models + "mistralai/Mistral-Nemo-Instruct-2407", + "mistralai/Devstral-Small-2505", + # moonshot models + "moonshotai/Kimi-K2-Instruct", + # nvidia models + "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", + "nvidia/Llama-3_3-Nemotron-Super-49B-v1", + # openai models + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + # qwen models + "Qwen/Qwen3-Coder-480B-A35B-Instruct", + "Qwen/Qwen3-235B-A22B-Instruct-2507", "Qwen/Qwen3-235B-A22B", - "Qwen/Qwen3-30B-A3B-fast", + "Qwen/Qwen3-30B-A3B", "Qwen/Qwen3-32B", "Qwen/Qwen3-14B", - "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", - "deepseek-ai/DeepSeek-V3-0324", - "deepseek-ai/DeepSeek-V3-0324-fast", - "deepseek-ai/DeepSeek-R1", - "deepseek-ai/DeepSeek-R1-fast", - "meta-llama/Llama-3.3-70B-Instruct-fast", - "Qwen/Qwen2.5-32B-Instruct-fast", - "Qwen/Qwen2.5-Coder-32B-Instruct-fast", + "Qwen/Qwen3-4B-fast", + "Qwen/Qwen2.5-Coder-7B", + "Qwen/Qwen2.5-Coder-32B-Instruct", + "Qwen/Qwen2.5-72B-Instruct", + "Qwen/QwQ-32B", + "Qwen/Qwen3-30B-A3B-Thinking-2507", + "Qwen/Qwen3-30B-A3B-Instruct-2507", + # zai models + "zai-org/GLM-4.5", + "zai-org/GLM-4.5-Air", + # other models + "aaditya/Llama3-OpenBioLLM-70B", + "ProdeusUnity/Stellar-Odyssey-12b-v0.0", + "all-hands/openhands-lm-32b-v0.1", ]) dashscope_models: set = set([ From e3d947abebefffbe63d76f2cbc48ce4c702d9680 Mon Sep 17 00:00:00 2001 From: Duc Tran Date: Tue, 26 Aug 2025 23:21:53 +0700 Subject: [PATCH 44/66] bump `orjson` version to "3.11.2" --- requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index aab643e78d4..db787b7822b 100644 --- a/requirements.txt +++ b/requirements.txt @@ -23,7 +23,7 @@ async_generator==1.10.0 # for async ollama calls langfuse==2.59.7 # for langfuse self-hosted logging prometheus_client==0.20.0 # for /metrics endpoint on proxy ddtrace==2.19.0 # for advanced DD tracing / profiling -orjson==3.10.12 # fast /embedding responses +orjson==3.11.2 # fast /embedding responses polars==1.31.0 # for data processing apscheduler==3.10.4 # for resetting budget in background fastapi-sso==0.16.0 # admin UI, SSO From 5647757ab3cb95de34d094793492bf0a6b12043a Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 26 Aug 2025 15:04:02 -0700 Subject: [PATCH 45/66] [Feat] New model `gemini-2.5-flash-image-preview` (#13979) * add gemini-2.5-flash-image-preview * add gemini-2.5-flash-image-preview --- ...odel_prices_and_context_window_backup.json | 110 ++++++++++++++++++ model_prices_and_context_window.json | 98 ++++++++++++++++ 2 files changed, 208 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 727e119e9da..0fd32d43aae 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -7909,6 +7909,55 @@ "cache_read_input_token_cost": 7.5e-08, "supports_prompt_caching": true }, + "gemini/gemini-2.5-flash-image-preview": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2.5e-06, + "output_cost_per_reasoning_token": 2.5e-06, + "output_cost_per_image": 0.039, + "litellm_provider": "gemini", + "mode": "chat", + "supports_reasoning": true, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview", + "supports_parallel_function_calling": true, + "supports_web_search": true, + "supports_url_context": true, + "tpm": 8000000, + "rpm": 100000, + "supports_pdf_input": true, + "cache_read_input_token_cost": 7.5e-08, + "supports_prompt_caching": true + }, "gemini-2.5-flash": { "max_tokens": 65535, "max_input_tokens": 1048576, @@ -8225,6 +8274,55 @@ "cache_read_input_token_cost": 2.5e-08, "supports_prompt_caching": true }, + "gemini-2.5-flash-image-preview": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2.5e-06, + "output_cost_per_reasoning_token": 2.5e-06, + "output_cost_per_image": 0.039, + "litellm_provider": "vertex_ai-language-models", + "mode": "chat", + "supports_reasoning": true, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview", + "supports_parallel_function_calling": true, + "supports_web_search": true, + "supports_url_context": true, + "tpm": 8000000, + "rpm": 100000, + "supports_pdf_input": true, + "cache_read_input_token_cost": 7.5e-08, + "supports_prompt_caching": true + }, "gemini-2.5-flash-preview-05-20": { "max_tokens": 65535, "max_input_tokens": 1048576, @@ -16688,6 +16786,18 @@ "supports_tool_choice": false, "supports_response_schema": true }, + "fireworks_ai/accounts/fireworks/models/deepseek-v3p1": { + "max_tokens": 8192, + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "input_cost_per_token": 5.6e-07, + "output_cost_per_token": 1.68e-06, + "litellm_provider": "fireworks_ai", + "mode": "chat", + "supports_response_schema": true, + "source": "https://fireworks.ai/pricing", + "supports_tool_choice": true + }, "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct": { "max_tokens": 131072, "max_input_tokens": 131072, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1b298fbbaee..0fd32d43aae 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -7909,6 +7909,55 @@ "cache_read_input_token_cost": 7.5e-08, "supports_prompt_caching": true }, + "gemini/gemini-2.5-flash-image-preview": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2.5e-06, + "output_cost_per_reasoning_token": 2.5e-06, + "output_cost_per_image": 0.039, + "litellm_provider": "gemini", + "mode": "chat", + "supports_reasoning": true, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview", + "supports_parallel_function_calling": true, + "supports_web_search": true, + "supports_url_context": true, + "tpm": 8000000, + "rpm": 100000, + "supports_pdf_input": true, + "cache_read_input_token_cost": 7.5e-08, + "supports_prompt_caching": true + }, "gemini-2.5-flash": { "max_tokens": 65535, "max_input_tokens": 1048576, @@ -8225,6 +8274,55 @@ "cache_read_input_token_cost": 2.5e-08, "supports_prompt_caching": true }, + "gemini-2.5-flash-image-preview": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 2.5e-06, + "output_cost_per_reasoning_token": 2.5e-06, + "output_cost_per_image": 0.039, + "litellm_provider": "vertex_ai-language-models", + "mode": "chat", + "supports_reasoning": true, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview", + "supports_parallel_function_calling": true, + "supports_web_search": true, + "supports_url_context": true, + "tpm": 8000000, + "rpm": 100000, + "supports_pdf_input": true, + "cache_read_input_token_cost": 7.5e-08, + "supports_prompt_caching": true + }, "gemini-2.5-flash-preview-05-20": { "max_tokens": 65535, "max_input_tokens": 1048576, From 5dba5822f23d4ed5bae622c287e2e808a46fd07d Mon Sep 17 00:00:00 2001 From: "codeflash-ai[bot]" <148906541+codeflash-ai[bot]@users.noreply.github.com> Date: Tue, 26 Aug 2025 17:25:45 -0700 Subject: [PATCH 46/66] =?UTF-8?q?=E2=9A=A1=EF=B8=8F=20Speed=20up=20functio?= =?UTF-8?q?n=20`=5Fis=5Fdebugging=5Fon`=20by=2045%=20(#13988)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The optimization eliminates unnecessary conditional branching by replacing the explicit `if-else` structure with a direct return of the boolean expression. Instead of evaluating the condition and then branching to return `True` or `False`, the optimized version directly returns the result of the boolean expression `verbose_logger.isEnabledFor(logging.DEBUG) or set_verbose is True`. This change removes Python's conditional jump overhead and reduces the number of executed bytecode instructions per function call. The line profiler shows the original version required 3 lines of execution (condition check, conditional return True, fallback return False) while the optimized version executes only 1 line. The 45% speedup is achieved by: - **Eliminating branching overhead**: No conditional jumps needed - **Reducing bytecode instructions**: From ~3 instructions to 1 instruction per call - **Leveraging Python's short-circuit evaluation**: The `or` operator still evaluates left-to-right and stops early when the first condition is True The optimization is particularly effective for this logging utility function which is likely called frequently throughout the application. All test cases show consistent 40-75% improvements across different scenarios (debug on/off, verbose flag variations, edge cases with different logging levels), demonstrating the optimization works well regardless of the boolean expression's outcome. Co-authored-by: codeflash-ai[bot] <148906541+codeflash-ai[bot]@users.noreply.github.com> --- litellm/_logging.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/litellm/_logging.py b/litellm/_logging.py index 8c23994f92a..73902d2fc5a 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -108,6 +108,7 @@ verbose_router_logger.addHandler(handler) verbose_proxy_logger.addHandler(handler) verbose_logger.addHandler(handler) + def _suppress_loggers(): """Suppress noisy loggers at INFO level""" # Suppress httpx request logging at INFO level @@ -120,6 +121,7 @@ def _suppress_loggers(): apscheduler_scheduler_logger = logging.getLogger("apscheduler.scheduler") apscheduler_scheduler_logger.setLevel(logging.WARNING) + # Call the suppression function _suppress_loggers() @@ -187,6 +189,4 @@ def _is_debugging_on() -> bool: """ Returns True if debugging is on """ - if verbose_logger.isEnabledFor(logging.DEBUG) or set_verbose is True: - return True - return False + return verbose_logger.isEnabledFor(logging.DEBUG) or set_verbose is True From 66969555062656f43b948a63c77a6552cc53674c Mon Sep 17 00:00:00 2001 From: Cole French Date: Tue, 26 Aug 2025 21:58:16 -0400 Subject: [PATCH 47/66] [Bug]: Fix tests to reference moved attributes in `braintrust_logging` module (#13978) * Mock instance attribute moved from globals * Support looking up mock message on mock choice --- .../integrations/test_braintrust_logging.py | 81 ++++++++++++------- .../integrations/test_braintrust_span_name.py | 48 ++++++----- 2 files changed, 79 insertions(+), 50 deletions(-) diff --git a/tests/test_litellm/integrations/test_braintrust_logging.py b/tests/test_litellm/integrations/test_braintrust_logging.py index cca13b4e9e6..cb227148ed9 100644 --- a/tests/test_litellm/integrations/test_braintrust_logging.py +++ b/tests/test_litellm/integrations/test_braintrust_logging.py @@ -44,17 +44,20 @@ class TestBraintrustLogger(unittest.TestCase): BraintrustLogger(api_key=None) self.assertIn("Missing keys=['BRAINTRUST_API_KEY']", str(context.exception)) - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_log_success_event_with_default_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_log_success_event_with_default_span_name(self, MockHTTPHandler): """Test log_success_event uses default span name when not provided.""" + # Mock HTTP response + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler = Mock() + mock_http_handler.post.return_value = mock_response + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - mock_response = Mock() - mock_response.json.return_value = {"id": "test-project-id"} - mock_http_handler.post.return_value = mock_response - # Create a mock response object message_mock = Mock() message_mock.json = Mock(return_value={"content": "test"}) @@ -62,6 +65,8 @@ class TestBraintrustLogger(unittest.TestCase): choice_mock = Mock() choice_mock.message = message_mock choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + # Mock the __getitem__ to support response_obj["choices"][0]["message"] + choice_mock.__getitem__ = Mock(return_value=message_mock) response_obj = Mock(spec=litellm.ModelResponse) response_obj.choices = [choice_mock] @@ -90,17 +95,20 @@ class TestBraintrustLogger(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_log_success_event_with_custom_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_log_success_event_with_custom_span_name(self, MockHTTPHandler): """Test log_success_event uses custom span name when provided.""" + # Mock HTTP response + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler = Mock() + mock_http_handler.post.return_value = mock_response + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - mock_response = Mock() - mock_response.json.return_value = {"id": "test-project-id"} - mock_http_handler.post.return_value = mock_response - # Create a mock response object message_mock = Mock() message_mock.json = Mock(return_value={"content": "test"}) @@ -108,6 +116,7 @@ class TestBraintrustLogger(unittest.TestCase): choice_mock = Mock() choice_mock.message = message_mock choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + choice_mock.__getitem__ = Mock(return_value=message_mock) response_obj = Mock(spec=litellm.ModelResponse) response_obj.choices = [choice_mock] @@ -135,17 +144,20 @@ class TestBraintrustLogger(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Custom Operation') - @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') - async def test_async_log_success_event_with_default_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.get_async_httpx_client') + async def test_async_log_success_event_with_default_span_name(self, mock_get_http_handler): """Test async_log_success_event uses default span name when not provided.""" + # Mock async HTTP response + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler = MagicMock() + mock_http_handler.post = MagicMock(return_value=mock_response) + mock_get_http_handler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - mock_response = Mock() - mock_response.json.return_value = {"id": "test-project-id"} - mock_http_handler.post = MagicMock(return_value=mock_response) - # Create a mock response object message_mock = Mock() message_mock.json = Mock(return_value={"content": "test"}) @@ -153,6 +165,7 @@ class TestBraintrustLogger(unittest.TestCase): choice_mock = Mock() choice_mock.message = message_mock choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + choice_mock.__getitem__ = Mock(return_value=message_mock) response_obj = Mock(spec=litellm.ModelResponse) response_obj.choices = [choice_mock] @@ -180,17 +193,20 @@ class TestBraintrustLogger(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') - @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') - async def test_async_log_success_event_with_custom_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.get_async_httpx_client') + async def test_async_log_success_event_with_custom_span_name(self, mock_get_http_handler): """Test async_log_success_event uses custom span name when provided.""" + # Mock async HTTP response + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler = MagicMock() + mock_http_handler.post = MagicMock(return_value=mock_response) + mock_get_http_handler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - mock_response = Mock() - mock_response.json.return_value = {"id": "test-project-id"} - mock_http_handler.post = MagicMock(return_value=mock_response) - # Create a mock response object message_mock = Mock() message_mock.json = Mock(return_value={"content": "test"}) @@ -198,6 +214,7 @@ class TestBraintrustLogger(unittest.TestCase): choice_mock = Mock() choice_mock.message = message_mock choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + choice_mock.__getitem__ = Mock(return_value=message_mock) response_obj = Mock(spec=litellm.ModelResponse) response_obj.choices = [choice_mock] @@ -225,17 +242,20 @@ class TestBraintrustLogger(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Async Custom Operation') - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_span_name_with_multiple_metadata_fields(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_span_name_with_multiple_metadata_fields(self, MockHTTPHandler): """Test that span_name works correctly alongside other metadata fields.""" + # Mock HTTP response + mock_response = Mock() + mock_response.json.return_value = {"id": "test-project-id"} + mock_http_handler = Mock() + mock_http_handler.post.return_value = mock_response + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - mock_response = Mock() - mock_response.json.return_value = {"id": "test-project-id"} - mock_http_handler.post.return_value = mock_response - # Create a mock response object message_mock = Mock() message_mock.json = Mock(return_value={"content": "test"}) @@ -243,6 +263,7 @@ class TestBraintrustLogger(unittest.TestCase): choice_mock = Mock() choice_mock.message = message_mock choice_mock.dict = Mock(return_value={"message": {"content": "test"}}) + choice_mock.__getitem__ = Mock(return_value=message_mock) response_obj = Mock(spec=litellm.ModelResponse) response_obj.choices = [choice_mock] diff --git a/tests/test_litellm/integrations/test_braintrust_span_name.py b/tests/test_litellm/integrations/test_braintrust_span_name.py index d3d98ea70af..10e512fc0ca 100644 --- a/tests/test_litellm/integrations/test_braintrust_span_name.py +++ b/tests/test_litellm/integrations/test_braintrust_span_name.py @@ -11,16 +11,18 @@ from litellm.integrations.braintrust_logging import BraintrustLogger class TestBraintrustSpanName(unittest.TestCase): """Test custom span_name functionality in Braintrust logging.""" - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_default_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_default_span_name(self, MockHTTPHandler): """Test that default span name is 'Chat Completion' when not provided.""" + # Mock HTTP response + mock_http_handler = Mock() + mock_http_handler.post.return_value = Mock() + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - # Mock HTTP response - mock_http_handler.post.return_value = Mock() - # Create a properly structured mock response response_obj = litellm.ModelResponse( id="test-id", @@ -52,16 +54,18 @@ class TestBraintrustSpanName(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Chat Completion') - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_custom_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_custom_span_name(self, MockHTTPHandler): """Test that custom span name is used when provided in metadata.""" + # Mock HTTP response + mock_http_handler = Mock() + mock_http_handler.post.return_value = Mock() + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - # Mock HTTP response - mock_http_handler.post.return_value = Mock() - # Create a properly structured mock response response_obj = litellm.ModelResponse( id="test-id", @@ -93,16 +97,18 @@ class TestBraintrustSpanName(unittest.TestCase): json_data = call_args.kwargs['json'] self.assertEqual(json_data['events'][0]['span_attributes']['name'], 'Custom Operation') - @patch('litellm.integrations.braintrust_logging.global_braintrust_sync_http_handler') - def test_span_name_with_other_metadata(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.HTTPHandler') + def test_span_name_with_other_metadata(self, MockHTTPHandler): """Test that span_name works alongside other metadata fields.""" + # Mock HTTP response + mock_http_handler = Mock() + mock_http_handler.post.return_value = Mock() + MockHTTPHandler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - # Mock HTTP response - mock_http_handler.post.return_value = Mock() - # Create a properly structured mock response response_obj = litellm.ModelResponse( id="test-id", @@ -153,16 +159,18 @@ class TestBraintrustSpanName(unittest.TestCase): # Span name should be in span_attributes, not in metadata self.assertIn('span_name', event_metadata) # span_name is also kept in metadata - @patch('litellm.integrations.braintrust_logging.global_braintrust_http_handler') - async def test_async_custom_span_name(self, mock_http_handler): + @patch('litellm.integrations.braintrust_logging.get_async_httpx_client') + async def test_async_custom_span_name(self, mock_get_http_handler): """Test async logging with custom span name.""" + # Mock async HTTP response + mock_http_handler = MagicMock() + mock_http_handler.post = MagicMock(return_value=Mock()) + mock_get_http_handler.return_value = mock_http_handler + # Setup logger = BraintrustLogger(api_key="test-key") logger.default_project_id = "test-project-id" - # Mock async HTTP response - mock_http_handler.post = MagicMock(return_value=Mock()) - # Create a properly structured mock response response_obj = litellm.ModelResponse( id="test-id", From ab7efaa832322312e254f6c1660b4568e9942698 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 26 Aug 2025 19:15:17 -0700 Subject: [PATCH 48/66] test_pre_process_non_default_params (#13990) --- litellm/main.py | 1 + litellm/utils.py | 9 +-------- tests/test_litellm/test_utils.py | 8 +++++++- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 6102fe3ccce..70f55125507 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1256,6 +1256,7 @@ def completion( # type: ignore # noqa: PLR0915 additional_drop_params=kwargs.get("additional_drop_params"), remove_sensitive_keys=True, add_provider_specific_params=True, + provider_config=provider_config, ) if litellm.add_function_to_prompt and optional_params.get( diff --git a/litellm/utils.py b/litellm/utils.py index aa3c00735ec..40a1438b3fe 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -3088,6 +3088,7 @@ def pre_process_non_default_params( model: str, remove_sensitive_keys: bool = False, add_provider_specific_params: bool = False, + provider_config: Optional[BaseConfig] = None, ) -> dict: """ Pre-process non-default params to a standardized format @@ -3103,14 +3104,6 @@ def pre_process_non_default_params( additional_endpoint_specific_params=["messages"], ) - provider_config: Optional[BaseConfig] = None - if custom_llm_provider is not None and custom_llm_provider in [ - provider.value for provider in LlmProviders - ]: - provider_config = ProviderConfigManager.get_provider_chat_config( - model=model, provider=LlmProviders(custom_llm_provider) - ) - if "response_format" in non_default_params: if provider_config is not None: non_default_params[ diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index bd39fbfc9c4..dba40e2214a 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -957,7 +957,12 @@ def test_get_model_info_shows_supports_computer_use(): def test_pre_process_non_default_params(model, custom_llm_provider): from pydantic import BaseModel - from litellm.utils import pre_process_non_default_params + from litellm.utils import ProviderConfigManager, pre_process_non_default_params + + provider_config = ProviderConfigManager.get_provider_chat_config( + model=model, + provider=LlmProviders(custom_llm_provider) + ) class ResponseFormat(BaseModel): x: str @@ -974,6 +979,7 @@ def test_pre_process_non_default_params(model, custom_llm_provider): special_params=special_params, custom_llm_provider=custom_llm_provider, additional_drop_params=None, + provider_config=provider_config, ) print(processed_non_default_params) assert processed_non_default_params == { From 0f5b31fd78592c7c8144efe1b372c53e51d27d11 Mon Sep 17 00:00:00 2001 From: Ifta Khairul Alam Adil Date: Wed, 27 Aug 2025 04:17:35 +0200 Subject: [PATCH 49/66] fix: resolve invalid model name error for Gemini Imagen models (#13851) - Fix URL construction in Gemini image generation to strip 'gemini/' prefix - Google AI API expects base model name without the prefix - Update model references and pricing information for consistency - Remove outdated image generation pricing entries Fixes issue where models like 'gemini/imagen-4.0-fast-generate-preview-06-06' were being rejected by the Google AI API due to incorrect URL formatting. --- .../providers/google_ai_studio/image_gen.md | 10 +++--- .../my-website/docs/providers/vertex_image.md | 4 +-- .../release_notes/v1.74.15-stable/index.md | 6 ++-- .../gemini/image_generation/transformation.py | 6 +++- ...odel_prices_and_context_window_backup.json | 36 ------------------- model_prices_and_context_window.json | 36 ------------------- .../image_gen_tests/test_image_generation.py | 2 +- tests/llm_translation/test_optional_params.py | 2 +- 8 files changed, 17 insertions(+), 85 deletions(-) diff --git a/docs/my-website/docs/providers/google_ai_studio/image_gen.md b/docs/my-website/docs/providers/google_ai_studio/image_gen.md index f4e96d5225a..31b1766e450 100644 --- a/docs/my-website/docs/providers/google_ai_studio/image_gen.md +++ b/docs/my-website/docs/providers/google_ai_studio/image_gen.md @@ -42,7 +42,7 @@ os.environ["GEMINI_API_KEY"] = "your-api-key-here" # Generate a single image response = litellm.image_generation( - model="gemini/imagen-4.0-generate-preview-06-06", + model="gemini/imagen-4.0-generate-001", prompt="A cute baby sea otter swimming in crystal clear water" ) @@ -64,7 +64,7 @@ async def generate_image(): # Generate image asynchronously response = await litellm.aimage_generation( - model="gemini/imagen-4.0-generate-preview-06-06", + model="gemini/imagen-4.0-generate-001", prompt="A beautiful sunset over mountains with vibrant colors", n=1, ) @@ -89,7 +89,7 @@ os.environ["GEMINI_API_KEY"] = "your-api-key-here" # Generate image with additional parameters response = litellm.image_generation( - model="gemini/imagen-4.0-generate-preview-06-06", + model="gemini/imagen-4.0-generate-001", prompt="A futuristic cityscape at night with neon lights", n=1, size="1024x1024", @@ -112,7 +112,7 @@ for image in response.data: model_list: - model_name: google-imagen litellm_params: - model: gemini/imagen-4.0-generate-preview-06-06 + model: gemini/imagen-4.0-generate-001 api_key: os.environ/GEMINI_API_KEY model_info: mode: image_generation @@ -198,7 +198,7 @@ Google AI Studio Image Generation supports the following OpenAI-compatible param | Parameter | Type | Description | Default | Example | |-----------|------|-------------|---------|---------| | `prompt` | string | Text description of the image to generate | Required | `"A sunset over the ocean"` | -| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-preview-06-06"` | +| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-001"` | | `n` | integer | Number of images to generate (1-4) | `1` | `2` | | `size` | string | Image dimensions | `"1024x1024"` | `"512x512"`, `"1024x1024"` | diff --git a/docs/my-website/docs/providers/vertex_image.md b/docs/my-website/docs/providers/vertex_image.md index 2434c3a9a57..27e584cb222 100644 --- a/docs/my-website/docs/providers/vertex_image.md +++ b/docs/my-website/docs/providers/vertex_image.md @@ -18,7 +18,7 @@ import litellm # Generate a single image response = await litellm.aimage_generation( prompt="An olympic size swimming pool with crystal clear water and modern architecture", - model="vertex_ai/imagen-4.0-generate-preview-06-06", + model="vertex_ai/imagen-4.0-generate-001", vertex_ai_project="your-project-id", vertex_ai_location="us-central1", ) @@ -34,7 +34,7 @@ print(response.data[0].url) model_list: - model_name: vertex-imagen litellm_params: - model: vertex_ai/imagen-4.0-generate-preview-06-06 + model: vertex_ai/imagen-4.0-generate-001 vertex_ai_project: "your-project-id" vertex_ai_location: "us-central1" vertex_ai_credentials: "path/to/service-account.json" # Optional if using environment auth diff --git a/docs/my-website/release_notes/v1.74.15-stable/index.md b/docs/my-website/release_notes/v1.74.15-stable/index.md index dd748f18ffa..9807a00b7e7 100644 --- a/docs/my-website/release_notes/v1.74.15-stable/index.md +++ b/docs/my-website/release_notes/v1.74.15-stable/index.md @@ -86,9 +86,9 @@ This is great to central AI Platform teams looking to track how they are helping | Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Cost per Image | | ----------- | -------------------------------------- | -------------- | ------------------- | -------------------- | -------------- | | OpenRouter | `openrouter/x-ai/grok-4` | 256k | $3 | $15 | N/A | -| Google AI Studio | `gemini/imagen-4.0-generate-preview-06-06` | N/A | N/A | N/A | $0.04 | -| Google AI Studio | `gemini/imagen-4.0-ultra-generate-preview-06-06` | N/A | N/A | N/A | $0.06 | -| Google AI Studio | `gemini/imagen-4.0-fast-generate-preview-06-06` | N/A | N/A | N/A | $0.02 | +| Google AI Studio | `gemini/imagen-4.0-generate-001` | N/A | N/A | N/A | $0.04 | +| Google AI Studio | `gemini/imagen-4.0-ultra-generate-001` | N/A | N/A | N/A | $0.06 | +| Google AI Studio | `gemini/imagen-4.0-fast-generate-001` | N/A | N/A | N/A | $0.02 | | Google AI Studio | `gemini/imagen-3.0-generate-002` | N/A | N/A | N/A | $0.04 | | Google AI Studio | `gemini/imagen-3.0-generate-001` | N/A | N/A | N/A | $0.04 | | Google AI Studio | `gemini/imagen-3.0-fast-generate-001` | N/A | N/A | N/A | $0.02 | diff --git a/litellm/llms/gemini/image_generation/transformation.py b/litellm/llms/gemini/image_generation/transformation.py index e57364fd288..30799991d35 100644 --- a/litellm/llms/gemini/image_generation/transformation.py +++ b/litellm/llms/gemini/image_generation/transformation.py @@ -88,14 +88,18 @@ class GoogleImageGenConfig(BaseImageGenerationConfig): Google AI API format: https://generativelanguage.googleapis.com/v1beta/models/{model}:predict """ + from litellm.llms.gemini.common_utils import GeminiModelInfo + complete_url: str = ( api_base or get_secret_str("GEMINI_API_BASE") or self.DEFAULT_BASE_URL ) + # Strip the "gemini/" prefix from the model name for the API call + base_model = GeminiModelInfo.get_base_model(model) or model complete_url = complete_url.rstrip("/") - complete_url = f"{complete_url}/models/{model}:predict" + complete_url = f"{complete_url}/models/{base_model}:predict" return complete_url def validate_environment( diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 0fd32d43aae..104b5bb7936 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -10180,18 +10180,6 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "vertex_ai/imagen-4.0-generate-preview-06-06": { - "output_cost_per_image": 0.04, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, - "vertex_ai/imagen-4.0-ultra-generate-preview-06-06": { - "output_cost_per_image": 0.06, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "vertex_ai/imagen-4.0-ultra-generate-001": { "output_cost_per_image": 0.06, "litellm_provider": "vertex_ai-image-models", @@ -10204,12 +10192,6 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "vertex_ai/imagen-4.0-fast-generate-preview-06-06": { - "output_cost_per_image": 0.02, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "vertex_ai/imagen-3.0-generate-002": { "output_cost_per_image": 0.04, "litellm_provider": "vertex_ai-image-models", @@ -10916,36 +10898,18 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-generate-preview-06-06": { - "output_cost_per_image": 0.04, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-4.0-ultra-generate-001": { "output_cost_per_image": 0.06, "litellm_provider": "gemini", "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-ultra-generate-preview-06-06": { - "output_cost_per_image": 0.06, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-4.0-fast-generate-001": { "output_cost_per_image": 0.02, "litellm_provider": "gemini", "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-fast-generate-preview-06-06": { - "output_cost_per_image": 0.02, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-3.0-generate-002": { "output_cost_per_image": 0.04, "litellm_provider": "gemini", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 0fd32d43aae..104b5bb7936 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -10180,18 +10180,6 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "vertex_ai/imagen-4.0-generate-preview-06-06": { - "output_cost_per_image": 0.04, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, - "vertex_ai/imagen-4.0-ultra-generate-preview-06-06": { - "output_cost_per_image": 0.06, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "vertex_ai/imagen-4.0-ultra-generate-001": { "output_cost_per_image": 0.06, "litellm_provider": "vertex_ai-image-models", @@ -10204,12 +10192,6 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "vertex_ai/imagen-4.0-fast-generate-preview-06-06": { - "output_cost_per_image": 0.02, - "litellm_provider": "vertex_ai-image-models", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "vertex_ai/imagen-3.0-generate-002": { "output_cost_per_image": 0.04, "litellm_provider": "vertex_ai-image-models", @@ -10916,36 +10898,18 @@ "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-generate-preview-06-06": { - "output_cost_per_image": 0.04, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-4.0-ultra-generate-001": { "output_cost_per_image": 0.06, "litellm_provider": "gemini", "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-ultra-generate-preview-06-06": { - "output_cost_per_image": 0.06, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-4.0-fast-generate-001": { "output_cost_per_image": 0.02, "litellm_provider": "gemini", "mode": "image_generation", "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, - "gemini/imagen-4.0-fast-generate-preview-06-06": { - "output_cost_per_image": 0.02, - "litellm_provider": "gemini", - "mode": "image_generation", - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" - }, "gemini/imagen-3.0-generate-002": { "output_cost_per_image": 0.04, "litellm_provider": "gemini", diff --git a/tests/image_gen_tests/test_image_generation.py b/tests/image_gen_tests/test_image_generation.py index 043a0a88fd1..1f51efa659f 100644 --- a/tests/image_gen_tests/test_image_generation.py +++ b/tests/image_gen_tests/test_image_generation.py @@ -175,7 +175,7 @@ class TestAimlImageGeneration(BaseImageGenTest): class TestGoogleImageGen(BaseImageGenTest): def get_base_image_generation_call_args(self) -> dict: - return {"model": "gemini/imagen-4.0-generate-preview-06-06"} + return {"model": "gemini/imagen-4.0-generate-001"} class TestAzureOpenAIDalle3(BaseImageGenTest): def get_base_image_generation_call_args(self) -> dict: diff --git a/tests/llm_translation/test_optional_params.py b/tests/llm_translation/test_optional_params.py index 0e1e34362b0..dd42d5e1c5b 100644 --- a/tests/llm_translation/test_optional_params.py +++ b/tests/llm_translation/test_optional_params.py @@ -1557,7 +1557,7 @@ def test_azure_ai_cohere_embed_input_type_param(): def test_optional_params_image_gen_with_aspect_ratio(): optional_params = get_optional_params_image_gen( - model="imagen-4.0-ultra-generate-preview-06-06", + model="imagen-4.0-ultra-generate-001", custom_llm_provider="vertex_ai", aspect_ratio="16:9", ) From 80d4cc12837a2144feeba57ed024038f4272ed73 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 26 Aug 2025 19:47:25 -0700 Subject: [PATCH 50/66] [Perf] Use fastuuid for fast UUID generations - 2.1x Faster (#13992) * use fastuuid * add fastuuid * add fastuuid==0.12.0 --- .circleci/requirements.txt | 3 ++- litellm/types/utils.py | 2 +- poetry.lock | 37 +++++++++++++++++++++++++++- pyproject.toml | 1 + requirements.txt | 1 + test_profile_mock_response.py.lprof | Bin 0 -> 3495 bytes 6 files changed, 41 insertions(+), 3 deletions(-) create mode 100644 test_profile_mock_response.py.lprof diff --git a/.circleci/requirements.txt b/.circleci/requirements.txt index f41a5291e50..8e0f1dfe7e9 100644 --- a/.circleci/requirements.txt +++ b/.circleci/requirements.txt @@ -14,4 +14,5 @@ google-cloud-iam==2.19.1 fastapi-sso==0.16.0 uvloop==0.21.0 mcp==1.10.1 # for MCP server -semantic_router==0.1.10 # for auto-routing with litellm \ No newline at end of file +semantic_router==0.1.10 # for auto-routing with litellm +fastuuid==0.12.0 \ No newline at end of file diff --git a/litellm/types/utils.py b/litellm/types/utils.py index fad46a8b9f5..30d46d61c93 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1,6 +1,5 @@ import json import time -import uuid from enum import Enum from typing import ( TYPE_CHECKING, @@ -14,6 +13,7 @@ from typing import ( Union, ) +import fastuuid as uuid from aiohttp import FormData from openai._models import BaseModel as OpenAIObject from openai.types.audio.transcription_create_params import FileTypes # type: ignore diff --git a/poetry.lock b/poetry.lock index 66bdc2a4ffb..29d1a877087 100644 --- a/poetry.lock +++ b/poetry.lock @@ -1250,6 +1250,41 @@ lz4 = ["lz4"] snappy = ["cramjam"] zstandard = ["zstandard"] +[[package]] +name = "fastuuid" +version = "0.12.0" +description = "Python bindings to Rust's UUID library." +optional = false +python-versions = ">=3.8" +groups = ["main"] +files = [ + {file = "fastuuid-0.12.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:22a900ef0956aacf862b460e20541fdae2d7c340594fe1bd6fdcb10d5f0791a9"}, + {file = "fastuuid-0.12.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0302f5acf54dc75de30103025c5a95db06d6c2be36829043a0aa16fc170076bc"}, + {file = "fastuuid-0.12.0-cp310-cp310-manylinux_2_34_x86_64.whl", hash = "sha256:7946b4a310cfc2d597dcba658019d72a2851612a2cebb949d809c0e2474cf0a6"}, + {file = "fastuuid-0.12.0-cp310-cp310-win_amd64.whl", hash = "sha256:a1b6764dd42bf0c46c858fb5ade7b7a3d93b7a27485a7a5c184909026694cd88"}, + {file = "fastuuid-0.12.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:2bced35269315d16fe0c41003f8c9d63f2ee16a59295d90922cad5e6a67d0418"}, + {file = "fastuuid-0.12.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:82106e4b0a24f4f2f73c88f89dadbc1533bb808900740ca5db9bbb17d3b0c824"}, + {file = "fastuuid-0.12.0-cp311-cp311-manylinux_2_34_x86_64.whl", hash = "sha256:4db1bc7b8caa1d7412e1bea29b016d23a8d219131cff825b933eb3428f044dca"}, + {file = "fastuuid-0.12.0-cp311-cp311-win_amd64.whl", hash = "sha256:07afc8e674e67ac3d35a608c68f6809da5fab470fb4ef4469094fdb32ba36c51"}, + {file = "fastuuid-0.12.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:328694a573fe9dce556b0b70c9d03776786801e028d82f0b6d9db1cb0521b4d1"}, + {file = "fastuuid-0.12.0-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:02acaea2c955bb2035a7d8e7b3fba8bd623b03746ae278e5fa932ef54c702f9f"}, + {file = "fastuuid-0.12.0-cp312-cp312-manylinux_2_34_x86_64.whl", hash = "sha256:ed9f449cba8cf16cced252521aee06e633d50ec48c807683f21cc1d89e193eb0"}, + {file = "fastuuid-0.12.0-cp312-cp312-win_amd64.whl", hash = "sha256:0df2ea4c9db96fd8f4fa38d0e88e309b3e56f8fd03675a2f6958a5b082a0c1e4"}, + {file = "fastuuid-0.12.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:7fe2407316a04ee8f06d3dbc7eae396d0a86591d92bafe2ca32fce23b1145786"}, + {file = "fastuuid-0.12.0-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b9b31dd488d0778c36f8279b306dc92a42f16904cba54acca71e107d65b60b0c"}, + {file = "fastuuid-0.12.0-cp313-cp313-manylinux_2_34_x86_64.whl", hash = "sha256:b19361ee649365eefc717ec08005972d3d1eb9ee39908022d98e3bfa9da59e37"}, + {file = "fastuuid-0.12.0-cp313-cp313-win_amd64.whl", hash = "sha256:8fc66b11423e6f3e1937385f655bedd67aebe56a3dcec0cb835351cfe7d358c9"}, + {file = "fastuuid-0.12.0-cp38-cp38-macosx_11_0_arm64.whl", hash = "sha256:7b15c54d300279ab20a9cc0579ada9c9f80d1bc92997fc61fb7bf3103d7cb26b"}, + {file = "fastuuid-0.12.0-cp38-cp38-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:458f1bc3ebbd76fdb89ad83e6b81ccd3b2a99fa6707cd3650b27606745cfb170"}, + {file = "fastuuid-0.12.0-cp38-cp38-manylinux_2_34_x86_64.whl", hash = "sha256:a8f0f83fbba6dc44271a11b22e15838641b8c45612cdf541b4822a5930f6893c"}, + {file = "fastuuid-0.12.0-cp38-cp38-win_amd64.whl", hash = "sha256:7cfd2092253d3441f6a8c66feff3c3c009da25a5b3da82bc73737558543632be"}, + {file = "fastuuid-0.12.0-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:9303617e887429c193d036d47d0b32b774ed3618431123e9106f610d601eb57e"}, + {file = "fastuuid-0.12.0-cp39-cp39-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:8790221325b376e1122e95f865753ebf456a9fb8faf0dca4f9bf7a3ff620e413"}, + {file = "fastuuid-0.12.0-cp39-cp39-manylinux_2_34_x86_64.whl", hash = "sha256:e4b12d3e23515e29773fa61644daa660ceb7725e05397a986c2109f512579a48"}, + {file = "fastuuid-0.12.0-cp39-cp39-win_amd64.whl", hash = "sha256:e41656457c34b5dcb784729537ea64c7d9bbaf7047b480c6c6a64c53379f455a"}, + {file = "fastuuid-0.12.0.tar.gz", hash = "sha256:d0bd4e5b35aad2826403f4411937c89e7c88857b1513fe10f696544c03e9bd8e"}, +] + [[package]] name = "filelock" version = "3.16.1" @@ -6541,4 +6576,4 @@ utils = ["numpydoc"] [metadata] lock-version = "2.1" python-versions = ">=3.8.1,<4.0, !=3.9.7" -content-hash = "17a23611c832b757244c5b5dfd3a6eadae4699602a823587ea115513bfca8e4d" +content-hash = "f41e6359109c5c52dba2a28f301b04030d865265f408974082b390bf45568a01" diff --git a/pyproject.toml b/pyproject.toml index 7c3f24c89c8..46b777b8f7c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -20,6 +20,7 @@ Documentation = "https://docs.litellm.ai" [tool.poetry.dependencies] python = ">=3.8.1,<4.0, !=3.9.7" +fastuuid = ">=0.12.0" httpx = ">=0.23.0" openai = ">=1.99.5" python-dotenv = ">=0.2.0" diff --git a/requirements.txt b/requirements.txt index aab643e78d4..598c1462b8d 100644 --- a/requirements.txt +++ b/requirements.txt @@ -7,6 +7,7 @@ backoff==2.2.1 # server dep pyyaml==6.0.2 # server dep uvicorn==0.29.0 # server dep gunicorn==23.0.0 # server dep +fastuuid==0.12.0 # for uuid4 uvloop==0.21.0 # uvicorn dep, gives us much better performance under load boto3==1.36.0 # aws bedrock/sagemaker calls redis==5.2.1 # redis caching diff --git a/test_profile_mock_response.py.lprof b/test_profile_mock_response.py.lprof new file mode 100644 index 0000000000000000000000000000000000000000..b99ff91148b35d291799e1aef47340275b7e5073 GIT binary patch literal 3495 zcmY+`ZERCz6u|KgR#pauj#l2u6NeOo zZp^Axi%uju#g`gQoRJTc__F8+mWVh-O?`%%!NK_cxGkC$4m1aX zyFy@Q*dGf2dworF%Z@!Yt^SbO5)AoE)jQ&DV{{4`|BJ*sQxsj=Dz^S>YR+!=03nUVg#Sg!NsPBgCWjQgWY$~K4mt)a3&sIjgt zxVx^Q!QWc8HV|%Xt1oK`g#ArT&Hw#xt_uXEEzj|>(d->9&1#cB9B2vZ{dmS?dMqWb zf#wozG0(V6k9EDvG1sZJ;>jXnIKMGdg9#{;_DNbcNVt9wOa#t(ACA#r2FW){B{B(^ zPAwEn2E(W32`pgb+-X4$h^(iBDWLa6kwkJqtZ7bLC#kjs*tZl&q#?>ESGC`fd1_n85(wyW?4kq zUf{TqMQ;a<<=plips$~_J3;?k8tnjyCpiB5^|ZHgWt{^cy(Nc$^IbOUFdg(=@=K%L z6472)?`%+o_mP;W*^{jy3c4Poq5Hv5oM$)&yd@m6IQ5bh^-|yns5Nhp0*{i`pHD-N z0cQv30R#Iu=8uEH}W(jmh-4*I7q6Fkd0zAmreIU?GX+zVB2i@IDbG zmk!Q=;SHRT4?t=@7y`-5-06qF{v`MsI4<#0O95X2ga45`ReH9{cz*(k)kjxqIHwmF z%S6u;@xB_Cv=LA{?jy5?pGmVVIw_6*s+ZPZ0vAYA%P&aeH$Ae1Tet{Pf6~DpV6=!Q z)nMe?*yIrGT| zm8-!7P-`Y_B8YVnF@xbjB9nmOdB!*ybXAaM0Yfi<9MDq@rhvXGkPBi)s!av<9wK?b zd6d%m`ZZqy1*FgKbdV}was_(Yrxch0oYO!dF!X|%p#LQ>3#9LKHb|GYg1#CSodX6p zbIn|!90Nrlc@Y!?LlKy#Ltk`0@CJyKfK-y#=M`*Z>;iAHg+`CPt=ri6^KVbd!B|Ah z@k6no3|M0gvg3Bpvu(u!4U4Il-tQ$~a0Q!x6=?jD0WSr%N>C0)RUX0$;JlkNvJ7+; z0|!W-k!$p8zTin&O9vyrZ;^JKM%T!~EQ8B>=47768%b>cnq%t%jwJQmz-a-=XSn5>-97j1$KbgZr*=8L8FIT@PpJYUZ4%YR83?T@QpCUMi5hp1VCgPUGLV@ z9AFQ-I{LR=#@j?Bxsz*xKzW-3)1o6r*D6Sz=C{&qAhwQnT7mUt?lc5!kMQ*p2JN?k zHjwyYl`Og!*iUiWx9e&9*phZ&ae#d^I=m()OS|SyBI!Nt0IH3&{UC9Po4X4n_ksh! z;^K)s2(0CFeK+X)o@0K9b$pRFIp+5Wx{bp}>YkD>zkBr(dPAL{=L{q10*)Ukei#gX z1iC?^AKVApALbns1 Date: Tue, 26 Aug 2025 19:51:02 -0700 Subject: [PATCH 51/66] installing_litellm_on_python --- .circleci/config.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.circleci/config.yml b/.circleci/config.yml index e49dca6faae..faf43ff0b8b 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -1291,6 +1291,7 @@ jobs: pip install jinja2 pip install "tokenizers==0.20.0" pip install "uvloop==0.21.0" + pip install "fastuuid==0.12.0" pip install jsonschema - setup_litellm_enterprise_pip - run: From ee324943d732ae3651d25010f98b18eb049acff9 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Tue, 26 Aug 2025 20:02:57 -0700 Subject: [PATCH 52/66] =?UTF-8?q?bump:=20version=201.76.0=20=E2=86=92=201.?= =?UTF-8?q?76.1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- pyproject.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 46b777b8f7c..2b9650a1527 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "litellm" -version = "1.76.0" +version = "1.76.1" description = "Library to easily interface with LLM API providers" authors = ["BerriAI"] license = "MIT" @@ -156,7 +156,7 @@ requires = ["poetry-core", "wheel"] build-backend = "poetry.core.masonry.api" [tool.commitizen] -version = "1.76.0" +version = "1.76.1" version_files = [ "pyproject.toml:^version" ] From 2d515e72ffa43ef6799118a839b320c262b65592 Mon Sep 17 00:00:00 2001 From: Ifta Khairul Alam Adil Date: Wed, 27 Aug 2025 11:53:47 +0200 Subject: [PATCH 53/66] refactor: simplify URL construction for Gemini image generation - Remove unnecessary model name prefix stripping - Directly use the model name in the API URL construction This change streamlines the URL generation process for the Google AI API, ensuring compatibility with model names without the 'gemini/' prefix. Signed-off-by: Ifta Khairul Alam Adil --- litellm/llms/gemini/image_generation/transformation.py | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/litellm/llms/gemini/image_generation/transformation.py b/litellm/llms/gemini/image_generation/transformation.py index 30799991d35..a78a5b9dda7 100644 --- a/litellm/llms/gemini/image_generation/transformation.py +++ b/litellm/llms/gemini/image_generation/transformation.py @@ -88,18 +88,15 @@ class GoogleImageGenConfig(BaseImageGenerationConfig): Google AI API format: https://generativelanguage.googleapis.com/v1beta/models/{model}:predict """ - from litellm.llms.gemini.common_utils import GeminiModelInfo - + complete_url: str = ( api_base or get_secret_str("GEMINI_API_BASE") or self.DEFAULT_BASE_URL ) - # Strip the "gemini/" prefix from the model name for the API call - base_model = GeminiModelInfo.get_base_model(model) or model complete_url = complete_url.rstrip("/") - complete_url = f"{complete_url}/models/{base_model}:predict" + complete_url = f"{complete_url}/models/{model}:predict" return complete_url def validate_environment( From ab4cd48c40583e54a2eb882e8d915f29cad5ccbb Mon Sep 17 00:00:00 2001 From: Ifta Khairul Alam Adil Date: Wed, 27 Aug 2025 11:55:19 +0200 Subject: [PATCH 54/66] refactor: remove unnecessary whitespace in GoogleImageGenConfig - Cleaned up the code by removing an extra blank line in the GoogleImageGenConfig class. - This minor adjustment improves code readability without affecting functionality. Signed-off-by: Ifta Khairul Alam Adil --- litellm/llms/gemini/image_generation/transformation.py | 1 - 1 file changed, 1 deletion(-) diff --git a/litellm/llms/gemini/image_generation/transformation.py b/litellm/llms/gemini/image_generation/transformation.py index a78a5b9dda7..e57364fd288 100644 --- a/litellm/llms/gemini/image_generation/transformation.py +++ b/litellm/llms/gemini/image_generation/transformation.py @@ -88,7 +88,6 @@ class GoogleImageGenConfig(BaseImageGenerationConfig): Google AI API format: https://generativelanguage.googleapis.com/v1beta/models/{model}:predict """ - complete_url: str = ( api_base or get_secret_str("GEMINI_API_BASE") From 76f78f7b2a0d4f1bc8df0972dbc1c95471e3baf7 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 08:25:24 -0700 Subject: [PATCH 55/66] fix: LITELLM_LOG_FILE test --- docs/my-website/docs/proxy/config_settings.md | 1 + ...odel_prices_and_context_window_backup.json | 1763 +++++++++-------- tests/test_litellm/test_utils.py | 1 + 3 files changed, 912 insertions(+), 853 deletions(-) diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index 73b16af2f95..1fb4cd55aa2 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -570,6 +570,7 @@ router_settings: | LITELLM_LICENSE | License key for LiteLLM usage | LITELLM_LOCAL_MODEL_COST_MAP | Local configuration for model cost mapping in LiteLLM | LITELLM_LOG | Enable detailed logging for LiteLLM +| LITELLM_LOG_FILE | File path to write LiteLLM logs to. When set, logs will be written to both console and the specified file | LITELLM_MASTER_KEY | Master key for proxy authentication | LITELLM_MODE | Operating mode for LiteLLM (e.g., production, development) | LITELLM_RATE_LIMIT_WINDOW_SIZE | Rate limit window size for LiteLLM. Default is 60 diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 0fd32d43aae..3e5dbeb2ae9 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11923,6 +11923,63 @@ "mode": "chat", "supports_tool_choice": true }, + "openrouter/openai/gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_tool_choice": true, + "supports_reasoning": true + }, + "openrouter/openai/gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_tool_choice": true, + "supports_reasoning": true + }, + "openrouter/openai/gpt-5-chat": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openrouter", + "mode": "chat", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_tool_choice": false, + "supports_reasoning": true + }, "openrouter/openai/gpt-oss-20b": { "max_tokens": 32768, "max_input_tokens": 131072, @@ -15207,236 +15264,16 @@ "litellm_provider": "ollama", "mode": "completion" }, - "deepinfra/deepseek-ai/DeepSeek-V3": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 3.8e-07, - "output_cost_per_token": 8.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Phind/Phind-CodeLlama-34B-v2": { + "deepinfra/Austism/chronos-hermes-13b-v2": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 6e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 1.5e-08, - "output_cost_per_token": 2e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemma-2-9b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2-7B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/QVQ-72B-Preview": { - "max_tokens": 32000, - "max_input_tokens": 32000, - "max_output_tokens": 32000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-3.3-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2.3e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/Phi-4-multimodal-instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mistralai/Devstral-Small-2507": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 2.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/WizardLM-2-7B": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-V3-0324": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 2.8e-07, - "output_cost_per_token": 8.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 8e-08, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/anthropic/claude-3-7-sonnet-latest": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 3.3e-06, - "output_cost_per_token": 1.65e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-235B-A22B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, + "output_cost_per_token": 1.3e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/WizardLM-2-8x22B": { - "max_tokens": 65536, - "max_input_tokens": 65536, - "max_output_tokens": 65536, - "input_cost_per_token": 4.8e-07, - "output_cost_per_token": 4.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Llama-Guard-4-12B": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1.8e-07, - "output_cost_per_token": 1.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, "deepinfra/Gryphe/MythoMax-L2-13b": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -15447,72 +15284,32 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Llama-3.2-1B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-09, - "output_cost_per_token": 1e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemma-2-27b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 2.7e-07, + "deepinfra/Gryphe/MythoMax-L2-13b-turbo": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 1.3e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 2e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": { - "max_tokens": 65536, - "max_input_tokens": 65536, - "max_output_tokens": 65536, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 6.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2.5-7B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 4e-08, + "deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1e-07, "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", - "supports_tool_choice": false + "supports_tool_choice": true }, - "deepinfra/google/gemini-1.5-flash-8b": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 3.75e-08, - "output_cost_per_token": 1.5e-07, + "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15527,6 +15324,376 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/NovaSky-AI/Sky-T1-32B-Preview": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Phind/Phind-CodeLlama-34B-v2": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 6e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/QVQ-72B-Preview": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "max_output_tokens": 32000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/QwQ-32B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/QwQ-32B-Preview": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2-72B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2-7B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2.5-72B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 3.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen2.5-7B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 4e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-Coder-7B": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.5e-08, + "output_cost_per_token": 5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-14B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 1.3e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-30B-A3B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 8e-08, + "output_cost_per_token": 2.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-32B": { + "max_tokens": 40960, + "max_input_tokens": 40960, + "max_output_tokens": 40960, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 4e-07, + "output_cost_per_token": 1.6e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": { + "max_tokens": 262144, + "max_input_tokens": 262144, + "max_output_tokens": 262144, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/Sao10K/L3-70B-Euryale-v2.1": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3-8B-Lunaris-v1": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/allenai/olmOCR-7B-0725-FP8": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 1.5e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/anthropic/claude-3-7-sonnet-latest": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 3.3e-06, + "output_cost_per_token": 1.65e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/anthropic/claude-4-opus": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 1.65e-05, + "output_cost_per_token": 8.25e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/anthropic/claude-4-sonnet": { + "max_tokens": 200000, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "input_cost_per_token": 3.3e-06, + "output_cost_per_token": 1.65e-05, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/bigcode/starcoder2-15b-instruct-v0.1": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.4e-07, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/deepinfra/airoboros-70b": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 7e-07, + "output_cost_per_token": 9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.18e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/deepseek-ai/DeepSeek-R1": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 4.5e-07, + "output_cost_per_token": 2.15e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-0528": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2.15e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { "max_tokens": 131072, "max_input_tokens": 131072, @@ -15537,22 +15704,182 @@ "mode": "chat", "supports_tool_choice": false }, - "deepinfra/meta-llama/Llama-Guard-3-8B": { + "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-R1-Turbo": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 3.8e-07, + "output_cost_per_token": 8.9e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3-0324": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 2.8e-07, + "output_cost_per_token": 8.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-06, + "output_cost_per_token": 3e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/deepseek-ai/DeepSeek-V3.1": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1e-06, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, + "deepinfra/google/codegemma-7b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 7e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemini-1.5-flash": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 7.5e-08, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-1.5-flash-8b": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 3.75e-08, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.0-flash-001": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.5-flash": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 2.1e-07, + "output_cost_per_token": 1.75e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemini-2.5-pro": { + "max_tokens": 1000000, + "max_input_tokens": 1000000, + "max_output_tokens": 1000000, + "input_cost_per_token": 8.75e-07, + "output_cost_per_token": 7e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-1.1-7b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 7e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-2-27b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 2.7e-07, + "output_cost_per_token": 2.7e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemma-2-9b-it": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/google/gemma-3-12b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 5e-08, - "output_cost_per_token": 8e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-3-27b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 1.7e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/google/gemma-3-4b-it": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 4e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15567,6 +15894,16 @@ "mode": "chat", "supports_tool_choice": false }, + "deepinfra/mattshumer/Reflection-Llama-3.1-70B": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, "deepinfra/meta-llama/Llama-2-13b-chat-hf": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -15577,182 +15914,32 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/anthropic/claude-4-opus": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 1.65e-05, - "output_cost_per_token": 8.25e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openchat/openchat-3.6-8b": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-3-27b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 1.7e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Austism/chronos-hermes-13b-v2": { + "deepinfra/meta-llama/Llama-2-70b-chat-hf": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 1.3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 7.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/QwQ-32B-Preview": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 1.8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/anthropic/claude-4-sonnet": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 200000, - "input_cost_per_token": 3.3e-06, - "output_cost_per_token": 1.65e-05, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/Phi-3-medium-4k-instruct": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 1.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/mattshumer/Reflection-Llama-3.1-70B": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/openchat/openchat_3.5": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5.5e-08, - "output_cost_per_token": 5.5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 6.5e-07, - "output_cost_per_token": 7.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2.3e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-V3.1": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 3.2e-07, - "output_cost_per_token": 1.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen2.5-Coder-7B": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.5e-08, - "output_cost_per_token": 5e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.4e-07, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 8e-07, + "input_cost_per_token": 6.4e-07, "output_cost_per_token": 8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2.18e-06, + "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 4.9e-08, + "output_cost_per_token": 4.9e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/zai-org/GLM-4.5": { + "deepinfra/meta-llama/Llama-3.2-1B-Instruct": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 5.5e-07, - "output_cost_per_token": 2e-06, + "input_cost_per_token": 5e-09, + "output_cost_per_token": 1e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -15767,276 +15954,36 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-07, + "deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 3.5e-07, + "output_cost_per_token": 4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Llama-3.3-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2.3e-07, "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemini-1.5-flash": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemini-2.5-pro": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 8.75e-07, - "output_cost_per_token": 7e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen3-30B-A3B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 8e-08, - "output_cost_per_token": 2.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/QwQ-32B": { + "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 1.5e-07, + "input_cost_per_token": 3.8e-08, + "output_cost_per_token": 1.2e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/moonshotai/Kimi-K2-Instruct": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Sao10K/L3-70B-Euryale-v2.1": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/microsoft/phi-4-reasoning-plus": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 3.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-3-12b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 1e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/gemini-2.5-flash": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 2.1e-07, - "output_cost_per_token": 1.75e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 4.5e-07, - "output_cost_per_token": 2.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-7B-Instruct-v0.3": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 2.8e-08, - "output_cost_per_token": 5.4e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen2.5-72B-Instruct": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 3.9e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/Qwen/Qwen3-14B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 2.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/allenai/olmOCR-7B-0725-FP8": { - "max_tokens": 16384, - "max_input_tokens": 16384, - "max_output_tokens": 16384, - "input_cost_per_token": 2.7e-07, - "output_cost_per_token": 1.5e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 4e-07, - "output_cost_per_token": 1.6e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/microsoft/phi-4": { - "max_tokens": 16384, - "max_input_tokens": 16384, - "max_output_tokens": 16384, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 1.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/zai-org/GLM-4.5-Air": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-07, - "output_cost_per_token": 1.1e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 7.5e-08, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openai/gpt-oss-120b": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 9e-08, - "output_cost_per_token": 4.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/google/codegemma-7b-it": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 7e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 3e-07, - "output_cost_per_token": 1.2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-Nemo-Instruct-2407": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 4e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/openbmb/MiniCPM-Llama3-V-2_5": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3.4e-07, - "output_cost_per_token": 3.4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/bigcode/starcoder2-15b-instruct-v0.1": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.5e-07, - "output_cost_per_token": 1.5e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "max_tokens": 1048576, "max_input_tokens": 1048576, @@ -16047,6 +15994,16 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, "deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": { "max_tokens": 327680, "max_input_tokens": 327680, @@ -16057,32 +16014,62 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemini-2.0-flash-001": { - "max_tokens": 1000000, - "max_input_tokens": 1000000, - "max_output_tokens": 1000000, - "input_cost_per_token": 1e-07, + "deepinfra/meta-llama/Llama-Guard-3-8B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Llama-Guard-4-12B": { + "max_tokens": 163840, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3e-07, "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Gryphe/MythoMax-L2-13b-turbo": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 1.3e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/google/gemma-1.1-7b-it": { + "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 7e-08, - "output_cost_per_token": 7e-08, + "input_cost_per_token": 3e-08, + "output_cost_per_token": 6e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 8e-07, + "output_cost_per_token": 8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2.3e-07, + "output_cost_per_token": 4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -16107,97 +16094,37 @@ "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen3-32B": { - "max_tokens": 40960, - "max_input_tokens": 40960, - "max_output_tokens": 40960, - "input_cost_per_token": 1e-07, - "output_cost_per_token": 3e-07, + "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 1.5e-08, + "output_cost_per_token": 2e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Llama-2-70b-chat-hf": { + "deepinfra/microsoft/Phi-3-medium-4k-instruct": { "max_tokens": 4096, "max_input_tokens": 4096, "max_output_tokens": 4096, - "input_cost_per_token": 6.4e-07, - "output_cost_per_token": 8e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/nvidia/Nemotron-4-340B-Instruct": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 4.2e-06, - "output_cost_per_token": 4.2e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-0528": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 2.15e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/deepseek-ai/DeepSeek-R1-Turbo": { - "max_tokens": 163840, - "max_input_tokens": 163840, - "max_output_tokens": 163840, - "input_cost_per_token": 1e-06, - "output_cost_per_token": 3e-06, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/NovaSky-AI/Sky-T1-32B-Preview": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 1.2e-07, - "output_cost_per_token": 1.8e-07, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 1.4e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507": { - "max_tokens": 262144, - "max_input_tokens": 262144, - "max_output_tokens": 262144, - "input_cost_per_token": 1.3e-07, - "output_cost_per_token": 6e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, + "deepinfra/microsoft/Phi-4-multimodal-instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, "input_cost_per_token": 5e-08, "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/mistralai/Mistral-7B-Instruct-v0.1": { + "deepinfra/microsoft/WizardLM-2-7B": { "max_tokens": 32768, "max_input_tokens": 32768, "max_output_tokens": 32768, @@ -16205,64 +16132,64 @@ "output_cost_per_token": 5.5e-08, "litellm_provider": "deepinfra", "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/microsoft/WizardLM-2-8x22B": { + "max_tokens": 65536, + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "input_cost_per_token": 4.8e-07, + "output_cost_per_token": 4.8e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/microsoft/phi-4": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 1.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", "supports_tool_choice": true }, - "deepinfra/Qwen/Qwen2-72B-Instruct": { + "deepinfra/microsoft/phi-4-reasoning-plus": { "max_tokens": 32768, "max_input_tokens": 32768, "max_output_tokens": 32768, - "input_cost_per_token": 3.5e-07, - "output_cost_per_token": 4e-07, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": true - }, - "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 5e-07, - "output_cost_per_token": 5e-07, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 3.5e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": false }, - "deepinfra/Sao10K/L3-8B-Lunaris-v1": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, - "litellm_provider": "deepinfra", - "mode": "chat", - "supports_tool_choice": false - }, - "deepinfra/deepinfra/airoboros-70b": { - "max_tokens": 4096, - "max_input_tokens": 4096, - "max_output_tokens": 4096, - "input_cost_per_token": 7e-07, - "output_cost_per_token": 9e-07, + "deepinfra/mistralai/Devstral-Small-2505": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 1.2e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/google/gemma-3-4b-it": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 2e-08, - "output_cost_per_token": 4e-08, + "deepinfra/mistralai/Devstral-Small-2507": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 2.8e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": { - "max_tokens": 8192, - "max_input_tokens": 8192, - "max_output_tokens": 8192, - "input_cost_per_token": 3e-08, - "output_cost_per_token": 6e-08, + "deepinfra/mistralai/Mistral-7B-Instruct-v0.1": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true @@ -16277,35 +16204,115 @@ "mode": "chat", "supports_tool_choice": false }, - "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 3.8e-08, - "output_cost_per_token": 1.2e-07, + "deepinfra/mistralai/Mistral-7B-Instruct-v0.3": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.8e-08, + "output_cost_per_token": 5.4e-08, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/mistralai/Devstral-Small-2505": { + "deepinfra/mistralai/Mistral-Nemo-Instruct-2407": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-08, + "output_cost_per_token": 4e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": { "max_tokens": 128000, "max_input_tokens": 128000, "max_output_tokens": 128000, - "input_cost_per_token": 6e-08, - "output_cost_per_token": 1.2e-07, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 1e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 1e-07, "litellm_provider": "deepinfra", "mode": "chat", "supports_tool_choice": true }, - "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct": { + "deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": { + "max_tokens": 65536, + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "input_cost_per_token": 6.5e-07, + "output_cost_per_token": 6.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 8e-08, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/moonshotai/Kimi-K2-Instruct": { "max_tokens": 131072, "max_input_tokens": 131072, "max_output_tokens": 131072, - "input_cost_per_token": 4.9e-08, - "output_cost_per_token": 4.9e-08, + "input_cost_per_token": 5e-07, + "output_cost_per_token": 2e-06, "litellm_provider": "deepinfra", "mode": "chat", - "supports_tool_choice": false + "supports_tool_choice": true + }, + "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 1.2e-07, + "output_cost_per_token": 3e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/nvidia/Nemotron-4-340B-Instruct": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 4.2e-06, + "output_cost_per_token": 4.2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/openai/gpt-oss-120b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 9e-08, + "output_cost_per_token": 4.5e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true }, "deepinfra/openai/gpt-oss-20b": { "max_tokens": 131072, @@ -16317,6 +16324,56 @@ "mode": "chat", "supports_tool_choice": true }, + "deepinfra/openbmb/MiniCPM-Llama3-V-2_5": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 3.4e-07, + "output_cost_per_token": 3.4e-07, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/openchat/openchat-3.6-8b": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": false + }, + "deepinfra/openchat/openchat_3.5": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 5.5e-08, + "output_cost_per_token": 5.5e-08, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/zai-org/GLM-4.5": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.5e-07, + "output_cost_per_token": 2e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, + "deepinfra/zai-org/GLM-4.5-Air": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 1.1e-06, + "litellm_provider": "deepinfra", + "mode": "chat", + "supports_tool_choice": true + }, "perplexity/codellama-34b-instruct": { "max_tokens": 16384, "max_input_tokens": 16384, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index dba40e2214a..fa7e00f4a25 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -847,6 +847,7 @@ async def test_supports_tool_choice(): or "o3" in model_name or "mistral" in model_name or "oci" in model_name + or "openrouter" in model_name ): continue From d5440f9614a8cfab76f9a2639d8f109d39b36f3b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 08:27:57 -0700 Subject: [PATCH 56/66] test_image_edit_array_handling --- tests/image_gen_tests/test_image_edits.py | 9 +-------- 1 file changed, 1 insertion(+), 8 deletions(-) diff --git a/tests/image_gen_tests/test_image_edits.py b/tests/image_gen_tests/test_image_edits.py index 745993499d2..33dfbc4d52e 100644 --- a/tests/image_gen_tests/test_image_edits.py +++ b/tests/image_gen_tests/test_image_edits.py @@ -651,14 +651,7 @@ async def test_image_edit_array_handling(): image=TEST_IMAGES, ) - # Test 3: Empty list (should fail validation) - with pytest.raises(Exception): - await aimage_edit( - prompt=prompt, - model="gpt-image-1", - image=[], - ) - + # Both valid calls should succeed ImageResponse.model_validate(result1) ImageResponse.model_validate(result2) From ed52b67fcf120951705629b6c05c7d718d0ccd76 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 09:51:22 -0700 Subject: [PATCH 57/66] test_image_edit_array_handling --- tests/image_gen_tests/test_image_edits.py | 114 ---------------------- 1 file changed, 114 deletions(-) diff --git a/tests/image_gen_tests/test_image_edits.py b/tests/image_gen_tests/test_image_edits.py index 33dfbc4d52e..74562c4648f 100644 --- a/tests/image_gen_tests/test_image_edits.py +++ b/tests/image_gen_tests/test_image_edits.py @@ -660,117 +660,3 @@ async def test_image_edit_array_handling(): assert mock_post.call_count == 2 -@pytest.mark.asyncio -async def test_openai_transformation_handles_multiple_images(): - """Test that OpenAI transformation correctly handles multiple images in request""" - from litellm.llms.openai.image_edit.transformation import OpenAIImageEditConfig - from litellm.types.router import GenericLiteLLMParams - - config = OpenAIImageEditConfig() - - # Test with multiple images - prompt = "Edit these images" - images = [b"fake_image_1", b"fake_image_2", b"fake_image_3"] - litellm_params = GenericLiteLLMParams(api_key="test_key") - - data, files = config.transform_image_edit_request( - model="gpt-image-1", - prompt=prompt, - image=images, - image_edit_optional_request_params={"n": 1}, - litellm_params=litellm_params, - headers={} - ) - - # Check that data contains the prompt and parameters - assert data["prompt"] == prompt - assert data["model"] == "gpt-image-1" - assert data["n"] == 1 - - # Check that files contains all images with correct field names - assert len(files) == len(images) - for i, file_entry in enumerate(files): - assert file_entry[0] == "image[]" # OpenAI uses image[] for multiple files - assert file_entry[1][1] == images[i] # Image data - assert file_entry[1][2] == "image/png" # Content type - - print(f"Successfully processed {len(images)} images in transformation") - - -@pytest.mark.asyncio -async def test_multiple_image_edit_parameter_validation(): - """Test parameter validation with multiple images""" - from litellm import aimage_edit - - # Mock response - mock_response = { - "created": 1589478378, - "data": [ - { - "b64_json": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==" - } - ] - } - - class MockResponse: - def __init__(self, json_data, status_code): - self._json_data = json_data - self.status_code = status_code - self.text = json.dumps(json_data) - - def json(self): - return self._json_data - - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - new_callable=AsyncMock, - ) as mock_post: - mock_post.return_value = MockResponse(mock_response, 200) - - # Test with valid parameters - result = await aimage_edit( - prompt="Test prompt", - model="gpt-image-1", - image=TEST_IMAGES, - n=1, - size="1024x1024", - response_format="b64_json" - ) - - ImageResponse.model_validate(result) - - # Verify the request was made with correct parameters - mock_post.assert_called_once() - call_args = mock_post.call_args - - # Check that the request contains the expected data - if 'data' in call_args.kwargs: - form_data = call_args.kwargs['data'] - assert 'model' in form_data - assert 'prompt' in form_data - assert 'n' in form_data - assert form_data['n'] == 1 # Could be int or string depending on implementation print("Parameter validation passed for multiple image edit") - - -@pytest.mark.asyncio -async def test_multiple_image_edit_error_handling(): - """Test error handling with multiple images""" - from litellm import aimage_edit - - # Test with None image (should raise error) - with pytest.raises(Exception): - await aimage_edit( - prompt="Test prompt", - model="gpt-image-1", - image=None, - ) - - # Test with invalid model (should raise error) - with pytest.raises(Exception): - await aimage_edit( - prompt="Test prompt", - model="invalid-model", - image=TEST_IMAGES, - ) - - print("Error handling tests passed for multiple image edit") From d9f8eb27c93c9ec6e08f504f1e88077e30511245 Mon Sep 17 00:00:00 2001 From: Ifta Khairul Alam Adil Date: Wed, 27 Aug 2025 19:46:53 +0200 Subject: [PATCH 58/66] fix: enable tool choice support for model prices and context window - Updated the "supports_tool_choice" field to true in the model_prices_and_context_window.json file, allowing for tool choice functionality in the specified model. Signed-off-by: Ifta Khairul Alam Adil --- model_prices_and_context_window.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b7297f07199..b33e3d9d9eb 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11941,7 +11941,7 @@ "supported_output_modalities": [ "text" ], - "supports_tool_choice": false, + "supports_tool_choice": true, "supports_reasoning": true }, "openrouter/openai/gpt-oss-20b": { From f35ce02475a0b341c21531ded111b4bb1993c1c7 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 10:51:24 -0700 Subject: [PATCH 59/66] [Bug Fix] LLM Translation - Allow using dynamic `api_key` for image generation requests (#14007) * fix - allow using dyanmic api key for img gen * test_aiml_image_generation_with_dynamic_api_key --- litellm/images/main.py | 3 +- litellm/llms/custom_httpx/llm_http_handler.py | 7 +- .../image_gen_tests/test_image_generation.py | 75 +++++++++++++++++++ 3 files changed, 82 insertions(+), 3 deletions(-) diff --git a/litellm/images/main.py b/litellm/images/main.py index 4e4dfa752f3..4993a48c724 100644 --- a/litellm/images/main.py +++ b/litellm/images/main.py @@ -1,7 +1,7 @@ import asyncio import contextvars from functools import partial -from typing import Any, Coroutine, Dict, Literal, Optional, Union, cast, overload, List +from typing import Any, Coroutine, Dict, List, Literal, Optional, Union, cast, overload import httpx @@ -347,6 +347,7 @@ def image_generation( # noqa: PLR0915 raise ValueError(f"image generation config is not supported for {custom_llm_provider}") return llm_http_handler.image_generation_handler( + api_key=api_key, model=model, prompt=prompt, image_generation_provider_config=image_generation_config, diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index d404077a5b6..2faea53901c 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -2681,6 +2681,7 @@ class BaseLLMHTTPHandler: _is_async: bool = False, fake_stream: bool = False, litellm_metadata: Optional[Dict[str, Any]] = None, + api_key: Optional[str] = None, ) -> Union[ ImageResponse, Coroutine[Any, Any, ImageResponse], @@ -2705,6 +2706,7 @@ class BaseLLMHTTPHandler: client=client if isinstance(client, AsyncHTTPHandler) else None, fake_stream=fake_stream, litellm_metadata=litellm_metadata, + api_key=api_key, ) if client is None or not isinstance(client, HTTPHandler): @@ -2715,7 +2717,7 @@ class BaseLLMHTTPHandler: sync_httpx_client = client headers = image_generation_provider_config.validate_environment( - api_key=litellm_params.get("api_key", None), + api_key=api_key, headers=image_generation_optional_request_params.get("extra_headers", {}) or {}, model=model, @@ -2798,6 +2800,7 @@ class BaseLLMHTTPHandler: client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, fake_stream: bool = False, litellm_metadata: Optional[Dict[str, Any]] = None, + api_key: Optional[str] = None, ) -> ImageResponse: """ Async version of the image generation handler. @@ -2812,7 +2815,7 @@ class BaseLLMHTTPHandler: async_httpx_client = client headers = image_generation_provider_config.validate_environment( - api_key=litellm_params.get("api_key", None), + api_key=api_key, headers=image_generation_optional_request_params.get("extra_headers", {}) or {}, model=model, diff --git a/tests/image_gen_tests/test_image_generation.py b/tests/image_gen_tests/test_image_generation.py index 043a0a88fd1..9aa81527164 100644 --- a/tests/image_gen_tests/test_image_generation.py +++ b/tests/image_gen_tests/test_image_generation.py @@ -330,3 +330,78 @@ async def test_gpt_image_1_with_input_fidelity(): assert captured_kwargs["quality"] == "medium" assert captured_kwargs["size"] == "1024x1024" + +@pytest.mark.asyncio +async def test_aiml_image_generation_with_dynamic_api_key(): + """ + Test that when api_key is passed as a dynamic parameter to aimage_generation, + it gets properly used for AIML provider authentication instead of falling back + to environment variables. + + This test validates the fix for ensuring dynamic API keys are respected + when making image generation requests to the AIML provider. + """ + from unittest.mock import AsyncMock, patch, MagicMock + import httpx + + # Mock AIML response + mock_aiml_response = { + "created": 1703658209, + "data": [ + { + "url": "https://example.com/generated_image.png" + } + ] + } + + # Track captured arguments + captured_headers = None + captured_url = None + captured_json_data = None + + def capture_post_call(*args, **kwargs): + nonlocal captured_headers, captured_url, captured_json_data + captured_url = kwargs.get('url') or (args[0] if args else None) + captured_headers = kwargs.get('headers', {}) + captured_json_data = kwargs.get('json', {}) + + # Create a mock response + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = mock_aiml_response + mock_response.text = json.dumps(mock_aiml_response) + return mock_response + + # Mock the HTTP client that actually makes the request (sync version for image generation) + with patch('litellm.llms.custom_httpx.http_handler.HTTPHandler.post') as mock_post: + mock_post.side_effect = capture_post_call + + # Test with dynamic api_key + test_api_key = "test-dynamic-api-key-12345" + + response = await litellm.aimage_generation( + prompt="A cute baby sea otter", + model="aiml/flux-pro/v1.1", + api_key=test_api_key, # This should be used instead of env vars + ) + + # Validate the response (mocked response processing might not populate data correctly) + assert response is not None + + # The most important validations: API key and endpoint usage + # These prove that the dynamic API key was properly used + assert captured_headers is not None + assert "Authorization" in captured_headers + assert captured_headers["Authorization"] == f"Bearer {test_api_key}" + print("TESTCAPTURED HEADERS", captured_headers) + # Validate the correct AIML endpoint was called + assert captured_url is not None + assert "api.aimlapi.com" in captured_url + assert "/v1/images/generations" in captured_url + + # Validate the request data + assert captured_json_data is not None + assert captured_json_data["prompt"] == "A cute baby sea otter" + assert captured_json_data["model"] == "flux-pro/v1.1" + + From 8404fdea5ba54f1ad55def608cd16c004266512c Mon Sep 17 00:00:00 2001 From: Kevin Turcios Date: Wed, 27 Aug 2025 13:35:48 -0500 Subject: [PATCH 60/66] Update in_memory_cache.py --- litellm/caching/in_memory_cache.py | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/litellm/caching/in_memory_cache.py b/litellm/caching/in_memory_cache.py index 47f911894a3..63869474d47 100644 --- a/litellm/caching/in_memory_cache.py +++ b/litellm/caching/in_memory_cache.py @@ -112,14 +112,15 @@ class InMemoryCache(BaseCache): - 3. the size of in-memory cache is bounded """ - for key in list(self.ttl_dict.keys()): - if self._is_key_expired(key): - self._remove_key(key) + current_time = time.time() + expired_keys = [key for key, ttl in self.ttl_dict.items() if current_time > ttl] + for key in expired_keys: + self._remove_key(key) - # de-reference the removed item - # https://www.geeksforgeeks.org/diagnosing-and-fixing-memory-leaks-in-python/ - # One of the most common causes of memory leaks in Python is the retention of objects that are no longer being used. - # This can occur when an object is referenced by another object, but the reference is never removed. + # de-reference the removed item + # https://www.geeksforgeeks.org/diagnosing-and-fixing-memory-leaks-in-python/ + # One of the most common causes of memory leaks in Python is the retention of objects that are no longer being used. + # This can occur when an object is referenced by another object, but the reference is never removed. def allow_ttl_override(self, key: str) -> bool: """ From 4fff05f1cce716876315ea0cf262f4e225abdcb9 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 12:12:47 -0700 Subject: [PATCH 61/66] [Feature]: Support Gemini requests with only system prompt (#14010) * _default_user_message_when_system_message_passed * test_system_prompt_only_adds_blank_user_message * test_system_message_with_no_user_message --- .../llms/vertex_ai/gemini/transformation.py | 15 ++++++++ tests/llm_translation/base_llm_unit_tests.py | 22 ++++++++++++ .../llms/vertex_ai/test_vertex.py | 34 +++++++++++++++++++ 3 files changed, 71 insertions(+) diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index 85e3f15364b..8ab212e2558 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -35,6 +35,7 @@ from litellm.types.llms.openai import ( ChatCompletionFileObject, ChatCompletionImageObject, ChatCompletionTextObject, + ChatCompletionUserMessage, ) from litellm.types.llms.vertex_ai import * from litellm.types.llms.vertex_ai import ( @@ -475,6 +476,13 @@ async def async_transform_request_body( optional_params=optional_params, ) +def _default_user_message_when_system_message_passed() -> ChatCompletionUserMessage: + """ + Returns a default user message when a "system" message is passed in gemini fails. + + This adds a blank user message to the messages list, to ensure that gemini doesn't fail the request. + """ + return ChatCompletionUserMessage(content=".", role="user") def _transform_system_message( supports_system_message: bool, messages: List[AllMessageValues] @@ -510,6 +518,13 @@ def _transform_system_message( messages.pop(idx) if len(system_content_blocks) > 0: + ######################################################### + # If no messages are passed in, add a blank user message + # Relevant Issue - https://github.com/BerriAI/litellm/issues/13769 + ######################################################### + if len(messages) == 0: + messages.append(_default_user_message_when_system_message_passed()) + ######################################################### return SystemInstructions(parts=system_content_blocks), messages return None, messages diff --git a/tests/llm_translation/base_llm_unit_tests.py b/tests/llm_translation/base_llm_unit_tests.py index 4c24d1fcd5b..9e2af871787 100644 --- a/tests/llm_translation/base_llm_unit_tests.py +++ b/tests/llm_translation/base_llm_unit_tests.py @@ -119,6 +119,28 @@ class BaseLLMChatTest(ABC): pytest.skip("Model is overloaded") assert response.choices[0].message.content is not None + + def test_system_message_with_no_user_message(self): + """ + Test that the system message is translated correctly for non-OpenAI providers. + """ + base_completion_call_args = self.get_base_completion_call_args() + messages = [ + { + "role": "system", + "content": "Be a good bot!", + }, + ] + try: + response = self.completion_function( + **base_completion_call_args, + messages=messages, + ) + assert response is not None + except litellm.InternalServerError: + pytest.skip("Model is overloaded") + + assert response.choices[0].message.content is not None def test_content_list_handling(self): """Check if content list is supported by LLM API""" diff --git a/tests/test_litellm/llms/vertex_ai/test_vertex.py b/tests/test_litellm/llms/vertex_ai/test_vertex.py index 07b0cbc6234..7e683d1f54e 100644 --- a/tests/test_litellm/llms/vertex_ai/test_vertex.py +++ b/tests/test_litellm/llms/vertex_ai/test_vertex.py @@ -1469,3 +1469,37 @@ def test_vertex_parallel_tool_calls_false_single_tool(): parallel_tool_calls=False, ) assert "tools" in optional_params + + +from litellm.llms.vertex_ai.gemini.transformation import _transform_request_body + + +def test_system_prompt_only_adds_blank_user_message(): + """ + Test that the system prompt only adds a blank user message when a system message is passed in. + + Relevant Issue - https://github.com/BerriAI/litellm/issues/13769 + """ + SYSTEM_INSTRUCTION = "System instructions for the model" + data = _transform_request_body( + messages=[{"role": "system", "content": SYSTEM_INSTRUCTION}], + model="gemini-2.5-flash", + optional_params={}, + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=None, + ) + print("Final data: ", data) + + # validate that a blank user message is added when a system message is passed in + assert len(data["contents"]) == 1 + first_content = data["contents"][0] + assert first_content["role"] == "user" + assert len(first_content["parts"]) == 1 + + + ######################################################### + # system message was passed in + ######################################################### + assert len(data["system_instruction"]) == 1 + assert data["system_instruction"]["parts"][0]["text"] == SYSTEM_INSTRUCTION From ee13c65701468c05653b7468ed504ae9011a8c82 Mon Sep 17 00:00:00 2001 From: Sam Xie Date: Wed, 27 Aug 2025 12:35:09 -0700 Subject: [PATCH 62/66] [Bug]: `/responses` endpoint proxy ignores extra_headers in GitHub Copilot (#13775) * fix: pass extra_headers parameter through responses API transformation chain Ensure extra_headers parameter is properly forwarded from the responses() function through the transformation handler and config to maintain header propagation in litellm_completion_request dict. * Add tests --- .../responses/litellm_completion_transformation/handler.py | 4 +++- .../litellm_completion_transformation/transformation.py | 2 ++ litellm/responses/main.py | 1 + .../test_litellm_completion_responses.py | 5 ++++- 4 files changed, 10 insertions(+), 2 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/handler.py b/litellm/responses/litellm_completion_transformation/handler.py index 7b3243309b2..9317bf26178 100644 --- a/litellm/responses/litellm_completion_transformation/handler.py +++ b/litellm/responses/litellm_completion_transformation/handler.py @@ -2,7 +2,7 @@ Handler for transforming responses api requests to litellm.completion requests """ -from typing import Any, Coroutine, Optional, Union +from typing import Any, Coroutine, Dict, Optional, Union import litellm from litellm.responses.litellm_completion_transformation.streaming_iterator import ( @@ -30,6 +30,7 @@ class LiteLLMCompletionTransformationHandler: custom_llm_provider: Optional[str] = None, _is_async: bool = False, stream: Optional[bool] = None, + extra_headers: Optional[Dict[str, Any]] = None, **kwargs, ) -> Union[ ResponsesAPIResponse, @@ -45,6 +46,7 @@ class LiteLLMCompletionTransformationHandler: responses_api_request=responses_api_request, custom_llm_provider=custom_llm_provider, stream=stream, + extra_headers=extra_headers, **kwargs, ) ) diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index ba453edbf66..82d3980b370 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -99,6 +99,7 @@ class LiteLLMCompletionResponsesConfig: responses_api_request: ResponsesAPIOptionalRequestParams, custom_llm_provider: Optional[str] = None, stream: Optional[bool] = None, + extra_headers: Optional[Dict[str, Any]] = None, **kwargs, ) -> dict: """ @@ -126,6 +127,7 @@ class LiteLLMCompletionResponsesConfig: "web_search_options": web_search_options, # litellm specific params "custom_llm_provider": custom_llm_provider, + "extra_headers": extra_headers, } # Responses API `Completed` events require usage, we pass `stream_options` to litellm.completion to include usage diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 6757e1ad31f..9584baf7368 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -455,6 +455,7 @@ def responses( custom_llm_provider=custom_llm_provider, _is_async=_is_async, stream=stream, + extra_headers=extra_headers, **kwargs, ) diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 8e38011279d..00fa0851a7b 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -541,7 +541,8 @@ class TestFunctionCallTransformation: result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request( model="gemini/gemini-2.0-flash", input=test_input, - responses_api_request=responses_api_request + responses_api_request=responses_api_request, + extra_headers={"X-Test-Header": "test-value"} ) assert "messages" in result @@ -563,6 +564,8 @@ class TestFunctionCallTransformation: tool_msg = messages[2] assert tool_msg["role"] == "tool" + assert result["extra_headers"] == {"X-Test-Header": "test-value"} + def test_function_call_without_call_id_fallback_to_id(self): """Test that function_call items can use 'id' field when 'call_id' is missing""" function_call_item = { From 165242e31f3282794790253bd3e865b6b2ce0291 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 14:43:47 -0700 Subject: [PATCH 63/66] [Feat] `langfuse_otel` logger - allow using LANGFUSE_OTEL_HOST for configuring host (#14013) * feat - add _get_langfuse_otel_host * test_get_langfuse_otel_config_with_otel_host_priority * docs: LANGFUSE_OTEL_HOST --- .../langfuse_otel_integration.md | 23 +++++++++++-------- .../integrations/langfuse/langfuse_otel.py | 13 ++++++++++- .../integrations/test_langfuse_otel.py | 13 +++++++++++ 3 files changed, 38 insertions(+), 11 deletions(-) diff --git a/docs/my-website/docs/observability/langfuse_otel_integration.md b/docs/my-website/docs/observability/langfuse_otel_integration.md index 4801fa8e1b0..b4c9a2bd1ad 100644 --- a/docs/my-website/docs/observability/langfuse_otel_integration.md +++ b/docs/my-website/docs/observability/langfuse_otel_integration.md @@ -35,14 +35,14 @@ The Langfuse OpenTelemetry integration allows you to send LiteLLM traces and obs |----------|----------|-------------|---------| | `LANGFUSE_PUBLIC_KEY` | Yes | Your Langfuse public key | `pk-lf-...` | | `LANGFUSE_SECRET_KEY` | Yes | Your Langfuse secret key | `sk-lf-...` | -| `LANGFUSE_HOST` | No | Langfuse host URL | `https://us.cloud.langfuse.com` (default) | +| `LANGFUSE_OTEL_HOST` | No | OTEL endpoint host | `https://otel.my-langfuse.com` | ### Endpoint Resolution -The integration automatically constructs the OTEL endpoint from the `LANGFUSE_HOST`: +The integration automatically constructs the OTEL endpoint from `LANGFUSE_OTEL_HOST` - **Default (US)**: `https://us.cloud.langfuse.com/api/public/otel` - **EU Region**: `https://cloud.langfuse.com/api/public/otel` -- **Self-hosted**: `{LANGFUSE_HOST}/api/public/otel` +- **Self-hosted**: `{LANGFUSE_OTEL_HOST}/api/public/otel` ## Usage @@ -77,11 +77,11 @@ os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..." os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..." # Use EU region -os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region -# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region (default) +os.environ["LANGFUSE_OTEL_HOST"] = "https://cloud.langfuse.com" # EU region +# os.environ["LANGFUSE_OTEL_HOST"] = "https://otel.my-langfuse.company.com" # custom OTEL endpoint # Or use self-hosted instance -# os.environ["LANGFUSE_HOST"] = "https://my-langfuse.company.com" +# os.environ["LANGFUSE_OTEL_HOST"] = "https://my-langfuse.company.com" litellm.callbacks = ["langfuse_otel"] ``` @@ -98,14 +98,16 @@ import litellm # Get keys for your project from the project settings page: https://cloud.langfuse.com os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..." os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..." -os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region -# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region +os.environ["LANGFUSE_OTEL_HOST"] = "https://cloud.langfuse.com" # EU region +# os.environ["LANGFUSE_OTEL_HOST"] = "https://us.cloud.langfuse.com" # US region +# os.environ["LANGFUSE_OTEL_HOST"] = "https://otel.my-langfuse.company.com" # custom OTEL endpoint LANGFUSE_AUTH = base64.b64encode( f"{os.environ.get('LANGFUSE_PUBLIC_KEY')}:{os.environ.get('LANGFUSE_SECRET_KEY')}".encode() ).decode() -os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = os.environ.get("LANGFUSE_HOST") + "/api/public/otel" +host = os.environ.get("LANGFUSE_OTEL_HOST") +os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = host + "/api/public/otel" os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = f"Authorization=Basic {LANGFUSE_AUTH}" litellm.callbacks = ["langfuse_otel"] @@ -120,7 +122,8 @@ Add the integration to your proxy configuration: ```bash export LANGFUSE_PUBLIC_KEY="pk-lf-..." export LANGFUSE_SECRET_KEY="sk-lf-..." -export LANGFUSE_HOST="https://us.cloud.langfuse.com" # Default US region +export LANGFUSE_OTEL_HOST="https://us.cloud.langfuse.com" # Default US region +# export LANGFUSE_OTEL_HOST="https://otel.my-langfuse.company.com" # custom OTEL endpoint ``` 2. Setup config.yaml diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index 8b90e123371..fbe480be95f 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -141,6 +141,17 @@ class LangfuseOtelLogger(OpenTelemetry): value = str(value) safe_set_attribute(span, enum_attr.value, value) + @staticmethod + def _get_langfuse_otel_host() -> Optional[str]: + """ + Returns the Langfuse OTEL host based on environment variables. + + Returned in the following order of precedence: + 1. LANGFUSE_OTEL_HOST + 2. LANGFUSE_HOST + """ + return os.environ.get("LANGFUSE_OTEL_HOST") or os.environ.get("LANGFUSE_HOST") + @staticmethod def get_langfuse_otel_config() -> LangfuseOtelConfig: """ @@ -166,7 +177,7 @@ class LangfuseOtelLogger(OpenTelemetry): ) # Determine endpoint - default to US cloud - langfuse_host = os.environ.get("LANGFUSE_HOST", None) + langfuse_host = LangfuseOtelLogger._get_langfuse_otel_host() if langfuse_host: # If LANGFUSE_HOST is provided, construct OTEL endpoint from it diff --git a/tests/test_litellm/integrations/test_langfuse_otel.py b/tests/test_litellm/integrations/test_langfuse_otel.py index 41b20c2a236..ceae019d344 100644 --- a/tests/test_litellm/integrations/test_langfuse_otel.py +++ b/tests/test_litellm/integrations/test_langfuse_otel.py @@ -229,6 +229,19 @@ class TestLangfuseOtelIntegration: # Should return an empty dict assert result == {} + + def test_get_langfuse_otel_config_with_otel_host_priority(self): + """LANGFUSE_OTEL_HOST should take priority over LANGFUSE_HOST.""" + with patch.dict(os.environ, { + 'LANGFUSE_PUBLIC_KEY': 'test_public_key', + 'LANGFUSE_SECRET_KEY': 'test_secret_key', + 'LANGFUSE_HOST': 'https://should-not-be-used.com', + 'LANGFUSE_OTEL_HOST': 'https://otel-host.com' + }, clear=False): + _ = LangfuseOtelLogger.get_langfuse_otel_config() + + assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://otel-host.com/api/public/otel" + if __name__ == "__main__": From 9acf80b8adc79361bcd3c436c5725d4452c8c530 Mon Sep 17 00:00:00 2001 From: Xingyao Wang Date: Wed, 27 Aug 2025 17:51:09 -0400 Subject: [PATCH 64/66] Fix issue #13995: Handle None metadata in batch requests (#13996) * Fix issue #13995: Handle None metadata in batch requests - Added null check in add_key_level_controls method to prevent NoneType error - Updated type hint to Optional[dict] for better type safety - Added comprehensive test suite to verify the fix works correctly - All existing tests pass, confirming no regression Fixes #13995 Co-authored-by: openhands * Move test file to tests/test_litellm/proxy/ directory - Moved test_batch_metadata_none_fix.py from tests/ to tests/test_litellm/proxy/ - Updated import structure to match existing test patterns - This ensures the test runs in GitHub Actions as requested by @krrishdholakia Co-authored-by: openhands --------- Co-authored-by: openhands --- litellm/proxy/litellm_pre_call_utils.py | 4 +- .../proxy/test_batch_metadata_none_fix.py | 147 ++++++++++++++++++ 2 files changed, 150 insertions(+), 1 deletion(-) create mode 100644 tests/test_litellm/proxy/test_batch_metadata_none_fix.py diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index adec337351c..d2bd7c8db29 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -489,8 +489,10 @@ class LiteLLMProxyRequestSetup: @staticmethod def add_key_level_controls( - key_metadata: dict, data: dict, _metadata_variable_name: str + key_metadata: Optional[dict], data: dict, _metadata_variable_name: str ): + if key_metadata is None: + return data if "cache" in key_metadata: data["cache"] = {} if isinstance(key_metadata["cache"], dict): diff --git a/tests/test_litellm/proxy/test_batch_metadata_none_fix.py b/tests/test_litellm/proxy/test_batch_metadata_none_fix.py new file mode 100644 index 00000000000..26744935037 --- /dev/null +++ b/tests/test_litellm/proxy/test_batch_metadata_none_fix.py @@ -0,0 +1,147 @@ +""" +Test for issue #13995: /batches request throws Internal Server Error when metadata=None + +This test verifies that the fix for handling None metadata in batch requests works correctly. +""" +import asyncio +import os +import sys +from unittest.mock import patch, MagicMock, AsyncMock + +import pytest +from openai import OpenAI + +import litellm +from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup +from litellm.proxy._types import UserAPIKeyAuth + +sys.path.insert( + 0, os.path.abspath("../../..") +) # Adds the parent directory to the system path + + +def test_add_key_level_controls_with_none_metadata(): + """ + Test that add_key_level_controls handles None metadata gracefully. + This is the core fix for issue #13995. + """ + # Test data + data = {"metadata": {}} + metadata_variable_name = "metadata" + + # Test with None key_metadata (this was causing the original error) + result = LiteLLMProxyRequestSetup.add_key_level_controls( + key_metadata=None, + data=data, + _metadata_variable_name=metadata_variable_name + ) + + # Should return the data unchanged without throwing an error + assert result == data + + # Test with empty dict key_metadata (should also work) + result = LiteLLMProxyRequestSetup.add_key_level_controls( + key_metadata={}, + data=data, + _metadata_variable_name=metadata_variable_name + ) + + # Should return the data unchanged + assert result == data + + # Test with valid key_metadata containing cache settings + key_metadata_with_cache = { + "cache": { + "ttl": 300, + "s-maxage": 600 + } + } + + result = LiteLLMProxyRequestSetup.add_key_level_controls( + key_metadata=key_metadata_with_cache, + data=data.copy(), + _metadata_variable_name=metadata_variable_name + ) + + # Should add cache settings to data + assert "cache" in result + assert result["cache"]["ttl"] == 300 + assert result["cache"]["s-maxage"] == 600 + + +def test_add_key_level_controls_simulates_original_issue(): + """ + Test that simulates the original issue scenario more directly. + This tests the exact code path that was failing in issue #13995. + """ + # This simulates the scenario where user_api_key_dict.metadata is None + # which was causing the original "'NoneType' object has no attribute 'get'" error + + data = {"metadata": {}} + metadata_variable_name = "metadata" + + # This is the exact call that was failing before the fix + # user_api_key_dict.metadata was None, causing the error in add_key_level_controls + try: + result = LiteLLMProxyRequestSetup.add_key_level_controls( + key_metadata=None, # This was the root cause of the issue + data=data, + _metadata_variable_name=metadata_variable_name + ) + + # If we get here, the fix is working + assert result == data + print("✓ Original issue scenario handled correctly - no NoneType error") + + except AttributeError as e: + if "'NoneType' object has no attribute 'get'" in str(e): + pytest.fail("The fix for issue #13995 is not working - still getting NoneType error") + else: + # Some other AttributeError, re-raise it + raise + + +def test_batch_create_with_litellm_sdk(): + """ + Test creating a batch using litellm SDK with metadata=None. + This is a more direct test of the original issue. + """ + # Mock the OpenAI batches instance to avoid actual API calls + with patch('litellm.batches.main.openai_batches_instance') as mock_openai_batches: + # Mock the response + mock_response = MagicMock() + mock_response.id = "batch_test123" + mock_openai_batches.create_batch.return_value = mock_response + + # This should not raise an exception + try: + response = litellm.create_batch( + completion_window="24h", + endpoint="/v1/chat/completions", + input_file_id="file-test123", + metadata=None, # This was causing the original issue + custom_llm_provider="openai" + ) + + assert response.id == "batch_test123" + + except Exception as e: + if "'NoneType' object has no attribute 'get'" in str(e): + pytest.fail("The fix for issue #13995 is not working - still getting NoneType error") + else: + # Some other exception, re-raise it + raise + + +if __name__ == "__main__": + # Run the tests + test_add_key_level_controls_with_none_metadata() + print("✓ test_add_key_level_controls_with_none_metadata passed") + + test_add_key_level_controls_simulates_original_issue() + print("✓ test_add_key_level_controls_simulates_original_issue passed") + + test_batch_create_with_litellm_sdk() + print("✓ test_batch_create_with_litellm_sdk passed") + + print("All tests passed! Issue #13995 fix is working correctly.") \ No newline at end of file From 04dc1a5351be4b926cc77e2e8a84cd89104ccdf0 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 16:16:19 -0700 Subject: [PATCH 65/66] [Feat] Add support for returning `images` with `gemini/gemini-2.5-flash-image-preview` with /chat/completions (#13983) * add gemini-2.5-flash-image-preview * add gemini-2.5-flash-image-preview * add image in ChatCompletionResponseMessage * test_gemini_image_generation_async * Revert "Merge pull request #13394 from Deviad/feature/enhance_logging_for_containers" This reverts commit 539b94ad4e97617ef5b1c052bd6da13f54400431, reversing changes made to 71af7bcf9c8373d349a54903695fcf26dbae93d9. * include `image` in Delta * fix _process_candidates should show the image response * fix: _handle_special_delta_attributes * test_gemini_image_generation_async_stream * image_generation_chat * UI - allow looking at generated images from /chat/completions * _create_streaming_choice * fix import StreamingChoices * fix ChatCompletionResponseMessage * test_gemini_image_generation * add gemini img migration * fix _extract_candidate_metadata * ui fix * fix batch endpoint test --- .dockerignore | 1 - .gitignore | 3 +- .../docs/completion/image_generation_chat.md | 232 +++++++ .../docs/extras/gemini_img_migration.md | 201 ++++++ docs/my-website/sidebars.js | 1 + litellm/_logging.py | 49 +- .../litellm_core_utils/streaming_handler.py | 92 ++- .../vertex_and_google_ai_studio_gemini.py | 158 ++++- litellm/types/utils.py | 17 + tests/llm_translation/test_gemini.py | 55 +- .../openai_endpoints_tests/input_azure.jsonl | 2 +- tests/test_litellm/conftest.py | 71 +- .../test_streaming_handler.py | 222 +++++- tests/test_litellm/test_logging_behavior.py | 638 ------------------ .../src/components/chat_ui.tsx | 47 +- .../chat_ui/llm_calls/chat_completion.tsx | 9 +- .../src/components/chat_ui/types.ts | 8 + 17 files changed, 991 insertions(+), 815 deletions(-) create mode 100644 docs/my-website/docs/completion/image_generation_chat.md create mode 100644 docs/my-website/docs/extras/gemini_img_migration.md delete mode 100644 tests/test_litellm/test_logging_behavior.py diff --git a/.dockerignore b/.dockerignore index 766b7a1db67..89c3c34bd71 100644 --- a/.dockerignore +++ b/.dockerignore @@ -10,4 +10,3 @@ tests *.tgz log.txt docker/Dockerfile.* -*.whl diff --git a/.gitignore b/.gitignore index 547734ddceb..ed8c88c8990 100644 --- a/.gitignore +++ b/.gitignore @@ -95,5 +95,4 @@ test.py litellm_config.yaml .cursor .vscode/launch.json -*.whl -litellm/proxy/to_delete_loadtest_work/* +litellm/proxy/to_delete_loadtest_work/* \ No newline at end of file diff --git a/docs/my-website/docs/completion/image_generation_chat.md b/docs/my-website/docs/completion/image_generation_chat.md new file mode 100644 index 00000000000..58ae70e2fff --- /dev/null +++ b/docs/my-website/docs/completion/image_generation_chat.md @@ -0,0 +1,232 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Image Generation in Chat Completions, Responses API + +This guide covers how to generate images when using the `chat/completions`. Note - if you want this on Responses API please file a Feature Request [here](https://github.com/BerriAI/litellm/issues/new). + +:::info + +Requires LiteLLM v1.76.1+ + +::: + +Supported Providers: +- Google AI Studio (`gemini`) +- Vertex AI (`vertex_ai/`) + +LiteLLM will standardize the `image` response in the assistant message for models that support image generation during chat completions. + +```python title="Example response from litellm" +"message": { + ... + "content": "Here's the image you requested:", + "image": { + "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...", + "detail": "auto" + } +} +``` + +## Quick Start + + + + +```python showLineNumbers title="Image generation with chat completion" +from litellm import completion +import os + +os.environ["GEMINI_API_KEY"] = "your-api-key" + +response = completion( + model="gemini/gemini-2.5-flash-image-preview", + messages=[ + {"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"} + ], +) + +print(response.choices[0].message.content) # Text response +print(response.choices[0].message.image) # Image data +``` + + + + +1. Setup config.yaml + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: gemini-image-gen + litellm_params: + model: gemini/gemini-2.5-flash-image-preview + api_key: os.environ/GEMINI_API_KEY +``` + +2. Run proxy server + +```bash showLineNumbers title="Start the proxy" +litellm --config config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Test it! + +```bash showLineNumbers title="Make request" +curl http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_KEY" \ + -d '{ + "model": "gemini-image-gen", + "messages": [ + { + "role": "user", + "content": "Generate an image of a banana wearing a costume that says LiteLLM" + } + ] + }' +``` + + + + +**Expected Response** + +```bash +{ + "id": "chatcmpl-3b66124d79a708e10c603496b363574c", + "choices": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "Here's the image you requested:", + "role": "assistant", + "image": { + "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...", + "detail": "auto" + } + } + } + ], + "created": 1723323084, + "model": "gemini/gemini-2.5-flash-image-preview", + "object": "chat.completion", + "usage": { + "completion_tokens": 12, + "prompt_tokens": 16, + "total_tokens": 28 + } +} +``` + +## Streaming Support + + + + +```python showLineNumbers title="Streaming image generation" +from litellm import completion +import os + +os.environ["GEMINI_API_KEY"] = "your-api-key" + +response = completion( + model="gemini/gemini-2.5-flash-image-preview", + messages=[ + {"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"} + ], + stream=True, +) + +for chunk in response: + if hasattr(chunk.choices[0].delta, "image") and chunk.choices[0].delta.image is not None: + print("Generated image:", chunk.choices[0].delta.image["url"]) + break +``` + + + + +```bash showLineNumbers title="Streaming request" +curl http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_KEY" \ + -d '{ + "model": "gemini-image-gen", + "messages": [ + { + "role": "user", + "content": "Generate an image of a banana wearing a costume that says LiteLLM" + } + ], + "stream": true + }' +``` + + + + +**Expected Streaming Response** + +```bash +data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":null}]} + +data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"content":"Here's the image you requested:"},"finish_reason":null}]} + +data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"image":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"}},"finish_reason":null}]} + +data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]} + +data: [DONE] +``` + +## Async Support + +```python showLineNumbers title="Async image generation" +from litellm import acompletion +import asyncio +import os + +os.environ["GEMINI_API_KEY"] = "your-api-key" + +async def generate_image(): + response = await acompletion( + model="gemini/gemini-2.5-flash-image-preview", + messages=[ + {"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"} + ], + ) + + print(response.choices[0].message.content) # Text response + print(response.choices[0].message.image) # Image data + + return response + +# Run the async function +asyncio.run(generate_image()) +``` + +## Supported Models + +| Provider | Model | +|----------|--------| +| Google AI Studio | `gemini/gemini-2.5-flash-image-preview` | +| Vertex AI | `vertex_ai/gemini-2.5-flash-image-preview` | + +## Spec + +The `image` field in the response follows this structure: + +```python +"image": { + "url": "data:image/png;base64,", + "detail": "auto" +} +``` + +- `url` - str: Base64 encoded image data in data URI format +- `detail` - str: Image detail level (always "auto" for generated images) + +The image is returned as a base64-encoded data URI that can be directly used in HTML `` tags or saved to a file. diff --git a/docs/my-website/docs/extras/gemini_img_migration.md b/docs/my-website/docs/extras/gemini_img_migration.md new file mode 100644 index 00000000000..ae02c89e6fd --- /dev/null +++ b/docs/my-website/docs/extras/gemini_img_migration.md @@ -0,0 +1,201 @@ +# Gemini Image Generation Migration Guide + +## Who is impacted by this change? + +Anyone using the following models with /chat/completions: +- `gemini/gemini-2.0-flash-exp-image-generation` +- `vertex_ai/gemini-2.5-flash-image-preview` + +## Key Change + +Gemini models now support image generation through chat completions. Images are returned in `response.choices[0].message.image` with base64 data URLs. + +## Before and After + +### Before +```python +from litellm import completion + +response = completion( + model="gemini/gemini-2.0-flash-exp-image-generation", + messages=[{"role": "user", "content": "Generate an image of a cat"}], + modalities=["image", "text"], +) + + +base_64_image_data = response.choices[0].message.content +``` + +### After +```python +from litellm import completion + +response = completion( + model="gemini/gemini-2.0-flash-exp-image-generation", + messages=[{"role": "user", "content": "Generate an image of a cat"}], + modalities=["image", "text"], +) + +# Image is now available in the response +image_url = response.choices[0].message.image["url"] # "data:image/png;base64,..." +``` + +## Usage + +### Using the Python SDK + +**Key Change:** +```diff +# Before +-- base_64_image_data = response.choices[0].message.content + +# After +++ image_url = response.choices[0].message.image["url"] +``` + +#### Basic Image Generation + +```python +from litellm import completion +import os + +# Set your API key +os.environ["GEMINI_API_KEY"] = "your-api-key" + +# Generate an image +response = completion( + model="gemini/gemini-2.0-flash-exp-image-generation", + messages=[{"role": "user", "content": "Generate an image of a cat"}], + modalities=["image", "text"], +) + +# Access the generated image +print(response.choices[0].message.content) # Text response (if any) +print(response.choices[0].message.image) # Image data +``` + +#### Response Format + +The image is returned in the `message.image` field: + +```python +{ + "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...", + "detail": "auto" +} +``` + +### Using the LiteLLM Proxy Server + +**Key Change:** +```diff +# Before +-- "content": "base64-image-data..." + +# After +++ "image": { +++ "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...", +++ "detail": "auto" +++ } +``` + +#### Configuration Setup + +1. **Configure your models in `config.yaml`:** + +```yaml +model_list: + - model_name: gemini-image-gen + litellm_params: + model: gemini/gemini-2.0-flash-exp-image-generation + api_key: os.environ/GEMINI_API_KEY + - model_name: vertex-image-gen + litellm_params: + model: vertex_ai/gemini-2.5-flash-image-preview + vertex_project: your-project-id + vertex_location: us-central1 + +general_settings: + master_key: sk-1234 # Your proxy API key +``` + +2. **Start the proxy server:** + +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +#### Making Requests + +**Using OpenAI SDK:** + +```python +from openai import OpenAI + +# Point to your proxy server +client = OpenAI( + api_key="sk-1234", # Your proxy API key + base_url="http://0.0.0.0:4000" +) + +response = client.chat.completions.create( + model="gemini-image-gen", + messages=[{"role": "user", "content": "Generate an image of a cat"}], + extra_body={"modalities": ["image", "text"]} +) + +# Access the generated image +print(response.choices[0].message.content) # Text response (if any) +print(response.choices[0].message.image) # Image data +``` + +**Using curl:** + +```bash +curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "gemini-image-gen", + "messages": [ + { + "role": "user", + "content": "Generate an image of a cat" + } + ], + "modalities": ["image", "text"] +}' +``` + +**Response format from proxy:** + +```json +{ + "id": "chatcmpl-123", + "object": "chat.completion", + "created": 1704089632, + "model": "gemini-image-gen", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "Here's an image of a cat for you!", + "image": { + "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...", + "detail": "auto" + } + }, + "finish_reason": "stop" + } + ], + "usage": { + "prompt_tokens": 10, + "completion_tokens": 8, + "total_tokens": 18 + } +} +``` + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 9af8a9b6f67..5e3de92ba83 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -493,6 +493,7 @@ const sidebars = { "guides/finetuned_models", "guides/security_settings", "completion/audio", + "completion/image_generation_chat", "completion/web_search", "completion/document_understanding", "completion/vision", diff --git a/litellm/_logging.py b/litellm/_logging.py index 1cf2a49832e..73902d2fc5a 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -4,41 +4,21 @@ import os import sys from datetime import datetime from logging import Formatter + set_verbose = False -def __strtobool(val: str) -> bool: - """Convert a string representation of truth to true (1) or false (0). - - True values are 'y', 'yes', 't', 'true', 'on', and '1'; false values - are 'n', 'no', 'f', 'false', 'off', and '0'. Raises ValueError if - 'val' is anything else. - """ - val = val.lower() - if val in ('y', 'yes', 't', 'true', 'on', '1'): - return True - elif val in ('n', 'no', 'f', 'false', 'off', '0'): - return False - else: - raise ValueError(f"invalid truth value {val!r}") - if set_verbose is True: logging.warning( "`litellm.set_verbose` is deprecated. Please set `os.environ['LITELLM_LOG'] = 'DEBUG'` for debug logs." ) - -json_logs = __strtobool(os.getenv("JSON_LOGS", "False")) +json_logs = bool(os.getenv("JSON_LOGS", False)) # Create a handler for the logger (you may need to adapt this based on your needs) log_level = os.getenv("LITELLM_LOG", "DEBUG") numeric_level: str = getattr(logging, log_level.upper()) handler = logging.StreamHandler() handler.setLevel(numeric_level) -log_file = os.getenv("LITELLM_LOG_FILE", "") -file_handler = None -if log_file: - file_handler = logging.FileHandler(log_file) - file_handler.setLevel(numeric_level) class JsonFormatter(Formatter): def __init__(self): super(JsonFormatter, self).__init__() @@ -60,7 +40,6 @@ class JsonFormatter(Formatter): return json.dumps(json_record) -json_formatter = JsonFormatter() # Function to set up exception handlers for JSON logging def _setup_json_exception_handlers(formatter): @@ -110,10 +89,8 @@ def _setup_json_exception_handlers(formatter): # Create a formatter and set it for the handler if json_logs: - handler.setFormatter(json_formatter) - if file_handler: - file_handler.setFormatter(json_formatter) - _setup_json_exception_handlers(json_formatter) + handler.setFormatter(JsonFormatter()) + _setup_json_exception_handlers(JsonFormatter()) else: formatter = logging.Formatter( "\033[92m%(asctime)s - %(name)s:%(levelname)s\033[0m: %(filename)s:%(lineno)s - %(message)s", @@ -121,18 +98,11 @@ else: ) handler.setFormatter(formatter) - if file_handler: - file_handler.setFormatter(formatter) verbose_proxy_logger = logging.getLogger("LiteLLM Proxy") verbose_router_logger = logging.getLogger("LiteLLM Router") verbose_logger = logging.getLogger("LiteLLM") -# Set logger levels -verbose_proxy_logger.setLevel(numeric_level) -verbose_router_logger.setLevel(numeric_level) -verbose_logger.setLevel(numeric_level) - # Add the handler to the logger verbose_router_logger.addHandler(handler) verbose_proxy_logger.addHandler(handler) @@ -155,13 +125,6 @@ def _suppress_loggers(): # Call the suppression function _suppress_loggers() -if file_handler: - verbose_router_logger.addHandler(file_handler) - verbose_proxy_logger.addHandler(file_handler) - verbose_logger.addHandler(file_handler) - - - ALL_LOGGERS = [ logging.getLogger(), verbose_logger, @@ -190,10 +153,10 @@ def _turn_on_json(): - Adds a JSON formatter to all loggers """ handler = logging.StreamHandler() - handler.setFormatter(json_formatter) + handler.setFormatter(JsonFormatter()) _initialize_loggers_with_handler(handler) # Set up exception handlers - _setup_json_exception_handlers(json_formatter) + _setup_json_exception_handlers(JsonFormatter()) def _turn_on_debug(): diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index b9433239271..01b2609d31d 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -20,7 +20,9 @@ from litellm.litellm_core_utils.redact_messages import LiteLLMLoggingObject from litellm.litellm_core_utils.thread_pool_executor import executor from litellm.types.llms.openai import ChatCompletionChunk from litellm.types.router import GenericLiteLLMParams -from litellm.types.utils import Delta +from litellm.types.utils import ( + Delta, +) from litellm.types.utils import GenericStreamingChunk as GChunk from litellm.types.utils import ( ModelResponse, @@ -35,6 +37,12 @@ from .exception_mapping_utils import exception_type from .llm_response_utils.get_api_base import get_api_base from .rules import Rules +# Constants for special delta attribute names +AUDIO_ATTRIBUTE = "audio" +IMAGE_ATTRIBUTE = "image" +TOOL_CALLS_ATTRIBUTE = "tool_calls" +FUNCTION_CALL_ATTRIBUTE = "function_call" + def is_async_iterable(obj: Any) -> bool: """ @@ -766,6 +774,66 @@ class CustomStreamWrapper: model_response.choices[0].delta = Delta(**_initial_delta) return model_response + def _has_special_delta_content(self, model_response: ModelResponseStream) -> bool: + """ + Check if the delta contains special content types (tool_calls, function_call, audio, or image). + """ + if len(model_response.choices) == 0: + return False + + delta = model_response.choices[0].delta + + # Check for tool_calls or function_call + if getattr(delta, TOOL_CALLS_ATTRIBUTE, None) is not None or getattr(delta, FUNCTION_CALL_ATTRIBUTE, None) is not None: + return True + + # Check for audio + if hasattr(delta, AUDIO_ATTRIBUTE) and getattr(delta, AUDIO_ATTRIBUTE, None) is not None: + return True + + # Check for image + if hasattr(delta, IMAGE_ATTRIBUTE) and getattr(delta, IMAGE_ATTRIBUTE, None) is not None: + return True + + return False + + def _handle_special_delta_content(self, model_response: ModelResponseStream) -> ModelResponseStream: + """ + Handle special delta content types by stripping role and returning the response. + """ + return self.strip_role_from_delta(model_response) + + def _has_special_delta_attribute(self, delta, attribute_name: str) -> bool: + """ + Check if delta has a specific attribute and it's not None. + """ + return delta is not None and getattr(delta, attribute_name, None) is not None + + def _copy_delta_attribute(self, source_delta, target_delta, attribute_name: str) -> None: + """ + Copy a specific attribute from source delta to target delta. + """ + setattr(target_delta, attribute_name, getattr(source_delta, attribute_name)) + + def _has_any_special_delta_attributes(self, delta) -> bool: + """ + Check if delta has any special attributes (audio, image). + """ + special_attributes = [AUDIO_ATTRIBUTE, IMAGE_ATTRIBUTE] + for attribute in special_attributes: + if self._has_special_delta_attribute(delta, attribute): + return True + return False + + def _handle_special_delta_attributes(self, delta, model_response: "ModelResponseStream") -> None: + """ + Handle special delta attributes (audio, image) by copying them to model_response. + """ + special_attributes = [AUDIO_ATTRIBUTE, IMAGE_ATTRIBUTE] + for attribute in special_attributes: + if self._has_special_delta_attribute(delta, attribute): + self._copy_delta_attribute(delta, model_response.choices[0].delta, attribute) + def return_processed_chunk_logic( # noqa self, completion_obj: Dict[str, Any], @@ -888,20 +956,8 @@ class CustomStreamWrapper: self.sent_last_chunk = True return model_response - elif ( - model_response.choices[0].delta.tool_calls is not None - or model_response.choices[0].delta.function_call is not None - ): - model_response = self.strip_role_from_delta(model_response) - - return model_response - elif ( - len(model_response.choices) > 0 - and hasattr(model_response.choices[0].delta, "audio") - and model_response.choices[0].delta.audio is not None - ): - model_response = self.strip_role_from_delta(model_response) - return model_response + elif self._has_special_delta_content(model_response): + return self._handle_special_delta_content(model_response) else: if hasattr(model_response, "usage"): self.chunks.append(model_response) @@ -1374,10 +1430,8 @@ class CustomStreamWrapper: ) ) model_response.choices[0].delta = Delta() - elif ( - delta is not None and getattr(delta, "audio", None) is not None - ): - model_response.choices[0].delta.audio = delta.audio + elif self._has_any_special_delta_attributes(delta): + self._handle_special_delta_attributes(delta, model_response) else: try: delta = ( diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index c5c4f46c92f..923e140a262 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -46,6 +46,7 @@ from litellm.types.llms.openai import ( ChatCompletionToolCallChunk, ChatCompletionToolCallFunctionChunk, ChatCompletionToolParamFunctionChunk, + ImageURLObject, OpenAIChatCompletionFinishReason, ) from litellm.types.llms.vertex_ai import ( @@ -89,11 +90,12 @@ from .transformation import ( if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj - from litellm.types.utils import ModelResponseStream + from litellm.types.utils import ModelResponseStream, StreamingChoices LoggingClass = LiteLLMLoggingObj else: LoggingClass = Any + StreamingChoices = Any class VertexAIBaseConfig: @@ -774,8 +776,9 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): elif "inlineData" in part: mime_type = part["inlineData"]["mimeType"] data = part["inlineData"]["data"] - # Check if inline data is audio - if so, exclude from text content - if mime_type.startswith("audio/"): + # Check if inline data is audio or image - if so, exclude from text content + # Images and audio are now handled separately in their respective response fields + if mime_type.startswith("audio/") or mime_type.startswith("image/"): continue _content_str += "data:{};base64,{}".format(mime_type, data) @@ -790,6 +793,23 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): content_str += _content_str return content_str, reasoning_content_str + + def _extract_image_response_from_parts( + self, parts: List[HttpxPartType] + ) -> Optional[ImageURLObject]: + """Extract image response from parts if present""" + for part in parts: + if "inlineData" in part: + mime_type = part["inlineData"]["mimeType"] + data = part["inlineData"]["data"] + if mime_type.startswith("image/"): + # Convert base64 data to data URI format + data_uri = f"data:{mime_type};base64,{data}" + return ImageURLObject( + url=data_uri, + detail="auto" + ) + return None def _extract_audio_response_from_parts( self, parts: List[HttpxPartType] @@ -1108,6 +1128,75 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): elif web_search_queries: web_search_requests = len(grounding_metadata) return web_search_requests + + @staticmethod + def _create_streaming_choice( + chat_completion_message: ChatCompletionResponseMessage, + candidate: Candidates, + idx: int, + tools: Optional[List[ChatCompletionToolCallChunk]], + functions: Optional[ChatCompletionToolCallFunctionChunk], + chat_completion_logprobs: Optional[ChoiceLogprobs], + image_response: Optional[ImageURLObject], + ) -> StreamingChoices: + """ + Helper method to create a streaming choice object for Vertex AI + """ + from litellm.types.utils import Delta, StreamingChoices + + # create a streaming choice object + choice = StreamingChoices( + finish_reason=VertexGeminiConfig._check_finish_reason( + chat_completion_message, candidate.get("finishReason") + ), + index=candidate.get("index", idx), + delta=Delta( + content=chat_completion_message.get("content"), + reasoning_content=chat_completion_message.get( + "reasoning_content" + ), + tool_calls=tools, + image=image_response, + function_call=functions, + ), + logprobs=chat_completion_logprobs, + enhancements=None, + ) + return choice + + @staticmethod + def _extract_candidate_metadata(candidate: Candidates) -> Tuple[List[dict], List[dict], List, List]: + """ + Extract metadata from a single candidate response. + + Returns: + grounding_metadata: List[dict] + url_context_metadata: List[dict] + safety_ratings: List + citation_metadata: List + """ + grounding_metadata: List[dict] = [] + url_context_metadata: List[dict] = [] + safety_ratings: List = [] + citation_metadata: List = [] + + if "groundingMetadata" in candidate: + if isinstance(candidate["groundingMetadata"], list): + grounding_metadata.extend(candidate["groundingMetadata"]) # type: ignore + else: + grounding_metadata.append(candidate["groundingMetadata"]) # type: ignore + + if "safetyRatings" in candidate: + safety_ratings.append(candidate["safetyRatings"]) + + if "citationMetadata" in candidate: + citation_metadata.append(candidate["citationMetadata"]) + + if "urlContextMetadata" in candidate: + # Add URL context metadata to grounding metadata + url_context_metadata.append(cast(dict, candidate["urlContextMetadata"])) + + return grounding_metadata, url_context_metadata, safety_ratings, citation_metadata @staticmethod def _process_candidates( @@ -1131,6 +1220,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): grounding_metadata: List[dict] = [] url_context_metadata: List[dict] = [] + image_response: Optional[ImageURLObject] = None safety_ratings: List = [] citation_metadata: List = [] chat_completion_message: ChatCompletionResponseMessage = {"role": "assistant"} @@ -1143,21 +1233,18 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): if "content" not in candidate: continue - if "groundingMetadata" in candidate: - if isinstance(candidate["groundingMetadata"], list): - grounding_metadata.extend(candidate["groundingMetadata"]) # type: ignore - else: - grounding_metadata.append(candidate["groundingMetadata"]) # type: ignore - - if "safetyRatings" in candidate: - safety_ratings.append(candidate["safetyRatings"]) - - if "citationMetadata" in candidate: - citation_metadata.append(candidate["citationMetadata"]) - - if "urlContextMetadata" in candidate: - # Add URL context metadata to grounding metadata - url_context_metadata.append(cast(dict, candidate["urlContextMetadata"])) + # Extract metadata using helper function + ( + candidate_grounding_metadata, + candidate_url_context_metadata, + candidate_safety_ratings, + candidate_citation_metadata, + ) = VertexGeminiConfig._extract_candidate_metadata(candidate) + + grounding_metadata.extend(candidate_grounding_metadata) + url_context_metadata.extend(candidate_url_context_metadata) + safety_ratings.extend(candidate_safety_ratings) + citation_metadata.extend(candidate_citation_metadata) if "parts" in candidate["content"]: ( @@ -1172,18 +1259,25 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): parts=candidate["content"]["parts"] ) ) + image_response = ( + VertexGeminiConfig()._extract_image_response_from_parts( + parts=candidate["content"]["parts"] + ) + ) if audio_response is not None: cast(Dict[str, Any], chat_completion_message)[ "audio" ] = audio_response chat_completion_message["content"] = None # OpenAI spec + elif image_response is not None: + # Handle image response - combine with text content into structured format + cast(Dict[str, Any], chat_completion_message)["image"] = image_response elif content is not None: chat_completion_message["content"] = content if reasoning_content is not None: chat_completion_message["reasoning_content"] = reasoning_content - ( functions, tools, @@ -1206,24 +1300,14 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): chat_completion_message["function_call"] = functions if isinstance(model_response, ModelResponseStream): - from litellm.types.utils import Delta, StreamingChoices - - # create a streaming choice object - choice = StreamingChoices( - finish_reason=VertexGeminiConfig._check_finish_reason( - chat_completion_message, candidate.get("finishReason") - ), - index=candidate.get("index", idx), - delta=Delta( - content=chat_completion_message.get("content"), - reasoning_content=chat_completion_message.get( - "reasoning_content" - ), - tool_calls=tools, - function_call=functions, - ), - logprobs=chat_completion_logprobs, - enhancements=None, + choice = VertexGeminiConfig._create_streaming_choice( + chat_completion_message=chat_completion_message, + candidate=candidate, + idx=idx, + tools=tools, + functions=functions, + chat_completion_logprobs=chat_completion_logprobs, + image_response=image_response ) model_response.choices.append(choice) elif isinstance(model_response, ModelResponse): diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 30d46d61c93..bcb6fb9fb69 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -51,6 +51,7 @@ from .llms.openai import ( ChatCompletionUsageBlock, FileSearchTool, FineTuningJob, + ImageURLObject, OpenAIChatCompletionChunk, OpenAIFileObject, OpenAIRealtimeStreamList, @@ -572,6 +573,7 @@ class Message(OpenAIObject): tool_calls: Optional[List[ChatCompletionMessageToolCall]] function_call: Optional[FunctionCall] audio: Optional[ChatCompletionAudioResponse] = None + image: Optional[ImageURLObject] = None reasoning_content: Optional[str] = None thinking_blocks: Optional[ List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]] @@ -588,6 +590,7 @@ class Message(OpenAIObject): function_call=None, tool_calls: Optional[list] = None, audio: Optional[ChatCompletionAudioResponse] = None, + image: Optional[ImageURLObject] = None, provider_specific_fields: Optional[Dict[str, Any]] = None, reasoning_content: Optional[str] = None, thinking_blocks: Optional[ @@ -621,6 +624,9 @@ class Message(OpenAIObject): if audio is not None: init_values["audio"] = audio + if image is not None: + init_values["image"] = image + if thinking_blocks is not None: init_values["thinking_blocks"] = thinking_blocks @@ -640,6 +646,10 @@ class Message(OpenAIObject): # OpenAI compatible APIs like mistral API will raise an error if audio is passed in if hasattr(self, "audio"): del self.audio + + if image is None: + if hasattr(self, "image"): + del self.image if annotations is None: # ensure default response matches OpenAI spec @@ -693,6 +703,7 @@ class Delta(OpenAIObject): function_call=None, tool_calls=None, audio: Optional[ChatCompletionAudioResponse] = None, + image: Optional[ImageURLObject] = None, reasoning_content: Optional[str] = None, thinking_blocks: Optional[ List[ @@ -710,6 +721,7 @@ class Delta(OpenAIObject): self.function_call: Optional[Union[FunctionCall, Any]] = None self.tool_calls: Optional[List[Union[ChatCompletionDeltaToolCall, Any]]] = None self.audio: Optional[ChatCompletionAudioResponse] = None + self.image: Optional[ImageURLObject] = None self.annotations: Optional[List[ChatCompletionAnnotation]] = None if reasoning_content is not None: @@ -729,6 +741,11 @@ class Delta(OpenAIObject): self.annotations = annotations else: del self.annotations + + if image is not None: + self.image = image + else: + del self.image if function_call is not None and isinstance(function_call, dict): self.function_call = FunctionCall(**function_call) diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index db403f81386..b25248ea765 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -261,7 +261,13 @@ def test_gemini_image_generation(): messages=[{"role": "user", "content": "Generate an image of a cat"}], modalities=["image", "text"], ) - assert response.choices[0].message.content is not None + + ######################################################### + # Important: Validate we did get an image in the response + ######################################################### + assert response.choices[0].message.image is not None + assert response.choices[0].message.image["url"] is not None + assert response.choices[0].message.image["url"].startswith("data:image/png;base64,") def test_gemini_thinking(): @@ -571,3 +577,50 @@ def test_gemini_tool_use(): stop_reason = chunk.choices[0].finish_reason assert stop_reason is not None assert stop_reason == "tool_calls" + +@pytest.mark.asyncio +async def test_gemini_image_generation_async(): + #litellm._turn_on_debug() + response = await litellm.acompletion( + messages=[{"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"}], + model="gemini/gemini-2.5-flash-image-preview", + ) + + CONTENT = response.choices[0].message.content + + IMAGE_URL = response.choices[0].message.image + print("IMAGE_URL: ", IMAGE_URL) + + assert CONTENT is not None + assert IMAGE_URL is not None + assert IMAGE_URL["url"] is not None + assert IMAGE_URL["url"].startswith("data:image/png;base64,") + + + +@pytest.mark.asyncio +async def test_gemini_image_generation_async_stream(): + #litellm._turn_on_debug() + response = await litellm.acompletion( + messages=[{"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"}], + model="gemini/gemini-2.5-flash-image-preview", + stream=True, + ) + + print("RESPONSE: ", response) + model_response_image = None + async for chunk in response: + print("CHUNK: ", chunk) + if hasattr(chunk.choices[0].delta, "image") and chunk.choices[0].delta.image is not None: + model_response_image = chunk.choices[0].delta.image + print("MODEL_RESPONSE_IMAGE: ", model_response_image) + assert model_response_image is not None + assert model_response_image["url"].startswith("data:image/png;base64,") + break + + ######################################################### + # Important: Validate we did get an image in the response + ######################################################### + assert model_response_image is not None + assert model_response_image["url"].startswith("data:image/png;base64,") + diff --git a/tests/openai_endpoints_tests/input_azure.jsonl b/tests/openai_endpoints_tests/input_azure.jsonl index 449bb88243c..e6178945e8d 100644 --- a/tests/openai_endpoints_tests/input_azure.jsonl +++ b/tests/openai_endpoints_tests/input_azure.jsonl @@ -1 +1 @@ -{"custom_id": "ae006110bb364606||/workspace/saved_models/meta-llama/Meta-Llama-3.1-8B-Instruct", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-mini", "temperature": 0, "max_tokens": 1024, "response_format": {"type": "json_object"}, "messages": [{"role": "user", "content": "# Instruction \n\nYou are an expert evaluator. Your task is to evaluate the quality of the responses generated by AI models. \nWe will provide you with the user query and an AI-generated responses.\nYo must respond in json"}]}} \ No newline at end of file +{"custom_id": "ae006110bb364606||/workspace/saved_models/meta-llama/Meta-Llama-3.1-8B-Instruct", "method": "POST", "url": "/chat/completions", "body": {"model": "gpt-4o-mini", "temperature": 0, "max_tokens": 1024, "response_format": {"type": "json_object"}, "messages": [{"role": "user", "content": "# Instruction \n\nYou are an expert evaluator. Your task is to evaluate the quality of the responses generated by AI models. \nWe will provide you with the user query and an AI-generated responses.\nYo must respond in json"}]}} \ No newline at end of file diff --git a/tests/test_litellm/conftest.py b/tests/test_litellm/conftest.py index dd0f03a8a37..ac8a00d850c 100644 --- a/tests/test_litellm/conftest.py +++ b/tests/test_litellm/conftest.py @@ -3,23 +3,9 @@ import importlib import os import sys -import tempfile -import random -import string import pytest -# Set up a temporary log directory and file BEFORE importing litellm -temp_dir = tempfile.mkdtemp(prefix="litellm_test_") -test_log_file = os.path.join(temp_dir, "test_litellm.log") - -# Store original log file for cleanup -orig_log_file = os.getenv("LITELLM_LOG_FILE") - -# Set environment variables to use temporary files BEFORE importing litellm -os.environ["LITELLM_LOG_FILE"] = test_log_file - -# Import litellm after setting up the environment sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path @@ -27,61 +13,6 @@ import asyncio import litellm -@pytest.fixture(scope="function") -def temp_log_file(): - """ - Creates a temporary log file in /tmp/litellm.log for testing. - Returns the path to the temporary log file and cleans it up after the test. - """ - # Generate a random number for the log file - random_number = ''.join(random.choices(string.digits, k=8)) - log_file_path = f"/tmp/litellm{random_number}.log" - - # Set the environment variable for litellm to use this temporary log file - original_log_file = os.environ.get("LITELLM_LOG_FILE") - os.environ["LITELLM_LOG_FILE"] = log_file_path - - yield log_file_path - - # Cleanup: Restore original environment variable and remove the temporary file - if original_log_file is not None: - os.environ["LITELLM_LOG_FILE"] = original_log_file - else: - os.environ.pop("LITELLM_LOG_FILE", None) - - # Remove the temporary log file if it exists - if os.path.exists(log_file_path): - try: - os.remove(log_file_path) - except OSError: - pass # Ignore errors if file can't be removed - - -@pytest.fixture(scope="session", autouse=True) -def cleanup_temp_log_dir(): - """ - Cleans up the temporary log directory created at module import time. - This runs once per test session after all tests are complete. - """ - yield - - if orig_log_file is not None: - os.environ["LITELLM_LOG_FILE"] = orig_log_file - else: - os.environ.pop("LITELLM_LOG_FILE", None) - - # Cleanup: Remove the temporary directory created at module import time - if os.path.exists(temp_dir): - try: - # Remove the test log file first - if os.path.exists(test_log_file): - os.remove(test_log_file) - - # Remove the temporary directory - import shutil - shutil.rmtree(temp_dir, ignore_errors=True) - except OSError: - pass # Ignore errors if cleanup fails @pytest.fixture(scope="session") def event_loop(): @@ -94,6 +25,7 @@ def event_loop(): + @pytest.fixture(scope="function", autouse=True) def setup_and_teardown(): """ @@ -145,4 +77,3 @@ def pytest_collection_modifyitems(config, items): # Reorder the items list items[:] = custom_logger_tests + other_tests - diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py index 127f3573cbc..422bbdc9c8b 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py @@ -15,7 +15,10 @@ from typing import Optional import litellm from litellm.litellm_core_utils.litellm_logging import Logging -from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper +from litellm.litellm_core_utils.streaming_handler import ( + AUDIO_ATTRIBUTE, + CustomStreamWrapper, +) from litellm.types.utils import ( CompletionTokensDetailsWrapper, Delta, @@ -813,3 +816,220 @@ def test_optional_combine_thinking_block_with_none_content( assert final_response.choices[0].delta.content == "The answer is 42" assert initialized_custom_stream_wrapper.sent_last_thinking_block is True assert not hasattr(final_response.choices[0].delta, "reasoning_content") + + +def test_has_special_delta_content( + initialized_custom_stream_wrapper: CustomStreamWrapper, +): + """Test the _has_special_delta_content helper method""" + + # Test empty choices + empty_response = ModelResponseStream( + id="test", created=1742056047, model=None, choices=[] + ) + assert not initialized_custom_stream_wrapper._has_special_delta_content(empty_response) + + # Test with tool_calls (simulate with mock object) + tool_call_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content=None, tool_calls=[{"id": "test"}]) + ) + ] + ) + assert initialized_custom_stream_wrapper._has_special_delta_content(tool_call_response) + + # Test with function_call (simulate with mock object) + function_call_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content=None, function_call={"name": "test_func"}) + ) + ] + ) + assert initialized_custom_stream_wrapper._has_special_delta_content(function_call_response) + + # Test with audio (simulate by adding audio attribute) + audio_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content=None) + ) + ] + ) + # Manually add audio attribute to delta + audio_response.choices[0].delta.audio = {"transcript": "test"} + assert initialized_custom_stream_wrapper._has_special_delta_content(audio_response) + + # Test with image (simulate by adding image attribute) + image_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content=None) + ) + ] + ) + # Manually add image attribute to delta + image_response.choices[0].delta.image = {"url": "test.jpg"} + assert initialized_custom_stream_wrapper._has_special_delta_content(image_response) + + # Test with regular content (should return False) + regular_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content="Hello world") + ) + ] + ) + assert not initialized_custom_stream_wrapper._has_special_delta_content(regular_response) + + +def test_handle_special_delta_content( + initialized_custom_stream_wrapper: CustomStreamWrapper, +): + """Test the _handle_special_delta_content helper method""" + test_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content="test", role="assistant") + ) + ] + ) + + # The method should call strip_role_from_delta + result = initialized_custom_stream_wrapper._handle_special_delta_content(test_response) + + # Should return the same response object (modified) + assert result is test_response + + # Should have set sent_first_chunk to True + assert initialized_custom_stream_wrapper.sent_first_chunk is True + + +def test_has_any_special_delta_attributes( + initialized_custom_stream_wrapper: CustomStreamWrapper, +): + """Test the _has_any_special_delta_attributes helper method""" + + # Test with delta that has audio attribute + class MockDelta: + def __init__(self): + self.audio = {"transcript": "Hello world"} + + audio_delta = MockDelta() + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(audio_delta) + assert result is True + + # Test with delta that has image attribute + class MockDeltaImage: + def __init__(self): + self.image = {"url": "test.jpg"} + + image_delta = MockDeltaImage() + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(image_delta) + assert result is True + + # Test with delta that has no special attributes + class MockDeltaRegular: + def __init__(self): + self.content = "regular content" + + regular_delta = MockDeltaRegular() + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(regular_delta) + assert result is False + + +def test_handle_special_delta_attributes( + initialized_custom_stream_wrapper: CustomStreamWrapper, +): + """Test the _handle_special_delta_attributes helper method""" + + # Create a model response + model_response = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content="test") + ) + ] + ) + + # Test with delta that has audio attribute + class MockDelta: + def __init__(self): + self.audio = {"transcript": "Hello world"} + + audio_delta = MockDelta() + initialized_custom_stream_wrapper._handle_special_delta_attributes(audio_delta, model_response) + + # Should copy the audio attribute + assert hasattr(model_response.choices[0].delta, "audio") + assert model_response.choices[0].delta.audio == {"transcript": "Hello world"} + + # Test with delta that has image attribute + class MockDeltaImage: + def __init__(self): + self.image = {"url": "test.jpg"} + + image_delta = MockDeltaImage() + model_response2 = ModelResponseStream( + id="test", created=1742056047, model=None, + choices=[ + StreamingChoices( + finish_reason=None, index=0, + delta=Delta(content="test") + ) + ] + ) + + initialized_custom_stream_wrapper._handle_special_delta_attributes(image_delta, model_response2) + + # Should copy the image attribute + assert hasattr(model_response2.choices[0].delta, "image") + assert model_response2.choices[0].delta.image == {"url": "test.jpg"} + + +def test_has_special_delta_attribute( + initialized_custom_stream_wrapper: CustomStreamWrapper, +): + """Test the _has_special_delta_attribute helper method""" + + # Test with None delta + assert not initialized_custom_stream_wrapper._has_special_delta_attribute(None, "audio") + + # Test with delta that has the attribute + class MockDelta: + def __init__(self): + self.audio = {"transcript": "test"} + + delta_with_audio = MockDelta() + assert initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_audio, "audio") + + # Test with delta that doesn't have the attribute + class MockDeltaNoAudio: + def __init__(self): + self.content = "test" + + delta_without_audio = MockDeltaNoAudio() + assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_without_audio, "audio") + + # Test with delta that has the attribute but it's None + class MockDeltaNone: + def __init__(self): + self.audio = None + + delta_with_none = MockDeltaNone() + assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_none, "audio") diff --git a/tests/test_litellm/test_logging_behavior.py b/tests/test_litellm/test_logging_behavior.py deleted file mode 100644 index 24f92838acc..00000000000 --- a/tests/test_litellm/test_logging_behavior.py +++ /dev/null @@ -1,638 +0,0 @@ -import os -import tempfile -import re -import json -from pathlib import Path -from datetime import datetime - -import pytest - -# Import the loggers from litellm._logging -from litellm._logging import verbose_logger, verbose_proxy_logger, verbose_router_logger - - -class TestLoggingBehavior: - """Test suite to verify logging behavior for all LiteLLM loggers.""" - - def read_log_file_contents(self, log_file_path): - """Helper method to read and return contents of log file.""" - if not os.path.exists(log_file_path): - return "" - - with open(log_file_path, 'r') as f: - return f.read() - - @pytest.fixture(autouse=True) - def setup_log_file(self, temp_log_file): - """Use the temp_log_file fixture to ensure proper isolation.""" - self.temp_log_path = temp_log_file - - # Set environment variable before importing/reloading - original_log_file = os.environ.get("LITELLM_LOG_FILE") - os.environ["LITELLM_LOG_FILE"] = temp_log_file - - # Force reload of the logging module to pick up new environment variable - import importlib - import litellm._logging - importlib.reload(litellm._logging) - - yield - - # Cleanup: Restore original environment variable - if original_log_file is not None: - os.environ["LITELLM_LOG_FILE"] = original_log_file - else: - os.environ.pop("LITELLM_LOG_FILE", None) - - # Reload again to restore original state - importlib.reload(litellm._logging) - - def test_verbose_logger_info_level(self): - """Test that verbose_logger writes to file with INFO level.""" - test_message = "INFO level test message from verbose_logger" - - # Log at INFO level - verbose_logger.info(test_message) - - # Force flush all handlers to ensure they write to disk - for handler in verbose_logger.handlers: - if hasattr(handler, 'flush'): - handler.flush() - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_verbose_logger_debug_level(self): - """Test that verbose_logger writes to file with DEBUG level.""" - test_message = "DEBUG level test message from verbose_logger" - - # Log at DEBUG level - verbose_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_verbose_proxy_logger_info_level(self): - """Test that verbose_proxy_logger writes to file with INFO level.""" - test_message = "INFO level test message from verbose_proxy_logger" - - # Log at INFO level - verbose_proxy_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_verbose_proxy_logger_debug_level(self): - """Test that verbose_proxy_logger writes to file with DEBUG level.""" - test_message = "DEBUG level test message from verbose_proxy_logger" - - # Log at DEBUG level - verbose_proxy_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_verbose_router_logger_info_level(self): - """Test that verbose_router_logger writes to file with INFO level.""" - test_message = "INFO level test message from verbose_router_logger" - - # Log at INFO level - verbose_router_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_verbose_router_logger_debug_level(self): - """Test that verbose_router_logger writes to file with DEBUG level.""" - test_message = "DEBUG level test message from verbose_router_logger" - - # Log at DEBUG level - verbose_router_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert test_message in log_contents, f"Message '{test_message}' should be found in log file" - - def test_log_format_includes_timestamp_and_level(self): - """Test that log entries include timestamp and level information.""" - test_message = "Format test message" - - # Log at INFO level - verbose_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - - # Check for timestamp format (should be in HH:MM:SS format based on _logging.py) - assert re.search(r'\d{2}:\d{2}:\d{2}', log_contents), "Log should contain timestamp in HH:MM:SS format" - - # Check for level information - assert 'INFO' in log_contents, "Log should contain INFO level indicator" - - # Check for logger name - assert 'LiteLLM' in log_contents, "Log should contain LiteLLM logger name" - - def test_multiple_loggers_write_to_same_file(self): - """Test that all loggers write to the same file.""" - messages = { - 'verbose_logger': "Message from verbose_logger", - 'verbose_proxy_logger': "Message from verbose_proxy_logger", - 'verbose_router_logger': "Message from verbose_router_logger" - } - - # Log messages from different loggers - verbose_logger.info(messages['verbose_logger']) - verbose_proxy_logger.info(messages['verbose_proxy_logger']) - verbose_router_logger.info(messages['verbose_router_logger']) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - - # Verify all messages are in the same file - for message in messages.values(): - assert message in log_contents, f"Message '{message}' should be found in log file" - - def test_log_file_is_not_empty(self): - """Test that the log file is not empty after logging.""" - # Log a message - verbose_logger.info("Test message to ensure file is not empty") - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - - # Verify file is not empty - assert len(log_contents.strip()) > 0, "Log file should not be empty after logging" - - -class TestJSONLoggingBehavior: - """Test suite to verify JSON logging behavior for all LiteLLM loggers.""" - - def read_log_file_contents(self, log_file_path): - """Helper method to read and return contents of log file.""" - if not os.path.exists(log_file_path): - return "" - - with open(log_file_path, 'r') as f: - return f.read() - - @pytest.fixture(autouse=True) - def setup_json_logging(self, temp_log_file): - """Set up JSON logging environment and ensure proper isolation.""" - self.temp_log_path = temp_log_file - - # Store original environment variables - original_log_file = os.environ.get("LITELLM_LOG_FILE") - original_json_logs = os.environ.get("JSON_LOGS") - - # Set environment variables for JSON logging - os.environ["LITELLM_LOG_FILE"] = temp_log_file - os.environ["JSON_LOGS"] = "True" - - # Force reload of the logging module to pick up new environment variables - import importlib - import litellm._logging - importlib.reload(litellm._logging) - - yield - - # Cleanup: Restore original environment variables - if original_log_file is not None: - os.environ["LITELLM_LOG_FILE"] = original_log_file - else: - os.environ.pop("LITELLM_LOG_FILE", None) - - if original_json_logs is not None: - os.environ["JSON_LOGS"] = original_json_logs - else: - os.environ.pop("JSON_LOGS", None) - - # Reload again to restore original state - importlib.reload(litellm._logging) - - def test_verbose_logger_json_info_level(self): - """Test that verbose_logger writes JSON formatted logs at INFO level.""" - test_message = "JSON INFO level test message from verbose_logger" - - # Log at INFO level - verbose_logger.info(test_message) - - # Force flush all handlers to ensure they write to disk - for handler in verbose_logger.handlers: - if hasattr(handler, 'flush'): - handler.flush() - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - assert len(log_lines) > 0, "Should have at least one log line" - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - - # Verify JSON structure - assert "message" in target_log, "JSON log should contain 'message' field" - assert "level" in target_log, "JSON log should contain 'level' field" - assert "timestamp" in target_log, "JSON log should contain 'timestamp' field" - - # Verify content - assert target_log["message"] == test_message - assert target_log["level"] == "INFO" - - # Verify timestamp is in ISO 8601 format - timestamp_str = target_log["timestamp"] - try: - datetime.fromisoformat(timestamp_str) - except ValueError: - pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") - - def test_verbose_logger_json_debug_level(self): - """Test that verbose_logger writes JSON formatted logs at DEBUG level.""" - test_message = "JSON DEBUG level test message from verbose_logger" - - # Log at DEBUG level - verbose_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - assert target_log["level"] == "DEBUG" - - def test_verbose_proxy_logger_json_info_level(self): - """Test that verbose_proxy_logger writes JSON formatted logs at INFO level.""" - test_message = "JSON INFO level test message from verbose_proxy_logger" - - # Log at INFO level - verbose_proxy_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - - # Verify JSON structure and content - assert target_log["message"] == test_message - assert target_log["level"] == "INFO" - - # Verify timestamp is in ISO 8601 format - timestamp_str = target_log["timestamp"] - try: - datetime.fromisoformat(timestamp_str) - except ValueError: - pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") - - def test_verbose_proxy_logger_json_debug_level(self): - """Test that verbose_proxy_logger writes JSON formatted logs at DEBUG level.""" - test_message = "JSON DEBUG level test message from verbose_proxy_logger" - - # Log at DEBUG level - verbose_proxy_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - assert target_log["level"] == "DEBUG" - - def test_verbose_router_logger_json_info_level(self): - """Test that verbose_router_logger writes JSON formatted logs at INFO level.""" - test_message = "JSON INFO level test message from verbose_router_logger" - - # Log at INFO level - verbose_router_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - - # Verify JSON structure and content - assert target_log["message"] == test_message - assert target_log["level"] == "INFO" - - # Verify timestamp is in ISO 8601 format - timestamp_str = target_log["timestamp"] - try: - datetime.fromisoformat(timestamp_str) - except ValueError: - pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format") - - def test_verbose_router_logger_json_debug_level(self): - """Test that verbose_router_logger writes JSON formatted logs at DEBUG level.""" - test_message = "JSON DEBUG level test message from verbose_router_logger" - - # Log at DEBUG level - verbose_router_logger.debug(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify structure - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - assert target_log["level"] == "DEBUG" - - def test_json_output_is_valid_json(self): - """Test that all JSON log output can be parsed as valid JSON.""" - test_messages = [ - "JSON test message 1", - "JSON test message 2", - "JSON test message 3" - ] - - # Log messages from all loggers - verbose_logger.info(test_messages[0]) - verbose_proxy_logger.info(test_messages[1]) - verbose_router_logger.info(test_messages[2]) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse each line as JSON - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - parsed_logs = [] - - for line in log_lines: - try: - parsed = json.loads(line) - parsed_logs.append(parsed) - except json.JSONDecodeError as e: - pytest.fail(f"Failed to parse JSON log line: {line}. Error: {e}") - - assert len(parsed_logs) >= len(test_messages), f"Should have at least {len(test_messages)} parsed log entries" - - # Verify each parsed log has required fields - for parsed_log in parsed_logs: - assert isinstance(parsed_log, dict), "Parsed log should be a dictionary" - assert "message" in parsed_log, "Each log should have a 'message' field" - assert "level" in parsed_log, "Each log should have a 'level' field" - assert "timestamp" in parsed_log, "Each log should have a 'timestamp' field" - - def test_json_timestamp_iso8601_format(self): - """Test that JSON log timestamps are in ISO 8601 format.""" - test_message = "Timestamp format test message" - - # Log a message - verbose_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify timestamp format - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - - timestamp_str = target_log["timestamp"] - - # Verify timestamp can be parsed as ISO 8601 - try: - parsed_timestamp = datetime.fromisoformat(timestamp_str) - assert isinstance(parsed_timestamp, datetime), "Parsed timestamp should be a datetime object" - except ValueError as e: - pytest.fail(f"Timestamp '{timestamp_str}' is not in valid ISO 8601 format. Error: {e}") - - # Verify timestamp format matches expected pattern (YYYY-MM-DDTHH:MM:SS.ffffff) - import re - iso8601_pattern = r'^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?$' - assert re.match(iso8601_pattern, timestamp_str), f"Timestamp '{timestamp_str}' does not match ISO 8601 pattern" - - def test_json_logs_contain_expected_fields(self): - """Test that JSON logs contain all expected fields with correct types.""" - test_message = "Field validation test message" - - # Log a message - verbose_logger.info(test_message) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse JSON and verify fields - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - - # Find the line containing our test message - target_log = None - for line in log_lines: - try: - parsed = json.loads(line) - if parsed.get("message") == test_message: - target_log = parsed - break - except json.JSONDecodeError: - continue - - assert target_log is not None, f"Could not find JSON log entry with message: {test_message}" - - # Verify required fields exist and have correct types - assert "message" in target_log, "JSON log should contain 'message' field" - assert "level" in target_log, "JSON log should contain 'level' field" - assert "timestamp" in target_log, "JSON log should contain 'timestamp' field" - - assert isinstance(target_log["message"], str), "'message' field should be a string" - assert isinstance(target_log["level"], str), "'level' field should be a string" - assert isinstance(target_log["timestamp"], str), "'timestamp' field should be a string" - - # Verify field values - assert target_log["message"] == test_message - assert target_log["level"] in ["DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"], "Level should be a valid log level" - - def test_multiple_json_loggers_write_to_same_file(self): - """Test that all loggers write JSON formatted logs to the same file.""" - messages = { - 'verbose_logger': "JSON message from verbose_logger", - 'verbose_proxy_logger': "JSON message from verbose_proxy_logger", - 'verbose_router_logger': "JSON message from verbose_router_logger" - } - - # Log messages from different loggers - verbose_logger.info(messages['verbose_logger']) - verbose_proxy_logger.info(messages['verbose_proxy_logger']) - verbose_router_logger.info(messages['verbose_router_logger']) - - # Read log file contents - log_file_path = os.environ.get("LITELLM_LOG_FILE") - assert log_file_path is not None, "LITELLM_LOG_FILE environment variable should be set" - - log_contents = self.read_log_file_contents(log_file_path) - assert log_contents.strip(), "Log file should not be empty" - - # Parse all JSON logs - log_lines = [line.strip() for line in log_contents.strip().split('\n') if line.strip()] - parsed_logs = [] - - for line in log_lines: - try: - parsed = json.loads(line) - parsed_logs.append(parsed) - except json.JSONDecodeError: - continue - - # Find logs for each message - found_messages = set() - for parsed_log in parsed_logs: - message = parsed_log.get("message", "") - if message in messages.values(): - found_messages.add(message) - - # Verify all messages are found in JSON format - for message in messages.values(): - assert message in found_messages, f"Message '{message}' should be found in JSON logs" \ No newline at end of file diff --git a/ui/litellm-dashboard/src/components/chat_ui.tsx b/ui/litellm-dashboard/src/components/chat_ui.tsx index b478336f9df..beccef260f9 100644 --- a/ui/litellm-dashboard/src/components/chat_ui.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui.tsx @@ -450,6 +450,38 @@ const ChatUI: React.FC = ({ ]); }; + const updateChatImageUI = (imageUrl: string, model?: string) => { + setChatHistory((prev) => { + const last = prev[prev.length - 1]; + // If the last message is from assistant and has content, add image to it + if (last && last.role === "assistant" && !last.isImage) { + const updated = { + ...last, + image: { + url: imageUrl, + detail: "auto" + }, + model: last.model ?? model + }; + return [...prev.slice(0, -1), updated]; + } else { + // Otherwise create a new assistant message with just the image + return [ + ...prev, + { + role: "assistant", + content: "", + model, + image: { + url: imageUrl, + detail: "auto" + } + } + ]; + } + }); + }; + const handleKeyDown = (event: React.KeyboardEvent) => { if (event.key === 'Enter' && !event.shiftKey) { event.preventDefault(); // Prevent default to avoid newline @@ -611,7 +643,8 @@ const ChatUI: React.FC = ({ traceId, selectedVectorStores.length > 0 ? selectedVectorStores : undefined, selectedGuardrails.length > 0 ? selectedGuardrails : undefined, - selectedMCPTools // Pass the selected tool directly + selectedMCPTools, // Pass the selected tool directly + updateChatImageUI // Pass the image callback ); } else if (endpointType === EndpointType.IMAGE) { // For image generation @@ -1057,6 +1090,18 @@ const ChatUI: React.FC = ({ > {typeof message.content === "string" ? message.content : ""} + + {/* Show generated image from chat completions */} + {message.image && ( +

+ Generated image +
+ )} )} diff --git a/ui/litellm-dashboard/src/components/chat_ui/llm_calls/chat_completion.tsx b/ui/litellm-dashboard/src/components/chat_ui/llm_calls/chat_completion.tsx index e611294c8de..70f5e36f863 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/llm_calls/chat_completion.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/llm_calls/chat_completion.tsx @@ -17,7 +17,8 @@ export async function makeOpenAIChatCompletionRequest( traceId?: string, vector_store_ids?: string[], guardrails?: string[], - selectedMCPTool?: string + selectedMCPTool?: string, + onImageGenerated?: (imageUrl: string, model?: string) => void ) { // base url should be the current base_url const isLocal = process.env.NODE_ENV === "development"; @@ -103,6 +104,12 @@ export async function makeOpenAIChatCompletionRequest( fullResponseContent += content; } + // Process image generation if present + if (delta && delta.image && onImageGenerated) { + console.log("Image generated:", delta.image); + onImageGenerated(delta.image.url, chunk.model); + } + // Process reasoning content if present - using type assertion if (delta && delta.reasoning_content) { const reasoningContent = delta.reasoning_content; diff --git a/ui/litellm-dashboard/src/components/chat_ui/types.ts b/ui/litellm-dashboard/src/components/chat_ui/types.ts index e7b25f8a38a..e0eba9f2aad 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/types.ts +++ b/ui/litellm-dashboard/src/components/chat_ui/types.ts @@ -7,6 +7,10 @@ export interface Delta { audio?: any; refusal?: any; provider_specific_fields?: any; + image?: { + url: string; + detail: string; + }; } export interface CompletionTokensDetails { @@ -67,6 +71,10 @@ export interface MessageType { }; toolName?: string; imagePreviewUrl?: string; // For storing image preview URL in chat history + image?: { + url: string; + detail: string; + }; } export interface MultimodalContent { From d5774d89ccc6045de8008362bd9f3ac08dffd785 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 27 Aug 2025 16:44:50 -0700 Subject: [PATCH 66/66] Update Docker image tag to v1.75.8-stable (#14014) Co-authored-by: Cursor Agent Co-authored-by: ishaan --- docs/my-website/release_notes/v1.75.8/index.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/my-website/release_notes/v1.75.8/index.md b/docs/my-website/release_notes/v1.75.8/index.md index 474a934743a..d7d4f37c4ee 100644 --- a/docs/my-website/release_notes/v1.75.8/index.md +++ b/docs/my-website/release_notes/v1.75.8/index.md @@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem'; docker run \ -e STORE_MODEL_IN_DB=True \ -p 4000:4000 \ -ghcr.io/berriai/litellm:v1.75.8 +ghcr.io/berriai/litellm:v1.75.8-stable ```