diff --git a/docs/my-website/docs/completion/drop_params.md b/docs/my-website/docs/completion/drop_params.md
index e79a88e14bb..590d9a45955 100644
--- a/docs/my-website/docs/completion/drop_params.md
+++ b/docs/my-website/docs/completion/drop_params.md
@@ -107,4 +107,76 @@ response = litellm.completion(
-**additional_drop_params**: List or null - Is a list of openai params you want to drop when making a call to the model.
\ No newline at end of file
+**additional_drop_params**: List or null - Is a list of openai params you want to drop when making a call to the model.
+
+## Specify allowed openai params in a request
+
+Tell litellm to allow specific openai params in a request. Use this if you get a `litellm.UnsupportedParamsError` and want to allow a param. LiteLLM will pass the param as is to the model.
+
+
+
+
+Each instance writes updates to redis +
+ + +### Stage 2. A single instance flushes the redis queue to the DB + +A single instance will acquire a lock on the DB and flush all elements in the redis queue to the DB. + + ++A single instance flushes the redis queue to the DB +
+ + +## Usage + +### Required components + +- Redis +- Postgres + +### Setup on LiteLLM config + +You can enable using the redis buffer by setting `use_redis_transaction_buffer: true` in the `general_settings` section of your `proxy_config.yaml` file. + +Note: This setup requires litellm to be connected to a redis instance. + +```yaml showLineNumbers title="litellm proxy_config.yaml" +general_settings: + use_redis_transaction_buffer: true + +litellm_settings: + cache: True + cache_params: + type: redis + supported_call_types: [] # Optional: Set cache for proxy, but not on the actual llm api call +``` + + diff --git a/docs/my-website/img/deadlock_fix_1.png b/docs/my-website/img/deadlock_fix_1.png new file mode 100644 index 00000000000..3dff86a4d20 Binary files /dev/null and b/docs/my-website/img/deadlock_fix_1.png differ diff --git a/docs/my-website/img/deadlock_fix_2.png b/docs/my-website/img/deadlock_fix_2.png new file mode 100644 index 00000000000..345acff586f Binary files /dev/null and b/docs/my-website/img/deadlock_fix_2.png differ diff --git a/docs/my-website/package-lock.json b/docs/my-website/package-lock.json index 6c07e67d912..06251b16bbc 100644 --- a/docs/my-website/package-lock.json +++ b/docs/my-website/package-lock.json @@ -12559,9 +12559,10 @@ } }, "node_modules/image-size": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/image-size/-/image-size-1.1.1.tgz", - "integrity": "sha512-541xKlUw6jr/6gGuk92F+mYM5zaFAc5ahphvkqvNe2bQ6gVBkd6bfrmVJ2t4KDAfikAYZyIqTnktX3i6/aQDrQ==", + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/image-size/-/image-size-1.2.1.tgz", + "integrity": "sha512-rH+46sQJ2dlwfjfhCyNx5thzrv+dtmBIhPHk0zgRUukHzZ/kRueTJXoYYsclBaKcSMBWuGbOFXtioLpzTb5euw==", + "license": "MIT", "dependencies": { "queue": "6.0.2" }, diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 1542edceb5d..1cb3e02707d 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -53,7 +53,7 @@ const sidebars = { { type: "category", label: "Architecture", - items: ["proxy/architecture", "proxy/db_info", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling"], + items: ["proxy/architecture", "proxy/db_info", "proxy/db_deadlocks", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling"], }, { type: "link", diff --git a/litellm/llms/azure/chat/o_series_transformation.py b/litellm/llms/azure/chat/o_series_transformation.py index 0ca3a28d23e..21aafce7fbf 100644 --- a/litellm/llms/azure/chat/o_series_transformation.py +++ b/litellm/llms/azure/chat/o_series_transformation.py @@ -14,6 +14,7 @@ Translations handled by LiteLLM: from typing import List, Optional +import litellm from litellm import verbose_logger from litellm.types.llms.openai import AllMessageValues from litellm.utils import get_model_info @@ -22,6 +23,27 @@ from ...openai.chat.o_series_transformation import OpenAIOSeriesConfig class AzureOpenAIO1Config(OpenAIOSeriesConfig): + def get_supported_openai_params(self, model: str) -> list: + """ + Get the supported OpenAI params for the Azure O-Series models + """ + all_openai_params = litellm.OpenAIGPTConfig().get_supported_openai_params( + model=model + ) + non_supported_params = [ + "logprobs", + "top_p", + "presence_penalty", + "frequency_penalty", + "top_logprobs", + ] + + o_series_only_param = ["reasoning_effort"] + all_openai_params.extend(o_series_only_param) + return [ + param for param in all_openai_params if param not in non_supported_params + ] + def should_fake_stream( self, model: Optional[str], diff --git a/litellm/main.py b/litellm/main.py index f69454aaad8..56b0aa36716 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1115,6 +1115,7 @@ def completion( # type: ignore # noqa: PLR0915 messages=messages, reasoning_effort=reasoning_effort, thinking=thinking, + allowed_openai_params=kwargs.get("allowed_openai_params"), **non_default_params, ) diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index 2ee830bca44..52948c927e9 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -13,5 +13,3 @@ litellm_settings: cache_params: type: redis supported_call_types: [] - callbacks: ["prometheus"] - service_callback: ["prometheus_system"] \ No newline at end of file diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 8716779d1f8..51a6ed17b16 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1950,6 +1950,7 @@ all_litellm_params = [ "use_in_pass_through", "merge_reasoning_content_in_choices", "litellm_credential_name", + "allowed_openai_params", ] + list(StandardCallbackDynamicParams.__annotations__.keys()) diff --git a/litellm/utils.py b/litellm/utils.py index 1a35e58dc1c..4283cf2df11 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2843,6 +2843,7 @@ def get_optional_params( # noqa: PLR0915 api_version=None, parallel_tool_calls=None, drop_params=None, + allowed_openai_params: Optional[List[str]] = None, reasoning_effort=None, additional_drop_params=None, messages: Optional[List[AllMessageValues]] = None, @@ -2928,6 +2929,7 @@ def get_optional_params( # noqa: PLR0915 "api_version": None, "parallel_tool_calls": None, "drop_params": None, + "allowed_openai_params": None, "additional_drop_params": None, "messages": None, "reasoning_effort": None, @@ -2944,6 +2946,7 @@ def get_optional_params( # noqa: PLR0915 and k != "custom_llm_provider" and k != "api_version" and k != "drop_params" + and k != "allowed_openai_params" and k != "additional_drop_params" and k != "messages" and k in default_params @@ -3053,6 +3056,12 @@ def get_optional_params( # noqa: PLR0915 tool_function["parameters"] = new_parameters def _check_valid_arg(supported_params: List[str]): + """ + Check if the params passed to completion() are supported by the provider + + Args: + supported_params: List[str] - supported params from the litellm config + """ verbose_logger.info( f"\nLiteLLM completion() model= {model}; provider = {custom_llm_provider}" ) @@ -3086,7 +3095,7 @@ def get_optional_params( # noqa: PLR0915 else: raise UnsupportedParamsError( status_code=500, - message=f"{custom_llm_provider} does not support parameters: {unsupported_params}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n", + message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.", ) supported_params = get_supported_openai_params( @@ -3096,7 +3105,14 @@ def get_optional_params( # noqa: PLR0915 supported_params = get_supported_openai_params( model=model, custom_llm_provider="openai" ) - _check_valid_arg(supported_params=supported_params or []) + + supported_params = supported_params or [] + allowed_openai_params = allowed_openai_params or [] + supported_params.extend(allowed_openai_params) + + _check_valid_arg( + supported_params=supported_params or [], + ) ## raise exception if provider doesn't support passed in param if custom_llm_provider == "anthropic": ## check if unsupported param passed in @@ -3735,6 +3751,26 @@ def get_optional_params( # noqa: PLR0915 if k not in default_params.keys(): optional_params[k] = passed_params[k] print_verbose(f"Final returned optional params: {optional_params}") + optional_params = _apply_openai_param_overrides( + optional_params=optional_params, + non_default_params=non_default_params, + allowed_openai_params=allowed_openai_params, + ) + return optional_params + + +def _apply_openai_param_overrides( + optional_params: dict, non_default_params: dict, allowed_openai_params: list +): + """ + If user passes in allowed_openai_params, apply them to optional_params + + These params will get passed as is to the LLM API since the user opted in to passing them in the request + """ + if allowed_openai_params: + for param in allowed_openai_params: + if param not in optional_params: + optional_params[param] = non_default_params.pop(param, None) return optional_params diff --git a/tests/image_gen_tests/test_image_generation.py b/tests/image_gen_tests/test_image_generation.py index 1f2c563f266..bee9d1f7d50 100644 --- a/tests/image_gen_tests/test_image_generation.py +++ b/tests/image_gen_tests/test_image_generation.py @@ -167,7 +167,9 @@ class TestAzureOpenAIDalle3(BaseImageGenTest): litellm.set_verbose = True return { "model": "azure/dall-e-3-test", - "api_version": "2023-09-01-preview", + "api_version": "2023-12-01-preview", + "api_base": os.getenv("AZURE_SWEDEN_API_BASE"), + "api_key": os.getenv("AZURE_SWEDEN_API_KEY"), "metadata": { "model_info": { "base_model": "azure/dall-e-3", diff --git a/tests/litellm_utils_tests/test_supports_tool_choice.py b/tests/litellm_utils_tests/test_supports_tool_choice.py index cfa190f74b1..e09be153195 100644 --- a/tests/litellm_utils_tests/test_supports_tool_choice.py +++ b/tests/litellm_utils_tests/test_supports_tool_choice.py @@ -137,6 +137,8 @@ async def test_supports_tool_choice(): or model_name in block_list or "azure/eu" in model_name or "azure/us" in model_name + or "o1" in model_name + or "o3" in model_name ): continue diff --git a/tests/llm_translation/test_azure_o_series.py b/tests/llm_translation/test_azure_o_series.py index 13ba4169ceb..3d3764f0492 100644 --- a/tests/llm_translation/test_azure_o_series.py +++ b/tests/llm_translation/test_azure_o_series.py @@ -21,9 +21,10 @@ from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest class TestAzureOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest): def get_base_completion_call_args(self): return { - "model": "azure/o1-preview", + "model": "azure/o1", "api_key": os.getenv("AZURE_OPENAI_O1_KEY"), - "api_base": "https://openai-gpt-4-test-v-1.openai.azure.com", + "api_base": "https://openai-prod-test.openai.azure.com", + "api_version": "2024-12-01-preview" } def get_client(self): @@ -31,7 +32,7 @@ class TestAzureOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest): return AzureOpenAI( api_key="my-fake-o1-key", - base_url="https://openai-gpt-4-test-v-1.openai.azure.com", + base_url="https://openai-prod-test.openai.azure.com", api_version="2024-02-15-preview", ) @@ -170,3 +171,54 @@ def test_openai_o_series_max_retries_0(mock_get_openai_client): mock_get_openai_client.assert_called_once() assert mock_get_openai_client.call_args.kwargs["max_retries"] == 0 + + +@pytest.mark.asyncio +async def test_azure_o1_series_response_format_extra_params(): + """ + Tool calling should work for all azure o_series models. + """ + litellm._turn_on_debug() + + from openai import AsyncAzureOpenAI + + litellm.set_verbose = True + + client = AsyncAzureOpenAI( + api_key="fake-api-key", + base_url="https://openai-prod-test.openai.azure.com/openai/deployments/o1/chat/completions?api-version=2025-01-01-preview", + api_version="2025-01-01-preview" + ) + + tools = [{'type': 'function', 'function': {'name': 'get_current_time', 'description': 'Get the current time in a given location.', 'parameters': {'type': 'object', 'properties': {'location': {'type': 'string', 'description': 'The city name, e.g. San Francisco'}}, 'required': ['location']}}}] + response_format = {'type': 'json_object'} + tool_choice = "auto" + with patch.object( + client.chat.completions.with_raw_response, "create" + ) as mock_client: + try: + await litellm.acompletion( + client=client, + model="azure/o_series/