From 443ae55904760b73ea7eeb9640722ae47b3e2d96 Mon Sep 17 00:00:00 2001 From: Krish Dholakia Date: Wed, 5 Feb 2025 19:38:58 -0800 Subject: [PATCH] Azure OpenAI improvements - o3 native streaming, improved tool call + response format handling (#8292) * fix(convert_dict_to_response.py): only convert if response is the response_format tool call passed in Fixes https://github.com/BerriAI/litellm/issues/8241 * fix(gpt_transformation.py): makes sure response format / tools conversion doesn't remove previous tool calls * refactor(gpt_transformation.py): refactor out json schema converstion to base config keeps logic consistent across providers * fix(o_series_transformation.py): support o3 mini native streaming Fixes https://github.com/BerriAI/litellm/issues/8274 * fix(gpt_transformation.py): remove unused variables * test: update test --- .../convert_dict_to_response.py | 25 ++- litellm/llms/azure/chat/gpt_transformation.py | 59 ++----- .../azure/chat/o_series_transformation.py | 7 +- litellm/llms/base_llm/chat/transformation.py | 58 ++++++- tests/llm_translation/test_azure_o_series.py | 2 +- tests/llm_translation/test_azure_openai.py | 153 ++++++++++++++++++ .../test_convert_dict_to_chat_completion.py | 11 +- 7 files changed, 260 insertions(+), 55 deletions(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 28d546796db..dacd21f426e 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -7,6 +7,7 @@ from typing import Dict, Iterable, List, Literal, Optional, Union import litellm from litellm._logging import verbose_logger +from litellm.constants import RESPONSE_FORMAT_TOOL_NAME from litellm.types.utils import ( ChatCompletionDeltaToolCall, ChatCompletionMessageToolCall, @@ -313,6 +314,23 @@ class LiteLLMResponseObjectHandler: return transformed_logprobs +def _should_convert_tool_call_to_json_mode( + tool_calls: Optional[List[ChatCompletionMessageToolCall]] = None, + convert_tool_call_to_json_mode: Optional[bool] = None, +) -> bool: + """ + Determine if tool calls should be converted to JSON mode + """ + if ( + convert_tool_call_to_json_mode + and tool_calls is not None + and len(tool_calls) == 1 + and tool_calls[0]["function"]["name"] == RESPONSE_FORMAT_TOOL_NAME + ): + return True + return False + + def convert_to_model_response_object( # noqa: PLR0915 response_object: Optional[dict] = None, model_response_object: Optional[ @@ -397,10 +415,9 @@ def convert_to_model_response_object( # noqa: PLR0915 message: Optional[Message] = None finish_reason: Optional[str] = None - if ( - convert_tool_call_to_json_mode - and tool_calls is not None - and len(tool_calls) == 1 + if _should_convert_tool_call_to_json_mode( + tool_calls=tool_calls, + convert_tool_call_to_json_mode=convert_tool_call_to_json_mode, ): # to support 'json_schema' logic on older models json_mode_content_str: Optional[str] = tool_calls[0][ diff --git a/litellm/llms/azure/chat/gpt_transformation.py b/litellm/llms/azure/chat/gpt_transformation.py index 00e336d69a9..53a7fdd6877 100644 --- a/litellm/llms/azure/chat/gpt_transformation.py +++ b/litellm/llms/azure/chat/gpt_transformation.py @@ -11,13 +11,7 @@ from litellm.types.utils import ModelResponse from litellm.utils import supports_response_schema from ....exceptions import UnsupportedParamsError -from ....types.llms.openai import ( - AllMessageValues, - ChatCompletionToolChoiceFunctionParam, - ChatCompletionToolChoiceObjectParam, - ChatCompletionToolParam, - ChatCompletionToolParamFunctionChunk, -) +from ....types.llms.openai import AllMessageValues from ...base_llm.chat.transformation import BaseConfig from ..common_utils import AzureOpenAIError @@ -174,49 +168,20 @@ class AzureOpenAIConfig(BaseConfig): else: optional_params["tool_choice"] = value elif param == "response_format" and isinstance(value, dict): - json_schema: Optional[dict] = None - schema_name: str = "" - if "response_schema" in value: - json_schema = value["response_schema"] - schema_name = "json_tool_call" - elif "json_schema" in value: - json_schema = value["json_schema"]["schema"] - schema_name = value["json_schema"]["name"] - """ - Follow similar approach to anthropic - translate to a single tool call. - - When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode - - You usually want to provide a single tool - - You should set tool_choice (see Forcing tool use) to instruct the model to explicitly use that tool - - Remember that the model will pass the input to the tool, so the name of the tool and description should be from the model’s perspective. - """ _is_response_format_supported_model = ( self._is_response_format_supported_model(model) ) - if json_schema is not None and ( - (api_version_year <= "2024" and api_version_month < "08") - or not _is_response_format_supported_model - ): # azure api version "2024-08-01-preview" onwards supports 'json_schema' only for gpt-4o/3.5 models - - _tool_choice = ChatCompletionToolChoiceObjectParam( - type="function", - function=ChatCompletionToolChoiceFunctionParam( - name=schema_name - ), - ) - - _tool = ChatCompletionToolParam( - type="function", - function=ChatCompletionToolParamFunctionChunk( - name=schema_name, parameters=json_schema - ), - ) - - optional_params["tools"] = [_tool] - optional_params["tool_choice"] = _tool_choice - optional_params["json_mode"] = True - else: - optional_params["response_format"] = value + should_convert_response_format_to_tool = ( + api_version_year <= "2024" and api_version_month < "08" + ) or not _is_response_format_supported_model + optional_params = self._add_response_format_to_tools( + optional_params=optional_params, + value=value, + should_convert_response_format_to_tool=should_convert_response_format_to_tool, + ) + elif param == "tools" and isinstance(value, list): + optional_params.setdefault("tools", []) + optional_params["tools"].extend(value) elif param in supported_openai_params: optional_params[param] = value diff --git a/litellm/llms/azure/chat/o_series_transformation.py b/litellm/llms/azure/chat/o_series_transformation.py index 2cae4c7cbb1..0ca3a28d23e 100644 --- a/litellm/llms/azure/chat/o_series_transformation.py +++ b/litellm/llms/azure/chat/o_series_transformation.py @@ -35,11 +35,16 @@ class AzureOpenAIO1Config(OpenAIOSeriesConfig): if stream is not True: return False + if ( + model and "o3" in model + ): # o3 models support streaming - https://github.com/BerriAI/litellm/issues/8274 + return False + if model is not None: try: model_info = get_model_info( model=model, custom_llm_provider=custom_llm_provider - ) + ) # allow user to override default with model_info={"supports_native_streaming": true} if ( model_info.get("supports_native_streaming") is True diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index a4c2335ebfa..1004cc90127 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -19,8 +19,17 @@ import httpx from pydantic import BaseModel from litellm._logging import verbose_logger +from litellm.constants import RESPONSE_FORMAT_TOOL_NAME +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionToolChoiceFunctionParam, + ChatCompletionToolChoiceObjectParam, + ChatCompletionToolParam, + ChatCompletionToolParamFunctionChunk, +) + from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler -from litellm.types.llms.openai import AllMessageValues + from litellm.types.utils import ModelResponse from litellm.utils import CustomStreamWrapper @@ -150,6 +159,53 @@ class BaseConfig(ABC): def get_supported_openai_params(self, model: str) -> list: pass + def _add_response_format_to_tools( + self, + optional_params: dict, + value: dict, + should_convert_response_format_to_tool: bool, + ) -> dict: + """ + Follow similar approach to anthropic - translate to a single tool call. + + When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode + - You usually want to provide a single tool + - You should set tool_choice (see Forcing tool use) to instruct the model to explicitly use that tool + - Remember that the model will pass the input to the tool, so the name of the tool and description should be from the model’s perspective. + + Add response format to tools + + This is used to translate response_format to a tool call, for models/APIs that don't support response_format directly. + """ + json_schema: Optional[dict] = None + if "response_schema" in value: + json_schema = value["response_schema"] + elif "json_schema" in value: + json_schema = value["json_schema"]["schema"] + + if json_schema and should_convert_response_format_to_tool: + _tool_choice = ChatCompletionToolChoiceObjectParam( + type="function", + function=ChatCompletionToolChoiceFunctionParam( + name=RESPONSE_FORMAT_TOOL_NAME + ), + ) + + _tool = ChatCompletionToolParam( + type="function", + function=ChatCompletionToolParamFunctionChunk( + name=RESPONSE_FORMAT_TOOL_NAME, parameters=json_schema + ), + ) + + optional_params.setdefault("tools", []) + optional_params["tools"].append(_tool) + optional_params["tool_choice"] = _tool_choice + optional_params["json_mode"] = True + else: + optional_params["response_format"] = value + return optional_params + @abstractmethod def map_openai_params( self, diff --git a/tests/llm_translation/test_azure_o_series.py b/tests/llm_translation/test_azure_o_series.py index 52a5fbd96df..21c4f4e04af 100644 --- a/tests/llm_translation/test_azure_o_series.py +++ b/tests/llm_translation/test_azure_o_series.py @@ -120,7 +120,7 @@ def test_azure_o3_streaming(): ) as e: # expect output translation error as mock response doesn't return a json print(e) assert mock_create.call_count == 1 - assert "stream" not in mock_create.call_args.kwargs + assert "stream" in mock_create.call_args.kwargs def test_azure_o_series_routing(): diff --git a/tests/llm_translation/test_azure_openai.py b/tests/llm_translation/test_azure_openai.py index 0e9be710e2a..6306f58cfbd 100644 --- a/tests/llm_translation/test_azure_openai.py +++ b/tests/llm_translation/test_azure_openai.py @@ -283,3 +283,156 @@ def test_azure_openai_gpt_4o_naming(monkeypatch): print(mock_post.call_args.kwargs) assert "tool_calls" not in mock_post.call_args.kwargs + + +def test_azure_gpt_4o_with_tool_call_and_response_format(): + from litellm import completion + from typing import Optional + from pydantic import BaseModel + import litellm + + class InvestigationOutput(BaseModel): + alert_explanation: Optional[str] = None + investigation: Optional[str] = None + conclusions_and_possible_root_causes: Optional[str] = None + next_steps: Optional[str] = None + related_logs: Optional[str] = None + app_or_infra: Optional[str] = None + external_links: Optional[str] = None + + tools = [ + { + "type": "function", + "function": { + "name": "get_current_time", + "description": "Returns the current date and time", + "strict": True, + "parameters": { + "properties": { + "timezone": { + "type": "string", + "description": "The timezone to get the current time for (e.g., 'UTC', 'America/New_York')", + } + }, + "required": ["timezone"], + "type": "object", + "additionalProperties": False, + }, + }, + } + ] + + response = litellm.completion( + model="azure/gpt-4o", + messages=[ + { + "role": "system", + "content": "You are a tool-calling AI assist provided with common devops and IT tools that you can use to troubleshoot problems or answer questions.\nWhenever possible you MUST first use tools to investigate then answer the question.", + }, + {"role": "user", "content": "What is the current date and time in NYC?"}, + ], + drop_params=True, + temperature=0.00000001, + tools=tools, + tool_choice="auto", + response_format=InvestigationOutput, # commenting this line will cause the output to be correct + ) + + assert response.choices[0].finish_reason == "tool_calls" + + print(response.to_json()) + + +def test_map_openai_params(): + """ + Ensure response_format does not override tools + """ + from litellm.llms.azure.chat.gpt_transformation import AzureOpenAIConfig + + azure_openai_config = AzureOpenAIConfig() + tools = [ + { + "type": "function", + "function": { + "name": "get_current_time", + "description": "Returns the current date and time", + "strict": True, + "parameters": { + "properties": { + "timezone": { + "type": "string", + "description": "The timezone to get the current time for (e.g., 'UTC', 'America/New_York')", + } + }, + "required": ["timezone"], + "type": "object", + "additionalProperties": False, + }, + }, + } + ] + received_args = { + "non_default_params": { + "temperature": 1e-08, + "response_format": { + "type": "json_schema", + "json_schema": { + "schema": { + "properties": { + "alert_explanation": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "Alert Explanation", + }, + "investigation": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "Investigation", + }, + "conclusions_and_possible_root_causes": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "Conclusions And Possible Root Causes", + }, + "next_steps": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "Next Steps", + }, + "related_logs": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "Related Logs", + }, + "app_or_infra": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "App Or Infra", + }, + "external_links": { + "anyOf": [{"type": "string"}, {"type": "null"}], + "title": "External Links", + }, + }, + "title": "InvestigationOutput", + "type": "object", + "additionalProperties": False, + "required": [ + "alert_explanation", + "investigation", + "conclusions_and_possible_root_causes", + "next_steps", + "related_logs", + "app_or_infra", + "external_links", + ], + }, + "name": "InvestigationOutput", + "strict": True, + }, + }, + "tools": tools, + "tool_choice": "auto", + }, + "optional_params": {}, + "model": "gpt-4o", + "drop_params": True, + "api_version": "2024-02-15-preview", + } + optional_params = azure_openai_config.map_openai_params(**received_args) + assert "tools" in optional_params + assert len(optional_params["tools"]) > 1 diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index f1b6250bb5b..3b0bd6ca821 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -327,12 +327,21 @@ def test_convert_to_model_response_object_json_mode(): This test is verifying that when convert_tool_call_to_json_mode is True, a single tool call's arguments are correctly converted into the message content of the response. """ model_response_object = ModelResponse(model="gpt-3.5-turbo") + from litellm.constants import RESPONSE_FORMAT_TOOL_NAME + response_object = { "choices": [ { "message": { "role": "assistant", - "tool_calls": [{"function": {"arguments": '{"key": "value"}'}}], + "tool_calls": [ + { + "function": { + "arguments": '{"key": "value"}', + "name": RESPONSE_FORMAT_TOOL_NAME, + } + } + ], }, "finish_reason": None, }