mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Azure OpenAI improvements - o3 native streaming, improved tool call + response format handling (#8292)
* fix(convert_dict_to_response.py): only convert if response is the response_format tool call passed in Fixes https://github.com/BerriAI/litellm/issues/8241 * fix(gpt_transformation.py): makes sure response format / tools conversion doesn't remove previous tool calls * refactor(gpt_transformation.py): refactor out json schema converstion to base config keeps logic consistent across providers * fix(o_series_transformation.py): support o3 mini native streaming Fixes https://github.com/BerriAI/litellm/issues/8274 * fix(gpt_transformation.py): remove unused variables * test: update test
This commit is contained in:
parent
515598114c
commit
443ae55904
7 changed files with 260 additions and 55 deletions
|
|
@ -7,6 +7,7 @@ from typing import Dict, Iterable, List, Literal, Optional, Union
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
from litellm.types.utils import (
|
||||
ChatCompletionDeltaToolCall,
|
||||
ChatCompletionMessageToolCall,
|
||||
|
|
@ -313,6 +314,23 @@ class LiteLLMResponseObjectHandler:
|
|||
return transformed_logprobs
|
||||
|
||||
|
||||
def _should_convert_tool_call_to_json_mode(
|
||||
tool_calls: Optional[List[ChatCompletionMessageToolCall]] = None,
|
||||
convert_tool_call_to_json_mode: Optional[bool] = None,
|
||||
) -> bool:
|
||||
"""
|
||||
Determine if tool calls should be converted to JSON mode
|
||||
"""
|
||||
if (
|
||||
convert_tool_call_to_json_mode
|
||||
and tool_calls is not None
|
||||
and len(tool_calls) == 1
|
||||
and tool_calls[0]["function"]["name"] == RESPONSE_FORMAT_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def convert_to_model_response_object( # noqa: PLR0915
|
||||
response_object: Optional[dict] = None,
|
||||
model_response_object: Optional[
|
||||
|
|
@ -397,10 +415,9 @@ def convert_to_model_response_object( # noqa: PLR0915
|
|||
|
||||
message: Optional[Message] = None
|
||||
finish_reason: Optional[str] = None
|
||||
if (
|
||||
convert_tool_call_to_json_mode
|
||||
and tool_calls is not None
|
||||
and len(tool_calls) == 1
|
||||
if _should_convert_tool_call_to_json_mode(
|
||||
tool_calls=tool_calls,
|
||||
convert_tool_call_to_json_mode=convert_tool_call_to_json_mode,
|
||||
):
|
||||
# to support 'json_schema' logic on older models
|
||||
json_mode_content_str: Optional[str] = tool_calls[0][
|
||||
|
|
|
|||
|
|
@ -11,13 +11,7 @@ from litellm.types.utils import ModelResponse
|
|||
from litellm.utils import supports_response_schema
|
||||
|
||||
from ....exceptions import UnsupportedParamsError
|
||||
from ....types.llms.openai import (
|
||||
AllMessageValues,
|
||||
ChatCompletionToolChoiceFunctionParam,
|
||||
ChatCompletionToolChoiceObjectParam,
|
||||
ChatCompletionToolParam,
|
||||
ChatCompletionToolParamFunctionChunk,
|
||||
)
|
||||
from ....types.llms.openai import AllMessageValues
|
||||
from ...base_llm.chat.transformation import BaseConfig
|
||||
from ..common_utils import AzureOpenAIError
|
||||
|
||||
|
|
@ -174,49 +168,20 @@ class AzureOpenAIConfig(BaseConfig):
|
|||
else:
|
||||
optional_params["tool_choice"] = value
|
||||
elif param == "response_format" and isinstance(value, dict):
|
||||
json_schema: Optional[dict] = None
|
||||
schema_name: str = ""
|
||||
if "response_schema" in value:
|
||||
json_schema = value["response_schema"]
|
||||
schema_name = "json_tool_call"
|
||||
elif "json_schema" in value:
|
||||
json_schema = value["json_schema"]["schema"]
|
||||
schema_name = value["json_schema"]["name"]
|
||||
"""
|
||||
Follow similar approach to anthropic - translate to a single tool call.
|
||||
|
||||
When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode
|
||||
- You usually want to provide a single tool
|
||||
- You should set tool_choice (see Forcing tool use) to instruct the model to explicitly use that tool
|
||||
- Remember that the model will pass the input to the tool, so the name of the tool and description should be from the model’s perspective.
|
||||
"""
|
||||
_is_response_format_supported_model = (
|
||||
self._is_response_format_supported_model(model)
|
||||
)
|
||||
if json_schema is not None and (
|
||||
(api_version_year <= "2024" and api_version_month < "08")
|
||||
or not _is_response_format_supported_model
|
||||
): # azure api version "2024-08-01-preview" onwards supports 'json_schema' only for gpt-4o/3.5 models
|
||||
|
||||
_tool_choice = ChatCompletionToolChoiceObjectParam(
|
||||
type="function",
|
||||
function=ChatCompletionToolChoiceFunctionParam(
|
||||
name=schema_name
|
||||
),
|
||||
)
|
||||
|
||||
_tool = ChatCompletionToolParam(
|
||||
type="function",
|
||||
function=ChatCompletionToolParamFunctionChunk(
|
||||
name=schema_name, parameters=json_schema
|
||||
),
|
||||
)
|
||||
|
||||
optional_params["tools"] = [_tool]
|
||||
optional_params["tool_choice"] = _tool_choice
|
||||
optional_params["json_mode"] = True
|
||||
else:
|
||||
optional_params["response_format"] = value
|
||||
should_convert_response_format_to_tool = (
|
||||
api_version_year <= "2024" and api_version_month < "08"
|
||||
) or not _is_response_format_supported_model
|
||||
optional_params = self._add_response_format_to_tools(
|
||||
optional_params=optional_params,
|
||||
value=value,
|
||||
should_convert_response_format_to_tool=should_convert_response_format_to_tool,
|
||||
)
|
||||
elif param == "tools" and isinstance(value, list):
|
||||
optional_params.setdefault("tools", [])
|
||||
optional_params["tools"].extend(value)
|
||||
elif param in supported_openai_params:
|
||||
optional_params[param] = value
|
||||
|
||||
|
|
|
|||
|
|
@ -35,11 +35,16 @@ class AzureOpenAIO1Config(OpenAIOSeriesConfig):
|
|||
if stream is not True:
|
||||
return False
|
||||
|
||||
if (
|
||||
model and "o3" in model
|
||||
): # o3 models support streaming - https://github.com/BerriAI/litellm/issues/8274
|
||||
return False
|
||||
|
||||
if model is not None:
|
||||
try:
|
||||
model_info = get_model_info(
|
||||
model=model, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
) # allow user to override default with model_info={"supports_native_streaming": true}
|
||||
|
||||
if (
|
||||
model_info.get("supports_native_streaming") is True
|
||||
|
|
|
|||
|
|
@ -19,8 +19,17 @@ import httpx
|
|||
from pydantic import BaseModel
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
ChatCompletionToolChoiceFunctionParam,
|
||||
ChatCompletionToolChoiceObjectParam,
|
||||
ChatCompletionToolParam,
|
||||
ChatCompletionToolParamFunctionChunk,
|
||||
)
|
||||
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
|
||||
from litellm.types.utils import ModelResponse
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
|
|
@ -150,6 +159,53 @@ class BaseConfig(ABC):
|
|||
def get_supported_openai_params(self, model: str) -> list:
|
||||
pass
|
||||
|
||||
def _add_response_format_to_tools(
|
||||
self,
|
||||
optional_params: dict,
|
||||
value: dict,
|
||||
should_convert_response_format_to_tool: bool,
|
||||
) -> dict:
|
||||
"""
|
||||
Follow similar approach to anthropic - translate to a single tool call.
|
||||
|
||||
When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode
|
||||
- You usually want to provide a single tool
|
||||
- You should set tool_choice (see Forcing tool use) to instruct the model to explicitly use that tool
|
||||
- Remember that the model will pass the input to the tool, so the name of the tool and description should be from the model’s perspective.
|
||||
|
||||
Add response format to tools
|
||||
|
||||
This is used to translate response_format to a tool call, for models/APIs that don't support response_format directly.
|
||||
"""
|
||||
json_schema: Optional[dict] = None
|
||||
if "response_schema" in value:
|
||||
json_schema = value["response_schema"]
|
||||
elif "json_schema" in value:
|
||||
json_schema = value["json_schema"]["schema"]
|
||||
|
||||
if json_schema and should_convert_response_format_to_tool:
|
||||
_tool_choice = ChatCompletionToolChoiceObjectParam(
|
||||
type="function",
|
||||
function=ChatCompletionToolChoiceFunctionParam(
|
||||
name=RESPONSE_FORMAT_TOOL_NAME
|
||||
),
|
||||
)
|
||||
|
||||
_tool = ChatCompletionToolParam(
|
||||
type="function",
|
||||
function=ChatCompletionToolParamFunctionChunk(
|
||||
name=RESPONSE_FORMAT_TOOL_NAME, parameters=json_schema
|
||||
),
|
||||
)
|
||||
|
||||
optional_params.setdefault("tools", [])
|
||||
optional_params["tools"].append(_tool)
|
||||
optional_params["tool_choice"] = _tool_choice
|
||||
optional_params["json_mode"] = True
|
||||
else:
|
||||
optional_params["response_format"] = value
|
||||
return optional_params
|
||||
|
||||
@abstractmethod
|
||||
def map_openai_params(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -120,7 +120,7 @@ def test_azure_o3_streaming():
|
|||
) as e: # expect output translation error as mock response doesn't return a json
|
||||
print(e)
|
||||
assert mock_create.call_count == 1
|
||||
assert "stream" not in mock_create.call_args.kwargs
|
||||
assert "stream" in mock_create.call_args.kwargs
|
||||
|
||||
|
||||
def test_azure_o_series_routing():
|
||||
|
|
|
|||
|
|
@ -283,3 +283,156 @@ def test_azure_openai_gpt_4o_naming(monkeypatch):
|
|||
print(mock_post.call_args.kwargs)
|
||||
|
||||
assert "tool_calls" not in mock_post.call_args.kwargs
|
||||
|
||||
|
||||
def test_azure_gpt_4o_with_tool_call_and_response_format():
|
||||
from litellm import completion
|
||||
from typing import Optional
|
||||
from pydantic import BaseModel
|
||||
import litellm
|
||||
|
||||
class InvestigationOutput(BaseModel):
|
||||
alert_explanation: Optional[str] = None
|
||||
investigation: Optional[str] = None
|
||||
conclusions_and_possible_root_causes: Optional[str] = None
|
||||
next_steps: Optional[str] = None
|
||||
related_logs: Optional[str] = None
|
||||
app_or_infra: Optional[str] = None
|
||||
external_links: Optional[str] = None
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_time",
|
||||
"description": "Returns the current date and time",
|
||||
"strict": True,
|
||||
"parameters": {
|
||||
"properties": {
|
||||
"timezone": {
|
||||
"type": "string",
|
||||
"description": "The timezone to get the current time for (e.g., 'UTC', 'America/New_York')",
|
||||
}
|
||||
},
|
||||
"required": ["timezone"],
|
||||
"type": "object",
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
response = litellm.completion(
|
||||
model="azure/gpt-4o",
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are a tool-calling AI assist provided with common devops and IT tools that you can use to troubleshoot problems or answer questions.\nWhenever possible you MUST first use tools to investigate then answer the question.",
|
||||
},
|
||||
{"role": "user", "content": "What is the current date and time in NYC?"},
|
||||
],
|
||||
drop_params=True,
|
||||
temperature=0.00000001,
|
||||
tools=tools,
|
||||
tool_choice="auto",
|
||||
response_format=InvestigationOutput, # commenting this line will cause the output to be correct
|
||||
)
|
||||
|
||||
assert response.choices[0].finish_reason == "tool_calls"
|
||||
|
||||
print(response.to_json())
|
||||
|
||||
|
||||
def test_map_openai_params():
|
||||
"""
|
||||
Ensure response_format does not override tools
|
||||
"""
|
||||
from litellm.llms.azure.chat.gpt_transformation import AzureOpenAIConfig
|
||||
|
||||
azure_openai_config = AzureOpenAIConfig()
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_time",
|
||||
"description": "Returns the current date and time",
|
||||
"strict": True,
|
||||
"parameters": {
|
||||
"properties": {
|
||||
"timezone": {
|
||||
"type": "string",
|
||||
"description": "The timezone to get the current time for (e.g., 'UTC', 'America/New_York')",
|
||||
}
|
||||
},
|
||||
"required": ["timezone"],
|
||||
"type": "object",
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
received_args = {
|
||||
"non_default_params": {
|
||||
"temperature": 1e-08,
|
||||
"response_format": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"schema": {
|
||||
"properties": {
|
||||
"alert_explanation": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "Alert Explanation",
|
||||
},
|
||||
"investigation": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "Investigation",
|
||||
},
|
||||
"conclusions_and_possible_root_causes": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "Conclusions And Possible Root Causes",
|
||||
},
|
||||
"next_steps": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "Next Steps",
|
||||
},
|
||||
"related_logs": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "Related Logs",
|
||||
},
|
||||
"app_or_infra": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "App Or Infra",
|
||||
},
|
||||
"external_links": {
|
||||
"anyOf": [{"type": "string"}, {"type": "null"}],
|
||||
"title": "External Links",
|
||||
},
|
||||
},
|
||||
"title": "InvestigationOutput",
|
||||
"type": "object",
|
||||
"additionalProperties": False,
|
||||
"required": [
|
||||
"alert_explanation",
|
||||
"investigation",
|
||||
"conclusions_and_possible_root_causes",
|
||||
"next_steps",
|
||||
"related_logs",
|
||||
"app_or_infra",
|
||||
"external_links",
|
||||
],
|
||||
},
|
||||
"name": "InvestigationOutput",
|
||||
"strict": True,
|
||||
},
|
||||
},
|
||||
"tools": tools,
|
||||
"tool_choice": "auto",
|
||||
},
|
||||
"optional_params": {},
|
||||
"model": "gpt-4o",
|
||||
"drop_params": True,
|
||||
"api_version": "2024-02-15-preview",
|
||||
}
|
||||
optional_params = azure_openai_config.map_openai_params(**received_args)
|
||||
assert "tools" in optional_params
|
||||
assert len(optional_params["tools"]) > 1
|
||||
|
|
|
|||
|
|
@ -327,12 +327,21 @@ def test_convert_to_model_response_object_json_mode():
|
|||
This test is verifying that when convert_tool_call_to_json_mode is True, a single tool call's arguments are correctly converted into the message content of the response.
|
||||
"""
|
||||
model_response_object = ModelResponse(model="gpt-3.5-turbo")
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
|
||||
response_object = {
|
||||
"choices": [
|
||||
{
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"tool_calls": [{"function": {"arguments": '{"key": "value"}'}}],
|
||||
"tool_calls": [
|
||||
{
|
||||
"function": {
|
||||
"arguments": '{"key": "value"}',
|
||||
"name": RESPONSE_FORMAT_TOOL_NAME,
|
||||
}
|
||||
}
|
||||
],
|
||||
},
|
||||
"finish_reason": None,
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue