mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
test(e2e): cover responses API gaps on gemini, azure, compact and context management (#45248)
* test(e2e): cover responses API gaps on gemini, azure, compact and context management * test(e2e): assert streamed usage cost and split anthropic strict schema test * test(e2e): assert compaction items and final stream event in responses e2e --------- Co-authored-by: yuneng <yuneng@berri.ai>
This commit is contained in:
parent
d6e453f74b
commit
da2bb5a6b1
3 changed files with 515 additions and 49 deletions
17
tests/e2e/llm_translation/responses_helpers.py
Normal file
17
tests/e2e/llm_translation/responses_helpers.py
Normal file
|
|
@ -0,0 +1,17 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
from models import LiteLLMParamsBody
|
||||
|
||||
AZURE_OPENAI_BACKEND: Final = "azure/gpt-5.4-nano"
|
||||
AZURE_OPENAI_API_VERSION: Final = "v1"
|
||||
|
||||
|
||||
def azure_openai_params(api_version: str = AZURE_OPENAI_API_VERSION) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(
|
||||
model=AZURE_OPENAI_BACKEND,
|
||||
api_base="os.environ/AZURE_API_BASE",
|
||||
api_key="os.environ/AZURE_API_KEY",
|
||||
api_version=api_version,
|
||||
)
|
||||
|
|
@ -42,6 +42,7 @@ from provider_edge import LiveEdge, start_provider_edge
|
|||
from provider_edge_bedrock import bedrock_signer
|
||||
from proxy_client import ProxyClient
|
||||
from pydantic import BaseModel, TypeAdapter
|
||||
from responses_helpers import AZURE_OPENAI_BACKEND, azure_openai_params
|
||||
from sdk_clients import NO_PROXY_CACHE, SdkClients
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
|
@ -58,8 +59,8 @@ OPENAI_VISION_BACKEND: Final = "openai/gpt-4o"
|
|||
ANTHROPIC_BACKEND: Final = "anthropic/claude-haiku-4-5"
|
||||
BEDROCK_CONVERSE_BACKEND: Final = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
VERTEX_BACKEND: Final = "vertex_ai/gemini-2.5-flash"
|
||||
AZURE_OPENAI_BACKEND: Final = "azure/gpt-5.4-nano"
|
||||
AZURE_OPENAI_API_VERSION: Final = "v1"
|
||||
GEMINI_BACKEND: Final = "gemini/gemini-2.5-flash"
|
||||
OPENAI_RESPONSES_BACKEND: Final = "openai/gpt-5.5"
|
||||
INSTRUCTIONS = "You are a helpful assistant"
|
||||
CAT_IMAGE_URL = "https://upload.wikimedia.org/wikipedia/commons/3/3a/Cat03.jpg"
|
||||
BEDROCK_EDGE_REGION: Final = "us-east-1"
|
||||
|
|
@ -102,11 +103,36 @@ WEATHER_TOOL: FunctionToolParam = {
|
|||
"strict": False,
|
||||
}
|
||||
|
||||
LOCATIONS_TOOL: Final[FunctionToolParam] = {
|
||||
"type": "function",
|
||||
"name": "get_locations",
|
||||
"description": "Return locations that need weather information",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"locations": {"type": "array", "items": {"type": "string"}}},
|
||||
"required": ["locations"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
"strict": True,
|
||||
}
|
||||
|
||||
|
||||
class LocationsArguments(BaseModel):
|
||||
locations: list[str]
|
||||
|
||||
|
||||
class ResponseUsageCost(BaseModel):
|
||||
cost: float | None = None
|
||||
|
||||
|
||||
def _openai_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=OPENAI_MINI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
|
||||
|
||||
|
||||
def _openai_responses_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=OPENAI_RESPONSES_BACKEND, api_key="os.environ/OPENAI_API_KEY")
|
||||
|
||||
|
||||
def _anthropic_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=ANTHROPIC_BACKEND, api_key="os.environ/ANTHROPIC_API_KEY")
|
||||
|
||||
|
|
@ -128,13 +154,8 @@ def _vertex_params() -> LiteLLMParamsBody:
|
|||
)
|
||||
|
||||
|
||||
def _azure_openai_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(
|
||||
model=AZURE_OPENAI_BACKEND,
|
||||
api_base="os.environ/AZURE_API_BASE",
|
||||
api_key="os.environ/AZURE_API_KEY",
|
||||
api_version=AZURE_OPENAI_API_VERSION,
|
||||
)
|
||||
def _gemini_params() -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=GEMINI_BACKEND, api_key="os.environ/GEMINI_API_KEY")
|
||||
|
||||
|
||||
def _register(
|
||||
|
|
@ -197,22 +218,147 @@ class TestResponses:
|
|||
def test_responses_streaming_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model = _register(proxy, resources, _openai_params())
|
||||
client = sdk.openai(resources.key())
|
||||
model: Final = _register(proxy, resources, _openai_params())
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
stream = client.responses.create(
|
||||
stream: Final = client.responses.create(
|
||||
model=model,
|
||||
input="reply with one word",
|
||||
instructions=INSTRUCTIONS,
|
||||
stream=True,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
events = tuple(stream)
|
||||
events: Final = tuple(stream)
|
||||
assert events, "responses stream returned no events"
|
||||
deltas = tuple(event.delta for event in events if event.type == "response.output_text.delta")
|
||||
deltas: Final = tuple(event.delta for event in events if event.type == "response.output_text.delta")
|
||||
assert any(delta for delta in deltas), "responses stream returned no text deltas"
|
||||
assert events[-1].type == "response.completed", (
|
||||
f"responses stream did not terminate with response.completed: {events[-1].type}"
|
||||
completed: Final = events[-1]
|
||||
assert isinstance(completed, ResponseCompletedEvent), (
|
||||
f"responses stream did not terminate with response.completed: {completed.type}"
|
||||
)
|
||||
usage: Final = completed.response.usage
|
||||
assert usage is not None, f"response.completed had no usage: {completed.response!r}"
|
||||
assert usage.input_tokens > 0, f"response.completed had no input tokens: {usage!r}"
|
||||
assert usage.output_tokens > 0, f"response.completed had no output tokens: {usage!r}"
|
||||
assert usage.total_tokens == usage.input_tokens + usage.output_tokens, (
|
||||
f"response.completed token totals were inconsistent: {usage!r}"
|
||||
)
|
||||
usage_cost: Final = TypeAdapter(ResponseUsageCost).validate_python(
|
||||
cast(object, usage.model_extra if usage.model_extra is not None else {})
|
||||
)
|
||||
assert usage_cost.cost is not None, f"response.completed usage had no cost: {usage.model_extra!r}"
|
||||
assert usage_cost.cost > 0, f"response.completed cost was not positive: {usage_cost.cost}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_gemini_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, _gemini_params(), prefix="e2e-responses-gemini")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model, input="reply with one word", instructions=INSTRUCTIONS, extra_body=NO_PROXY_CACHE
|
||||
)
|
||||
assert response.output_text.strip(), f"/responses over gemini returned no output text: {response.output!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_gemini_streaming_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, _gemini_params(), prefix="e2e-responses-gemini")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
stream: Final = client.responses.create(
|
||||
model=model,
|
||||
input="reply with one word",
|
||||
instructions=INSTRUCTIONS,
|
||||
stream=True,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
events: Final = tuple(stream)
|
||||
deltas: Final = tuple(event.delta for event in events if event.type == "response.output_text.delta")
|
||||
assert any(deltas), "responses stream over gemini returned no text deltas"
|
||||
assert isinstance(events[-1], ResponseCompletedEvent), (
|
||||
f"responses stream over gemini did not end with response.completed: {events[-1].type}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.GEMINI,),
|
||||
models=(GEMINI_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_gemini_replays_legacy_function_call_output(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, _gemini_params(), prefix="e2e-responses-gemini-tool")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
function_call_id: Final = f"fc_{unique_marker()}"
|
||||
input_items: Final[ResponseInputParam] = [
|
||||
{
|
||||
"type": "message",
|
||||
"role": "user",
|
||||
"content": "What is the temperature in Paris today?",
|
||||
},
|
||||
{
|
||||
"type": "function_call",
|
||||
"arguments": '{"location": "Paris, France"}',
|
||||
"call_id": function_call_id,
|
||||
"name": "get_temperature",
|
||||
"id": function_call_id,
|
||||
"status": "completed",
|
||||
},
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": function_call_id,
|
||||
"output": "Temperature is exactly 31 Celsius.",
|
||||
},
|
||||
]
|
||||
tools: Final[tuple[FunctionToolParam, ...]] = (
|
||||
{
|
||||
"type": "function",
|
||||
"name": "get_temperature",
|
||||
"description": "Get the current temperature for a location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"location": {"type": "string"}},
|
||||
"required": ["location"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
"strict": False,
|
||||
},
|
||||
)
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model,
|
||||
input=input_items,
|
||||
tools=tools,
|
||||
store=False,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert response.status == "completed", f"legacy tool replay was not completed: {response.status}"
|
||||
assert "31" in response.output_text, (
|
||||
f"legacy tool result was missing from output text: {response.output_text!r}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.openai.basic.nonstream.cost_logged")
|
||||
|
|
@ -361,6 +507,94 @@ class TestResponses:
|
|||
)
|
||||
_assert_weather_call(response)
|
||||
|
||||
@pytest.mark.covers("llm.responses.anthropic.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_anthropic_strict_array_schema_tool_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, _anthropic_params())
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model,
|
||||
input="Find the weather locations for Tokyo and Paris using get_locations.",
|
||||
instructions=INSTRUCTIONS,
|
||||
tools=[LOCATIONS_TOOL],
|
||||
tool_choice="required",
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
function_call: Final = next(
|
||||
(call for call in _function_calls(response) if call.name == "get_locations"),
|
||||
None,
|
||||
)
|
||||
assert function_call is not None, f"response had no get_locations call: {response.output!r}"
|
||||
arguments: Final = LocationsArguments.model_validate_json(function_call.arguments)
|
||||
assert arguments.locations, f"get_locations returned no locations: {function_call.arguments}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.anthropic.multi_turn.nonstream.works")
|
||||
@pytest.mark.skip(
|
||||
reason="stage red: product gap, Anthropic previous_response_id continuation sends invalid unmatched tool_use history"
|
||||
)
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.ANTHROPIC,),
|
||||
models=(ANTHROPIC_BACKEND,),
|
||||
capabilities=(Capability.FUNCTION_CALLING,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_anthropic_tool_output_continues_with_previous_response_id(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, _anthropic_params())
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
first: Final = client.responses.create(
|
||||
model=model,
|
||||
input="Find the weather locations for Tokyo and Paris using get_locations.",
|
||||
instructions=INSTRUCTIONS,
|
||||
tools=[LOCATIONS_TOOL],
|
||||
tool_choice="required",
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
function_call: Final = next(
|
||||
(call for call in _function_calls(first) if call.name == "get_locations"),
|
||||
None,
|
||||
)
|
||||
assert function_call is not None, f"response had no get_locations call: {first.output!r}"
|
||||
assert function_call.call_id, f"get_locations call had no call_id: {function_call!r}"
|
||||
arguments: Final = LocationsArguments.model_validate_json(function_call.arguments)
|
||||
assert arguments.locations, f"get_locations call had no locations: {function_call.arguments}"
|
||||
|
||||
tool_result: Final = "Distinctive forecast: 47 degrees Celsius"
|
||||
follow_up_input: Final[ResponseInputParam] = [
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": function_call.call_id,
|
||||
"output": tool_result,
|
||||
}
|
||||
]
|
||||
second: Final = client.responses.create(
|
||||
model=model,
|
||||
previous_response_id=first.id,
|
||||
input=follow_up_input,
|
||||
instructions=INSTRUCTIONS,
|
||||
tools=[LOCATIONS_TOOL],
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert tool_result in second.output_text, f"follow-up omitted tool result: {second.output_text!r}"
|
||||
|
||||
@pytest.mark.covers("llm.responses.bedrock_converse.basic.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
|
|
@ -469,7 +703,7 @@ class TestResponses:
|
|||
def test_responses_azure_openai_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model = _register(proxy, resources, _azure_openai_params(), prefix="e2e-responses-azure-openai")
|
||||
model = _register(proxy, resources, azure_openai_params(), prefix="e2e-responses-azure-openai")
|
||||
client = sdk.openai(resources.key())
|
||||
|
||||
response = client.responses.create(
|
||||
|
|
@ -479,6 +713,150 @@ class TestResponses:
|
|||
f"/responses over azure openai returned no output text: {response.output!r}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
mode=Mode.STREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_azure_openai_streaming_returns_completion(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(proxy, resources, azure_openai_params(), prefix="e2e-responses-azure-stream")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
stream: Final = client.responses.create(
|
||||
model=model,
|
||||
input="reply with one word",
|
||||
instructions=INSTRUCTIONS,
|
||||
stream=True,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
events: Final = tuple(stream)
|
||||
deltas: Final = tuple(event.delta for event in events if event.type == "response.output_text.delta")
|
||||
assert any(deltas), "responses stream over azure openai returned no text deltas"
|
||||
assert isinstance(events[-1], ResponseCompletedEvent), (
|
||||
f"responses stream over azure openai did not end with response.completed: {events[-1].type}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.AZURE,),
|
||||
models=(AZURE_OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_azure_openai_preview_api_version_accepts_truncation(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(
|
||||
proxy,
|
||||
resources,
|
||||
azure_openai_params(api_version="preview"),
|
||||
prefix="e2e-responses-azure-preview",
|
||||
)
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model,
|
||||
input="reply with one word",
|
||||
instructions=INSTRUCTIONS,
|
||||
truncation="auto",
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert response.output_text.strip(), (
|
||||
f"/responses over azure openai preview returned no output text: {response.output!r}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_RESPONSES_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_compact_returns_compacted_conversation(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(
|
||||
proxy,
|
||||
resources,
|
||||
_openai_responses_params(),
|
||||
prefix="e2e-responses-compact",
|
||||
)
|
||||
client: Final = sdk.openai(resources.key())
|
||||
conversation: Final[ResponseInputParam] = [
|
||||
{"role": "user", "content": "Remember that my favorite color is blue."},
|
||||
{"role": "assistant", "content": "I will remember that your favorite color is blue."},
|
||||
]
|
||||
|
||||
compacted: Final = client.responses.compact(
|
||||
model=model,
|
||||
input=conversation,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert compacted.id, f"/responses/compact returned no id: {compacted!r}"
|
||||
assert any(item.type == "compaction" for item in compacted.output), (
|
||||
f"/responses/compact returned no compaction item: {compacted.output!r}"
|
||||
)
|
||||
compacted_input: Final[ResponseInputParam] = TypeAdapter(ResponseInputParam).validate_python(
|
||||
[item.model_dump(exclude_none=True) for item in compacted.output]
|
||||
+ [{"role": "user", "content": "What is my favorite color?"}]
|
||||
)
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model,
|
||||
input=compacted_input,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert "blue" in response.output_text.lower(), (
|
||||
f"compacted conversation did not retain the favorite color: {response.output_text!r}"
|
||||
)
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_RESPONSES_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
)
|
||||
def test_responses_context_management_compacts_server_side(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model: Final = _register(
|
||||
proxy,
|
||||
resources,
|
||||
_openai_responses_params(),
|
||||
prefix="e2e-responses-context-compaction",
|
||||
)
|
||||
client: Final = sdk.openai(resources.key())
|
||||
filler: Final = "The archive record has a blue marker beside every stored entry. " * 350
|
||||
conversation: Final[ResponseInputParam] = [
|
||||
{"role": "user", "content": filler},
|
||||
{"role": "assistant", "content": "I have read the archive and retained its details."},
|
||||
{"role": "user", "content": "Reply with one word to verify server-side compaction."},
|
||||
]
|
||||
|
||||
response: Final = client.responses.create(
|
||||
model=model,
|
||||
input=conversation,
|
||||
context_management=[{"type": "compaction", "compact_threshold": 1000}],
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
assert response.status == "completed", f"context management did not complete: {response.status}"
|
||||
assert any(item.type == "compaction" for item in response.output), (
|
||||
f"context management returned no compaction item: {response.output!r}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("llm.responses.azure_openai.tool_use.nonstream.works")
|
||||
@meta(
|
||||
Subject(
|
||||
|
|
@ -493,7 +871,7 @@ class TestResponses:
|
|||
def test_responses_azure_openai_returns_function_call(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
) -> None:
|
||||
model = _register(proxy, resources, _azure_openai_params(), prefix="e2e-responses-azure-openai-tool")
|
||||
model = _register(proxy, resources, azure_openai_params(), prefix="e2e-responses-azure-openai-tool")
|
||||
client = sdk.openai(resources.key())
|
||||
|
||||
response = client.responses.create(
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ Creates a stored response, retrieves it by id, and pins invalid-id error handlin
|
|||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Final
|
||||
from typing import Final, Literal
|
||||
|
||||
import openai
|
||||
import pytest
|
||||
|
|
@ -15,7 +15,9 @@ from e2e_http import NoBody, Success, UnknownApiError, unwrap
|
|||
from e2e_metadata import Domain, Mode, Provider, Route, Subject, meta
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from responses_helpers import AZURE_OPENAI_BACKEND, azure_openai_params
|
||||
from openai.types.responses import (
|
||||
ResponseCompletedEvent,
|
||||
ResponseCreatedEvent,
|
||||
ResponseInputMessageItem,
|
||||
ResponseInputText,
|
||||
|
|
@ -138,12 +140,42 @@ class TestResponsesRetrieve:
|
|||
|
||||
|
||||
def _register_openai(proxy: ProxyClient, resources: ResourceManager, prefix: str) -> str:
|
||||
model = f"{prefix}-{unique_marker()}"
|
||||
model_id = proxy.create_model(model, LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY"))
|
||||
return _register_response_deployment(proxy, resources, "openai", prefix)
|
||||
|
||||
|
||||
def _register_response_deployment(
|
||||
proxy: ProxyClient,
|
||||
resources: ResourceManager,
|
||||
deployment: Literal["openai", "azure"],
|
||||
prefix: str,
|
||||
) -> str:
|
||||
model: Final = f"{prefix}-{unique_marker()}"
|
||||
params: Final = (
|
||||
azure_openai_params()
|
||||
if deployment == "azure"
|
||||
else LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY")
|
||||
)
|
||||
model_id: Final = proxy.create_model(model, params)
|
||||
resources.defer(lambda: proxy.delete_model(model_id))
|
||||
return model
|
||||
|
||||
|
||||
def _deployment_param(deployment: Literal["openai", "azure"], provider: Provider, backend: str, mode: Mode) -> object:
|
||||
return pytest.param(
|
||||
deployment,
|
||||
id=deployment,
|
||||
marks=meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(provider,),
|
||||
models=(backend,),
|
||||
mode=mode,
|
||||
)
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _input_texts(item: object) -> tuple[str, ...]:
|
||||
if not isinstance(item, ResponseInputMessageItem):
|
||||
return ()
|
||||
|
|
@ -176,57 +208,96 @@ class TestStoredResponseLifecycle:
|
|||
texts = tuple(text for item in items for text in _input_texts(item))
|
||||
assert any(marker in text for text in texts), f"input_items did not list the stored prompt: {items!r}"
|
||||
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"deployment",
|
||||
[
|
||||
_deployment_param("openai", Provider.OPENAI, OPENAI_BACKEND, Mode.NONSTREAM),
|
||||
_deployment_param("azure", Provider.AZURE, AZURE_OPENAI_BACKEND, Mode.NONSTREAM),
|
||||
],
|
||||
)
|
||||
def test_deleted_response_is_no_longer_retrievable(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
resources: ResourceManager,
|
||||
sdk: SdkClients,
|
||||
deployment: Literal["openai", "azure"],
|
||||
) -> None:
|
||||
model = _register_openai(proxy, resources, "e2e-resp-delete")
|
||||
client = sdk.openai(resources.key())
|
||||
model: Final = _register_response_deployment(proxy, resources, deployment, "e2e-resp-delete")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
created = client.responses.create(
|
||||
created: Final = client.responses.create(
|
||||
model=model, input=f"Reply with one word. {unique_marker()}", store=True, extra_body=NO_PROXY_CACHE
|
||||
)
|
||||
retrieved = client.responses.retrieve(created.id)
|
||||
retrieved: Final = client.responses.retrieve(created.id)
|
||||
assert retrieved.status == "completed", f"stored response not retrievable as completed: {retrieved!r}"
|
||||
assert retrieved.output_text == created.output_text, (
|
||||
f"retrieved output changed: created={created.output_text!r}, retrieved={retrieved.output_text!r}"
|
||||
)
|
||||
|
||||
client.responses.delete(created.id)
|
||||
|
||||
with pytest.raises(openai.APIStatusError) as gone:
|
||||
client.responses.retrieve(created.id)
|
||||
gone: Final = pytest.raises(openai.APIStatusError, client.responses.retrieve, created.id)
|
||||
assert 400 <= gone.value.status_code < 500, f"retrieve after delete expected a 4xx: {gone.value!r}"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"deployment",
|
||||
[
|
||||
_deployment_param("openai", Provider.OPENAI, OPENAI_BACKEND, Mode.STREAM),
|
||||
_deployment_param("azure", Provider.AZURE, AZURE_OPENAI_BACKEND, Mode.STREAM),
|
||||
],
|
||||
)
|
||||
def test_streamed_response_can_be_deleted(
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
resources: ResourceManager,
|
||||
sdk: SdkClients,
|
||||
deployment: Literal["openai", "azure"],
|
||||
) -> None:
|
||||
model: Final = _register_response_deployment(proxy, resources, deployment, "e2e-resp-delete-stream")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
events: Final = tuple(
|
||||
client.responses.create(
|
||||
model=model,
|
||||
input=f"Reply with one word. {unique_marker()}",
|
||||
store=True,
|
||||
stream=True,
|
||||
extra_body=NO_PROXY_CACHE,
|
||||
)
|
||||
)
|
||||
completed: Final = next((event for event in events if isinstance(event, ResponseCompletedEvent)), None)
|
||||
assert completed is not None, f"stream did not complete: {events!r}"
|
||||
response_id: Final = completed.response.id
|
||||
|
||||
client.responses.delete(response_id)
|
||||
gone: Final = pytest.raises(openai.APIStatusError, client.responses.retrieve, response_id)
|
||||
assert 400 <= gone.value.status_code < 500, f"retrieve after streamed delete expected a 4xx: {gone.value!r}"
|
||||
|
||||
|
||||
@pytest.mark.provider_live
|
||||
class TestBackgroundResponseCancel:
|
||||
@meta(
|
||||
Subject(
|
||||
domain=Domain.LLM_TRANSLATION,
|
||||
route=Route.RESPONSES,
|
||||
providers=(Provider.OPENAI,),
|
||||
models=(OPENAI_BACKEND,),
|
||||
mode=Mode.NONSTREAM,
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"deployment",
|
||||
[
|
||||
_deployment_param("openai", Provider.OPENAI, OPENAI_BACKEND, Mode.NONSTREAM),
|
||||
_deployment_param("azure", Provider.AZURE, AZURE_OPENAI_BACKEND, Mode.NONSTREAM),
|
||||
],
|
||||
)
|
||||
def test_cancel_background_response(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients
|
||||
self,
|
||||
proxy: ProxyClient,
|
||||
resources: ResourceManager,
|
||||
sdk: SdkClients,
|
||||
deployment: Literal["openai", "azure"],
|
||||
) -> None:
|
||||
model = _register_openai(proxy, resources, "e2e-resp-cancel")
|
||||
client = sdk.openai(resources.key())
|
||||
model: Final = _register_response_deployment(proxy, resources, deployment, "e2e-resp-cancel")
|
||||
client: Final = sdk.openai(resources.key())
|
||||
|
||||
created = client.responses.create(
|
||||
created: Final = client.responses.create(
|
||||
model=model, input=f"{LONG_TASK} {unique_marker()}", background=True, extra_body=NO_PROXY_CACHE
|
||||
)
|
||||
assert created.status in CANCELLABLE_STATUSES, f"background response was not queued: {created.status}"
|
||||
|
||||
cancelled = client.responses.cancel(created.id)
|
||||
cancelled: Final = client.responses.cancel(created.id)
|
||||
assert cancelled.status == "cancelled", f"cancel did not stop the response: {cancelled.status}"
|
||||
|
||||
@meta(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue