From c168199e33af584e33198714fd4bff716c866ae0 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Wed, 30 Sep 2026 19:19:59 -0700 Subject: [PATCH] test(ci): repair stale tests and move retired OpenAI text-completion fixtures (#43958) * test(ci): repair stale request fakes, spend-log golden, auto-router labels, and Interactions spec lookups Request fakes now carry the scope a real Starlette request has, the GCS pub/sub spend-log golden gains the agent identity keys from #43722, the auto-router session tests follow the baseline_models contract from #43348, and the Interactions spec checks resolve the create body and resource paths from the live spec instead of hardcoded names * test(ci): move retired OpenAI text-completion fixtures to live vehicles OpenAI still serves native /v1/completions on the gpt-5.4 family, so the single-prompt cases move to text-completion-openai/gpt-5.4-nano. Multi-prompt batches and echo with logprobs now 500 on every OpenAI model, so those cases keep the same text-completion-openai transport pointed at Fireworks, which documents both. The optional-params test asserts the request body actually sent instead of a success callback whose assertions were swallowed * test(ci): use a serverless Fireworks model for the text-completion batch and echo cases gpt-oss-20b is on-demand only on Fireworks, so the CI key got 404 model not deployed; glm-5p3-flash is listed as serverless * test(ci): skip the ROI calculator repository listing in the security route sweep GET /roi-calculator/repositories (#43669) lists repositories from the configured GitHub API, api.github.com by default, so the S2 sweep's GET of every route made the owned proxy reach an external host and failed the egress check in 31 integration-security tests. It joins /get/latest_release_info in the deny list --- tests/integration/security/_sweeps.py | 1 + tests/local_testing/test_completion.py | 90 ++++++------------- .../local_testing/test_http_parsing_utils.py | 78 +++++++--------- tests/local_testing/test_text_completion.py | 31 ++++--- .../gcs_pub_sub_body/spend_logs_payload.json | 5 +- .../proxy/batches_endpoints/test_endpoints.py | 1 + .../interactions/test_openapi_compliance.py | 44 +++++---- tests/unit/models/test_models.py | 15 ++-- 8 files changed, 122 insertions(+), 143 deletions(-) diff --git a/tests/integration/security/_sweeps.py b/tests/integration/security/_sweeps.py index 617bf4c9bae..f97a0a7fcc6 100644 --- a/tests/integration/security/_sweeps.py +++ b/tests/integration/security/_sweeps.py @@ -105,6 +105,7 @@ ROUTE_DENY_LIST: Final = MappingProxyType( "/plugin-proxy/{plugin_name}/{path:path}": "reverse proxy to a plugin process", "/openai_passthrough/{endpoint:path}": "forwards to a provider, not a proxy read", "/get/latest_release_info": "fetches the latest release from api.github.com", + "/roi-calculator/repositories": "lists repositories from the configured GitHub API, api.github.com by default", } ) diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index c6dd78c73b4..2d8983c2fc8 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -11,7 +11,9 @@ import io from unittest.mock import AsyncMock, MagicMock, patch +import httpx import pytest +from openai import OpenAI import litellm from litellm import RateLimitError, Timeout, completion, completion_cost, embedding @@ -1580,7 +1582,7 @@ def test_completion_openai_pydantic(model, api_version): def test_completion_text_openai(): try: # litellm.set_verbose =True - response = completion(model="gpt-3.5-turbo-instruct", messages=messages) + response = completion(model="text-completion-openai/gpt-5.4-nano", messages=messages) print(response["choices"][0]["message"]["content"]) except Exception as e: print(e) @@ -1592,7 +1594,7 @@ async def test_completion_text_openai_async(): try: # litellm.set_verbose =True response = await litellm.acompletion( - model="gpt-3.5-turbo-instruct", messages=messages + model="text-completion-openai/gpt-5.4-nano", messages=messages ) print(response["choices"][0]["message"]["content"]) except Exception as e: @@ -1600,67 +1602,33 @@ async def test_completion_text_openai_async(): pytest.fail(f"Error occurred: {e}") -def custom_callback( - kwargs, # kwargs to completion - completion_response, # response from completion - start_time, - end_time, # start/end time -): - # Your custom code here - try: - print("LITELLM: in custom callback function") - print("\nkwargs\n", kwargs) - model = kwargs["model"] - messages = kwargs["messages"] - user = kwargs.get("user") - - ################################################# - - print( - f""" - Model: {model}, - Messages: {messages}, - User: {user}, - Seed: {kwargs["seed"]}, - temperature: {kwargs["temperature"]}, - """ - ) - - assert kwargs["user"] == "ishaans app" - assert kwargs["model"] == "gpt-3.5-turbo-1106" - assert kwargs["seed"] == 12 - assert kwargs["temperature"] == 0.5 - except Exception as e: - pytest.fail(f"Error occurred: {e}") - - def test_completion_openai_with_optional_params(): # [Proxy PROD TEST] WARNING: DO NOT DELETE THIS TEST - # assert that `user` gets passed to the completion call - # Note: This tests that we actually send the optional params to the completion call - # We use custom callbacks to test this - try: - litellm.set_verbose = True - litellm.success_callback = [custom_callback] - response = completion( - model="gpt-3.5-turbo-1106", - messages=[ - {"role": "user", "content": "respond in valid, json - what is the day"} - ], - temperature=0.5, - top_p=0.1, - seed=12, - response_format={"type": "json_object"}, - logit_bias=None, - user="ishaans app", - ) - # Add any assertions here to check the response + on_request = MagicMock() + client = OpenAI(http_client=httpx.Client(event_hooks={"request": [on_request]})) + response = completion( + model="gpt-6-luna", + reasoning_effort="none", + messages=[{"role": "user", "content": "respond in valid, json - what is the day"}], + temperature=0.5, + top_p=0.1, + seed=12, + response_format={"type": "json_object"}, + logit_bias=None, + user="ishaans app", + client=client, + ) - print(response) - litellm.success_callback = [] # unset callbacks - - except Exception as e: - pytest.fail(f"Error occurred: {e}") + assert response.choices[0].message.content + on_request.assert_called_once() + sent = json.loads(on_request.call_args.args[0].content) + assert sent["model"] == "gpt-6-luna" + assert sent["user"] == "ishaans app" + assert sent["seed"] == 12 + assert sent["temperature"] == 0.5 + assert sent["top_p"] == 0.1 + assert sent["response_format"] == {"type": "json_object"} + assert "logit_bias" not in sent # test_completion_openai_with_optional_params() @@ -4008,7 +3976,7 @@ def test_deepseek_reasoning_content_completion(): def test_qwen_text_completion(): # litellm._turn_on_debug() resp = litellm.completion( - model="gpt-3.5-turbo-instruct", + model="text-completion-openai/gpt-5.4-nano", messages=[{"content": "hello", "role": "user"}], stream=False, logprobs=1, diff --git a/tests/local_testing/test_http_parsing_utils.py b/tests/local_testing/test_http_parsing_utils.py index db282d6d4be..59efe883c5d 100644 --- a/tests/local_testing/test_http_parsing_utils.py +++ b/tests/local_testing/test_http_parsing_utils.py @@ -1,75 +1,61 @@ +from collections.abc import Awaitable, Callable + import pytest from fastapi import Request -from fastapi.testclient import TestClient -from starlette.datastructures import Headers -from starlette.requests import HTTPConnection +from starlette.types import Message - -from litellm.proxy.common_utils.http_parsing_utils import _read_request_body from litellm.proxy._types import ProxyException +from litellm.proxy.common_utils.http_parsing_utils import _read_request_body + + +def _request(receive: Callable[[], Awaitable[Message]]) -> Request: + return Request( + { + "type": "http", + "method": "POST", + "path": "/v1/chat/completions", + "headers": [(b"content-type", b"application/json")], + }, + receive, + ) + + +def _request_with_body(body: bytes) -> Request: + async def receive() -> Message: + return {"type": "http.request", "body": body, "more_body": False} + + return _request(receive) @pytest.mark.asyncio async def test_read_request_body_valid_json(): - """Test the function with a valid JSON payload.""" - - class MockRequest: - async def body(self): - return b'{"key": "value"}' - - request = MockRequest() - result = await _read_request_body(request) + result = await _read_request_body(_request_with_body(b'{"key": "value"}')) assert result == {"key": "value"} @pytest.mark.asyncio async def test_read_request_body_empty_body(): - """Test the function with an empty body.""" - - class MockRequest: - async def body(self): - return b"" - - request = MockRequest() - result = await _read_request_body(request) + result = await _read_request_body(_request_with_body(b"")) assert result == {} @pytest.mark.asyncio async def test_read_request_body_invalid_json(): - """Test the function with an invalid JSON payload.""" - - class MockRequest: - async def body(self): - return b'{"key": value}' # Missing quotes around `value` - - request = MockRequest() with pytest.raises(ProxyException): - await _read_request_body(request) + await _read_request_body(_request_with_body(b'{"key": value}')) @pytest.mark.asyncio async def test_read_request_body_large_payload(): - """Test the function with a very large payload.""" - large_payload = '{"key":' + '"a"' * 10**6 + "}" # Large payload - - class MockRequest: - async def body(self): - return large_payload.encode() - - request = MockRequest() + large_payload = '{"key":' + '"a"' * 10**6 + "}" with pytest.raises(ProxyException): - await _read_request_body(request) + await _read_request_body(_request_with_body(large_payload.encode())) @pytest.mark.asyncio async def test_read_request_body_unexpected_error(): - """Test the function when an unexpected error occurs.""" + async def receive() -> Message: + raise ValueError("Unexpected error") - class MockRequest: - async def body(self): - raise ValueError("Unexpected error") - - request = MockRequest() - result = await _read_request_body(request) - assert result == {} # Ensure fallback behavior + result = await _read_request_body(_request(receive)) + assert result == {} diff --git a/tests/local_testing/test_text_completion.py b/tests/local_testing/test_text_completion.py index e49d3818d45..ea34b2dd21a 100644 --- a/tests/local_testing/test_text_completion.py +++ b/tests/local_testing/test_text_completion.py @@ -1,7 +1,9 @@ import asyncio from typing import Final import json +import os import traceback +from types import MappingProxyType from dotenv import load_dotenv @@ -26,6 +28,14 @@ from litellm import ( litellm.num_retries = 3 +FIREWORKS_TEXT_COMPLETION: Final = MappingProxyType( + { + "model": "text-completion-openai/accounts/fireworks/models/glm-5p3-flash", + "api_base": "https://api.fireworks.ai/inference/v1", + "api_key": os.environ.get("FIREWORKS_AI_API_KEY"), + } +) + token_prompt = [ [ 32, @@ -3778,8 +3788,9 @@ def test_completion_openai_prompt(): try: print("\n text 003 test\n") response = text_completion( - model="gpt-3.5-turbo-instruct", prompt=["What's the weather in SF?", "How is Manchester?"], + max_tokens=5, + **FIREWORKS_TEXT_COMPLETION, ) print(response) assert len(response.choices) == 2 @@ -3841,9 +3852,9 @@ def test_completion_chatgpt_prompt(): def test_completion_gpt_instruct(): try: response = text_completion( - model="gpt-3.5-turbo-instruct-0914", + model="gpt-5.4-nano", prompt="What's the weather in SF?", - custom_llm_provider="openai", + custom_llm_provider="text-completion-openai", ) print(response) response_str = response["choices"][0]["text"] @@ -3862,7 +3873,7 @@ def test_text_completion_basic(): print("\n test 003 with logprobs \n") litellm.set_verbose = False response = text_completion( - model="gpt-3.5-turbo-instruct", + model="text-completion-openai/gpt-5.4-nano", prompt="good morning", max_tokens=10, logprobs=10, @@ -3886,13 +3897,11 @@ def test_completion_text_003_prompt_array(): try: litellm.set_verbose = False response = text_completion( - model="gpt-3.5-turbo-instruct", prompt=token_prompt, # token prompt is a 2d list + max_tokens=5, + **FIREWORKS_TEXT_COMPLETION, ) - print("\n\n response") - - print(response) - # response_str = response["choices"][0]["text"] + assert len(response.choices) == len(token_prompt) except Exception as e: pytest.fail(f"Error occurred: {e}") @@ -4151,8 +4160,8 @@ def test_completion_fireworks_ai_multiple_choices(): def test_text_completion_with_echo(stream): litellm.set_verbose = True response = litellm.text_completion( - model="davinci-002", prompt="hello", + **FIREWORKS_TEXT_COMPLETION, max_tokens=1, # only see the first token stop="\n", # stop at the first newline logprobs=1, # return log prob @@ -4166,6 +4175,8 @@ def test_text_completion_with_echo(stream): print(chunk) else: assert isinstance(response, TextCompletionResponse) + assert response.choices[0].text.startswith("hello") + assert response.choices[0].logprobs.token_logprobs def test_text_completion_ollama(): diff --git a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json index 1d2d2bb336e..21c3d41c238 100644 --- a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json +++ b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json @@ -11,7 +11,7 @@ "user": "", "team_id": "", "organization_id": "", - "metadata": "{\"applied_guardrails\": [], \"attempted_fallbacks\": null, \"original_model_group\": null, \"batch_models\": null, \"batch_successful_requests\": null, \"batch_failed_requests\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"routing_decision\": null, \"internal_call_origin\": null, \"router_metadata\": null, \"autorouter_savings_estimate\": null, \"autorouter_baseline_observation\": null, \"azure_spillover\": null, \"guardrail_information\": null, \"compression_savings\": null, \"litellm_gateway_injected_cache\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_project_alias\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"user_agent\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}", + "metadata": "{\"actor_agent_id\": null, \"target_agent_id\": null, \"billing_agent_id\": null, \"agent_execution_mode\": null, \"verified_human_user_id\": null, \"applied_guardrails\": [], \"attempted_fallbacks\": null, \"original_model_group\": null, \"batch_models\": null, \"batch_successful_requests\": null, \"batch_failed_requests\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"routing_decision\": null, \"internal_call_origin\": null, \"router_metadata\": null, \"autorouter_savings_estimate\": null, \"autorouter_baseline_observation\": null, \"azure_spillover\": null, \"guardrail_information\": null, \"compression_savings\": null, \"litellm_gateway_injected_cache\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_project_alias\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"user_agent\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}", "cache_key": "Cache OFF", "spend": 0.00022500000000000002, "total_tokens": 30, @@ -29,5 +29,6 @@ "proxy_server_request": "{}", "status": "success", "mcp_namespaced_tool_name": null, - "agent_id": null + "agent_id": null, + "billing_agent_id": null } \ No newline at end of file diff --git a/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py b/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py index 2d597abf3b8..3bf51f02d34 100644 --- a/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/batches_endpoints/test_endpoints.py @@ -1034,6 +1034,7 @@ def _raw_batches_request(body: Dict[str, Any]) -> MagicMock: request.url.__str__.return_value = "http://localhost/v1/batches" request.url.path = "/v1/batches" request.method = "POST" + request.scope = {"type": "http", "method": "POST", "path": "/v1/batches"} request.query_params = {} request.headers = {"Content-Type": "application/json"} request.client = MagicMock() diff --git a/tests/unit/interactions/test_openapi_compliance.py b/tests/unit/interactions/test_openapi_compliance.py index d3f1183cea6..247d02298aa 100644 --- a/tests/unit/interactions/test_openapi_compliance.py +++ b/tests/unit/interactions/test_openapi_compliance.py @@ -9,6 +9,7 @@ Run with: pytest tests/unit/interactions/test_openapi_compliance.py -v import json import os +import re from typing import Any, Dict from unittest.mock import MagicMock, patch @@ -37,6 +38,25 @@ def _load_openapi_spec_dict() -> Dict[str, Any]: ) +def _model_create_request_schema(spec_dict: Dict[str, Any]) -> Dict[str, Any]: + schemas = spec_dict["components"]["schemas"] + create_path = next(path for path in spec_dict["paths"] if path.endswith("/interactions")) + body_schema = spec_dict["paths"][create_path]["post"]["requestBody"]["content"]["application/json"]["schema"] + variants = [schemas[option["$ref"].split("/")[-1]] for option in body_schema.get("oneOf", []) if "$ref" in option] + return next(variant for variant in variants if "model" in variant.get("properties", {})) + + +def _interaction_resource_path(spec_dict: Dict[str, Any], method: str) -> str | None: + return next( + ( + path + for path, methods in spec_dict["paths"].items() + if re.search(r"/interactions/\{[^}]+\}$", path) and method in methods + ), + None, + ) + + def _declared_type_value(variant_schema: Dict[str, Any]) -> Any: """The single `type` value a union variant pins, whether spelled as a const or a 1-item enum.""" type_property = variant_schema.get("properties", {}).get("type", {}) @@ -60,12 +80,10 @@ class TestRequestCompliance: """Tests that our request bodies match the OpenAPI spec.""" def test_create_model_interaction_request_schema(self, spec_dict): - """Verify CreateModelInteractionParams schema fields.""" - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] + schema = _model_create_request_schema(spec_dict) - # Required fields per spec assert "model" in schema["required"] - assert "input" in schema["required"] + assert "input" in schema["properties"] # Check our supported optional fields exist in spec our_optional_fields = [ @@ -88,7 +106,7 @@ class TestRequestCompliance: def test_input_types_match_spec(self, spec_dict): """Verify input field supports string, Content, Content[], Turn[].""" - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] + schema = _model_create_request_schema(spec_dict) input_schema = schema["properties"]["input"] # The input property may be inline oneOf or a $ref to InteractionsInput @@ -309,26 +327,14 @@ class TestEndpointCompliance: def test_get_endpoint_exists(self, spec_dict): """Verify GET /interactions/{id} endpoint exists.""" - paths = spec_dict["paths"] - - get_path = None - for path, methods in paths.items(): - if "{id}" in path and "interactions" in path and "get" in methods: - get_path = path - break + get_path = _interaction_resource_path(spec_dict, "get") assert get_path is not None, "GET /interactions/{id} endpoint not found" print(f"✓ Get endpoint: GET {get_path}") def test_delete_endpoint_exists(self, spec_dict): """Verify DELETE /interactions/{id} endpoint exists.""" - paths = spec_dict["paths"] - - delete_path = None - for path, methods in paths.items(): - if "{id}" in path and "interactions" in path and "delete" in methods: - delete_path = path - break + delete_path = _interaction_resource_path(spec_dict, "delete") assert delete_path is not None, "DELETE /interactions/{id} endpoint not found" print(f"✓ Delete endpoint: DELETE {delete_path}") diff --git a/tests/unit/models/test_models.py b/tests/unit/models/test_models.py index ab456bb1624..7b8953bd1a0 100644 --- a/tests/unit/models/test_models.py +++ b/tests/unit/models/test_models.py @@ -605,7 +605,7 @@ class TestManagedTables: class TestAutoRouterSession: @staticmethod - def _row(estimated_baseline_models: dict[str, int]) -> LiteLLM_AutoRouterSession: + def _row(baseline_models: dict[str, int], estimated_turns: int = 3) -> LiteLLM_AutoRouterSession: return LiteLLM_AutoRouterSession( api_key="k", session_id="s", @@ -619,9 +619,8 @@ class TestAutoRouterSession: saved_spend=0.24, classifier_cost=0.0, tier_turns={}, - baseline_models={"legacy-baseline": 100}, - savings_estimated_turns=sum(estimated_baseline_models.values()), - savings_estimated_baseline_models=estimated_baseline_models, + baseline_models=baseline_models, + savings_estimated_turns=estimated_turns, ) def test_the_baseline_label_is_the_one_most_turns_were_priced_against(self): @@ -633,5 +632,11 @@ class TestAutoRouterSession: assert self._row({"b-model": 1, "a-model": 1}).baseline_model == "b-model" assert self._row({"a-model": 1, "b-model": 1}).baseline_model == "b-model" - def test_a_row_without_current_estimates_has_no_baseline_label(self) -> None: + def test_a_row_without_recorded_baselines_has_no_baseline_label(self) -> None: assert self._row({}).baseline_model is None + + def test_a_partial_comparison_across_baselines_has_no_baseline_label(self) -> None: + assert self._row({"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1}, estimated_turns=2).baseline_model is None + + def test_a_partial_comparison_against_one_baseline_keeps_its_label(self) -> None: + assert self._row({"anthropic/claude-opus-5": 3}, estimated_turns=1).baseline_model == "anthropic/claude-opus-5"