test(ci): repair stale tests and move retired OpenAI text-completion fixtures (#43958)

* test(ci): repair stale request fakes, spend-log golden, auto-router labels, and Interactions spec lookups

Request fakes now carry the scope a real Starlette request has, the GCS pub/sub
spend-log golden gains the agent identity keys from #43722, the auto-router
session tests follow the baseline_models contract from #43348, and the
Interactions spec checks resolve the create body and resource paths from the
live spec instead of hardcoded names

* test(ci): move retired OpenAI text-completion fixtures to live vehicles

OpenAI still serves native /v1/completions on the gpt-5.4 family, so the
single-prompt cases move to text-completion-openai/gpt-5.4-nano. Multi-prompt
batches and echo with logprobs now 500 on every OpenAI model, so those cases
keep the same text-completion-openai transport pointed at Fireworks, which
documents both. The optional-params test asserts the request body actually
sent instead of a success callback whose assertions were swallowed

* test(ci): use a serverless Fireworks model for the text-completion batch and echo cases

gpt-oss-20b is on-demand only on Fireworks, so the CI key got 404 model not
deployed; glm-5p3-flash is listed as serverless

* test(ci): skip the ROI calculator repository listing in the security route sweep

GET /roi-calculator/repositories (#43669) lists repositories from the configured
GitHub API, api.github.com by default, so the S2 sweep's GET of every route made
the owned proxy reach an external host and failed the egress check in 31
integration-security tests. It joins /get/latest_release_info in the deny list
This commit is contained in:
yuneng-jiang 2026-09-30 19:19:59 -07:00 • committed by GitHub
parent 0c515ed7a8
commit c168199e33
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 122 additions and 143 deletions

View file

@ -105,6 +105,7 @@ ROUTE_DENY_LIST: Final = MappingProxyType(
"/plugin-proxy/{plugin_name}/{path:path}": "reverse proxy to a plugin process",
"/openai_passthrough/{endpoint:path}": "forwards to a provider, not a proxy read",
"/get/latest_release_info": "fetches the latest release from api.github.com",
"/roi-calculator/repositories": "lists repositories from the configured GitHub API, api.github.com by default",
}
)

View file

@ -11,7 +11,9 @@ import io
from unittest.mock import AsyncMock, MagicMock, patch
import httpx
import pytest
from openai import OpenAI
import litellm
from litellm import RateLimitError, Timeout, completion, completion_cost, embedding
@ -1580,7 +1582,7 @@ def test_completion_openai_pydantic(model, api_version):
def test_completion_text_openai():
try:
# litellm.set_verbose =True
response = completion(model="gpt-3.5-turbo-instruct", messages=messages)
response = completion(model="text-completion-openai/gpt-5.4-nano", messages=messages)
print(response["choices"][0]["message"]["content"])
except Exception as e:
print(e)
@ -1592,7 +1594,7 @@ async def test_completion_text_openai_async():
try:
# litellm.set_verbose =True
response = await litellm.acompletion(
model="gpt-3.5-turbo-instruct", messages=messages
model="text-completion-openai/gpt-5.4-nano", messages=messages
)
print(response["choices"][0]["message"]["content"])
except Exception as e:
@ -1600,67 +1602,33 @@ async def test_completion_text_openai_async():
pytest.fail(f"Error occurred: {e}")
def custom_callback(
kwargs, # kwargs to completion
completion_response, # response from completion
start_time,
end_time, # start/end time
):
# Your custom code here
try:
print("LITELLM: in custom callback function")
print("\nkwargs\n", kwargs)
model = kwargs["model"]
messages = kwargs["messages"]
user = kwargs.get("user")
#################################################
print(
f"""
Model: {model},
Messages: {messages},
User: {user},
Seed: {kwargs["seed"]},
temperature: {kwargs["temperature"]},
"""
)
assert kwargs["user"] == "ishaans app"
assert kwargs["model"] == "gpt-3.5-turbo-1106"
assert kwargs["seed"] == 12
assert kwargs["temperature"] == 0.5
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def test_completion_openai_with_optional_params():
# [Proxy PROD TEST] WARNING: DO NOT DELETE THIS TEST
# assert that `user` gets passed to the completion call
# Note: This tests that we actually send the optional params to the completion call
# We use custom callbacks to test this
try:
litellm.set_verbose = True
litellm.success_callback = [custom_callback]
response = completion(
model="gpt-3.5-turbo-1106",
messages=[
{"role": "user", "content": "respond in valid, json - what is the day"}
],
temperature=0.5,
top_p=0.1,
seed=12,
response_format={"type": "json_object"},
logit_bias=None,
user="ishaans app",
)
# Add any assertions here to check the response
on_request = MagicMock()
client = OpenAI(http_client=httpx.Client(event_hooks={"request": [on_request]}))
response = completion(
model="gpt-6-luna",
reasoning_effort="none",
messages=[{"role": "user", "content": "respond in valid, json - what is the day"}],
temperature=0.5,
top_p=0.1,
seed=12,
response_format={"type": "json_object"},
logit_bias=None,
user="ishaans app",
client=client,
)
print(response)
litellm.success_callback = [] # unset callbacks
except Exception as e:
pytest.fail(f"Error occurred: {e}")
assert response.choices[0].message.content
on_request.assert_called_once()
sent = json.loads(on_request.call_args.args[0].content)
assert sent["model"] == "gpt-6-luna"
assert sent["user"] == "ishaans app"
assert sent["seed"] == 12
assert sent["temperature"] == 0.5
assert sent["top_p"] == 0.1
assert sent["response_format"] == {"type": "json_object"}
assert "logit_bias" not in sent
# test_completion_openai_with_optional_params()
@ -4008,7 +3976,7 @@ def test_deepseek_reasoning_content_completion():
def test_qwen_text_completion():
# litellm._turn_on_debug()
resp = litellm.completion(
model="gpt-3.5-turbo-instruct",
model="text-completion-openai/gpt-5.4-nano",
messages=[{"content": "hello", "role": "user"}],
stream=False,
logprobs=1,

View file

@ -1,75 +1,61 @@
from collections.abc import Awaitable, Callable
import pytest
from fastapi import Request
from fastapi.testclient import TestClient
from starlette.datastructures import Headers
from starlette.requests import HTTPConnection
from starlette.types import Message
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
from litellm.proxy._types import ProxyException
from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
def _request(receive: Callable[[], Awaitable[Message]]) -> Request:
return Request(
{
"type": "http",
"method": "POST",
"path": "/v1/chat/completions",
"headers": [(b"content-type", b"application/json")],
},
receive,
)
def _request_with_body(body: bytes) -> Request:
async def receive() -> Message:
return {"type": "http.request", "body": body, "more_body": False}
return _request(receive)
@pytest.mark.asyncio
async def test_read_request_body_valid_json():
"""Test the function with a valid JSON payload."""
class MockRequest:
async def body(self):
return b'{"key": "value"}'
request = MockRequest()
result = await _read_request_body(request)
result = await _read_request_body(_request_with_body(b'{"key": "value"}'))
assert result == {"key": "value"}
@pytest.mark.asyncio
async def test_read_request_body_empty_body():
"""Test the function with an empty body."""
class MockRequest:
async def body(self):
return b""
request = MockRequest()
result = await _read_request_body(request)
result = await _read_request_body(_request_with_body(b""))
assert result == {}
@pytest.mark.asyncio
async def test_read_request_body_invalid_json():
"""Test the function with an invalid JSON payload."""
class MockRequest:
async def body(self):
return b'{"key": value}' # Missing quotes around `value`
request = MockRequest()
with pytest.raises(ProxyException):
await _read_request_body(request)
await _read_request_body(_request_with_body(b'{"key": value}'))
@pytest.mark.asyncio
async def test_read_request_body_large_payload():
"""Test the function with a very large payload."""
large_payload = '{"key":' + '"a"' * 10**6 + "}" # Large payload
class MockRequest:
async def body(self):
return large_payload.encode()
request = MockRequest()
large_payload = '{"key":' + '"a"' * 10**6 + "}"
with pytest.raises(ProxyException):
await _read_request_body(request)
await _read_request_body(_request_with_body(large_payload.encode()))
@pytest.mark.asyncio
async def test_read_request_body_unexpected_error():
"""Test the function when an unexpected error occurs."""
async def receive() -> Message:
raise ValueError("Unexpected error")
class MockRequest:
async def body(self):
raise ValueError("Unexpected error")
request = MockRequest()
result = await _read_request_body(request)
assert result == {} # Ensure fallback behavior
result = await _read_request_body(_request(receive))
assert result == {}

View file

@ -1,7 +1,9 @@
import asyncio
from typing import Final
import json
import os
import traceback
from types import MappingProxyType
from dotenv import load_dotenv
@ -26,6 +28,14 @@ from litellm import (
litellm.num_retries = 3
FIREWORKS_TEXT_COMPLETION: Final = MappingProxyType(
{
"model": "text-completion-openai/accounts/fireworks/models/glm-5p3-flash",
"api_base": "https://api.fireworks.ai/inference/v1",
"api_key": os.environ.get("FIREWORKS_AI_API_KEY"),
}
)
token_prompt = [
[
32,
@ -3778,8 +3788,9 @@ def test_completion_openai_prompt():
try:
print("\n text 003 test\n")
response = text_completion(
model="gpt-3.5-turbo-instruct",
prompt=["What's the weather in SF?", "How is Manchester?"],
max_tokens=5,
**FIREWORKS_TEXT_COMPLETION,
)
print(response)
assert len(response.choices) == 2
@ -3841,9 +3852,9 @@ def test_completion_chatgpt_prompt():
def test_completion_gpt_instruct():
try:
response = text_completion(
model="gpt-3.5-turbo-instruct-0914",
model="gpt-5.4-nano",
prompt="What's the weather in SF?",
custom_llm_provider="openai",
custom_llm_provider="text-completion-openai",
)
print(response)
response_str = response["choices"][0]["text"]
@ -3862,7 +3873,7 @@ def test_text_completion_basic():
print("\n test 003 with logprobs \n")
litellm.set_verbose = False
response = text_completion(
model="gpt-3.5-turbo-instruct",
model="text-completion-openai/gpt-5.4-nano",
prompt="good morning",
max_tokens=10,
logprobs=10,
@ -3886,13 +3897,11 @@ def test_completion_text_003_prompt_array():
try:
litellm.set_verbose = False
response = text_completion(
model="gpt-3.5-turbo-instruct",
prompt=token_prompt, # token prompt is a 2d list
max_tokens=5,
**FIREWORKS_TEXT_COMPLETION,
)
print("\n\n response")
print(response)
# response_str = response["choices"][0]["text"]
assert len(response.choices) == len(token_prompt)
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@ -4151,8 +4160,8 @@ def test_completion_fireworks_ai_multiple_choices():
def test_text_completion_with_echo(stream):
litellm.set_verbose = True
response = litellm.text_completion(
model="davinci-002",
prompt="hello",
**FIREWORKS_TEXT_COMPLETION,
max_tokens=1, # only see the first token
stop="\n", # stop at the first newline
logprobs=1, # return log prob
@ -4166,6 +4175,8 @@ def test_text_completion_with_echo(stream):
print(chunk)
else:
assert isinstance(response, TextCompletionResponse)
assert response.choices[0].text.startswith("hello")
assert response.choices[0].logprobs.token_logprobs
def test_text_completion_ollama():

View file

@ -11,7 +11,7 @@
"user": "",
"team_id": "",
"organization_id": "",
"metadata": "{\"applied_guardrails\": [], \"attempted_fallbacks\": null, \"original_model_group\": null, \"batch_models\": null, \"batch_successful_requests\": null, \"batch_failed_requests\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"routing_decision\": null, \"internal_call_origin\": null, \"router_metadata\": null, \"autorouter_savings_estimate\": null, \"autorouter_baseline_observation\": null, \"azure_spillover\": null, \"guardrail_information\": null, \"compression_savings\": null, \"litellm_gateway_injected_cache\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_project_alias\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"user_agent\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}",
"metadata": "{\"actor_agent_id\": null, \"target_agent_id\": null, \"billing_agent_id\": null, \"agent_execution_mode\": null, \"verified_human_user_id\": null, \"applied_guardrails\": [], \"attempted_fallbacks\": null, \"original_model_group\": null, \"batch_models\": null, \"batch_successful_requests\": null, \"batch_failed_requests\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"routing_decision\": null, \"internal_call_origin\": null, \"router_metadata\": null, \"autorouter_savings_estimate\": null, \"autorouter_baseline_observation\": null, \"azure_spillover\": null, \"guardrail_information\": null, \"compression_savings\": null, \"litellm_gateway_injected_cache\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_project_alias\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"user_agent\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}",
"cache_key": "Cache OFF",
"spend": 0.00022500000000000002,
"total_tokens": 30,
@ -29,5 +29,6 @@
"proxy_server_request": "{}",
"status": "success",
"mcp_namespaced_tool_name": null,
"agent_id": null
"agent_id": null,
"billing_agent_id": null
}

View file

@ -1034,6 +1034,7 @@ def _raw_batches_request(body: Dict[str, Any]) -> MagicMock:
request.url.__str__.return_value = "http://localhost/v1/batches"
request.url.path = "/v1/batches"
request.method = "POST"
request.scope = {"type": "http", "method": "POST", "path": "/v1/batches"}
request.query_params = {}
request.headers = {"Content-Type": "application/json"}
request.client = MagicMock()

View file

@ -9,6 +9,7 @@ Run with: pytest tests/unit/interactions/test_openapi_compliance.py -v
import json
import os
import re
from typing import Any, Dict
from unittest.mock import MagicMock, patch
@ -37,6 +38,25 @@ def _load_openapi_spec_dict() -> Dict[str, Any]:
)
def _model_create_request_schema(spec_dict: Dict[str, Any]) -> Dict[str, Any]:
schemas = spec_dict["components"]["schemas"]
create_path = next(path for path in spec_dict["paths"] if path.endswith("/interactions"))
body_schema = spec_dict["paths"][create_path]["post"]["requestBody"]["content"]["application/json"]["schema"]
variants = [schemas[option["$ref"].split("/")[-1]] for option in body_schema.get("oneOf", []) if "$ref" in option]
return next(variant for variant in variants if "model" in variant.get("properties", {}))
def _interaction_resource_path(spec_dict: Dict[str, Any], method: str) -> str | None:
return next(
(
path
for path, methods in spec_dict["paths"].items()
if re.search(r"/interactions/\{[^}]+\}$", path) and method in methods
),
None,
)
def _declared_type_value(variant_schema: Dict[str, Any]) -> Any:
"""The single `type` value a union variant pins, whether spelled as a const or a 1-item enum."""
type_property = variant_schema.get("properties", {}).get("type", {})
@ -60,12 +80,10 @@ class TestRequestCompliance:
"""Tests that our request bodies match the OpenAPI spec."""
def test_create_model_interaction_request_schema(self, spec_dict):
"""Verify CreateModelInteractionParams schema fields."""
schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"]
schema = _model_create_request_schema(spec_dict)
# Required fields per spec
assert "model" in schema["required"]
assert "input" in schema["required"]
assert "input" in schema["properties"]
# Check our supported optional fields exist in spec
our_optional_fields = [
@ -88,7 +106,7 @@ class TestRequestCompliance:
def test_input_types_match_spec(self, spec_dict):
"""Verify input field supports string, Content, Content[], Turn[]."""
schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"]
schema = _model_create_request_schema(spec_dict)
input_schema = schema["properties"]["input"]
# The input property may be inline oneOf or a $ref to InteractionsInput
@ -309,26 +327,14 @@ class TestEndpointCompliance:
def test_get_endpoint_exists(self, spec_dict):
"""Verify GET /interactions/{id} endpoint exists."""
paths = spec_dict["paths"]
get_path = None
for path, methods in paths.items():
if "{id}" in path and "interactions" in path and "get" in methods:
get_path = path
break
get_path = _interaction_resource_path(spec_dict, "get")
assert get_path is not None, "GET /interactions/{id} endpoint not found"
print(f"✓ Get endpoint: GET {get_path}")
def test_delete_endpoint_exists(self, spec_dict):
"""Verify DELETE /interactions/{id} endpoint exists."""
paths = spec_dict["paths"]
delete_path = None
for path, methods in paths.items():
if "{id}" in path and "interactions" in path and "delete" in methods:
delete_path = path
break
delete_path = _interaction_resource_path(spec_dict, "delete")
assert delete_path is not None, "DELETE /interactions/{id} endpoint not found"
print(f"✓ Delete endpoint: DELETE {delete_path}")

View file

@ -605,7 +605,7 @@ class TestManagedTables:
class TestAutoRouterSession:
@staticmethod
def _row(estimated_baseline_models: dict[str, int]) -> LiteLLM_AutoRouterSession:
def _row(baseline_models: dict[str, int], estimated_turns: int = 3) -> LiteLLM_AutoRouterSession:
return LiteLLM_AutoRouterSession(
api_key="k",
session_id="s",
@ -619,9 +619,8 @@ class TestAutoRouterSession:
saved_spend=0.24,
classifier_cost=0.0,
tier_turns={},
baseline_models={"legacy-baseline": 100},
savings_estimated_turns=sum(estimated_baseline_models.values()),
savings_estimated_baseline_models=estimated_baseline_models,
baseline_models=baseline_models,
savings_estimated_turns=estimated_turns,
)
def test_the_baseline_label_is_the_one_most_turns_were_priced_against(self):
@ -633,5 +632,11 @@ class TestAutoRouterSession:
assert self._row({"b-model": 1, "a-model": 1}).baseline_model == "b-model"
assert self._row({"a-model": 1, "b-model": 1}).baseline_model == "b-model"
def test_a_row_without_current_estimates_has_no_baseline_label(self) -> None:
def test_a_row_without_recorded_baselines_has_no_baseline_label(self) -> None:
assert self._row({}).baseline_model is None
def test_a_partial_comparison_across_baselines_has_no_baseline_label(self) -> None:
assert self._row({"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1}, estimated_turns=2).baseline_model is None
def test_a_partial_comparison_against_one_baseline_keeps_its_label(self) -> None:
assert self._row({"anthropic/claude-opus-5": 3}, estimated_turns=1).baseline_model == "anthropic/claude-opus-5"