test(together_ai): select the live model from the cost map

`together_ai/openai/gpt-oss-20b` was hardcoded in two live tests and is no
longer served, so both failed on a vendor catalog change rather than on
anything litellm did.

Both call sites now resolve the cheapest non-deprecated together_ai chat
entry at runtime, filtered on the capabilities the tests actually exercise,
mirroring what tests/e2e/llm_translation/test_together_ai_e2e.py already
does. The selector lives in tests/_live_test_helpers.py so both lanes share
one implementation.
This commit is contained in:
Yuneng Jiang 2026-09-16 14:20:49 -07:00
parent f10d95fb95
commit b478131701
3 changed files with 48 additions and 2 deletions

View file

@ -1,4 +1,7 @@
import os
from collections.abc import Mapping
from datetime import date
from typing import Any
import pytest
@ -8,3 +11,36 @@ def _skip_live_prompt_caching_test():
pytest.skip("Live prompt-caching E2E tests are opt-in")
if os.environ.get("CASSETTE_REDIS_URL"):
pytest.skip("Live prompt-caching E2E tests cannot run under VCR replay")
def cheapest_together_chat_model(*capability_flags: str) -> str:
import litellm
today = date.today().isoformat()
def qualifies(name: str, entry: Mapping[str, Any]) -> bool:
deprecation_date = entry.get("deprecation_date")
return (
name.startswith("together_ai/")
and entry.get("litellm_provider") == "together_ai"
and entry.get("mode") == "chat"
and (deprecation_date is None or deprecation_date > today)
and (entry.get("input_cost_per_token") or 0.0) > 0
and (entry.get("output_cost_per_token") or 0.0) > 0
and all(bool(entry.get(flag)) for flag in capability_flags)
)
candidates = sorted(
(
name
for name, entry in litellm.model_cost.items()
if isinstance(entry, Mapping) and qualifies(name, entry)
),
key=lambda name: (
litellm.model_cost[name].get("input_cost_per_token") or 0.0,
litellm.model_cost[name].get("output_cost_per_token") or 0.0,
name,
),
)
assert candidates, f"no live together_ai chat model in the cost map satisfies {capability_flags}"
return candidates[0]

View file

@ -3,6 +3,7 @@ Test TogetherAI LLM
"""
from base_llm_unit_tests import BaseLLMChatTest
from tests._live_test_helpers import cheapest_together_chat_model
import json
import os
from datetime import datetime
@ -16,7 +17,11 @@ import pytest
class TestTogetherAI(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
litellm.set_verbose = True
return {"model": "together_ai/openai/gpt-oss-20b"}
return {
"model": cheapest_together_chat_model(
"supports_function_calling", "supports_response_schema"
)
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""

View file

@ -1,7 +1,11 @@
import asyncio
import json
import os
import sys
import traceback
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
from dotenv import load_dotenv
load_dotenv()
@ -12,6 +16,7 @@ from unittest.mock import MagicMock, patch
import pytest
import litellm
from tests._live_test_helpers import cheapest_together_chat_model
from litellm import (
RateLimitError,
TextCompletionResponse,
@ -4030,7 +4035,7 @@ def test_async_text_completion_together_ai():
async def test_get_response():
try:
response = await litellm.atext_completion(
model="together_ai/openai/gpt-oss-20b",
model=cheapest_together_chat_model(),
prompt="good morning",
max_tokens=10,
)