mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
test(together_ai): select the live model from the cost map
`together_ai/openai/gpt-oss-20b` was hardcoded in two live tests and is no longer served, so both failed on a vendor catalog change rather than on anything litellm did. Both call sites now resolve the cheapest non-deprecated together_ai chat entry at runtime, filtered on the capabilities the tests actually exercise, mirroring what tests/e2e/llm_translation/test_together_ai_e2e.py already does. The selector lives in tests/_live_test_helpers.py so both lanes share one implementation.
This commit is contained in:
parent
f10d95fb95
commit
b478131701
3 changed files with 48 additions and 2 deletions
|
|
@ -1,4 +1,7 @@
|
|||
import os
|
||||
from collections.abc import Mapping
|
||||
from datetime import date
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -8,3 +11,36 @@ def _skip_live_prompt_caching_test():
|
|||
pytest.skip("Live prompt-caching E2E tests are opt-in")
|
||||
if os.environ.get("CASSETTE_REDIS_URL"):
|
||||
pytest.skip("Live prompt-caching E2E tests cannot run under VCR replay")
|
||||
|
||||
|
||||
def cheapest_together_chat_model(*capability_flags: str) -> str:
|
||||
import litellm
|
||||
|
||||
today = date.today().isoformat()
|
||||
|
||||
def qualifies(name: str, entry: Mapping[str, Any]) -> bool:
|
||||
deprecation_date = entry.get("deprecation_date")
|
||||
return (
|
||||
name.startswith("together_ai/")
|
||||
and entry.get("litellm_provider") == "together_ai"
|
||||
and entry.get("mode") == "chat"
|
||||
and (deprecation_date is None or deprecation_date > today)
|
||||
and (entry.get("input_cost_per_token") or 0.0) > 0
|
||||
and (entry.get("output_cost_per_token") or 0.0) > 0
|
||||
and all(bool(entry.get(flag)) for flag in capability_flags)
|
||||
)
|
||||
|
||||
candidates = sorted(
|
||||
(
|
||||
name
|
||||
for name, entry in litellm.model_cost.items()
|
||||
if isinstance(entry, Mapping) and qualifies(name, entry)
|
||||
),
|
||||
key=lambda name: (
|
||||
litellm.model_cost[name].get("input_cost_per_token") or 0.0,
|
||||
litellm.model_cost[name].get("output_cost_per_token") or 0.0,
|
||||
name,
|
||||
),
|
||||
)
|
||||
assert candidates, f"no live together_ai chat model in the cost map satisfies {capability_flags}"
|
||||
return candidates[0]
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ Test TogetherAI LLM
|
|||
"""
|
||||
|
||||
from base_llm_unit_tests import BaseLLMChatTest
|
||||
from tests._live_test_helpers import cheapest_together_chat_model
|
||||
import json
|
||||
import os
|
||||
from datetime import datetime
|
||||
|
|
@ -16,7 +17,11 @@ import pytest
|
|||
class TestTogetherAI(BaseLLMChatTest):
|
||||
def get_base_completion_call_args(self) -> dict:
|
||||
litellm.set_verbose = True
|
||||
return {"model": "together_ai/openai/gpt-oss-20b"}
|
||||
return {
|
||||
"model": cheapest_together_chat_model(
|
||||
"supports_function_calling", "supports_response_schema"
|
||||
)
|
||||
}
|
||||
|
||||
def test_tool_call_no_arguments(self, tool_call_no_arguments):
|
||||
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
|
||||
|
|
|
|||
|
|
@ -1,7 +1,11 @@
|
|||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
|
||||
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
|
@ -12,6 +16,7 @@ from unittest.mock import MagicMock, patch
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from tests._live_test_helpers import cheapest_together_chat_model
|
||||
from litellm import (
|
||||
RateLimitError,
|
||||
TextCompletionResponse,
|
||||
|
|
@ -4030,7 +4035,7 @@ def test_async_text_completion_together_ai():
|
|||
async def test_get_response():
|
||||
try:
|
||||
response = await litellm.atext_completion(
|
||||
model="together_ai/openai/gpt-oss-20b",
|
||||
model=cheapest_together_chat_model(),
|
||||
prompt="good morning",
|
||||
max_tokens=10,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue