mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Every live together_ai call in CI has answered 503 Service unavailable since 2026-08-20, across two runs 2.5 hours apart, while Together's status page reported no incident in either window. These are real calls, not replayed cassettes: the VCR layer runs filter_non_2xx_response, so a 503 is never written to a cassette and cannot be replayed back. Qwen/Qwen2.5-7B-Instruct-Turbo does not appear anywhere on Together's monitored component list, whose Qwen entries are all Qwen3.x, so a model-level outage there would never surface as an incident. The same 503 already forced test_basic_rerank_together_ai to be skipped on a different together_ai model, so per-model 503s are an established failure mode here rather than a platform outage. openai/gpt-oss-20b is the cheapest together_ai entry that carries real pricing and the capabilities these suites exercise, at $0.05/$0.20 per 1M tokens with function calling, response schema and tool choice. Together monitors it as a served component. The retired model also carries null pricing in the cost map, which is its own liability now that unpriced models are blocked. test_multiple_deployments.py keeps the old id: it is a router fallback list that is green today, and busting its cassette to prove a point would trade a passing test for a live call this change cannot vouch for.
52 lines
1.6 KiB
Python
52 lines
1.6 KiB
Python
"""
|
|
Test TogetherAI LLM
|
|
"""
|
|
|
|
from base_llm_unit_tests import BaseLLMChatTest
|
|
import json
|
|
import os
|
|
import sys
|
|
from datetime import datetime
|
|
from unittest.mock import AsyncMock
|
|
|
|
sys.path.insert(
|
|
0, os.path.abspath("../..")
|
|
) # Adds the parent directory to the system path
|
|
|
|
import litellm
|
|
import pytest
|
|
|
|
|
|
class TestTogetherAI(BaseLLMChatTest):
|
|
def get_base_completion_call_args(self) -> dict:
|
|
litellm.set_verbose = True
|
|
return {"model": "together_ai/openai/gpt-oss-20b"}
|
|
|
|
def test_tool_call_no_arguments(self, tool_call_no_arguments):
|
|
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
|
|
pass
|
|
|
|
@pytest.mark.parametrize(
|
|
"model, expected_bool",
|
|
[
|
|
("meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", True),
|
|
("nvidia/Llama-3.1-Nemotron-70B-Instruct-HF", False),
|
|
],
|
|
)
|
|
def test_get_supported_response_format_together_ai(
|
|
self, model: str, expected_bool: bool
|
|
) -> None:
|
|
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
|
litellm.model_cost = litellm.get_model_cost_map(url="")
|
|
optional_params = litellm.get_supported_openai_params(
|
|
model, custom_llm_provider="together_ai"
|
|
)
|
|
# Mapped provider
|
|
assert isinstance(optional_params, list)
|
|
|
|
if expected_bool:
|
|
assert "response_format" in optional_params
|
|
assert "tools" in optional_params
|
|
else:
|
|
assert "response_format" not in optional_params
|
|
assert "tools" not in optional_params
|