test: fix five tests left stale by #41311, #41337, #39996 and #41310

Every one of these fails on main's own scheduled CircleCI run with the same
assertion as on any PR, and each traces to a merged behavior change that
never updated the test that pinned the old behavior

- tests/integration/_support/client.py: #41311 made /key/info serve deleted
  keys from the archive with status deleted, so the scenario teardown asserts
  the live row is gone and the readback reports deleted instead of a 404.
  This alone accounts for nine integration-management and one
  integration-providers failure
- tests/integration/authorization/test_warmed_policy.py: #39996 made team
  admins unable to edit any team field unless a proxy admin allow-lists it,
  and tpm_limit is the only field it accepts today. The demotion test now
  enables tpm_limit for the scenario and edits that instead of team_alias
- tests/llm_responses_api_testing/test_base_responses_api_streaming_iterator.py:
  #41337 reads usage off the terminal response and copies the event when it
  is missing, which a Mock(spec=ResponsesAPIResponse) cannot survive. The
  four mocks now carry a usage object
- tests/test_openai_endpoints.py: #41310 lengthened the access-denied
  message, and the test matched against the ExceptionInfo repr, which
  saferepr truncates in the middle. It now matches the exception text
- tests/local_testing/test_text_completion.py: Together no longer serves
  Qwen2-1.5B serverless, the cheapest cost-map row. The test mocks the
  completions call and asserts the request litellm builds, so a vendor
  catalog rotation cannot fail it again

test_router_fallbacks_with_cooldowns_and_dynamic_credentials is deliberately
untouched: it passes and fails on main with identical code, and the failing
path is a product question about whether dynamic-credential 429s cool down
This commit is contained in:
Yuneng Jiang 2026-09-16 17:47:27 -07:00
parent 9e1eb546e4
commit 02ced74540
No known key found for this signature in database
5 changed files with 58 additions and 31 deletions

View file

@ -132,8 +132,10 @@ class Scenario:
def delete_key(self, token: str) -> None:
self.gateway.post("/key/delete", {"keys": [token]})
response: Final = self.gateway.request("GET", "/key/info", params={"key": sha256(token.encode()).hexdigest()})
assert response.status_code == 404, f"Deleted key remains readable: {response.status_code}"
hashed: Final = sha256(token.encode()).hexdigest()
assert read_rows('SELECT token FROM "LiteLLM_VerificationToken" WHERE token = %s', (hashed,)) == []
info: Final = object_value(self.gateway.get("/key/info", {"key": hashed})["info"])
assert info["status"] == "deleted", f"Deleted key still served as live: {info['status']}"
def delete_model(self, identity: str) -> None:
self.gateway.post("/model/delete", {"id": identity})

View file

@ -1,10 +1,12 @@
from contextlib import ExitStack
from collections.abc import Iterator
from contextlib import ExitStack, contextmanager
from hashlib import sha256
from typing import Final
import os
import psycopg
import pytest
from pydantic import JsonValue
from hypothesis import strategies as st
from hypothesis.stateful import RuleBasedStateMachine, invariant, rule, run_state_machine_as_test
@ -134,37 +136,58 @@ def test_scim_deactivation_blocks_null_and_false_keys_but_preserves_other_owners
assert_serving(gateway, model, token, 200)
def _set_team_admin_editable_fields(gateway: Gateway, fields: list[JsonValue]) -> None:
response: Final = gateway.request("PATCH", "/update/ui_settings", {"team_admin_editable_team_fields": fields})
assert response.status_code == 200, response.text
@contextmanager
def _team_admins_may_edit(gateway: Gateway, fields: list[JsonValue]) -> Iterator[None]:
original: Final = object_value(gateway.get("/get/ui_settings")["values"]).get("team_admin_editable_team_fields")
_set_team_admin_editable_fields(gateway, fields)
try:
yield
finally:
_set_team_admin_editable_fields(gateway, original if isinstance(original, list) else [])
@pytest.mark.covers("mgmt.team.member_update.demoted_role_cannot_write")
def test_warmed_team_role_demotion_prevents_later_management_writes(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
with gateway.scenario() as scenario, _team_admins_may_edit(gateway, ["tpm_limit"]):
model: Final = scenario.model()
user: Final = scenario.user(user_role="internal_user")
team: Final = scenario.team(models=[model], members_with_roles=[{"user_id": user, "role": "admin"}])
control_team: Final = scenario.team(models=[model])
team: Final = scenario.team(
models=[model], tpm_limit=1000, members_with_roles=[{"user_id": user, "role": "admin"}]
)
control_team: Final = scenario.team(models=[model], tpm_limit=1000)
caller: Final = scenario.key(
user_id=user, team_id=team, models=[model], allowed_routes=["/team/update", "/v1/chat/completions"]
)
gateway.chat(model, key=caller)
changed: Final = gateway.request("POST", "/team/update", {"team_id": team, "team_alias": "permitted"}, key=caller)
changed: Final = gateway.request("POST", "/team/update", {"team_id": team, "tpm_limit": 5000}, key=caller)
assert changed.status_code == 200, changed.text
assert read_rows('SELECT tpm_limit FROM "LiteLLM_TeamTable" WHERE team_id = %s', (team,)) == [
{"tpm_limit": 5000}
]
unrelated_before: Final = read_rows(
'SELECT team_alias FROM "LiteLLM_TeamTable" WHERE team_id = %s', (control_team,)
'SELECT tpm_limit FROM "LiteLLM_TeamTable" WHERE team_id = %s', (control_team,)
)
unrelated: Final = gateway.request(
"POST", "/team/update", {"team_id": control_team, "team_alias": "must-not-persist"}, key=caller
"POST", "/team/update", {"team_id": control_team, "tpm_limit": 7000}, key=caller
)
assert unrelated.status_code == 403, unrelated.text
assert read_rows(
'SELECT team_alias FROM "LiteLLM_TeamTable" WHERE team_id = %s', (control_team,)
'SELECT tpm_limit FROM "LiteLLM_TeamTable" WHERE team_id = %s', (control_team,)
) == unrelated_before
gateway.post("/team/member_update", {"team_id": team, "user_id": user, "role": "user"})
for target in (team, control_team):
before: Final = read_rows('SELECT team_alias FROM "LiteLLM_TeamTable" WHERE team_id = %s', (target,))
before: Final = read_rows('SELECT tpm_limit FROM "LiteLLM_TeamTable" WHERE team_id = %s', (target,))
denied: Final = gateway.request(
"POST", "/team/update", {"team_id": target, "team_alias": "must-not-persist"}, key=caller
"POST", "/team/update", {"team_id": target, "tpm_limit": 9000}, key=caller
)
assert denied.status_code == 403, denied.text
assert read_rows('SELECT team_alias FROM "LiteLLM_TeamTable" WHERE team_id = %s', (target,)) == before
after: Final = read_rows('SELECT tpm_limit FROM "LiteLLM_TeamTable" WHERE team_id = %s', (target,))
assert after == before
roster: Final = read_rows('SELECT members_with_roles FROM "LiteLLM_TeamTable" WHERE team_id = %s', (team,))
members: Final = roster[0]["members_with_roles"]
assert isinstance(members, list)

View file

@ -26,6 +26,7 @@ from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfi
from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator
from litellm.responses.utils import ResponsesAPIRequestUtils
from litellm.types.llms.openai import (
ResponseAPIUsage,
ResponseCompletedEvent,
ResponseFailedEvent,
ResponseIncompleteEvent,
@ -69,6 +70,7 @@ class TestBaseResponsesAPIStreamingIterator:
mock_responses_api_response = Mock(spec=ResponsesAPIResponse)
mock_responses_api_response.id = "resp_u2028"
mock_responses_api_response.usage = ResponseAPIUsage(input_tokens=3, output_tokens=2, total_tokens=5)
mock_completed_event = Mock(spec=ResponseCompletedEvent)
mock_completed_event.type = ResponsesAPIStreamEvents.RESPONSE_COMPLETED
mock_completed_event.response = mock_responses_api_response
@ -123,6 +125,7 @@ class TestBaseResponsesAPIStreamingIterator:
# Mock the _update_responses_api_response_id_with_model_id method
updated_response = Mock(spec=ResponsesAPIResponse)
updated_response.id = "updated_response_id"
updated_response.usage = ResponseAPIUsage(input_tokens=3, output_tokens=2, total_tokens=5)
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
@ -524,7 +527,7 @@ class TestBaseResponsesAPIStreamingIterator:
"type": "server_error",
"message": "The model encountered an error",
}
mock_responses_api_response.usage = None
mock_responses_api_response.usage = ResponseAPIUsage(input_tokens=3, output_tokens=2, total_tokens=5)
mock_failed_event = Mock(spec=ResponseFailedEvent)
mock_failed_event.type = ResponsesAPIStreamEvents.RESPONSE_FAILED
@ -604,7 +607,7 @@ class TestBaseResponsesAPIStreamingIterator:
mock_responses_api_response = Mock(spec=ResponsesAPIResponse)
mock_responses_api_response.id = "resp_incomplete_123"
mock_responses_api_response.incomplete_details = {"reason": "max_output_tokens"}
mock_responses_api_response.usage = None
mock_responses_api_response.usage = ResponseAPIUsage(input_tokens=3, output_tokens=2, total_tokens=5)
mock_incomplete_event = Mock(spec=ResponseIncompleteEvent)
mock_incomplete_event.type = ResponsesAPIStreamEvents.RESPONSE_INCOMPLETE

View file

@ -12,7 +12,6 @@ from unittest.mock import MagicMock, patch
import pytest
import litellm
from tests._live_test_helpers import cheapest_together_chat_model
from litellm import (
RateLimitError,
TextCompletionResponse,
@ -4023,27 +4022,27 @@ def test_async_text_completion():
asyncio.run(test_get_response())
@pytest.mark.flaky(retries=6, delay=1)
def test_async_text_completion_together_ai():
litellm.set_verbose = True
print("test_async_text_completion")
from openai import AsyncOpenAI
async def test_get_response():
try:
client = AsyncOpenAI(api_key="my-fake-key")
async def run_call():
with patch.object(client.completions.with_raw_response, "create", side_effect=mock_post) as mock_call:
response = await litellm.atext_completion(
model=cheapest_together_chat_model(),
model="together_ai/Qwen/Qwen2-1.5B-Instruct",
prompt="good morning",
max_tokens=10,
client=client,
)
print(f"response: {response}")
except litellm.RateLimitError as e:
print(e)
except litellm.Timeout as e:
print(e)
except Exception as e:
pytest.fail("An unexpected error occurred")
return response, mock_call.call_args.kwargs
asyncio.run(test_get_response())
response, sent = asyncio.run(run_call())
assert sent["model"] == "Qwen/Qwen2-1.5B-Instruct"
assert sent["prompt"] == "good morning"
assert sent["max_tokens"] == 10
assert response.choices[0].text == ") might be faster than then answering, and the added time it takes for the"
assert response.usage.total_tokens == 18
# test_async_text_completion()

View file

@ -307,7 +307,7 @@ async def test_chat_completion():
model="gpt-4",
messages=[{"role": "user", "content": "Hello!"}],
)
assert "is not available for this API key" in str(e)
assert "is not available for this API key" in str(e.value)
@pytest.mark.asyncio