Merge branch 'litellm_internal_staging' into litellm_grid-v4-e2e-tests-cZRwz

Resolve conflicts in the five unrelated CI-flake fixes I previously landed
on this branch -- staging shipped stronger versions (mocked HTTP for the
Fireworks tests, mocked image-fetch for the Gemini size-limit test, switched
the openapi-compliance test to the Interaction response schema instead of
dropping the assertion). Take staging's version of all five files and drop
my now-unreachable 429-skip lines from the Gemini test that the auto-merge
left behind.
This commit is contained in:
Mateo Wang 2026-05-16 16:19:38 +00:00
commit 4e8ac3151d
No known key found for this signature in database
6 changed files with 190 additions and 56 deletions

View file

@ -93,20 +93,32 @@ class TestFireworksAIAudioTranscription(BaseLLMAudioTranscriptionTest):
[True, False],
)
def test_document_inlining_example(disable_add_transform_inline_image_block):
litellm.set_verbose = True
if disable_add_transform_inline_image_block is True:
with pytest.raises(Exception):
completion = litellm.completion(
model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct",
"""
Document inlining appends ``#transform=inline`` to image/PDF URLs in the
outgoing request unless explicitly disabled. Assert the transform on the
serialized payload rather than making a live Fireworks call — the live
call only proved the model responded and broke whenever Fireworks rotated
its serverless model catalog.
"""
from unittest.mock import patch
from litellm import completion
from litellm.llms.custom_httpx.http_handler import HTTPHandler
client = HTTPHandler()
pdf_url = "https://storage.googleapis.com/fireworks-public/test/sample_resume.pdf"
with patch.object(client, "post") as mock_post:
try:
completion(
model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1",
messages=[
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {
"url": "https://storage.googleapis.com/fireworks-public/test/sample_resume.pdf"
},
"image_url": {"url": pdf_url},
},
{
"type": "text",
@ -116,22 +128,19 @@ def test_document_inlining_example(disable_add_transform_inline_image_block):
}
],
disable_add_transform_inline_image_block=disable_add_transform_inline_image_block,
client=client,
)
else:
try:
completion = litellm.completion(
model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct",
messages=[
{
"role": "user",
"content": "this is a test request, write a short poem",
},
],
disable_add_transform_inline_image_block=disable_add_transform_inline_image_block,
)
except litellm.NotFoundError as e:
pytest.skip(f"Fireworks model unavailable upstream (404): {e}")
print(completion)
except Exception as e:
print(e)
mock_post.assert_called_once()
json_data = json.loads(mock_post.call_args.kwargs["data"])
sent_url = json_data["messages"][0]["content"][0]["image_url"]["url"]
if disable_add_transform_inline_image_block is True:
assert sent_url == pdf_url
assert "#transform=inline" not in sent_url
else:
assert sent_url == pdf_url + "#transform=inline"
@pytest.mark.parametrize(
@ -218,7 +227,7 @@ def test_global_disable_flag_with_transform_messages_helper(monkeypatch):
) as mock_post:
try:
completion(
model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct",
model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1",
messages=[
{
"role": "user",

View file

@ -1605,13 +1605,49 @@ def test_gemini_31_flash_lite_reasoning_effort_minimal():
), "gemini-3.1-flash-lite-preview should use thinkingLevel, not thinkingBudget"
def test_gemini_image_size_limit_exceeded():
def test_gemini_image_size_limit_exceeded(monkeypatch):
"""
Test that large images exceeding MAX_IMAGE_URL_DOWNLOAD_SIZE_MB are rejected.
This validates that the 50MB default limit prevents downloading very large images
that could cause memory issues and pod crashes.
The image fetch is mocked (mirroring the LargeImageClient pattern in
tests/test_litellm/litellm_core_utils/test_image_handling.py) so the test
deterministically exercises the size-limit rejection path without any
external network dependency.
"""
from httpx import Request, Response
from litellm.litellm_core_utils.prompt_templates import image_handling
class LargeImageClient:
"""Returns a response whose Content-Length exceeds the 50MB limit."""
def get(self, url, follow_redirects=True):
size_bytes = int(100 * 1024 * 1024) # 100MB > 50MB default limit
return Response(
status_code=200,
headers={
"Content-Type": "image/jpeg",
"Content-Length": str(size_bytes),
},
# Empty body: the Content-Length header check in
# _process_image_response rejects the image before the body
# is ever streamed, so there's no need to allocate 100MB.
content=b"",
request=Request("GET", url),
)
# Bypass SSRF validation (which would resolve DNS / hit the network) and
# route straight to our mocked client.
monkeypatch.setattr(
image_handling,
"safe_get",
lambda client, url, **kw: client.get(url, follow_redirects=True),
)
monkeypatch.setattr(litellm, "module_level_client", LargeImageClient())
messages = [
{
"role": "user",
@ -1619,7 +1655,7 @@ def test_gemini_image_size_limit_exceeded():
{"type": "text", "text": "What is in this image?"},
{
"type": "image_url",
"image_url": "https://upload.wikimedia.org/wikipedia/commons/5/51/Blue_Marble_2002.jpg",
"image_url": "https://example.com/large-image.jpg",
},
],
}
@ -1629,8 +1665,6 @@ def test_gemini_image_size_limit_exceeded():
completion(model="gemini/gemini-2.5-flash-lite", messages=messages)
error_message = str(excinfo.value)
if "Status code: 429" in error_message or "Too Many Requests" in error_message:
pytest.skip(f"Wikimedia rate-limited the test fixture image: {error_message}")
assert "Image size" in error_message
assert "exceeds maximum allowed size" in error_message

View file

@ -1047,24 +1047,50 @@ def test_completion_openai_params(model):
def test_completion_fireworks_ai():
try:
litellm.set_verbose = True
messages = [
{"role": "system", "content": "You're a good bot"},
"""
Mocked so it does not depend on Fireworks' rotating serverless catalog
(no externally-verifiable model list exists). Asserts the request is
built correctly and the OpenAI-compatible response is parsed back.
"""
litellm.set_verbose = True
messages = [
{"role": "system", "content": "You're a good bot"},
{"role": "user", "content": "Hey"},
]
mock_response = MagicMock()
mock_response.status_code = 200
mock_response.headers = {"content-type": "application/json"}
mock_response.json.return_value = {
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1234567890,
"model": "accounts/fireworks/models/deepseek-v3p1",
"choices": [
{
"role": "user",
"content": "Hey",
},
]
"index": 0,
"message": {"role": "assistant", "content": "Hello there!"},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12},
}
mock_response.text = json.dumps(mock_response.json.return_value)
client = HTTPHandler()
with patch.object(client, "post", return_value=mock_response) as mock_post:
response = completion(
model="fireworks_ai/llama-v3p3-70b-instruct",
model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1",
messages=messages,
client=client,
)
print(response)
except litellm.NotFoundError as e:
pytest.skip(f"Fireworks model unavailable upstream (404): {e}")
except Exception as e:
pytest.fail(f"Error occurred: {e}")
mock_post.assert_called_once()
request_body = json.loads(mock_post.call_args.kwargs["data"])
assert "deepseek-v3p1" in request_body["model"]
assert request_body["messages"] == messages
assert response.choices[0].message.content == "Hello there!"
assert response.usage.total_tokens == 12
@pytest.mark.parametrize(

View file

@ -1171,7 +1171,7 @@ from litellm.llms.fireworks_ai.cost_calculator import get_base_model_for_pricing
@pytest.mark.parametrize(
"model, base_model",
[
("fireworks_ai/llama-v3p3-70b-instruct", "fireworks-ai-above-16b"),
("fireworks_ai/llama-v3p1-70b-instruct", "fireworks-ai-above-16b"),
],
)
def test_get_model_params_fireworks_ai(model, base_model):
@ -1182,21 +1182,47 @@ def test_get_model_params_fireworks_ai(model, base_model):
@pytest.mark.parametrize(
"model",
[
"fireworks_ai/llama-v3p3-70b-instruct",
"fireworks_ai/accounts/fireworks/models/deepseek-v3p1",
],
)
def test_completion_cost_fireworks_ai(model):
"""
Mocked so it does not depend on Fireworks' rotating serverless catalog.
Validates the Fireworks cost path: a parsed response with usage yields a
non-zero cost against the local cost map.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
messages = [{"role": "user", "content": "Hey, how's it going?"}]
try:
resp = litellm.completion(model=model, messages=messages)
except litellm.NotFoundError as e:
pytest.skip(f"Fireworks model unavailable upstream (404): {e}")
mock_response_data = {
"id": "chatcmpl-test",
"object": "chat.completion",
"created": 1234567890,
"model": model.split("fireworks_ai/")[-1],
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "Going great, thanks!"},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 8, "completion_tokens": 5, "total_tokens": 13},
}
mock_response = MagicMock()
mock_response.status_code = 200
mock_response.headers = {"content-type": "application/json"}
mock_response.json.return_value = mock_response_data
mock_response.text = json.dumps(mock_response_data)
sync_handler = HTTPHandler()
messages = [{"role": "user", "content": "Hey, how's it going?"}]
with patch.object(HTTPHandler, "post", return_value=mock_response):
resp = litellm.completion(model=model, messages=messages, client=sync_handler)
print(resp)
cost = completion_cost(completion_response=resp)
assert cost > 0
def test_cost_azure_openai_prompt_caching():

View file

@ -153,16 +153,20 @@ class TestResponseCompliance:
def test_interaction_response_fields(self, spec_dict):
"""Verify our InteractionsAPIResponse has correct fields."""
# The response is the Interaction schema
# Check CreateModelInteractionParams which includes output fields
schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"]
# The response is the dedicated `Interaction` schema. Google moved the
# output-only fields (notably the `steps` array, formerly `outputs`)
# off `CreateModelInteractionParams` and onto `Interaction`; the request
# schema no longer carries `steps`. Keep this aligned with the live spec.
schema = spec_dict["components"]["schemas"]["Interaction"]
# Output fields (readOnly).
output_fields = [
"id",
"status",
"created",
"updated",
"role",
"steps",
"usage",
]
@ -172,7 +176,8 @@ class TestResponseCompliance:
def test_status_enum_values(self, spec_dict):
"""Verify status enum values match spec."""
schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"]
# `status` is an output-only field; validate against the response schema.
schema = spec_dict["components"]["schemas"]["Interaction"]
status_prop = schema["properties"]["status"]
# Google Interactions API uses lowercase status values (updated Feb 2026)
expected_statuses = [

View file

@ -538,7 +538,13 @@ class TestProxyInitializationHelpers:
@patch("uvicorn.run")
@patch("builtins.print")
def test_keepalive_timeout_flag(self, mock_print, mock_uvicorn_run):
@patch("litellm.proxy.db.prisma_client.PrismaManager.setup_database")
@patch(
"litellm.proxy.db.prisma_client.should_update_prisma_schema", return_value=False
)
def test_keepalive_timeout_flag(
self, mock_should_update, mock_setup_db, mock_print, mock_uvicorn_run
):
"""Test that the keepalive_timeout flag is properly passed to uvicorn"""
from click.testing import CliRunner
@ -551,7 +557,18 @@ class TestProxyInitializationHelpers:
mock_key_mgmt = MagicMock()
mock_save_worker_config = MagicMock()
# Strip DATABASE_URL/DIRECT_URL so run_server doesn't enter the prisma
# DB-setup block (un-timeout'd `subprocess.run(["prisma"])` +
# migrate-deploy retry loop) — same isolation every other run_server
# test in this file uses.
clean_env = {
k: v
for k, v in os.environ.items()
if k not in ("DATABASE_URL", "DIRECT_URL")
}
with (
patch.dict(os.environ, clean_env, clear=True),
patch.dict(
"sys.modules",
{
@ -596,7 +613,13 @@ class TestProxyInitializationHelpers:
@patch("uvicorn.run")
@patch("builtins.print")
def test_timeout_worker_healthcheck_flag(self, mock_print, mock_uvicorn_run):
@patch("litellm.proxy.db.prisma_client.PrismaManager.setup_database")
@patch(
"litellm.proxy.db.prisma_client.should_update_prisma_schema", return_value=False
)
def test_timeout_worker_healthcheck_flag(
self, mock_should_update, mock_setup_db, mock_print, mock_uvicorn_run
):
"""Test that the --timeout_worker_healthcheck flag is threaded through to the uvicorn init helper."""
from click.testing import CliRunner
@ -609,7 +632,18 @@ class TestProxyInitializationHelpers:
mock_key_mgmt = MagicMock()
mock_save_worker_config = MagicMock()
# Strip DATABASE_URL/DIRECT_URL so run_server doesn't enter the prisma
# DB-setup block (un-timeout'd `subprocess.run(["prisma"])` +
# migrate-deploy retry loop) — same isolation every other run_server
# test in this file uses.
clean_env = {
k: v
for k, v in os.environ.items()
if k not in ("DATABASE_URL", "DIRECT_URL")
}
with (
patch.dict(os.environ, clean_env, clear=True),
patch.dict(
"sys.modules",
{