From 953a8b1b8552dcd72a97a669f9985156b4206bab Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 16:51:45 -0700 Subject: [PATCH 1/8] test(proxy): isolate run_server CLI tests from prisma DB-setup path test_keepalive_timeout_flag and test_timeout_worker_healthcheck_flag were the only run_server tests in test_proxy_cli.py that neither stripped DATABASE_URL/DIRECT_URL nor mocked the prisma DB path. When a DATABASE_URL is present (CI/env leak), run_server --local enters the DB block and blocks in the un-timeout'd subprocess.run(["prisma"]) at proxy_cli.py:987 plus the ProxyExtrasDBManager migrate-deploy retry loops, ~370s per test on the CI runner. --dist=loadscope pins both to one xdist worker, so the proxy-infra job appears stuck at 99% and hits the 20-min timeout. Apply the same isolation every other run_server test in this file already uses: mock PrismaManager.setup_database + should_update_prisma_schema and strip DATABASE_URL/DIRECT_URL. Full module drops from 31.7s to 2.9s locally; both tests fall off the slow list. --- tests/test_litellm/proxy/test_proxy_cli.py | 38 ++++++++++++++++++++-- 1 file changed, 36 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/proxy/test_proxy_cli.py b/tests/test_litellm/proxy/test_proxy_cli.py index 327200a6a95..46a55fc7468 100644 --- a/tests/test_litellm/proxy/test_proxy_cli.py +++ b/tests/test_litellm/proxy/test_proxy_cli.py @@ -538,7 +538,13 @@ class TestProxyInitializationHelpers: @patch("uvicorn.run") @patch("builtins.print") - def test_keepalive_timeout_flag(self, mock_print, mock_uvicorn_run): + @patch("litellm.proxy.db.prisma_client.PrismaManager.setup_database") + @patch( + "litellm.proxy.db.prisma_client.should_update_prisma_schema", return_value=False + ) + def test_keepalive_timeout_flag( + self, mock_should_update, mock_setup_db, mock_print, mock_uvicorn_run + ): """Test that the keepalive_timeout flag is properly passed to uvicorn""" from click.testing import CliRunner @@ -551,7 +557,18 @@ class TestProxyInitializationHelpers: mock_key_mgmt = MagicMock() mock_save_worker_config = MagicMock() + # Strip DATABASE_URL/DIRECT_URL so run_server doesn't enter the prisma + # DB-setup block (un-timeout'd `subprocess.run(["prisma"])` + + # migrate-deploy retry loop) — same isolation every other run_server + # test in this file uses. + clean_env = { + k: v + for k, v in os.environ.items() + if k not in ("DATABASE_URL", "DIRECT_URL") + } + with ( + patch.dict(os.environ, clean_env, clear=True), patch.dict( "sys.modules", { @@ -596,7 +613,13 @@ class TestProxyInitializationHelpers: @patch("uvicorn.run") @patch("builtins.print") - def test_timeout_worker_healthcheck_flag(self, mock_print, mock_uvicorn_run): + @patch("litellm.proxy.db.prisma_client.PrismaManager.setup_database") + @patch( + "litellm.proxy.db.prisma_client.should_update_prisma_schema", return_value=False + ) + def test_timeout_worker_healthcheck_flag( + self, mock_should_update, mock_setup_db, mock_print, mock_uvicorn_run + ): """Test that the --timeout_worker_healthcheck flag is threaded through to the uvicorn init helper.""" from click.testing import CliRunner @@ -609,7 +632,18 @@ class TestProxyInitializationHelpers: mock_key_mgmt = MagicMock() mock_save_worker_config = MagicMock() + # Strip DATABASE_URL/DIRECT_URL so run_server doesn't enter the prisma + # DB-setup block (un-timeout'd `subprocess.run(["prisma"])` + + # migrate-deploy retry loop) — same isolation every other run_server + # test in this file uses. + clean_env = { + k: v + for k, v in os.environ.items() + if k not in ("DATABASE_URL", "DIRECT_URL") + } + with ( + patch.dict(os.environ, clean_env, clear=True), patch.dict( "sys.modules", { From aebb6061f86a21b933749fa89d097c05daa8a4da Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 17:40:14 -0700 Subject: [PATCH 2/8] chore: retrigger CI From 11393f86f5b83e249a43dd53cf22f59414f43f0f Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 21:13:36 -0700 Subject: [PATCH 3/8] test(interactions): validate response fields against Interaction schema Google restructured the live spec at ai.google.dev/static/api/ interactions.openapi.json: the output-only fields (notably the `steps` array, formerly `outputs`) moved off the request schema `CreateModelInteractionParams` onto a dedicated `Interaction` response schema. The response-side tests still read from the request schema, so `test_interaction_response_fields` failed with "Output field 'steps' not in spec". Point `test_interaction_response_fields` and `test_status_enum_values` at the `Interaction` schema (the semantic response object; all output fields incl. `steps` present there). Request-side tests keep using `CreateModelInteractionParams` (all request fields verified still present). 13/13 pass against the current live spec. --- .../interactions/test_openapi_compliance.py | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/tests/test_litellm/interactions/test_openapi_compliance.py b/tests/test_litellm/interactions/test_openapi_compliance.py index 11d61d4c82e..1d3b6b8ae1e 100644 --- a/tests/test_litellm/interactions/test_openapi_compliance.py +++ b/tests/test_litellm/interactions/test_openapi_compliance.py @@ -153,12 +153,13 @@ class TestResponseCompliance: def test_interaction_response_fields(self, spec_dict): """Verify our InteractionsAPIResponse has correct fields.""" - # The response is the Interaction schema - # Check CreateModelInteractionParams which includes output fields - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] + # The response is the dedicated `Interaction` schema. Google moved the + # output-only fields (notably the `steps` array, formerly `outputs`) + # off `CreateModelInteractionParams` and onto `Interaction`; the request + # schema no longer carries `steps`. Keep this aligned with the live spec. + schema = spec_dict["components"]["schemas"]["Interaction"] - # Output fields (readOnly). Google renamed `outputs` → `steps` in the - # upstream spec; keep this list aligned with the live schema. + # Output fields (readOnly). output_fields = [ "id", "status", @@ -175,7 +176,8 @@ class TestResponseCompliance: def test_status_enum_values(self, spec_dict): """Verify status enum values match spec.""" - schema = spec_dict["components"]["schemas"]["CreateModelInteractionParams"] + # `status` is an output-only field; validate against the response schema. + schema = spec_dict["components"]["schemas"]["Interaction"] status_prop = schema["properties"]["status"] # Google Interactions API uses lowercase status values (updated Feb 2026) expected_statuses = [ From 39a1d438f23f88d1c88f3e74930ab221b3e450de Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 22:17:10 -0700 Subject: [PATCH 4/8] test(fireworks): replace deprecated llama-v3p3-70b-instruct model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fireworks removed llama-v3p3-70b-instruct from serverless, so every live test using it now fails with NotFoundError ("Model not found, inaccessible, and/or not deployed"). Swap the 6 references (3 files) to the currently-served accounts/fireworks/models/deepseek-v3p1 — the canonical model in Fireworks' current docs examples and present in LiteLLM's cost map. test_get_model_params_fireworks_ai is a pure pricing-heuristic test (no network) asserting the >16b branch, so it uses llama-v3p1-70b- instruct instead to keep the "fireworks-ai-above-16b" assertion and branch coverage intact. --- tests/llm_translation/test_fireworks_ai_translation.py | 6 +++--- tests/local_testing/test_completion.py | 2 +- tests/local_testing/test_completion_cost.py | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/llm_translation/test_fireworks_ai_translation.py b/tests/llm_translation/test_fireworks_ai_translation.py index 1cc6aabdca8..de065dc451b 100644 --- a/tests/llm_translation/test_fireworks_ai_translation.py +++ b/tests/llm_translation/test_fireworks_ai_translation.py @@ -97,7 +97,7 @@ def test_document_inlining_example(disable_add_transform_inline_image_block): if disable_add_transform_inline_image_block is True: with pytest.raises(Exception): completion = litellm.completion( - model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct", + model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=[ { "role": "user", @@ -119,7 +119,7 @@ def test_document_inlining_example(disable_add_transform_inline_image_block): ) else: completion = litellm.completion( - model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct", + model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=[ { "role": "user", @@ -215,7 +215,7 @@ def test_global_disable_flag_with_transform_messages_helper(monkeypatch): ) as mock_post: try: completion( - model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct", + model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=[ { "role": "user", diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index 6341fa78006..cff3fdef45b 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -1057,7 +1057,7 @@ def test_completion_fireworks_ai(): }, ] response = completion( - model="fireworks_ai/llama-v3p3-70b-instruct", + model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=messages, ) print(response) diff --git a/tests/local_testing/test_completion_cost.py b/tests/local_testing/test_completion_cost.py index 618287e1955..a9bbcffd8e7 100644 --- a/tests/local_testing/test_completion_cost.py +++ b/tests/local_testing/test_completion_cost.py @@ -1171,7 +1171,7 @@ from litellm.llms.fireworks_ai.cost_calculator import get_base_model_for_pricing @pytest.mark.parametrize( "model, base_model", [ - ("fireworks_ai/llama-v3p3-70b-instruct", "fireworks-ai-above-16b"), + ("fireworks_ai/llama-v3p1-70b-instruct", "fireworks-ai-above-16b"), ], ) def test_get_model_params_fireworks_ai(model, base_model): @@ -1182,7 +1182,7 @@ def test_get_model_params_fireworks_ai(model, base_model): @pytest.mark.parametrize( "model", [ - "fireworks_ai/llama-v3p3-70b-instruct", + "fireworks_ai/accounts/fireworks/models/deepseek-v3p1", ], ) def test_completion_cost_fireworks_ai(model): From 9770efe9e1f7dc850127dbf5fa92b2e4337b8bae Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 22:25:37 -0700 Subject: [PATCH 5/8] test(fireworks): mock document-inlining test instead of live call The live deepseek-v3p1 call kept hitting Fireworks NOT_FOUND because Fireworks rotates its serverless catalog and no externally-verifiable list exists. The [False] branch also never sent an image, so it only proved the model responded. Mock the HTTP post (mirrors test_global_disable_flag_with_transform_ messages_helper) and assert the real behavior: #transform=inline is appended to the PDF URL unless disabled. No network, no model dependency, and stronger coverage than the old live test. --- .../test_fireworks_ai_translation.py | 50 ++++++++++++------- 1 file changed, 31 insertions(+), 19 deletions(-) diff --git a/tests/llm_translation/test_fireworks_ai_translation.py b/tests/llm_translation/test_fireworks_ai_translation.py index de065dc451b..47b95c27ab7 100644 --- a/tests/llm_translation/test_fireworks_ai_translation.py +++ b/tests/llm_translation/test_fireworks_ai_translation.py @@ -93,10 +93,24 @@ class TestFireworksAIAudioTranscription(BaseLLMAudioTranscriptionTest): [True, False], ) def test_document_inlining_example(disable_add_transform_inline_image_block): - litellm.set_verbose = True - if disable_add_transform_inline_image_block is True: - with pytest.raises(Exception): - completion = litellm.completion( + """ + Document inlining appends ``#transform=inline`` to image/PDF URLs in the + outgoing request unless explicitly disabled. Assert the transform on the + serialized payload rather than making a live Fireworks call — the live + call only proved the model responded and broke whenever Fireworks rotated + its serverless model catalog. + """ + from unittest.mock import patch + + from litellm import completion + from litellm.llms.custom_httpx.http_handler import HTTPHandler + + client = HTTPHandler() + pdf_url = "https://storage.googleapis.com/fireworks-public/test/sample_resume.pdf" + + with patch.object(client, "post") as mock_post: + try: + completion( model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=[ { @@ -104,9 +118,7 @@ def test_document_inlining_example(disable_add_transform_inline_image_block): "content": [ { "type": "image_url", - "image_url": { - "url": "https://storage.googleapis.com/fireworks-public/test/sample_resume.pdf" - }, + "image_url": {"url": pdf_url}, }, { "type": "text", @@ -116,19 +128,19 @@ def test_document_inlining_example(disable_add_transform_inline_image_block): } ], disable_add_transform_inline_image_block=disable_add_transform_inline_image_block, + client=client, ) - else: - completion = litellm.completion( - model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", - messages=[ - { - "role": "user", - "content": "this is a test request, write a short poem", - }, - ], - disable_add_transform_inline_image_block=disable_add_transform_inline_image_block, - ) - print(completion) + except Exception as e: + print(e) + + mock_post.assert_called_once() + json_data = json.loads(mock_post.call_args.kwargs["data"]) + sent_url = json_data["messages"][0]["content"][0]["image_url"]["url"] + if disable_add_transform_inline_image_block is True: + assert sent_url == pdf_url + assert "#transform=inline" not in sent_url + else: + assert sent_url == pdf_url + "#transform=inline" @pytest.mark.parametrize( From b5db7ed37da21818c4defe030e3762447fe62e15 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 22:28:27 -0700 Subject: [PATCH 6/8] test(fireworks): mock remaining live smoke tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit test_completion_fireworks_ai and test_completion_cost_fireworks_ai made real Fireworks calls and broke whenever Fireworks rotated its serverless catalog (no externally-verifiable model list exists). They also asserted nothing — just printed. Mock the HTTP post and assert real behavior instead: the request is built with the right model/messages and the OpenAI-compatible response parses back; the cost path yields a non-zero cost against the local cost map. No network, no model dependency, stronger than the old smoke checks. --- tests/local_testing/test_completion.py | 50 ++++++++++++++++----- tests/local_testing/test_completion_cost.py | 35 +++++++++++++-- 2 files changed, 71 insertions(+), 14 deletions(-) diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index cff3fdef45b..cce6d33e799 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -1047,22 +1047,50 @@ def test_completion_openai_params(model): def test_completion_fireworks_ai(): - try: - litellm.set_verbose = True - messages = [ - {"role": "system", "content": "You're a good bot"}, + """ + Mocked so it does not depend on Fireworks' rotating serverless catalog + (no externally-verifiable model list exists). Asserts the request is + built correctly and the OpenAI-compatible response is parsed back. + """ + litellm.set_verbose = True + messages = [ + {"role": "system", "content": "You're a good bot"}, + {"role": "user", "content": "Hey"}, + ] + + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.headers = {"content-type": "application/json"} + mock_response.json.return_value = { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1234567890, + "model": "accounts/fireworks/models/deepseek-v3p1", + "choices": [ { - "role": "user", - "content": "Hey", - }, - ] + "index": 0, + "message": {"role": "assistant", "content": "Hello there!"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12}, + } + mock_response.text = json.dumps(mock_response.json.return_value) + + client = HTTPHandler() + with patch.object(client, "post", return_value=mock_response) as mock_post: response = completion( model="fireworks_ai/accounts/fireworks/models/deepseek-v3p1", messages=messages, + client=client, ) - print(response) - except Exception as e: - pytest.fail(f"Error occurred: {e}") + + mock_post.assert_called_once() + request_body = json.loads(mock_post.call_args.kwargs["data"]) + assert "deepseek-v3p1" in request_body["model"] + assert request_body["messages"] == messages + assert response.choices[0].message.content == "Hello there!" + assert response.usage.total_tokens == 12 @pytest.mark.parametrize( diff --git a/tests/local_testing/test_completion_cost.py b/tests/local_testing/test_completion_cost.py index a9bbcffd8e7..cf0c645615d 100644 --- a/tests/local_testing/test_completion_cost.py +++ b/tests/local_testing/test_completion_cost.py @@ -1186,14 +1186,43 @@ def test_get_model_params_fireworks_ai(model, base_model): ], ) def test_completion_cost_fireworks_ai(model): + """ + Mocked so it does not depend on Fireworks' rotating serverless catalog. + Validates the Fireworks cost path: a parsed response with usage yields a + non-zero cost against the local cost map. + """ os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") - messages = [{"role": "user", "content": "Hey, how's it going?"}] - resp = litellm.completion(model=model, messages=messages) # works fine + mock_response_data = { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1234567890, + "model": model.split("fireworks_ai/")[-1], + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Going great, thanks!"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 8, "completion_tokens": 5, "total_tokens": 13}, + } + + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.headers = {"content-type": "application/json"} + mock_response.json.return_value = mock_response_data + mock_response.text = json.dumps(mock_response_data) + + sync_handler = HTTPHandler() + messages = [{"role": "user", "content": "Hey, how's it going?"}] + + with patch.object(HTTPHandler, "post", return_value=mock_response): + resp = litellm.completion(model=model, messages=messages, client=sync_handler) - print(resp) cost = completion_cost(completion_response=resp) + assert cost > 0 def test_cost_azure_openai_prompt_caching(): From 1cf2d7f286a71a2df4d1244fa06343ff4c0e073e Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 22:34:57 -0700 Subject: [PATCH 7/8] test(gemini): de-flake test_gemini_image_size_limit_exceeded Mock the image fetch instead of downloading a 50MB+ image from upload.wikimedia.org. The runner was intermittently rate-limited (HTTP 429), so the code raised "Unable to fetch image ... Status code: 429" and the size-limit assertions failed even though pytest.raises(litellm.ImageFetchError) still matched. Mirror the established LargeImageClient pattern in tests/test_litellm/litellm_core_utils/test_image_handling.py: stub litellm.module_level_client with a response whose Content-Length exceeds the 50MB limit and bypass SSRF validation, so the size-limit rejection path is exercised deterministically with no external network dependency. --- tests/llm_translation/test_gemini.py | 37 ++++++++++++++++++++++++++-- 1 file changed, 35 insertions(+), 2 deletions(-) diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index 97b0aaee86b..037b8dad304 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -1594,13 +1594,46 @@ def test_gemini_31_flash_lite_reasoning_effort_minimal(): ), "gemini-3.1-flash-lite-preview should use thinkingLevel, not thinkingBudget" -def test_gemini_image_size_limit_exceeded(): +def test_gemini_image_size_limit_exceeded(monkeypatch): """ Test that large images exceeding MAX_IMAGE_URL_DOWNLOAD_SIZE_MB are rejected. This validates that the 50MB default limit prevents downloading very large images that could cause memory issues and pod crashes. + + The image fetch is mocked (mirroring the LargeImageClient pattern in + tests/test_litellm/litellm_core_utils/test_image_handling.py) so the test + deterministically exercises the size-limit rejection path without any + external network dependency. """ + from httpx import Request, Response + + from litellm.litellm_core_utils.prompt_templates import image_handling + + class LargeImageClient: + """Returns a response whose Content-Length exceeds the 50MB limit.""" + + def get(self, url, follow_redirects=True): + size_bytes = int(100 * 1024 * 1024) # 100MB > 50MB default limit + return Response( + status_code=200, + headers={ + "Content-Type": "image/jpeg", + "Content-Length": str(size_bytes), + }, + content=b"x" * size_bytes, + request=Request("GET", url), + ) + + # Bypass SSRF validation (which would resolve DNS / hit the network) and + # route straight to our mocked client. + monkeypatch.setattr( + image_handling, + "safe_get", + lambda client, url, **kw: client.get(url, follow_redirects=True), + ) + monkeypatch.setattr(litellm, "module_level_client", LargeImageClient()) + messages = [ { "role": "user", @@ -1608,7 +1641,7 @@ def test_gemini_image_size_limit_exceeded(): {"type": "text", "text": "What is in this image?"}, { "type": "image_url", - "image_url": "https://upload.wikimedia.org/wikipedia/commons/5/51/Blue_Marble_2002.jpg", + "image_url": "https://example.com/large-image.jpg", }, ], } From 8c537f70a3e9939856cd1954f941642e1c2a982a Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 15 May 2026 23:10:38 -0700 Subject: [PATCH 8/8] test(gemini): drop 100MB allocation in size-limit mock The Content-Length header check in _process_image_response rejects the image before the body is streamed, so the mock body never needs to be materialized. Use an empty body instead of b"x" * 100MB (addresses greptile/cursor review feedback). --- tests/llm_translation/test_gemini.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index 037b8dad304..01e40628441 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -1621,7 +1621,10 @@ def test_gemini_image_size_limit_exceeded(monkeypatch): "Content-Type": "image/jpeg", "Content-Length": str(size_bytes), }, - content=b"x" * size_bytes, + # Empty body: the Content-Length header check in + # _process_image_response rejects the image before the body + # is ever streamed, so there's no need to allocate 100MB. + content=b"", request=Request("GET", url), )