From b0cb6557bd40387c3f9e31148ceb253ad363ec92 Mon Sep 17 00:00:00 2001 From: Harshit28j Date: Sat, 7 Mar 2026 03:52:17 +0530 Subject: [PATCH] fix(ci): stream chunk builder, GCS cache patch, files endpoint isolation - Fix usage extraction in stream_chunk_builder test to check for non-None - Fix GCS cache test to use patch.object avoiding litellm.caching shadowing - Add test isolation (monkeypatch, dependency overrides) to 3 files endpoint tests to prevent xdist parallel execution failures Co-Authored-By: Claude Opus 4.6 --- .../test_stream_chunk_builder.py | 2 +- tests/test_litellm/caching/test_gcs_cache.py | 7 +- .../test_files_endpoint.py | 268 ++++++++++-------- 3 files changed, 157 insertions(+), 120 deletions(-) diff --git a/tests/local_testing/test_stream_chunk_builder.py b/tests/local_testing/test_stream_chunk_builder.py index ddb1546097c..b14104a0399 100644 --- a/tests/local_testing/test_stream_chunk_builder.py +++ b/tests/local_testing/test_stream_chunk_builder.py @@ -666,7 +666,7 @@ def test_stream_chunk_builder_openai_audio_output_usage(): usage_obj: Optional[litellm.Usage] = None for index, chunk in enumerate(chunks): - if hasattr(chunk, "usage"): + if hasattr(chunk, "usage") and chunk.usage is not None: usage_obj = chunk.usage print(f"chunk usage: {chunk.usage}") print(f"index: {index}") diff --git a/tests/test_litellm/caching/test_gcs_cache.py b/tests/test_litellm/caching/test_gcs_cache.py index c346570bb05..dafc9ced2f0 100644 --- a/tests/test_litellm/caching/test_gcs_cache.py +++ b/tests/test_litellm/caching/test_gcs_cache.py @@ -6,6 +6,7 @@ import pytest sys.path.insert(0, os.path.abspath("../../..")) +import litellm.caching.gcs_cache as gcs_cache_module from litellm.caching.gcs_cache import GCSCache @@ -15,9 +16,9 @@ def mock_gcs_dependencies(): mock_sync_client = MagicMock() mock_async_client = AsyncMock() - with patch("litellm.caching.gcs_cache._get_httpx_client", return_value=mock_sync_client), \ - patch("litellm.caching.gcs_cache.get_async_httpx_client", return_value=mock_async_client), \ - patch("litellm.caching.gcs_cache.GCSBucketBase.sync_construct_request_headers", return_value={}): + with patch.object(gcs_cache_module, "_get_httpx_client", return_value=mock_sync_client), \ + patch.object(gcs_cache_module, "get_async_httpx_client", return_value=mock_async_client), \ + patch.object(gcs_cache_module.GCSBucketBase, "sync_construct_request_headers", return_value={}): yield { "sync_client": mock_sync_client, "async_client": mock_async_client, diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py index 8c732e0596e..161750864be 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py @@ -902,30 +902,35 @@ def test_managed_files_with_loadbalancing(mocker: MockerFixture, monkeypatch, ll """ Test that managed files work with loadbalancing when both target_model_names and enable_loadbalancing_on_batch_endpoints are enabled. - + This ensures that the priority order is correct: - managed files should take precedence over deprecated loadbalancing - managed files internally use llm_router.acreate_file() which provides loadbalancing """ from litellm.llms.base_llm.files.transformation import BaseFileEndpoints + from litellm.proxy._types import LitellmUserRoles from litellm.types.llms.openai import OpenAIFileObject + import litellm.proxy.proxy_server as ps + + monkeypatch.setattr("litellm.proxy.proxy_server.master_key", None) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) # Enable loadbalancing on batch endpoints monkeypatch.setattr("litellm.enable_loadbalancing_on_batch_endpoints", True) - + proxy_logging_obj = ProxyLogging( user_api_key_cache=DualCache(default_in_memory_ttl=1) ) proxy_logging_obj._add_proxy_hooks(llm_router) - + # Track calls to verify loadbalancing through router router_acreate_file_calls = [] - + class ManagedFilesWithLoadbalancing(BaseFileEndpoints): async def acreate_file(self, llm_router, create_file_request, target_model_names_list, litellm_parent_otel_span, user_api_key_dict): # Verify we receive the target model names assert len(target_model_names_list) > 0, "Should have target_model_names_list" - + # Simulate what managed files does - call llm_router.acreate_file for each model # This is where loadbalancing happens internally for model in target_model_names_list: @@ -933,7 +938,7 @@ def test_managed_files_with_loadbalancing(mocker: MockerFixture, monkeypatch, ll "model": model, "via_router": True }) - + # Return a managed file ID (base64 encoded) return OpenAIFileObject( id="litellm_managed_file_abc123", @@ -944,52 +949,59 @@ def test_managed_files_with_loadbalancing(mocker: MockerFixture, monkeypatch, ll purpose="batch", status="uploaded", ) - + async def afile_retrieve(self, file_id, litellm_parent_otel_span, llm_router): raise NotImplementedError("Not implemented for test") - + async def afile_list(self, purpose, litellm_parent_otel_span): raise NotImplementedError("Not implemented for test") - + async def afile_delete(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + async def afile_content(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + proxy_logging_obj.proxy_hook_mapping["managed_files"] = ManagedFilesWithLoadbalancing() monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", llm_router) monkeypatch.setattr( "litellm.proxy.proxy_server.proxy_logging_obj", proxy_logging_obj ) - - # Create batch file content - test_file_content = b'{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello"}]}}' - test_file = ("batch_data.jsonl", test_file_content, "application/jsonl") - - # Make request with both target_model_names AND enable_loadbalancing_on_batch_endpoints - response = client.post( - "/v1/files", - files={"file": test_file}, - data={ - "purpose": "batch", - "target_model_names": "azure-gpt-3-5-turbo,gpt-3.5-turbo", # Multiple models - }, - headers={"Authorization": "Bearer test-key"}, + + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="test-user" ) - - # Verify success - assert response.status_code == 200 - result = response.json() - assert result["id"] == "litellm_managed_file_abc123" - assert result["purpose"] == "batch" - - # Verify that managed files was called (via router for loadbalancing) - # This proves that managed files took precedence over deprecated loadbalancing - assert len(router_acreate_file_calls) == 2, "Should have called router for both models" - assert router_acreate_file_calls[0]["model"] == "azure-gpt-3-5-turbo" - assert router_acreate_file_calls[1]["model"] == "gpt-3.5-turbo" - assert all(call["via_router"] for call in router_acreate_file_calls), "All calls should go through router" + + try: + # Create batch file content + test_file_content = b'{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello"}]}}' + test_file = ("batch_data.jsonl", test_file_content, "application/jsonl") + + # Make request with both target_model_names AND enable_loadbalancing_on_batch_endpoints + response = client.post( + "/v1/files", + files={"file": test_file}, + data={ + "purpose": "batch", + "target_model_names": "azure-gpt-3-5-turbo,gpt-3.5-turbo", # Multiple models + }, + headers={"Authorization": "Bearer test-key"}, + ) + + # Verify success + assert response.status_code == 200 + result = response.json() + assert result["id"] == "litellm_managed_file_abc123" + assert result["purpose"] == "batch" + + # Verify that managed files was called (via router for loadbalancing) + # This proves that managed files took precedence over deprecated loadbalancing + assert len(router_acreate_file_calls) == 2, "Should have called router for both models" + assert router_acreate_file_calls[0]["model"] == "azure-gpt-3-5-turbo" + assert router_acreate_file_calls[1]["model"] == "gpt-3.5-turbo" + assert all(call["via_router"] for call in router_acreate_file_calls), "All calls should go through router" + finally: + app.dependency_overrides.clear() def test_create_file_with_nested_litellm_metadata( @@ -997,20 +1009,25 @@ def test_create_file_with_nested_litellm_metadata( ): """ Test that nested litellm_metadata is correctly parsed from form data in bracket notation. - + Regression test for: litellm_metadata[spend_logs_metadata][owner] format should be correctly parsed into nested dictionary structure. """ from litellm.llms.base_llm.files.transformation import BaseFileEndpoints + from litellm.proxy._types import LitellmUserRoles from litellm.types.llms.openai import OpenAIFileObject - + import litellm.proxy.proxy_server as ps + + monkeypatch.setattr("litellm.proxy.proxy_server.master_key", None) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) + proxy_logging_obj = ProxyLogging( user_api_key_cache=DualCache(default_in_memory_ttl=1) ) proxy_logging_obj._add_proxy_hooks(llm_router) - + captured_litellm_metadata = {} - + class DummyManagedFiles(BaseFileEndpoints): async def acreate_file(self, llm_router, create_file_request, target_model_names_list, litellm_parent_otel_span, user_api_key_dict): # Capture litellm_metadata for verification @@ -1022,7 +1039,7 @@ def test_create_file_with_nested_litellm_metadata( captured_litellm_metadata.update( getattr(create_file_request, "litellm_metadata", {}) ) - + return OpenAIFileObject( id="file-test-123", object="file", @@ -1032,54 +1049,61 @@ def test_create_file_with_nested_litellm_metadata( purpose="fine-tune", status="uploaded", ) - + async def afile_retrieve(self, file_id, litellm_parent_otel_span, llm_router): raise NotImplementedError("Not implemented for test") - + async def afile_list(self, purpose, litellm_parent_otel_span): raise NotImplementedError("Not implemented for test") - + async def afile_delete(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + async def afile_content(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + proxy_logging_obj.proxy_hook_mapping["managed_files"] = DummyManagedFiles() monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", llm_router) monkeypatch.setattr( "litellm.proxy.proxy_server.proxy_logging_obj", proxy_logging_obj ) - - test_file_content = b'{"prompt": "Hello", "completion": "Hi"}' - test_file = ("test.jsonl", test_file_content, "application/jsonl") - - # Test with nested litellm_metadata in bracket notation - response = client.post( - "/v1/files", - files={"file": test_file}, - data={ - "purpose": "fine-tune", - "target_model_names": "gpt-3.5-turbo", - "litellm_metadata[spend_logs_metadata][owner]": "john_doe", - "litellm_metadata[spend_logs_metadata][team]": "engineering", - "litellm_metadata[tags]": "production", - "litellm_metadata[environment]": "prod", - }, - headers={"Authorization": "Bearer test-key"}, + + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="test-user" ) - - # Verify success - assert response.status_code == 200 - result = response.json() - assert result["id"] == "file-test-123" - - # Verify nested metadata was correctly parsed - assert "spend_logs_metadata" in captured_litellm_metadata - assert captured_litellm_metadata["spend_logs_metadata"]["owner"] == "john_doe" - assert captured_litellm_metadata["spend_logs_metadata"]["team"] == "engineering" - assert captured_litellm_metadata["tags"] == "production" - assert captured_litellm_metadata["environment"] == "prod" + + try: + test_file_content = b'{"prompt": "Hello", "completion": "Hi"}' + test_file = ("test.jsonl", test_file_content, "application/jsonl") + + # Test with nested litellm_metadata in bracket notation + response = client.post( + "/v1/files", + files={"file": test_file}, + data={ + "purpose": "fine-tune", + "target_model_names": "gpt-3.5-turbo", + "litellm_metadata[spend_logs_metadata][owner]": "john_doe", + "litellm_metadata[spend_logs_metadata][team]": "engineering", + "litellm_metadata[tags]": "production", + "litellm_metadata[environment]": "prod", + }, + headers={"Authorization": "Bearer test-key"}, + ) + + # Verify success + assert response.status_code == 200 + result = response.json() + assert result["id"] == "file-test-123" + + # Verify nested metadata was correctly parsed + assert "spend_logs_metadata" in captured_litellm_metadata + assert captured_litellm_metadata["spend_logs_metadata"]["owner"] == "john_doe" + assert captured_litellm_metadata["spend_logs_metadata"]["team"] == "engineering" + assert captured_litellm_metadata["tags"] == "production" + assert captured_litellm_metadata["environment"] == "prod" + finally: + app.dependency_overrides.clear() def test_create_file_with_deep_nested_litellm_metadata( @@ -1087,19 +1111,24 @@ def test_create_file_with_deep_nested_litellm_metadata( ): """ Test that deeply nested litellm_metadata is correctly parsed from form data. - + Regression test for: litellm_metadata[a][b][c] format should be correctly parsed. """ from litellm.llms.base_llm.files.transformation import BaseFileEndpoints + from litellm.proxy._types import LitellmUserRoles from litellm.types.llms.openai import OpenAIFileObject - + import litellm.proxy.proxy_server as ps + + monkeypatch.setattr("litellm.proxy.proxy_server.master_key", None) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) + proxy_logging_obj = ProxyLogging( user_api_key_cache=DualCache(default_in_memory_ttl=1) ) proxy_logging_obj._add_proxy_hooks(llm_router) - + captured_litellm_metadata = {} - + class DummyManagedFiles(BaseFileEndpoints): async def acreate_file(self, llm_router, create_file_request, target_model_names_list, litellm_parent_otel_span, user_api_key_dict): if isinstance(create_file_request, dict): @@ -1110,7 +1139,7 @@ def test_create_file_with_deep_nested_litellm_metadata( captured_litellm_metadata.update( getattr(create_file_request, "litellm_metadata", {}) ) - + return OpenAIFileObject( id="file-test-456", object="file", @@ -1120,54 +1149,61 @@ def test_create_file_with_deep_nested_litellm_metadata( purpose="batch", status="uploaded", ) - + async def afile_retrieve(self, file_id, litellm_parent_otel_span, llm_router): raise NotImplementedError("Not implemented for test") - + async def afile_list(self, purpose, litellm_parent_otel_span): raise NotImplementedError("Not implemented for test") - + async def afile_delete(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + async def afile_content(self, file_id, litellm_parent_otel_span, llm_router, **data): raise NotImplementedError("Not implemented for test") - + proxy_logging_obj.proxy_hook_mapping["managed_files"] = DummyManagedFiles() monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", llm_router) monkeypatch.setattr( "litellm.proxy.proxy_server.proxy_logging_obj", proxy_logging_obj ) - - test_file_content = b'{"custom_id": "req-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo"}}' - test_file = ("nested.jsonl", test_file_content, "application/jsonl") - - # Test with deeply nested metadata - response = client.post( - "/v1/files", - files={"file": test_file}, - data={ - "purpose": "batch", - "target_model_names": "gpt-3.5-turbo", - "litellm_metadata[config][database][host]": "localhost", - "litellm_metadata[config][database][port]": "5432", - "litellm_metadata[config][cache][enabled]": "true", - }, - headers={"Authorization": "Bearer test-key"}, - ) - - # Verify success - assert response.status_code == 200, f"Expected 200, got {response.status_code}. Response: {response.text}" - result = response.json() - assert result["id"] == "file-test-456" - # Verify deeply nested metadata was correctly parsed - assert "config" in captured_litellm_metadata - assert "database" in captured_litellm_metadata["config"] - assert captured_litellm_metadata["config"]["database"]["host"] == "localhost" - assert captured_litellm_metadata["config"]["database"]["port"] == "5432" - assert "cache" in captured_litellm_metadata["config"] - assert captured_litellm_metadata["config"]["cache"]["enabled"] == "true" + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN, user_id="test-user" + ) + + try: + test_file_content = b'{"custom_id": "req-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo"}}' + test_file = ("nested.jsonl", test_file_content, "application/jsonl") + + # Test with deeply nested metadata + response = client.post( + "/v1/files", + files={"file": test_file}, + data={ + "purpose": "batch", + "target_model_names": "gpt-3.5-turbo", + "litellm_metadata[config][database][host]": "localhost", + "litellm_metadata[config][database][port]": "5432", + "litellm_metadata[config][cache][enabled]": "true", + }, + headers={"Authorization": "Bearer test-key"}, + ) + + # Verify success + assert response.status_code == 200, f"Expected 200, got {response.status_code}. Response: {response.text}" + result = response.json() + assert result["id"] == "file-test-456" + + # Verify deeply nested metadata was correctly parsed + assert "config" in captured_litellm_metadata + assert "database" in captured_litellm_metadata["config"] + assert captured_litellm_metadata["config"]["database"]["host"] == "localhost" + assert captured_litellm_metadata["config"]["database"]["port"] == "5432" + assert "cache" in captured_litellm_metadata["config"] + assert captured_litellm_metadata["config"]["cache"]["enabled"] == "true" + finally: + app.dependency_overrides.clear() # ---------------------------------------------------------------------------