mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix: stabilize proxy unit tests for parallel execution
- test_response_polling_handler: add xdist_group to prevent heavy import OOM - test_db_schema_migration: use temp dir for worker isolation, sync schema.prisma index - test_custom_tokenizer_bug: use lighter tokenizer to prevent OOM in parallel Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
parent
92745a650b
commit
fe8d609d3c
4 changed files with 27 additions and 27 deletions
|
|
@ -505,6 +505,7 @@ model LiteLLM_SpendLogs {
|
|||
agent_id String?
|
||||
proxy_server_request Json? @default("{}")
|
||||
@@index([startTime])
|
||||
@@index([startTime, request_id])
|
||||
@@index([end_user])
|
||||
@@index([session_id])
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6,6 +6,10 @@ causing token_counter to always use OpenAI tokenizer instead of the configured c
|
|||
|
||||
import pytest
|
||||
import litellm
|
||||
|
||||
# These tests load HuggingFace tokenizers which can cause OOM when run in parallel with -n 8.
|
||||
# Use lighter tokenizer (Xenova/llama-3-tokenizer) to reduce memory; isolate to prevent crashes.
|
||||
pytestmark = pytest.mark.xdist_group("heavy_tokenizer")
|
||||
import litellm.proxy.proxy_server
|
||||
from litellm.proxy.proxy_server import token_counter
|
||||
from litellm.proxy._types import TokenCountRequest
|
||||
|
|
@ -44,7 +48,7 @@ async def test_custom_tokenizer_from_model_info():
|
|||
"model_info": {
|
||||
"mode": "embedding",
|
||||
"custom_tokenizer": {
|
||||
"identifier": "intfloat/multilingual-e5-large-instruct",
|
||||
"identifier": "Xenova/llama-3-tokenizer", # Lighter for CI
|
||||
"revision": "main",
|
||||
"auth_token": None,
|
||||
},
|
||||
|
|
@ -71,9 +75,9 @@ async def test_custom_tokenizer_from_model_info():
|
|||
print("Model used:", response.model_used)
|
||||
print("Total tokens:", response.total_tokens)
|
||||
|
||||
# Verify that custom tokenizer (intfloat/multilingual-e5-large-instruct) was used
|
||||
# Verify that custom tokenizer (Xenova/llama-3-tokenizer) was used
|
||||
assert response.tokenizer_type == "huggingface_tokenizer", (
|
||||
f"Expected 'huggingface_tokenizer' (intfloat/multilingual-e5-large-instruct) "
|
||||
f"Expected 'huggingface_tokenizer' (custom_tokenizer from model_info) "
|
||||
f"but got '{response.tokenizer_type}'. "
|
||||
"This indicates the custom_tokenizer from model_info was not used."
|
||||
)
|
||||
|
|
@ -104,7 +108,7 @@ async def test_custom_tokenizer_with_llamacpp():
|
|||
},
|
||||
"model_info": {
|
||||
"custom_tokenizer": {
|
||||
"identifier": "intfloat/multilingual-e5-large-instruct",
|
||||
"identifier": "Xenova/llama-3-tokenizer",
|
||||
"revision": "main",
|
||||
"auth_token": None,
|
||||
},
|
||||
|
|
@ -129,15 +133,10 @@ async def test_custom_tokenizer_with_llamacpp():
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_multilingual_e5_embedding_model():
|
||||
async def test_custom_tokenizer_embedding_model():
|
||||
"""
|
||||
Test the exact real-world use case: intfloat/multilingual-e5-large-instruct
|
||||
tokenizer with a custom embedding endpoint.
|
||||
|
||||
This is the user's actual production scenario:
|
||||
- Custom embedding model endpoint (could be llama.cpp, vLLM, etc.)
|
||||
- Using intfloat/multilingual-e5-large-instruct for tokenization
|
||||
- Model served via OpenAI-compatible API
|
||||
Test custom tokenizer with embedding model (simulates intfloat/multilingual-e5
|
||||
or similar). Uses Xenova/llama-3-tokenizer for CI stability (lighter than e5).
|
||||
"""
|
||||
|
||||
llm_router = Router(
|
||||
|
|
@ -151,7 +150,7 @@ async def test_multilingual_e5_embedding_model():
|
|||
"model_info": {
|
||||
"mode": "embedding",
|
||||
"custom_tokenizer": {
|
||||
"identifier": "intfloat/multilingual-e5-large-instruct",
|
||||
"identifier": "Xenova/llama-3-tokenizer",
|
||||
"revision": "main",
|
||||
"auth_token": None,
|
||||
},
|
||||
|
|
@ -162,14 +161,13 @@ async def test_multilingual_e5_embedding_model():
|
|||
|
||||
setattr(litellm.proxy.proxy_server, "llm_router", llm_router)
|
||||
|
||||
# Test with multilingual content (what e5-large-instruct is designed for)
|
||||
response = await token_counter(
|
||||
request=TokenCountRequest(
|
||||
model="my-embedding-model",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "This is a multilingual test. C'est un test multilingue. 这是一个多语言测试。",
|
||||
"content": "This is a multilingual test. C'est un test multilingue.",
|
||||
}
|
||||
],
|
||||
)
|
||||
|
|
@ -179,10 +177,8 @@ async def test_multilingual_e5_embedding_model():
|
|||
f"Embedding model test - Tokenizer: {response.tokenizer_type}, Tokens: {response.total_tokens}"
|
||||
)
|
||||
|
||||
# Must use HuggingFace tokenizer with intfloat/multilingual-e5-large-instruct
|
||||
assert response.tokenizer_type == "huggingface_tokenizer", (
|
||||
f"The intfloat/multilingual-e5-large-instruct tokenizer was not used! "
|
||||
f"Got: {response.tokenizer_type}"
|
||||
f"Custom tokenizer from model_info was not used! Got: {response.tokenizer_type}"
|
||||
)
|
||||
assert response.total_tokens > 0
|
||||
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ def schema_setup(postgresql_my):
|
|||
return postgresql_my
|
||||
|
||||
|
||||
@pytest.mark.xdist_group("schema_migration")
|
||||
def test_aaaasschema_migration_check(schema_setup, monkeypatch):
|
||||
"""Test to check if schema requires migration"""
|
||||
# Set test database URL
|
||||
|
|
@ -26,15 +27,16 @@ def test_aaaasschema_migration_check(schema_setup, monkeypatch):
|
|||
|
||||
deploy_dir = Path("./litellm-proxy-extras/litellm_proxy_extras")
|
||||
source_migrations_dir = deploy_dir / "migrations"
|
||||
schema_path = Path("./schema.prisma")
|
||||
source_schema_path = Path("./schema.prisma")
|
||||
|
||||
# Create temporary migrations directory next to schema.prisma
|
||||
temp_migrations_dir = schema_path.parent / "migrations"
|
||||
# Use worker-specific temp directory to avoid races when running with -n 8.
|
||||
# Prisma expects migrations in <schema_dir>/migrations, so we create that layout.
|
||||
temp_base = Path(tempfile.mkdtemp(prefix="litellm_schema_migration_"))
|
||||
temp_migrations_dir = temp_base / "migrations"
|
||||
schema_path = temp_base / "schema.prisma"
|
||||
|
||||
try:
|
||||
# Copy migrations to correct location
|
||||
if temp_migrations_dir.exists():
|
||||
shutil.rmtree(temp_migrations_dir)
|
||||
shutil.copy(source_schema_path, schema_path)
|
||||
shutil.copytree(source_migrations_dir, temp_migrations_dir)
|
||||
|
||||
if not temp_migrations_dir.exists() or not any(temp_migrations_dir.iterdir()):
|
||||
|
|
@ -80,6 +82,6 @@ def test_aaaasschema_migration_check(schema_setup, monkeypatch):
|
|||
print("No schema changes detected. Migration not needed.")
|
||||
|
||||
finally:
|
||||
# Clean up: remove temporary migrations directory
|
||||
if temp_migrations_dir.exists():
|
||||
shutil.rmtree(temp_migrations_dir)
|
||||
# Clean up: remove temporary directory
|
||||
if temp_base.exists():
|
||||
shutil.rmtree(temp_base)
|
||||
|
|
|
|||
|
|
@ -626,6 +626,7 @@ class TestStreamingEventProcessing:
|
|||
assert UPDATE_INTERVAL * 1000 == 150 # 150 milliseconds
|
||||
|
||||
|
||||
@pytest.mark.xdist_group("heavy_imports")
|
||||
class TestBackgroundStreamingModule:
|
||||
"""Test cases for background_streaming module imports and structure"""
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue