test_litellm_overhead_stream

This commit is contained in:
Ishaan Jaffer 2025-09-29 13:15:43 -07:00
parent 957991b33f
commit 6fee7eff07

View file

@ -23,7 +23,7 @@ import litellm
"vertex_ai/gemini-1.0-pro-vision-001", "vertex_ai/gemini-1.0-pro-vision-001",
], ],
) )
async def test_litellm_overhead(model): async def test_litellm_overhead_non_streaming(model):
""" """
- Test we can see the litellm overhead and that it is less than 40% of the total request time - Test we can see the litellm overhead and that it is less than 40% of the total request time
""" """
@ -37,20 +37,15 @@ async def test_litellm_overhead(model):
######################################################### #########################################################
# Specific cases for models # Specific cases for models
######################################################### #########################################################
if model == "vertex_ai/gemini-1.0-pro-vision-001": if model == "vertex_ai/gemini-1.0-pro-vision-001" or model == "openai/self_hosted":
kwargs["api_base"] = "https://exampleopenaiendpoint-production.up.railway.app/" kwargs["api_base"] = "https://exampleopenaiendpoint-production.up.railway.app/"
# warmup call for auth validation on vertex_ai models # warmup call for auth validation on vertex_ai models
await litellm.acompletion(**kwargs) await litellm.acompletion(**kwargs)
if model == "openai/self_hosted":
response = await litellm.acompletion(
model=model, response = await litellm.acompletion(
messages=[{"role": "user", "content": "Hello, world!"}], **kwargs
api_base="https://exampleopenaiendpoint-production.up.railway.app/", )
)
else:
response = await litellm.acompletion(
**kwargs
)
######################################################### #########################################################
# End of specific cases for models # End of specific cases for models
######################################################### #########################################################
@ -92,19 +87,22 @@ async def test_litellm_overhead_stream(model):
litellm._turn_on_debug() litellm._turn_on_debug()
start_time = datetime.now() start_time = datetime.now()
kwargs ={
"messages": [{"role": "user", "content": "Hello, world!"}],
"model": model,
"stream": True,
}
#########################################################
# Specific cases for models
#########################################################
if model == "openai/self_hosted": if model == "openai/self_hosted":
response = await litellm.acompletion( kwargs["api_base"] = "https://exampleopenaiendpoint-production.up.railway.app/"
model=model, # warmup call for auth validation on vertex_ai models
messages=[{"role": "user", "content": "Hello, world!"}], await litellm.acompletion(**kwargs)
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
stream=True, response = await litellm.acompletion(
) **kwargs
else: )
response = await litellm.acompletion(
model=model,
messages=[{"role": "user", "content": "Hello, world!"}],
stream=True,
)
async for chunk in response: async for chunk in response:
print() print()