mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
* test(e2e): move live-provider legacy tests into tests/e2e Port legacy tests that exercise real providers into the tests/e2e suites that own them, using the harness (/model/new plus deferred cleanup) and asserting on what the caller receives. Delete legacy tests already covered at equal or stronger strength by e2e, integration or unit tests, and drop the now empty ocr_testing CircleCI job * test(e2e): address review on the live-provider test move Assert the SSE error frame a client actually receives when a post_call guardrail blocks a stream, and require a tool call for every requested city before checking the answer. Restore the OCR matrix and its CircleCI job, the Claude Agent SDK streaming test, and test_async_create_batch, since their SDK-level and callback assertions have no equivalent in tests/e2e * test(e2e): accept both guardrail block shapes on a blocked stream A post_call block before the first chunk reaches the client as HTTP 400 with either a JSON error body or a single SSE error frame, depending on whether the block surfaced as an exception or an error chunk. Assert the policy message is present and the blocked output is absent in both * test(realtime): restore direct SDK realtime tests against OpenAI The e2e realtime tests go through the proxy and the remaining SDK tests either mock the upstream or assert less, so keep the direct litellm._arealtime tests with and without intent, and TestOpenAIRealtime::test_realtime_connection, in place * test: make realtime and Nova stream checks deterministic The direct SDK realtime tests now fail on a refused connection instead of skipping. The with-intent test asserts OpenAI rejects the exact intent value sent, which only happens when the intent is forwarded. The Nova /v1/messages stream test asserts stream structure, stop reason and usage instead of model wording * test(realtime): own intent forwarding with a unit test instead of a live rejection Assert litellm._arealtime passes the intent query param into the OpenAI realtime websocket URL, which is the behavior LiteLLM owns, and drop the live test that depended on OpenAI's rejection wording
157 lines
4.9 KiB
Python
157 lines
4.9 KiB
Python
import traceback
|
|
from litellm._uuid import uuid
|
|
import pytest
|
|
from dotenv import load_dotenv
|
|
from fastapi import Request
|
|
from fastapi.routing import APIRoute
|
|
|
|
load_dotenv()
|
|
import io
|
|
import time
|
|
import json
|
|
|
|
# this file is to test litellm/proxy
|
|
|
|
import litellm
|
|
import asyncio
|
|
from typing import Optional
|
|
from litellm.types.utils import StandardLoggingPayload, Usage, ModelInfoBase
|
|
from litellm.integrations.custom_logger import CustomLogger
|
|
|
|
|
|
class TestCustomLogger(CustomLogger):
|
|
def __init__(self):
|
|
self.recorded_usage: Optional[Usage] = None
|
|
self.standard_logging_payload: Optional[StandardLoggingPayload] = None
|
|
|
|
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
|
standard_logging_payload = kwargs.get("standard_logging_object")
|
|
self.standard_logging_payload = standard_logging_payload
|
|
print(
|
|
"standard_logging_payload",
|
|
json.dumps(standard_logging_payload, indent=4, default=str),
|
|
)
|
|
|
|
self.recorded_usage = Usage(
|
|
prompt_tokens=standard_logging_payload.get("prompt_tokens"),
|
|
completion_tokens=standard_logging_payload.get("completion_tokens"),
|
|
total_tokens=standard_logging_payload.get("total_tokens"),
|
|
)
|
|
pass
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_stream_token_counting_gpt_4o():
|
|
"""
|
|
When stream_options={"include_usage": True} logging callback tracks Usage == Usage from llm API
|
|
"""
|
|
custom_logger = TestCustomLogger()
|
|
litellm.logging_callback_manager.add_litellm_callback(custom_logger)
|
|
|
|
response = await litellm.acompletion(
|
|
model="gpt-5.5",
|
|
messages=[{"role": "user", "content": "Hello, how are you?" * 100}],
|
|
stream=True,
|
|
stream_options={"include_usage": True},
|
|
)
|
|
|
|
actual_usage = None
|
|
async for chunk in response:
|
|
if "usage" in chunk:
|
|
actual_usage = chunk["usage"]
|
|
print("chunk.usage", json.dumps(chunk["usage"], indent=4, default=str))
|
|
pass
|
|
|
|
await asyncio.sleep(2)
|
|
|
|
print("\n\n\n\n\n")
|
|
print(
|
|
"recorded_usage",
|
|
json.dumps(custom_logger.recorded_usage, indent=4, default=str),
|
|
)
|
|
print("\n\n\n\n\n")
|
|
|
|
assert actual_usage.prompt_tokens == custom_logger.recorded_usage.prompt_tokens
|
|
assert (
|
|
actual_usage.completion_tokens == custom_logger.recorded_usage.completion_tokens
|
|
)
|
|
assert actual_usage.total_tokens == custom_logger.recorded_usage.total_tokens
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_stream_token_counting_without_include_usage():
|
|
"""
|
|
When stream_options={"include_usage": True} is not passed, the usage tracked == usage from llm api chunk
|
|
|
|
by default, litellm passes `include_usage=True` for OpenAI API
|
|
"""
|
|
custom_logger = TestCustomLogger()
|
|
litellm.logging_callback_manager.add_litellm_callback(custom_logger)
|
|
|
|
response = await litellm.acompletion(
|
|
model="gpt-5.5",
|
|
messages=[{"role": "user", "content": "Hello, how are you?" * 100}],
|
|
stream=True,
|
|
)
|
|
|
|
actual_usage = None
|
|
async for chunk in response:
|
|
if "usage" in chunk:
|
|
actual_usage = chunk["usage"]
|
|
print("chunk.usage", json.dumps(chunk["usage"], indent=4, default=str))
|
|
pass
|
|
|
|
await asyncio.sleep(2)
|
|
|
|
print("\n\n\n\n\n")
|
|
print(
|
|
"recorded_usage",
|
|
json.dumps(custom_logger.recorded_usage, indent=4, default=str),
|
|
)
|
|
print("\n\n\n\n\n")
|
|
|
|
assert actual_usage.prompt_tokens == custom_logger.recorded_usage.prompt_tokens
|
|
assert (
|
|
actual_usage.completion_tokens == custom_logger.recorded_usage.completion_tokens
|
|
)
|
|
assert actual_usage.total_tokens == custom_logger.recorded_usage.total_tokens
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_stream_token_counting_with_redaction():
|
|
"""
|
|
When litellm.turn_off_message_logging=True is used, the usage tracked == usage from llm api chunk
|
|
"""
|
|
litellm.turn_off_message_logging = True
|
|
custom_logger = TestCustomLogger()
|
|
litellm.logging_callback_manager.add_litellm_callback(custom_logger)
|
|
|
|
response = await litellm.acompletion(
|
|
model="gpt-5.5",
|
|
messages=[{"role": "user", "content": "Hello, how are you?" * 100}],
|
|
stream=True,
|
|
)
|
|
|
|
actual_usage = None
|
|
async for chunk in response:
|
|
if "usage" in chunk:
|
|
actual_usage = chunk["usage"]
|
|
print("chunk.usage", json.dumps(chunk["usage"], indent=4, default=str))
|
|
pass
|
|
|
|
await asyncio.sleep(2)
|
|
|
|
print("\n\n\n\n\n")
|
|
print(
|
|
"recorded_usage",
|
|
json.dumps(custom_logger.recorded_usage, indent=4, default=str),
|
|
)
|
|
print("\n\n\n\n\n")
|
|
|
|
assert actual_usage.prompt_tokens == custom_logger.recorded_usage.prompt_tokens
|
|
assert (
|
|
actual_usage.completion_tokens == custom_logger.recorded_usage.completion_tokens
|
|
)
|
|
assert actual_usage.total_tokens == custom_logger.recorded_usage.total_tokens
|
|
|
|
|