Merge pull request #22628 from BerriAI/litellm_oss_staging_03_02_2026

Litellm oss staging 03 02 2026
This commit is contained in:
Sameer Kankute 2026-03-10 17:40:31 +05:30 • committed by GitHub
commit d0d1291c15
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 208 additions and 3 deletions

View file

@ -346,6 +346,8 @@ class DualCache(BaseCache):
)
try:
if self.in_memory_cache is not None:
if "ttl" not in kwargs and self.default_in_memory_ttl is not None:
kwargs["ttl"] = self.default_in_memory_ttl
await self.in_memory_cache.async_set_cache(key, value, **kwargs)
if self.redis_cache is not None and local_only is False:
@ -367,6 +369,8 @@ class DualCache(BaseCache):
)
try:
if self.in_memory_cache is not None:
if "ttl" not in kwargs and self.default_in_memory_ttl is not None:
kwargs["ttl"] = self.default_in_memory_ttl
await self.in_memory_cache.async_set_cache_pipeline(
cache_list=cache_list, **kwargs
)

View file

@ -1095,6 +1095,12 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
finish_reason = "tool_calls" if has_function_calls else "stop"
usage = None
if response_data.get("usage"):
from litellm.responses.utils import ResponseAPILoggingUtils
usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
response_data.get("usage")
)
return ModelResponseStream(
choices=[
StreamingChoices(
@ -1102,7 +1108,8 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
delta=Delta(content=""),
finish_reason=finish_reason,
)
]
],
usage=usage
)
else:
pass

View file

@ -2062,7 +2062,8 @@ class InitPassThroughEndpointHelpers:
"""
## CHECK IF MAPPED PASS THROUGH ENDPOINT
for mapped_route in LiteLLMRoutes.mapped_pass_through_routes.value:
if route.startswith(mapped_route):
full_mapped_route = InitPassThroughEndpointHelpers._build_full_path_with_root(mapped_route)
if route.startswith(full_mapped_route):
return True
# Fast path: check if any registered route key contains this path

View file

@ -1,9 +1,11 @@
import asyncio
import time
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from litellm.caching.dual_cache import DualCache
from litellm.caching.in_memory_cache import InMemoryCache
from litellm.caching.redis_cache import RedisCache
@ -56,3 +58,104 @@ async def test_dual_cache_async_batch_get_cache_rolls_back_redis_reservation_on_
assert mock_async_batch_get_cache.call_count == 2
assert "shared_a" not in dual_cache.last_redis_batch_access_time
assert "shared_b" not in dual_cache.last_redis_batch_access_time
@pytest.mark.asyncio
async def test_dual_cache_async_set_cache_injects_default_in_memory_ttl():
"""
Test that async_set_cache injects default_in_memory_ttl into kwargs
when no explicit ttl is provided, matching the sync set_cache behavior.
Regression test for: async_set_cache was missing the TTL injection that
sync set_cache has, causing InMemoryCache to use its own default_ttl (600s)
instead of DualCache's default_in_memory_ttl.
"""
in_memory_cache = InMemoryCache(default_ttl=600)
dual_cache = DualCache(
in_memory_cache=in_memory_cache,
default_in_memory_ttl=60,
)
before = time.time()
await dual_cache.async_set_cache(key="test_key", value="test_value")
after = time.time()
# The TTL stored should reflect default_in_memory_ttl (60s), not
# InMemoryCache's default_ttl (600s)
expiry = in_memory_cache.ttl_dict["test_key"]
assert expiry >= before + 60
assert expiry <= after + 60
@pytest.mark.asyncio
async def test_dual_cache_async_set_cache_respects_explicit_ttl():
"""
Test that async_set_cache does NOT override an explicitly provided ttl.
"""
in_memory_cache = InMemoryCache(default_ttl=600)
dual_cache = DualCache(
in_memory_cache=in_memory_cache,
default_in_memory_ttl=60,
)
before = time.time()
await dual_cache.async_set_cache(key="test_key", value="test_value", ttl=30)
after = time.time()
# The explicit ttl=30 should be used, not default_in_memory_ttl (60)
expiry = in_memory_cache.ttl_dict["test_key"]
assert expiry >= before + 30
assert expiry <= after + 30
@pytest.mark.asyncio
async def test_dual_cache_async_set_cache_pipeline_injects_default_in_memory_ttl():
"""
Test that async_set_cache_pipeline injects default_in_memory_ttl into kwargs
when no explicit ttl is provided.
"""
in_memory_cache = InMemoryCache(default_ttl=600)
dual_cache = DualCache(
in_memory_cache=in_memory_cache,
default_in_memory_ttl=60,
)
cache_list = [("key_a", "value_a"), ("key_b", "value_b")]
before = time.time()
await dual_cache.async_set_cache_pipeline(cache_list=cache_list)
after = time.time()
for key in ["key_a", "key_b"]:
expiry = in_memory_cache.ttl_dict[key]
assert expiry >= before + 60
assert expiry <= after + 60
@pytest.mark.asyncio
async def test_dual_cache_sync_and_async_set_cache_use_same_ttl():
"""
Test that sync set_cache and async async_set_cache produce the same TTL
when no explicit ttl is provided, ensuring parity between the two paths.
"""
in_memory_sync = InMemoryCache(default_ttl=600)
dual_cache_sync = DualCache(
in_memory_cache=in_memory_sync,
default_in_memory_ttl=60,
)
in_memory_async = InMemoryCache(default_ttl=600)
dual_cache_async = DualCache(
in_memory_cache=in_memory_async,
default_in_memory_ttl=60,
)
dual_cache_sync.set_cache(key="test_key", value="test_value")
await dual_cache_async.async_set_cache(key="test_key", value="test_value")
sync_expiry = in_memory_sync.ttl_dict["test_key"]
async_expiry = in_memory_async.ttl_dict["test_key"]
# Both should use default_in_memory_ttl=60, so their expiry times
# should be within a small tolerance of each other
assert abs(sync_expiry - async_expiry) < 1.0

View file

@ -738,7 +738,58 @@ def test_response_completed_with_message_only_emits_stop_finish_reason():
)
def test_function_call_done_does_not_emit_finish_reason():
def test_response_completed_preserves_usage_with_cached_tokens():
"""
Test that response.completed correctly translates Responses API usage
(input_tokens_details) to chat completion usage (prompt_tokens_details).
This is a regression test for an issue where streaming with models that
use the Responses API bridge (e.g. gpt-5.2-codex) would drop
prompt_tokens_details, causing cached_tokens to always be None.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.completed",
"response": {
"id": "resp_789",
"status": "completed",
"output": [
{
"type": "message",
"id": "msg_abc",
"role": "assistant",
"content": [{"type": "output_text", "text": "Six"}],
"status": "completed",
}
],
"usage": {
"input_tokens": 1226,
"output_tokens": 5,
"total_tokens": 1231,
"input_tokens_details": {"cached_tokens": 1024},
"output_tokens_details": {"reasoning_tokens": 0},
},
},
}
result = iterator.chunk_parser(chunk)
assert result.usage is not None, "usage should be set on response.completed chunk"
assert result.usage.prompt_tokens == 1226, "prompt_tokens should map from input_tokens"
assert result.usage.completion_tokens == 5, "completion_tokens should map from output_tokens"
assert result.usage.prompt_tokens_details is not None, "prompt_tokens_details should be set"
assert result.usage.prompt_tokens_details.cached_tokens == 1024, (
"cached_tokens should be preserved from input_tokens_details"
)
def test_function_call_done_emits_is_finished():
"""
Test that OUTPUT_ITEM_DONE for a function_call does NOT emit finish_reason.
The response.completed event handles the terminal finish_reason correctly.

View file

@ -2369,3 +2369,42 @@ def test_get_registered_pass_through_route_with_custom_root():
# Clean up
_registered_pass_through_routes.clear()
def test_mapped_pass_through_routes_with_server_root_path():
"""
Mapped passthrough routes (vertex_ai, bedrock, etc) should match
even when SERVER_ROOT_PATH is set and the incoming route is prefixed.
Regression test for https://github.com/BerriAI/litellm/issues/22272
"""
from litellm.proxy.pass_through_endpoints.pass_through_endpoints import (
InitPassThroughEndpointHelpers,
)
with patch(
"litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_server_root_path"
) as mock_get_root:
mock_get_root.return_value = "/litellm"
# prefixed route should match mapped routes like /vertex_ai
assert (
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
"/litellm/vertex_ai/v1/projects/foo"
)
is True
)
assert (
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
"/litellm/bedrock/model/invoke"
)
is True
)
# bare route without prefix should not match when root is set
assert (
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
"/vertex_ai/v1/projects/foo"
)
is False
)