mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #22628 from BerriAI/litellm_oss_staging_03_02_2026
Litellm oss staging 03 02 2026
This commit is contained in:
commit
d0d1291c15
6 changed files with 208 additions and 3 deletions
|
|
@ -346,6 +346,8 @@ class DualCache(BaseCache):
|
|||
)
|
||||
try:
|
||||
if self.in_memory_cache is not None:
|
||||
if "ttl" not in kwargs and self.default_in_memory_ttl is not None:
|
||||
kwargs["ttl"] = self.default_in_memory_ttl
|
||||
await self.in_memory_cache.async_set_cache(key, value, **kwargs)
|
||||
|
||||
if self.redis_cache is not None and local_only is False:
|
||||
|
|
@ -367,6 +369,8 @@ class DualCache(BaseCache):
|
|||
)
|
||||
try:
|
||||
if self.in_memory_cache is not None:
|
||||
if "ttl" not in kwargs and self.default_in_memory_ttl is not None:
|
||||
kwargs["ttl"] = self.default_in_memory_ttl
|
||||
await self.in_memory_cache.async_set_cache_pipeline(
|
||||
cache_list=cache_list, **kwargs
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1095,6 +1095,12 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
|
||||
finish_reason = "tool_calls" if has_function_calls else "stop"
|
||||
|
||||
usage = None
|
||||
if response_data.get("usage"):
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
response_data.get("usage")
|
||||
)
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
|
|
@ -1102,7 +1108,8 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
|
|||
delta=Delta(content=""),
|
||||
finish_reason=finish_reason,
|
||||
)
|
||||
]
|
||||
],
|
||||
usage=usage
|
||||
)
|
||||
else:
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -2062,7 +2062,8 @@ class InitPassThroughEndpointHelpers:
|
|||
"""
|
||||
## CHECK IF MAPPED PASS THROUGH ENDPOINT
|
||||
for mapped_route in LiteLLMRoutes.mapped_pass_through_routes.value:
|
||||
if route.startswith(mapped_route):
|
||||
full_mapped_route = InitPassThroughEndpointHelpers._build_full_path_with_root(mapped_route)
|
||||
if route.startswith(full_mapped_route):
|
||||
return True
|
||||
|
||||
# Fast path: check if any registered route key contains this path
|
||||
|
|
|
|||
|
|
@ -1,9 +1,11 @@
|
|||
import asyncio
|
||||
import time
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.caching.redis_cache import RedisCache
|
||||
|
||||
|
||||
|
|
@ -56,3 +58,104 @@ async def test_dual_cache_async_batch_get_cache_rolls_back_redis_reservation_on_
|
|||
assert mock_async_batch_get_cache.call_count == 2
|
||||
assert "shared_a" not in dual_cache.last_redis_batch_access_time
|
||||
assert "shared_b" not in dual_cache.last_redis_batch_access_time
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_async_set_cache_injects_default_in_memory_ttl():
|
||||
"""
|
||||
Test that async_set_cache injects default_in_memory_ttl into kwargs
|
||||
when no explicit ttl is provided, matching the sync set_cache behavior.
|
||||
|
||||
Regression test for: async_set_cache was missing the TTL injection that
|
||||
sync set_cache has, causing InMemoryCache to use its own default_ttl (600s)
|
||||
instead of DualCache's default_in_memory_ttl.
|
||||
"""
|
||||
in_memory_cache = InMemoryCache(default_ttl=600)
|
||||
dual_cache = DualCache(
|
||||
in_memory_cache=in_memory_cache,
|
||||
default_in_memory_ttl=60,
|
||||
)
|
||||
|
||||
before = time.time()
|
||||
await dual_cache.async_set_cache(key="test_key", value="test_value")
|
||||
after = time.time()
|
||||
|
||||
# The TTL stored should reflect default_in_memory_ttl (60s), not
|
||||
# InMemoryCache's default_ttl (600s)
|
||||
expiry = in_memory_cache.ttl_dict["test_key"]
|
||||
assert expiry >= before + 60
|
||||
assert expiry <= after + 60
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_async_set_cache_respects_explicit_ttl():
|
||||
"""
|
||||
Test that async_set_cache does NOT override an explicitly provided ttl.
|
||||
"""
|
||||
in_memory_cache = InMemoryCache(default_ttl=600)
|
||||
dual_cache = DualCache(
|
||||
in_memory_cache=in_memory_cache,
|
||||
default_in_memory_ttl=60,
|
||||
)
|
||||
|
||||
before = time.time()
|
||||
await dual_cache.async_set_cache(key="test_key", value="test_value", ttl=30)
|
||||
after = time.time()
|
||||
|
||||
# The explicit ttl=30 should be used, not default_in_memory_ttl (60)
|
||||
expiry = in_memory_cache.ttl_dict["test_key"]
|
||||
assert expiry >= before + 30
|
||||
assert expiry <= after + 30
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_async_set_cache_pipeline_injects_default_in_memory_ttl():
|
||||
"""
|
||||
Test that async_set_cache_pipeline injects default_in_memory_ttl into kwargs
|
||||
when no explicit ttl is provided.
|
||||
"""
|
||||
in_memory_cache = InMemoryCache(default_ttl=600)
|
||||
dual_cache = DualCache(
|
||||
in_memory_cache=in_memory_cache,
|
||||
default_in_memory_ttl=60,
|
||||
)
|
||||
|
||||
cache_list = [("key_a", "value_a"), ("key_b", "value_b")]
|
||||
|
||||
before = time.time()
|
||||
await dual_cache.async_set_cache_pipeline(cache_list=cache_list)
|
||||
after = time.time()
|
||||
|
||||
for key in ["key_a", "key_b"]:
|
||||
expiry = in_memory_cache.ttl_dict[key]
|
||||
assert expiry >= before + 60
|
||||
assert expiry <= after + 60
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dual_cache_sync_and_async_set_cache_use_same_ttl():
|
||||
"""
|
||||
Test that sync set_cache and async async_set_cache produce the same TTL
|
||||
when no explicit ttl is provided, ensuring parity between the two paths.
|
||||
"""
|
||||
in_memory_sync = InMemoryCache(default_ttl=600)
|
||||
dual_cache_sync = DualCache(
|
||||
in_memory_cache=in_memory_sync,
|
||||
default_in_memory_ttl=60,
|
||||
)
|
||||
|
||||
in_memory_async = InMemoryCache(default_ttl=600)
|
||||
dual_cache_async = DualCache(
|
||||
in_memory_cache=in_memory_async,
|
||||
default_in_memory_ttl=60,
|
||||
)
|
||||
|
||||
dual_cache_sync.set_cache(key="test_key", value="test_value")
|
||||
await dual_cache_async.async_set_cache(key="test_key", value="test_value")
|
||||
|
||||
sync_expiry = in_memory_sync.ttl_dict["test_key"]
|
||||
async_expiry = in_memory_async.ttl_dict["test_key"]
|
||||
|
||||
# Both should use default_in_memory_ttl=60, so their expiry times
|
||||
# should be within a small tolerance of each other
|
||||
assert abs(sync_expiry - async_expiry) < 1.0
|
||||
|
|
|
|||
|
|
@ -738,7 +738,58 @@ def test_response_completed_with_message_only_emits_stop_finish_reason():
|
|||
)
|
||||
|
||||
|
||||
def test_function_call_done_does_not_emit_finish_reason():
|
||||
|
||||
def test_response_completed_preserves_usage_with_cached_tokens():
|
||||
"""
|
||||
Test that response.completed correctly translates Responses API usage
|
||||
(input_tokens_details) to chat completion usage (prompt_tokens_details).
|
||||
|
||||
This is a regression test for an issue where streaming with models that
|
||||
use the Responses API bridge (e.g. gpt-5.2-codex) would drop
|
||||
prompt_tokens_details, causing cached_tokens to always be None.
|
||||
"""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
OpenAiResponsesToChatCompletionStreamIterator,
|
||||
)
|
||||
|
||||
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
|
||||
|
||||
chunk = {
|
||||
"type": "response.completed",
|
||||
"response": {
|
||||
"id": "resp_789",
|
||||
"status": "completed",
|
||||
"output": [
|
||||
{
|
||||
"type": "message",
|
||||
"id": "msg_abc",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "Six"}],
|
||||
"status": "completed",
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 1226,
|
||||
"output_tokens": 5,
|
||||
"total_tokens": 1231,
|
||||
"input_tokens_details": {"cached_tokens": 1024},
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
result = iterator.chunk_parser(chunk)
|
||||
|
||||
assert result.usage is not None, "usage should be set on response.completed chunk"
|
||||
assert result.usage.prompt_tokens == 1226, "prompt_tokens should map from input_tokens"
|
||||
assert result.usage.completion_tokens == 5, "completion_tokens should map from output_tokens"
|
||||
assert result.usage.prompt_tokens_details is not None, "prompt_tokens_details should be set"
|
||||
assert result.usage.prompt_tokens_details.cached_tokens == 1024, (
|
||||
"cached_tokens should be preserved from input_tokens_details"
|
||||
)
|
||||
|
||||
|
||||
def test_function_call_done_emits_is_finished():
|
||||
"""
|
||||
Test that OUTPUT_ITEM_DONE for a function_call does NOT emit finish_reason.
|
||||
The response.completed event handles the terminal finish_reason correctly.
|
||||
|
|
|
|||
|
|
@ -2369,3 +2369,42 @@ def test_get_registered_pass_through_route_with_custom_root():
|
|||
|
||||
# Clean up
|
||||
_registered_pass_through_routes.clear()
|
||||
|
||||
|
||||
def test_mapped_pass_through_routes_with_server_root_path():
|
||||
"""
|
||||
Mapped passthrough routes (vertex_ai, bedrock, etc) should match
|
||||
even when SERVER_ROOT_PATH is set and the incoming route is prefixed.
|
||||
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/22272
|
||||
"""
|
||||
from litellm.proxy.pass_through_endpoints.pass_through_endpoints import (
|
||||
InitPassThroughEndpointHelpers,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_server_root_path"
|
||||
) as mock_get_root:
|
||||
mock_get_root.return_value = "/litellm"
|
||||
|
||||
# prefixed route should match mapped routes like /vertex_ai
|
||||
assert (
|
||||
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
|
||||
"/litellm/vertex_ai/v1/projects/foo"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
|
||||
"/litellm/bedrock/model/invoke"
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
# bare route without prefix should not match when root is set
|
||||
assert (
|
||||
InitPassThroughEndpointHelpers.is_registered_pass_through_route(
|
||||
"/vertex_ai/v1/projects/foo"
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue