diff --git a/litellm/proxy/response_polling/background_streaming.py b/litellm/proxy/response_polling/background_streaming.py index b0dcb69a82e..1e37b42f0ca 100644 --- a/litellm/proxy/response_polling/background_streaming.py +++ b/litellm/proxy/response_polling/background_streaming.py @@ -9,6 +9,7 @@ https://platform.openai.com/docs/api-reference/responses-streaming """ import asyncio import json +from typing import Any from fastapi import Request, Response @@ -85,7 +86,7 @@ async def background_streaming_task( # noqa: PLR0915 # Process streaming response following OpenAI events format # https://platform.openai.com/docs/api-reference/responses-streaming - output_items = {} # Track output items by ID + output_items: dict[str, dict[str, Any]] = {} # Track output items by ID accumulated_text = {} # Track accumulated text deltas by (item_id, content_index) # ResponsesAPIResponse fields to extract from response.completed diff --git a/litellm/proxy/response_polling/polling_handler.py b/litellm/proxy/response_polling/polling_handler.py index 121b128f06d..c47578c8d7b 100644 --- a/litellm/proxy/response_polling/polling_handler.py +++ b/litellm/proxy/response_polling/polling_handler.py @@ -2,8 +2,8 @@ Response Polling Handler for Background Responses with Cache """ import json -from typing import Any, Dict, Optional from datetime import datetime, timezone +from typing import Any, Dict, Optional from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid4 @@ -246,13 +246,9 @@ class ResponsePollingHandler: return False cache_key = self.get_cache_key(polling_id) - # Redis client's delete method - if hasattr(self.redis_cache, 'redis_async_client'): - async_client = self.redis_cache.init_async_client() - await async_client.delete(cache_key) - return True - - return False + # Use RedisCache's async_delete_cache method which handles Redis/RedisCluster + await self.redis_cache.async_delete_cache(cache_key) + return True def should_use_polling_for_request(