diff --git a/litellm/constants.py b/litellm/constants.py index 487c1a1e020..5751e6e46af 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -2039,7 +2039,3 @@ BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY: Final = "batch_enqueued_token_limit" # Shared read-only empty mapping, for defaulting optional Mapping parameters without # constructing a fresh mutable dict at each call site. EMPTY_MAPPING: Final = MappingProxyType({}) - -# Marks a fallback re-entry as a mid-stream continuation, read by the deployment -# pre-call filter. Lives here so router and the filter share it without an import cycle. -MID_STREAM_CONTINUATION_KWARG: Final = "_mid_stream_continuation" diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index b6301929ce3..cd560f4c0d1 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -2414,8 +2414,8 @@ class CustomStreamWrapper: # Shared by the sync, async, and non-aiohttp iteration sites so answer # text and the disqualifying-content latch stay in step across all three. get: Final = getattr(delta, "get", None) - content: Final = get("content", "") if callable(get) else "" - self.response_uptil_now += content or "" + content: Final = get("content") if callable(get) else None + self.response_uptil_now += content if isinstance(content, str) else "" if not self._emitted_disqualifying_content and self._delta_disqualifies_continuation(delta): self._emitted_disqualifying_content = True diff --git a/litellm/router.py b/litellm/router.py index 64db136a791..226b5da7a83 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -62,7 +62,6 @@ from litellm.constants import ( DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER, DEFAULT_MAX_LRU_CACHE_SIZE, INTERNAL_CALL_ORIGIN_METADATA_KEY, - MID_STREAM_CONTINUATION_KWARG, OUTPUT_TOKEN_CEILING_PARAMS, ROUTING_REQUEST_TAGS_METADATA_KEY, RUNTIME_UPDATABLE_ROUTER_SETTINGS, @@ -2853,6 +2852,10 @@ class Router: ) initial_kwargs["original_function"] = self._acompletion if continue_after_content: + from litellm.router_utils.pre_call_checks.continuation_prefill_check import ( + MID_STREAM_CONTINUATION_KWARG, + ) + initial_kwargs["messages"] = self._build_completion_continuation_input( messages, e.generated_content ) @@ -3634,6 +3637,10 @@ class Router: "client": model_client, **kwargs, } + from litellm.router_utils.pre_call_checks.continuation_prefill_check import ( + MID_STREAM_CONTINUATION_KWARG, + ) + input_kwargs.pop("silent_model", None) input_kwargs.pop("include_fallback_errors", None) input_kwargs.pop(MID_STREAM_CONTINUATION_KWARG, None) diff --git a/litellm/router_utils/pre_call_checks/continuation_prefill_check.py b/litellm/router_utils/pre_call_checks/continuation_prefill_check.py index 688470b2b77..a4e05672879 100644 --- a/litellm/router_utils/pre_call_checks/continuation_prefill_check.py +++ b/litellm/router_utils/pre_call_checks/continuation_prefill_check.py @@ -11,11 +11,14 @@ from typing import Final from pydantic import TypeAdapter, ValidationError -from litellm.constants import MID_STREAM_CONTINUATION_KWARG from litellm.integrations.custom_logger import CustomLogger, Span from litellm.types.llms.openai import AllMessageValues from litellm.utils import supports_assistant_prefill +# Marks a fallback re-entry as a mid-stream continuation. Router sets it (via a +# lazy import) and this filter reads it; kept here to avoid a module-level cycle. +MID_STREAM_CONTINUATION_KWARG: Final = "_mid_stream_continuation" + _STR_KEYED_DICT_ADAPTER: Final = TypeAdapter(dict[str, object])