diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 5b4187846ff..4a27aac9769 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -640,8 +640,8 @@ class Logging(LiteLLMLoggingBaseClass): self._own_session_id: str = session_id_var.get() self.function_id = function_id - self.streaming_chunks: list[object] = [] # for generating complete stream response - self.sync_streaming_chunks: list[object] = [] # for generating complete stream response + self.streaming_chunks: list[object] = [] + self.sync_streaming_chunks: list[object] = [] self.log_raw_request_response = log_raw_request_response self.raw_request_only = raw_request_only diff --git a/litellm/rust_bridge/public_call.py b/litellm/rust_bridge/public_call.py index 17ff7e52165..d107483ecc0 100644 --- a/litellm/rust_bridge/public_call.py +++ b/litellm/rust_bridge/public_call.py @@ -6,6 +6,33 @@ import inspect from collections.abc import Callable, Mapping, Sequence from typing import Final, cast # noqa: TID251 # narrows caller-owned containers without copying them +import litellm + +_INFERENCE_CONTEXT: Final = frozenset( + { + "model", + "messages", + "input", + "api_key", + "api_base", + "base_url", + "custom_llm_provider", + "extra_headers", + "timeout", + "request_timeout", + "callbacks", + "success_callback", + "failure_callback", + "metadata", + "litellm_metadata", + "litellm_call_id", + "litellm_trace_id", + "litellm_logging_obj", + "litellm_credential_name", + "proxy_server_request", + } +) + def signature(legacy: Callable[..., object]) -> inspect.Signature: return inspect.signature(legacy) @@ -43,37 +70,11 @@ def optional_sequence(value: object) -> Sequence[object] | None: def inference_decline_reason(parameters: tuple[str, ...], kwargs: Mapping[str, object]) -> str | None: - import litellm - if litellm.cache is not None or litellm.drop_params or litellm.modify_params: return "native inference does not implement the configured cache or parameter rewrites" - context: Final = frozenset( - { - "model", - "messages", - "input", - "api_key", - "api_base", - "base_url", - "custom_llm_provider", - "extra_headers", - "timeout", - "request_timeout", - "callbacks", - "success_callback", - "failure_callback", - "metadata", - "litellm_metadata", - "litellm_call_id", - "litellm_trace_id", - "litellm_logging_obj", - "litellm_credential_name", - "proxy_server_request", - } - ) for name, value in kwargs.items(): if value is None: continue - if name not in parameters and name not in context: + if name not in parameters and name not in _INFERENCE_CONTEXT: return f"native inference does not implement {name}" return None diff --git a/litellm/rust_bridge/responses/route_host.py b/litellm/rust_bridge/responses/route_host.py index 110dd02ab22..4f491064185 100644 --- a/litellm/rust_bridge/responses/route_host.py +++ b/litellm/rust_bridge/responses/route_host.py @@ -34,7 +34,7 @@ def decline_reason(request: LiteLLMResponsesRequest) -> str | None: if request.custom_llm_provider is None and "/" not in request.model: try: _, provider, _, _ = get_llm_provider(model=request.model) - except Exception: # noqa: BLE001 # unresolved models stay on the existing Python dispatch path + except litellm.exceptions.BadRequestError: return "native Responses could not resolve the provider" if provider != "openai": return "native HTTP responses provider"