diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index f04ff8b6e78..2e854ced473 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -527,7 +527,7 @@ class LLMCachingHandler: model=model_name, data=[None] * len(kwargs_input_as_list), ) - final_embedding_cached_response._hidden_params["cache_hit"] = True + final_embedding_cached_response.hidden_params["cache_hit"] = True prompt_tokens = 0 aggregated_details: dict | None = None @@ -712,7 +712,7 @@ class LLMCachingHandler: ], usage=merged_usage, hidden_params={ - **cached._hidden_params, + **cached.hidden_params, "cache_hit": True, }, _response_headers=cached._response_headers, @@ -971,10 +971,10 @@ class LLMCachingHandler: response_obj: Final = ResponsesAPIResponse(**cached_result) if ( hasattr(response_obj, "_hidden_params") - and response_obj._hidden_params is not None - and isinstance(response_obj._hidden_params, dict) + and response_obj.hidden_params is not None + and isinstance(response_obj.hidden_params, dict) ): - response_obj._hidden_params["cache_hit"] = True + response_obj.hidden_params["cache_hit"] = True if _stream_replay_requested(kwargs): cached_result = CachedResponsesAPIStreamingIterator( diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 93d79bb3ac8..e1a3d685eb9 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -994,18 +994,18 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): # which contain important provider information like x-request-id raw_response_hidden_params: Final = getattr(raw_response, "_hidden_params", {}) if raw_response_hidden_params: - if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None: - model_response._hidden_params = {} + if not hasattr(model_response, "_hidden_params") or model_response.hidden_params is None: + model_response.hidden_params = {} # Merge the raw_response hidden params with model_response hidden params # Preserve existing keys in model_response but add/override with raw_response params for key, value in raw_response_hidden_params.items(): - if key == "additional_headers" and key in model_response._hidden_params: + if key == "additional_headers" and key in model_response.hidden_params: # Merge additional_headers to preserve both sets - existing_additional_headers = model_response._hidden_params.get("additional_headers", {}) + existing_additional_headers = model_response.hidden_params.get("additional_headers", {}) merged_headers = {**value, **existing_additional_headers} - model_response._hidden_params[key] = merged_headers + model_response.hidden_params[key] = merged_headers else: - model_response._hidden_params[key] = value + model_response.hidden_params[key] = value return model_response diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index f3abf50bbce..9fe192bc4d4 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2024,7 +2024,7 @@ def response_cost_calculator( else: if isinstance(response_object, BaseModel): if hasattr(response_object, "_hidden_params"): - provider_response_cost: Final = get_response_cost_from_hidden_params(response_object._hidden_params) + provider_response_cost: Final = get_response_cost_from_hidden_params(response_object.hidden_params) if provider_response_cost is not None: return provider_response_cost diff --git a/litellm/decisions/main.py b/litellm/decisions/main.py index 4864fcaa0e4..44a685e3235 100644 --- a/litellm/decisions/main.py +++ b/litellm/decisions/main.py @@ -206,7 +206,7 @@ def _parse_response( response.raise_for_status() payload: Final[object] = _DECISIONS_PAYLOAD_ADAPTER.validate_json(response.content) result: Final = _DECISIONS_RESPONSE_ADAPTER.validate_python(prepared.config.unwrap_response(payload)) - result._hidden_params.update( + result.hidden_params.update( { "model": f"{prepared.provider}/{prepared.upstream_model}", "custom_llm_provider": prepared.provider, diff --git a/litellm/images/main.py b/litellm/images/main.py index 7dc68dafecc..03d62dd75f5 100644 --- a/litellm/images/main.py +++ b/litellm/images/main.py @@ -230,7 +230,7 @@ def image_generation( else: model = "dall-e-2" custom_llm_provider = "openai" # default to dall-e-2 on openai - model_response._hidden_params["model"] = model + model_response.hidden_params["model"] = model openai_params: Final = [ "user", "request_timeout", diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index 39e95fbf687..924de8bd152 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -758,12 +758,18 @@ _NO_HEADERS: Final[Mapping[str, object]] = MappingProxyType({}) class _CarriesHiddenParams(Protocol): _hidden_params: dict[str, object] # mutable-ok: the responses billed here keep hidden params in a plain dict + @property + def hidden_params(self) -> dict[str, object]: ... # mutable-ok: API requires mutation + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: ... # mutable-ok: API requires mutation + def set_response_cost_in_hidden_params(response: _CarriesHiddenParams, cost: float | None) -> None: """Record a provider-reported cost where the cost calculator looks before the price map.""" if cost is None: return - hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor + hidden_params: Final = response.hidden_params additional_headers: Final[object] = hidden_params.get("additional_headers") merged: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params **(additional_headers if isinstance(additional_headers, Mapping) else _NO_HEADERS), @@ -779,7 +785,7 @@ _PROVIDER_HEADERS_ADAPTER: Final = TypeAdapter(Mapping[str, str]) def set_provider_response_headers_in_hidden_params( response: _CarriesHiddenParams, headers: httpx.Headers | Mapping[str, str] ) -> None: - hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor + hidden_params: Final = response.hidden_params existing_additional_headers: Final[object] = hidden_params.get("additional_headers") raw_headers: Final[dict[str, str]] = dict(headers) # mutable-ok: stored as the plain-dict hidden param additional_headers: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index b2c22880f5a..a4946dae5ae 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3253,8 +3253,8 @@ class Logging(LiteLLMLoggingBaseClass): should_compute_batch_data: Final = not has_explicit_batch_data and batch_cost_is_final(result) if has_explicit_batch_data: - result._hidden_params["response_cost"] = batch_cost - result._hidden_params["batch_models"] = batch_models + result.hidden_params["response_cost"] = batch_cost + result.hidden_params["batch_models"] = batch_models result._hidden_params["batch_successful_requests"] = batch_successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same result._hidden_params pattern as response_cost/batch_models above result._hidden_params["batch_failed_requests"] = batch_failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above result.usage = batch_usage @@ -3281,8 +3281,8 @@ class Logging(LiteLLMLoggingBaseClass): model_info=self.get_router_deployment_model_info(), ) - result._hidden_params["response_cost"] = batch_result.cost - result._hidden_params["batch_models"] = batch_result.models + result.hidden_params["response_cost"] = batch_result.cost + result.hidden_params["batch_models"] = batch_result.models result._hidden_params["batch_successful_requests"] = batch_result.successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above result._hidden_params["batch_failed_requests"] = batch_result.failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above result.usage = batch_result.usage diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 9ea730a873f..22bff04590d 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -512,7 +512,7 @@ class LiteLLMResponseObjectHandler: text_completion_response["choices"] = choices_list text_completion_response["usage"] = response.get("usage", None) - text_completion_response._hidden_params = HiddenParams(**response._hidden_params) + text_completion_response.hidden_params = HiddenParams(**response.hidden_params) return text_completion_response @staticmethod @@ -740,9 +740,9 @@ def convert_to_model_response_object( model_response_object._response_ms = (end_time - start_time).total_seconds() * 1000 if hidden_params is not None: - if model_response_object._hidden_params is None: - model_response_object._hidden_params = {} - model_response_object._hidden_params.update(hidden_params) + if model_response_object.hidden_params is None: + model_response_object.hidden_params = {} + model_response_object.hidden_params.update(hidden_params) if _response_headers is not None: model_response_object._response_headers = _response_headers @@ -780,7 +780,7 @@ def convert_to_model_response_object( ).total_seconds() * 1000 # return response latency in ms like openai if hidden_params is not None: - model_response_object._hidden_params = hidden_params + model_response_object.hidden_params = hidden_params if _response_headers is not None: model_response_object._response_headers = _response_headers @@ -826,13 +826,13 @@ def convert_to_model_response_object( setattr(model_response_object, "usage", tr_usage_object) if hidden_params is not None: - model_response_object._hidden_params = hidden_params + model_response_object.hidden_params = hidden_params # Store internally-calculated duration in _hidden_params for cost # tracking without exposing it in the response body. Must be set # after hidden_params assignment to avoid being overwritten. if "_audio_transcription_duration" in response_object: - model_response_object._hidden_params["audio_transcription_duration"] = response_object[ + model_response_object.hidden_params["audio_transcription_duration"] = response_object[ "_audio_transcription_duration" ] diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index be9a17a5dd2..9f27a52439e 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -265,7 +265,7 @@ class ChunkProcessor: return model_response # set hidden params from chunk to model_response if model_response is not None and hasattr(model_response, "_hidden_params"): - model_response._hidden_params = chunk.get("_hidden_params", {}) + model_response.hidden_params = chunk.get("_hidden_params", {}) return model_response @staticmethod @@ -841,7 +841,7 @@ class ChunkProcessor: elif (isinstance(chunk, ModelResponse) or isinstance(chunk, ModelResponseStream)) and hasattr( chunk, "_hidden_params" ): - usage_chunk = chunk._hidden_params.get("usage", None) + usage_chunk = chunk.hidden_params.get("usage", None) if isinstance(usage_chunk, dict): return Usage(**usage_chunk) diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 9d33f86d841..028401c6c11 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -205,6 +205,14 @@ def _provider_hidden_params( class CustomStreamWrapper: + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__( self, completion_stream, @@ -757,14 +765,14 @@ class CustomStreamWrapper: # must win over both caller-supplied hidden_params and the computed # custom_llm_provider/created_at values, so it comes last. if hidden_params is not None: - model_response._hidden_params = { + model_response.hidden_params = { **hidden_params, "custom_llm_provider": _logging_obj_llm_provider, "created_at": time.time(), **self._base_hidden_params, } else: - model_response._hidden_params = { + model_response.hidden_params = { "custom_llm_provider": _logging_obj_llm_provider, "created_at": time.time(), **self._base_hidden_params, @@ -795,7 +803,7 @@ class CustomStreamWrapper: self.response_id = id if id and isinstance(id, str) and id.strip(): - model_response._hidden_params["received_model_id"] = id + model_response.hidden_params["received_model_id"] = id if self.response_id is not None and isinstance(self.response_id, str): model_response.id = self.response_id @@ -1761,9 +1769,9 @@ class CustomStreamWrapper: _usage: Final[Usage | None] = getattr(response, "usage", None) _cost: Final = CustomStreamWrapper._resolve_provider_reported_cost(getattr(_usage, "cost", None)) if _cost is not None: - if "additional_headers" not in response._hidden_params: - response._hidden_params["additional_headers"] = {} - response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost + if "additional_headers" not in response.hidden_params: + response.hidden_params["additional_headers"] = {} + response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost def __next__(self) -> "ModelResponseStream": cache_hit = False @@ -1822,14 +1830,14 @@ class CustomStreamWrapper: if getattr(response, "usage", None) is not None: usage_to_preserve = response.usage if usage_to_preserve: - response._hidden_params["usage"] = usage_to_preserve + response.hidden_params["usage"] = usage_to_preserve obj_dict = response.model_dump() if "usage" in obj_dict: del obj_dict["usage"] - response = self.model_response_creator(chunk=obj_dict, hidden_params=response._hidden_params) + response = self.model_response_creator(chunk=obj_dict, hidden_params=response.hidden_params) ## check if empty is_empty = is_model_response_stream_empty(model_response=cast(ModelResponseStream, response)) @@ -1838,8 +1846,8 @@ class CustomStreamWrapper: # add usage as hidden param if self.sent_last_chunk is True and self.stream_options is None: usage = calculate_total_usage(chunks=self.chunks) - response._hidden_params["usage"] = usage - self._last_returned_hidden_params = response._hidden_params + response.hidden_params["usage"] = usage + self._last_returned_hidden_params = response.hidden_params # Add MCP metadata to final chunk if present response = self._add_mcp_metadata_to_final_chunk(response) # RETURN RESULT @@ -1936,7 +1944,7 @@ class CustomStreamWrapper: self.chunks.append(processed_chunk) if self.stream_options is None: # add usage as hidden param usage = calculate_total_usage(chunks=self.chunks) - processed_chunk._hidden_params["usage"] = usage + processed_chunk.hidden_params["usage"] = usage ## LOGGING executor.submit( self.run_success_logging_and_cache_storage, @@ -2034,7 +2042,7 @@ class CustomStreamWrapper: if "usage" in obj_dict: del obj_dict["usage"] processed_chunk = self.model_response_creator( - chunk=obj_dict, hidden_params=processed_chunk._hidden_params + chunk=obj_dict, hidden_params=processed_chunk.hidden_params ) is_empty = is_model_response_stream_empty( model_response=cast(ModelResponseStream, processed_chunk) @@ -2048,8 +2056,8 @@ class CustomStreamWrapper: # add usage as hidden param if self.sent_last_chunk is True and self.stream_options is None: usage = calculate_total_usage(chunks=self.chunks) - processed_chunk._hidden_params["usage"] = usage - self._last_returned_hidden_params = processed_chunk._hidden_params + processed_chunk.hidden_params["usage"] = usage + self._last_returned_hidden_params = processed_chunk.hidden_params # Call post-call streaming deployment hook for final chunk if self.sent_last_chunk is True: diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index fd67ebd5293..fc05472ef7e 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -2668,7 +2668,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): _message = json_mode_message model_response.choices[0].message = _message - model_response._hidden_params["original_response"] = completion_response["content"] + model_response.hidden_params["original_response"] = completion_response["content"] model_response.choices[0].finish_reason = cast( OpenAIChatCompletionFinishReason, map_finish_reason(completion_response["stop_reason"]), @@ -2686,7 +2686,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response.model = completion_response["model"] _hidden_params["provider_specific_fields"] = provider_specific_fields - model_response._hidden_params = _hidden_params + model_response.hidden_params = _hidden_params return model_response def get_prefix_prompt(self, messages: list[AllMessageValues]) -> str | None: diff --git a/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py index 4ef6c305cf5..24bc6b6f4a3 100644 --- a/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py @@ -165,7 +165,7 @@ class _CombinedChunkSplitter: chunk.usage = None hidden_params: Final = getattr(chunk, "_hidden_params", None) if isinstance(hidden_params, dict) and "usage" in hidden_params: - chunk._hidden_params = {key: value for key, value in hidden_params.items() if key != "usage"} + chunk.hidden_params = {key: value for key, value in hidden_params.items() if key != "usage"} @staticmethod def _split_by_payload_kind(chunk: "ModelResponseStream") -> "tuple[ModelResponseStream, ...]": diff --git a/litellm/llms/anthropic/pass_through/adapters/transformation.py b/litellm/llms/anthropic/pass_through/adapters/transformation.py index 022bc6337b5..cea381f31ff 100644 --- a/litellm/llms/anthropic/pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/pass_through/adapters/transformation.py @@ -1739,8 +1739,8 @@ class LiteLLMAnthropicMessagesAdapter: ) if getattr(response, "usage", None) is not None: litellm_usage_chunk: Usage | None = response.usage - elif hasattr(response, "_hidden_params") and "usage" in response._hidden_params: - litellm_usage_chunk = response._hidden_params["usage"] + elif hasattr(response, "_hidden_params") and "usage" in response.hidden_params: + litellm_usage_chunk = response.hidden_params["usage"] else: litellm_usage_chunk = None if litellm_usage_chunk is not None: diff --git a/litellm/llms/anthropic/pass_through/messages/response_cache.py b/litellm/llms/anthropic/pass_through/messages/response_cache.py index 5dc26c934ab..7f9ce8bea7a 100644 --- a/litellm/llms/anthropic/pass_through/messages/response_cache.py +++ b/litellm/llms/anthropic/pass_through/messages/response_cache.py @@ -46,7 +46,7 @@ class AnthropicMessagesStreamCacheWriter: self.collected_chunks: list[bytes] = [] # mutable-ok: rebuilding a tuple per SSE chunk is quadratic self.persisted = False self._hidden_params: dict[str, object] = dict( # mutable-ok: callers stamp cache_key in here - stream._hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING + stream.hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING ) @property diff --git a/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py b/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py index 417017cfb6e..60fa3373357 100644 --- a/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py +++ b/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py @@ -365,6 +365,14 @@ class AnthropicMessagesStreamingResponse: self.completion_stream = completion_stream self._hidden_params = hidden_params + @property + def hidden_params(self) -> AnthropicMessagesStreamHiddenParams: + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: AnthropicMessagesStreamHiddenParams) -> None: + self._hidden_params = hidden_params + @property def has_buffered_provider_output(self) -> bool: return getattr(self.completion_stream, "has_buffered_provider_output", False) is True diff --git a/litellm/llms/azure/audio_transcription/transformation.py b/litellm/llms/azure/audio_transcription/transformation.py index 17597da3895..858f122f65a 100644 --- a/litellm/llms/azure/audio_transcription/transformation.py +++ b/litellm/llms/azure/audio_transcription/transformation.py @@ -142,7 +142,7 @@ class AzureSpeechAudioTranscriptionConfig(BaseAudioTranscriptionConfig): text: Final = self._extract_text(payload) response: Final = TranscriptionResponse(text=text) - response._hidden_params = response_json + response.hidden_params = response_json return response def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index 35a41304e1b..64441fb0a43 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -1254,7 +1254,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): and litellm_params is not None and litellm_params.get("base_model", None) is not None ): - model_response._hidden_params["model"] = litellm_params.get("base_model", None) + model_response.hidden_params["model"] = litellm_params.get("base_model", None) # Azure image generation API doesn't support extra_body parameter extra_body: Final = optional_params.pop("extra_body", {}) diff --git a/litellm/llms/azure/search/transformation.py b/litellm/llms/azure/search/transformation.py index 45ad78df687..4981724f664 100644 --- a/litellm/llms/azure/search/transformation.py +++ b/litellm/llms/azure/search/transformation.py @@ -421,7 +421,7 @@ class BingGroundingSearchConfig(BaseSearchConfig): ) if get_secret_str(CONNECTION_ID_ENV): return response - response._hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0} + response.hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0} return response def get_error_class( diff --git a/litellm/llms/azure_ai/agents/handler.py b/litellm/llms/azure_ai/agents/handler.py index f7382190fca..6d5b754969f 100644 --- a/litellm/llms/azure_ai/agents/handler.py +++ b/litellm/llms/azure_ai/agents/handler.py @@ -231,9 +231,9 @@ class AzureAIAgentsHandler: model_response.model = model # Store thread_id for conversation continuity - if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None: - model_response._hidden_params = {} - model_response._hidden_params["thread_id"] = thread_id + if not hasattr(model_response, "_hidden_params") or model_response.hidden_params is None: + model_response.hidden_params = {} + model_response.hidden_params["thread_id"] = thread_id # Estimate token usage try: @@ -660,7 +660,7 @@ class AzureAIAgentsHandler: ], ) if thread_id: - final_chunk._hidden_params = {"thread_id": thread_id} + final_chunk.hidden_params = {"thread_id": thread_id} yield final_chunk return @@ -706,7 +706,7 @@ class AzureAIAgentsHandler: ], ) if thread_id: - chunk._hidden_params = {"thread_id": thread_id} + chunk.hidden_params = {"thread_id": thread_id} yield chunk diff --git a/litellm/llms/azure_ai/embed/cohere_transformation.py b/litellm/llms/azure_ai/embed/cohere_transformation.py index e60d419255c..453e48d6819 100644 --- a/litellm/llms/azure_ai/embed/cohere_transformation.py +++ b/litellm/llms/azure_ai/embed/cohere_transformation.py @@ -68,7 +68,7 @@ class AzureAICohereConfig: return image_embeddings_request, v1_embeddings_request, image_embedding_idx def _transform_response(self, response: EmbeddingResponse) -> EmbeddingResponse: - additional_headers: Final[dict | None] = response._hidden_params.get("additional_headers") + additional_headers: Final[dict | None] = response.hidden_params.get("additional_headers") if additional_headers: # CALCULATE USAGE input_tokens: Final[str | None] = additional_headers.get("llm_provider-num_tokens") diff --git a/litellm/llms/azure_ai/rerank/transformation.py b/litellm/llms/azure_ai/rerank/transformation.py index 64372c53f09..dfc709154cf 100644 --- a/litellm/llms/azure_ai/rerank/transformation.py +++ b/litellm/llms/azure_ai/rerank/transformation.py @@ -105,8 +105,8 @@ class AzureAIRerankConfig(CohereRerankConfig): optional_params=optional_params, litellm_params=litellm_params, ) - base_model: Final = self._get_base_model(rerank_response._hidden_params.get("llm_provider-azureml-model-group")) - rerank_response._hidden_params["model"] = base_model + base_model: Final = self._get_base_model(rerank_response.hidden_params.get("llm_provider-azureml-model-group")) + rerank_response.hidden_params["model"] = base_model return rerank_response def _get_base_model(self, azure_model_group: str | None) -> str | None: diff --git a/litellm/llms/base_llm/ocr/transformation.py b/litellm/llms/base_llm/ocr/transformation.py index f0c85dd379e..aeec06d4263 100644 --- a/litellm/llms/base_llm/ocr/transformation.py +++ b/litellm/llms/base_llm/ocr/transformation.py @@ -91,6 +91,14 @@ class OCRResponse(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def set_provider_native_response(self, native_response: Mapping[str, builtins.object]) -> None: """Keep the provider's own response payload alongside the normalized one.""" self._hidden_params[PROVIDER_NATIVE_RESPONSE_KEY] = native_response diff --git a/litellm/llms/base_llm/sandbox/transformation.py b/litellm/llms/base_llm/sandbox/transformation.py index d19c0744b08..24b228cf9b5 100644 --- a/litellm/llms/base_llm/sandbox/transformation.py +++ b/litellm/llms/base_llm/sandbox/transformation.py @@ -27,6 +27,14 @@ class ContainerHandle(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class CodeExecutionResult(LiteLLMPydanticObjectBase): """Passthrough of the sandbox's own execution output.""" @@ -42,6 +50,14 @@ class CodeExecutionResult(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class BaseSandboxConfig: """Provider-agnostic sandbox operations.""" diff --git a/litellm/llms/base_llm/search/transformation.py b/litellm/llms/base_llm/search/transformation.py index 4edbf260d99..85188c896f5 100644 --- a/litellm/llms/base_llm/search/transformation.py +++ b/litellm/llms/base_llm/search/transformation.py @@ -77,6 +77,14 @@ class SearchResponse(LiteLLMPydanticObjectBase): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class BaseSearchConfig: """ diff --git a/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py b/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py index acb0cc8dcb7..c0a9afca94f 100644 --- a/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py +++ b/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py @@ -449,15 +449,15 @@ class BedrockAmazonNovaCanvasImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} try: model_info: Final = get_model_info(model, custom_llm_provider="bedrock") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None and model_response.data: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( cost_per_image ) * len(model_response.data) except Exception: diff --git a/litellm/llms/bedrock/image_edit/stability_transformation.py b/litellm/llms/bedrock/image_edit/stability_transformation.py index f66dcf130b8..2f01c7de5ff 100644 --- a/litellm/llms/bedrock/image_edit/stability_transformation.py +++ b/litellm/llms/bedrock/image_edit/stability_transformation.py @@ -335,15 +335,15 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} # Set cost based on model model_info: Final = get_model_info(model, custom_llm_provider="bedrock") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( cost_per_image ) diff --git a/litellm/llms/bytez/chat/transformation.py b/litellm/llms/bytez/chat/transformation.py index 12338d83017..6a5f97e46b4 100644 --- a/litellm/llms/bytez/chat/transformation.py +++ b/litellm/llms/bytez/chat/transformation.py @@ -231,7 +231,7 @@ class BytezChatConfig(BaseConfig): model_response.usage = usage - model_response._hidden_params["additional_headers"] = raw_response.headers + model_response.hidden_params["additional_headers"] = raw_response.headers message.provider_specific_fields = { "ratelimit-limit": raw_response.headers.get("ratelimit-limit"), "ratelimit-remaining": raw_response.headers.get("ratelimit-remaining"), diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index 78ee8a86a65..a0ee907dd63 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -239,8 +239,8 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_headers) if not hasattr(completed_response, "_hidden_params"): setattr(completed_response, "_hidden_params", {}) - completed_response._hidden_params["additional_headers"] = processed_headers - completed_response._hidden_params["headers"] = raw_headers + completed_response.hidden_params["additional_headers"] = processed_headers + completed_response.hidden_params["headers"] = raw_headers def get_complete_url( self, diff --git a/litellm/llms/deepgram/audio_transcription/transformation.py b/litellm/llms/deepgram/audio_transcription/transformation.py index c5c9fe12423..a3812dac8ef 100644 --- a/litellm/llms/deepgram/audio_transcription/transformation.py +++ b/litellm/llms/deepgram/audio_transcription/transformation.py @@ -123,7 +123,7 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig): ] # Store full response in hidden params - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/deepinfra/rerank/transformation.py b/litellm/llms/deepinfra/rerank/transformation.py index a3d0482af0a..5196c44d504 100644 --- a/litellm/llms/deepinfra/rerank/transformation.py +++ b/litellm/llms/deepinfra/rerank/transformation.py @@ -213,7 +213,7 @@ class DeepinfraRerankConfig(BaseRerankConfig): rerank_response = RerankResponse(id=request_id or str(uuid.uuid4()), results=results, meta=meta) # Store additional information in hidden params - rerank_response._hidden_params = { + rerank_response.hidden_params = { "status": status, "runtime_ms": runtime_ms, "cost": cost, @@ -237,7 +237,7 @@ class DeepinfraRerankConfig(BaseRerankConfig): litellm_params=litellm_params, ) - rerank_response._hidden_params["model"] = model + rerank_response.hidden_params["model"] = model return rerank_response def get_supported_cohere_rerank_params(self, model: str) -> list: diff --git a/litellm/llms/e2b/sandbox/transformation.py b/litellm/llms/e2b/sandbox/transformation.py index d2702b116dc..64fd1857de2 100644 --- a/litellm/llms/e2b/sandbox/transformation.py +++ b/litellm/llms/e2b/sandbox/transformation.py @@ -90,7 +90,7 @@ class E2BSandboxConfig(BaseSandboxConfig): "domain": data.get("domain") or E2B_DEFAULT_DOMAIN, } ) - handle._hidden_params = { + handle.hidden_params = { "envd_access_token": data.get("envdAccessToken"), "traffic_access_token": data.get("trafficAccessToken"), "api_key": key, @@ -110,7 +110,7 @@ class E2BSandboxConfig(BaseSandboxConfig): ) -> CodeExecutionResult: handle: Final = self._as_handle(container) - token: Final = handle._hidden_params.get("envd_access_token") + token: Final = handle.hidden_params.get("envd_access_token") if not token: raise ValueError( "Cannot run code from a sandbox id alone. e2b secure sandboxes " @@ -119,7 +119,7 @@ class E2BSandboxConfig(BaseSandboxConfig): ) headers: Final = {"Content-Type": "application/json", "X-Access-Token": token} - traffic_token: Final = handle._hidden_params.get("traffic_access_token") + traffic_token: Final = handle.hidden_params.get("traffic_access_token") if traffic_token: headers["E2B-Traffic-Access-Token"] = traffic_token @@ -143,8 +143,8 @@ class E2BSandboxConfig(BaseSandboxConfig): **kwargs, ) -> bool: handle: Final = self._as_handle(container) - key: Final = api_key or handle._hidden_params.get("api_key") or self.validate_environment() - base: Final = api_base or handle._hidden_params.get("api_base") or E2B_API_BASE + key: Final = api_key or handle.hidden_params.get("api_key") or self.validate_environment() + base: Final = api_base or handle.hidden_params.get("api_base") or E2B_API_BASE try: response: Final = await self._http(client).delete( url=f"{base}/sandboxes/{handle.id}", @@ -161,7 +161,7 @@ class E2BSandboxConfig(BaseSandboxConfig): if isinstance(container, ContainerHandle): return container handle: Final = ContainerHandle(id=str(container), provider="e2b", domain=E2B_DEFAULT_DOMAIN) - handle._hidden_params = {} + handle.hidden_params = {} return handle @staticmethod diff --git a/litellm/llms/elevenlabs/audio_transcription/transformation.py b/litellm/llms/elevenlabs/audio_transcription/transformation.py index 0dbf61917e2..04d2717d64f 100644 --- a/litellm/llms/elevenlabs/audio_transcription/transformation.py +++ b/litellm/llms/elevenlabs/audio_transcription/transformation.py @@ -147,7 +147,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): ) # Store full response in hidden params - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py b/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py index d4d2941db43..81a1c4a8e72 100644 --- a/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py +++ b/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py @@ -239,10 +239,10 @@ class FalAIFluxProV11UltraConfig(FalAIBaseConfig): # Add additional metadata from Flux Pro response if hasattr(model_response, "_hidden_params"): if "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + model_response.hidden_params["seed"] = response_object["seed"] if "timings" in response_object: - model_response._hidden_params["timings"] = response_object["timings"] + model_response.hidden_params["timings"] = response_object["timings"] if "has_nsfw_concepts" in response_object: - model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] + model_response.hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py b/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py index 970addbd5a2..26406b3a2c5 100644 --- a/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py +++ b/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py @@ -190,6 +190,6 @@ class FalAIIdeogramV3Config(FalAIBaseConfig): ) if hasattr(model_response, "_hidden_params") and "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + model_response.hidden_params["seed"] = response_object["seed"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/imagen4_transformation.py b/litellm/llms/fal_ai/image_generation/imagen4_transformation.py index 3624b76a4a3..a6d170de614 100644 --- a/litellm/llms/fal_ai/image_generation/imagen4_transformation.py +++ b/litellm/llms/fal_ai/image_generation/imagen4_transformation.py @@ -236,6 +236,6 @@ class FalAIImagen4Config(FalAIBaseConfig): # Add seed metadata from Imagen4 response if hasattr(model_response, "_hidden_params"): if "seed" in response_data: - model_response._hidden_params["seed"] = response_data["seed"] + model_response.hidden_params["seed"] = response_data["seed"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py b/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py index 659011ae537..d850f337d45 100644 --- a/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py +++ b/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py @@ -270,10 +270,10 @@ class FalAIStableDiffusionConfig(FalAIBaseConfig): # Add additional metadata from Stable Diffusion response if hasattr(model_response, "_hidden_params"): if "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + model_response.hidden_params["seed"] = response_object["seed"] if "timings" in response_object: - model_response._hidden_params["timings"] = response_object["timings"] + model_response.hidden_params["timings"] = response_object["timings"] if "has_nsfw_concepts" in response_object: - model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] + model_response.hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] return model_response diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index 29a989a5cb0..26e75e2b4a6 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -756,7 +756,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig): tool_calls=optional_params.get("tools", None), ) - response._hidden_params = { + response.hidden_params = { "additional_headers": additional_headers, **_extract_fireworks_hidden_params(completion_response), } diff --git a/litellm/llms/gemini/interactions/transformation.py b/litellm/llms/gemini/interactions/transformation.py index 0e898147d90..f0e6bb12d39 100644 --- a/litellm/llms/gemini/interactions/transformation.py +++ b/litellm/llms/gemini/interactions/transformation.py @@ -274,8 +274,8 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig): verbose_logger.debug("Google AI Interactions response: %s", raw_json) response: Final = InteractionsAPIResponse(**raw_json) - response._hidden_params["headers"] = dict(raw_response.headers) - response._hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers)) + response.hidden_params["headers"] = dict(raw_response.headers) + response.hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers)) return response @@ -328,7 +328,7 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig): headers=dict(raw_response.headers), ) response: Final = InteractionsAPIResponse(**raw_json) - response._hidden_params["headers"] = dict(raw_response.headers) + response.hidden_params["headers"] = dict(raw_response.headers) return response def transform_delete_interaction_request( diff --git a/litellm/llms/huggingface/embedding/transformation.py b/litellm/llms/huggingface/embedding/transformation.py index 3fdd4abda73..334fc9a5d9a 100644 --- a/litellm/llms/huggingface/embedding/transformation.py +++ b/litellm/llms/huggingface/embedding/transformation.py @@ -465,7 +465,7 @@ class HuggingFaceEmbeddingConfig(BaseConfig): total_tokens=prompt_tokens + completion_tokens, ) setattr(model_response, "usage", usage) - model_response._hidden_params["original_response"] = completion_response + model_response.hidden_params["original_response"] = completion_response return model_response def transform_response( diff --git a/litellm/llms/manus/responses/transformation.py b/litellm/llms/manus/responses/transformation.py index f5733a152ba..ae4ad01bb08 100644 --- a/litellm/llms/manus/responses/transformation.py +++ b/litellm/llms/manus/responses/transformation.py @@ -225,8 +225,8 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def supports_native_websocket(self) -> bool: @@ -315,6 +315,6 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response diff --git a/litellm/llms/mistral/audio_transcription/transformation.py b/litellm/llms/mistral/audio_transcription/transformation.py index 9b1d343b94c..4fce7a95350 100644 --- a/litellm/llms/mistral/audio_transcription/transformation.py +++ b/litellm/llms/mistral/audio_transcription/transformation.py @@ -147,5 +147,5 @@ class MistralAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if "language" in response_json: response["language"] = response_json["language"] - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index 01b39ba3c40..53716f5a577 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -626,7 +626,7 @@ class OCIChatConfig(BaseConfig): else: model_response = handle_generic_response(response_json, model, model_response, raw_response) - model_response._hidden_params["additional_headers"] = raw_response.headers + model_response.hidden_params["additional_headers"] = raw_response.headers return model_response @track_llm_api_timing() diff --git a/litellm/llms/openai/completion/handler.py b/litellm/llms/openai/completion/handler.py index c7b59509eb0..66b13f2d1fd 100644 --- a/litellm/llms/openai/completion/handler.py +++ b/litellm/llms/openai/completion/handler.py @@ -204,7 +204,7 @@ class OpenAITextCompletion(BaseLLM): ) ## RESPONSE OBJECT response_obj: Final = TextCompletionResponse(**response_json) - response_obj._hidden_params.original_response = json.dumps(response_json) + response_obj.hidden_params.original_response = json.dumps(response_json) return response_obj except Exception as e: status_code: Final = getattr(e, "status_code", 500) diff --git a/litellm/llms/openai/completion/transformation.py b/litellm/llms/openai/completion/transformation.py index 383a67fd913..c638a88d5ca 100644 --- a/litellm/llms/openai/completion/transformation.py +++ b/litellm/llms/openai/completion/transformation.py @@ -111,7 +111,7 @@ class OpenAITextCompletionConfig(BaseTextCompletionConfig, OpenAIGPTConfig): if "model" in response_object: model_response_object.model = response_object["model"] - model_response_object._hidden_params["original_response"] = ( + model_response_object.hidden_params["original_response"] = ( response_object # track original response, if users make a litellm.text_completion() request, we can return the original response ) return model_response_object diff --git a/litellm/llms/openai/containers/transformation.py b/litellm/llms/openai/containers/transformation.py index 1a5211d5ff5..a3736116c89 100644 --- a/litellm/llms/openai/containers/transformation.py +++ b/litellm/llms/openai/containers/transformation.py @@ -165,11 +165,11 @@ class OpenAIContainerConfig(BaseContainerConfig): provider="openai", ) - if not hasattr(container_obj, "_hidden_params") or container_obj._hidden_params is None: - container_obj._hidden_params = {} - if "additional_headers" not in container_obj._hidden_params: - container_obj._hidden_params["additional_headers"] = {} - container_obj._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = container_cost + if not hasattr(container_obj, "_hidden_params") or container_obj.hidden_params is None: + container_obj.hidden_params = {} + if "additional_headers" not in container_obj.hidden_params: + container_obj.hidden_params["additional_headers"] = {} + container_obj.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = container_cost return container_obj diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 1452ecebca9..72d30876f19 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -591,8 +591,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def validate_environment(self, headers: dict, model: str, litellm_params: GenericLiteLLMParams | None) -> dict: @@ -839,8 +839,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): raw_response_headers: Final = dict(raw_response.headers) processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response @@ -921,8 +921,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response @@ -991,7 +991,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): ) response = ResponsesAPIResponse.model_construct(**raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response diff --git a/litellm/llms/openai_like/chat/transformation.py b/litellm/llms/openai_like/chat/transformation.py index e5d6cbb7e5e..b0770f1a7e1 100644 --- a/litellm/llms/openai_like/chat/transformation.py +++ b/litellm/llms/openai_like/chat/transformation.py @@ -117,7 +117,7 @@ class OpenAILikeChatConfig(OpenAIGPTConfig): returned_response.model = custom_llm_provider + "/" + (returned_response.model or "") if base_model is not None: - returned_response._hidden_params["model"] = base_model + returned_response.hidden_params["model"] = base_model return returned_response def transform_response( diff --git a/litellm/llms/openrouter/chat/transformation.py b/litellm/llms/openrouter/chat/transformation.py index 08d43c169f4..f8a551cd640 100644 --- a/litellm/llms/openrouter/chat/transformation.py +++ b/litellm/llms/openrouter/chat/transformation.py @@ -217,10 +217,10 @@ class OpenrouterConfig(OpenAIGPTConfig): if response_cost is not None: # Store cost in hidden params for the cost calculator to use if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( response_cost ) except Exception: diff --git a/litellm/llms/openrouter/image_edit/transformation.py b/litellm/llms/openrouter/image_edit/transformation.py index 3d46277a69e..d41e1303032 100644 --- a/litellm/llms/openrouter/image_edit/transformation.py +++ b/litellm/llms/openrouter/image_edit/transformation.py @@ -334,20 +334,18 @@ class OpenRouterImageEditConfig(BaseImageEditConfig): cost: Final = usage_data.get("cost") if cost is not None: if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost - ) + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(cost) cost_details: Final = usage_data.get("cost_details", {}) if cost_details: - if "response_cost_details" not in model_response._hidden_params: - model_response._hidden_params["response_cost_details"] = {} - model_response._hidden_params["response_cost_details"].update(cost_details) + if "response_cost_details" not in model_response.hidden_params: + model_response.hidden_params["response_cost_details"] = {} + model_response.hidden_params["response_cost_details"].update(cost_details) - model_response._hidden_params["model"] = response_json.get("model", model) + model_response.hidden_params["model"] = response_json.get("model", model) def _read_image_bytes(self, image: FileTypes) -> bytes: """Read raw bytes from various image input types.""" diff --git a/litellm/llms/openrouter/image_generation/transformation.py b/litellm/llms/openrouter/image_generation/transformation.py index cfca853e280..e18cfcc153d 100644 --- a/litellm/llms/openrouter/image_generation/transformation.py +++ b/litellm/llms/openrouter/image_generation/transformation.py @@ -227,20 +227,18 @@ class OpenRouterImageGenerationConfig(BaseImageGenerationConfig): cost: Final = usage_data.get("cost") if cost is not None: if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost - ) + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(cost) cost_details: Final = usage_data.get("cost_details", {}) if cost_details: - if "response_cost_details" not in model_response._hidden_params: - model_response._hidden_params["response_cost_details"] = {} - model_response._hidden_params["response_cost_details"].update(cost_details) + if "response_cost_details" not in model_response.hidden_params: + model_response.hidden_params["response_cost_details"] = {} + model_response.hidden_params["response_cost_details"].update(cost_details) - model_response._hidden_params["model"] = response_json.get("model", model) + model_response.hidden_params["model"] = response_json.get("model", model) def get_complete_url( self, diff --git a/litellm/llms/opensandbox/sandbox/transformation.py b/litellm/llms/opensandbox/sandbox/transformation.py index 5126db9bbc9..bf1c3314c0d 100644 --- a/litellm/llms/opensandbox/sandbox/transformation.py +++ b/litellm/llms/opensandbox/sandbox/transformation.py @@ -115,7 +115,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): ) handle: Final = ContainerHandle(id=sandbox_id, provider="opensandbox", domain=base) - handle._hidden_params = { + handle.hidden_params = { "api_base": base, "api_key": key, "execd_endpoint": endpoint, @@ -147,9 +147,9 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): poll_interval=(float(poll_interval) if poll_interval is not None else DEFAULT_POLL_INTERVAL), client=client, ) - endpoint: Final = str(handle._hidden_params["execd_endpoint"]) - endpoint_headers: Final = self._as_str_dict(handle._hidden_params.get("execd_headers")) - base: Final = str(handle._hidden_params.get("api_base") or handle.domain or self._api_base(api_base)) + endpoint: Final = str(handle.hidden_params["execd_endpoint"]) + endpoint_headers: Final = self._as_str_dict(handle.hidden_params.get("execd_headers")) + base: Final = str(handle.hidden_params.get("api_base") or handle.domain or self._api_base(api_base)) lines: Final = await self._post_code( url=f"{self._endpoint_base_url(endpoint, base)}/code", headers={ @@ -176,7 +176,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): **kwargs, ) -> bool: handle: Final = self._as_handle(container, api_base=api_base) - base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base)) + base: Final = str(handle.hidden_params.get("api_base") or self._api_base(api_base)) key: Final = self._api_key(api_key=api_key, handle=handle) try: response: Final = await self._http(client).delete( @@ -201,12 +201,12 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): client: AsyncHTTPHandler | None, ) -> ContainerHandle: handle: Final = self._as_handle(container, api_base=api_base) - if handle._hidden_params.get("execd_endpoint"): + if handle.hidden_params.get("execd_endpoint"): return handle - base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base)) + base: Final = str(handle.hidden_params.get("api_base") or self._api_base(api_base)) key: Final = self._api_key(api_key=api_key, handle=handle) - resolved_use_server_proxy: Final = bool(handle._hidden_params.get("use_server_proxy", use_server_proxy)) + resolved_use_server_proxy: Final = bool(handle.hidden_params.get("use_server_proxy", use_server_proxy)) endpoint, endpoint_headers = await self._wait_for_execd_endpoint( sandbox_id=handle.id, api_base=base, @@ -217,8 +217,8 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): poll_interval=poll_interval, ) handle.domain = base - handle._hidden_params = { - **handle._hidden_params, + handle.hidden_params = { + **handle.hidden_params, "api_base": base, "api_key": key, "execd_endpoint": endpoint, @@ -329,8 +329,8 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): def _api_key(self, *, api_key: str | None, handle: ContainerHandle) -> str: if api_key is not None: return api_key - if "api_key" in handle._hidden_params: - return str(handle._hidden_params["api_key"]) + if "api_key" in handle.hidden_params: + return str(handle.hidden_params["api_key"]) return self.validate_environment() @staticmethod @@ -421,7 +421,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): provider="opensandbox", domain=OpenSandboxSandboxConfig._api_base(api_base), ) - handle._hidden_params = {} + handle.hidden_params = {} return handle @staticmethod diff --git a/litellm/llms/ovhcloud/audio_transcription/transformation.py b/litellm/llms/ovhcloud/audio_transcription/transformation.py index 086c1afbcd4..3ae5f6fa69c 100644 --- a/litellm/llms/ovhcloud/audio_transcription/transformation.py +++ b/litellm/llms/ovhcloud/audio_transcription/transformation.py @@ -163,5 +163,5 @@ class OVHCloudAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if duration is not None: response_json["duration"] = duration - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/predibase/chat/transformation.py b/litellm/llms/predibase/chat/transformation.py index 69924396a1e..bb9b808df0a 100644 --- a/litellm/llms/predibase/chat/transformation.py +++ b/litellm/llms/predibase/chat/transformation.py @@ -260,7 +260,7 @@ class PredibaseConfig(BaseConfig): if k.startswith("x-"): response_headers[f"llm_provider-{k}"] = v - model_response._hidden_params["additional_headers"] = response_headers + model_response.hidden_params["additional_headers"] = response_headers return model_response diff --git a/litellm/llms/scaleway/audio_transcription/transformation.py b/litellm/llms/scaleway/audio_transcription/transformation.py index 525a43ad86d..7ebf4475355 100644 --- a/litellm/llms/scaleway/audio_transcription/transformation.py +++ b/litellm/llms/scaleway/audio_transcription/transformation.py @@ -147,5 +147,5 @@ class ScalewayAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if "language" in response_json: response["language"] = response_json["language"] - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/snowflake/chat/transformation.py b/litellm/llms/snowflake/chat/transformation.py index cc51e1162e3..a6c060732c5 100644 --- a/litellm/llms/snowflake/chat/transformation.py +++ b/litellm/llms/snowflake/chat/transformation.py @@ -558,7 +558,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig): returned_response.model = "snowflake/" + (returned_response.model or "") if model is not None: - returned_response._hidden_params["model"] = model + returned_response.hidden_params["model"] = model return returned_response @@ -623,7 +623,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig): model_response.id = response_json.get("id", "") if model is not None: - model_response._hidden_params["model"] = model + model_response.hidden_params["model"] = model return model_response diff --git a/litellm/llms/snowflake/embedding/transformation.py b/litellm/llms/snowflake/embedding/transformation.py index 6aa66de1db5..ec482337e52 100644 --- a/litellm/llms/snowflake/embedding/transformation.py +++ b/litellm/llms/snowflake/embedding/transformation.py @@ -62,7 +62,7 @@ class SnowflakeEmbeddingConfig(SnowflakeBaseConfig, BaseEmbeddingConfig): returned_response.model = "snowflake/" + (returned_response.model or "") if model is not None: - returned_response._hidden_params["model"] = model + returned_response.hidden_params["model"] = model return returned_response def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: diff --git a/litellm/llms/soniox/audio_transcription/handler.py b/litellm/llms/soniox/audio_transcription/handler.py index a335caa65c2..a9723125f17 100644 --- a/litellm/llms/soniox/audio_transcription/handler.py +++ b/litellm/llms/soniox/audio_transcription/handler.py @@ -458,7 +458,7 @@ class SonioxAudioTranscriptionHandler: self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload) audio_duration_ms: Final = transcription_meta.get("audio_duration_ms") - response._hidden_params.update( + response.hidden_params.update( { "model": model, "custom_llm_provider": "soniox", @@ -692,7 +692,7 @@ class SonioxAudioTranscriptionHandler: self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload) audio_duration_ms: Final = transcription_meta.get("audio_duration_ms") - response._hidden_params.update( + response.hidden_params.update( { "model": model, "custom_llm_provider": "soniox", diff --git a/litellm/llms/soniox/audio_transcription/transformation.py b/litellm/llms/soniox/audio_transcription/transformation.py index 8507ae73305..48510015f44 100644 --- a/litellm/llms/soniox/audio_transcription/transformation.py +++ b/litellm/llms/soniox/audio_transcription/transformation.py @@ -260,7 +260,7 @@ class SonioxAudioTranscriptionConfig(BaseAudioTranscriptionConfig): # Stash the raw Soniox payload so power-users can read tokens, segments, # speaker/language data, etc. - response._hidden_params.update( + response.hidden_params.update( { "soniox_raw": { "transcription": transcription_meta, diff --git a/litellm/llms/stability/image_edit/transformations.py b/litellm/llms/stability/image_edit/transformations.py index 94711d21b50..fd184f4141d 100644 --- a/litellm/llms/stability/image_edit/transformations.py +++ b/litellm/llms/stability/image_edit/transformations.py @@ -300,14 +300,14 @@ class StabilityImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + if "additional_headers" not in model_response.hidden_params: + model_response.hidden_params["additional_headers"] = {} # Override: fetch model-cost from model_cost map based on the provided model name model_info: Final = get_model_info(model, custom_llm_provider="stability") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( + model_response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( cost_per_image ) return model_response diff --git a/litellm/llms/vertex_ai/audio_transcription/transformation.py b/litellm/llms/vertex_ai/audio_transcription/transformation.py index b1284e15def..5603be81095 100644 --- a/litellm/llms/vertex_ai/audio_transcription/transformation.py +++ b/litellm/llms/vertex_ai/audio_transcription/transformation.py @@ -187,7 +187,7 @@ class VertexAIAudioTranscriptionConfig(BaseAudioTranscriptionConfig, VertexBase) billed_duration = _parse_duration_seconds(parsed.metadata.totalBilledDuration if parsed.metadata else None) if billed_duration is not None: response["duration"] = billed_duration - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b5f32d57061..d3d77c8c304 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -2098,10 +2098,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): ) -> None: setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) if grounding_metadata: - model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata + model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) if url_context_metadata: - model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata + model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata setattr(model_response, "vertex_ai_safety_ratings", safety_ratings) setattr(model_response, "vertex_ai_safety_results", safety_ratings) if safety_ratings: @@ -2128,7 +2128,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): merged.append(value) if merged: setattr(response, field_name, merged) - response._hidden_params[field_name] = merged + response.hidden_params[field_name] = merged @staticmethod def _convert_grounding_metadata_to_annotations( @@ -2470,27 +2470,27 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): ## ADD METADATA TO RESPONSE ## setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) - model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata + model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) - model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata + model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata setattr(model_response, "vertex_ai_safety_results", safety_ratings) - model_response._hidden_params["vertex_ai_safety_results"] = ( + model_response.hidden_params["vertex_ai_safety_results"] = ( safety_ratings # older approach - maintaining to prevent regressions ) ## ADD CITATION METADATA ## setattr(model_response, "vertex_ai_citation_metadata", citation_metadata) - model_response._hidden_params["vertex_ai_citation_metadata"] = ( + model_response.hidden_params["vertex_ai_citation_metadata"] = ( citation_metadata # older approach - maintaining to prevent regressions ) ## ADD TRAFFIC TYPE ## traffic_type: Final = completion_response.get("usageMetadata", {}).get("trafficType") if traffic_type: - model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type + model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type ## ADD SERVICE TIER ## if getattr(raw_response, "headers", None): @@ -3221,7 +3221,7 @@ class ModelResponseIterator: traffic_type: Final = processed_chunk.get("usageMetadata", {}).get("trafficType") if traffic_type: - model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type + model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type service_tier: Final = self.response_headers.get("x-gemini-service-tier") if service_tier: @@ -3278,7 +3278,7 @@ class ModelResponseIterator: setattr(model_response, "usage", usage) - model_response._hidden_params["is_finished"] = False + model_response.hidden_params["is_finished"] = False return model_response except json.JSONDecodeError: diff --git a/litellm/llms/volcengine/responses/transformation.py b/litellm/llms/volcengine/responses/transformation.py index 82c4f7f61a7..e3318c39067 100644 --- a/litellm/llms/volcengine/responses/transformation.py +++ b/litellm/llms/volcengine/responses/transformation.py @@ -259,8 +259,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): construct_response: Final[Callable[..., ResponsesAPIResponse]] = ResponsesAPIResponse.model_construct response = construct_response(**raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response ######################################################### @@ -325,8 +325,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response ######################################################### @@ -398,8 +398,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def should_fake_stream( diff --git a/litellm/main.py b/litellm/main.py index 122de9a02c0..b2dbcf98edc 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1035,11 +1035,11 @@ def mock_completion( ) if custom_llm_provider is not None: - model_response._hidden_params["custom_llm_provider"] = custom_llm_provider + model_response.hidden_params["custom_llm_provider"] = custom_llm_provider else: try: _, inferred_provider, _, _ = litellm.utils.get_llm_provider(model=model) - model_response._hidden_params["custom_llm_provider"] = inferred_provider + model_response.hidden_params["custom_llm_provider"] = inferred_provider except Exception: # dont let setting a hidden param block a mock_respose pass @@ -5516,8 +5516,8 @@ def completion( ) if model_response is not None and hasattr(model_response, "_hidden_params"): - model_response._hidden_params["custom_llm_provider"] = custom_llm_provider - model_response._hidden_params["region_name"] = kwargs.get( + model_response.hidden_params["custom_llm_provider"] = custom_llm_provider + model_response.hidden_params["region_name"] = kwargs.get( "aws_region_name", None ) # support region-based pricing for bedrock @@ -6243,7 +6243,7 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse: elif asyncio.iscoroutine(init_response): response = await init_response if response is not None and isinstance(response, EmbeddingResponse) and hasattr(response, "_hidden_params"): - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise ValueError("Unable to get Embedding Response. Please pass a valid llm_provider.") @@ -7360,7 +7360,7 @@ def embedding( else: raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider) if response is not None and hasattr(response, "_hidden_params") and isinstance(response, EmbeddingResponse): - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider) @@ -7925,7 +7925,7 @@ async def atranscription(*args, **kwargs) -> TranscriptionResponse: if existing_duration is None: calculated_duration: Final = calculate_request_duration(file) if calculated_duration is not None: - response._hidden_params["audio_transcription_duration"] = calculated_duration + response.hidden_params["audio_transcription_duration"] = calculated_duration return response except Exception as e: @@ -8198,7 +8198,7 @@ def transcription( if existing_duration is None: calculated_duration: Final = calculate_request_duration(file) if calculated_duration is not None: - response._hidden_params["audio_transcription_duration"] = calculated_duration + response.hidden_params["audio_transcription_duration"] = calculated_duration if response is None: raise ValueError("Unmapped provider passed in. Unable to get the response.") @@ -9074,7 +9074,7 @@ def stream_chunk_builder( else: hidden = getattr(chunk, "_hidden_params", None) if isinstance(hidden, dict) and "provider_specific_fields" in hidden: - response._hidden_params.setdefault("provider_specific_fields", {}).update( + response.hidden_params.setdefault("provider_specific_fields", {}).update( hidden["provider_specific_fields"] ) break @@ -9253,7 +9253,7 @@ def stream_chunk_builder( else: hidden = getattr(chunk, "_hidden_params", None) if isinstance(hidden, dict) and "provider_specific_fields" in hidden: - response._hidden_params.setdefault("provider_specific_fields", {}).update( + response.hidden_params.setdefault("provider_specific_fields", {}).update( hidden["provider_specific_fields"] ) break diff --git a/litellm/proxy/batches_endpoints/endpoints.py b/litellm/proxy/batches_endpoints/endpoints.py index 1a5386b20d0..c5c8ea78426 100644 --- a/litellm/proxy/batches_endpoints/endpoints.py +++ b/litellm/proxy/batches_endpoints/endpoints.py @@ -211,7 +211,7 @@ async def _create_provider_batch_for_managed_file( } response: Final = await llm_router.acreate_batch(**request) response.input_file_id = input_file_id - response._hidden_params["unified_file_id"] = unified_file_id + response.hidden_params["unified_file_id"] = unified_file_id return response @@ -484,7 +484,7 @@ async def create_batch( **_create_batch_data, ) - response._hidden_params[BATCH_CREATE_HIDDEN_PARAM] = True + response.hidden_params[BATCH_CREATE_HIDDEN_PARAM] = True ### CALL HOOKS ### - modify outgoing data response = await proxy_logging_obj.post_call_success_hook( @@ -736,11 +736,11 @@ async def retrieve_batch( ) response = await llm_router.aretrieve_batch(**data) - response._hidden_params["unified_batch_id"] = unified_batch_id + response.hidden_params["unified_batch_id"] = unified_batch_id if unified_batch_id: model_id_from_batch: Final = get_model_id_from_unified_batch_id(unified_batch_id) if model_id_from_batch: - response._hidden_params["model_id"] = model_id_from_batch + response.hidden_params["model_id"] = model_id_from_batch # SCENARIO 3: Fallback to custom_llm_provider (uses env variables) else: @@ -1168,10 +1168,10 @@ async def cancel_batch( data["model"] = model_id_from_batch data["batch_id"] = get_batch_id_from_unified_batch_id(unified_batch_id) response = await llm_router.acancel_batch(**data) - response._hidden_params["unified_batch_id"] = unified_batch_id + response.hidden_params["unified_batch_id"] = unified_batch_id - if not response._hidden_params.get("model_id") and data.get("model"): - response._hidden_params["model_id"] = data["model"] + if not response.hidden_params.get("model_id") and data.get("model"): + response.hidden_params["model_id"] = data["model"] # SCENARIO 3: Fallback to custom_llm_provider (uses env variables) else: diff --git a/litellm/proxy/fine_tuning_endpoints/endpoints.py b/litellm/proxy/fine_tuning_endpoints/endpoints.py index e09f1ec8ba8..a22c2db084b 100644 --- a/litellm/proxy/fine_tuning_endpoints/endpoints.py +++ b/litellm/proxy/fine_tuning_endpoints/endpoints.py @@ -161,7 +161,7 @@ async def create_fine_tuning_job( response = cast(LiteLLMFineTuningJob, await llm_router.acreate_fine_tuning_job(**data)) response.training_file = unified_file_id - response._hidden_params["unified_file_id"] = unified_file_id + response.hidden_params["unified_file_id"] = unified_file_id ## ELSE, Route based on custom_llm_provider elif fine_tuning_request.custom_llm_provider: # get configs for custom_llm_provider @@ -304,7 +304,7 @@ async def retrieve_fine_tuning_job( **data, ), ) - response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id + response.hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id elif custom_llm_provider: # get configs for custom_llm_provider llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider) @@ -577,7 +577,7 @@ async def cancel_fine_tuning_job( **data, ), ) - response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id + response.hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id else: # get configs for custom_llm_provider llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider) diff --git a/litellm/proxy/hooks/dynamic_rate_limiter.py b/litellm/proxy/hooks/dynamic_rate_limiter.py index 8c41eb8d2d3..e7b2e989213 100644 --- a/litellm/proxy/hooks/dynamic_rate_limiter.py +++ b/litellm/proxy/hooks/dynamic_rate_limiter.py @@ -243,9 +243,9 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger): async def async_post_call_success_hook(self, data: dict, user_api_key_dict: UserAPIKeyAuth, response): try: if isinstance(response, ModelResponse): - model_info: Final = self.llm_router.get_model_info(id=response._hidden_params["model_id"]) + model_info: Final = self.llm_router.get_model_info(id=response.hidden_params["model_id"]) assert model_info is not None, "Model info for model with id={} is None".format( - response._hidden_params["model_id"] + response.hidden_params["model_id"] ) key_priority: Final[str | None] = user_api_key_dict.metadata.get("priority", None) ( @@ -255,7 +255,7 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger): model_rpm, active_projects, ) = await self.check_available_usage(model=model_info["model_name"], priority=key_priority) - response._hidden_params["additional_headers"] = { # Add additional response headers - easier debugging + response.hidden_params["additional_headers"] = { # Add additional response headers - easier debugging "x-litellm-model_group": model_info["model_name"], "x-ratelimit-remaining-litellm-project-tokens": available_tpm, "x-ratelimit-remaining-litellm-project-requests": available_rpm, diff --git a/litellm/proxy/openai_files_endpoints/storage_backend_service.py b/litellm/proxy/openai_files_endpoints/storage_backend_service.py index 53f48d93aa2..ea38d589266 100644 --- a/litellm/proxy/openai_files_endpoints/storage_backend_service.py +++ b/litellm/proxy/openai_files_endpoints/storage_backend_service.py @@ -170,9 +170,9 @@ class StorageBackendFileService: ) # Store storage metadata in hidden params - if not hasattr(file_object, "_hidden_params") or file_object._hidden_params is None: - file_object._hidden_params = {} - file_object._hidden_params.update( + if not hasattr(file_object, "_hidden_params") or file_object.hidden_params is None: + file_object.hidden_params = {} + file_object.hidden_params.update( { "storage_backend": target_storage, "storage_url": storage_url, diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py index 597c2e742b3..d9fd69a29cf 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py @@ -115,8 +115,8 @@ class CoherePassthroughLoggingHandler(BasePassthroughLoggingHandler): # Set the calculated cost in _hidden_params to prevent recalculation if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + litellm_model_response.hidden_params = {} + litellm_model_response.hidden_params["response_cost"] = response_cost kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py index d97ddb9a909..7d7ec6da18a 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py @@ -79,8 +79,8 @@ class GeminiPassthroughLoggingHandler: # Set response_cost in _hidden_params to prevent recalculation if not hasattr(litellm_video_response, "_hidden_params"): - litellm_video_response._hidden_params = {} - litellm_video_response._hidden_params["response_cost"] = response_cost + litellm_video_response.hidden_params = {} + litellm_video_response.hidden_params["response_cost"] = response_cost kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 6aef278963d..ef321bfdf8b 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -414,7 +414,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): model=model, custom_llm_provider=custom_llm_provider, ) - litellm_model_response._hidden_params["response_cost"] = response_cost + litellm_model_response.hidden_params["response_cost"] = response_cost elif is_image_generation: # Handle image generation cost calculation response_cost = OpenAIPassthroughLoggingHandler._calculate_image_generation_cost( @@ -434,8 +434,8 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): ) # Set the calculated cost in _hidden_params to prevent recalculation if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + litellm_model_response.hidden_params = {} + litellm_model_response.hidden_params["response_cost"] = response_cost elif is_image_editing: # Handle image editing cost calculation response_cost = OpenAIPassthroughLoggingHandler._calculate_image_editing_cost( @@ -455,8 +455,8 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): ) # Set the calculated cost in _hidden_params to prevent recalculation if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + litellm_model_response.hidden_params = {} + litellm_model_response.hidden_params["response_cost"] = response_cost elif is_responses: # Responses-API cost tracking — see # `_build_responses_api_response_and_cost` for why this needs diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py index be40e225cde..b1b84973774 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py @@ -174,8 +174,8 @@ class VertexPassthroughLoggingHandler: # Set response_cost in _hidden_params to prevent recalculation if not hasattr(litellm_video_response, "_hidden_params"): - litellm_video_response._hidden_params = {} - litellm_video_response._hidden_params["response_cost"] = response_cost + litellm_video_response.hidden_params = {} + litellm_video_response.hidden_params["response_cost"] = response_cost kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/responses/file_search/emulated_handler.py b/litellm/responses/file_search/emulated_handler.py index 887d1a9ff93..e2bb2ae40ea 100644 --- a/litellm/responses/file_search/emulated_handler.py +++ b/litellm/responses/file_search/emulated_handler.py @@ -380,7 +380,7 @@ def _synthesize_responses_api_response( if first_cost is not None: current_cost: Final = hidden.get("response_cost") if isinstance(hidden, dict) else 0 hidden["response_cost"] = (current_cost or 0) + first_cost - synthesized._hidden_params = hidden + synthesized.hidden_params = hidden return synthesized diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index f01e515d356..b6419ddb735 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -649,9 +649,9 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): ), ) if response is not None and self._accumulated_provider_specific_fields: - if not hasattr(response, "_hidden_params") or response._hidden_params is None: - response._hidden_params = {} - response._hidden_params.setdefault("provider_specific_fields", {}).update( + if not hasattr(response, "_hidden_params") or response.hidden_params is None: + response.hidden_params = {} + response.hidden_params.setdefault("provider_specific_fields", {}).update( self._accumulated_provider_specific_fields ) return response diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 13b353dbdd0..b1e0e47329a 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -2486,10 +2486,10 @@ class LiteLLMCompletionResponsesConfig: user=echoed.get("user"), store=echoed.get("store"), ) - responses_api_response._hidden_params = getattr(chat_completion_response, "_hidden_params", {}) + responses_api_response.hidden_params = getattr(chat_completion_response, "_hidden_params", {}) # Surface provider-specific fields (generic passthrough from any provider) - provider_fields: Final = responses_api_response._hidden_params.get("provider_specific_fields") + provider_fields: Final = responses_api_response.hidden_params.get("provider_specific_fields") if provider_fields: setattr(responses_api_response, "provider_specific_fields", provider_fields) diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 5e2793ed40d..f589a4f68f6 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -748,7 +748,7 @@ async def aresponses( ) # Stamp custom_llm_provider so callbacks can identify the provider # (mirrors litellm/main.py:1371 for chat completions) - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise ValueError(f"Got an unexpected None response from the Responses API: {response}") @@ -1453,7 +1453,7 @@ def responses( ) # Stamp custom_llm_provider so callbacks can identify the provider # (mirrors litellm/main.py:1371 for chat completions) - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider return response except Exception as e: diff --git a/litellm/responses/mcp/chat_completions_handler.py b/litellm/responses/mcp/chat_completions_handler.py index 8aa4181cc8a..5e40b887397 100644 --- a/litellm/responses/mcp/chat_completions_handler.py +++ b/litellm/responses/mcp/chat_completions_handler.py @@ -43,7 +43,7 @@ def _add_mcp_metadata_to_response( # CustomStreamWrapper._add_mcp_metadata_to_final_chunk() will automatically # add it to the final chunk's delta.provider_specific_fields if not hasattr(response, "_hidden_params"): - response._hidden_params = {} + response.hidden_params = {} mcp_metadata: Final = {} if openai_tools: @@ -54,7 +54,7 @@ def _add_mcp_metadata_to_response( mcp_metadata["mcp_call_results"] = tool_results if mcp_metadata: - response._hidden_params["mcp_metadata"] = mcp_metadata + response.hidden_params["mcp_metadata"] = mcp_metadata return if not isinstance(response, ModelResponse): diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index c1dabf8a5c9..6d950f641ac 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -588,7 +588,7 @@ class BaseResponsesAPIStreamingIterator: target: Final[object] = getattr(logging_response, "response", None) if not isinstance(target, ResponsesAPIResponse): return - existing: Final[Mapping[str, object]] = target._hidden_params + existing: Final[Mapping[str, object]] = target.hidden_params source_hidden: Final[object] = getattr( getattr(self.completed_response, "response", None), "_hidden_params", None ) @@ -599,7 +599,7 @@ class BaseResponsesAPIStreamingIterator: raw_headers: Final[Mapping[str, object]] = raw if isinstance(raw, Mapping) else EMPTY_MAPPING # rebuild by value and let existing keys win: sharing the source dicts would alias what the proxy # splats into the client's HTTP headers, and copying non-header keys would carry response_cost - target._hidden_params = { + target.hidden_params = { "additional_headers": {**headers}, "headers": {**raw_headers}, **existing, diff --git a/litellm/router.py b/litellm/router.py index afc842a05a0..d7211495db4 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -2909,7 +2909,7 @@ class Router: if not isinstance(item_headers, dict): item_headers = {} - cast(_HiddenParamsHost, fallback_item)._hidden_params = { + cast(_HiddenParamsHost, fallback_item).hidden_params = { **item_hidden_params, **fallback_hidden_params, "additional_headers": {**item_headers, **fallback_headers}, @@ -4393,7 +4393,7 @@ class Router: if result is not None: # Return the first successful result - result._hidden_params["fastest_response_batch_completion"] = True + result.hidden_params["fastest_response_batch_completion"] = True return result # If we exit the loop without returning, all tasks failed @@ -4462,8 +4462,8 @@ class Router: if make_request: try: _response: Final = await self.acompletion(model=model, messages=messages, stream=stream, **kwargs) - _response._hidden_params.setdefault("additional_headers", {}) - _response._hidden_params["additional_headers"].update({"x-litellm-request-prioritization-used": True}) + _response.hidden_params.setdefault("additional_headers", {}) + _response.hidden_params["additional_headers"].update({"x-litellm-request-prioritization-used": True}) return _response except Exception as e: setattr(e, "priority", priority) @@ -4522,9 +4522,9 @@ class Router: if make_request: try: _response: Final = await original_function(*args, **kwargs) - if isinstance(_response._hidden_params, dict): - _response._hidden_params.setdefault("additional_headers", {}) - _response._hidden_params["additional_headers"].update( + if isinstance(_response.hidden_params, dict): + _response.hidden_params.setdefault("additional_headers", {}) + _response.hidden_params["additional_headers"].update( {"x-litellm-request-prioritization-used": True} ) return _response @@ -6151,7 +6151,7 @@ class Router: healthy_deployments=healthy_deployments, responses=responses ) returned_response: Final = cast(OpenAIFileObject, responses[0]) - returned_response._hidden_params["model_file_id_mapping"] = model_file_id_mapping + returned_response.hidden_params["model_file_id_mapping"] = model_file_id_mapping return returned_response except Exception as e: verbose_router_logger.exception( diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 43200271f9f..cae2b70bfaa 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -410,7 +410,7 @@ def _parent_session_kwargs(request_kwargs: Mapping[str, object] | None) -> Mappi def _response_cost_or_none(response: ModelResponse | ResponsesAPIResponse) -> float | None: - hidden_params: Final = response._hidden_params + hidden_params: Final = response.hidden_params if not isinstance(hidden_params, dict): return None cost: Final = hidden_params.get("response_cost") diff --git a/litellm/router_utils/add_retry_fallback_headers.py b/litellm/router_utils/add_retry_fallback_headers.py index 6e07693b7ea..5178d62e0ad 100644 --- a/litellm/router_utils/add_retry_fallback_headers.py +++ b/litellm/router_utils/add_retry_fallback_headers.py @@ -17,6 +17,12 @@ class FallbackErrorInfo(TypedDict): class _HiddenParamsHost(Protocol): _hidden_params: dict[str, object] + @property + def hidden_params(self) -> dict[str, object]: ... # mutable-ok: API requires mutation + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: ... # mutable-ok: API requires mutation + _EMPTY_OBJECT_MAPPING: Final[Mapping[str, object]] = MappingProxyType({}) _ROUTING_HEADER_MAPPING: Final = TypeAdapter(Mapping[str, object]) @@ -213,7 +219,9 @@ def _write_hidden_params(response: object, hidden_params: dict[str, object]) -> if isinstance(response, dict): response["_hidden_params"] = hidden_params elif hasattr(response, "_hidden_params"): - cast(_HiddenParamsHost, response)._hidden_params = hidden_params + host: Final = cast(_HiddenParamsHost, response) + if get_hidden_params_dict(response) is not hidden_params: + host.hidden_params = hidden_params def _ensure_additional_headers_dict( diff --git a/litellm/types/agents.py b/litellm/types/agents.py index 94adb9f7c4a..d67e90e21e0 100644 --- a/litellm/types/agents.py +++ b/litellm/types/agents.py @@ -360,6 +360,14 @@ class AgentCreateResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentDeleteResult(LiteLLMPydanticObjectBase): """Result of a provider-side agent deletion (e.g. Gemini DELETE /v1beta/agents/{name}). @@ -374,6 +382,14 @@ class AgentDeleteResult(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentListResponse(LiteLLMPydanticObjectBase): """Response from listing agents on the provider side (e.g. Gemini GET /v1beta/agents). @@ -388,6 +404,14 @@ class AgentListResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentVersionsResponse(LiteLLMPydanticObjectBase): """Response from listing versions of an agent (e.g. Gemini GET /v1beta/agents/{name}/versions). @@ -402,6 +426,14 @@ class AgentVersionsResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentMakePublicResponse(BaseModel): message: str @@ -465,6 +497,14 @@ class LiteLLMSendMessageResponse(LiteLLMPydanticObjectBase): # LiteLLM private attributes for logging/cost tracking _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @classmethod def from_a2a_response( cls, diff --git a/litellm/types/containers/main.py b/litellm/types/containers/main.py index 62ef524a435..e8d6ef09cde 100644 --- a/litellm/types/containers/main.py +++ b/litellm/types/containers/main.py @@ -25,6 +25,14 @@ class ContainerObject(BaseModel): name: str | None = None _hidden_params: dict[str, Any] = {} + @property + def hidden_params(self) -> dict[str, Any]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, Any]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key: str) -> bool: # Define custom behavior for the 'in' operator return hasattr(self, key) @@ -142,6 +150,14 @@ class ContainerFileObject(BaseModel): source: str _hidden_params: dict[str, builtins.object] = {} + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key: str) -> bool: return hasattr(self, key) diff --git a/litellm/types/decisions.py b/litellm/types/decisions.py index 80db26f201d..2745e00b6c6 100644 --- a/litellm/types/decisions.py +++ b/litellm/types/decisions.py @@ -121,3 +121,7 @@ class DecisionsResponse(LiteLLMPydanticObjectBase): model_config = ConfigDict(extra="allow", frozen=True) _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params diff --git a/litellm/types/google_genai/main.py b/litellm/types/google_genai/main.py index 13ba9423a53..9ae541374a1 100644 --- a/litellm/types/google_genai/main.py +++ b/litellm/types/google_genai/main.py @@ -23,6 +23,14 @@ if TYPE_CHECKING: class GenerateContentResponse(GoogleGenAIGenerateContentResponse, BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + else: # Fallback types when google.genai is not available ContentListUnion = Any @@ -50,6 +58,14 @@ else: super().__init__(**kwargs) class GenerateContentResponse(BaseLiteLLMOpenAIResponseObject): + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, **kwargs) -> None: super().__init__(**kwargs) self._hidden_params = kwargs.get("_hidden_params", {}) diff --git a/litellm/types/interactions/generated.py b/litellm/types/interactions/generated.py index a666d236e4b..3bb7c1c9588 100644 --- a/litellm/types/interactions/generated.py +++ b/litellm/types/interactions/generated.py @@ -1081,6 +1081,14 @@ class InteractionsAPIResponse(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject): """ @@ -1119,6 +1127,14 @@ class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject): """Result of deleting an interaction.""" @@ -1128,6 +1144,14 @@ class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject): """Result of cancelling an interaction.""" @@ -1137,6 +1161,14 @@ class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + # Backwards compatibility aliases InteractionTool = Tool diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 6db7fd68292..73ccaab99df 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -121,6 +121,14 @@ class BinaryResponseSummary(TypedDict): class HttpxBinaryResponseContent(_HttpxBinaryResponseContent): _hidden_params: dict + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, response: httpx.Response) -> None: super().__init__(response) self._hidden_params = {} @@ -407,6 +415,14 @@ class OpenAIFileObject(BaseModel): _hidden_params: dict = {"response_cost": 0.0} # no cost for writing a file + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @model_serializer(mode="wrap") def _omit_absent_batch_guardrail( # noqa: ANN202 # annotating it replaces the model's serialization schema self, handler: SerializerFunctionWrapHandler @@ -1423,6 +1439,14 @@ class ResponsesAPIResponse(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @field_validator("reasoning", mode="before") @classmethod def validate_reasoning_to_dict(cls, value: Any) -> dict[str, Any] | None: @@ -1598,6 +1622,14 @@ class ResponseCompletedEvent(BaseLiteLLMOpenAIResponseObject): response: ResponsesAPIResponse _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class ResponseFailedEvent(BaseLiteLLMOpenAIResponseObject): type: Literal[ResponsesAPIStreamEvents.RESPONSE_FAILED] @@ -2398,6 +2430,14 @@ class OpenAIModerationResponse(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class OpenAIChatCompletionLogprobs(TypedDict, total=False): content: list[OpenAIChatCompletionLogprobsContent] @@ -2560,6 +2600,14 @@ class OpenAIVideoObject(BaseModel): _hidden_params: dict[str, _JsonValue] = {} + @property + def hidden_params(self) -> dict[str, _JsonValue]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, _JsonValue]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: return hasattr(self, key) diff --git a/litellm/types/rerank.py b/litellm/types/rerank.py index a76e6cf1187..f99aaf37fd7 100644 --- a/litellm/types/rerank.py +++ b/litellm/types/rerank.py @@ -85,6 +85,14 @@ class RerankResponse(BaseModel): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __getitem__(self, key): return self.__dict__[key] diff --git a/litellm/types/responses/main.py b/litellm/types/responses/main.py index 2381c7ff3a1..2880b19f589 100644 --- a/litellm/types/responses/main.py +++ b/litellm/types/responses/main.py @@ -164,6 +164,14 @@ class DeleteResponseResult(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class DecodedResponseId(TypedDict, total=False): """Structure representing a decoded response ID""" diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 142f9b14a72..c9fc18e5e65 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2087,6 +2087,14 @@ class ModelResponseBase(OpenAIObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + _response_headers: dict | None = None def set_provider_response_headers(self, headers: httpx.Headers) -> None: @@ -2309,6 +2317,15 @@ class EmbeddingResponse(OpenAIObject): """Usage statistics for the embedding request.""" _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + _response_headers: dict | None = None _response_ms: float | None = None @@ -2449,6 +2466,14 @@ class TextCompletionResponse(OpenAIObject): _response_ms: int | None = None _hidden_params: HiddenParams + @property + def hidden_params(self) -> HiddenParams: + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: HiddenParams) -> None: + self._hidden_params = hidden_params + def __init__( self, id=None, @@ -2613,6 +2638,14 @@ from openai.types.images_response import ImagesResponse as OpenAIImageResponse class ImageResponse(OpenAIImageResponse, BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + usage: ImageUsage | None = None """ Users might use litellm with older python versions, we don't want this to break for them. @@ -2748,6 +2781,14 @@ class TranscriptionResponse(OpenAIObject): _hidden_params: dict = {} _response_headers: dict | None = None + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, text=None) -> None: super().__init__(text=text) @@ -4314,6 +4355,15 @@ class SelectTokenizerResponse(TypedDict): class LiteLLMFineTuningJob(FineTuningJob): _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + seed: int | None = None def __init__(self, **kwargs) -> None: @@ -4327,6 +4377,15 @@ class LiteLLMFineTuningJob(FineTuningJob): class LiteLLMBatch(Batch): _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + usage: Usage | None = None def __contains__(self, key) -> bool: @@ -4359,6 +4418,14 @@ class LiteLLMRealtimeStreamLoggingObject(LiteLLMPydanticObjectBase): service_tier: str | None = None _hidden_params: dict = {} + @property + def hidden_params(self) -> dict: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @field_serializer("results") def _serialize_results(self, results: OpenAIRealtimeStreamList) -> list[dict[str, Any]]: return [dict(event) for event in results] diff --git a/litellm/types/videos/main.py b/litellm/types/videos/main.py index f4369fd95af..81f5b8242a3 100644 --- a/litellm/types/videos/main.py +++ b/litellm/types/videos/main.py @@ -24,6 +24,14 @@ class VideoObject(BaseModel): usage: dict[str, Any] | None = None _hidden_params: dict[str, builtins.object] = {} + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: # Define custom behavior for the 'in' operator return hasattr(self, key) @@ -113,6 +121,14 @@ class CharacterObject(BaseModel): name: str _hidden_params: dict[str, builtins.object] = {} + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: return hasattr(self, key) diff --git a/tests/unit/decisions/test_main.py b/tests/unit/decisions/test_main.py index 106710d328f..684248b3561 100644 --- a/tests/unit/decisions/test_main.py +++ b/tests/unit/decisions/test_main.py @@ -267,6 +267,15 @@ def test_decisions_cost_uses_litellm_token_pricing() -> None: assert cost == pytest.approx(expected_cost) +def test_decisions_response_hidden_params_getter_preserves_mutable_identity() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + + assert response.hidden_params is response._hidden_params + + response.hidden_params["mutation"] = "visible" + assert response._hidden_params["mutation"] == "visible" + + @pytest.mark.asyncio async def test_decisions_cost_is_in_standard_logging_object(respx_mock: respx.MockRouter) -> None: respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE) diff --git a/tests/unit/router_strategy/test_complexity_router.py b/tests/unit/router_strategy/test_complexity_router.py index dfeb9f8d961..c0e14df3351 100644 --- a/tests/unit/router_strategy/test_complexity_router.py +++ b/tests/unit/router_strategy/test_complexity_router.py @@ -2710,6 +2710,7 @@ def _llm_response(content: str, response_cost: float | None = None): response.choices = [MagicMock()] response.choices[0].message.content = content response._hidden_params = {} if response_cost is None else {"response_cost": response_cost} + response.hidden_params = response._hidden_params return response diff --git a/tests/unit/router_utils/test_add_retry_fallback_headers.py b/tests/unit/router_utils/test_add_retry_fallback_headers.py index eb7490d76f8..ae487d29917 100644 --- a/tests/unit/router_utils/test_add_retry_fallback_headers.py +++ b/tests/unit/router_utils/test_add_retry_fallback_headers.py @@ -1,6 +1,6 @@ import json from collections.abc import Mapping -from typing import Literal +from typing import Final, Literal import pytest from pydantic import BaseModel @@ -13,6 +13,7 @@ from litellm.router_utils.add_retry_fallback_headers import ( get_hidden_params_dict, replace_complexity_router_headers, ) +from litellm.types.decisions import DecisionsResponse class StreamingWrapper: @@ -230,6 +231,15 @@ def test_add_fallback_headers_when_no_existing_additional_headers(): assert response._hidden_params["additional_headers"]["x-litellm-attempted-fallbacks"] == 2 +def test_add_fallback_headers_to_frozen_decisions_response() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + + result: Final = add_fallback_headers_to_response(response=response, attempted_fallbacks=1) + + assert result is response + assert response.hidden_params["additional_headers"] == {"x-litellm-attempted-fallbacks": 1} + + def test_add_fallback_headers_returns_none_when_response_is_none(): result = add_fallback_headers_to_response(response=None, attempted_fallbacks=1) assert result is None diff --git a/tests/unit/test_router_streaming_fallback_metadata.py b/tests/unit/test_router_streaming_fallback_metadata.py index 6ed70dc7cfe..85d5f6dec87 100644 --- a/tests/unit/test_router_streaming_fallback_metadata.py +++ b/tests/unit/test_router_streaming_fallback_metadata.py @@ -1,4 +1,5 @@ import json +from typing import Final from unittest.mock import MagicMock import pytest @@ -125,10 +126,12 @@ def test_apply_fallback_hidden_params_to_item_none_item(): def test_apply_fallback_hidden_params_to_item_no_existing_additional_headers(): - class FakeChunk: - _hidden_params = {"model_id": "test-id"} - - chunk = FakeChunk() + chunk: Final = litellm.ModelResponseStream( + id="test", + model="openai/internal-fallback", + choices=[], + ) + chunk.hidden_params["model_id"] = "test-id" Router._apply_fallback_hidden_params_to_item( chunk, ( @@ -137,9 +140,9 @@ def test_apply_fallback_hidden_params_to_item_no_existing_additional_headers(): ), ) - assert chunk._hidden_params["api_base"] == "http://fallback.example" - assert chunk._hidden_params["model_id"] == "test-id" - assert chunk._hidden_params["additional_headers"] == { + assert chunk.hidden_params["api_base"] == "http://fallback.example" + assert chunk.hidden_params["model_id"] == "test-id" + assert chunk.hidden_params["additional_headers"] == { "x-litellm-attempted-fallbacks": 1 } diff --git a/tests/unit/types/llms/test_types_llms_openai.py b/tests/unit/types/llms/test_types_llms_openai.py index 64ec09838e8..2e96fd99d40 100644 --- a/tests/unit/types/llms/test_types_llms_openai.py +++ b/tests/unit/types/llms/test_types_llms_openai.py @@ -1,5 +1,5 @@ import asyncio -from typing import Optional +from typing import Final, Optional from unittest.mock import AsyncMock, patch import pytest @@ -7,7 +7,7 @@ import pytest import json import litellm -from litellm.types.llms.openai import HttpxBinaryResponseContent +from litellm.types.llms.openai import HttpxBinaryResponseContent, ResponsesAPIResponse @pytest.mark.parametrize("stream", (False, True)) @@ -589,3 +589,20 @@ def test_set_response_cost_none_leaves_hidden_params_empty(): binary_response.set_response_cost(None) assert "response_cost" not in binary_response._hidden_params + + +def test_responses_api_response_hidden_params_public_accessor_is_instance_scoped() -> None: + first: Final = ResponsesAPIResponse(id="resp_first", created_at=1, output=[]) + second: Final = ResponsesAPIResponse(id="resp_second", created_at=2, output=[]) + + assert first.hidden_params is first._hidden_params + + first.hidden_params["public_key"] = "visible" + assert first._hidden_params["public_key"] == "visible" + assert first.hidden_params is not second.hidden_params + assert "public_key" not in second.hidden_params + + replacement: Final = {"replacement_key": "replacement_value"} + first.hidden_params = replacement + assert first._hidden_params is replacement + assert first.hidden_params is replacement diff --git a/tests/unit/types/test_types_utils.py b/tests/unit/types/test_types_utils.py index 0a8c9414a0d..c032e3f3ddf 100644 --- a/tests/unit/types/test_types_utils.py +++ b/tests/unit/types/test_types_utils.py @@ -5,9 +5,12 @@ import pytest from litellm.types.utils import ( + EmbeddingResponse, HiddenParams, ImageObject, ImageResponse, + ModelResponse, + ModelResponseStream, all_litellm_params, text_tokens_without_nested_reasoning, ) @@ -24,6 +27,26 @@ def test_hidden_params_response_ms(): assert hidden_params_dict.get("_response_ms") == 100 +@pytest.mark.parametrize("response_type", (ModelResponse, ModelResponseStream, EmbeddingResponse)) +def test_hidden_params_public_accessor_preserves_identity_and_instance_isolation( + response_type: type[ModelResponse] | type[ModelResponseStream] | type[EmbeddingResponse], +) -> None: + response: Final = response_type() + other_response: Final = response_type() + + assert response.hidden_params is response._hidden_params + + response.hidden_params["public_key"] = "visible" + assert response._hidden_params["public_key"] == "visible" + assert response.hidden_params is not other_response.hidden_params + assert "public_key" not in other_response.hidden_params + + replacement: Final = {"replacement_key": "replacement_value"} + response.hidden_params = replacement + assert response._hidden_params is replacement + assert response.hidden_params is replacement + + def test_chat_completion_delta_tool_call(): from litellm.types.utils import ChatCompletionDeltaToolCall, Function